Files
Christian ManivongandClaude Opus 5.5 369f66afdc feat(screenshots): cropped shots, sizes for the site, a site check
- shots.py crops to a region (clip) or to what one or more elements cover
  (element, pad), per-shot viewport; capture.py writes the published sizes to
  src/data/screenshots.json so the page reserves the right space.
- anonymize.py no longer empties secrets that netOrk compares with each other
  (Wi-Fi keys on an SSID against the key read from the access point). Emptying
  them invented passphrase "drift" that never existed; a keyed hash keeps equal
  equal, reverses nothing, and its key lives for one run.
- scripts/check/site.py checks the built site in both languages at four widths:
  sideways overflow, one h1, images with alt and size, console errors, requests
  to other origins, links to unknown routes, old-URL redirects, language
  detection, and word counts against the budgets.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-29 23:24:49 +02:00

245 lines
10 KiB
Python

#!/usr/bin/env python3
"""Take real screenshots of a running netOrk instance for the website.
Normally that instance is the local demo copy from scripts/demo (anonymized
production data), which this script logs into on its own:
capture.py --list-devices # prints IDs to pick for --var
capture.py --var ap=<id> [--only name ...]
Against a real instance, log in by hand and cover what must not be seen:
NETORK_URL=https://... capture.py --login
NETORK_URL=https://... capture.py --mask --var ...
While capturing, every request to the API that is not a GET is aborted, so
taking screenshots cannot change anything on the instance.
"""
import argparse
import io
import json
import os
import re
import sys
import urllib.error
import urllib.parse
import urllib.request
from pathlib import Path
from PIL import Image
from playwright.sync_api import Page, sync_playwright
from shots import SHOTS
BASE = os.environ.get("NETORK_URL", "http://127.0.0.1:5173").rstrip("/")
LOCAL = re.match(r"https?://(127\.0\.0\.1|localhost)[:/]", BASE + "/") is not None
# The demo instance's admin (see scripts/demo/anonymize.py).
USER = os.environ.get("NETORK_USER", "netork")
PASSWORD = os.environ.get("NETORK_PASSWORD", "netork-demo")
STATE = Path(os.environ.get(
"NETORK_STATE", Path.home() / ".cache" / "netork-screenshots" / "state.json"))
# One term per line: site names, customer names, domains ... never committed.
MASK_FILE = Path(os.environ.get(
"NETORK_MASK_FILE", Path.home() / ".config" / "netork-screenshots" / "mask.txt"))
SITE = Path(__file__).resolve().parents[2]
OUT = SITE / "public" / "screenshots"
VIEWPORT = {"width": 1600, "height": 1000}
# Any IPv4 address that is not RFC 1918, loopback or link-local.
PUBLIC_IPV4 = re.compile(
r"\b(?!10\.)(?!127\.)(?!169\.254\.)(?!192\.168\.)(?!172\.(?:1[6-9]|2\d|3[01])\.)"
r"(?:25[0-5]|2[0-4]\d|1?\d?\d)(?:\.(?:25[0-5]|2[0-4]\d|1?\d?\d)){3}\b")
EMAIL = re.compile(r"[\w.+-]+@[\w-]+\.[\w.-]+")
def mask_terms() -> list[str]:
if not MASK_FILE.exists():
return []
return [t.strip() for t in MASK_FILE.read_text().splitlines()
if t.strip() and not t.startswith("#")]
def login() -> None:
STATE.parent.mkdir(parents=True, exist_ok=True)
with sync_playwright() as p:
browser = p.chromium.launch(headless=False)
ctx = browser.new_context(ignore_https_errors=True, viewport=VIEWPORT)
page = ctx.new_page()
page.goto(f"{BASE}/login")
print("Log in in the browser window (10 minutes) ...", flush=True)
page.wait_for_function(
"() => localStorage.getItem('token') && !location.pathname.startsWith('/login')",
timeout=600_000)
ctx.storage_state(path=STATE)
STATE.chmod(0o600)
browser.close()
print(f"Session saved to {STATE}")
def token() -> str:
if LOCAL:
body = urllib.parse.urlencode({"username": USER, "password": PASSWORD}).encode()
try:
with urllib.request.urlopen(f"{BASE}/api/v1/auth/token", body) as res:
tok = json.load(res).get("access_token")
if not tok:
sys.exit(f"Login as {USER} needs MFA; the demo copy should have none (anonymize.py)")
return tok
except urllib.error.URLError as e:
sys.exit(f"Login as {USER} at {BASE} failed: {e} (is scripts/demo/up.sh start running?)")
state = json.loads(STATE.read_text())
for origin in state.get("origins", []):
for item in origin.get("localStorage", []):
if item["name"] == "token":
return item["value"]
sys.exit("No token in the saved session; run --login first.")
def list_devices() -> None:
with sync_playwright() as p:
req = p.request.new_context(
base_url=BASE, ignore_https_errors=True,
extra_http_headers={"Authorization": f"Bearer {token()}"})
res = req.get("/api/v1/devices/")
if not res.ok:
sys.exit(f"{res.status}: {res.text()[:200]} (session expired? run --login)")
for d in res.json():
print(f"{d.get('id')} {d.get('driver') or '-':18} "
f"{d.get('device_type') or '-':20} {d.get('hostname')}")
def settle(page: Page) -> None:
"""Wait until the page has finished loading its data."""
try:
page.wait_for_load_state("networkidle", timeout=15_000)
except Exception:
pass # pages that poll never go fully idle
try:
page.wait_for_function(
"() => !document.querySelector('.animate-spin, .animate-pulse')", timeout=15_000)
except Exception:
print(" still loading after 15 s, taking the shot anyway")
page.wait_for_timeout(800)
def publish(png: bytes, path: Path, width: int) -> dict[str, int]:
"""Scale the 2x capture down to its published width and store it as WebP."""
img = Image.open(io.BytesIO(png)).convert("RGB")
if img.width > width:
img = img.resize((width, round(img.height * width / img.width)), Image.LANCZOS)
img.save(path, "WEBP", quality=85, method=6)
print(f" -> {path.name} {img.width}x{img.height}, {path.stat().st_size // 1024} KB")
return {"width": img.width, "height": img.height}
def capture(variables: dict[str, str], only: set[str], mask: bool) -> None:
OUT.mkdir(parents=True, exist_ok=True)
terms = mask_terms()
tok = token() if LOCAL else None
blocked: list[str] = []
sizes_file = SITE / "src" / "data" / "screenshots.json"
sizes: dict[str, dict[str, int]] = json.loads(sizes_file.read_text()) if sizes_file.exists() else {}
def guard(route):
if route.request.method in ("GET", "HEAD", "OPTIONS"):
route.continue_()
else:
blocked.append(f"{route.request.method} {route.request.url}")
route.abort()
with sync_playwright() as p:
browser = p.chromium.launch()
ctx = browser.new_context(
storage_state=None if LOCAL else STATE, ignore_https_errors=True,
viewport=VIEWPORT, device_scale_factor=2, color_scheme="dark")
if tok:
ctx.add_init_script(f"localStorage.setItem('token', {json.dumps(tok)})")
ctx.route("**/api/**", guard)
page = ctx.new_page()
for shot in SHOTS:
if only and shot.name not in only:
continue
try:
path = shot.path.format(**variables)
except KeyError as e:
print(f"skip {shot.name}: needs --var {e.args[0]}=<id>")
continue
print(f"{shot.name}: {path}")
vw, vh = shot.viewport or (VIEWPORT["width"], VIEWPORT["height"])
page.set_viewport_size({"width": vw, "height": vh})
page.goto(f"{BASE}{path}")
if page.url.rstrip("/").endswith("/login"):
sys.exit("Session expired; run --login again.")
page.wait_for_selector(shot.wait_for, timeout=20_000)
settle(page)
# The release notes dialog after an upgrade; dismissing it only
# writes localStorage in this throwaway browser context.
got_it = page.get_by_role("button", name="Got it")
if got_it.is_visible():
got_it.click()
page.wait_for_timeout(300)
for sel in shot.clicks:
page.locator(sel).first.click()
settle(page)
masks = [page.locator(s) for s in shot.mask]
if mask:
masks += [page.get_by_text(PUBLIC_IPV4), page.get_by_text(EMAIL)]
masks += [page.get_by_text(t) for t in terms]
clip = None
if shot.clip:
x, y, w, h = shot.clip
clip = {"x": x, "y": y, "width": w, "height": h}
elif shot.element:
boxes = []
for sel in ([shot.element] if isinstance(shot.element, str) else shot.element):
el = page.locator(sel).first
el.scroll_into_view_if_needed()
b = el.bounding_box()
if not b:
sys.exit(f"{shot.name}: element not found: {sel}")
boxes.append(b)
left = min(b["x"] for b in boxes)
top = min(b["y"] for b in boxes)
right = max(b["x"] + b["width"] for b in boxes)
bottom = max(b["y"] + b["height"] for b in boxes)
box = {"x": left, "y": top, "width": right - left, "height": bottom - top}
x0, y0 = max(0, box["x"] - shot.pad), max(0, box["y"] - shot.pad)
clip = {"x": x0, "y": y0,
"width": min(vw - x0, box["width"] + 2 * shot.pad),
"height": min(vh - y0, box["height"] + 2 * shot.pad)}
png = page.screenshot(full_page=shot.full_page, clip=clip, mask=masks,
mask_color="#334155", animations="disabled")
sizes[shot.name] = publish(png, OUT / f"{shot.name}.webp", shot.width)
browser.close()
# The site reads these to reserve the right space for each image.
sizes_file.write_text(json.dumps(dict(sorted(sizes.items())), indent=2) + "\n")
if blocked:
print("Blocked non-GET requests (nothing was sent):")
for b in sorted(set(blocked)):
print(f" {b}")
def main() -> None:
ap = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--login", action="store_true", help="log in and save the session")
ap.add_argument("--list-devices", action="store_true", help="print device IDs")
ap.add_argument("--var", action="append", default=[], metavar="NAME=VALUE",
help="fill a {placeholder} in the shot paths")
ap.add_argument("--only", nargs="*", default=[], help="only these shot names")
ap.add_argument("--mask", action="store_true",
help="cover public IPs, e-mails and the mask-file terms (real instances)")
args = ap.parse_args()
if args.login:
login()
elif args.list_devices:
list_devices()
else:
capture(dict(v.split("=", 1) for v in args.var), set(args.only), args.mask)
if __name__ == "__main__":
main()