From 1863de33c2b3513d6118f0b05cd59a6e1f9c3fa0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?L=C3=A9o?= Date: Fri, 25 Sep 2026 11:27:18 +0200 Subject: [PATCH 1/3] test(css): pin computed styles against a committed baseline static/style.css is 51,425 lines in one file. Hundreds of selectors are declared more than once and !important is used throughout, so the rendered result is a function of source order. Extracting a block into its own file changes that order, and nothing in the suite would notice - which makes a 51k-line split unfalsifiable and "looks fine to me" the only available evidence. This moves no CSS. It captures getComputedStyle over a fixed inventory of 676 elements across three pages, four viewports, both themes and the three density modes - 16,224 element snapshots - hashes them, and compares against tests/css_snapshot/baseline.json. A capture takes about 21 seconds. The bench page synthesises one element per selector from an evidence-driven list: every selector declared more than once in style.css that can be expressed as a static compound chain, plus a curated set per feature area. Redeclared selectors are the ones a reorder can flip. The bench loads whatever stylesheets index.html ships, so it keeps measuring the real set once the file is split. tests/test_css_computed_style_snapshot.py also carries a self-test that swaps two conflicting .attach-strip declarations and asserts the digest moves, so the harness cannot silently stop watching. The second half is the asset-manifest check specs/frontend.md asks for, scoped to stylesheets: every stylesheet referenced by shipped HTML and by the sw.js precache exists, and index.html and sw.js agree on the ?v= string. They hardcode it independently today, so a split that updates one and not the other ships an offline cache nobody notices until a plane. --- scripts/css_snapshot.py | 290 +++++ specs/frontend.md | 8 +- tests/README.md | 19 + tests/css_snapshot/README.md | 129 ++ tests/css_snapshot/baseline.json | 767 ++++++++++++ tests/css_snapshot/bench.html | 20 + tests/css_snapshot/capture.mjs | 266 ++++ tests/css_snapshot/inventory.json | 1349 +++++++++++++++++++++ tests/test_css_computed_style_snapshot.py | 104 ++ tests/test_static_stylesheet_manifest.py | 102 ++ 10 files changed, 3051 insertions(+), 3 deletions(-) create mode 100644 scripts/css_snapshot.py create mode 100644 tests/css_snapshot/README.md create mode 100644 tests/css_snapshot/baseline.json create mode 100644 tests/css_snapshot/bench.html create mode 100644 tests/css_snapshot/capture.mjs create mode 100644 tests/css_snapshot/inventory.json create mode 100644 tests/test_css_computed_style_snapshot.py create mode 100644 tests/test_static_stylesheet_manifest.py diff --git a/scripts/css_snapshot.py b/scripts/css_snapshot.py new file mode 100644 index 000000000..ba927775f --- /dev/null +++ b/scripts/css_snapshot.py @@ -0,0 +1,290 @@ +#!/usr/bin/env python3 +"""Computed-style snapshot harness for ``static/style.css``. + +``static/style.css`` is a single 51k-line stylesheet whose rendered result +depends on source order: hundreds of selectors are declared more than once and +``!important`` is used throughout. Any restructuring - extracting a block into +its own file, reordering ```` tags, moving an ``@media`` rule - can +silently change which declaration wins, and nothing else in the suite would +notice. + +This module captures ``getComputedStyle`` for a fixed inventory of elements +across pages, viewports, themes and density modes, hashes the result, and +compares it against a committed baseline. It moves no CSS. It only makes a move +falsifiable. + +Usage:: + + python scripts/css_snapshot.py --check # compare to the baseline + python scripts/css_snapshot.py --write-baseline # re-record it + python scripts/css_snapshot.py --dump before.json # raw values, for diffing + +With no ``--origin`` the script serves the repository over loopback on an +ephemeral port for the duration of the run, so it works standalone. Under +pytest the session static server is reused instead. + +To see *which property* moved rather than just which element:: + + python scripts/css_snapshot.py --dump after.json + git stash && python scripts/css_snapshot.py --dump before.json && git stash pop + diff <(python -m json.tool before.json) <(python -m json.tool after.json) +""" +import argparse +import hashlib +import http.server +import json +import os +import shutil +import socketserver +import subprocess +import sys +import threading +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +SNAPSHOT_DIR = ROOT / "tests" / "css_snapshot" +INVENTORY_PATH = SNAPSHOT_DIR / "inventory.json" +BASELINE_PATH = SNAPSHOT_DIR / "baseline.json" +CAPTURE_SCRIPT = SNAPSHOT_DIR / "capture.mjs" + +# A capture is ~70 page loads; on a warm checkout it runs in well under a +# minute, but a cold `npx playwright install` machine can be slow to start +# Chromium the first time. +CAPTURE_TIMEOUT_SECONDS = 900 + +# Hash prefix length. 16 hex characters is 64 bits - far past any accidental +# collision risk for a few thousand entries, and short enough that the baseline +# stays readable in a diff. +HASH_LENGTH = 16 + + +def load_inventory(path=INVENTORY_PATH): + """Load the checked-in element inventory.""" + return json.loads(Path(path).read_text(encoding="utf-8")) + + +def load_baseline(path=BASELINE_PATH): + """Load the committed baseline digest.""" + return json.loads(Path(path).read_text(encoding="utf-8")) + + +def _canonical(value): + return json.dumps(value, sort_keys=True, separators=(",", ":"), ensure_ascii=False) + + +def _hash(value): + return hashlib.sha256(_canonical(value).encode("utf-8")).hexdigest()[:HASH_LENGTH] + + +def node_available(node="node"): + """True when the node binary is on PATH.""" + return shutil.which(node) is not None + + +def playwright_available(node="node", cwd=ROOT): + """True when node can resolve the playwright package from the repo root. + + Playwright is a devDependency installed by ``npm ci``; a clean checkout + that has not run it cannot drive a browser at all. + """ + if not node_available(node): + return False + result = subprocess.run( + [node, "-e", "require.resolve('playwright')"], + cwd=str(cwd), capture_output=True, text=True, check=False, + ) + return result.returncode == 0 + + +def capture(origin, inventory=None, *, swap_rule=None, variants=None, + node="node", cwd=ROOT, timeout=CAPTURE_TIMEOUT_SECONDS): + """Drive the browser capture and return ``{"snapshot": ..., "missing": ...}``. + + ``swap_rule`` swaps the first two top-level declarations of one selector + before the stylesheet reaches the browser. It exists for the harness + self-test: a snapshot that does not move when two conflicting rules trade + places is not evidence of anything. + + ``variants`` restricts the run to the named variants, for a faster + focused capture. + """ + inventory = inventory or load_inventory() + selected = inventory["variants"] + if variants: + wanted = set(variants) + selected = [v for v in selected if v["name"] in wanted] + unknown = wanted - {v["name"] for v in inventory["variants"]} + if unknown: + raise ValueError(f"unknown variants: {sorted(unknown)}") + job = { + "origin": origin.rstrip("/"), + "properties": inventory["properties"], + "variants": selected, + "pages": inventory["pages"], + "swapRule": swap_rule, + } + result = subprocess.run( + [node, str(CAPTURE_SCRIPT)], + input=json.dumps(job), cwd=str(cwd), + capture_output=True, text=True, check=False, timeout=timeout, + ) + if result.returncode != 0: + raise RuntimeError(f"css snapshot capture failed:\n{result.stderr.strip()}") + return json.loads(result.stdout) + + +def summarize(snapshot): + """Reduce a raw capture to the committed digest shape. + + Two orthogonal projections are stored rather than one hash per + (element, variant) pair: hashing every pair would commit ~5,000 lines that + nobody reads, while a single global digest would only ever say "something + moved". Per-element and per-variant hashes localise a failure from both + directions - which element drifted, and in which variant - for a file small + enough to review. + """ + elements = {} + variants = {} + for page, per_variant in snapshot.items(): + element_values = {} + variants[page] = {} + for variant, measured in per_variant.items(): + variants[page][variant] = _hash(measured) + for key, values in measured.items(): + element_values.setdefault(key, {})[variant] = values + elements[page] = {key: _hash(values) for key, values in element_values.items()} + return { + "digest": _hash(snapshot), + "elements": elements, + "variants": variants, + } + + +def compare(baseline, current): + """Return the drift between a committed baseline and a fresh summary.""" + drift = {"digest_changed": baseline.get("digest") != current["digest"], + "elements": [], "variants": []} + for section in ("elements", "variants"): + old = baseline.get(section, {}) + new = current.get(section, {}) + for page in sorted(set(old) | set(new)): + old_page = old.get(page, {}) + new_page = new.get(page, {}) + for key in sorted(set(old_page) | set(new_page)): + if old_page.get(key) != new_page.get(key): + drift[section].append(f"{page}/{key}") + return drift + + +def serve_repository(root=ROOT): + """Serve the repository over loopback on an ephemeral port. + + Mirrors the browser-test static server in ``tests/conftest.py`` so the CLI + can run outside pytest. Returns ``(origin, shutdown)``. + """ + root = Path(root).resolve() + + class Handler(http.server.SimpleHTTPRequestHandler): + def __init__(self, *args, **kwargs): + super().__init__(*args, directory=str(root), **kwargs) + + def log_message(self, fmt, *args): + pass + + def guess_type(self, path): + if path.endswith(".js") or path.endswith(".mjs"): + return "application/javascript" + if path.endswith(".css"): + return "text/css" + return super().guess_type(path) + + class Server(socketserver.TCPServer): + allow_reuse_address = True + + server = Server(("127.0.0.1", 0), Handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + + def shutdown(): + server.shutdown() + server.server_close() + + return f"http://127.0.0.1:{server.server_address[1]}", shutdown + + +def _describe(drift, limit=25): + lines = [] + for section in ("elements", "variants"): + items = drift[section] + if not items: + continue + shown = items[:limit] + suffix = f" (+{len(items) - limit} more)" if len(items) > limit else "" + lines.append(f" {section} that moved ({len(items)}): {', '.join(shown)}{suffix}") + return "\n".join(lines) or " (no per-element drift; the digest itself changed)" + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--origin", help="static server origin to capture against; " + "one is started on an ephemeral port when omitted") + parser.add_argument("--write-baseline", action="store_true", + help=f"re-record {BASELINE_PATH.relative_to(ROOT)}") + parser.add_argument("--check", action="store_true", + help="compare against the committed baseline (default)") + parser.add_argument("--dump", metavar="PATH", + help="write the raw computed values, for property-level diffing") + parser.add_argument("--swap-rule", metavar="SELECTOR", + help="swap the first two top-level declarations of SELECTOR " + "before capturing (harness self-test)") + parser.add_argument("--variants", help="comma-separated variant names to restrict the run to") + parser.add_argument("--node", default="node", help="node binary to use") + args = parser.parse_args(argv) + + if not playwright_available(args.node): + parser.error("node with the playwright package is required; run `npm ci` first") + + variants = [v.strip() for v in args.variants.split(",")] if args.variants else None + shutdown = None + origin = args.origin or os.environ.get("ODYSSEUS_TEST_STATIC_ORIGIN") + if not origin: + origin, shutdown = serve_repository() + try: + captured = capture(origin, swap_rule=args.swap_rule, variants=variants, node=args.node) + finally: + if shutdown: + shutdown() + + if captured["missing"]: + print("inventory entries that matched no element:", file=sys.stderr) + for scope, keys in sorted(captured["missing"].items()): + print(f" {scope}: {', '.join(keys)}", file=sys.stderr) + + summary = summarize(captured["snapshot"]) + + if args.dump: + Path(args.dump).write_text(json.dumps(captured["snapshot"], indent=1, sort_keys=True) + "\n", + encoding="utf-8") + print(f"raw values written to {args.dump}") + + if args.write_baseline: + if variants or args.swap_rule: + parser.error("--write-baseline needs a full, unmutated capture: " + "drop --variants and --swap-rule") + BASELINE_PATH.write_text(json.dumps(summary, indent=1, sort_keys=True) + "\n", + encoding="utf-8") + print(f"baseline written: digest {summary['digest']}") + return 0 + + baseline = load_baseline() + drift = compare(baseline, summary) + if not drift["digest_changed"] and not drift["elements"] and not drift["variants"]: + print(f"computed styles match the baseline (digest {summary['digest']})") + return 0 + print(f"computed styles moved: baseline {baseline.get('digest')} -> {summary['digest']}") + print(_describe(drift)) + return 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/specs/frontend.md b/specs/frontend.md index 4bd58d490..cd3eea540 100644 --- a/specs/frontend.md +++ b/specs/frontend.md @@ -136,6 +136,8 @@ The Settings finder and navigation are registry-backed, hide admin-only destinat Existing frontend coverage is a mix of Node-executed helper tests, `.mjs` tests, static DOM/CSS/source-shape tests, browser exploration specs, and app/static tests. Many tests are useful source-shape regressions but do not replace browser/module-graph execution. +`tests/test_css_computed_style_snapshot.py` pins `getComputedStyle` for a fixed element inventory across pages, viewports, themes and density modes, so a `static/style.css` restructuring that changes which declaration wins fails a test instead of shipping; see `tests/css_snapshot/README.md` for what it does and does not cover. + Recent focused coverage includes model-key matching under Node, document-library counters, chat resend/delete/mobile Enter/ArrowUp, scoped approval continuation and compare routing, route provenance, live-thinking throttling, startup shell/history hydration, shared app-config caching/invalidation, settings registry/navigation/finder/lifecycle, lazy panel loading/offline editor precache, vendored lazy KaTeX/Mermaid rendering, email read dedup/prewarm, Markdown restoration, malformed keybinds, currency-safe inline math, notes/calendar/modal/manifest/admin-log behavior, Markdown XSS helpers, and CardDAV unchanged-password handling. Missing coverage includes: @@ -143,7 +145,7 @@ Missing coverage includes: - SPA route/static auth and no-cache headers; - CSP header contents and nonce injection for `/` and `/login`; - service-worker API/non-GET bypass and cache strategy; -- service-worker precache versus `index.html` script/module tags, including query strings; +- service-worker precache versus `index.html` script/module tags, including query strings (stylesheet links and their `?v=` strings are covered by `tests/test_static_stylesheet_manifest.py`; script and module tags are not); - ongoing manifest/icon reference drift; - module graph/load-order validation; - degraded vendor-library/browser API behavior, including Pyodide's remaining CDN path. @@ -151,8 +153,8 @@ Missing coverage includes: ## Current Gaps - `static/style.css` and large coordinators remain high-risk owners: `static/js/document.js`, `static/js/settings.js`, `static/js/chat.js`, and `static/app.js`. -- There is no build-time type checking, module graph validation, script-order validation, or service-worker precache validation. +- There is no build-time type checking, module graph validation, or script-order validation. Service-worker precache validation exists for stylesheets only. - Frontend state is mostly module/global/localStorage driven, so cross-session and cross-user behavior needs explicit care. - `window.*` compatibility bridges remain widespread. - PWA/static-serving behavior may deserve a separate spec if service worker, manifests, route-specific icons, and cache policy keep growing. -- A static asset/route manifest regression should verify files referenced by `index.html`, `manifest.json`, `sw.js`, and app-owned HTML routes actually exist. +- The static asset/route manifest regression covers stylesheets referenced by app-owned HTML and `sw.js`; scripts, modules and `manifest.json` icon references are still unverified. diff --git a/tests/README.md b/tests/README.md index 83e2c10c2..770cd4c49 100644 --- a/tests/README.md +++ b/tests/README.md @@ -137,6 +137,25 @@ The runner propagates pytest's exit code, so it composes with normal local workflows; "report-only" means it is not a CI gate, not that failures are swallowed. +## CSS computed-style snapshot + +`tests/test_css_computed_style_snapshot.py` pins the rendered result of +`static/style.css` - one 51k-line file whose behavior depends on source order - +by hashing `getComputedStyle` over a fixed element inventory across pages, +viewports, themes and density modes. Any PR that moves CSS has to produce an +identical digest or explain why it did not. + +```bash +./venv/bin/python -m pytest tests/test_css_computed_style_snapshot.py +./venv/bin/python scripts/css_snapshot.py --check # standalone, no pytest +./venv/bin/python scripts/css_snapshot.py --write-baseline # re-record, deliberately +``` + +The inventory, the baseline and the capture live in `tests/css_snapshot/`; +`tests/css_snapshot/README.md` documents what is covered, what is deliberately +not, and how to find the property that moved when it fails. The run takes about +21 seconds and skips when `npm ci` has not been run. + ## Core principles - Keep PRs small and homogeneous: one kind of change per PR. diff --git a/tests/css_snapshot/README.md b/tests/css_snapshot/README.md new file mode 100644 index 000000000..c4691dc7c --- /dev/null +++ b/tests/css_snapshot/README.md @@ -0,0 +1,129 @@ +# Computed-style snapshot harness + +`static/style.css` is 51,425 lines in one file. Hundreds of selectors are +declared more than once and `!important` appears throughout, so the rendered +result is a function of **source order**. Extracting a block into its own file, +reordering `` tags, or moving an `@media` rule can silently change which +declaration wins, and nothing else in the suite would notice. + +This harness makes that falsifiable. It captures `getComputedStyle` over a +fixed element inventory, hashes the result, and compares it to a committed +baseline. It moves no CSS itself. + +## What it covers + +| Dimension | Values | +|---|---| +| Pages | `static/index.html` (app shell, 76 elements), `static/login.html` (14), the bench (586 selectors) | +| Viewports | 1440x900, 820x1000, 768x1024 (touch), 390x844 (touch) | +| Themes | dark (default) and `:root.light` | +| Density | default, `:root.density-compact`, `:root.density-spacious` | +| Properties | 122 pinned properties per element, plus every custom property on `:root` and `body` | + +That is 676 elements x 24 variants = 16,224 element snapshots per run, in +about 21 seconds. + +The **app shell** page measures real elements in the markup the server sends, +including modals - each one revealed on its own and re-hidden straight after, +so the measurements stay independent. + +The **bench** page measures one synthesised element per selector, built from +the selector itself. Its selector list is evidence-driven: every selector +declared **more than once** in `style.css` that can be expressed as a static +compound chain (551 of them), plus a curated set covering chat, documents, +email, notes, calendar, settings, cookbook and gallery. Redeclared selectors +are the ones a reorder can actually flip, so they are the ones worth benching. +A bench element pins the cascade for that class combination; it does not pin +the markup that the JS produces. + +Selectors the bench grammar cannot express are the gap: selector lists +(`a, b`), pseudo-elements, pseudo-classes, `:not()` and `:has()`. They are +skipped rather than approximated. + +## Files + +| File | Role | +|---|---| +| `inventory.json` | The fixed inventory: properties, variants, pages, elements, bench selectors | +| `baseline.json` | The committed digest plus per-element and per-variant hashes | +| `capture.mjs` | Playwright capture; raw values on stdout | +| `bench.html` | Empty page that loads the stylesheet; the capture mounts bench nodes into it | +| `../test_css_computed_style_snapshot.py` | The regression test | +| `../../scripts/css_snapshot.py` | Hashing, comparison, and the CLI | + +## Running it + +```bash +./venv/bin/python -m pytest tests/test_css_computed_style_snapshot.py +./venv/bin/python scripts/css_snapshot.py --check # same comparison, standalone +./venv/bin/python scripts/css_snapshot.py --write-baseline # re-record +``` + +The CLI serves the repository on an ephemeral port itself, so it does not need +pytest. Under pytest the session static server is reused through +`ODYSSEUS_TEST_STATIC_ORIGIN`. + +`npm ci` is required: the capture drives Playwright's Chromium. Without it the +browser tests skip. + +## When the test fails + +The failure names the elements and the variants whose hashes moved. To see +which *property* moved, capture both sides and diff: + +```bash +./venv/bin/python scripts/css_snapshot.py --dump after.json +git stash && ./venv/bin/python scripts/css_snapshot.py --dump before.json && git stash pop +diff <(python -m json.tool before.json) <(python -m json.tool after.json) +``` + +Re-record the baseline only when the change in rendered style is **intended** +and reviewed. On a mechanical CSS extraction it never should be: an extraction +that preserves order produces an identical digest, and one that does not has +changed the UI. + +## Determinism + +The digest is only worth having if an unchanged stylesheet always produces the +same bytes, so the capture: + +- strips every `