test: stabilize final PR 40 validation gates

This commit is contained in:
Alexandre Teixeira
2026-10-01 06:39:22 +01:00
parent f74a262f73
commit 5dfe1c353c
10 changed files with 424 additions and 146 deletions
+6 -1
View File
@@ -97,7 +97,8 @@ def playwright_available(node="node", cwd=ROOT):
def capture(origin, inventory=None, *, swap_rule=None, variants=None,
node="node", cwd=ROOT, timeout=CAPTURE_TIMEOUT_SECONDS):
measurement_delay_ms=0, node="node", cwd=ROOT,
timeout=CAPTURE_TIMEOUT_SECONDS):
"""Drive the browser capture and return ``{"snapshot": ..., "missing": ...}``.
``swap_rule`` swaps the first two top-level declarations of one selector
@@ -107,6 +108,9 @@ def capture(origin, inventory=None, *, swap_rule=None, variants=None,
``variants`` restricts the run to the named variants, for a faster
focused capture.
``measurement_delay_ms`` perturbs the capture timing for the determinism
self-test; elapsed wall time must not change an idle-state snapshot.
"""
inventory = inventory or load_inventory()
selected = inventory["variants"]
@@ -122,6 +126,7 @@ def capture(origin, inventory=None, *, swap_rule=None, variants=None,
"variants": selected,
"pages": inventory["pages"],
"swapRule": swap_rule,
"measurementDelayMs": measurement_delay_ms,
}
result = subprocess.run(
[node, str(CAPTURE_SCRIPT)],
+29 -5
View File
@@ -92,6 +92,20 @@ same bytes, so the capture:
- injects the theme and density classes into `<html>` *before* first paint
rather than toggling them afterwards, so no CSS transition is ever
mid-interpolation while `getComputedStyle` runs;
- removes `autofocus` before parsing: focus states are outside this inventory,
and the browser's asynchronous autofocus step otherwise races the capture;
- pauses CSS animations at time zero and finishes CSS transitions before each
measurement, including newly revealed modals and newly mounted bench nodes.
Animation and transition declarations are still captured; the harness does
not inject `animation: none` or `transition: none`;
- pins Chromium's standard font preference to `Times New Roman` via CDP,
without overriding any author declaration;
- canonicalizes only the `BlinkMacSystemFont` family token to `"system-ui"`,
the spelling Chromium uses for that alias on macOS. Other family names and
their order remain significant;
- measures the `custom-system-prompt` element's `max-height` in `lh`, as opted
into by its inventory entry. Its authored `30lh` resolves to different pixel
heights with different fallback fonts; the line count remains significant;
- aborts images, fonts and media, which cost time and change nothing in the
pinned property set;
- hides scrollbars, so a platform's scrollbar width cannot change the width
@@ -111,11 +125,11 @@ same bytes, so the capture:
- **JS-applied classes.** State the app adds at runtime (collapsed sidebar,
open panels, active tabs) is not represented beyond what the served markup
and the bench selectors already carry.
- **Cross-platform equality has not been measured.** The baseline was recorded
on macOS. The self-hosted Fira Code face means text metrics should not differ
from CI's Linux Chromium, and the layout-derived properties are excluded, but
until a Linux run confirms it, treat a CI-only drift as a possible harness
artifact and diff the dumps before assuming the CSS moved.
- **Browser upgrades and additional platforms.** The original macOS baseline
and Linux captures were compared property by property through exact hash
recovery; the proven platform differences are now controlled above. A new
capture on macOS has not been performed. New browser serialization changes
still need investigation rather than automatic baseline regeneration.
- **The stylesheet is only one of the inputs.** `static/login.html` styles
itself from an inline `<style>` block; it is in the inventory so the
hand-mirrored token values there are pinned too.
@@ -127,3 +141,13 @@ Add an entry to `inventory.json` - an `{key, selector}` object under a page's
include custom properties), or a selector string under `bench` - then
re-record the baseline. `test_baseline_covers_every_inventory_entry` fails if
the two go out of sync.
`test_capture_is_independent_of_elapsed_time_and_font_metrics` perturbs capture
timing and font metrics, checks autofocus suppression, and proves that animation
keyframes, metadata and relative line counts still affect measurements. The
existing cascade-order self-test still detects a reordered declaration.
The PR #40 canonical baseline and its exact justification are documented in
[`pr40-validation.md`](pr40-validation.md). All 122 properties and 676 inventory
elements remain covered; only one element/property opts into line-relative
measurement.
+84 -84
View File
@@ -1,5 +1,5 @@
{
"digest": "e8ef2a81ebd7a4ab",
"digest": "9868d50a542b7ad1",
"elements": {
"app-shell": {
"app-loader": "f04ad5bf6312d53f",
@@ -12,7 +12,7 @@
"cookbook-modal": "918b17b2e2a311dc",
"cookbook-modal-content": "73700bbe8be47774",
"custom-preset-modal": "9345624780009d14",
"custom-system-prompt": "ef839a436cd42975",
"custom-system-prompt": "3ebe8cb381e35746",
"export-dl-btn": "d3fe28e931b66974",
"export-dropdown-item": "d90f6f513d06a6f7",
"export-dropdown-menu": "0793e06a6aae6a7a",
@@ -29,7 +29,7 @@
"memory-search-input": "8b0fbb33401750d4",
"memory-toolbar-btn": "9f985efa7c896f90",
"message-ghost": "ce9144a7705ee903",
"message-textarea": "ae703a5c12022f08",
"message-textarea": "6a0ce314ee2fb9cf",
"mobile-backdrop": "b2ed3a84835d4916",
"mode-toggle-active": "9dfc0fe8866d86c2",
"mode-toggle-idle": "4a221dded6bec4d6",
@@ -40,9 +40,9 @@
"overflow-menu-item": "1a7cad32047dc976",
"pinned-tools-bar": "bb0f88ce29ae7637",
"rail-new-chat": "f8f1c2f6be426ab9",
"reasoning-effort-btn": "800d3d3edff6bf85",
"reasoning-effort-btn": "2a07cdd8ec2b4d26",
"rename-session-modal": "9345624780009d14",
"root": "e3897afe51e821cc",
"root": "820190c144fda419",
"save-custom-preset": "443da70ef5fe9b88",
"scroll-bottom-btn": "24602d72e2949991",
"search-input": "53b94f3e5fdb384d",
@@ -75,7 +75,7 @@
"tool-indicator": "0dee45eaca8b703a",
"user-bar-avatar": "b989bbab5aab3988",
"user-bar-settings": "2d91a65e0ebdd049",
"welcome-screen": "a1dc93c81a7d1899",
"welcome-screen": "7024d434f65886c9",
"welcome-sub": "89cdafbb95b7c845",
"welcome-tip": "875045a9d2ec1055"
},
@@ -289,7 +289,7 @@
".doc-rich-slash-menu": "2e31f63219c116aa",
".doc-save-button[data-save-state=\"saving\"] .doc-save-state-saving": "84ecc420b557e35d",
".doc-selection-clear": "b0d9428152a09e39",
".doc-suggestion-card": "b1db7167c4231257",
".doc-suggestion-card": "1b4f2e6dc6aa2ec6",
".doc-suggestion-close": "8fbd4d5f6ba460b6",
".doc-suggestion-nav-btn": "befb117a2fd3fa7d",
".doc-tab": "8f563b5e946c20be",
@@ -434,8 +434,8 @@
".ge-layer-lock-menu": "528deb6e31022e89",
".ge-layer-lock-option": "38a3da4a8785330a",
".ge-layers-grab": "094e764489b1b069",
".ge-layers-header": "03ee9ca6c3fc912b",
".ge-layers-list": "18de5e01e079682f",
".ge-layers-header": "459d17627f2e83e6",
".ge-layers-list": "0b4bfbd17d414d6d",
".ge-main-canvas": "274650d62945ac12",
".ge-mask-sub-item": "546f03f9c46eb507",
".ge-right-panel": "3d0aed961f2df331",
@@ -574,7 +574,7 @@
".preset-btn.active": "2fc1631701e917c2",
".preset-range": "6c05bf7f8a880758",
".private-browser-preview-frame img": "b1866aa1b67e4626",
".reasoning-effort-prefix": "51c4c138641d61e2",
".reasoning-effort-prefix": "bb0f88ce29ae7637",
".recipient-chip": "f88380e8f4c09cc5",
".recipient-chips": "83fad85374d3c475",
".recording-content": "d5e0c053ceaf4f29",
@@ -676,92 +676,92 @@
"pw-toggle": "9e5f0475eb7b6c57",
"remember-dot": "62b8fa66ceaf590e",
"remember-toggle": "87bed84dfe9061e8",
"root": "07533d1e7958a57a",
"root": "6f8529bca11d0e80",
"setup-note": "028127c742bf5f87",
"submit": "5e371dbe68196bfd",
"toggle-link": "289ae7be0ef6e564",
"username": "a8d417a9a08b204c",
"username": "42f0c35485f59dea",
"version-label": "cc443cc7790ff3b4"
}
},
"variants": {
"app-shell": {
"desktop-dark-comfortable": "21afac735b96f31e",
"desktop-dark-compact": "c1690c3291f79cef",
"desktop-dark-spacious": "7dbd8d97ad28a540",
"desktop-light-comfortable": "9e88a1c7a8dfc1d2",
"desktop-light-compact": "4c882ebf427c8ff7",
"desktop-light-spacious": "f74cff517a61dad9",
"laptop-dark-comfortable": "95c85f14bfde85e3",
"laptop-dark-compact": "d7c8522e5f70cc07",
"laptop-dark-spacious": "b7245ea652815c27",
"laptop-light-comfortable": "b34f4282a5ef51b7",
"laptop-light-compact": "38887a9d17175a65",
"laptop-light-spacious": "8f3ad89ae046dfe9",
"phone-dark-comfortable": "1a8c2a51ed18c0cc",
"phone-dark-compact": "2a11e3b08826edef",
"phone-dark-spacious": "1a8c2a51ed18c0cc",
"phone-light-comfortable": "5b0af830cb78676b",
"phone-light-compact": "3ffe9d863f213a49",
"phone-light-spacious": "5b0af830cb78676b",
"tablet-dark-comfortable": "f2ddb9a92ac7c0f6",
"tablet-dark-compact": "8ef6c7958cabb3ad",
"tablet-dark-spacious": "f2ddb9a92ac7c0f6",
"tablet-light-comfortable": "2c846ae8e56f881b",
"tablet-light-compact": "f4bfea7dbcd7df5f",
"tablet-light-spacious": "2c846ae8e56f881b"
"desktop-dark-comfortable": "65b23a35b5a34126",
"desktop-dark-compact": "1bac3a0aa4467555",
"desktop-dark-spacious": "944583b1854fecbb",
"desktop-light-comfortable": "d279432f8c120b58",
"desktop-light-compact": "aa768c34ba3c7abc",
"desktop-light-spacious": "b74f53e95bfa1d82",
"laptop-dark-comfortable": "d98d16f3d60ed275",
"laptop-dark-compact": "c3eaf3c4278a0120",
"laptop-dark-spacious": "a085fe0f7d9eb66c",
"laptop-light-comfortable": "0d7c64f324bebe87",
"laptop-light-compact": "da50bf03b0f1be58",
"laptop-light-spacious": "a6789fd05a66bf5a",
"phone-dark-comfortable": "21b94f88da549eb5",
"phone-dark-compact": "b9aa7def2b85414e",
"phone-dark-spacious": "21b94f88da549eb5",
"phone-light-comfortable": "4a0cf45db775ea7d",
"phone-light-compact": "f7fbcb94d99c8730",
"phone-light-spacious": "4a0cf45db775ea7d",
"tablet-dark-comfortable": "e18256ce58398f11",
"tablet-dark-compact": "3cf3e52a10313d44",
"tablet-dark-spacious": "e18256ce58398f11",
"tablet-light-comfortable": "c2a45155c391f26c",
"tablet-light-compact": "9fa7dc03dd02b723",
"tablet-light-spacious": "c2a45155c391f26c"
},
"bench": {
"desktop-dark-comfortable": "2471a82979dfd8d8",
"desktop-dark-compact": "b2807c0c54345c3e",
"desktop-dark-spacious": "41a6d588192a02ed",
"desktop-light-comfortable": "631778e05aca46ef",
"desktop-light-compact": "340c1bbce11d0b32",
"desktop-light-spacious": "e21051317e9d346d",
"laptop-dark-comfortable": "4c931615f7151fc3",
"laptop-dark-compact": "181a9bb9f0add9a0",
"laptop-dark-spacious": "ad24cee14e456d04",
"laptop-light-comfortable": "d635b3b801ff803d",
"laptop-light-compact": "dbe6446fad9cc2f0",
"laptop-light-spacious": "c81a4f935d0270a3",
"phone-dark-comfortable": "ae0aa5695982a188",
"phone-dark-compact": "8ca1949b9817e3c4",
"phone-dark-spacious": "912fe8e2d490162f",
"phone-light-comfortable": "ccd790243858a150",
"phone-light-compact": "2b008b092bec6fb1",
"phone-light-spacious": "c2465ac50f71a035",
"tablet-dark-comfortable": "1d96addb759bc3e3",
"tablet-dark-compact": "53294ba127a61959",
"tablet-dark-spacious": "c26663bc163aa446",
"tablet-light-comfortable": "b66904a1f69417b7",
"tablet-light-compact": "9f0836a37f087c2e",
"tablet-light-spacious": "645c09fa0e1ddb59"
"desktop-dark-comfortable": "f9fa9d54db5ad45a",
"desktop-dark-compact": "7603a804cb2e23a2",
"desktop-dark-spacious": "b0c69306349fc078",
"desktop-light-comfortable": "d3e98c4c1e90686f",
"desktop-light-compact": "a04ae70088bc8b35",
"desktop-light-spacious": "4aa01a125f541429",
"laptop-dark-comfortable": "025e60ebe84e0f98",
"laptop-dark-compact": "53d6555bb7a6b5bb",
"laptop-dark-spacious": "c124d9ce1633d027",
"laptop-light-comfortable": "e8328c45f5127bb0",
"laptop-light-compact": "dc2da3534534c888",
"laptop-light-spacious": "abe5483724c987e0",
"phone-dark-comfortable": "67ca3c78e6b9495a",
"phone-dark-compact": "ad6f3a9f5c8d81cd",
"phone-dark-spacious": "1e6f66e56519ba5d",
"phone-light-comfortable": "ef876073cc2dc45e",
"phone-light-compact": "2ef6f5dc7a646be9",
"phone-light-spacious": "b0d4cd9bd0361966",
"tablet-dark-comfortable": "58922240b216e786",
"tablet-dark-compact": "3991c22f99a394e0",
"tablet-dark-spacious": "9848c3b7debac974",
"tablet-light-comfortable": "12dfbad36172de20",
"tablet-light-compact": "80d4843e622f1353",
"tablet-light-spacious": "c420b8d46c991a44"
},
"login": {
"desktop-dark-comfortable": "d3d0512f5223397e",
"desktop-dark-compact": "d3d0512f5223397e",
"desktop-dark-spacious": "d3d0512f5223397e",
"desktop-light-comfortable": "d3d0512f5223397e",
"desktop-light-compact": "d3d0512f5223397e",
"desktop-light-spacious": "d3d0512f5223397e",
"laptop-dark-comfortable": "d614fc150b017e6e",
"laptop-dark-compact": "d614fc150b017e6e",
"laptop-dark-spacious": "d614fc150b017e6e",
"laptop-light-comfortable": "d614fc150b017e6e",
"laptop-light-compact": "d614fc150b017e6e",
"laptop-light-spacious": "d614fc150b017e6e",
"phone-dark-comfortable": "c7f59e5d8c979f02",
"phone-dark-compact": "c7f59e5d8c979f02",
"phone-dark-spacious": "c7f59e5d8c979f02",
"phone-light-comfortable": "c7f59e5d8c979f02",
"phone-light-compact": "c7f59e5d8c979f02",
"phone-light-spacious": "c7f59e5d8c979f02",
"tablet-dark-comfortable": "ee0918e00caa6875",
"tablet-dark-compact": "ee0918e00caa6875",
"tablet-dark-spacious": "ee0918e00caa6875",
"tablet-light-comfortable": "ee0918e00caa6875",
"tablet-light-compact": "ee0918e00caa6875",
"tablet-light-spacious": "ee0918e00caa6875"
"desktop-dark-comfortable": "3cf4809db0f8e119",
"desktop-dark-compact": "3cf4809db0f8e119",
"desktop-dark-spacious": "3cf4809db0f8e119",
"desktop-light-comfortable": "3cf4809db0f8e119",
"desktop-light-compact": "3cf4809db0f8e119",
"desktop-light-spacious": "3cf4809db0f8e119",
"laptop-dark-comfortable": "1e6b6522bcb4a942",
"laptop-dark-compact": "1e6b6522bcb4a942",
"laptop-dark-spacious": "1e6b6522bcb4a942",
"laptop-light-comfortable": "1e6b6522bcb4a942",
"laptop-light-compact": "1e6b6522bcb4a942",
"laptop-light-spacious": "1e6b6522bcb4a942",
"phone-dark-comfortable": "e4b7998ae024fdaf",
"phone-dark-compact": "e4b7998ae024fdaf",
"phone-dark-spacious": "e4b7998ae024fdaf",
"phone-light-comfortable": "e4b7998ae024fdaf",
"phone-light-compact": "e4b7998ae024fdaf",
"phone-light-spacious": "e4b7998ae024fdaf",
"tablet-dark-comfortable": "321622b649b2f151",
"tablet-dark-compact": "321622b649b2f151",
"tablet-dark-spacious": "321622b649b2f151",
"tablet-light-comfortable": "321622b649b2f151",
"tablet-light-compact": "321622b649b2f151",
"tablet-light-spacious": "321622b649b2f151"
}
}
}
+49 -3
View File
@@ -103,10 +103,47 @@ function pageMeasure(job) {
return { root, leaf };
}
function readStyle(el, pseudo, properties, wantCustom) {
function readStyle(el, pseudo, properties, wantCustom, lineRelativeProperties = []) {
// Style/layout is flushed when querying animations. Sample CSS animations
// at the start, and settle transitions to their destination. WAAPI timing
// overrides leave animation/transition declarations in getComputedStyle.
for (const animation of document.getAnimations()) {
if (animation instanceof CSSTransition) animation.finish();
else {
animation.pause();
animation.currentTime = 0;
}
}
const cs = getComputedStyle(el, pseudo || undefined);
const values = {};
for (const prop of properties) values[prop] = cs.getPropertyValue(prop);
for (const prop of properties) {
let value = cs.getPropertyValue(prop);
// Chromium on macOS serializes this alias as a quoted system-ui family;
// Linux preserves the alias spelling. Keep every other family and order.
if (prop === 'font-family') {
value = value.replace(/(^|,\s*)BlinkMacSystemFont(?=\s*(?:,|$))/g, '$1"system-ui"');
}
values[prop] = value;
}
if (lineRelativeProperties.length) {
// `normal` line-height uses the installed fallback font's metrics. Measure
// one lh with this element's font, then retain the authored line count
// rather than the platform's pixel height. Only inventory opt-ins use it.
const ruler = document.createElement('div');
ruler.style.cssText = 'all:initial;position:absolute;left:-10000px;height:1lh;';
for (const prop of ['font-family', 'font-size', 'font-weight', 'font-style',
'font-stretch', 'font-variant', 'line-height']) {
ruler.style.setProperty(prop, cs.getPropertyValue(prop));
}
document.body.appendChild(ruler);
const lineHeight = parseFloat(getComputedStyle(ruler).height);
ruler.remove();
for (const prop of lineRelativeProperties) {
if (values[prop]?.endsWith('px')) {
values[prop] = `${Number((parseFloat(values[prop]) / lineHeight).toFixed(6))}lh`;
}
}
}
if (wantCustom) {
const names = [];
for (let i = 0; i < cs.length; i += 1) {
@@ -151,7 +188,8 @@ function pageMeasure(job) {
// Reading a layout property forces the style and layout pass before the
// computed values are read back.
void document.body.offsetHeight;
measured[entry.key] = readStyle(el, entry.pseudo, job.properties, !!entry.custom);
measured[entry.key] = readStyle(el, entry.pseudo, job.properties, !!entry.custom,
entry.lineRelativeProperties || []);
if (restore) restore();
}
@@ -211,6 +249,10 @@ async function main() {
javaScriptEnabled: true,
});
const tab = await context.newPage();
const cdp = await context.newCDPSession(tab);
// Pin the UA standard font preference rather than overriding author
// CSS. macOS defaults to Times; Linux defaults to Times New Roman.
await cdp.send('Page.setFontFamilies', { fontFamilies: { standard: 'Times New Roman' } });
// Registered first so the document/stylesheet handlers below win:
// Playwright matches the most recently registered route.
@@ -239,6 +281,9 @@ async function main() {
const response = await route.fetch();
let html = await response.text();
html = html.replace(/<script\b[^>]*>[\s\S]*?<\/script>/gi, '');
// Focus states are outside this idle-state inventory. Autofocus can
// run after load, racing the measurement and changing outline-offset.
html = html.replace(/(<[^>]*?)\sautofocus(?=[\s=>])(?:\s*=\s*(?:"[^"]*"|'[^']*'|[^\s>]+))?/gi, '$1');
if (shippedStylesheets !== null) {
html = html.replace(/<link\b[^>]*rel=["']stylesheet["'][^>]*>/gi, '');
html = html.replace(/<\/head>/i, ` ${shippedStylesheets}\n</head>`);
@@ -253,6 +298,7 @@ async function main() {
if (!response || !response.ok()) {
throw new Error(`${page.url} returned ${response ? response.status() : 'no response'}`);
}
if (job.measurementDelayMs) await tab.waitForTimeout(job.measurementDelayMs);
const result = await tab.evaluate(pageMeasure, {
elements: page.elements || [],
bench: page.bench || [],
+1
View File
@@ -660,6 +660,7 @@
{
"key": "custom-system-prompt",
"selector": "#custom-system-prompt",
"lineRelativeProperties": ["max-height"],
"unhide": true
},
{
+190
View File
@@ -0,0 +1,190 @@
# PR #40 final harness validation
Starting branch: `review/pr40-final-lab-validation`.
Starting HEAD: `6f14439e4179b8ad674660e59ca0bea96f1a6ead` (clean).
This cleanup changes snapshot tooling, test harnesses, baselines and documentation.
It changes no application production implementation.
## Reproduction and root cause
Both the computed-style baseline failure and the obsolete markdown VM loader
failure reproduced on the reconciled branch and preserved lab source at
`fff55a786c5c40daebf64d1a12b576d6c889b1a3`. The snapshot mismatch also reproduced
on the original recording revision, `73b1726d`. No canonical lab checkout was used.
Chromium: `151.0.7922.34`; Node: `22.23.1`; Python: `3.12.3`.
All 42 pre-existing Linux/macOS element-hash mismatches were explained by exact
recovery of the committed hashes from raw captures:
| Elements | Proven difference |
|---|---|
| Two document roots | Default font-family `Times` on macOS versus `"Times New Roman"` on Linux |
| 39 elements | `BlinkMacSystemFont` serializes as `"system-ui"` on macOS |
| `custom-system-prompt` | The same alias difference, plus authored `30lh` resolving to `480px` on macOS and `450px` on Linux |
Independent repeated unmodified captures also changed the message textarea outline
offset from `0px` to `2px`: asynchronous autofocus races measurement. Animations
also depend on elapsed time; an idle cascade inventory needs a defined sample time.
## Canonical snapshot contract
- Pin the UA standard font preference through CDP; author declarations still win.
- Normalize only the proven unquoted BlinkMacSystemFont family token; preserve all other families and order.
- Remove autofocus before parsing; focus states were already outside this inventory.
- Pause CSS animations at time zero and finish transitions; keep their computed declarations.
- Opt only `custom-system-prompt/max-height` into line-relative measurement; the `30lh` line count remains significant.
- Preserve all 122 properties, 676 elements and 24 variants.
Two full controlled captures, separated by a 150ms measurement delay on every
page/variant, were byte-identical. Their canonical digest is
`9868d50a542b7ad1`. There were no missing inventory entries. The controlled
lab/merged comparison still differs on exactly five intended PR #40 elements:
| Element | Intended reconciled change |
|---|---|
| app-shell/reasoning-effort-btn | Moved control to the Chat Context home; visibility changes |
| bench/.reasoning-effort-prefix | Removed old narrow composer hiding rule |
| bench/.doc-suggestion-card | Responsive overflow, display and minimum sizing |
| bench/.ge-layers-header | Wrapping layer controls |
| bench/.ge-layers-list | Minimum height and bottom padding |
Only 11 element hashes change from the old committed baseline: these five, the
two pinned root fonts, the two autofocus controls (message and login username),
welcome-screen at the defined animation start, and the line-relative textarea cap.
The other 665 element hashes are unchanged. The new baseline was written from the
proven identical full captures, not from an unexplained failing local capture.
The new browser self-test perturbs timing and font metrics, checks autofocus
suppression, retains animation/transition metadata, distinguishes 30lh from 31lh,
and detects changed animation keyframes. The existing cascade-order test still
detects reordering conflicting declarations.
## Markdown and environment reference
The standalone codefence script previously stripped imports/exports with regexes
and evaluated the result as a classic VM script. Its ui.js pattern missed the
versioned ES-module import. It now uses the existing streaming markdown ESM loader
and keeps its original regression assertions. A Python wrapper gates it in normal pytest.
No production markdown.js change was needed.
The env-reference suite passed before edits (13 tests). Adding capture timing
support moved the snapshot tooling environment read from line 249 to 254. The
generator was rerun only after the resulting reference mismatch was proven; its
diff changes exactly that one location.
The current fetched-page contract was verified before edits: the provider-facing
schema omits top-level anyOf for provider compatibility; compact preview validation
requires url or urls. `test_fetch_requires_a_page_in_full_and_compact_contracts` and
its whole targeted module passed (10 tests).
## Validation
Tests use a fresh isolated virtual environment installed from `requirements.txt`:
`/tmp/pr40-final-validation-venv`. Dependency consistency: all 107 packages compatible.
Optional PyMuPDF and openpyxl remain absent. Their attachment skips are explicit
importorskip contracts; PyMuPDF and spreadsheet extraction extras are optional in
requirements-optional.txt. Live endpoint tests require explicit opt-in fixtures.
- Required snapshot/CSS/markdown/env/schema/attachment gates: **138 passed, 3 intentional skips**.
- Additional CSS and streaming Python gates: **11 passed**.
- Reconstructed runtime suite: **3543 passed, 32 intentional skips, 2 expected xfails**, 173 files.
- Runtime selection: changed reconciliation test modules, tests matching changed production stems, turn contract/tool/browser/runtime/markdown/CSS suites and model-tool-mode, preview recovery, form roundtrip and env-reference gates.
- Expected xfails: two pre-existing negative-web-wording cases in test_runtime_behavior_regressions.py; their markers document partially detected negative instructions on lab.
- Python compileall: 1680 tracked Python files passed.
- Node syntax checks: 279 tracked .js files and 82 tracked .mjs files passed.
- git diff --check passed; conflict-marker scan found no tracked files with conflict markers.
## Every executable tests .mjs gate
Invoked with `node --experimental-vm-modules`; browser fixtures route locally or
use isolated DOMs. The live email UI fixture was not supplied.
| File | Result |
|---|---|
| `tests/backgroundToolJobs.test.mjs` | PASS |
| `tests/chatEditorProgress.test.mjs` | PASS |
| `tests/chatImageDeletion.test.mjs` | PASS |
| `tests/chatProcessingHandoff.test.mjs` | PASS |
| `tests/documentSelectionCaret.mjs` | PASS |
| `tests/editor-ai-cancel.mjs` | PASS |
| `tests/editor-layer-styles.mjs` | PASS |
| `tests/editorRichUpdate.mjs` | PASS |
| `tests/editorSuggestionApply.mjs` | PASS |
| `tests/editorSuggestionButtons.mjs` | PASS |
| `tests/emailReplyBrowser.test.mjs` | PASS; skipped 1 (opt-in UI fixture absent) |
| `tests/emailReplyStream.test.mjs` | PASS |
| `tests/generatedImageResult.test.mjs` | PASS |
| `tests/historyResumeRendering.test.mjs` | PASS |
| `tests/live_thinking_scheduler.test.mjs` | PASS |
| `tests/markdown_codefence_placeholder_regression.mjs` | PASS |
| `tests/noteTestOracle.test.mjs` | PASS |
| `tests/notesDraftAutosave.test.mjs` | PASS |
| `tests/researchMobileButtons.test.mjs` | PASS |
| `tests/schemaThinkingProbe.test.mjs` | PASS |
| `tests/sidebarNewChat.test.mjs` | PASS |
| `tests/skillsApproval.test.mjs` | PASS |
| `tests/toolFollowupOracle.test.mjs` | PASS |
| `tests/tool_followup_oracle.test.mjs` | PASS |
| `tests/turnRendering.test.mjs` | PASS |
| `tests/streaming/invariant.test.mjs` | PASS |
| `tests/streaming/segmenter.test.mjs` | PASS |
| `tests/helpers/test_settings_shell_coordinator.mjs` | PASS |
`tests/css_snapshot/capture.mjs` ran repeatedly with valid capture jobs through
the snapshot gate, including all 72 page/variant combinations. Other helpers
(document_source.mjs, stylesheets.mjs, streaming/corpus.mjs and markdownHarness.mjs)
are imported support modules, not standalone gates.
## Remaining limits
A new macOS capture was not available. Cross-platform normalization is backed by
exact old macOS hash recovery and controlled Linux captures, with no broad property
exclusions. Browser upgrades may introduce new serialization differences requiring
fresh investigation. Optional/live fixtures were intentionally skipped. Unused
`.session-run-state` CSS remains follow-up debt; it was not removed.
Full canonical pytest was run after every code/test edit in the fresh requirements environment.
Command: `DATABASE_URL=sqlite:///:memory: /tmp/pr40-final-validation-venv/bin/python -m pytest -q -p no:cacheprovider -rsx`.
Result: **11396 passed, 53 skipped, 2 xfailed, 185 warnings, 6 subtests passed in 451.08s (0:07:31)**.
Declared skip/xfail contracts:
```text
SKIPPED [1] tests/smoke/test_calendar_smoke.py:16: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_chat_smoke.py:20: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_chat_smoke.py:35: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_chat_smoke.py:59: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_compare_smoke.py:39: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_cookbook_smoke.py:18: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_cookbook_smoke.py:31: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_cookbook_smoke.py:41: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_documents_rag_smoke.py:31: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_documents_smoke.py:12: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_email_smoke.py:75: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_memory_smoke.py:19: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_notes_smoke.py:11: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_settings_smoke.py:16: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_tasks_smoke.py:16: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/smoke/test_uploads_smoke.py:11: APP_PORT is not set, so there is no instance to drive. Run the suite with `scripts/odysseus-smoke`, which boots this worktree and exports it.
SKIPPED [1] tests/test_ajax_email_live.py:21: opt-in live Ajax endpoint
SKIPPED [5] tests/test_ajax_email_live.py:55: opt-in live Ajax endpoint
SKIPPED [12] tests/test_ajax_email_live.py:74: opt-in live Ajax endpoint
SKIPPED [2] tests/test_ajax_email_live.py:141: opt-in live Ajax endpoint
SKIPPED [8] tests/test_ajax_email_live.py:175: opt-in live Ajax endpoint
SKIPPED [1] tests/test_cookbook_helpers.py:875: Windows Ollama CLI startup guard
SKIPPED [1] tests/test_email_attachment_text.py:19: could not import 'fitz': No module named 'fitz'
SKIPPED [1] tests/test_email_attachment_text.py:41: could not import 'openpyxl': No module named 'openpyxl'
SKIPPED [1] tests/test_email_attachment_text.py:110: Opt-in Ajax fixture test
SKIPPED [1] tests/test_inspect_media_tool.py:574: needs an ffmpeg built without webp
SKIPPED [1] tests/test_inspect_media_tool.py:934: rsvg-convert required
SKIPPED [1] tests/test_markitdown_runtime.py:64: could not import 'markitdown': No module named 'markitdown'
SKIPPED [1] tests/test_result_reference_followup.py:96: Opt-in live Ajax replay
SKIPPED [1] tests/test_upload_content_detection_magic.py:41: libmagic/python-magic not installed in this environment
XFAIL tests/test_runtime_behavior_regressions.py::test_negative_web_wording_withholds_the_web_tools_unhandled[Summarise what you already know. Do not search the web.] - negative web wording is only partially detected on lab@c499c01b; these phrasings still get the web tools offered
XFAIL tests/test_runtime_behavior_regressions.py::test_negative_web_wording_withholds_the_web_tools_unhandled[No web search please, just tell me what you know about Python decorators.] - negative web wording is only partially detected on lab@c499c01b; these phrasings still get the web tools offered
```
Final commit SHA and clean status are reported in the session completion message.
@@ -1,56 +1,9 @@
import assert from 'node:assert/strict';
import fs from 'node:fs';
import path from 'node:path';
import vm from 'node:vm';
import { fileURLToPath } from 'node:url';
import { loadMarkdown } from './streaming/markdownHarness.mjs';
const __dirname = path.dirname(fileURLToPath(import.meta.url));
const markdownPath = path.join(__dirname, '..', 'static', 'js', 'markdown.js');
let src = fs.readFileSync(markdownPath, 'utf8');
src = src.replace(
/import uiModule from '\.\/ui\.js';/,
'const uiModule = { esc: (s) => String(s).replace(/&/g, "&amp;").replace(/</g, "&lt;").replace(/>/g, "&gt;").replace(/\\"/g, "&quot;") };'
);
src = src.replace(
/import \{ splitTableRow \} from '\.\/markdown\/tableRow\.js';/,
'const splitTableRow = (row) => row.split("|").filter((cell) => cell.trim() !== "");'
);
src = src.replace(
/import \{ replaceEmojiShortcodes, hasEmojiShortcode \} from '\.\/emojiShortcodes\.js';/,
'const hasEmojiShortcode = (t) => !!t && t.indexOf(":") !== -1 && /:[a-z0-9_+-]{1,40}:/i.test(t); const replaceEmojiShortcodes = (t) => t;'
);
src = src.replace(/export function /g, 'function ');
src = src.replace(/export const /g, 'const ');
src = src.replace(/export default markdownModule;?/g, '');
src += '\nthis.__mdToHtml = mdToHtml;';
class MutationObserver {
observe() {}
disconnect() {}
}
const sandbox = {
console,
URL,
MutationObserver,
localStorage: { getItem() { return '[]'; }, setItem() {} },
document: {
body: { classList: { contains() { return true; } } },
addEventListener() {},
querySelectorAll() { return []; },
getElementById() { return null; },
contains() { return true; },
},
window: {
location: { origin: 'http://localhost' },
katex: null,
mermaid: null,
},
};
vm.createContext(sandbox);
vm.runInContext(src, sandbox, { filename: markdownPath });
// Use the same ES-module loader as the streaming renderer tests. It keeps the
// production exports intact and handles the browser's versioned sibling imports.
const { mdToHtml } = await loadMarkdown();
const input = [
'> ```html',
@@ -62,7 +15,7 @@ const input = [
'> ```',
].join('\n');
const html = sandbox.__mdToHtml(input);
const html = mdToHtml(input);
assert.equal(html.includes('___ALLOWED_HTML_'), false, html);
assert.equal(html.includes('appendChild'), true, html);
+51
View File
@@ -88,6 +88,57 @@ def test_computed_styles_match_the_committed_baseline():
)
@_requires_browser
def test_capture_is_independent_of_elapsed_time_and_font_metrics(tmp_path):
page = tmp_path / "fixture.html"
source = """<!doctype html><html><head><style>
@keyframes fade { from { opacity: .25; } to { opacity: .75; } }
.sample { animation: fade .4s linear infinite; transition: color .1s;
font: 13px monospace; max-height: 30lh; }
#other { font: 23px serif; max-height: 31lh; }
textarea { outline-offset: 0; }
textarea:focus { outline-offset: 7px; }
#alias { font-family: Inter, -apple-system, BlinkMacSystemFont, sans-serif; }
#serialized { font-family: Inter, -apple-system, "system-ui", sans-serif; }
</style></head><body>
<textarea id="focus" autofocus></textarea><div class="sample" id="sample"></div>
<div class="sample" id="other"></div><div id="alias"></div><div id="serialized"></div>
</body></html>"""
page.write_text(source, encoding="utf-8")
inventory = {
"properties": ["font-family", "opacity", "max-height", "outline-offset",
"animation-name", "animation-duration", "animation-play-state",
"transition-duration"],
"variants": [snapshot.load_inventory()["variants"][0]],
"pages": [{"name": "fixture", "url": "/fixture.html", "elements": [
{"key": key, "selector": f"#{key}",
"lineRelativeProperties": ["max-height"] if key in {"sample", "other"} else []}
for key in ("focus", "sample", "other", "alias", "serialized")
]}],
}
origin, shutdown = snapshot.serve_repository(tmp_path)
try:
early = snapshot.capture(origin, inventory)["snapshot"]
late = snapshot.capture(origin, inventory, measurement_delay_ms=150)["snapshot"]
assert early == late
values = early["fixture"][inventory["variants"][0]["name"]]
assert values["focus"]["outline-offset"] == "0px"
assert values["sample"]["opacity"] == "0.25"
assert values["sample"]["animation-name"] == "fade"
assert values["sample"]["animation-duration"] == "0.4s"
assert values["sample"]["animation-play-state"] == "running"
assert values["sample"]["transition-duration"] == "0.1s"
assert values["sample"]["max-height"] == "30lh"
assert values["other"]["max-height"] == "31lh"
assert values["alias"]["font-family"] == values["serialized"]["font-family"]
page.write_text(source.replace("opacity: .25", "opacity: .5"), encoding="utf-8")
changed = snapshot.capture(origin, inventory)["snapshot"]
assert changed["fixture"][inventory["variants"][0]["name"]]["sample"]["opacity"] == "0.5"
assert snapshot.summarize(changed)["digest"] != snapshot.summarize(early)["digest"]
finally:
shutdown()
@_requires_browser
def test_reordering_two_conflicting_declarations_moves_the_digest():
"""The harness has to fail when the cascade changes, or it proves nothing.
+8
View File
@@ -18,6 +18,14 @@ def node_available():
pytest.skip("node binary not on PATH")
def test_blockquoted_html_codefence_does_not_leak_placeholders(node_available):
result = subprocess.run(
["node", "tests/markdown_codefence_placeholder_regression.mjs"],
cwd=_REPO, capture_output=True, text=True, timeout=15,
)
assert result.returncode == 0, result.stderr + result.stdout
def _run_markdown_case(markdown: str, render_expr: str = "mod.mdToHtml(input)", with_katex: bool = False):
script = textwrap.dedent(
r"""
+1 -1
View File
@@ -226,7 +226,7 @@ Listed for completeness. Setting one of these on a real install is either a no-o
| `ODYSSEUS_SFT_TRACE_CAPTURE` | `'1'` | `routes/chat_helpers.py:161` (+1 more) | On by default, but only for owners whose name starts with `sft_`. Set 0, false, no or off to stop writing training traces. |
| `ODYSSEUS_SFT_TRACE_DIR` | *unset* | `routes/chat_helpers.py:195` (+2 more) | Directory the SFT trace JSONL files are written to. Defaults to `sft_traces` under the data directory. |
| `ODYSSEUS_SKIP_RUN_HINT` | *unset* | `setup.py:284` | Any non-empty value suppresses the `start the server with` hint at the end of setup. `start-macos.sh` sets it because it starts the server itself. |
| `ODYSSEUS_TEST_STATIC_ORIGIN` | *unset* | `scripts/css_snapshot.py:249` (+6 more) | Origin an already-running static server is serving the repository from, so snapshot tooling reuses it instead of starting its own. |
| `ODYSSEUS_TEST_STATIC_ORIGIN` | *unset* | `scripts/css_snapshot.py:254` (+6 more) | Origin an already-running static server is serving the repository from, so snapshot tooling reuses it instead of starting its own. |
| `ODYSSEUS_TEST_STATIC_PORT` | *unset* | `tests/conftest.py:137` | Fixed port for the test suite's static server. Unset takes an ephemeral port, which is what keeps parallel runs from colliding. |
### Build and release metadata