Merge lab into refactor/split-style-css

This commit is contained in:
Alexandre Teixeira
2026-10-01 00:48:58 +01:00
96 changed files with 2210 additions and 389 deletions
+288
View File
@@ -0,0 +1,288 @@
"""Read the document editor's JavaScript the way the browser loads it.
``static/js/document.js`` is being decomposed. It stays the entry point the
browser requests -- ``static/index.html`` names it, ``static/sw.js`` precaches
it, and five modules import it -- but the implementation moves into modules
under ``static/js/document/``. The implementation set is the entry plus that
directory.
Two habits in the existing tests do not survive that move, and this module
exists to replace both.
**Reading the entry file alone.** A membership assertion against
``document.js`` silently covers less the moment the behaviour it names moves
out. Use :func:`document_source` for those: it is the whole implementation set,
so a test keeps finding what it asserts on wherever the code lands.
**Slicing between two adjacent functions.** ``function_body("a")`` means "the region between a and b", which is only
the body of ``a`` while ``a`` and ``b`` happen to be neighbours in one file.
After a split they may sit in different modules, and then the slice runs to the
end of the concatenation and quietly grows: an ``assert "x" in region`` passes
against code it was never meant to see. Several of these also hard-code the
entry file's two-space indentation (``"\\n function showDocTabMenu"``), which
no extracted module reproduces. Use :func:`function_body` or
:func:`declaration` instead -- they find the construct by name, in whichever
module defines it, and end at its real closing brace.
"""
from __future__ import annotations
import re
from pathlib import Path
_STATIC = Path(__file__).resolve().parents[2] / "static"
_ENTRY = _STATIC / "js" / "document.js"
# Extracted implementation modules get one home, so the set is discoverable
# without a manifest anyone has to remember to update.
_IMPL_DIR = _STATIC / "js" / "document"
def document_source_paths() -> list[Path]:
"""Every file holding document-editor implementation, entry first.
The entry comes first so a concatenation reads in the order the browser
evaluates the graph's root; the rest are sorted for determinism.
"""
if not _ENTRY.is_file():
raise AssertionError(f"document editor entry point is missing: {_ENTRY}")
extracted = sorted(_IMPL_DIR.rglob("*.js")) if _IMPL_DIR.is_dir() else []
return [_ENTRY, *extracted]
def document_source() -> str:
"""The whole implementation set as one string, entry first.
For membership assertions (``assert "..." in document_source()``). For
anything positional use :func:`function_body` or :func:`declaration`.
"""
return "\n".join(p.read_text(encoding="utf-8") for p in document_source_paths())
# --- Locating a construct by name, not by what follows it ------------------
def _defining_source(pattern: re.Pattern[str], what: str) -> tuple[str, int]:
"""The source text that defines ``what``, and the offset of the match."""
hits = []
for path in document_source_paths():
src = path.read_text(encoding="utf-8")
for m in pattern.finditer(src):
hits.append((path, src, m.start()))
if not hits:
raise AssertionError(f"{what} is not defined anywhere in {_describe_set()}")
if len(hits) > 1:
where = ", ".join(
f"{p.relative_to(_STATIC.parent)}:{s.count(chr(10), 0, o) + 1}"
for p, s, o in hits
)
raise AssertionError(f"{what} is defined more than once ({where})")
_path, src, offset = hits[0]
return src, offset
def _describe_set() -> str:
return ", ".join(str(p.relative_to(_STATIC.parent)) for p in document_source_paths())
def function_body(name: str) -> str:
"""The full text of function ``name``, signature through closing brace.
Matches ``function name``, optionally prefixed by ``export`` and/or
``async``, at any indentation, in whichever module of the implementation
set defines it. The end is found by matching braces rather than by naming
whatever declaration follows, so moving the function -- or the one after
it -- does not change the region a test sees.
"""
pattern = re.compile(
r"^[ \t]*(?:export\s+)?(?:async\s+)?function\s+" + re.escape(name) + r"\s*\(",
re.M,
)
src, offset = _defining_source(pattern, f"function {name}")
# Skip the parameter list before looking for the body. A destructured
# parameter -- `function f(table, { headerRow, headerColumn })` -- opens a
# brace that is not the body, and matching it would return the signature
# alone.
body_start = _end_of_params(src, src.index("(", offset))
return src[offset : _end_of_block(src, body_start)]
def declaration(name: str) -> str:
"""The full text of a top-level ``const``/``let``/``var`` named ``name``.
For the array and object tables the tests assert on (toolbar groups, slash
commands, input rules). Ends at the declaration's closing bracket or brace,
or at the end of the statement for a simple initialiser.
"""
pattern = re.compile(
r"^[ \t]*(?:export\s+)?(?:const|let|var)\s+" + re.escape(name) + r"\b",
re.M,
)
src, offset = _defining_source(pattern, f"declaration {name}")
return src[offset : _end_of_statement(src, offset)]
# --- A brace matcher that is not fooled by braces inside literals ----------
#
# `document.js` is full of template literals building DOM, regexes containing
# braces, and apostrophes inside comments. Counting raw `{`/`}` mis-slices on
# all three, so the scan tracks what kind of text it is inside.
# After one of these, `/` starts a regex literal; after a value it is division.
_REGEX_OK_BEFORE = re.compile(r"[({\[,;:=!&|?+\-*~^%<>]\s*$|\b(?:return|typeof|case|in|of|new|delete|void|do|else|yield|await)\s*$")
def _scan(src: str, start: int, stop):
"""Walk ``src`` from ``start``, skipping literals and comments.
Calls ``stop(index, depth_delta_applied)``-free: instead it yields
``(index, char)`` for code positions only, so callers can track nesting.
"""
i, n = start, len(src)
# Stack of template-literal depths: entering `${` pushes brace depth.
template_stack: list[int] = []
while i < n:
c = src[i]
two = src[i : i + 2]
if two == "//":
j = src.find("\n", i)
i = n if j == -1 else j + 1
continue
if two == "/*":
j = src.find("*/", i + 2)
i = n if j == -1 else j + 2
continue
if c in "'\"":
i = _skip_quoted(src, i, c)
continue
if c == "`":
i += 1
i, entered = _skip_template(src, i)
if entered:
template_stack.append(0)
continue
if c == "/" and _REGEX_OK_BEFORE.search(src[max(0, i - 24) : i]):
j = _skip_regex(src, i)
if j is not None:
i = j
continue
if template_stack:
# Inside `${ ... }`: a `}` that closes it returns to template text.
if c == "{":
template_stack[-1] += 1
elif c == "}":
if template_stack[-1] == 0:
template_stack.pop()
i += 1
i, entered = _skip_template(src, i)
if entered:
template_stack.append(0)
continue
template_stack[-1] -= 1
yield i, c
i += 1
def _skip_quoted(src: str, i: int, quote: str) -> int:
i += 1
n = len(src)
while i < n:
if src[i] == "\\":
i += 2
continue
if src[i] == quote:
return i + 1
if src[i] == "\n": # unterminated; do not run away
return i
i += 1
return n
def _skip_template(src: str, i: int) -> tuple[int, bool]:
"""From inside template text, advance to the backtick end or a ``${``.
Returns the new index and whether an interpolation was entered.
"""
n = len(src)
while i < n:
if src[i] == "\\":
i += 2
continue
if src[i] == "`":
return i + 1, False
if src[i : i + 2] == "${":
return i + 2, True
i += 1
return n, False
def _skip_regex(src: str, i: int) -> int | None:
"""Past a regex literal starting at ``i``, or None if it is not one."""
i += 1
n = len(src)
in_class = False
while i < n:
c = src[i]
if c == "\\":
i += 2
continue
if c == "\n":
return None
if in_class:
if c == "]":
in_class = False
elif c == "[":
in_class = True
elif c == "/":
i += 1
while i < n and src[i].isalpha(): # flags
i += 1
return i
i += 1
return None
def _end_of_block(src: str, start: int) -> int:
"""Index just past the ``}`` closing the first ``{`` at or after ``start``."""
depth = 0
seen = False
for i, c in _scan(src, start, None):
if c == "{":
depth += 1
seen = True
elif c == "}":
depth -= 1
if seen and depth == 0:
return i + 1
raise AssertionError(f"unbalanced braces from offset {start}")
def _end_of_statement(src: str, start: int) -> int:
"""Index just past the end of the declaration statement at ``start``.
Ends on the ``;`` or newline that closes it at nesting depth zero, so an
array or object initialiser is returned whole.
"""
depth = 0
for i, c in _scan(src, start, None):
if c in "{[(":
depth += 1
elif c in "}])":
depth -= 1
elif depth == 0 and c == ";":
return i + 1
elif depth == 0 and c == "\n" and i > start:
return i
return len(src)
def _end_of_params(src: str, open_paren: int) -> int:
"""Index just past the ``)`` closing the parameter list at ``open_paren``."""
depth = 0
for i, c in _scan(src, open_paren, None):
if c == "(":
depth += 1
elif c == ")":
depth -= 1
if depth == 0:
return i + 1
raise AssertionError(f"unbalanced parameter list at offset {open_paren}")
+31
View File
@@ -31,6 +31,7 @@ safe for callers that pass both a parent package and a child module.
"""
import sys
import types
from contextlib import contextmanager
_ABSENT = object()
@@ -167,3 +168,33 @@ def preserve_import_state(*module_names):
# Phase 2: restore all parent-package attributes.
for name, (_, saved_attr) in saved.items():
_restore_parent_attr(name, saved_attr)
# Names under these prefixes are the ones a leaked stub actually breaks: a
# later test doing ``import src.x`` or ``import core.x`` silently gets the
# empty stub instead of the real module.
_GUARDED_PREFIXES = ("src.", "core.")
def bare_module_stubs():
"""Return the ``src.*``/``core.*`` names currently bound to a bare stub.
A bare stub is a plain :class:`types.ModuleType` with no on-disk
``__file__`` — the object ``types.ModuleType(name)`` produces. That is the
same "is this a fake?" test the ``clear_fake_*`` helpers above use, so a
module imported from disk is never reported.
``MagicMock`` stand-ins are deliberately out of scope: they answer every
attribute, so they fail loudly at use rather than silently, and several
test modules install them on purpose.
"""
found = set()
for name, mod in list(sys.modules.items()):
if not name.startswith(_GUARDED_PREFIXES):
continue
if type(mod) is not types.ModuleType:
continue
if getattr(mod, "__file__", None):
continue
found.add(name)
return found
+67
View File
@@ -0,0 +1,67 @@
import { readFile } from 'node:fs/promises';
import { dirname, join } from 'node:path';
import { fileURLToPath } from 'node:url';
const HERE = dirname(fileURLToPath(import.meta.url));
const STATIC = join(HERE, '..', '..', 'static');
const INDEX = join(STATIC, 'index.html');
const LINK = /<link\b[^>]*\brel\s*=\s*["']stylesheet["'][^>]*\bhref\s*=\s*["']\/static\/([^"'?]+)([^"']*)["']/gi;
async function entries() {
const html = await readFile(INDEX, 'utf8');
const out = [];
for (const match of html.matchAll(LINK)) {
const rel = match[1];
const query = match[2];
if (rel.startsWith('lib/')) continue;
out.push({
path: join(STATIC, rel),
url: `/static/${rel}${query}`,
});
}
if (!out.length) {
throw new Error(`no app stylesheet <link> tags found in ${INDEX}`);
}
for (const entry of out) {
try {
await readFile(entry.path);
} catch {
throw new Error(
`index.html links a stylesheet that does not exist: ${entry.path}`,
);
}
}
return out;
}
export async function stylesheetPaths() {
return (await entries()).map(entry => entry.path);
}
export async function stylesheetUrls() {
return (await entries()).map(entry => entry.url);
}
export async function stylesheetLinkTags() {
return (await stylesheetUrls())
.map(url => `<link rel="stylesheet" href="${url}">`)
.join('');
}
export async function appCss() {
const paths = await stylesheetPaths();
const parts = [];
for (const path of paths) {
parts.push(await readFile(path, 'utf8'));
}
return parts.join('\n');
}
+8 -9
View File
@@ -1,11 +1,11 @@
"""Read the app's CSS the way the browser does.
``static/style.css`` no longer holds every rule: panel styles live in separate
files that ``static/index.html`` loads eagerly, in a fixed order, right after
it. The cascade is the concatenation of those files in that order.
The former ``static/style.css`` is now an ordered set of numbered fragments,
followed by the existing panel stylesheets. ``static/index.html`` loads the
complete cascade eagerly in the order the browser must apply it.
A test that asserts on a rule must therefore look at all of them. Reading
``static/style.css`` alone ties the test to whichever file a rule happens to
A test that asserts on a rule must therefore look at the complete cascade.
Reading one fragment alone ties the test to whichever file a rule happens to
sit in today, so it goes red the next time a rule moves without anything about
the rendered page having changed.
"""
@@ -53,8 +53,7 @@ def stylesheet_urls() -> list[str]:
def stylesheet_link_tags() -> str:
"""The <link> tags to drop into a synthetic page so it gets the whole
cascade, not just style.css."""
"""The <link> tags for a synthetic page that needs the whole cascade."""
return "".join(f'<link rel="stylesheet" href="{u}">' for u in stylesheet_urls())
@@ -68,8 +67,8 @@ def stylesheet_cache_version() -> str:
The stylesheet is split across several files that must be busted together:
shipping one fragment under a stale token serves a browser half of an old
cascade and half of a new one. Tests that used to read the version off
``style.css`` ask for it here instead, so they keep checking the invariant
cascade and half of a new one. Tests ask for the shared version here instead of deriving it from one
stylesheet filename, so they keep checking the invariant
rather than a filename.
"""
versions = set()
+46
View File
@@ -0,0 +1,46 @@
"""Bind an AF_UNIX socket at a path the kernel will actually accept.
``sun_path`` is 104 bytes on macOS, terminator included, so a bind path longer
than 103 characters fails with ``OSError: AF_UNIX path too long``. pytest's
``tmp_path`` is rooted at ``$TMPDIR``, which on stock macOS is a 49-character
``/var/folders/<2>/<30>/T/`` path; adding ``pytest-of-<user>/pytest-<n>/`` and
the test's own (truncated) name spends the rest of the budget before the
filename is appended.
That is why this reads as flaky rather than broken. Linux allows 108 bytes and
roots ``$TMPDIR`` at ``/tmp``, so it never bites there; on macOS whether it
bites depends on the length of ``$TMPDIR``, the test's name, and how many
digits pytest's run counter is currently using. A run under a shortened
``$TMPDIR`` passes, the same checkout under the default one does not.
The path is resolved before it is handed back, for the same reason the rest of
this change resolves temp paths: on macOS ``/tmp`` is a symlink to
``/private/tmp``, and a test that binds one spelling and asserts on the other
is comparing two names for the same socket.
"""
import os
import shutil
import socket
import tempfile
from contextlib import contextmanager
# Short enough to leave room for the socket's own name under every platform's
# sun_path budget. A relative root would depend on the working directory.
_SHORT_ROOT = os.path.realpath(tempfile.gettempdir() if os.name == "nt" else "/tmp")
@contextmanager
def bound_unix_socket(name="docker.sock"):
"""Yield the path of a listening AF_UNIX socket, cleaned up on exit."""
directory = os.path.realpath(tempfile.mkdtemp(prefix="odysseus-sock-", dir=_SHORT_ROOT))
path = os.path.join(directory, name)
if len(path) > 103: # pragma: no cover - guards the guard
raise AssertionError(f"socket path is {len(path)} bytes, over the limit: {path}")
sock = socket.socket(socket.AF_UNIX)
try:
sock.bind(path)
yield path
finally:
sock.close()
shutil.rmtree(directory, ignore_errors=True)