feat(config): T-1282 — the validate domain, and a move that broke a root

reach validate content / checklist / ron / name-collisions. The three old
scripts are retired, their make targets with them.

Print statements go through the logging sink rather than a collector. The
validators emit their findings as console events as they run, so a long content
validation streams instead of going quiet and dumping at the end — the message
strings and their order are unchanged, only the destination. That also
satisfies the conformance rule forbidding print() in the package, which is what
forced the question.

validate-ron was three languages deep: bash dispatching on a flag, a Python
heredoc doing collision detection, cargo run for schema validation. Logic
embedded in a shell string cannot be imported, tested, or found by anything
that indexes Python, so it became Python; the cargo call became a guarded exec.
It also split into two verbs, because --check-name-collisions answered a
different question from the default path: whether the SET of cultures is
coherent, versus whether ONE file is well-formed.

The move broke something, quietly, which is the point of doing these one at a
time. validate-checklist computed ROOT as Path(__file__).parent.parent — the
repo root while it lived at tooling/validate-checklist, and tooling/domains
once moved. Both its schema and gauntlet paths silently repointed at nothing,
the gauntlet directory "did not exist", and it reported success having checked
zero files. Caught by running it beside the original: old exit 1, new exit 0.
Now config.repo_root(), and load_schema raises ReachError instead of calling
sys.exit, which a service must not do.

Parity on the live tree: content reproduces the original byte for byte
including its counts, name-collisions likewise. Tests pin what those runs
cannot reach — the detection path, since the repo currently has no collisions,
and the argument errors.

Two things found and left alone: validate-content FAILS on the live tree with
13 missing schemas, pre-existing and unrelated to this port; and the ticket's
claim that validate-content sits in the pre-commit hook is wrong — that hook
runs only check-fact-ids and pql decisions validate, so there was no shared
edit to coordinate.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-09-02 12:48:56 +02:00
co-authored by Claude Opus 5
parent cb5d3f1335
commit 7f20bd303b
15 changed files with 798 additions and 252 deletions
+6
View File
@@ -0,0 +1,6 @@
"""The `validate` domain — content, checklists and RON against their schemas.
Three validators with three shapes: `content` and `checklist` were Python and
moved; `ron` was bash wrapping a Python heredoc wrapping a Rust binary, so its
logic became Python and its binary call became a guarded exec (D-263).
"""
@@ -9,22 +9,37 @@ Exit code 0 = all valid, 1 = validation errors found.
"""
import json
import sys
from pathlib import Path
from tooling.core import config, console
from tooling.core.errors import ReachError
import jsonschema
import yaml
ROOT = Path(__file__).resolve().parent.parent
# config.repo_root(), NOT __file__-relative. The original computed
# Path(__file__).parent.parent, which meant the repo root while this file lived
# at tooling/validate-checklist and means tooling/domains now. Moving the file
# silently repointed both paths at nothing, the gauntlet directory "did not
# exist", and the validator reported success having checked zero files — the
# exact shape of failure this whole initiative keeps finding.
ROOT = config.repo_root()
SCHEMA_PATH = ROOT / "server" / "content" / "_schema" / "checklist.schema.json"
GAUNTLET_DIR = ROOT / "server" / "content" / "gauntlet"
def load_schema():
if not SCHEMA_PATH.exists():
print(f"ERROR: Schema file not found at {_rel(SCHEMA_PATH)}")
print("Expected: content/_schema/checklist.schema.json")
sys.exit(1)
# ReachError, not sys.exit: a service must not decide to end the
# process, and the caller gets a remedy rather than a bare 1.
raise ReachError(
f"checklist schema not found at {_rel(SCHEMA_PATH)}",
fix="expected server/content/_schema/checklist.schema.json — "
"restore it or correct the path",
)
with open(SCHEMA_PATH) as f:
return json.load(f)
@@ -59,12 +74,12 @@ def validate_schema(files, schema):
with open(path) as f:
data = yaml.safe_load(f)
except yaml.YAMLError as e:
print(f"YAML ERROR: {rel}: {e}")
console.event(f"YAML ERROR: {rel}: {e}")
errors += 1
continue
if data is None:
print(f"EMPTY: {rel}")
console.event(f"EMPTY: {rel}")
errors += 1
continue
@@ -72,10 +87,10 @@ def validate_schema(files, schema):
jsonschema.validate(instance=data, schema=schema)
validated += 1
except jsonschema.ValidationError as e:
print(f"INVALID: {rel}")
print(f" Error: {e.message}")
console.event(f"INVALID: {rel}")
console.event(f" Error: {e.message}")
if e.absolute_path:
print(f" Path: {'.'.join(str(p) for p in e.absolute_path)}")
console.event(f" Path: {'.'.join(str(p) for p in e.absolute_path)}")
errors += 1
return validated, errors
@@ -102,14 +117,14 @@ def check_id_uniqueness(files):
if not cid:
continue
if cid in local_seen:
print(f'ID ERROR: duplicate condition id "{cid}" in {rel}')
console.event(f'ID ERROR: duplicate condition id "{cid}" in {rel}')
errors += 1
local_seen.add(cid)
if cid in global_ids and global_ids[cid] != path:
print(f'ID ERROR: condition id "{cid}" used in multiple files')
print(f" First: {_rel(global_ids[cid])}")
print(f" Also: {rel}")
console.event(f'ID ERROR: condition id "{cid}" used in multiple files')
console.event(f" First: {_rel(global_ids[cid])}")
console.event(f" Also: {rel}")
errors += 1
elif cid not in global_ids:
global_ids[cid] = path
@@ -135,60 +150,57 @@ def summarize(files):
count = len(conditions)
total += count
room = data.get("room_id", "cross_room")
print(f" {room}: {count} conditions")
console.event(f" {room}: {count} conditions")
for cond in conditions:
ct = cond.get("condition_type", "unknown")
type_counts[ct] = type_counts.get(ct, 0) + 1
print(f"\n Total: {total} conditions across {len(files)} files")
console.event(f"\n Total: {total} conditions across {len(files)} files")
if type_counts:
print(" By type:")
console.event(" By type:")
for ct in sorted(type_counts):
print(f" {ct}: {type_counts[ct]}")
console.event(f" {ct}: {type_counts[ct]}")
def main():
check_only = "--check" in sys.argv
def validate(check_only: bool = False) -> int:
if not GAUNTLET_DIR.exists():
print(f"No gauntlet directory at {_rel(GAUNTLET_DIR)}")
print("Checklist validation skipped (no content yet).")
console.event(f"No gauntlet directory at {_rel(GAUNTLET_DIR)}")
console.event("Checklist validation skipped (no content yet).")
return 0
files = find_checklists()
if not files:
print("No checklist files found under content/gauntlet/.")
print("Checklist validation skipped.")
console.event("No checklist files found under content/gauntlet/.")
console.event("Checklist validation skipped.")
return 0
schema = load_schema()
# Pass 1: Schema validation
validated, schema_errors = validate_schema(files, schema)
print(f"Pass 1 (schema): {validated} valid, {schema_errors} errors")
console.event(f"Pass 1 (schema): {validated} valid, {schema_errors} errors")
if schema_errors > 0:
print(f"\nSchema validation failed ({schema_errors} errors) — skipping ID checks")
console.event(f"\nSchema validation failed ({schema_errors} errors) — skipping ID checks")
return 1
# Pass 2: Condition ID uniqueness
id_errors = check_id_uniqueness(files)
if id_errors > 0:
print(f"Pass 2 (IDs): {id_errors} errors")
console.event(f"Pass 2 (IDs): {id_errors} errors")
total_errors = schema_errors + id_errors
print(f"\nChecklist validation: {validated} valid, {total_errors} errors")
console.event(f"\nChecklist validation: {validated} valid, {total_errors} errors")
if total_errors > 0:
return 1
if not check_only:
print("\nChecklist summary:")
console.event("\nChecklist summary:")
summarize(files)
return 0
if __name__ == "__main__":
sys.exit(main())
@@ -23,13 +23,17 @@ Exit code 0 = all valid, 1 = validation errors found.
import json
import re
import sys
from pathlib import Path
import jsonschema
import yaml
CONTENT_DIR = Path(__file__).resolve().parent.parent / "server" / "content"
from tooling.core import config, console
CONTENT_DIR = config.path("server", "content")
SCHEMA_DIR = CONTENT_DIR / "_schema"
# Map directory parent name (or filename) to schema file
@@ -222,9 +226,9 @@ class ContentIndex:
if not cid:
continue
if cid in seen:
print(f'XREF ERROR: duplicate canonical_id "{cid}"')
print(f" Defined in: {_rel(seen[cid])}")
print(f" Duplicate in: {_rel(path)}")
console.event(f'XREF ERROR: duplicate canonical_id "{cid}"')
console.event(f" Defined in: {_rel(seen[cid])}")
console.event(f" Duplicate in: {_rel(path)}")
errors += 1
else:
seen[cid] = path
@@ -243,9 +247,9 @@ class ContentIndex:
target = rel.get("target")
if target and target not in self.npcs:
known = sorted(self.npcs.keys())
print(f'XREF ERROR: unresolved relationship target "{target}"')
print(f" In: {_rel(path)}")
print(f" Known canonical_ids: {', '.join(known[:10])}"
console.event(f'XREF ERROR: unresolved relationship target "{target}"')
console.event(f" In: {_rel(path)}")
console.event(f" Known canonical_ids: {', '.join(known[:10])}"
+ (f" (and {len(known) - 10} more)" if len(known) > 10 else ""))
errors += 1
return errors
@@ -261,9 +265,9 @@ class ContentIndex:
for slug in locs:
expected = locations_dir / f"{slug}.yaml"
if not expected.exists():
print(f'XREF ERROR: location slug "{slug}" not found')
print(f" In: {_rel(path)}")
print(f" Expected file: {_rel(expected)}")
console.event(f'XREF ERROR: location slug "{slug}" not found')
console.event(f" In: {_rel(path)}")
console.event(f" Expected file: {_rel(expected)}")
errors += 1
return errors
@@ -280,21 +284,21 @@ class ContentIndex:
district_key = str(district_dir)
district_locs = self.locations.get(district_key, set())
if not district_locs:
print(f'XREF ERROR: dialogue location "{location}" references district with no locations declared')
print(f" In: {_rel(path)}")
print(f" District: {_rel(district_dir / 'district.yaml')}")
console.event(f'XREF ERROR: dialogue location "{location}" references district with no locations declared')
console.event(f" In: {_rel(path)}")
console.event(f" District: {_rel(district_dir / 'district.yaml')}")
errors += 1
elif location not in district_locs:
print(f'XREF ERROR: dialogue location "{location}" not in district')
print(f" In: {_rel(path)}")
print(f" District locations: {', '.join(sorted(district_locs))}")
console.event(f'XREF ERROR: dialogue location "{location}" not in district')
console.event(f" In: {_rel(path)}")
console.event(f" District locations: {', '.join(sorted(district_locs))}")
errors += 1
return errors
def _check_5_fact_ids(self) -> int:
"""Check that all referenced fact_ids resolve to knowledge catalogs."""
if not self.fact_ids:
print(" Check 5 (fact_ids): SKIPPED — no canonical fact_ids in knowledge catalogs yet")
console.event(" Check 5 (fact_ids): SKIPPED — no canonical fact_ids in knowledge catalogs yet")
return 0
errors = 0
@@ -303,12 +307,12 @@ class ContentIndex:
for fact_id, sources in sorted(refs.items()):
if fact_id not in self.fact_ids:
print(f'XREF ERROR: unknown fact_id "{fact_id}"')
console.event(f'XREF ERROR: unknown fact_id "{fact_id}"')
for src in sources[:3]:
print(f" In: {_rel(src)}")
console.event(f" In: {_rel(src)}")
if len(sources) > 3:
print(f" ...and {len(sources) - 3} more files")
print(f" Canonical fact_ids: {len(self.fact_ids)} defined")
console.event(f" ...and {len(sources) - 3} more files")
console.event(f" Canonical fact_ids: {len(self.fact_ids)} defined")
errors += 1
return errors
@@ -380,12 +384,12 @@ class ContentIndex:
if not isinstance(slug, str):
continue
if slug not in available:
print(f'XREF ERROR: triangle "{slug}" not found')
print(f" In: {_rel(path)}")
console.event(f'XREF ERROR: triangle "{slug}" not found')
console.event(f" In: {_rel(path)}")
if available:
print(f" Available triangles: {', '.join(sorted(available))}")
console.event(f" Available triangles: {', '.join(sorted(available))}")
else:
print(f" No triangles found in {_rel(district_dir / 'triangles')}")
console.event(f" No triangles found in {_rel(district_dir / 'triangles')}")
errors += 1
return errors
@@ -401,10 +405,10 @@ class ContentIndex:
continue
actual = len(list(npcs_dir.glob("*.yaml")))
if declared != actual:
print(f"XREF WARNING: npc_count mismatch")
print(f" Declared: {declared}")
print(f" Actual NPC files: {actual}")
print(f" In: {_rel(path)}")
console.event("XREF WARNING: npc_count mismatch")
console.event(f" Declared: {declared}")
console.event(f" Actual NPC files: {actual}")
console.event(f" In: {_rel(path)}")
warnings += 1
return warnings
@@ -426,18 +430,18 @@ class ContentIndex:
continue
# Per-file duplicate
if line_id in local_seen:
print(f'XREF ERROR: duplicate dialogue line id "{line_id}"')
print(f" In: {_rel(path)}")
print(f" First: line {local_seen[line_id]}")
print(f" Duplicate: line {i}")
console.event(f'XREF ERROR: duplicate dialogue line id "{line_id}"')
console.event(f" In: {_rel(path)}")
console.event(f" First: line {local_seen[line_id]}")
console.event(f" Duplicate: line {i}")
errors += 1
else:
local_seen[line_id] = i
# Cross-file duplicate
if line_id in global_seen and global_seen[line_id] != path:
print(f'XREF ERROR: dialogue line id "{line_id}" used in multiple files')
print(f" First: {_rel(global_seen[line_id])}")
print(f" Also in: {_rel(path)}")
console.event(f'XREF ERROR: dialogue line id "{line_id}" used in multiple files')
console.event(f" First: {_rel(global_seen[line_id])}")
console.event(f" Also in: {_rel(path)}")
errors += 1
elif line_id not in global_seen:
global_seen[line_id] = path
@@ -473,7 +477,7 @@ class ContentIndex:
continue
checked.add(pair)
if target in self.npcs and cid not in npc_rels.get(target, set()):
print(f"XREF WARNING: {cid} has relationship to {target} but {target} has no reciprocal entry")
console.event(f"XREF WARNING: {cid} has relationship to {target} but {target} has no reciprocal entry")
warnings += 1
return warnings
@@ -487,7 +491,7 @@ def schema_validate(campaigns_dir: Path) -> tuple[int, int, int]:
yaml_files = sorted(campaigns_dir.rglob("*.yaml"))
if not yaml_files:
print("No YAML files found under campaigns/", file=sys.stderr)
console.event("No YAML files found under campaigns/")
return 0, 0, 1
for yaml_path in yaml_files:
@@ -497,7 +501,7 @@ def schema_validate(campaigns_dir: Path) -> tuple[int, int, int]:
continue
if not schema_path.exists():
print(f"MISSING SCHEMA: {schema_path.name} for {yaml_path.relative_to(CONTENT_DIR)}")
console.event(f"MISSING SCHEMA: {schema_path.name} for {yaml_path.relative_to(CONTENT_DIR)}")
errors += 1
continue
@@ -511,7 +515,7 @@ def schema_validate(campaigns_dir: Path) -> tuple[int, int, int]:
with open(yaml_path) as f:
data = yaml.safe_load(f)
except yaml.YAMLError as e:
print(f"YAML ERROR: {yaml_path.relative_to(CONTENT_DIR)}: {e}")
console.event(f"YAML ERROR: {yaml_path.relative_to(CONTENT_DIR)}: {e}")
errors += 1
continue
@@ -524,40 +528,43 @@ def schema_validate(campaigns_dir: Path) -> tuple[int, int, int]:
validated += 1
except jsonschema.ValidationError as e:
rel = yaml_path.relative_to(CONTENT_DIR)
print(f"INVALID: {rel}")
print(f" Schema: {schema_path.name}")
print(f" Error: {e.message}")
console.event(f"INVALID: {rel}")
console.event(f" Schema: {schema_path.name}")
console.event(f" Error: {e.message}")
if e.absolute_path:
print(f" Path: {'.'.join(str(p) for p in e.absolute_path)}")
console.event(f" Path: {'.'.join(str(p) for p in e.absolute_path)}")
errors += 1
return validated, skipped, errors
def main() -> int:
def validate() -> int:
"""Run both passes. Returns the exit code.
Replaces the old main(). Same logic, same messages, in the same order — but
they go out as EVENTS through core/console rather than through print, so
they stream as the validation runs, carry the invocation's job id, and are
rendered by the sink. A service does not decide to print (D-263).
"""
campaigns_dir = CONTENT_DIR / "campaigns"
if not campaigns_dir.exists():
print(f"No campaigns directory at {campaigns_dir}", file=sys.stderr)
console.event(f"No campaigns directory at {campaigns_dir}")
return 1
# Pass 1: Schema validation
validated, skipped, schema_errors = schema_validate(campaigns_dir)
print(f"\nPass 1 (schema): {validated} validated, {skipped} skipped, {schema_errors} errors")
console.event(f"\nPass 1 (schema): {validated} validated, {skipped} skipped, {schema_errors} errors")
if schema_errors > 0:
print(f"\nSchema validation failed ({schema_errors} errors) -- skipping cross-references")
console.event(f"\nSchema validation failed ({schema_errors} errors) -- skipping cross-references")
return 1
# Pass 2: Cross-reference validation
print("\nPass 2 (cross-references):")
console.event("\nPass 2 (cross-references):")
index = ContentIndex(campaigns_dir)
index.build()
xref_errors, xref_warnings = index.validate_references()
total_errors = schema_errors + xref_errors
print(f"\nValidated {validated} files: {total_errors} errors, {xref_warnings} warnings")
return 1 if total_errors > 0 else 0
if __name__ == "__main__":
sys.exit(main())
console.event(f"\nValidated {validated} files: {total_errors} errors, {xref_warnings} warnings")
return (1 if total_errors > 0 else 0)
+112
View File
@@ -0,0 +1,112 @@
"""RON validation — ported from the bash `validate-ron` (D-263).
The old script was three languages deep: bash dispatching on a flag, a Python
heredoc doing collision detection, and `cargo run` for schema validation. The
heredoc is the reason this is a rewrite rather than a move — logic embedded in
a shell string cannot be imported, cannot be tested, and cannot be read by
anything that indexes Python.
What stayed shell-shaped is the one genuine OS interaction: running the Rust
validator. That goes through `core.process.run`, which is the sanctioned exec.
"""
from __future__ import annotations
import re
from pathlib import Path
from tooling.core import config, console, process
from tooling.core.errors import ReachError
SCHEMAS = ("zone", "zone_type", "culture")
def name_collisions(directory: Path) -> int:
"""Report names shared between culture name pools. Returns an exit code.
A name appearing in two cultures' pools makes generated NPCs ambiguous
about where they are from, which is invisible until someone notices two
cultures producing the same surnames.
"""
if not directory.is_dir():
raise ReachError(
f"directory not found: {directory}",
fix="pass a directory containing culture-*.ron files",
)
files = sorted(directory.glob("culture-*.ron"))
if not files:
console.event(f"No culture-*.ron files found in: {directory}")
return 0
given: dict[str, set[str]] = {}
family: dict[str, set[str]] = {}
for path in files:
text = path.read_text(encoding="utf-8")
match = re.search(r'\bid\s*:\s*"([^"]+)"', text)
culture = match.group(1) if match else path.name
given[culture] = set(_names(text, "given_names"))
family[culture] = set(_names(text, "family_names"))
collisions = False
for field, pools in (("given_names", given), ("family_names", family)):
for name, cultures in sorted(_shared(pools).items()):
collisions = True
console.event(
f'COLLISION {field}: "{name}" in {", ".join(sorted(cultures))}',
level="error",
)
if collisions:
return 1
console.verdict(
f"OK: no name collisions across {len(given)} culture(s): "
f"{', '.join(sorted(given))}"
)
return 0
def ron_file(path: Path, schema: str) -> int:
"""Validate one .ron file against a Rust struct schema."""
if schema not in SCHEMAS:
raise ReachError(
f"unknown schema {schema!r}",
fix=f"choose one of: {', '.join(SCHEMAS)}",
exit_code=2,
)
if not path.is_file():
# Checked before resolving, because realpath on a missing file gives an
# error about the path rather than about the file — the old script made
# the same distinction and it is worth keeping.
raise ReachError(
f"file not found: {path}",
fix="check the path, or pass a directory to `reach validate name-collisions`",
)
result = process.run(
["cargo", "run", "--quiet", "--bin", "validate_ron", "--", str(path.resolve()), schema],
cwd=config.path("server"),
check=False,
capture=False,
missing_fix="install Rust — make setup-rust",
)
return result.returncode
def _names(text: str, field: str) -> list[str]:
"""Quoted strings from a named RON array field, comments stripped."""
match = re.search(rf"\b{re.escape(field)}\s*:\s*\[([^\]]*)\]", text, re.DOTALL)
if not match:
return []
block = re.sub(r"//[^\n]*", "", match.group(1))
return re.findall(r'"([^"]+)"', block)
def _shared(pools: dict[str, set[str]]) -> dict[str, list[str]]:
"""name -> the cultures claiming it, for names claimed more than once."""
owners: dict[str, list[str]] = {}
for culture, names in pools.items():
for name in names:
owners.setdefault(name, []).append(culture)
return {name: cultures for name, cultures in owners.items() if len(cultures) > 1}
+100
View File
@@ -0,0 +1,100 @@
"""Transport for the `validate` domain — args in, delegate, format out.
Zero logic. Note what these commands do NOT do: collect output. The validators
emit their findings as events through `core/console` as they run, so a long
content validation streams rather than going quiet and dumping at the end. The
router's job is the verdict and the exit code.
"""
from __future__ import annotations
from pathlib import Path
import typer
from tooling.core import cli, console
from tooling.core.command import command
from tooling.core.errors import ReachError
from tooling.domains.validate import checklist as checklist_module
from tooling.domains.validate import content as content_module
from tooling.domains.validate import ron as ron_module
app = cli.domain("validate", "Content, checklists and RON against their schemas.")
@app.callback()
def _domain() -> None:
"""Keeps `validate` a group (Typer collapses a single-command app)."""
@app.command("content")
@command
def content() -> None:
"""Validate content YAML against JSON schemas, then cross-references."""
code = content_module.validate()
if code != 0:
raise ReachError(
"validate-content: validation failed",
fix="the errors above name each file and what is wrong with it; "
"schemas live in server/content/_schema/",
exit_code=code,
)
console.verdict("validate-content: OK")
@app.command("checklist")
@command
def checklist(
check: bool = typer.Option(
False, "--check", help="Schema validation only, for the pre-PR chain."
),
) -> None:
"""Validate checklist YAML against its schema, and ids for uniqueness."""
code = checklist_module.validate(check_only=check)
if code != 0:
raise ReachError(
"validate-checklist: validation failed",
fix="the errors above name each checklist and the field at fault",
exit_code=code,
)
console.verdict("validate-checklist: OK")
@app.command("ron")
@command
def ron(
path: Path = typer.Argument(..., help="The .ron file to validate."),
schema: str = typer.Argument(..., help=f"One of: {', '.join(ron_module.SCHEMAS)}"),
) -> None:
"""Validate a RON file against its Rust struct schema."""
code = ron_module.ron_file(path, schema)
if code != 0:
raise ReachError(
f"validate-ron: {path} does not match the {schema} schema",
fix="the validator's output above names the field; the struct is in "
"server/src/ — compare field names and types",
exit_code=code,
)
console.verdict(f"validate-ron: OK — {path} matches {schema}")
@app.command("name-collisions")
@command
def name_collisions(
directory: Path = typer.Argument(..., help="Directory holding culture-*.ron files."),
) -> None:
"""Report names shared between two cultures' name pools.
A separate verb rather than a flag on `ron`, because it answers a different
question: `ron` asks whether ONE file is well-formed, this asks whether the
SET of them is coherent. The old script fused them behind
--check-name-collisions and had to branch on it before doing anything.
"""
code = ron_module.name_collisions(directory)
if code != 0:
raise ReachError(
"validate-ron: name pools collide across cultures",
fix="rename the colliding entries so each name belongs to one culture — "
"a shared name makes a generated NPC's origin ambiguous",
exit_code=code,
)
+4
View File
@@ -51,6 +51,10 @@ DOMAINS: dict[str, tuple[str, str]] = {
"tooling.domains.check.router:app",
"Consistency gates — the checks the push hook runs",
),
"validate": (
"tooling.domains.validate.router:app",
"Content, checklists and RON against their schemas",
),
"jobs": (
"tooling.domains.jobs.router:app",
"Detached runs — status, logs and outcomes",
+180
View File
@@ -0,0 +1,180 @@
#!/usr/bin/env python3
"""Behaviour of the `reach validate` verbs (T-1282).
Parity against the scripts these replaced was established on the live tree —
`content` reproduces the original's output byte for byte including its counts,
`name-collisions` likewise — and recorded in the ticket. What is pinned here is
the behaviour those runs could not reach: the failure paths.
That gap matters more than usual for this domain. `validate-content` currently
FAILS on the live tree (13 missing schemas, pre-existing), so its success path
is the one nothing exercises; `name-collisions` currently PASSES, so its
detection path is the one nothing exercises. A live run proves whichever half
the repo happens to be in.
Run: python3 tooling/test_validate.py
"""
import os
import shutil
import subprocess
import sys
import tempfile
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parent.parent
CULTURE = '''(
id: "{cid}",
naming: (
given_names: [
// a comment mentioning "decoy" which must not be collected
{given}
],
family_names: [
{family}
],
),
)
'''
def _reach(*args: str, root: Path | None = None) -> subprocess.CompletedProcess[str]:
env = {**os.environ, "SR_OUTPUT_FORMAT": "text"}
if root is not None:
env["SR_REPO_ROOT"] = str(root)
return subprocess.run(
["reach", "validate", *args],
capture_output=True,
text=True,
cwd=REPO_ROOT,
env=env,
)
def _cultures(directory: Path, pools: dict[str, tuple[list[str], list[str]]]) -> None:
directory.mkdir(parents=True, exist_ok=True)
for cid, (given, family) in pools.items():
(directory / f"culture-{cid}.ron").write_text(
CULTURE.format(
cid=cid,
given=", ".join(f'"{n}"' for n in given),
family=", ".join(f'"{n}"' for n in family),
),
encoding="utf-8",
)
def test_collision_detected(failures: list[str]) -> None:
"""Two cultures sharing a name is a failure that names both."""
with tempfile.TemporaryDirectory() as tmp:
directory = Path(tmp) / "global"
_cultures(
directory,
{
"alpha": (["Ada", "Shared"], ["Alpha"]),
"beta": (["Bo", "Shared"], ["Beta"]),
},
)
result = _reach("name-collisions", str(directory))
combined = result.stdout + result.stderr
if result.returncode == 0:
failures.append(
"[collision] a shared name exited 0 — a collision that reports "
"success makes generated NPCs ambiguous about their origin, "
"invisibly"
)
if "Shared" not in combined:
failures.append("[collision] the colliding name is not in the output")
for culture in ("alpha", "beta"):
if culture not in combined:
failures.append(
f"[collision] {culture} is not named — a collision report that "
"omits an owner cannot be acted on"
)
if "decoy" in combined:
failures.append(
"[collision] a name inside a // comment was collected; comments "
"must be stripped before extracting quoted strings"
)
def test_no_collision(failures: list[str]) -> None:
with tempfile.TemporaryDirectory() as tmp:
directory = Path(tmp) / "global"
_cultures(directory, {"alpha": (["Ada"], ["Alpha"]), "beta": (["Bo"], ["Beta"])})
result = _reach("name-collisions", str(directory))
if result.returncode != 0:
failures.append(
f"[no-collision] exited {result.returncode} with disjoint pools"
)
def test_empty_directory(failures: list[str]) -> None:
"""No culture files is not a failure — there is nothing to contradict."""
with tempfile.TemporaryDirectory() as tmp:
directory = Path(tmp) / "global"
directory.mkdir(parents=True)
result = _reach("name-collisions", str(directory))
if result.returncode != 0:
failures.append(
f"[empty] exited {result.returncode} on a directory with no cultures"
)
def test_missing_directory(failures: list[str]) -> None:
result = _reach("name-collisions", "/definitely/not/a/directory")
if result.returncode == 0:
failures.append("[missing-dir] a nonexistent directory exited 0")
def test_ron_argument_errors(failures: list[str]) -> None:
"""Bad arguments fail before anything is executed.
Both cases matter because the alternative is invoking cargo to discover
them, which is slow and reports the mistake in the validator's vocabulary
rather than the caller's.
"""
unknown = _reach("ron", "server/content/global/culture-osse.ron", "not_a_schema")
if unknown.returncode != 2:
failures.append(
f"[ron-schema] unknown schema exited {unknown.returncode}, expected 2"
)
if "zone_type" not in (unknown.stdout + unknown.stderr):
failures.append(
"[ron-schema] the rejection does not list the accepted schemas — the "
"closed-set rule D-263 exists for"
)
missing = _reach("ron", "/definitely/not/a/file.ron", "culture")
if missing.returncode == 0:
failures.append("[ron-file] a nonexistent file exited 0")
def main() -> int:
if shutil.which("reach") is None:
print(
"test_validate: `reach` is not on PATH.\n Fix: make install-reach",
file=sys.stderr,
)
return 1
failures: list[str] = []
test_collision_detected(failures)
test_no_collision(failures)
test_empty_directory(failures)
test_missing_directory(failures)
test_ron_argument_errors(failures)
if failures:
print("test_validate: FAIL", file=sys.stderr)
for failure in failures:
print(f" - {failure}", file=sys.stderr)
return 1
print("test_validate: OK — collisions detected and named, argument errors caught")
return 0
if __name__ == "__main__":
sys.exit(main())
-137
View File
@@ -1,137 +0,0 @@
#!/usr/bin/env bash
# RON content validator — wrapper for the Rust validate_ron binary (#611).
#
# Usage:
# tooling/validate-ron <file.ron> <zone|zone_type|culture>
# tooling/validate-ron --check-name-collisions <content/dir/>
#
# Examples:
# tooling/validate-ron server/content/global/zone-types/rural_agricultural.ron zone_type
# tooling/validate-ron server/content/global/culture-van-maanens-star.example.ron culture
# tooling/validate-ron server/content/global/zone-identity-spec.example.ron zone
# tooling/validate-ron --check-name-collisions server/content/global/
#
# --check-name-collisions:
# Scans all culture-*.ron files in the given directory. Extracts naming.given_names
# and naming.family_names pools from each file. Reports any name that appears in more
# than one culture's pool. Exits 1 on collision; exits 0 if no collisions found.
# Output is machine-readable (one collision per line).
set -euo pipefail
# --- Name collision mode --------------------------------------------------
if [ "${1:-}" = "--check-name-collisions" ]; then
if [ $# -lt 2 ]; then
echo "Usage: tooling/validate-ron --check-name-collisions <directory>"
exit 1
fi
SCAN_DIR="$2"
if [ ! -d "$SCAN_DIR" ]; then
echo "Error: directory not found: $SCAN_DIR"
exit 1
fi
python3 - "$SCAN_DIR" <<'PYEOF'
import sys
import re
import os
import glob
scan_dir = sys.argv[1]
pattern = os.path.join(scan_dir, "culture-*.ron")
files = sorted(glob.glob(pattern))
if not files:
print(f"No culture-*.ron files found in: {scan_dir}")
sys.exit(0)
def extract_names_from_block(text, field):
"""Extract quoted string list from a named RON array field."""
# Match: field: [\n "name", "name", ...\n]
# Handles multi-line arrays with comments inside.
pat = re.compile(
r'\b' + re.escape(field) + r'\s*:\s*\[([^\]]*)\]',
re.DOTALL
)
m = pat.search(text)
if not m:
return []
block = m.group(1)
# Strip line comments before extracting quoted strings
block = re.sub(r'//[^\n]*', '', block)
return re.findall(r'"([^"]+)"', block)
# culture_id -> set of names (given + family, tracked separately for reporting)
given_by_culture = {}
family_by_culture = {}
for fpath in files:
with open(fpath, 'r', encoding='utf-8') as f:
text = f.read()
# Extract culture id from the id: "..." field
id_m = re.search(r'\bid\s*:\s*"([^"]+)"', text)
culture_id = id_m.group(1) if id_m else os.path.basename(fpath)
given_by_culture[culture_id] = set(extract_names_from_block(text, 'given_names'))
family_by_culture[culture_id] = set(extract_names_from_block(text, 'family_names'))
cultures = list(given_by_culture.keys())
collisions_found = False
# Check given_names collisions
all_given_names = {} # name -> list of cultures
for culture, names in given_by_culture.items():
for name in names:
all_given_names.setdefault(name, []).append(culture)
for name, cultures_with_name in sorted(all_given_names.items()):
if len(cultures_with_name) > 1:
collisions_found = True
print(f"COLLISION given_names: \"{name}\" in {', '.join(sorted(cultures_with_name))}")
# Check family_names collisions
all_family_names = {} # name -> list of cultures
for culture, names in family_by_culture.items():
for name in names:
all_family_names.setdefault(name, []).append(culture)
for name, cultures_with_name in sorted(all_family_names.items()):
if len(cultures_with_name) > 1:
collisions_found = True
print(f"COLLISION family_names: \"{name}\" in {', '.join(sorted(cultures_with_name))}")
if not collisions_found:
print(f"OK: no name collisions across {len(cultures)} culture(s): {', '.join(sorted(cultures))}")
sys.exit(0)
else:
sys.exit(1)
PYEOF
exit $?
fi
# --- End name collision mode ----------------------------------------------
if [ $# -lt 2 ]; then
echo "Usage: tooling/validate-ron <file.ron> <zone|zone_type|culture>"
echo " tooling/validate-ron --check-name-collisions <directory>"
echo ""
echo "Validates a RON file against the Rust struct schema."
echo "Schema types:"
echo " zone_type — ZoneTypeTemplate (D-142 zone-type template)"
echo " zone — ZoneSpec (legacy zone identity spec)"
echo " culture — CultureProfile (culture profile)"
echo ""
echo "Name collision check:"
echo " --check-name-collisions <dir> Scan culture-*.ron files for shared name pool entries"
exit 1
fi
# Check file exists before resolving — realpath gives unhelpful errors otherwise
if [ ! -f "$1" ]; then
echo "Error: file not found: $1"
exit 1
fi
FILE="$(realpath "$1")"
SCHEMA="$2"
cd "$(dirname "$0")/../server"
exec cargo run --quiet --bin validate_ron -- "$FILE" "$SCHEMA"