refactor(tooling): T-1286 — generate, pr and dev become reach domains
Twelve scripts retired, three domains registered. `reach` now covers nine. generate: `generate-brands` and `generate-corporations` were the second and third copies of the same 24-line build-if-missing-then-exec bash `tooling/atlas` carried, so they collapsed into `core.process.cargo_binary` rather than being ported. `import_economics` shelled out to the first of those, so it now calls that helper — `generated_brands.toml` comes back byte-identical, and the stamp registry swaps the retired wrapper for `core/process.py`. pr: `watchlist-diff` derives its watched set from `generator_sources.py` instead of restating it, so it cannot drift from the stamp check. dev: the environment scripts split decision from performing, per D-263's guarded-exec rule. `godot_plan()` and `worktree_plan()` decide what would happen; `install_godot()`, `install_rust()` and `setup_worktree()` do it. `tooling/test_environment.py` pins the version pin, both override precedences, the already-current skip, the platform refusal and both worktree refusals — none of them performed. `make setup` now installs reach first, since the targets that install rust and godot are reach verbs. Two live bugs found while porting: - The clerk read its decision index from `decisions/README.md`, a path that stopped existing when the DQR tree moved to `governance/`. Every clerk agent has been grepping blind; its prompt pointed at the same dead directory. - The conformance exec-check matched any `x.system()` regardless of receiver, so `platform.system()` read as `os.system()`. Narrowed and re-proved against a real mutant. `process.run` gains `input=`, `timeout=` and a `ProcessTimeout` subclass so a killed run stays distinguishable from a verdict. The pre-push hook no longer merges the clerk's stderr into its stdout — under streaming the last merged line is a JSONL event, which would read as an unrecognised verdict and block. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Executable
+260
@@ -0,0 +1,260 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Clerk pre-push review — checks D-record consistency, ticket drift, and decision
|
||||
contradictions against the diff that would be pushed.
|
||||
|
||||
Reviewed one commit at a time — commits are the logical units (each /git-commit is
|
||||
one coherent change), so each clerk agent reviews a self-contained change *and its
|
||||
commit message*. Commits are reviewed by a bounded pool of parallel clerk agents
|
||||
and the per-commit verdicts aggregated.
|
||||
|
||||
Three outcomes per commit:
|
||||
APPROVED no contradiction found.
|
||||
REJECTED a concrete contradiction with a named, active D-record. HARD BLOCK.
|
||||
INCOMPLETE the review could not finish — timed out, ran out of turns, or emitted
|
||||
no clear verdict. This is NOT a contradiction; it is non-blocking by
|
||||
default (the push proceeds with a warning). Raise the budget knobs or
|
||||
add a `Clerk-Skip:` trailer to avoid it.
|
||||
|
||||
Why INCOMPLETE exists: an unfinished review must never read as "hard contradiction
|
||||
found." The previous version defaulted no-verdict / max-turns to REJECTED, which
|
||||
turned every slow review into a false block (e.g. a CHANGELOG-only commit).
|
||||
|
||||
Safety valve: a commit message with a `Clerk-Skip:` trailer line is auto-approved
|
||||
without spawning an agent — for bulk content commits (e.g. shipping thousands of
|
||||
generated planetary description files) where D-record review is moot. A *trailer*
|
||||
(a line starting with `Clerk-Skip:`) is required, so merely mentioning the token in
|
||||
prose or a subject line does not trip the valve.
|
||||
|
||||
Outputs exactly one word as the LAST stdout line: APPROVED, REJECTED, or INCOMPLETE.
|
||||
Writes verbose findings to .cache/pre-push-review.md.
|
||||
|
||||
Exit codes:
|
||||
0 = APPROVED or INCOMPLETE (non-blocking)
|
||||
1 = REJECTED (>=1 commit contradicts an active D-record)
|
||||
|
||||
Env knobs:
|
||||
SR_CLERK_COMMIT_BUDGET per-commit diff char cap (default 120000, truncates)
|
||||
SR_CLERK_WORKERS parallel clerk agents (default 4)
|
||||
SR_CLERK_MAX_TURNS turns per clerk agent (default 15)
|
||||
SR_CLERK_TIMEOUT per-commit timeout seconds (default 300)
|
||||
|
||||
Usage:
|
||||
reach dev clerk run the review, print verdict
|
||||
reach dev clerk --plan print the per-commit plan only (no clerk spawned)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
|
||||
from tooling.core import config, console, process
|
||||
from tooling.core.errors import ReachError
|
||||
|
||||
REPO_ROOT = config.repo_root()
|
||||
|
||||
CACHE_DIR = REPO_ROOT / ".cache"
|
||||
FINDINGS_FILE = CACHE_DIR / "pre-push-review.md"
|
||||
|
||||
COMMIT_BUDGET = int(os.environ.get("SR_CLERK_COMMIT_BUDGET", "120000"))
|
||||
WORKERS = int(os.environ.get("SR_CLERK_WORKERS", "4"))
|
||||
MAX_TURNS = os.environ.get("SR_CLERK_MAX_TURNS", "15")
|
||||
TIMEOUT_SECONDS = int(os.environ.get("SR_CLERK_TIMEOUT", "300"))
|
||||
|
||||
# Safety valve: a commit message with a "Clerk-Skip:" trailer line is auto-approved
|
||||
# (no agent). A trailer (line-start) is required so that merely mentioning the token
|
||||
# in prose or a subject line does not trip the valve.
|
||||
SKIP_TRAILER = re.compile(r"(?im)^[ \t]*clerk-skip[ \t]*:")
|
||||
|
||||
CLERK_PROMPT = """You are the CLERK, the institutional guardrail for The Settled Reach.
|
||||
|
||||
You are reviewing ONE COMMIT ({label}) from a larger pre-push for D-record
|
||||
consistency, ticket drift, and decision contradictions.
|
||||
|
||||
## governance/ domain index
|
||||
|
||||
{index}
|
||||
|
||||
## The commit (message + diff)
|
||||
|
||||
{diff}
|
||||
|
||||
## Your task
|
||||
|
||||
Be efficient — a few targeted greps of governance/decisions/*.md, then conclude. Check:
|
||||
- Does this commit contradict any active D-record?
|
||||
- If the commit's code/text cites a D/Q/R-ID, does that ID exist and is it active?
|
||||
- If the commit message cites ticket #NNN, does the change match the ticket?
|
||||
Note any relevant open Q-records.
|
||||
|
||||
## Output format (REQUIRED)
|
||||
|
||||
Write brief findings, then on the VERY LAST LINE output exactly one word:
|
||||
APPROVED or REJECTED.
|
||||
|
||||
- REJECTED *only* for a concrete contradiction with a named, active D-record.
|
||||
- Drift, stale references, open Q-records, and suggestions are findings, NOT blocks -> APPROVED.
|
||||
- If you did not find a concrete contradiction, output APPROVED — even if you could
|
||||
not check everything. "Unsure / didn't finish" means APPROVED, never REJECTED.
|
||||
"""
|
||||
|
||||
|
||||
def _git(*args: str) -> str:
|
||||
return process.run(["git", *args], cwd=REPO_ROOT).stdout
|
||||
|
||||
|
||||
def base_range() -> str:
|
||||
"""The commit range that would be pushed (base..HEAD)."""
|
||||
branch = _git("branch", "--show-current").strip()
|
||||
for ref in [f"origin/{branch}", "origin/main"]:
|
||||
probe = process.run(
|
||||
["git", "rev-parse", "--verify", ref], cwd=REPO_ROOT, check=False
|
||||
)
|
||||
if probe.returncode == 0:
|
||||
return f"{ref}..HEAD"
|
||||
return "HEAD~1..HEAD"
|
||||
|
||||
|
||||
def list_commits(rng: str) -> list[str]:
|
||||
"""SHAs in the push range, oldest first."""
|
||||
return [s for s in _git("rev-list", "--reverse", rng).splitlines() if s.strip()]
|
||||
|
||||
|
||||
def commit_unit(sha):
|
||||
"""Return (subject, text, skip) for one commit.
|
||||
|
||||
`skip` is True when the commit message has a `Clerk-Skip:` trailer (the safety
|
||||
valve); `text` is the message + diff, truncated to COMMIT_BUDGET.
|
||||
"""
|
||||
subject = _git("show", "-s", "--format=%h %s", sha).strip()
|
||||
message = _git("show", "-s", "--format=%B", sha)
|
||||
skip = bool(SKIP_TRAILER.search(message))
|
||||
text = _git("show", "--format=fuller", sha)
|
||||
if len(text) > COMMIT_BUDGET:
|
||||
text = text[:COMMIT_BUDGET] + f"\n\n... (commit diff truncated; {len(text)} total chars)\n"
|
||||
return subject, text, skip
|
||||
|
||||
|
||||
def run_clerk(label, diff_text, index):
|
||||
"""Spawn one clerk agent over a commit; return (verdict, findings).
|
||||
|
||||
verdict is APPROVED, REJECTED, or INCOMPLETE. INCOMPLETE covers timeout, empty
|
||||
output, and no-clear-verdict — none of which is a contradiction.
|
||||
"""
|
||||
prompt = CLERK_PROMPT.format(label=label, index=index, diff=diff_text)
|
||||
try:
|
||||
result = process.run(
|
||||
["claude", "-p", "--model", "sonnet", "--max-turns", str(MAX_TURNS)],
|
||||
cwd=REPO_ROOT,
|
||||
input=prompt,
|
||||
timeout=TIMEOUT_SECONDS,
|
||||
check=False,
|
||||
missing_fix="install the claude CLI, or set SR_CLERK=0 to skip the review",
|
||||
)
|
||||
output = result.stdout.strip()
|
||||
except process.ProcessTimeout:
|
||||
return "INCOMPLETE", f"{label}: review timed out after {TIMEOUT_SECONDS}s (not a contradiction)."
|
||||
|
||||
if not output:
|
||||
return "INCOMPLETE", f"{label}: clerk produced no output (not a contradiction)."
|
||||
|
||||
last_line = output.splitlines()[-1].strip().upper()
|
||||
if last_line == "APPROVED":
|
||||
return "APPROVED", output
|
||||
if last_line == "REJECTED":
|
||||
return "REJECTED", output
|
||||
return "INCOMPLETE", output + "\n\n(No clear verdict on last line — recorded as INCOMPLETE, not a contradiction.)"
|
||||
|
||||
|
||||
def load_index() -> str:
|
||||
"""The decision-domain index the clerk greps from.
|
||||
|
||||
`governance/README.md` since the DQR tree moved; the old `decisions/README.md`
|
||||
path silently resolved to "not found", which handed every clerk an empty
|
||||
index and made it grep blind.
|
||||
"""
|
||||
readme = REPO_ROOT / "governance" / "README.md"
|
||||
return readme.read_text()[:8000] if readme.exists() else "(governance/README.md not found)"
|
||||
|
||||
|
||||
def run(plan_only: bool = False) -> str:
|
||||
"""Review the push range commit by commit. Returns the overall verdict."""
|
||||
CACHE_DIR.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
rng = base_range()
|
||||
commits = list_commits(rng)
|
||||
if not commits:
|
||||
FINDINGS_FILE.write_text("# Clerk Review\n\nNo commits to review.\n\nVerdict: APPROVED\n")
|
||||
console.event("clerk: no commits to review")
|
||||
console.out("APPROVED")
|
||||
return "APPROVED"
|
||||
|
||||
units = [commit_unit(sha) for sha in commits] # [(subject, text, skip), ...]
|
||||
|
||||
if plan_only:
|
||||
console.out(f"clerk plan: {len(commits)} commit(s) over {rng} "
|
||||
f"(budget {COMMIT_BUDGET} chars, {WORKERS} workers, "
|
||||
f"{MAX_TURNS} turns each)")
|
||||
for i, (subject, text, skip) in enumerate(units):
|
||||
tag = "SKIP (Clerk-Skip:)" if skip else "review"
|
||||
console.out(
|
||||
f" commit {i + 1}/{len(commits)}: {len(text):>9} chars [{tag}] {subject}"
|
||||
)
|
||||
return "PLAN"
|
||||
|
||||
index = load_index()
|
||||
reviewable = sum(1 for _, _, skip in units if not skip)
|
||||
console.event(
|
||||
f"clerk: {len(commits)} commit(s) — {reviewable} to review, "
|
||||
f"{len(commits) - reviewable} auto-approved (Clerk-Skip:); {WORKERS} parallel",
|
||||
phase="clerk",
|
||||
)
|
||||
|
||||
def task(i, subject, text, skip):
|
||||
label = f"commit {i + 1}/{len(commits)} ({subject})"
|
||||
if skip:
|
||||
return i, label, "APPROVED", f"{label}: auto-approved via Clerk-Skip: trailer."
|
||||
verdict, findings = run_clerk(label, text, index)
|
||||
return i, label, verdict, findings
|
||||
|
||||
results = [None] * len(commits)
|
||||
with ThreadPoolExecutor(max_workers=WORKERS) as ex:
|
||||
futures = [ex.submit(task, i, s, t, skip) for i, (s, t, skip) in enumerate(units)]
|
||||
for fut in as_completed(futures):
|
||||
i, label, verdict, findings = fut.result()
|
||||
results[i] = (label, verdict, findings)
|
||||
console.event(f"clerk: {label} — {verdict}", phase="clerk")
|
||||
|
||||
verdicts = [r[1] for r in results]
|
||||
if "REJECTED" in verdicts:
|
||||
overall = "REJECTED"
|
||||
elif "INCOMPLETE" in verdicts:
|
||||
overall = "INCOMPLETE"
|
||||
else:
|
||||
overall = "APPROVED"
|
||||
|
||||
n_rej = verdicts.count("REJECTED")
|
||||
n_inc = verdicts.count("INCOMPLETE")
|
||||
parts = [
|
||||
f"# Clerk Pre-Push Review\n\n**Overall verdict: {overall}**\n",
|
||||
f"Reviewed {len(commits)} commit(s) over {rng} — "
|
||||
f"{verdicts.count('APPROVED')} approved, {n_rej} rejected, {n_inc} incomplete.\n",
|
||||
]
|
||||
for label, verdict, findings in results:
|
||||
parts.append(f"\n---\n\n## {label} — {verdict}\n\n{findings}\n")
|
||||
FINDINGS_FILE.write_text("\n".join(parts))
|
||||
|
||||
console.event(
|
||||
f"clerk: overall verdict — {overall} (details: {FINDINGS_FILE})", phase="clerk"
|
||||
)
|
||||
console.out(overall)
|
||||
|
||||
if overall == "REJECTED":
|
||||
raise ReachError(
|
||||
f"clerk rejected {n_rej} commit(s) — a named, active D-record is contradicted",
|
||||
fix=f"read {FINDINGS_FILE.relative_to(REPO_ROOT)}; amend the commit or "
|
||||
"the decision, or add a `Clerk-Skip:` trailer if the clerk is wrong",
|
||||
)
|
||||
return overall
|
||||
@@ -0,0 +1,203 @@
|
||||
"""Environment setup — the decisions, separated from the performing (D-263).
|
||||
|
||||
These three ported worst and mattered most to get right. `install-godot`
|
||||
downloads and unzips a pinned build, `install-rust` drives `rustup`,
|
||||
`worktree-setup` manipulates git worktrees — none of which can be exercised in
|
||||
a gate without actually doing it.
|
||||
|
||||
So the shape here is deliberate: **every function that decides is pure and
|
||||
importable, and every function that acts is a thin call through
|
||||
`core.process.run`.** `godot_plan()` answers "what would be downloaded, and is
|
||||
it needed" without touching the network; `install_godot()` performs it. The
|
||||
test of a correct port is whether the decisions can be exercised WITHOUT
|
||||
performing them — a rewrite that cannot be tested has to be trusted instead,
|
||||
and trusting an installer is how a working environment becomes an
|
||||
unreproducible one.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import platform
|
||||
import shutil
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
from tooling.core import config, console, process
|
||||
from tooling.core.errors import ReachError
|
||||
|
||||
DEFAULT_GODOT_VERSION = "4.6"
|
||||
INSTALL_DIR = Path.home() / "bin"
|
||||
GODOT_BINARY = INSTALL_DIR / "godot4"
|
||||
|
||||
# Godot publishes one archive per platform triple; an unsupported pair is a
|
||||
# clear failure rather than a download that 404s.
|
||||
PLATFORMS = {
|
||||
("Linux", "x86_64"): "linux.x86_64",
|
||||
("Linux", "aarch64"): "linux.arm64",
|
||||
("Darwin", "x86_64"): "macos.universal",
|
||||
("Darwin", "arm64"): "macos.universal",
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class GodotPlan:
|
||||
"""What installing Godot would do — decided without doing any of it."""
|
||||
|
||||
wanted: str
|
||||
installed: str | None
|
||||
platform_tag: str
|
||||
url: str
|
||||
filename: str
|
||||
|
||||
@property
|
||||
def already_current(self) -> bool:
|
||||
return self.installed == self.wanted
|
||||
|
||||
|
||||
def godot_plan(wanted: str | None = None) -> GodotPlan:
|
||||
"""Decide what an install would fetch. Pure apart from reading the binary."""
|
||||
wanted = wanted or os.environ.get("GODOT_VERSION", DEFAULT_GODOT_VERSION)
|
||||
system, machine = platform.system(), platform.machine()
|
||||
tag = PLATFORMS.get((system, machine))
|
||||
if tag is None:
|
||||
raise ReachError(
|
||||
f"unsupported platform: {system} {machine}",
|
||||
fix="install Godot manually from https://godotengine.org/download",
|
||||
)
|
||||
|
||||
filename = f"Godot_v{wanted}-stable_{tag}.zip"
|
||||
return GodotPlan(
|
||||
wanted=wanted,
|
||||
installed=installed_godot_version(),
|
||||
platform_tag=tag,
|
||||
filename=filename,
|
||||
url=(
|
||||
"https://github.com/godotengine/godot/releases/download/"
|
||||
f"{wanted}-stable/{filename}"
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def installed_godot_version() -> str | None:
|
||||
"""The major.minor already on disk, or None. Never raises."""
|
||||
if not GODOT_BINARY.is_file() or not os.access(GODOT_BINARY, os.X_OK):
|
||||
return None
|
||||
result = process.run([str(GODOT_BINARY), "--version"], check=False)
|
||||
first = (result.stdout or "").splitlines()
|
||||
if not first:
|
||||
return None
|
||||
return ".".join(first[0].split(".")[:2])
|
||||
|
||||
|
||||
def install_godot(wanted: str | None = None) -> GodotPlan:
|
||||
"""Perform the install. The decision is `godot_plan`; this is the doing."""
|
||||
plan = godot_plan(wanted)
|
||||
if plan.already_current:
|
||||
console.verdict(f"Godot {plan.wanted} already installed at {GODOT_BINARY}")
|
||||
return plan
|
||||
if plan.installed:
|
||||
console.event(
|
||||
f"Godot found but version is {plan.installed}, want {plan.wanted}",
|
||||
level="warn",
|
||||
)
|
||||
|
||||
INSTALL_DIR.mkdir(parents=True, exist_ok=True)
|
||||
import tempfile
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
archive = Path(tmp) / plan.filename
|
||||
console.event(f"downloading Godot {plan.wanted} for {plan.platform_tag}", phase="godot")
|
||||
process.run(
|
||||
["curl", "-fSL", "--progress-bar", "-o", str(archive), plan.url],
|
||||
capture=False,
|
||||
fix=f"check that GODOT_VERSION={plan.wanted} is a real release",
|
||||
missing_fix="install curl",
|
||||
)
|
||||
console.event("extracting", phase="godot")
|
||||
process.run(
|
||||
["unzip", "-q", "-o", str(archive), "-d", tmp],
|
||||
fix="the archive may be truncated — re-run to download it again",
|
||||
missing_fix="install unzip",
|
||||
)
|
||||
extracted = next(
|
||||
(p for p in Path(tmp).iterdir() if p.is_file() and p.name.startswith("Godot")),
|
||||
None,
|
||||
)
|
||||
if extracted is None:
|
||||
raise ReachError(
|
||||
"the Godot archive contained no binary",
|
||||
fix="the download may be corrupt — delete it and re-run",
|
||||
)
|
||||
shutil.move(str(extracted), GODOT_BINARY)
|
||||
GODOT_BINARY.chmod(0o755)
|
||||
|
||||
console.verdict(f"Godot {plan.wanted} installed at {GODOT_BINARY}")
|
||||
return plan
|
||||
|
||||
|
||||
def install_rust() -> None:
|
||||
"""Ensure rustup, clippy and rustfmt are present."""
|
||||
if shutil.which("cargo") is None:
|
||||
raise ReachError(
|
||||
"cargo is not installed",
|
||||
fix="install Rust from https://rustup.rs, then re-run — this command "
|
||||
"adds the components but does not bootstrap the toolchain",
|
||||
)
|
||||
for component in ("clippy", "rustfmt"):
|
||||
console.event(f"ensuring {component}", phase="rust")
|
||||
process.run(
|
||||
["rustup", "component", "add", component],
|
||||
check=False,
|
||||
missing_fix="install rustup from https://rustup.rs",
|
||||
)
|
||||
console.verdict("Rust toolchain ready — clippy and rustfmt present")
|
||||
|
||||
|
||||
def worktree_plan(branch: str) -> Path:
|
||||
"""Where a worktree for `branch` would go. Raises if it cannot be created.
|
||||
|
||||
Both refusals are the original's and both are worth keeping: a worktree
|
||||
created from inside a worktree nests confusingly, and silently reusing an
|
||||
existing directory is how two branches end up sharing one tree.
|
||||
"""
|
||||
root = config.repo_root()
|
||||
if (root / ".git").is_file():
|
||||
raise ReachError(
|
||||
"this is a worktree, not the main checkout",
|
||||
fix="run this from the main checkout — nesting worktrees confuses "
|
||||
"both git and the tools that resolve the repo root",
|
||||
)
|
||||
target = root / ".worktrees" / branch
|
||||
if target.exists():
|
||||
raise ReachError(
|
||||
f"{target} already exists",
|
||||
fix=f"reuse it, or remove it first with: git worktree remove {target}",
|
||||
)
|
||||
return target
|
||||
|
||||
|
||||
def setup_worktree(branch: str, start: str = "HEAD") -> Path:
|
||||
"""Create a worktree and link the venv into it."""
|
||||
target = worktree_plan(branch)
|
||||
root = config.repo_root()
|
||||
|
||||
process.run(
|
||||
["git", "worktree", "add", str(target), "-b", branch, start],
|
||||
cwd=root,
|
||||
fix=f"check that {start} is a valid ref and {branch} is not already a branch",
|
||||
)
|
||||
|
||||
venv = root / ".venv"
|
||||
if venv.is_dir() and not (target / ".venv").exists():
|
||||
(target / ".venv").symlink_to(venv)
|
||||
console.event(f"linked .venv -> {venv}")
|
||||
|
||||
console.verdict(
|
||||
f"worktree ready: {target}\n"
|
||||
f" pql in this worktree needs --vault {target} (FR-4)\n"
|
||||
" tea runs from the main checkout — its go-git cannot read a linked worktree\n"
|
||||
" reach follows whichever checkout it was installed from; `make reach-repoint` "
|
||||
"if you want it to follow this one"
|
||||
)
|
||||
return target
|
||||
Executable
+274
@@ -0,0 +1,274 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Performance baseline — tick timing, memory, and shadowcast scaling.
|
||||
|
||||
Builds the server in release, runs the benchmark suite, and either writes
|
||||
tests/perf/baseline.json or compares against the committed one. The compare
|
||||
path exits non-zero on a >20% regression or a p95 over the D-026 tick budget.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from datetime import datetime, timezone
|
||||
|
||||
from tooling.core import config, console, process
|
||||
from tooling.core.errors import ReachError
|
||||
|
||||
ROOT = config.repo_root()
|
||||
PERF_DIR = ROOT / "tests" / "perf"
|
||||
BASELINE_FILE = PERF_DIR / "baseline.json"
|
||||
|
||||
# Tick budget from D-026: 100ms per tick at 10 tps floor (D-031).
|
||||
# If tick rate changes, update this constant.
|
||||
TICK_BUDGET_US = 100_000 # 100ms
|
||||
|
||||
|
||||
def run_command(cmd, **kwargs):
|
||||
"""Run a command in the server directory and return the result."""
|
||||
return process.run(cmd, cwd=ROOT / "server", check=False, **kwargs)
|
||||
|
||||
|
||||
def get_git_info():
|
||||
"""Current commit and branch. check=False — a shallow or detached tree is
|
||||
not an error here, it just means the stamp carries less."""
|
||||
commit = process.run(
|
||||
["git", "rev-parse", "--short", "HEAD"], cwd=ROOT, check=False
|
||||
).stdout.strip()
|
||||
branch = process.run(
|
||||
["git", "rev-parse", "--abbrev-ref", "HEAD"], cwd=ROOT, check=False
|
||||
).stdout.strip()
|
||||
return {"commit": commit, "branch": branch}
|
||||
|
||||
|
||||
def run_tick_benchmark():
|
||||
"""Run perf_tick_timing test and parse PERF_RESULT JSON."""
|
||||
console.event("running tick timing benchmark (release mode)", phase="bench")
|
||||
result = run_command([
|
||||
"cargo", "test", "--release", "--test", "perf_bench",
|
||||
"--", "--ignored", "--nocapture", "perf_tick_timing",
|
||||
])
|
||||
|
||||
if result.returncode != 0:
|
||||
console.event(f"tick benchmark exited {result.returncode}", level="error")
|
||||
if result.stderr:
|
||||
# Last 20 lines of stderr are the diagnostic; the rest is build noise.
|
||||
for line in result.stderr.strip().splitlines()[-20:]:
|
||||
console.event(line, level="error")
|
||||
return None
|
||||
|
||||
# Parse PERF_RESULT: line from stdout
|
||||
for line in result.stdout.splitlines():
|
||||
if line.startswith("PERF_RESULT:"):
|
||||
json_str = line[len("PERF_RESULT:"):]
|
||||
return json.loads(json_str)
|
||||
|
||||
console.event("no PERF_RESULT found in test output", level="warn")
|
||||
return None
|
||||
|
||||
|
||||
def run_shadowcast_benchmark():
|
||||
"""Run shadowcast benchmark and parse structured output."""
|
||||
console.event("running shadowcast benchmark (release mode)", phase="bench")
|
||||
result = run_command([
|
||||
"cargo", "test", "--release", "--test", "shadowcast_bench",
|
||||
"--", "--ignored", "--nocapture", "benchmark_symmetric_vs_recursive",
|
||||
])
|
||||
|
||||
if result.returncode != 0:
|
||||
console.event(
|
||||
f"shadowcast benchmark exited {result.returncode}", level="error"
|
||||
)
|
||||
return None
|
||||
|
||||
configs = []
|
||||
current = {}
|
||||
for line in result.stdout.splitlines():
|
||||
line = line.strip()
|
||||
|
||||
m = re.match(
|
||||
r"Map: (\d+)x(\d+), Density: (.+), Range: (\d+), Iterations: (\d+)",
|
||||
line,
|
||||
)
|
||||
if m:
|
||||
# New config block — flush previous if complete
|
||||
if current.get("map_size"):
|
||||
configs.append(current)
|
||||
current = {
|
||||
"map_size": int(m.group(1)),
|
||||
"density": m.group(3),
|
||||
"range": int(m.group(4)),
|
||||
"iterations": int(m.group(5)),
|
||||
}
|
||||
continue
|
||||
|
||||
m = re.match(r"Symmetric:\s+([0-9.]+)ms total, ([0-9.]+).s/call", line)
|
||||
if m:
|
||||
current["symmetric_total_ms"] = float(m.group(1))
|
||||
current["symmetric_per_call_us"] = float(m.group(2))
|
||||
continue
|
||||
|
||||
m = re.match(r"Recursive:\s+([0-9.]+)ms total, ([0-9.]+).s/call", line)
|
||||
if m:
|
||||
current["recursive_total_ms"] = float(m.group(1))
|
||||
current["recursive_per_call_us"] = float(m.group(2))
|
||||
continue
|
||||
|
||||
# Flush last config
|
||||
if current.get("map_size"):
|
||||
configs.append(current)
|
||||
|
||||
return {"configs": configs} if configs else None
|
||||
|
||||
|
||||
def compare_baselines(old, new):
|
||||
"""Compare two baselines and report regressions. Returns list of regression strings."""
|
||||
regressions = []
|
||||
improvements = []
|
||||
|
||||
old_tick = old.get("tick_timing", {})
|
||||
new_tick = new.get("tick_timing", {})
|
||||
|
||||
if old_tick and new_tick:
|
||||
# Mean tick time regression (>20% = warning)
|
||||
old_mean = old_tick.get("mean_us", 0)
|
||||
new_mean = new_tick.get("mean_us", 0)
|
||||
if old_mean > 0:
|
||||
change = (new_mean - old_mean) / old_mean * 100
|
||||
if change > 20:
|
||||
regressions.append(
|
||||
f"mean tick time {old_mean}us -> {new_mean}us (+{change:.1f}%)"
|
||||
)
|
||||
elif change < -20:
|
||||
improvements.append(
|
||||
f"mean tick time {old_mean}us -> {new_mean}us ({change:.1f}%)"
|
||||
)
|
||||
|
||||
# p95 tick time regression
|
||||
old_p95 = old_tick.get("p95_us", 0)
|
||||
new_p95 = new_tick.get("p95_us", 0)
|
||||
if old_p95 > 0:
|
||||
change = (new_p95 - old_p95) / old_p95 * 100
|
||||
if change > 20:
|
||||
regressions.append(
|
||||
f"p95 tick time {old_p95}us -> {new_p95}us (+{change:.1f}%)"
|
||||
)
|
||||
elif change < -20:
|
||||
improvements.append(
|
||||
f"p95 tick time {old_p95}us -> {new_p95}us ({change:.1f}%)"
|
||||
)
|
||||
|
||||
# Absolute budget check
|
||||
new_p95 = new.get("tick_timing", {}).get("p95_us", 0)
|
||||
if new_p95 > TICK_BUDGET_US:
|
||||
regressions.append(
|
||||
f"p95 {new_p95}us exceeds {TICK_BUDGET_US}us tick budget (D-026)"
|
||||
)
|
||||
|
||||
return regressions, improvements
|
||||
|
||||
|
||||
def run(compare_mode: bool = False) -> None:
|
||||
"""Build release, run the benchmarks, then write or compare the baseline."""
|
||||
console.event("building server (release)", phase="build")
|
||||
build = run_command(["cargo", "build", "--release"])
|
||||
if build.returncode != 0:
|
||||
for line in build.stderr.strip().splitlines()[-20:]:
|
||||
console.event(line, level="error")
|
||||
raise ReachError(
|
||||
"the release build failed, so there is nothing to benchmark",
|
||||
fix="fix the build errors above, then re-run — a perf number from a "
|
||||
"stale binary is worse than no number",
|
||||
)
|
||||
|
||||
tick_results = run_tick_benchmark()
|
||||
shadowcast_results = run_shadowcast_benchmark()
|
||||
|
||||
if not tick_results:
|
||||
raise ReachError(
|
||||
"the tick benchmark produced no result — no baseline generated",
|
||||
fix="run `cargo test --release --test perf_bench -- --ignored "
|
||||
"--nocapture perf_tick_timing` in server/ to see why it failed",
|
||||
)
|
||||
|
||||
# Assemble baseline
|
||||
git_info = get_git_info()
|
||||
baseline = {
|
||||
"timestamp": datetime.now(timezone.utc).isoformat(),
|
||||
"git": git_info,
|
||||
"tick_timing": tick_results.get("tick_timing", {}),
|
||||
"entities": tick_results.get("entities", {}),
|
||||
"memory": tick_results.get("memory", {}),
|
||||
}
|
||||
if shadowcast_results:
|
||||
baseline["shadowcast"] = shadowcast_results
|
||||
|
||||
# Report
|
||||
tt = baseline["tick_timing"]
|
||||
console.out("--- Results ---")
|
||||
console.out(f"Git: {git_info['commit']} ({git_info['branch']})")
|
||||
console.out(f"Tick timing ({tt.get('measured_ticks', '?')} ticks, "
|
||||
f"{tt.get('warmup_ticks', '?')} warmup):")
|
||||
console.out(f" min: {tt.get('min_us', '?')}us")
|
||||
console.out(f" mean: {tt.get('mean_us', '?')}us")
|
||||
console.out(f" p95: {tt.get('p95_us', '?')}us")
|
||||
console.out(f" max: {tt.get('max_us', '?')}us")
|
||||
|
||||
ent = baseline["entities"]
|
||||
console.out(f"Entities: avg {ent.get('avg_per_snapshot', '?')}, "
|
||||
f"max {ent.get('max_per_snapshot', '?')}")
|
||||
|
||||
rss = baseline["memory"].get("rss_kb")
|
||||
if rss:
|
||||
console.out(f"Memory: {rss} KB RSS ({rss / 1024:.1f} MB)")
|
||||
|
||||
if shadowcast_results:
|
||||
n = len(shadowcast_results.get("configs", []))
|
||||
console.out(f"Shadowcast: {n} configurations benchmarked")
|
||||
|
||||
# Budget check
|
||||
p95 = tt.get("p95_us", 0)
|
||||
if p95 > TICK_BUDGET_US:
|
||||
console.out(f"BUDGET EXCEEDED: p95 {p95}us > {TICK_BUDGET_US}us (D-026)")
|
||||
else:
|
||||
budget_pct = p95 / TICK_BUDGET_US * 100 if TICK_BUDGET_US else 0
|
||||
console.out(
|
||||
f"Budget: {budget_pct:.1f}% of {TICK_BUDGET_US}us tick budget (D-026)"
|
||||
)
|
||||
|
||||
# Compare with previous baseline if it exists
|
||||
regressions: list[str] = []
|
||||
if BASELINE_FILE.exists():
|
||||
old_baseline = json.loads(BASELINE_FILE.read_text())
|
||||
old_commit = old_baseline.get("git", {}).get("commit", "?")
|
||||
console.out(f"--- Comparison vs {old_commit} ---")
|
||||
regressions, improvements = compare_baselines(old_baseline, baseline)
|
||||
for r in regressions:
|
||||
console.out(f" REGRESSION: {r}")
|
||||
for i in improvements:
|
||||
console.out(f" IMPROVEMENT: {i}")
|
||||
if not regressions and not improvements:
|
||||
console.out(" No significant changes.")
|
||||
elif compare_mode:
|
||||
raise ReachError(
|
||||
f"--compare needs a saved baseline at {BASELINE_FILE.relative_to(ROOT)}",
|
||||
fix="run `reach dev perf` once without --compare to create one",
|
||||
)
|
||||
|
||||
if compare_mode:
|
||||
if regressions:
|
||||
raise ReachError(
|
||||
f"{len(regressions)} performance regression(s) detected",
|
||||
fix="see the REGRESSION lines above; if the change is intended, "
|
||||
"re-run without --compare to accept it as the new baseline",
|
||||
)
|
||||
console.verdict("perf: no regressions against the saved baseline")
|
||||
return
|
||||
|
||||
# Save baseline (strip per-tick array — too noisy for git diffs)
|
||||
PERF_DIR.mkdir(parents=True, exist_ok=True)
|
||||
committed = json.loads(json.dumps(baseline))
|
||||
committed["tick_timing"].pop("all_us", None)
|
||||
BASELINE_FILE.write_text(json.dumps(committed, indent=2) + "\n")
|
||||
|
||||
console.verdict(f"baseline written to {BASELINE_FILE.relative_to(ROOT)}")
|
||||
@@ -7,7 +7,7 @@ import typer
|
||||
from tooling.core import cli, console
|
||||
from tooling.core.command import command
|
||||
from tooling.core.errors import ReachError
|
||||
from tooling.domains.dev import service
|
||||
from tooling.domains.dev import environment, service
|
||||
|
||||
app = cli.domain("dev", "Developer environment and self-diagnosis.")
|
||||
|
||||
@@ -17,6 +17,79 @@ def _domain() -> None:
|
||||
"""Keeps `dev` a group (Typer collapses a single-command app)."""
|
||||
|
||||
|
||||
@app.command("install-godot")
|
||||
@command
|
||||
def install_godot(
|
||||
version: str = typer.Option(None, "--version", help="Godot version to install."),
|
||||
plan_only: bool = typer.Option(
|
||||
False, "--plan", help="Report what would be downloaded without doing it."
|
||||
),
|
||||
) -> None:
|
||||
"""Install the pinned Godot build to ~/bin/godot4, if it is not already there."""
|
||||
if plan_only:
|
||||
plan = environment.godot_plan(version)
|
||||
console.out(f"wanted {plan.wanted}")
|
||||
console.out(f"installed {plan.installed or '(none)'}")
|
||||
console.out(f"platform {plan.platform_tag}")
|
||||
console.out(f"url {plan.url}")
|
||||
console.verdict(
|
||||
"already current — nothing to do"
|
||||
if plan.already_current
|
||||
else "would download and install"
|
||||
)
|
||||
return
|
||||
environment.install_godot(version)
|
||||
|
||||
|
||||
@app.command("install-rust")
|
||||
@command
|
||||
def install_rust() -> None:
|
||||
"""Ensure clippy and rustfmt are present on the Rust toolchain."""
|
||||
environment.install_rust()
|
||||
|
||||
|
||||
@app.command("worktree")
|
||||
@command
|
||||
def worktree(
|
||||
branch: str = typer.Argument(..., help="Branch name for the new worktree."),
|
||||
start: str = typer.Argument("HEAD", help="Start point."),
|
||||
plan_only: bool = typer.Option(
|
||||
False, "--plan", help="Report where it would go without creating it."
|
||||
),
|
||||
) -> None:
|
||||
"""Create a worktree under .worktrees/ with the venv linked in."""
|
||||
if plan_only:
|
||||
console.verdict(f"would create {environment.worktree_plan(branch)}")
|
||||
return
|
||||
environment.setup_worktree(branch, start)
|
||||
|
||||
|
||||
@app.command("perf")
|
||||
@command
|
||||
def perf(
|
||||
compare: bool = typer.Option(
|
||||
False, "--compare", help="Compare against the saved baseline instead of writing one."
|
||||
),
|
||||
) -> None:
|
||||
"""Benchmark the server and write or check tests/perf/baseline.json."""
|
||||
from tooling.domains.dev import perf as perf_module
|
||||
|
||||
perf_module.run(compare)
|
||||
|
||||
|
||||
@app.command("clerk")
|
||||
@command
|
||||
def clerk(
|
||||
plan_only: bool = typer.Option(
|
||||
False, "--plan", help="Show the per-commit plan without spawning any clerk."
|
||||
),
|
||||
) -> None:
|
||||
"""Review the push range for decision contradictions, one agent per commit."""
|
||||
from tooling.domains.dev import clerk as clerk_module
|
||||
|
||||
clerk_module.run(plan_only)
|
||||
|
||||
|
||||
@app.command("selftest")
|
||||
@command
|
||||
def selftest(
|
||||
|
||||
Reference in New Issue
Block a user