Merge remote-tracking branch 'origin/main' into sprint-37/server

# Conflicts:
#	CHANGELOG.md
#	Makefile
#	server/data/systems.db
This commit is contained in:
2026-04-22 09:01:43 +02:00
15 changed files with 827 additions and 12 deletions
+154
View File
@@ -0,0 +1,154 @@
#!/usr/bin/env python3
"""
check-systems-db-stamp — verify that server/data/systems.db is up to date.
Reads the meta table from systems.db and checks that the stored SHA-1 of each
generator's source file(s) matches the current file content on disk.
Exit codes:
0 — DB is stamped and all generator SHAs match current sources
1 — DB is stale, has an unknown generator, or references a missing source file
2 — DB does not have a meta table (treat as unstamped — run make regen-db)
Usage (called by .config/hooks/pre-push):
tooling/check-systems-db-stamp
Usage (interactive):
tooling/check-systems-db-stamp --verbose
Decision refs: #855 (generator versioning), #857 (pre-push hook)
"""
import hashlib
import sqlite3
import sys
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parent.parent
DB_PATH = REPO_ROOT / "server" / "data" / "systems.db"
# Maps generator_name (as stored in meta.generator_name) to the source
# file(s) whose SHA is stamped. The SHA is computed as SHA-1 of the
# concatenated bytes of all files in sorted order.
#
# import_economics' source set includes the Rust generate_brands binary it now
# invokes as a subroutine (#136 review T2/H3). Keep this list in sync with
# IMPORT_ECONOMICS_SOURCES in tooling/economy-db/import_economics.py.
GENERATOR_SOURCES: dict[str, list[Path]] = {
"import_economics": [
REPO_ROOT / "tooling" / "economy-db" / "import_economics.py",
REPO_ROOT / "server" / "src" / "bin" / "generate_brands" / "main.rs",
REPO_ROOT / "server" / "src" / "bin" / "generate_brands" / "names.rs",
REPO_ROOT / "tooling" / "generate-brands",
],
"generate_atlas": [
REPO_ROOT / "tooling" / "planet-gen" / "generate_atlas.py",
],
}
def file_sha1(*paths: Path) -> str:
"""SHA-1 of concatenated file contents (sorted paths).
Missing files raise FileNotFoundError rather than silently contributing
an empty-string hash (H2): a ghost SHA could mask real breakage when
stored and current SHAs converge on the empty-bytes digest.
"""
h = hashlib.sha1()
for p in sorted(paths):
if not p.exists():
raise FileNotFoundError(f"generator source not found: {p}")
h.update(p.read_bytes())
return h.hexdigest()
def check(verbose: bool = False) -> int:
"""Return exit code: 0 = fresh, 1 = stale, 2 = no meta table."""
if not DB_PATH.exists():
if verbose:
print(f"check-systems-db-stamp: {DB_PATH} not found — skipping check")
return 0
try:
conn = sqlite3.connect(str(DB_PATH))
rows = conn.execute(
"SELECT generator_name, generator_sha FROM meta"
).fetchall()
conn.close()
except sqlite3.OperationalError:
# meta table does not exist
if verbose:
print("check-systems-db-stamp: no meta table — systems.db has not been stamped")
print(" Run: make regen-db")
return 2
if not rows:
if verbose:
print("check-systems-db-stamp: meta table is empty — systems.db has not been stamped")
print(" Run: make regen-db")
return 2
stale: list[str] = []
unknown: list[str] = []
for generator_name, stored_sha in rows:
sources = GENERATOR_SOURCES.get(generator_name)
if sources is None:
# Unknown generator — fail closed (T6). A future branch adding a
# new generator without registering it here must update this map
# before the check will pass, preventing the "silent no-op" trap.
unknown.append(generator_name)
continue
try:
current_sha = file_sha1(*sources)
except FileNotFoundError as exc:
# Source file moved/deleted — explicit failure instead of
# silent empty-hash (H2).
print(
f"check-systems-db-stamp: BROKEN — {generator_name}: {exc}",
file=sys.stderr,
)
return 1
if current_sha != stored_sha:
stale.append(generator_name)
if verbose:
print(
f"check-systems-db-stamp: STALE — {generator_name}"
f"\n stored: {stored_sha}"
f"\n current: {current_sha}"
)
if unknown:
print(
"check-systems-db-stamp: UNKNOWN generator(s) in meta table: "
f"{unknown}",
file=sys.stderr,
)
print(
" Update GENERATOR_SOURCES in tooling/check-systems-db-stamp to "
"register them before pushing.",
file=sys.stderr,
)
return 1
if stale:
if not verbose:
print(
"systems.db is stale — run `make regen-db` before pushing.",
file=sys.stderr,
)
print(f" Stale generators: {stale}", file=sys.stderr)
return 1
if verbose:
print(f"check-systems-db-stamp: OK — {len(rows)} generator(s) up to date")
return 0
def main() -> None:
verbose = "--verbose" in sys.argv or "-v" in sys.argv
sys.exit(check(verbose=verbose))
if __name__ == "__main__":
main()
+49
View File
@@ -414,6 +414,49 @@ def claim_id(cfg, prefix, domain, title):
conn.close()
def show_decision(cfg, decision_id):
"""Show a single decision with full details including linked tickets and cross-refs."""
conn = get_connection(cfg)
try:
row = conn.execute(
"SELECT * FROM decisions WHERE id = ?",
(decision_id,),
).fetchone()
if not row:
return {"ok": False, "error": f"Decision not found: {decision_id}"}
decision = dict(row)
# Implementing tickets: tickets where decision_ref = this ID
ticket_rows = conn.execute(
"SELECT id, title, status, type FROM tickets"
" WHERE decision_ref = ? ORDER BY id",
(decision_id,),
).fetchall()
decision["implementing_tickets"] = [dict(t) for t in ticket_rows]
# Cross-refs outbound: references from this decision to others
refs_out = conn.execute(
"SELECT target_id, ref_type, note FROM decision_refs"
" WHERE source_id = ? ORDER BY target_id",
(decision_id,),
).fetchall()
decision["refs_out"] = [dict(r) for r in refs_out]
# Cross-refs inbound: other decisions referencing this one
refs_in = conn.execute(
"SELECT source_id, ref_type, note FROM decision_refs"
" WHERE target_id = ? ORDER BY source_id",
(decision_id,),
).fetchall()
decision["refs_in"] = [dict(r) for r in refs_in]
return {"ok": True, "decision": decision}
finally:
conn.close()
def check_dupes(cfg):
"""Check for duplicate decision IDs across all markdown files."""
# Pre-existing collisions too deeply embedded to renumber (139+ references).
@@ -461,6 +504,7 @@ Settled Reach Decisions Sync & ID Management
Usage:
decisions_sync.py sync Parse decisions/*.md and upsert into SQLite
decisions_sync.py show <D-NNN> Show a decision with linked tickets + refs
decisions_sync.py next [D|Q|R] Show next available ID (all prefixes or one)
decisions_sync.py claim <D|Q|R> <domain> [title] Claim next ID and insert placeholder
decisions_sync.py check-dupes Check for duplicate IDs across markdown files
@@ -493,6 +537,11 @@ def main():
if cmd == "sync":
result = sync(cfg)
elif cmd == "show":
if len(sys.argv) < 3:
result = {"ok": False, "error": "Usage: show <decision_id> e.g. show D-159"}
else:
result = show_decision(cfg, sys.argv[2])
elif cmd == "next":
result = next_id(cfg, sys.argv[2] if len(sys.argv) > 2 else None)
elif cmd == "claim":
+137 -1
View File
@@ -24,6 +24,7 @@ Usage:
"""
import argparse
import hashlib
import json
import re
import sqlite3
@@ -41,6 +42,101 @@ SCHEMA_SQL = REPO_ROOT / "server" / "data" / "systems-schema.sql"
CORPORATIONS_DIR = REPO_ROOT / "wiki" / "corporations"
BRANDS_TOML = REPO_ROOT / "wiki" / "economics" / "corporations" / "brands.toml"
GENERATED_BRANDS_TOML = REPO_ROOT / "wiki" / "economics" / "corporations" / "generated_brands.toml"
# Rust sources for the generate_brands subroutine. import_economics shells out to
# tooling/generate-brands as part of its normal flow (see regenerate_brands()), so
# both Rust files contribute to this script's effective source SHA: any change to
# either must invalidate the meta stamp even though Python hasn't changed.
GENERATE_BRANDS_RS = REPO_ROOT / "server" / "src" / "bin" / "generate_brands" / "main.rs"
GENERATE_BRANDS_NAMES_RS = REPO_ROOT / "server" / "src" / "bin" / "generate_brands" / "names.rs"
GENERATE_BRANDS_WRAPPER = REPO_ROOT / "tooling" / "generate-brands"
def _file_sha1(*paths: Path) -> str:
"""Return SHA-1 hex of the concatenated content of one or more files.
Files are sorted by path for determinism. Missing files raise FileNotFoundError
rather than silently contributing an empty-string hash — a ghost SHA masks real
breakage (review comment H2: da39a3ee… convergence could produce vacuous passes).
"""
h = hashlib.sha1()
for p in sorted(paths):
if not p.exists():
raise FileNotFoundError(f"generator source not found: {p}")
h.update(p.read_bytes())
return h.hexdigest()
# Canonical source set for import_economics' meta stamp. Covers its own .py file
# plus the Rust binary it invokes (generate_brands main.rs + names.rs + wrapper
# script) so any change to the brand generation pipeline flips the stamp. Keep
# this list in sync with GENERATOR_SOURCES["import_economics"] in
# tooling/check-systems-db-stamp.
IMPORT_ECONOMICS_SOURCES: tuple[Path, ...] = (
Path(__file__),
GENERATE_BRANDS_RS,
GENERATE_BRANDS_NAMES_RS,
GENERATE_BRANDS_WRAPPER,
)
def _write_stamp(conn: sqlite3.Connection, generator_name: str, *source_files: Path) -> None:
"""Upsert a row in the meta table recording this generator's current source SHA.
Called after every successful non-dry-run commit. Idempotent: running
twice on the same sources writes the same sha with an updated timestamp.
Only one stamp is written by this module: ``import_economics``, whose source
set includes the Rust binary it invokes (see IMPORT_ECONOMICS_SOURCES).
generate_atlas writes its own stamp. generate_brands does NOT write a stamp
of its own — it's a subroutine of import_economics, not an independent DB
writer (PR #136 review T2/H3).
The meta table is created by the MIGRATION_SQL block above; this
function assumes it exists (caller must run migrations first).
"""
schema_sha = _file_sha1(SCHEMA_SQL)
generator_sha = _file_sha1(*source_files)
conn.execute(
"""INSERT OR REPLACE INTO meta (generator_name, schema_version, generator_sha, generated_at)
VALUES (?, ?, ?, datetime('now'))""",
(generator_name, schema_sha, generator_sha),
)
def regenerate_brands() -> None:
"""Run the Rust generate_brands binary to refresh generated_brands.toml.
Invoked as the first step of import_economics' main flow so the TOML on disk
always matches the current Rust source before the Python import reads it.
This replaces the former split (tooling/generate-brands run separately by
make regen-db) with a single, coherent brand pipeline owned by one stamp.
The wrapper script builds the binary on demand and runs it with the default
canonical seed=1; callers that need non-canonical seeds must still invoke
the wrapper directly (experimentation only — committed output must be seed=1).
"""
import subprocess
if not GENERATE_BRANDS_WRAPPER.exists():
raise FileNotFoundError(
f"generate_brands wrapper not found at {GENERATE_BRANDS_WRAPPER}"
)
print(" [pre/10] Running generate_brands (Rust) to refresh generated_brands.toml...")
result = subprocess.run(
[str(GENERATE_BRANDS_WRAPPER)],
cwd=str(REPO_ROOT),
capture_output=True,
text=True,
)
if result.returncode != 0:
print(result.stdout, file=sys.stderr)
print(result.stderr, file=sys.stderr)
raise _ImportAborted()
# Print the Rust binary's own summary lines (brands generated, coverage).
# Indent so they fold under the pre-step heading.
for line in result.stdout.splitlines():
if line.strip():
print(f" {line}")
class _ImportAborted(Exception):
@@ -170,6 +266,21 @@ CREATE INDEX IF NOT EXISTS idx_production_chains_output ON production_chains(out
CREATE INDEX IF NOT EXISTS idx_chain_inputs_commodity ON chain_inputs(input_commodity_id);
CREATE INDEX IF NOT EXISTS idx_corp_presence_corp ON corp_presence(corp_id);
CREATE INDEX IF NOT EXISTS idx_corp_presence_location ON corp_presence(location_id);
-- Generator metadata stamp (#855, #856)
CREATE TABLE IF NOT EXISTS meta (
generator_name TEXT PRIMARY KEY,
schema_version TEXT NOT NULL,
generator_sha TEXT NOT NULL,
generated_at TEXT NOT NULL DEFAULT (datetime('now'))
);
-- Drop the pre-merge 'generate_brands' stamp row if it exists (PR #136 review T2/H3).
-- The Rust brand binary is now a subroutine of import_economics — its source
-- SHA contributes to the 'import_economics' stamp — so it no longer merits its
-- own meta row. This DELETE makes the check-systems-db-stamp "unknown generator"
-- path (fail-closed per T6) compatible with older DBs that still have the row.
DELETE FROM meta WHERE generator_name = 'generate_brands';
"""
# Columns to add to existing tables (ALTER TABLE is idempotent via try/except)
@@ -1010,6 +1121,18 @@ def main():
wiki_corps = load_wiki_corps()
print(f" {len(wiki_corps)} corporation files parsed")
# Regenerate generated_brands.toml via the Rust binary before the Python
# import reads it. Single pipeline, single stamp — resolves review T2/H3
# ("on-behalf stamping" coupling) by folding brand generation into
# import_economics' flow rather than having the caller (Makefile / user)
# remember to run it first. Skipped on --dry-run to avoid a disk
# side-effect during validation.
if not args.dry_run:
try:
regenerate_brands()
except _ImportAborted:
sys.exit(1)
conn = sqlite3.connect(str(db_path))
conn.execute("PRAGMA foreign_keys=ON")
@@ -1136,6 +1259,19 @@ def main():
conn.close()
raise
# Stamp generator metadata (#855, #856): record source SHAs so the
# pre-push hook can detect stale DB snapshots. Written BEFORE the
# coverage gate — the stamp records generator execution (code version),
# not data completeness. Coverage gaps (#860) are pre-existing data
# issues and must not prevent the stamp from landing.
if not args.dry_run:
try:
_write_stamp(conn, "import_economics", *IMPORT_ECONOMICS_SOURCES)
conn.commit()
print(" Stamped: import_economics (covers brand pipeline Rust sources)")
except Exception as exc: # noqa: BLE001
print(f" WARNING: failed to write generator stamp: {exc}", file=sys.stderr)
# Validate coverage (hard errors per D-175, but after commit so data is usable).
print("\n Validating coverage (D-175 Phase 2 gate)...")
coverage_errors: list[str] = []
@@ -1154,7 +1290,7 @@ def main():
print("\n Data committed but Phase 2 gate is NOT met. "
"Add corporations to meet coverage thresholds and re-run.")
conn.close()
sys.exit(1)
sys.exit(2) # exit 2 = coverage warning (data+stamp committed); exit 1 = real error
else:
print(" All coverage thresholds met — Phase 2 gate PASSED.")
+67 -1
View File
@@ -88,6 +88,48 @@ _ATLAS_SCHEMA_BEGIN_MARKER = "-- BEGIN ATLAS INDEX"
_ATLAS_SCHEMA_END_MARKER = "-- END ATLAS INDEX"
# ---------------------------------------------------------------------------
# Generator metadata stamp (#855, #856)
# ---------------------------------------------------------------------------
def _file_sha1(*paths: Path) -> str:
"""Return SHA-1 hex of the concatenated content of one or more files.
Files are sorted by path for determinism. Missing files raise
FileNotFoundError rather than silently skip — a ghost hash (empty-bytes
digest) can mask real breakage when stored and current SHAs converge
(#136 review H2).
"""
h = hashlib.sha1()
for p in sorted(paths):
if not p.exists():
raise FileNotFoundError(f"generator source not found: {p}")
h.update(p.read_bytes())
return h.hexdigest()
def _write_stamp(conn: sqlite3.Connection) -> None:
"""Upsert a meta row for generate_atlas after a successful run.
Idempotent: running twice on the same source files writes the same SHA
with an updated timestamp. The meta table is created by the atlas schema
migration executed in ensure_atlas_schema(); this function assumes it
exists.
Transaction ownership stays with the caller (matches the import_economics
pattern) — no inner commit here. Review H1 flagged the prior behaviour as
a double-commit with the atlas data write that precedes it.
"""
schema_sha = _file_sha1(SYSTEMS_SCHEMA_PATH)
generator_sha = _file_sha1(Path(__file__))
conn.execute(
"""INSERT OR REPLACE INTO meta
(generator_name, schema_version, generator_sha, generated_at)
VALUES ('generate_atlas', ?, ?, datetime('now'))""",
(schema_sha, generator_sha),
)
def _load_atlas_schema() -> str:
"""Return the atlas_* DDL block from systems-schema.sql.
@@ -121,8 +163,20 @@ def ensure_atlas_schema(conn: sqlite3.Connection) -> None:
Idempotent: all statements inside the block use CREATE TABLE / INDEX
IF NOT EXISTS, so running this on an already-migrated DB is a no-op.
Also ensures the meta stamp table exists (#855, #856).
"""
conn.executescript(_load_atlas_schema())
conn.executescript(
"""
CREATE TABLE IF NOT EXISTS meta (
generator_name TEXT PRIMARY KEY,
schema_version TEXT NOT NULL,
generator_sha TEXT NOT NULL,
generated_at TEXT NOT NULL DEFAULT (datetime('now'))
);
"""
)
def _first_int(values, default: int = 0) -> int:
@@ -1333,7 +1387,19 @@ def main():
print(f" [{i+1}/{len(bodies)}] {body_id:20s} ERROR: {result.get('message', '')}")
if not args.dry_run:
conn.commit()
# Stamp generator metadata (#855, #856) together with the atlas data
# in a single commit — atlas data + stamp land atomically, and the
# stamp function itself no longer commits (H1). Failure of the stamp
# write rolls back the atlas data too rather than leaving a stamped-
# but-missing-data intermediate state.
try:
_write_stamp(conn)
conn.commit()
print(" Stamped: generate_atlas")
except Exception as exc: # noqa: BLE001
conn.rollback()
print(f" WARNING: failed to write generator stamp: {exc}", file=sys.stderr)
print(" Atlas data NOT committed — regen required.", file=sys.stderr)
conn.close()
elapsed_total = time.time() - t_total