Files
settled-reach/tooling/planet-gen/prune_atlas_features.py
T
jpmschweitzerandClaude Opus 4.6 8ba3a1d0a7 fix(tooling): address PR #130 review findings (#833)
Hoshe:
- Mark --dump-prompts / name_feature() as vestigial with TODO note
- Fix --refresh help string: 200 → 1000 (matches actual default)
- Fix _RIVER_POOLS comment numbering: Pool 6 before Pool 5 → correct order
- Remove dead first-pass code in fix_fewshot_bleed.py
- _CAPTURE_FILE leak noted in vestigial TODO

Tyre:
- Fix stale "Gemma 2" strings in banner, argparse description, model help
- Note dead code for cleanup pass (name_feature ~700 lines)

Hoshe (prune):
- prune_atlas_features.py: named features sort before unnamed, preventing
  silent discard of hand-authored names during pruning

naming_core:
- v0.2: few-shot blocklist, stricter is_valid_name (min 3 chars, no digits,
  no brackets), prompt fragment rejection expanded

Miri clarification: the 261 "empty-string" files contain only roads (37)
and railroads (37) — infrastructure features never in naming scope. All
cities/rivers/oceans/mountains/POIs are clean.

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-04-17 23:27:44 +02:00

204 lines
7.3 KiB
Python
Executable File
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
prune_atlas_features.py — Cap per-body feature counts so markers.json
files stay readable at atlas scale.
Background: the upstream terrain pipeline detects every distinct mountain
cluster as a separate `mountain_range` entry, which produces bodies with
40-80 ranges at 512×256 grid resolution. Similarly for rivers. At atlas
zoom those are noise, not information — a planet doesn't need 48 named
ridges for the player to recognise the continent shape.
This pass ranks each feature type by a size proxy and keeps the top N:
- mountain_ranges: sorted by `area_cells` descending, top 8
- rivers: sorted by `len(path)` descending, top 6
- oceans/seas/lakes: untouched (already small per body)
- cities/pois: untouched (generated by generate_atlas.py, not here)
Excluded: anything under `wiki/star-systems/GJ-0/` (Sol). Sol bodies
will be hand-authored and must not be touched by automated pruning.
Each pruned body gets its atlas_* rows re-synced via sync_markers_to_db
so the DB mirror stays consistent with the on-disk JSON.
Usage:
tooling/planet-gen/prune_atlas_features.py
tooling/planet-gen/prune_atlas_features.py --max-mtns 8 --max-rivers 6
tooling/planet-gen/prune_atlas_features.py --dry-run
tooling/planet-gen/prune_atlas_features.py --body GJ144d
Safe to re-run — idempotent when a body is already within the caps.
"""
import argparse
import json
import sqlite3
import sys
import time
from pathlib import Path
TOOLING_DIR = Path(__file__).resolve().parent
REPO_ROOT = (TOOLING_DIR / ".." / "..").resolve()
DB_PATH = REPO_ROOT / "server" / "data" / "systems.db"
WIKI_SYSTEMS = REPO_ROOT / "wiki" / "star-systems"
# Path fragments that are never pruned. Sol is hand-authored.
EXCLUDED_SYSTEMS = {"GJ-0"}
sys.path.insert(0, str(TOOLING_DIR))
from generate_atlas import ensure_atlas_schema, sync_markers_to_db # noqa: E402
def _system_slug(markers_path: Path) -> str:
# wiki/star-systems/GJ-244A/bodies/GJ244Ab/markers.json → GJ-244A
return markers_path.parent.parent.parent.name
def _body_id(markers_path: Path) -> str:
return markers_path.parent.name
def prune_mountains(markers: dict, cap: int) -> int:
mtns = markers.get("mountain_ranges") or []
if len(mtns) <= cap:
return 0
# Named features sort first (preserve hand-authored names),
# then by area_cells descending.
ranked = sorted(
mtns,
key=lambda m: (0 if m.get("name") else 1, -(int(m.get("area_cells") or 0))),
)
markers["mountain_ranges"] = ranked[:cap]
return len(mtns) - cap
def prune_rivers(markers: dict, cap: int) -> int:
rivers = markers.get("rivers") or []
if len(rivers) <= cap:
return 0
# Named features sort first (preserve hand-authored names),
# then by path length descending.
ranked = sorted(
rivers,
key=lambda r: (0 if r.get("name") else 1, -len(r.get("path") or [])),
)
markers["rivers"] = ranked[:cap]
return len(rivers) - cap
def main():
parser = argparse.ArgumentParser(
description="Prune oversized mountain_ranges / rivers in every "
"markers.json (except Sol)"
)
parser.add_argument("--db", default=str(DB_PATH), help="Path to systems.db")
parser.add_argument("--body", help="Process only this body_id")
parser.add_argument("--max-mtns", type=int, default=8,
help="Max mountain_ranges per body (default: 8)")
parser.add_argument("--max-rivers", type=int, default=6,
help="Max rivers per body (default: 6)")
parser.add_argument("--dry-run", action="store_true",
help="Report counts without writing")
parser.add_argument("--verbose", action="store_true",
help="Print every body's before/after counts")
args = parser.parse_args()
db_path = Path(args.db)
if not db_path.exists():
print(f"error: {db_path} not found", file=sys.stderr)
sys.exit(1)
conn = sqlite3.connect(str(db_path), timeout=30.0)
conn.execute("PRAGMA journal_mode=WAL")
conn.execute("PRAGMA busy_timeout=15000")
conn.execute("PRAGMA foreign_keys=ON")
ensure_atlas_schema(conn)
# Preload the set of valid body_ids so we can skip orphan wiki
# folders that have no matching row in the bodies table. Otherwise
# sync_markers_to_db hits a FK violation on atlas_body_grids insert.
valid_body_ids = {
r[0] for r in conn.execute("SELECT body_id FROM bodies").fetchall()
}
all_markers = sorted(WIKI_SYSTEMS.glob("*/bodies/*/markers.json"))
if args.body:
all_markers = [p for p in all_markers if _body_id(p) == args.body]
total_bodies = 0
skipped_excluded = 0
touched_bodies = 0
mtns_dropped = 0
rivers_dropped = 0
t0 = time.time()
print(f"\n prune_atlas_features.py")
print(f" DB: {db_path}")
print(f" Max mtns: {args.max_mtns}")
print(f" Max rivers: {args.max_rivers}")
if args.dry_run:
print(f" Mode: DRY RUN")
print(f" {len(all_markers)} markers.json files to scan")
print()
for markers_path in all_markers:
total_bodies += 1
slug = _system_slug(markers_path)
if slug in EXCLUDED_SYSTEMS:
skipped_excluded += 1
if args.verbose:
print(f" SKIP (excluded system) {markers_path}")
continue
try:
markers = json.loads(markers_path.read_text())
except json.JSONDecodeError as e:
print(f" ERROR: invalid JSON in {markers_path}: {e}")
continue
body_id = _body_id(markers_path)
dropped_mtns = prune_mountains(markers, args.max_mtns)
dropped_rivers = prune_rivers(markers, args.max_rivers)
if dropped_mtns or dropped_rivers:
touched_bodies += 1
mtns_dropped += dropped_mtns
rivers_dropped += dropped_rivers
if args.verbose or dropped_mtns >= 20:
print(f" {body_id:14s} {slug:8s} "
f"{dropped_mtns} mtns, {dropped_rivers} rivers")
if not args.dry_run:
markers_path.write_text(
json.dumps(markers, indent=2) + "\n"
)
if body_id in valid_body_ids:
sync_markers_to_db(conn, body_id, markers)
elif args.verbose:
print(f" (no DB row for {body_id} — skipping sync)")
# Periodic log line so the user sees progress on a long run.
if total_bodies % 250 == 0:
rate = total_bodies / max(time.time() - t0, 1e-6)
print(f" scanned {total_bodies}/{len(all_markers)} bodies "
f"({rate:.0f}/s) touched {touched_bodies}")
if not args.dry_run:
conn.commit()
conn.close()
elapsed = time.time() - t0
print()
print(f" Done: {elapsed:.1f}s")
print(f" bodies scanned: {total_bodies}")
print(f" excluded (Sol etc.): {skipped_excluded}")
print(f" bodies pruned: {touched_bodies}")
print(f" mountain ranges dropped: {mtns_dropped}")
print(f" rivers dropped: {rivers_dropped}")
if args.dry_run:
print(f"\n Dry run — no files written, no DB changes.")
print()
if __name__ == "__main__":
main()