diff --git a/tooling/planet-gen/fix_fewshot_bleed.py b/tooling/planet-gen/fix_fewshot_bleed.py new file mode 100644 index 000000000..4b0851a7a --- /dev/null +++ b/tooling/planet-gen/fix_fewshot_bleed.py @@ -0,0 +1,246 @@ +#!/usr/bin/env python3 +"""Replace few-shot example names that bled into the output. + +The batch naming prompt uses Scottish Highland and Dutch colonial +examples. The Scottish ones (Glen Moray, Dunvegan Ridge, Torridon, +Cairn Brae, The Kelpie's Spine) leaked into 270 features. This script +replaces them with unique names from a combined Scottish/Welsh/Irish +pool, ensuring no collisions with the existing corpus. +""" + +import json +import sqlite3 +import sys +from collections import defaultdict +from pathlib import Path + +TOOLING_DIR = Path(__file__).resolve().parent +REPO_ROOT = (TOOLING_DIR / ".." / "..").resolve() +DB_PATH = REPO_ROOT / "server" / "data" / "systems.db" +WIKI_SYSTEMS = REPO_ROOT / "wiki" / "star-systems" + +sys.path.insert(0, str(TOOLING_DIR)) +from generate_atlas import sync_markers_to_db + +# The few-shot names to replace +FEWSHOT_NAMES = { + "glen moray", "dunvegan ridge", "torridon", "cairn brae", "the kelpie's spine", + "kloosterbeek", "nieuw rijn", "hoogland run", "van diemen's creek", +} + +# Scottish / Welsh / Irish replacement pool — 300+ names to cover 270 replacements +# with room for Levenshtein filtering. Mix of geographic feature styles. +REPLACEMENT_POOL = [ + # Scottish + "Glenfinnan", "Dalwhinnie Pass", "Cairngorm", "Loch Maree", + "Kinlochleven", "Strathspey", "Brae Morar", "Skye Reach", + "Ardnamurchan", "Kintail", "Glen Affric", "Lochaber", + "Killiecrankie", "Rannoch Moor", "Glen Coe", "Strathnaver", + "Applecross", "Torrisdale", "Durness", "Assynt", + "Coigach", "Inverpolly", "Sandwood", "Cape Wrath", + "Sutherland", "Helmsdale", "Brora", "Golspie", + "Cromarty", "Dornoch", "Nairn", "Forres", + "Culbin", "Findhorn", "Spey Bay", "Buckie", + "Banff", "Fraserburgh", "Peterhead", "Cruden Bay", + "Slains", "Ythan", "Bennachie", "Morven", + "Lochnagar", "Braemar", "Balmoral", "Crathie", + "Ballater", "Dinnet", "Tarland", "Lumphanan", + "Corgarff", "Tomintoul", "Glenlivet", "Dufftown", + "Craigellachie", "Aberlour", "Knockando", "Archiestown", + "Rothes", "Elgin", "Lossiemouth", "Burghead", + "Kinloss", "Alves", "Pluscarden", "Dallas", + # Welsh + "Cwm Idwal", "Beddgelert", "Crib Goch", "Tryfan", + "Ogwen", "Llyn Padarn", "Dolgellau", "Harlech", + "Rhinog", "Cader Idris", "Barmouth", "Aberdovey", + "Tywyn", "Machynlleth", "Pumlumon", "Hafren", + "Elan Valley", "Claerwen", "Llandrindod", "Brecon", + "Pen y Fan", "Corn Du", "Crickhowell", "Llangorse", + "Talgarth", "Hay Bluff", "Mynydd Troed", "Mynydd Llangorse", + "Skirrid", "Blorenge", "Llanfoist", "Govilon", + "Gilwern", "Llangattock", "Crug Hywel", "Cwm Clydach", + "Pontneddfechan", "Ystradfellte", "Sgwd yr Eira", "Henrhyd", + "Carreg Cennen", "Dinefwr", "Llandeilo", "Dryslwyn", + "Tywi Valley", "Carmarthen", "Kidwelly", "Pembrey", + "Gower", "Rhossili", "Oxwich", "Port Eynon", + "Pennard", "Langland", "Caswell", "Mumbles", + "Merthyr Mawr", "Ogmore", "Dunraven", "Llantwit", + "Monknash", "Nash Point", "Aberthaw", "Fonmon", + # Irish + "Glendalough", "Lugnaquilla", "Glen Imaal", "Wicklow Gap", + "Sally Gap", "Kippure", "Djuce", "Maulin", + "Djouce", "Great Sugar Loaf", "Bray Head", "Killiney", + "Dalkey", "Howth", "Lambay", "Ireland's Eye", + "Malahide", "Portmarnock", "Donabate", "Skerries", + "Balbriggan", "Gormanston", "Bettystown", "Laytown", + "Slane", "Newgrange", "Dowth", "Knowth", + "Tara", "Trim", "Navan", "Kells", + "Loughcrew", "Oldcastle", "Castlepollard", "Fore", + "Delvin", "Mullingar", "Kilbeggan", "Tullamore", + "Clara", "Ferbane", "Banagher", "Shannonbridge", + "Clonmacnoise", "Ballinasloe", "Aughrim", "Loughrea", + "Portumna", "Mountshannon", "Killaloe", "Ballina", + "Nenagh", "Roscrea", "Templemore", "Thurles", + "Cashel", "Cahir", "Clonmel", "Carrick-on-Suir", + "Piltown", "Mooncoin", "Waterford", "Tramore", + "Bunmahon", "Ardmore", "Youghal", "Midleton", + "Cobh", "Crosshaven", "Kinsale", "Clonakilty", + "Skibbereen", "Bantry", "Glengarriff", "Kenmare", + "Sneem", "Caherdaniel", "Waterville", "Cahersiveen", + "Valentia", "Portmagee", "Skellig", "Dingle", + "Brandon", "Castlegregory", "Fenit", "Tralee", + "Listowel", "Ballybunion", "Tarbert", "Glin", + "Foynes", "Askeaton", "Adare", "Patrickswell", + # More Scottish/Gaelic to fill + "Stornoway", "Tarbert", "Scalpay", "Eriskay", + "Barra", "Vatersay", "Minguilay", "Pabbay", + "Berneray", "Monach Isles", "Balranald", "Lochmaddy", + "Benbecula", "Grimsay", "Ronay", "Wiay", + "Canna", "Rum", "Eigg", "Muck", + "Ardnish", "Arisaig", "Morar", "Mallaig", + "Knoydart", "Barrisdale", "Arnisdale", "Glenelg", + "Sandaig", "Brochs of Borve", "Callanish", "Garenin", + "Carloway", "Arnol", "Barvas", "Tolsta", + "Ness", "Europie", "Swainbost", "Skigersta", + # Additional Welsh/Irish + "Aberystwyth", "Llanberis", "Betws-y-Coed", "Conwy", + "Caernarfon", "Pwllheli", "Abersoch", "Nefyn", + "Llanbedrog", "Criccieth", "Porthmadog", "Portmeirion", + "Trawsfynydd", "Ffestiniog", "Blaenau", "Llyn Tegid", + "Corwen", "Llangollen", "Chirk", "Oswestry", +] + + +def load_global_names(conn): + """Load all existing names globally for uniqueness checking.""" + names = set() + for table in ['atlas_cities', 'atlas_rivers', 'atlas_mountain_ranges', + 'atlas_oceans', 'atlas_pois']: + rows = conn.execute( + f"SELECT lower(name) FROM {table} WHERE name IS NOT NULL AND name != ''" + ).fetchall() + names.update(r[0] for r in rows) + return names + + +def main(): + conn = sqlite3.connect(str(DB_PATH), timeout=30.0) + conn.execute("PRAGMA journal_mode=WAL") + conn.execute("PRAGMA busy_timeout=15000") + + global_names = load_global_names(conn) + print(f"Loaded {len(global_names)} existing names") + + # Build available replacements (not already in corpus) + available = [n for n in REPLACEMENT_POOL if n.lower() not in global_names] + print(f"Available replacements: {len(available)} (from pool of {len(REPLACEMENT_POOL)})") + + # Find all features that need replacement + replacements_needed = [] + for markers_path in sorted(WIKI_SYSTEMS.glob("*/bodies/*/markers.json")): + body_id = markers_path.parent.name + m = json.loads(markers_path.read_text()) + for section in ("cities", "rivers", "oceans", "mountain_ranges", "pois"): + for feat in m.get(section, []): + name = feat.get("name", "") + if name and name.lower() in FEWSHOT_NAMES: + replacements_needed.append((markers_path, body_id, section, feat)) + + print(f"Features to replace: {len(replacements_needed)}") + + if len(available) < len(replacements_needed): + print(f"WARNING: only {len(available)} replacements for {len(replacements_needed)} features") + print(" some features will keep their few-shot names") + + # Assign replacements deterministically — hash body_id + feature_id + # to pick from the pool, ensuring each body gets different names + used_per_body = defaultdict(set) + replacement_idx = 0 + changed_files = set() + total_replaced = 0 + + for markers_path, body_id, section, feat in replacements_needed: + old_name = feat["name"] + + # Find next available name not yet used on this body + assigned = None + for attempt in range(len(available)): + candidate = available[(replacement_idx + attempt) % len(available)] + if candidate.lower() not in used_per_body[body_id]: + assigned = candidate + replacement_idx = (replacement_idx + attempt + 1) % len(available) + break + + if assigned is None: + print(f" SKIP {body_id}/{section}: no unique replacement for \"{old_name}\"") + continue + + feat["name"] = assigned + used_per_body[body_id].add(assigned.lower()) + global_names.add(assigned.lower()) + changed_files.add(markers_path) + total_replaced += 1 + + # Write changed files + for markers_path in changed_files: + body_id = markers_path.parent.name + m = json.loads(markers_path.read_text()) + + # Re-apply changes (re-read since we modified feat objects in memory) + # Actually the feat dicts are still referenced — just rewrite + # But we need to reload and re-match since we didn't track which file + # has which changes... + + # Simpler approach: reload, replace, write + # Reset and do it properly + replacement_idx = 0 + used_per_body = defaultdict(set) + changed_bodies = [] + + # Group by file + by_file = defaultdict(list) + for markers_path, body_id, section, feat in replacements_needed: + by_file[markers_path].append((body_id, section, feat["id"] if "id" in feat else None)) + + for markers_path, entries in by_file.items(): + body_id = markers_path.parent.name + m = json.loads(markers_path.read_text()) + changed = False + + for _, section, feat_id in entries: + for feat in m.get(section, []): + name = feat.get("name") or "" + if not name or name.lower() not in FEWSHOT_NAMES: + continue + + assigned = None + for attempt in range(len(available)): + candidate = available[(replacement_idx + attempt) % len(available)] + if candidate.lower() not in used_per_body[body_id]: + assigned = candidate + replacement_idx = (replacement_idx + attempt + 1) % len(available) + break + + if assigned: + feat["name"] = assigned + used_per_body[body_id].add(assigned.lower()) + changed = True + + if changed: + markers_path.write_text(json.dumps(m, indent=2) + "\n") + changed_bodies.append(body_id) + # Sync to DB if body exists + try: + sync_markers_to_db(conn, body_id, m) + except Exception: + pass # orphan body + + conn.commit() + conn.close() + + print(f"\nReplaced few-shot names on {len(changed_bodies)} bodies") + print("Done.") + + +if __name__ == "__main__": + main() diff --git a/tooling/planet-gen/gemma_naming.py b/tooling/planet-gen/gemma_naming.py index f76205f90..07a1e73cc 100755 --- a/tooling/planet-gen/gemma_naming.py +++ b/tooling/planet-gen/gemma_naming.py @@ -1,7 +1,7 @@ #!/usr/bin/env python3 """ gemma_naming.py — Batch-name every empty name field in the reach's -markers.json files using the Gemma 2 voice pipeline (#833, D-191 §4). +markers.json files using the Gemma 4 E2B tooling pipeline (#833, D-191 §4). Pipeline per body: 1. Load markers.json; identify feature records whose `name` is empty @@ -40,7 +40,9 @@ import argparse import datetime import hashlib import json +import os import re +import signal import subprocess import sys import time @@ -60,6 +62,10 @@ from generate_atlas import ( # noqa: E402 ) import sqlite3 # noqa: E402 +from naming_core import ( # noqa: E402 + name_features_batch, + mood_for_body, +) DB_PATH = REPO_ROOT / "server" / "data" / "systems.db" WIKI_SYSTEMS = REPO_ROOT / "wiki" / "star-systems" @@ -91,6 +97,7 @@ _CAPTURE_FILE = None # set in main() when --dump-prompts is used # across worktrees (too large to duplicate). HOME_PROJECTS = Path.home() / "Projects" / "settled-reach" BINARIES_DIR = HOME_PROJECTS / "binaries" +MODELS_DIR = HOME_PROJECTS / "models" MAIN_WORKDIR = Path("/var/mnt/data/projects/settled-reach/main") @@ -98,19 +105,31 @@ def _find_sr_voice() -> Path: """Resolve the default sr-voice binary path. Preference order: - 1. $HOME/Projects/settled-reach/binaries/sr-voice-rocm — persistent - across worktree lifetimes, the canonical dev location. - 2. main workdir's target/release/sr-voice — legacy, for - backward-compatibility with older layouts. + 1. $HOME/Projects/settled-reach/binaries/sr-voice-tooling — Gemma 4 + tooling binary, preferred for content generation. + 2. $HOME/Projects/settled-reach/binaries/sr-voice-rocm — Gemma 2 + ROCm binary, fallback. + 3. main workdir's target/release/sr-voice — legacy. """ + tooling_bin = BINARIES_DIR / "sr-voice-tooling" + if tooling_bin.exists(): + return tooling_bin rocm_bin = BINARIES_DIR / "sr-voice-rocm" if rocm_bin.exists(): return rocm_bin return MAIN_WORKDIR / "server" / "sr-voice" / "target" / "release" / "sr-voice" +def _find_default_model() -> Path: + """Resolve the default model path. Prefers Gemma 4 over Gemma 2.""" + gemma4 = MODELS_DIR / "gemma-4.gguf" + if gemma4.exists(): + return gemma4 + return MAIN_WORKDIR / "server" / "models" / "gemma2.gguf" + + DEFAULT_SR_VOICE = _find_sr_voice() -DEFAULT_MODEL = MAIN_WORKDIR / "server" / "models" / "gemma2.gguf" +DEFAULT_MODEL = _find_default_model() MOCK_STDIO = REPO_ROOT / "server" / "sr-voice" / "mock-stdio.sh" @@ -285,12 +304,181 @@ def palette_for(corridor: str | None, system_id: str = "") -> dict[str, str]: All bodies in the same system get the same sub-style (consistent cultural register per star system). Different systems rotate through the sub-style list via hash(system_id). + + This is the FALLBACK path — the preferred path is select_register() + which asks Gemma to pick the register based on wiki/GTTR content. """ substyles = CORRIDOR_SUBSTYLES.get(corridor or "core", DEFAULT_SUBSTYLES) idx = int(hashlib.sha256(system_id.encode()).hexdigest()[:8], 16) % len(substyles) return substyles[idx] +def _system_slug(system_id: str) -> str: + """Convert system_id ('GJ 411') to wiki directory slug ('GJ-411').""" + if system_id.startswith("GJ "): + return "GJ-" + system_id[3:] + return system_id + + +def load_wiki_context(system_id: str) -> tuple[str | None, str | None]: + """Read index.md and gttr.md for a system from wiki/star-systems/. + + Returns (index_text, gttr_text). Either or both may be None if the + file doesn't exist. + """ + slug = _system_slug(system_id) + sys_dir = WIKI_SYSTEMS / slug + index_path = sys_dir / "index.md" + gttr_path = sys_dir / "gttr.md" + index_text = index_path.read_text() if index_path.exists() else None + gttr_text = gttr_path.read_text() if gttr_path.exists() else None + return index_text, gttr_text + + +def _extract_cultural_lines(wiki_text: str, max_lines: int = 8) -> str: + """Pull the most culturally relevant lines from a wiki index.md. + + Scans for lines mentioning heritage, founding identity, language, + cultural texture, or corridor affiliation. Falls back to the first + prose paragraphs if no keyword hits. Keeps the excerpt short enough + for Gemma 2 2B's 1024-token context. + """ + keywords = ( + "cultural", "heritage", "founding", "settler", "surname", + "language", "tradition", "diaspora", "population carried", + "portuguese", "iberian", "japanese", "korean", "chinese", + "filipino", "german", "dutch", "nordic", "scandinavian", + "polish", "czech", "finnish", "baltic", "swahili", "african", + "angolan", "cape verde", "irish", "scottish", "australian", + "british", "brazilian", "mozambic", "norwegian", "frisian", + "afrikaans", "lusophone", "corridor", + ) + hits: list[str] = [] + prose: list[str] = [] + for line in wiki_text.splitlines(): + stripped = line.strip() + if not stripped or stripped.startswith("#") or stripped.startswith("|") or stripped.startswith("---") or stripped.startswith("