Replace the one-at-a-time Gemma 2 naming pipeline with a batch-oriented Gemma 4 E2B pipeline. Key changes: - naming_core.py: shared library with Levenshtein distinctiveness ranking, batch prompt building, mood injection pool, name validation, and adjacent-register refill logic - Wiki-grounded register selection: per-system LLM call picks the cultural register based on wiki/GTTR content instead of hash randomizer - Batch naming: requests N*2 names per call, ranks by word-average Levenshtein distance, fills quota from most-distinct candidates - Mood pool: 13 emotional seeds randomized per-body for vocabulary divergence (ambition, fear, isolation, defiance, etc.) - Adjacent-register refill: when primary register exhausts, automatically switches to next corridor substyle - Inhabited-first body ordering: habitable worlds get first pick of register vocabulary, barren moons get leftovers - Process group cleanup: SIGTERM/SIGKILL the full distrobox chain on subprocess refresh to prevent GPU zombie processes - qa_naming.py: QA report, fix_fewshot_bleed.py: post-hoc fix script - test_batch_naming.py, test_register_selection.py: test harnesses Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
247 lines
10 KiB
Python
247 lines
10 KiB
Python
#!/usr/bin/env python3
|
|
"""Replace few-shot example names that bled into the output.
|
|
|
|
The batch naming prompt uses Scottish Highland and Dutch colonial
|
|
examples. The Scottish ones (Glen Moray, Dunvegan Ridge, Torridon,
|
|
Cairn Brae, The Kelpie's Spine) leaked into 270 features. This script
|
|
replaces them with unique names from a combined Scottish/Welsh/Irish
|
|
pool, ensuring no collisions with the existing corpus.
|
|
"""
|
|
|
|
import json
|
|
import sqlite3
|
|
import sys
|
|
from collections import defaultdict
|
|
from pathlib import Path
|
|
|
|
TOOLING_DIR = Path(__file__).resolve().parent
|
|
REPO_ROOT = (TOOLING_DIR / ".." / "..").resolve()
|
|
DB_PATH = REPO_ROOT / "server" / "data" / "systems.db"
|
|
WIKI_SYSTEMS = REPO_ROOT / "wiki" / "star-systems"
|
|
|
|
sys.path.insert(0, str(TOOLING_DIR))
|
|
from generate_atlas import sync_markers_to_db
|
|
|
|
# The few-shot names to replace
|
|
FEWSHOT_NAMES = {
|
|
"glen moray", "dunvegan ridge", "torridon", "cairn brae", "the kelpie's spine",
|
|
"kloosterbeek", "nieuw rijn", "hoogland run", "van diemen's creek",
|
|
}
|
|
|
|
# Scottish / Welsh / Irish replacement pool — 300+ names to cover 270 replacements
|
|
# with room for Levenshtein filtering. Mix of geographic feature styles.
|
|
REPLACEMENT_POOL = [
|
|
# Scottish
|
|
"Glenfinnan", "Dalwhinnie Pass", "Cairngorm", "Loch Maree",
|
|
"Kinlochleven", "Strathspey", "Brae Morar", "Skye Reach",
|
|
"Ardnamurchan", "Kintail", "Glen Affric", "Lochaber",
|
|
"Killiecrankie", "Rannoch Moor", "Glen Coe", "Strathnaver",
|
|
"Applecross", "Torrisdale", "Durness", "Assynt",
|
|
"Coigach", "Inverpolly", "Sandwood", "Cape Wrath",
|
|
"Sutherland", "Helmsdale", "Brora", "Golspie",
|
|
"Cromarty", "Dornoch", "Nairn", "Forres",
|
|
"Culbin", "Findhorn", "Spey Bay", "Buckie",
|
|
"Banff", "Fraserburgh", "Peterhead", "Cruden Bay",
|
|
"Slains", "Ythan", "Bennachie", "Morven",
|
|
"Lochnagar", "Braemar", "Balmoral", "Crathie",
|
|
"Ballater", "Dinnet", "Tarland", "Lumphanan",
|
|
"Corgarff", "Tomintoul", "Glenlivet", "Dufftown",
|
|
"Craigellachie", "Aberlour", "Knockando", "Archiestown",
|
|
"Rothes", "Elgin", "Lossiemouth", "Burghead",
|
|
"Kinloss", "Alves", "Pluscarden", "Dallas",
|
|
# Welsh
|
|
"Cwm Idwal", "Beddgelert", "Crib Goch", "Tryfan",
|
|
"Ogwen", "Llyn Padarn", "Dolgellau", "Harlech",
|
|
"Rhinog", "Cader Idris", "Barmouth", "Aberdovey",
|
|
"Tywyn", "Machynlleth", "Pumlumon", "Hafren",
|
|
"Elan Valley", "Claerwen", "Llandrindod", "Brecon",
|
|
"Pen y Fan", "Corn Du", "Crickhowell", "Llangorse",
|
|
"Talgarth", "Hay Bluff", "Mynydd Troed", "Mynydd Llangorse",
|
|
"Skirrid", "Blorenge", "Llanfoist", "Govilon",
|
|
"Gilwern", "Llangattock", "Crug Hywel", "Cwm Clydach",
|
|
"Pontneddfechan", "Ystradfellte", "Sgwd yr Eira", "Henrhyd",
|
|
"Carreg Cennen", "Dinefwr", "Llandeilo", "Dryslwyn",
|
|
"Tywi Valley", "Carmarthen", "Kidwelly", "Pembrey",
|
|
"Gower", "Rhossili", "Oxwich", "Port Eynon",
|
|
"Pennard", "Langland", "Caswell", "Mumbles",
|
|
"Merthyr Mawr", "Ogmore", "Dunraven", "Llantwit",
|
|
"Monknash", "Nash Point", "Aberthaw", "Fonmon",
|
|
# Irish
|
|
"Glendalough", "Lugnaquilla", "Glen Imaal", "Wicklow Gap",
|
|
"Sally Gap", "Kippure", "Djuce", "Maulin",
|
|
"Djouce", "Great Sugar Loaf", "Bray Head", "Killiney",
|
|
"Dalkey", "Howth", "Lambay", "Ireland's Eye",
|
|
"Malahide", "Portmarnock", "Donabate", "Skerries",
|
|
"Balbriggan", "Gormanston", "Bettystown", "Laytown",
|
|
"Slane", "Newgrange", "Dowth", "Knowth",
|
|
"Tara", "Trim", "Navan", "Kells",
|
|
"Loughcrew", "Oldcastle", "Castlepollard", "Fore",
|
|
"Delvin", "Mullingar", "Kilbeggan", "Tullamore",
|
|
"Clara", "Ferbane", "Banagher", "Shannonbridge",
|
|
"Clonmacnoise", "Ballinasloe", "Aughrim", "Loughrea",
|
|
"Portumna", "Mountshannon", "Killaloe", "Ballina",
|
|
"Nenagh", "Roscrea", "Templemore", "Thurles",
|
|
"Cashel", "Cahir", "Clonmel", "Carrick-on-Suir",
|
|
"Piltown", "Mooncoin", "Waterford", "Tramore",
|
|
"Bunmahon", "Ardmore", "Youghal", "Midleton",
|
|
"Cobh", "Crosshaven", "Kinsale", "Clonakilty",
|
|
"Skibbereen", "Bantry", "Glengarriff", "Kenmare",
|
|
"Sneem", "Caherdaniel", "Waterville", "Cahersiveen",
|
|
"Valentia", "Portmagee", "Skellig", "Dingle",
|
|
"Brandon", "Castlegregory", "Fenit", "Tralee",
|
|
"Listowel", "Ballybunion", "Tarbert", "Glin",
|
|
"Foynes", "Askeaton", "Adare", "Patrickswell",
|
|
# More Scottish/Gaelic to fill
|
|
"Stornoway", "Tarbert", "Scalpay", "Eriskay",
|
|
"Barra", "Vatersay", "Minguilay", "Pabbay",
|
|
"Berneray", "Monach Isles", "Balranald", "Lochmaddy",
|
|
"Benbecula", "Grimsay", "Ronay", "Wiay",
|
|
"Canna", "Rum", "Eigg", "Muck",
|
|
"Ardnish", "Arisaig", "Morar", "Mallaig",
|
|
"Knoydart", "Barrisdale", "Arnisdale", "Glenelg",
|
|
"Sandaig", "Brochs of Borve", "Callanish", "Garenin",
|
|
"Carloway", "Arnol", "Barvas", "Tolsta",
|
|
"Ness", "Europie", "Swainbost", "Skigersta",
|
|
# Additional Welsh/Irish
|
|
"Aberystwyth", "Llanberis", "Betws-y-Coed", "Conwy",
|
|
"Caernarfon", "Pwllheli", "Abersoch", "Nefyn",
|
|
"Llanbedrog", "Criccieth", "Porthmadog", "Portmeirion",
|
|
"Trawsfynydd", "Ffestiniog", "Blaenau", "Llyn Tegid",
|
|
"Corwen", "Llangollen", "Chirk", "Oswestry",
|
|
]
|
|
|
|
|
|
def load_global_names(conn):
|
|
"""Load all existing names globally for uniqueness checking."""
|
|
names = set()
|
|
for table in ['atlas_cities', 'atlas_rivers', 'atlas_mountain_ranges',
|
|
'atlas_oceans', 'atlas_pois']:
|
|
rows = conn.execute(
|
|
f"SELECT lower(name) FROM {table} WHERE name IS NOT NULL AND name != ''"
|
|
).fetchall()
|
|
names.update(r[0] for r in rows)
|
|
return names
|
|
|
|
|
|
def main():
|
|
conn = sqlite3.connect(str(DB_PATH), timeout=30.0)
|
|
conn.execute("PRAGMA journal_mode=WAL")
|
|
conn.execute("PRAGMA busy_timeout=15000")
|
|
|
|
global_names = load_global_names(conn)
|
|
print(f"Loaded {len(global_names)} existing names")
|
|
|
|
# Build available replacements (not already in corpus)
|
|
available = [n for n in REPLACEMENT_POOL if n.lower() not in global_names]
|
|
print(f"Available replacements: {len(available)} (from pool of {len(REPLACEMENT_POOL)})")
|
|
|
|
# Find all features that need replacement
|
|
replacements_needed = []
|
|
for markers_path in sorted(WIKI_SYSTEMS.glob("*/bodies/*/markers.json")):
|
|
body_id = markers_path.parent.name
|
|
m = json.loads(markers_path.read_text())
|
|
for section in ("cities", "rivers", "oceans", "mountain_ranges", "pois"):
|
|
for feat in m.get(section, []):
|
|
name = feat.get("name", "")
|
|
if name and name.lower() in FEWSHOT_NAMES:
|
|
replacements_needed.append((markers_path, body_id, section, feat))
|
|
|
|
print(f"Features to replace: {len(replacements_needed)}")
|
|
|
|
if len(available) < len(replacements_needed):
|
|
print(f"WARNING: only {len(available)} replacements for {len(replacements_needed)} features")
|
|
print(" some features will keep their few-shot names")
|
|
|
|
# Assign replacements deterministically — hash body_id + feature_id
|
|
# to pick from the pool, ensuring each body gets different names
|
|
used_per_body = defaultdict(set)
|
|
replacement_idx = 0
|
|
changed_files = set()
|
|
total_replaced = 0
|
|
|
|
for markers_path, body_id, section, feat in replacements_needed:
|
|
old_name = feat["name"]
|
|
|
|
# Find next available name not yet used on this body
|
|
assigned = None
|
|
for attempt in range(len(available)):
|
|
candidate = available[(replacement_idx + attempt) % len(available)]
|
|
if candidate.lower() not in used_per_body[body_id]:
|
|
assigned = candidate
|
|
replacement_idx = (replacement_idx + attempt + 1) % len(available)
|
|
break
|
|
|
|
if assigned is None:
|
|
print(f" SKIP {body_id}/{section}: no unique replacement for \"{old_name}\"")
|
|
continue
|
|
|
|
feat["name"] = assigned
|
|
used_per_body[body_id].add(assigned.lower())
|
|
global_names.add(assigned.lower())
|
|
changed_files.add(markers_path)
|
|
total_replaced += 1
|
|
|
|
# Write changed files
|
|
for markers_path in changed_files:
|
|
body_id = markers_path.parent.name
|
|
m = json.loads(markers_path.read_text())
|
|
|
|
# Re-apply changes (re-read since we modified feat objects in memory)
|
|
# Actually the feat dicts are still referenced — just rewrite
|
|
# But we need to reload and re-match since we didn't track which file
|
|
# has which changes...
|
|
|
|
# Simpler approach: reload, replace, write
|
|
# Reset and do it properly
|
|
replacement_idx = 0
|
|
used_per_body = defaultdict(set)
|
|
changed_bodies = []
|
|
|
|
# Group by file
|
|
by_file = defaultdict(list)
|
|
for markers_path, body_id, section, feat in replacements_needed:
|
|
by_file[markers_path].append((body_id, section, feat["id"] if "id" in feat else None))
|
|
|
|
for markers_path, entries in by_file.items():
|
|
body_id = markers_path.parent.name
|
|
m = json.loads(markers_path.read_text())
|
|
changed = False
|
|
|
|
for _, section, feat_id in entries:
|
|
for feat in m.get(section, []):
|
|
name = feat.get("name") or ""
|
|
if not name or name.lower() not in FEWSHOT_NAMES:
|
|
continue
|
|
|
|
assigned = None
|
|
for attempt in range(len(available)):
|
|
candidate = available[(replacement_idx + attempt) % len(available)]
|
|
if candidate.lower() not in used_per_body[body_id]:
|
|
assigned = candidate
|
|
replacement_idx = (replacement_idx + attempt + 1) % len(available)
|
|
break
|
|
|
|
if assigned:
|
|
feat["name"] = assigned
|
|
used_per_body[body_id].add(assigned.lower())
|
|
changed = True
|
|
|
|
if changed:
|
|
markers_path.write_text(json.dumps(m, indent=2) + "\n")
|
|
changed_bodies.append(body_id)
|
|
# Sync to DB if body exists
|
|
try:
|
|
sync_markers_to_db(conn, body_id, m)
|
|
except Exception:
|
|
pass # orphan body
|
|
|
|
conn.commit()
|
|
conn.close()
|
|
|
|
print(f"\nReplaced few-shot names on {len(changed_bodies)} bodies")
|
|
print("Done.")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|