Files
settled-reach/tooling/planet-gen/naming_core.py
T
jpmschweitzerandClaude Opus 4.6 9ad9b88d7c feat(tooling): Gemma 4 batch naming pipeline with wiki-grounded register selection (#833)
Replace the one-at-a-time Gemma 2 naming pipeline with a batch-oriented
Gemma 4 E2B pipeline. Key changes:

- naming_core.py: shared library with Levenshtein distinctiveness ranking,
  batch prompt building, mood injection pool, name validation, and
  adjacent-register refill logic
- Wiki-grounded register selection: per-system LLM call picks the cultural
  register based on wiki/GTTR content instead of hash randomizer
- Batch naming: requests N*2 names per call, ranks by word-average
  Levenshtein distance, fills quota from most-distinct candidates
- Mood pool: 13 emotional seeds randomized per-body for vocabulary
  divergence (ambition, fear, isolation, defiance, etc.)
- Adjacent-register refill: when primary register exhausts, automatically
  switches to next corridor substyle
- Inhabited-first body ordering: habitable worlds get first pick of
  register vocabulary, barren moons get leftovers
- Process group cleanup: SIGTERM/SIGKILL the full distrobox chain on
  subprocess refresh to prevent GPU zombie processes
- qa_naming.py: QA report, fix_fewshot_bleed.py: post-hoc fix script
- test_batch_naming.py, test_register_selection.py: test harnesses

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-04-17 16:09:23 +02:00

396 lines
14 KiB
Python

"""naming_core.py — Shared algorithms for the atlas naming pipeline.
Contains: Levenshtein distance, distinctiveness ranking, batch prompt
building, mood pool, name validation, and response parsing. Used by
both gemma_naming.py (production pipeline) and test scripts.
Version history:
0.1 2026-04-16 Initial extraction from test_batch_naming.py
- levenshtein, word_avg_distance, select_distinct
- build_batch_prompt with mood injection
- parse_batch_response with dedup
- name_features_batch with adjacent-register refill
- is_valid_name prompt-fragment filter
0.2 2026-04-17 Post-QA hardening
- few-shot example blocklist (prevents prompt bleed)
- minimum name length raised to 3 chars
- bracket/number rejection in is_valid_name
- parse_batch_response filters few-shot examples
"""
__version__ = "0.2"
import hashlib
# ---------------------------------------------------------------------------
# Mood pool — randomized per-body emotional seed for vocabulary divergence
# ---------------------------------------------------------------------------
MOOD_POOL = [
"ambition", "family", "wealth", "community", "industry",
"pride", "fleeting", "hope", "fear", "isolation",
"devotion", "defiance", "loss",
]
def mood_for_body(body_id: str, world_seed: int = 42) -> str:
"""Deterministic mood selection per body."""
h = int(hashlib.sha256(f"mood|{body_id}|{world_seed}".encode()).hexdigest()[:8], 16)
return MOOD_POOL[h % len(MOOD_POOL)]
# ---------------------------------------------------------------------------
# Levenshtein distance and distinctiveness ranking
# ---------------------------------------------------------------------------
def levenshtein(a: str, b: str) -> int:
"""Standard Levenshtein edit distance."""
if len(a) < len(b):
return levenshtein(b, a)
if not b:
return len(a)
prev = list(range(len(b) + 1))
for i, ca in enumerate(a):
curr = [i + 1]
for j, cb in enumerate(b):
cost = 0 if ca == cb else 1
curr.append(min(curr[j] + 1, prev[j + 1] + 1, prev[j] + cost))
prev = curr
return prev[-1]
def _words(name: str) -> list[str]:
"""Split a name into lowercase words for per-word comparison."""
return [w for w in name.lower().split() if w]
def word_avg_distance(a: str, b: str) -> float:
"""Average Levenshtein distance across word pairs.
Compares each word in the shorter name against the closest word in
the longer name, then averages. Shared structural words (Serra,
The, Mount) lower the score but don't block — the unique words
pull the average up.
Returns 0.0 for identical, higher = more distinct.
"""
wa, wb = _words(a), _words(b)
if not wa or not wb:
return float(levenshtein(a.lower(), b.lower()))
shorter, longer = (wa, wb) if len(wa) <= len(wb) else (wb, wa)
total = 0.0
for sw in shorter:
best = min(levenshtein(sw, lw) for lw in longer)
total += best
return total / len(shorter)
def min_avg_distance_to_set(name: str, existing: list[str]) -> float:
"""Minimum word-average distance from name to any name in the set."""
if not existing:
return 999.0
return min(word_avg_distance(name, e) for e in existing)
def select_distinct(
candidates: list[str],
count: int,
taken: list[str],
) -> list[str]:
"""Greedily select the N most distinct names from candidates.
No hard rejection — all candidates are eligible except exact
matches to taken names. Ranked by distinctiveness (word-average
Levenshtein distance) against both taken names and previously
selected names. The most distinct candidate is picked first, then
the next most distinct relative to the growing set, until the
quota is filled.
Shared words (Serra, The, Forum) lower a candidate's preference
but never block it. The greedy approach naturally spaces out
selections.
"""
# Hard-filter exact matches to taken (case-insensitive)
taken_lower = {t.lower() for t in taken}
pool = [c for c in candidates if c.lower() not in taken_lower]
selected: list[str] = []
reference: list[str] = list(taken)
for _ in range(min(count, len(pool))):
if not pool:
break
best_idx = 0
best_score = -1.0
for i, name in enumerate(pool):
score = min_avg_distance_to_set(name, reference) if reference else 999.0
if score > best_score:
best_score = score
best_idx = i
chosen = pool.pop(best_idx)
selected.append(chosen)
reference.append(chosen)
return selected
# ---------------------------------------------------------------------------
# Name validation
# ---------------------------------------------------------------------------
# Few-shot examples used in batch prompts. These must be blocked from
# appearing as output — Gemma pattern-completes them verbatim, and
# without this blocklist they appear 50-100x across the reach.
FEWSHOT_BLOCKLIST = {
"glen moray", "dunvegan ridge", "torridon", "cairn brae",
"the kelpie's spine", "kloosterbeek", "nieuw rijn",
"hoogland run", "van diemen's creek",
}
def is_valid_name(name: str) -> bool:
"""Filter out garbage: too short, too long, contains periods/brackets,
looks like a prompt fragment, matches a few-shot example, or contains
digits."""
if not name or len(name) < 3 or len(name) > 50:
return False
# Brackets, periods, digits — structural garbage
if "." in name or "(" in name or ")" in name or "[" in name or "]" in name:
return False
if any(c.isdigit() for c in name):
return False
low = name.lower()
# Prompt fragment echoes
reject_phrases = [
"names", "style:", "answer:", "must be", "avoid", "distinct",
"already used", "do not", "need", "generate", "list",
"number only", "comma-separated", "best:",
]
if any(phrase in low for phrase in reject_phrases):
return False
# Few-shot example bleed
if low in FEWSHOT_BLOCKLIST:
return False
return True
def parse_batch_response(raw: str) -> list[str]:
"""Parse a batch naming response into a list of clean, unique name strings.
Takes the first line only (model often continues with explanations
or more styles), splits on commas, strips quotes/whitespace,
filters invalid names, and deduplicates (preserving order).
"""
first_line = raw.strip().split("\n")[0] if raw.strip() else ""
candidates = [
n.strip().strip('"').strip("'").strip()
for n in first_line.split(",")
]
# Deduplicate preserving order (model often repeats names in batch)
seen: set[str] = set()
unique: list[str] = []
for n in candidates:
if is_valid_name(n) and n.lower() not in seen:
unique.append(n)
seen.add(n.lower())
return unique
# ---------------------------------------------------------------------------
# Batch prompt building
# ---------------------------------------------------------------------------
def build_batch_prompt(
feature_type: str,
inflection: str,
count: int,
taken: list[str],
prompt_config: dict,
system_name: str | None = None,
body_name: str | None = None,
system_hook: str | None = None,
mood: str | None = None,
ctx_size: int = 1024,
) -> str:
"""Build a batch naming prompt asking for N names in one call.
Uses the same preamble structure as the single-name prompts but
with few-shot examples showing comma-separated lists. The model
pattern-completes the list.
The prompt is truncated to fit within ctx_size tokens (rough
estimate: 1 token ≈ 4 chars).
"""
cfg = prompt_config.get(feature_type)
if not cfg:
return ""
subject = cfg["subject"]
verb = "called" if subject.startswith("their ") else "named"
mood_clause = ""
if mood:
mood_clause = f" The settlers here had a sense of {mood}. "
preamble = (
f"Settlers {verb} {subject} after themselves, after what they saw, "
f"or after places back home. Most names are mundane, short, and "
f"direct — a surname, a compass direction, a feature, a practical "
f"description.{mood_clause}Avoid the obvious choice. Each name must "
f"be distinct — no two names may share a root word.\n"
f"Reply with ONLY a comma-separated list, no numbering, no markdown."
)
lines = [preamble, ""]
if system_name or body_name:
ident = []
if system_name:
ident.append(f"System: {system_name}")
if body_name:
ident.append(f"Planet: {body_name}")
lines.append(". ".join(ident) + ".")
if system_hook:
lines.append(f"About the system: {system_hook}")
if taken:
lines.append(f"Already used (do NOT repeat): {', '.join(taken)}")
if system_name or body_name or system_hook or taken:
lines.append("")
# Few-shot examples showing batch format
lines.append("Style: Scottish Highland. 5 names: Glen Moray, Dunvegan Ridge, Torridon, Cairn Brae, The Kelpie's Spine")
lines.append("Style: Dutch colonial. 4 names: Kloosterbeek, Nieuw Rijn, Hoogland Run, Van Diemen's Creek")
lines.append("")
ask_for = count * 2 # oversample, then rank by distinctiveness
tail = f"Style: {inflection}. {ask_for} names:"
# Truncate context to fit within ctx_size
fixed = "\n".join(lines)
max_chars = (ctx_size - 16) * 4 # 16 tokens headroom for output
budget = max_chars - len(fixed) - len(tail) - 2 # 2 for newlines
if budget < 0:
# Trim the taken list to fit
while taken and budget < 0:
taken = taken[:-1]
lines_rebuild = [preamble, ""]
if system_name or body_name:
ident = []
if system_name:
ident.append(f"System: {system_name}")
if body_name:
ident.append(f"Planet: {body_name}")
lines_rebuild.append(". ".join(ident) + ".")
if system_hook:
lines_rebuild.append(f"About the system: {system_hook}")
if taken:
lines_rebuild.append(f"Already used (do NOT repeat): {', '.join(taken)}")
lines_rebuild.append("")
lines_rebuild.append("Style: Scottish Highland. 5 names: Glen Moray, Dunvegan Ridge, Torridon, Cairn Brae, The Kelpie's Spine")
lines_rebuild.append("Style: Dutch colonial. 4 names: Kloosterbeek, Nieuw Rijn, Hoogland Run, Van Diemen's Creek")
lines_rebuild.append("")
fixed = "\n".join(lines_rebuild)
budget = max_chars - len(fixed) - len(tail) - 2
return fixed + "\n" + tail
# ---------------------------------------------------------------------------
# Batch naming with refill
# ---------------------------------------------------------------------------
def name_features_batch(
voice, # VoiceSubprocess — not typed to avoid circular import
feature_type: str,
count: int,
inflection: str,
corridor: str,
corridor_substyles: list[dict[str, str]],
taken: list[str],
prompt_config: dict,
system_name: str | None,
body_name: str | None,
system_hook: str | None,
mood: str | None,
body_id: str,
world_seed: int,
ctx_size: int = 1024,
) -> list[str]:
"""Generate `count` names for a feature type using batch prompting.
Flow:
1. Build a batch prompt asking for count*2 names.
2. Send to voice, parse response.
3. Rank by distinctiveness via Levenshtein, pick top `count`.
4. If short, refill from the next adjacent register in the corridor.
5. Return the final list of names.
"""
prompt = build_batch_prompt(
feature_type=feature_type,
inflection=inflection,
count=count,
taken=taken,
prompt_config=prompt_config,
system_name=system_name,
body_name=body_name,
system_hook=system_hook,
mood=mood,
ctx_size=ctx_size,
)
seed = int(hashlib.sha256(
f"batch|{body_id}|{feature_type}|{world_seed}".encode()
).hexdigest()[:8], 16)
try:
raw = voice.request(prompt, seed)
except RuntimeError:
raw = ""
candidates = parse_batch_response(raw)
selected = select_distinct(candidates, count, taken)
# Refill from adjacent register if we didn't fill the quota
if len(selected) < count and corridor_substyles:
shortfall = count - len(selected)
refill_taken = taken + selected
# Find the primary register's index and pick the next one
primary_idx = next(
(i for i, s in enumerate(corridor_substyles)
if s["inflection"] == inflection),
0,
)
refill_idx = (primary_idx + 1) % len(corridor_substyles)
refill_inflection = corridor_substyles[refill_idx]["inflection"]
refill_prompt = build_batch_prompt(
feature_type=feature_type,
inflection=refill_inflection,
count=shortfall * 3,
taken=refill_taken,
prompt_config=prompt_config,
system_name=system_name,
body_name=body_name,
system_hook=system_hook,
mood=mood,
ctx_size=ctx_size,
)
refill_seed = int(hashlib.sha256(
f"refill|{body_id}|{feature_type}|{world_seed}".encode()
).hexdigest()[:8], 16)
try:
refill_raw = voice.request(refill_prompt, refill_seed)
except RuntimeError:
refill_raw = ""
refill_candidates = parse_batch_response(refill_raw)
extra = select_distinct(refill_candidates, shortfall, refill_taken)
selected.extend(extra)
return selected