Files
settled-reach/tooling/planet-gen/run-atlas-naming.sh
T
jpmschweitzerandClaude Opus 4.6 9ad9b88d7c feat(tooling): Gemma 4 batch naming pipeline with wiki-grounded register selection (#833)
Replace the one-at-a-time Gemma 2 naming pipeline with a batch-oriented
Gemma 4 E2B pipeline. Key changes:

- naming_core.py: shared library with Levenshtein distinctiveness ranking,
  batch prompt building, mood injection pool, name validation, and
  adjacent-register refill logic
- Wiki-grounded register selection: per-system LLM call picks the cultural
  register based on wiki/GTTR content instead of hash randomizer
- Batch naming: requests N*2 names per call, ranks by word-average
  Levenshtein distance, fills quota from most-distinct candidates
- Mood pool: 13 emotional seeds randomized per-body for vocabulary
  divergence (ambition, fear, isolation, defiance, etc.)
- Adjacent-register refill: when primary register exhausts, automatically
  switches to next corridor substyle
- Inhabited-first body ordering: habitable worlds get first pick of
  register vocabulary, barren moons get leftovers
- Process group cleanup: SIGTERM/SIGKILL the full distrobox chain on
  subprocess refresh to prevent GPU zombie processes
- qa_naming.py: QA report, fix_fewshot_bleed.py: post-hoc fix script
- test_batch_naming.py, test_register_selection.py: test harnesses

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-04-17 16:09:23 +02:00

81 lines
2.8 KiB
Bash
Executable File

#!/usr/bin/env bash
# Run the full Gemma 2 batch naming pipeline across every markers.json
# in wiki/star-systems. Uses the gfx1201 ROCm binary via distrobox.
#
# Safe to interrupt and resume: the pipeline preserves bodies that
# already have non-empty name fields, so re-running picks up where it
# left off. A crash loses at most the current body's in-progress work.
#
# Wall time on an RX 9070 is roughly 4-6 h for the ~26k features in
# 2394 bodies. Run with `nohup` and tail the log file:
#
# nohup tooling/planet-gen/run-atlas-naming.sh > /tmp/atlas.out 2>&1 &
# tail -f .tmp/atlas-naming-*.log
set -euo pipefail
# No arguments accepted — everything is hardcoded for the overnight
# run. Guard against typos/--help slipping through to gemma_naming.py
# and starting a real run when the caller expected a help screen.
if [[ $# -gt 0 ]]; then
echo "usage: $0" >&2
echo " (no arguments; edit this script to change binary/model/container)" >&2
exit 2
fi
REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
cd "$REPO_ROOT"
LOG_DIR="$REPO_ROOT/.tmp"
mkdir -p "$LOG_DIR"
STAMP="$(date +%Y%m%d-%H%M%S)"
LOG="$LOG_DIR/atlas-naming-$STAMP.log"
BIN="$HOME/Projects/settled-reach/binaries/sr-voice-tooling"
MODEL="$HOME/Projects/settled-reach/models/gemma-4.gguf"
DISTROBOX_NAME="reach-build"
if [[ ! -x "$BIN" ]]; then
echo "error: sr-voice binary not found at $BIN" >&2
echo " build with --features rocm inside $DISTROBOX_NAME" >&2
exit 1
fi
if [[ ! -f "$MODEL" ]]; then
echo "error: Gemma 2 model not found at $MODEL" >&2
exit 1
fi
if ! distrobox list 2>/dev/null | grep -q "^[[:xdigit:]]\+ *| *$DISTROBOX_NAME "; then
echo "error: distrobox container '$DISTROBOX_NAME' not found" >&2
exit 1
fi
# Sanity-check the binary is compiled for the host GPU. The llama.cpp
# HIP kernels embed their target architecture as substrings like
# "amdgcn-amd-amdhsa--gfx1201". If gfx1201 is missing and e.g. only
# gfx906 is present, the build shipped kernels for a different arch
# and every inference will crash with "invalid device function".
#
# Uses `grep -a` to scan the binary directly so we don't depend on
# `strings` being on PATH (not present on a stock Bazzite host).
if ! grep -a -q gfx1201 "$BIN"; then
echo "error: $BIN does not contain gfx1201 kernels" >&2
echo " expected target for AMD Radeon RX 9070 (Navi 48)" >&2
echo " rebuild with CMAKE_HIP_ARCHITECTURES=gfx1201 inside $DISTROBOX_NAME" >&2
exit 1
fi
echo "atlas naming run starting"
echo " bin: $BIN"
echo " model: $MODEL"
echo " distrobox: $DISTROBOX_NAME"
echo " log: $LOG"
echo " bodies: ~2394 (resume-safe — already-named skipped)"
echo " estimate: ~4-6 h on an RX 9070"
echo
exec python3 tooling/planet-gen/gemma_naming.py \
--sr-voice "$BIN" \
--model "$MODEL" \
--distrobox "$DISTROBOX_NAME" \
--log "$LOG"