Pruning happens at spawn time rather than on a schedule: a retention pass that depends on someone remembering to run it is one that silently never happens. reach jobs prune is the explicit escape hatch for reclaiming space now. The cap was measured rather than guessed, which is why this ticket ran last. A chatty short job writes ~1.8 KB across its three files, so 100 jobs is single-digit megabytes even if a generator emits per-body progress — inside .cache/, where being wrong costs disk and never data. SR_JOB_KEEP overrides it. The interesting part is what "a running job is never pruned" has to mean. Not "the file says running" — a process killed outright never updates its own status, so that reading would make every crashed job immortal. Those are exactly the ones that accumulate, so the naive rule produces the opposite of retention: the only logs that never go away are the ones nobody wants. The check consults the process table instead. Verified both directions. Live, a running 30-second job survived a prune to --keep 1. Pinned with a fixture holding a finished job, a corpse (record says running, pid gone), and a genuinely live one — asserting the live one survives and the corpse does not. Proven to fail by dropping the liveness check. One false alarm worth recording: my first live test looked exactly like the bug, showing a running job pruned. It was not — my commands ran two minutes apart, so the "20-second" job had finished long before. The test was invalid, not the guard. A timing-sensitive check across separate shell turns proves nothing. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
232 lines
8.1 KiB
Python
232 lines
8.1 KiB
Python
"""Detached execution: spawn a child that outlives its parent (D-263).
|
|
|
|
Substrate, not a domain — this has no verbs of its own. The verbs (`list`,
|
|
`status`, `log`, `wait`) have logic and state and are therefore
|
|
`reach jobs …`, which is the `core/` bound doing its job.
|
|
|
|
**Three things here are easy to get subtly wrong, and each has a comment where
|
|
it is handled rather than only here:**
|
|
|
|
1. *The child must genuinely outlive the parent.* `start_new_session=True` puts
|
|
it in its own session and process group, so a signal to the parent's group —
|
|
or the parent simply being killed on a timeout — does not take the work with
|
|
it. A background shell job would not survive that, which is the whole reason
|
|
detach exists.
|
|
2. *The child re-execs `reach` by BARE NAME.* Never an interpreter path, never
|
|
`.venv/bin/reach` (T-1261). An absolute path would freeze the child to
|
|
whichever checkout was current at spawn time, so after `make reach-repoint`
|
|
a detached job would silently run the wrong source with no error anywhere —
|
|
exactly the failure that command exists to fix.
|
|
3. *The exit code is recorded by the CHILD as its last act.* Not polled by a
|
|
parent that has already returned. A parent cannot observe an exit it is no
|
|
longer around for, and a runner that loses the failure is the exit-0 trap
|
|
from D-263 relocated somewhere nothing is watching.
|
|
|
|
Streams stay separated exactly as they are in the foreground: the event stream
|
|
to `<id>.jsonl`, the command's real output to `<id>.out`. Merging them would
|
|
make the log unparseable for the sake of one fewer file.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
import subprocess
|
|
import sys
|
|
import time
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from tooling.core import config, jobs
|
|
|
|
# Under .cache/, which is gitignored and already the repo's scratch space — so a
|
|
# wrong answer about retention costs disk, never data.
|
|
JOBS_SUBPATH = (".cache", "reach", "jobs")
|
|
|
|
|
|
def jobs_dir() -> Path:
|
|
path = config.path(*JOBS_SUBPATH)
|
|
path.mkdir(parents=True, exist_ok=True)
|
|
return path
|
|
|
|
|
|
def log_path(job_id: str) -> Path:
|
|
return jobs_dir() / f"{job_id}.jsonl"
|
|
|
|
|
|
def output_path(job_id: str) -> Path:
|
|
return jobs_dir() / f"{job_id}.out"
|
|
|
|
|
|
def meta_path(job_id: str) -> Path:
|
|
return jobs_dir() / f"{job_id}.json"
|
|
|
|
|
|
def spawn_detached(argv: list[str]) -> str:
|
|
"""Run `reach <argv>` in a detached child. Returns the job id immediately.
|
|
|
|
The caller is expected to report the id and exit — it must NOT wait, since
|
|
not waiting is the entire point.
|
|
"""
|
|
job_id = jobs.new_id()
|
|
directory = jobs_dir()
|
|
|
|
# Opened here and inherited by the child, which then owns them. The parent
|
|
# closes its copies below; the child keeps writing after the parent is gone.
|
|
log_file = open(directory / f"{job_id}.jsonl", "wb")
|
|
out_file = open(directory / f"{job_id}.out", "wb")
|
|
|
|
child_env = {
|
|
**os.environ,
|
|
jobs.ENV_JOB_ID: job_id,
|
|
# Force machine format: the child's stderr is a file, so isatty would
|
|
# already say JSONL — but being explicit means a future TTY-inheriting
|
|
# spawn cannot silently start writing prose into a log meant to be read
|
|
# back as events.
|
|
"SR_OUTPUT_FORMAT": "json",
|
|
}
|
|
|
|
try:
|
|
process = subprocess.Popen(
|
|
["reach", *argv], # BARE NAME — see note 2 in the module docstring
|
|
stdout=out_file,
|
|
stderr=log_file,
|
|
stdin=subprocess.DEVNULL,
|
|
start_new_session=True, # note 1: its own session, survives the parent
|
|
env=child_env,
|
|
cwd=config.repo_root(),
|
|
)
|
|
finally:
|
|
log_file.close()
|
|
out_file.close()
|
|
|
|
_write_meta(
|
|
job_id,
|
|
{
|
|
"job": job_id,
|
|
"argv": argv,
|
|
"command": " ".join(["reach", *argv]),
|
|
"pid": process.pid,
|
|
"started_at": _now(),
|
|
"status": "running",
|
|
},
|
|
)
|
|
|
|
# At write time, not on a schedule. A retention pass that depends on someone
|
|
# remembering to run it is one that silently never happens.
|
|
prune()
|
|
return job_id
|
|
|
|
|
|
# Measured 2026-08-31: a chatty short job (10 progress events) writes ~1.8 KB
|
|
# across its three files. A generator emitting per-body progress might reach
|
|
# 100 KB. 100 jobs is therefore single-digit megabytes at worst, inside .cache/
|
|
# which is gitignored scratch — so the failure mode of this number being wrong
|
|
# is disk, never data. Raise it freely if a real batch proves it tight.
|
|
DEFAULT_KEEP = 100
|
|
|
|
ENV_KEEP = "SR_JOB_KEEP"
|
|
|
|
|
|
def prune(keep: int | None = None) -> list[str]:
|
|
"""Delete all but the `keep` most recent jobs. Returns the ids removed.
|
|
|
|
**A genuinely running job is never pruned** — but note what that has to
|
|
mean. Skipping anything whose *file* says "running" would make every
|
|
crashed job immortal, because a process killed outright never gets to
|
|
update its own status. Those are precisely the jobs that accumulate, so the
|
|
check is against the process table, not the record.
|
|
|
|
Called at spawn time rather than by a sweep: a cleanup nothing invokes is a
|
|
cleanup that does not happen.
|
|
"""
|
|
if keep is None:
|
|
raw = os.environ.get(ENV_KEEP)
|
|
keep = int(raw) if raw and raw.isdigit() else DEFAULT_KEEP
|
|
|
|
directory = jobs_dir()
|
|
# Ids sort chronologically because they are timestamp-first (T-1276).
|
|
ids = sorted((path.stem for path in directory.glob("*.json")), reverse=True)
|
|
|
|
removed: list[str] = []
|
|
for job_id in ids[keep:]:
|
|
meta = read_meta(job_id)
|
|
if meta and meta.get("status") == "running" and is_alive(int(meta.get("pid", -1))):
|
|
continue # actually running — a long generator may outlive the window
|
|
for path in (log_path(job_id), output_path(job_id), meta_path(job_id)):
|
|
path.unlink(missing_ok=True)
|
|
removed.append(job_id)
|
|
return removed
|
|
|
|
|
|
def finish_if_detached(exit_code: int) -> None:
|
|
"""Record completion — called by the CHILD, from the outermost decorator.
|
|
|
|
A no-op in a foreground run, which has no metadata file to update. Note 3
|
|
in the module docstring is why this lives on the child's exit path rather
|
|
than in whatever spawned it.
|
|
"""
|
|
job_id = os.environ.get(jobs.ENV_JOB_ID)
|
|
if not job_id:
|
|
return
|
|
meta = read_meta(job_id)
|
|
if meta is None:
|
|
return
|
|
meta.update(
|
|
{
|
|
"status": "done" if exit_code == 0 else "failed",
|
|
"exit_code": exit_code,
|
|
"ended_at": _now(),
|
|
}
|
|
)
|
|
_write_meta(job_id, meta)
|
|
|
|
|
|
def read_meta(job_id: str) -> dict[str, Any] | None:
|
|
path = meta_path(job_id)
|
|
if not path.is_file():
|
|
return None
|
|
try:
|
|
return json.loads(path.read_text(encoding="utf-8"))
|
|
except json.JSONDecodeError:
|
|
return None
|
|
|
|
|
|
def is_alive(pid: int) -> bool:
|
|
"""Whether a recorded pid is still running.
|
|
|
|
Needed because a child killed outright — SIGKILL, OOM, a crash in the
|
|
interpreter itself — never gets to record its own completion, and its
|
|
metadata would otherwise say "running" forever. Reconciling against the
|
|
process table is what stops a dead job from looking like a busy one.
|
|
"""
|
|
try:
|
|
os.kill(pid, 0)
|
|
except ProcessLookupError:
|
|
return False
|
|
except PermissionError:
|
|
return True # exists, owned by someone else
|
|
return True
|
|
|
|
|
|
def _write_meta(job_id: str, meta: dict[str, Any]) -> None:
|
|
# Written via a temporary file and renamed, because `jobs list` may read
|
|
# this at any moment and a half-written JSON file is an unreadable job.
|
|
target = meta_path(job_id)
|
|
temporary = target.with_suffix(".json.tmp")
|
|
temporary.write_text(json.dumps(meta, indent=2), encoding="utf-8")
|
|
temporary.replace(target)
|
|
|
|
|
|
def _now() -> str:
|
|
return time.strftime("%Y-%m-%dT%H:%M:%S", time.gmtime())
|
|
|
|
|
|
def current_argv() -> list[str]:
|
|
"""The invocation's arguments with `--detach` removed.
|
|
|
|
Removed because the child must not detach again — it would fork forever,
|
|
each generation spawning another and none doing the work.
|
|
"""
|
|
return [arg for arg in sys.argv[1:] if arg != "--detach"]
|