feat(config): T-1280 — job log retention, and the corpse that would never die
Pruning happens at spawn time rather than on a schedule: a retention pass that depends on someone remembering to run it is one that silently never happens. reach jobs prune is the explicit escape hatch for reclaiming space now. The cap was measured rather than guessed, which is why this ticket ran last. A chatty short job writes ~1.8 KB across its three files, so 100 jobs is single-digit megabytes even if a generator emits per-body progress — inside .cache/, where being wrong costs disk and never data. SR_JOB_KEEP overrides it. The interesting part is what "a running job is never pruned" has to mean. Not "the file says running" — a process killed outright never updates its own status, so that reading would make every crashed job immortal. Those are exactly the ones that accumulate, so the naive rule produces the opposite of retention: the only logs that never go away are the ones nobody wants. The check consults the process table instead. Verified both directions. Live, a running 30-second job survived a prune to --keep 1. Pinned with a fixture holding a finished job, a corpse (record says running, pid gone), and a genuinely live one — asserting the live one survives and the corpse does not. Proven to fail by dropping the liveness check. One false alarm worth recording: my first live test looked exactly like the bug, showing a running job pruned. It was not — my commands ran two minutes apart, so the "20-second" job had finished long before. The test was invalid, not the guard. A timing-sensitive check across separate shell turns proves nothing. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -111,9 +111,54 @@ def spawn_detached(argv: list[str]) -> str:
|
||||
"status": "running",
|
||||
},
|
||||
)
|
||||
|
||||
# At write time, not on a schedule. A retention pass that depends on someone
|
||||
# remembering to run it is one that silently never happens.
|
||||
prune()
|
||||
return job_id
|
||||
|
||||
|
||||
# Measured 2026-08-31: a chatty short job (10 progress events) writes ~1.8 KB
|
||||
# across its three files. A generator emitting per-body progress might reach
|
||||
# 100 KB. 100 jobs is therefore single-digit megabytes at worst, inside .cache/
|
||||
# which is gitignored scratch — so the failure mode of this number being wrong
|
||||
# is disk, never data. Raise it freely if a real batch proves it tight.
|
||||
DEFAULT_KEEP = 100
|
||||
|
||||
ENV_KEEP = "SR_JOB_KEEP"
|
||||
|
||||
|
||||
def prune(keep: int | None = None) -> list[str]:
|
||||
"""Delete all but the `keep` most recent jobs. Returns the ids removed.
|
||||
|
||||
**A genuinely running job is never pruned** — but note what that has to
|
||||
mean. Skipping anything whose *file* says "running" would make every
|
||||
crashed job immortal, because a process killed outright never gets to
|
||||
update its own status. Those are precisely the jobs that accumulate, so the
|
||||
check is against the process table, not the record.
|
||||
|
||||
Called at spawn time rather than by a sweep: a cleanup nothing invokes is a
|
||||
cleanup that does not happen.
|
||||
"""
|
||||
if keep is None:
|
||||
raw = os.environ.get(ENV_KEEP)
|
||||
keep = int(raw) if raw and raw.isdigit() else DEFAULT_KEEP
|
||||
|
||||
directory = jobs_dir()
|
||||
# Ids sort chronologically because they are timestamp-first (T-1276).
|
||||
ids = sorted((path.stem for path in directory.glob("*.json")), reverse=True)
|
||||
|
||||
removed: list[str] = []
|
||||
for job_id in ids[keep:]:
|
||||
meta = read_meta(job_id)
|
||||
if meta and meta.get("status") == "running" and is_alive(int(meta.get("pid", -1))):
|
||||
continue # actually running — a long generator may outlive the window
|
||||
for path in (log_path(job_id), output_path(job_id), meta_path(job_id)):
|
||||
path.unlink(missing_ok=True)
|
||||
removed.append(job_id)
|
||||
return removed
|
||||
|
||||
|
||||
def finish_if_detached(exit_code: int) -> None:
|
||||
"""Record completion — called by the CHILD, from the outermost decorator.
|
||||
|
||||
|
||||
@@ -10,7 +10,7 @@ import time
|
||||
|
||||
import typer
|
||||
|
||||
from tooling.core import cli, console
|
||||
from tooling.core import cli, console, process
|
||||
from tooling.core.command import command
|
||||
from tooling.core.errors import ReachExit
|
||||
from tooling.domains.jobs import service
|
||||
@@ -93,6 +93,26 @@ def log(
|
||||
console.out(console.render(event).rstrip("\n"))
|
||||
|
||||
|
||||
@app.command("prune")
|
||||
@command
|
||||
def prune(
|
||||
keep: int = typer.Option(
|
||||
process.DEFAULT_KEEP, "--keep", help="How many recent jobs to keep."
|
||||
),
|
||||
) -> None:
|
||||
"""Delete old job logs. Running jobs are kept regardless of age.
|
||||
|
||||
Pruning also happens automatically whenever a job is started, so this is the
|
||||
explicit escape hatch rather than the only path — reach for it to reclaim
|
||||
space now, or with --keep 0 to clear everything.
|
||||
"""
|
||||
removed = service.prune(keep)
|
||||
if not removed:
|
||||
console.verdict(f"nothing to prune — {keep} or fewer jobs on record")
|
||||
return
|
||||
console.verdict(f"pruned {len(removed)} job(s), kept the {keep} most recent")
|
||||
|
||||
|
||||
@app.command("wait")
|
||||
@command
|
||||
def wait(
|
||||
|
||||
@@ -96,6 +96,22 @@ def wait(job_id: str, timeout: float | None = None) -> Job:
|
||||
time.sleep(POLL_SECONDS)
|
||||
|
||||
|
||||
def prune(keep: int) -> list[str]:
|
||||
"""Delete all but the `keep` most recent jobs; return the ids removed.
|
||||
|
||||
Genuinely running jobs survive regardless of age — see `process.prune`,
|
||||
which checks the process table rather than the recorded status so that a
|
||||
crashed job is prunable and a live generator is not.
|
||||
"""
|
||||
if keep < 0:
|
||||
raise ReachError(
|
||||
f"cannot keep {keep} jobs",
|
||||
fix="pass --keep 0 to remove everything, or a positive number to keep some",
|
||||
exit_code=2,
|
||||
)
|
||||
return process.prune(keep)
|
||||
|
||||
|
||||
def duration(job: Job) -> str:
|
||||
"""Human-readable elapsed time, or how long it has been running so far."""
|
||||
start = _parse(job.started_at)
|
||||
|
||||
@@ -110,12 +110,85 @@ def test_died_never_relays_success(failures: list[str]) -> None:
|
||||
failures.append("failed job did not relay its own exit code")
|
||||
|
||||
|
||||
def _fixture_jobs(root: Path, specs: list[tuple[str, str, int]]) -> None:
|
||||
"""Write job metadata into a throwaway repo root. specs: (id, status, pid)."""
|
||||
directory = root / ".cache" / "reach" / "jobs"
|
||||
directory.mkdir(parents=True, exist_ok=True)
|
||||
for job_id, status, pid in specs:
|
||||
(directory / f"{job_id}.json").write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"job": job_id,
|
||||
"command": "reach dev selftest",
|
||||
"argv": ["dev", "selftest"],
|
||||
"pid": pid,
|
||||
"started_at": "2026-01-01T00:00:00",
|
||||
"status": status,
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
(directory / f"{job_id}.jsonl").write_text("", encoding="utf-8")
|
||||
|
||||
|
||||
def test_prune_spares_running_but_not_corpses(failures: list[str]) -> None:
|
||||
"""The retention rule, and the half of it that is easy to get wrong.
|
||||
|
||||
A job that is genuinely running survives pruning however old it is — a
|
||||
generator can outlive the window. But a job whose FILE says running while
|
||||
its process is gone must be prunable, or every crashed job becomes immortal
|
||||
and those are exactly what accumulates. The check has to consult the process
|
||||
table, not the record.
|
||||
"""
|
||||
import os
|
||||
|
||||
from tooling.core import process
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
root = Path(tmp)
|
||||
(root / "project.yaml").write_text("version: 0.0.0\n", encoding="utf-8")
|
||||
_fixture_jobs(
|
||||
root,
|
||||
[
|
||||
("20260101T000001-aaaa", "done", 1),
|
||||
("20260101T000002-bbbb", "running", 4_000_000), # corpse: dead pid
|
||||
("20260101T000003-cccc", "running", os.getpid()), # genuinely alive
|
||||
("20260101T000004-dddd", "done", 1),
|
||||
],
|
||||
)
|
||||
|
||||
previous = os.environ.get("SR_REPO_ROOT")
|
||||
os.environ["SR_REPO_ROOT"] = str(root)
|
||||
try:
|
||||
removed = set(process.prune(keep=1))
|
||||
finally:
|
||||
if previous is None:
|
||||
del os.environ["SR_REPO_ROOT"]
|
||||
else:
|
||||
os.environ["SR_REPO_ROOT"] = previous
|
||||
|
||||
if "20260101T000003-cccc" in removed:
|
||||
failures.append(
|
||||
"prune: removed a job whose process is alive — a long generator "
|
||||
"would lose its own log while still writing to it"
|
||||
)
|
||||
if "20260101T000002-bbbb" not in removed:
|
||||
failures.append(
|
||||
"prune: kept a job whose file says running but whose process is "
|
||||
"gone — a crashed job must not be immortal, and those are exactly "
|
||||
"the ones that accumulate"
|
||||
)
|
||||
if "20260101T000001-aaaa" not in removed:
|
||||
failures.append("prune: kept a finished job beyond the keep window")
|
||||
|
||||
|
||||
def main() -> int:
|
||||
failures: list[str] = []
|
||||
test_partial_trailing_line(failures)
|
||||
test_offset_resume_is_stable(failures)
|
||||
test_dead_pid_is_reconciled(failures)
|
||||
test_died_never_relays_success(failures)
|
||||
test_prune_spares_running_but_not_corpses(failures)
|
||||
|
||||
if failures:
|
||||
print("test_jobs: FAIL", file=sys.stderr)
|
||||
|
||||
Reference in New Issue
Block a user