Pruning happens at spawn time rather than on a schedule: a retention pass that depends on someone remembering to run it is one that silently never happens. reach jobs prune is the explicit escape hatch for reclaiming space now. The cap was measured rather than guessed, which is why this ticket ran last. A chatty short job writes ~1.8 KB across its three files, so 100 jobs is single-digit megabytes even if a generator emits per-body progress — inside .cache/, where being wrong costs disk and never data. SR_JOB_KEEP overrides it. The interesting part is what "a running job is never pruned" has to mean. Not "the file says running" — a process killed outright never updates its own status, so that reading would make every crashed job immortal. Those are exactly the ones that accumulate, so the naive rule produces the opposite of retention: the only logs that never go away are the ones nobody wants. The check consults the process table instead. Verified both directions. Live, a running 30-second job survived a prune to --keep 1. Pinned with a fixture holding a finished job, a corpse (record says running, pid gone), and a genuinely live one — asserting the live one survives and the corpse does not. Proven to fail by dropping the liveness check. One false alarm worth recording: my first live test looked exactly like the bug, showing a running job pruned. It was not — my commands ran two minutes apart, so the "20-second" job had finished long before. The test was invalid, not the guard. A timing-sensitive check across separate shell turns proves nothing. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
137 lines
4.9 KiB
Python
137 lines
4.9 KiB
Python
"""Transport for the `jobs` domain — args in, delegate, format out.
|
|
|
|
Zero logic. Reconciliation, offsets and polling all live in `service.py`; what
|
|
happens here is turning a `Job` into lines and an exit code.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import time
|
|
|
|
import typer
|
|
|
|
from tooling.core import cli, console, process
|
|
from tooling.core.command import command
|
|
from tooling.core.errors import ReachExit
|
|
from tooling.domains.jobs import service
|
|
from tooling.domains.jobs.schemas import Status
|
|
|
|
app = cli.domain("jobs", "Detached runs — what is running, what it printed, how it ended.")
|
|
|
|
|
|
@app.callback()
|
|
def _domain() -> None:
|
|
"""Keeps `jobs` a group (Typer collapses a single-command app)."""
|
|
|
|
|
|
@app.command("list")
|
|
@command
|
|
def list_jobs(
|
|
limit: int = typer.Option(20, "--limit", "-n", help="How many recent jobs to show."),
|
|
) -> None:
|
|
"""Recent detached runs, newest first."""
|
|
jobs = service.list_jobs(limit)
|
|
if not jobs:
|
|
console.out("no jobs recorded — start one with: reach --detach <command>")
|
|
return
|
|
for job in jobs:
|
|
console.out(
|
|
f"{job.job} {job.status.value:<8} {service.duration(job):>7} {job.command}"
|
|
)
|
|
|
|
|
|
@app.command("status")
|
|
@command
|
|
def status(job_id: str = typer.Argument(..., help="Job id, as printed by --detach.")) -> None:
|
|
"""One job's outcome, reconciled against whether its process is alive."""
|
|
job = service.get(job_id)
|
|
console.out(f"job {job.job}")
|
|
console.out(f"command {job.command}")
|
|
console.out(f"status {job.status.value}")
|
|
console.out(f"started {job.started_at}")
|
|
console.out(f"elapsed {service.duration(job)}")
|
|
if job.exit_code is not None:
|
|
console.out(f"exit {job.exit_code}")
|
|
if job.status is Status.DIED:
|
|
# Said in words, because "died" alone reads like a synonym for "failed"
|
|
# and the distinction matters: nothing recorded an outcome here.
|
|
console.out("")
|
|
console.out("This job's process is gone but it never recorded an ending —")
|
|
console.out("killed outright (SIGKILL, OOM, or a crash). Its log holds")
|
|
console.out("whatever it managed to emit before that.")
|
|
|
|
|
|
@app.command("log")
|
|
@command
|
|
def log(
|
|
job_id: str = typer.Argument(..., help="Job id, as printed by --detach."),
|
|
follow: bool = typer.Option(False, "--follow", "-f", help="Keep printing as it runs."),
|
|
) -> None:
|
|
"""Replay a job's event stream, rendered as it appeared live."""
|
|
job = service.get(job_id)
|
|
events, offset = service.read_events(job_id)
|
|
for event in events:
|
|
# Rendered through console, not a local formatter, so a stored log and a
|
|
# live run are one artefact in two presentations rather than two
|
|
# renderers that drift apart precisely when someone is debugging.
|
|
console.out(console.render(event).rstrip("\n"))
|
|
|
|
if not follow:
|
|
return
|
|
|
|
while not job.status.finished:
|
|
time.sleep(service.POLL_SECONDS)
|
|
events, offset = service.read_events(job_id, offset)
|
|
for event in events:
|
|
console.out(console.render(event).rstrip("\n"))
|
|
job = service.get(job_id)
|
|
|
|
# One last read: the job may have written its final events between the last
|
|
# poll and its exit, and stopping at the status flip would drop them.
|
|
events, offset = service.read_events(job_id, offset)
|
|
for event in events:
|
|
console.out(console.render(event).rstrip("\n"))
|
|
|
|
|
|
@app.command("prune")
|
|
@command
|
|
def prune(
|
|
keep: int = typer.Option(
|
|
process.DEFAULT_KEEP, "--keep", help="How many recent jobs to keep."
|
|
),
|
|
) -> None:
|
|
"""Delete old job logs. Running jobs are kept regardless of age.
|
|
|
|
Pruning also happens automatically whenever a job is started, so this is the
|
|
explicit escape hatch rather than the only path — reach for it to reclaim
|
|
space now, or with --keep 0 to clear everything.
|
|
"""
|
|
removed = service.prune(keep)
|
|
if not removed:
|
|
console.verdict(f"nothing to prune — {keep} or fewer jobs on record")
|
|
return
|
|
console.verdict(f"pruned {len(removed)} job(s), kept the {keep} most recent")
|
|
|
|
|
|
@app.command("wait")
|
|
@command
|
|
def wait(
|
|
job_id: str = typer.Argument(..., help="Job id, as printed by --detach."),
|
|
timeout: float = typer.Option(None, "--timeout", help="Give up after N seconds."),
|
|
) -> None:
|
|
"""Block until a job finishes, then exit with ITS exit code.
|
|
|
|
That relay is the point: a Makefile or a hook can gate on a detached run
|
|
exactly as it would on a foreground one. `ReachExit` rather than
|
|
`ReachError` because waiting successfully for a job that failed is not a
|
|
failure of `wait`.
|
|
"""
|
|
job = service.wait(job_id, timeout)
|
|
console.verdict(
|
|
f"job {job.job} {job.status.value} after {service.duration(job)}"
|
|
+ (f" (exit {job.exit_code})" if job.exit_code is not None else ""),
|
|
ok=job.status is Status.DONE,
|
|
fix=None if job.status is Status.DONE else f"reach jobs log {job.job}",
|
|
)
|
|
raise ReachExit(job.effective_exit_code)
|