Files
odysseus/src/process_reaper.py
T
Alexandre Teixeira f9aa2818c4 refactor(runtime): centralize verified process lifecycle
Extract the generic process lifecycle layer (src/process_lifecycle.py)
shared by runtime-owned subprocesses: process identity (pid + boot-bound
start token), identity-bound observation, group and pidfd probes, the
TERM -> verify -> KILL -> verify escalation with re-gating before
escalation, identity-scoped sweeps, and the termination receipt.

Containment, the PTY shell, the Cookbook survivor sweep, the browser
lifecycle, web_tools browser cleanup, kill_process_tree and the startup
reaper consume it while keeping their own ownership semantics.

Safety corrections:
- browser membership and identity are bound in one snapshot; no identity
  is recaptured after membership is decided
- web_tools legacy pid-file and profile-match kills signal only verified
  identities; browser CLI groups only while their spawn identity verifies
- Cookbook and legacy-tmux descendant capture bind membership to identity
- PTY teardown never signals the server's own process group
- unverifiable processes are reported, never signalled
2026-10-02 01:27:08 +01:00

307 lines
14 KiB
Python

"""Startup reconciliation for processes a previous run left behind.
Two stores in this tree outlive the process that wrote them, on purpose:
``data/containment_grants.json`` so a restart can reap rather than orphan, and
``data/bg_jobs.json`` so a restart never loses a detached job or its result.
Until now nothing read either of them at startup. A crashed or restarted server
therefore left every grant permanently "active" and every background job
permanently "running", and the first thing to touch one of those records was a
teardown aimed at a pid that had been reassigned in the meantime.
This module runs once, during startup, before anything of this run exists. That
timing is what makes its rules safe: every record it sees was written by an
earlier run, so "I cannot identify this process" is information about a previous
run's child and not about one of ours.
The two stores get **opposite** treatment, which is the whole reason this is a
module and not a loop:
* A **containment grant** is tied to a tool call that no longer has a caller.
A live process under an abandoned grant is by definition an orphan, so it is
torn down.
* A **background job** is detached deliberately and is documented to survive a
uvicorn restart. Killing one here would break the feature, so its record is
only corrected, never reaped. What gets fixed is identity: a job whose pid now
belongs to someone else is retired so that nothing later signals the stranger.
Fail closed in both: a signal requires a positive identity from
:mod:`src.process_ownership`, and every other verdict is recorded rather than
acted on. Containment that cannot identify its target is not containment, and
the honest failure is a visible orphan rather than a dead bystander.
"""
from __future__ import annotations
import logging
from typing import Any, Dict
from src import process_lifecycle, process_ownership
logger = logging.getLogger(__name__)
def reap_containment_grants() -> Dict[str, Any]:
"""Tear down or retire every grant a previous run left active.
Per grant: a verified live process is torn down through
:func:`src.containment.reap_record`; a grant whose process is gone is
dropped; a grant naming a pid that is now someone else's is dropped
*without a signal*, because the only thing left to do with it is stop
believing it. A grant that cannot be verified at all is **kept**, so the
orphan stays visible in ``active_grants()`` instead of being quietly
written off as handled.
"""
from src import containment
report: Dict[str, Any] = {
"seen": 0, "torn_down": 0, "already_gone": 0,
"foreign": 0, "unverifiable": 0, "failed": 0,
}
try:
records = containment.active_grants()
except Exception:
logger.warning("process_reaper: containment grant store unreadable", exc_info=True)
return report
for record in records:
report["seen"] += 1
grant_id = str(record.get("id") or "")
if record.get("external"):
# Nothing local ever ran, so there is nothing local to reap.
containment.forget(grant_id)
report["already_gone"] += 1
continue
# A grant's holder — the detached supervisor of a background job, or
# the server process that acquired it — is an identity like any
# other: a live pid in its slot proves nothing without its token.
supervisor = process_lifecycle.ProcessIdentity.from_record(
record, pid_key="supervisor_pid", token_key="supervisor_token")
if record.get("lifetime") == "background" and supervisor and supervisor.owned():
# Detached jobs deliberately survive a server restart. Their
# supervisor owns the wall clock and teardown, independently.
report["background_kept"] = report.get("background_kept", 0) + 1
continue
manager = process_lifecycle.ProcessIdentity.from_record(
record, pid_key="manager_pid", token_key="manager_token")
if record.get("lifetime") != "cleanup" and manager and manager.owned():
report["manager_kept"] = report.get("manager_kept", 0) + 1
continue
verdict = process_ownership.verify_record(record)
if verdict == process_ownership.GONE:
if containment._group_present(record.get("pgid")):
# Leader death does not prove tree death. Without a surviving
# identity we cannot signal the group, so retain the evidence.
report["failed"] += 1
logger.error("process_reaper: grant %s leader is gone but group survives", grant_id)
continue
containment.forget(grant_id)
report["already_gone"] += 1
continue
if verdict == process_ownership.FOREIGN:
logger.warning(
"process_reaper: grant %s named pid %s, which now belongs to a "
"different process; dropping the record unsignalled",
grant_id, record.get("pid"),
)
containment.forget(grant_id)
report["foreign"] += 1
continue
if verdict == process_ownership.UNVERIFIABLE:
logger.error(
"process_reaper: grant %s (pid %s, owner %s) cannot be verified "
"via %s; leaving it active and unsignalled — this is a "
"containment failure, not a clean start",
grant_id, record.get("pid"), record.get("owner"),
process_ownership.inspection_mechanism(),
)
report["unverifiable"] += 1
continue
try:
outcome = containment.reap_record(record)
except Exception:
logger.warning("process_reaper: tearing down grant %s failed", grant_id, exc_info=True)
report["failed"] += 1
continue
if outcome.dead:
containment.forget(grant_id)
report["torn_down"] += 1
else:
logger.error(
"process_reaper: grant %s survived teardown; survivors=%s",
grant_id, list(outcome.survivors),
)
report["failed"] += 1
return report
def reap_bg_jobs() -> Dict[str, Any]:
"""Correct the identity of background jobs a previous run launched.
Deliberately kills nothing: a ``#!bg`` job is detached so that it outlives
the request *and* the server, and the store exists so its result is still
collected afterwards. The defect being closed is narrower — a record whose
pid has been reassigned will be signalled by the max-runtime reaper an hour
later, and that signal lands on whatever now holds the pid.
"""
from src import bg_jobs
try:
return bg_jobs.disown_unverified()
except Exception:
logger.warning("process_reaper: background job store unreadable", exc_info=True)
return {"seen": 0, "retired": 0, "kept": 0}
def reap_legacy_agent_tmux() -> Dict[str, Any]:
"""Retire this runtime's legacy agent shells; a name prefix is not ownership.
Match the original clean Bash launcher and this runtime's HOME marker on
every pane. Snapshot session/server identities and process start tokens
before teardown; ambiguous sessions remain visible and unsignalled.
"""
import os
import re
import shlex
import shutil
import subprocess
import uuid
from src import containment
from src.constants import DATA_DIR
report = {"seen": 0, "torn_down": 0, "unverifiable": 0, "failed": 0}
tmux = shutil.which("tmux")
if os.name == "nt" or not tmux:
return report
pattern = "#{session_id}\t#{session_name}\t#{session_created}\t#{pane_pid}\t#{pane_id}\t#{pid}\t#{pane_start_command}"
def snapshot():
result = subprocess.run([tmux, "list-panes", "-a", "-F", pattern],
capture_output=True, text=True, timeout=5)
if result.returncode:
if not result.stdout and any(message in result.stderr.lower() for message in ("no server", "no sessions", "error connecting")):
return {}
raise RuntimeError("tmux pane discovery failed")
sessions = {}
for line in result.stdout.splitlines():
fields = line.split("\t", 6)
if len(fields) != 7 or not fields[1].startswith("ody-agent-"):
continue
sessions.setdefault(fields[0], []).append(tuple(fields))
return {key: sorted(rows) for key, rows in sessions.items()}
def launcher_is_ours(command):
try:
argv = shlex.split(command)
except ValueError:
return False
if not argv or argv.pop(0) != "env":
return False
env = {}
while argv and "=" in argv[0]:
key, value = argv.pop(0).split("=", 1)
if key not in {"PATH", "VIRTUAL_ENV", "HOME", "TMPDIR", "TERM", "COLUMNS", "LINES"}:
return False
env[key] = value
return argv == ["/bin/bash", "--noprofile", "--norc"] and env.get("HOME") == DATA_DIR
try:
sessions = snapshot()
for session_id, panes in sessions.items():
report["seen"] += 1
if not re.fullmatch(r"\$\d+", session_id) or not all(launcher_is_ours(row[6]) for row in panes):
report["unverifiable"] += 1
continue
server_pid = int(panes[0][5])
server_token = process_ownership.start_token(server_pid)
roots = [int(row[3]) for row in panes]
table = process_ownership.process_table()
if not all(pid in table and table[pid].ppid == server_pid and shlex.split(table[pid].command) == [
"/bin/bash", "--noprofile", "--norc",
] for pid in roots):
# A stale pane PID can now name a bystander. Its parent and
# current launcher must still match the observed tmux server.
report["unverifiable"] += 1
continue
# Membership and identity bound together: a descendant that changed
# hands after the table was read is dropped, not recorded under a
# stranger's token. One this host cannot identify keeps the whole
# session visible and unsignalled.
bound = process_lifecycle.bind_descendants(roots)
targets = [seen.identity.pid for seen in bound]
identities = {seen.identity.pid: seen.identity.start_token for seen in bound}
if any(token is None for token in identities.values()):
report["unverifiable"] += 1
continue
if snapshot().get(session_id) != panes or process_ownership.verify(server_pid, server_token) != process_ownership.OWNED or any(
process_ownership.verify(pid, identities.get(pid)) != process_ownership.OWNED for pid in roots
):
report["unverifiable"] += 1
continue
# Persist every positively identified tree before touching it. A
# failed teardown then remains discoverable even if its pane dies.
tracked = []
for pid in reversed(targets):
if process_ownership.verify(pid, identities[pid]) != process_ownership.OWNED:
continue
spec = containment.ContainmentSpec(workspace=os.getcwd(), env={}, wall_clock_s=1,
required=frozenset())
grant = containment.ContainmentGrant(
id=uuid.uuid4().hex[:12], mechanism="process_group", workspace=spec.workspace,
enforced=frozenset(), degraded=(containment.FILESYSTEM, containment.PROCESS_TREE), unenforced_required=(),
owner=f"legacy-tmux:{session_id}", mode=containment.MODE_ENFORCING,
spec=spec, pid=pid, pgid=containment._pgid_of(pid),
)
containment._write_record(grant)
containment._update_record(grant.id, lifetime="cleanup", start_token=identities[pid])
receipt = containment._load_records().get(grant.id, {})
if receipt.get("start_token") != identities[pid] or receipt.get("lifetime") != "cleanup":
raise RuntimeError("legacy tmux cleanup receipt was not persisted")
tracked.append((grant, identities[pid]))
dead = True
for grant, token in tracked:
outcome = containment.release(grant, start_token=token, require_identity=True)
dead = dead and outcome.dead
remaining = snapshot().get(session_id)
if remaining and dead:
# Use the immutable tmux session id, not its reusable name.
if remaining != panes or process_ownership.verify(server_pid, server_token) != process_ownership.OWNED:
dead = False
else:
result = subprocess.run([tmux, "kill-session", "-t", session_id],
capture_output=True, timeout=5)
dead = result.returncode == 0 and session_id not in snapshot()
report["torn_down" if dead else "failed"] += 1
except Exception:
report["failed"] += 1
logger.warning("process_reaper: legacy agent tmux cleanup failed", exc_info=True)
if report["unverifiable"]:
logger.warning("process_reaper: left %s legacy tmux sessions without positive ownership", report["unverifiable"])
return report
def reap_orphans() -> Dict[str, Any]:
"""Run both reconciliations. Returns a report; raises nothing.
Blocking: a teardown escalates SIGTERM → grace → SIGKILL and waits for the
process to actually go. Call it off the event loop.
"""
report = {
"mechanism": process_ownership.inspection_mechanism(),
"grants": reap_containment_grants(),
"bg_jobs": reap_bg_jobs(),
"agent_tmux": reap_legacy_agent_tmux(),
}
if report["mechanism"] == process_ownership.MECHANISM_NONE:
logger.error(
"process_reaper: this host offers no process inspection; no orphan "
"from a previous run can be identified or reaped"
)
logger.info("process_reaper: startup reconciliation %s", report)
return report
async def reap_orphans_at_startup() -> Dict[str, Any]:
""":func:`reap_orphans` off the event loop, for an app startup task."""
import asyncio
return await asyncio.to_thread(reap_orphans)