Merge frozen lab b1666951 (Wave 3) into Wave 4 effects provenance

Integrates the merged and frozen Wave 3 lab commit
b1666951faf8285054e1ca90f11533b0fb53fb57 with a normal merge, preserving
every Wave 4 commit unchanged.

Conflict: src/agent_runtime/resources.py. Wave 3's _control_plane_snapshot()
/ _control_plane_path(path, *, snapshot=None) split is kept. The snapshot adds
the effect-store directories to its prefix set after the recursive job-dir
inventory and no longer references path (the auto-merged prefix check would
have raised NameError there). _control_plane_path calls _aliases_effect_store
after its os.stat, only for multiply linked files, so single-link files never
list the store.

Semantic reconciliation (no textual conflict): bg_monitor keeps launch
settlement right after the first successful validate_job and before the
authority check, with Wave 3's post-drain revalidation intact. The
deleted-session branch, terminal before linkage validation, now settles a
validated launch too: that job is later pruned and its publication retired,
which would otherwise leave its effect RUNNING. Regression tests cover the
snapshot form of the effect-store check and both deleted-session linkage
outcomes.
This commit is contained in:
Alexandre Teixeira
2026-10-03 01:13:36 +01:00
31 changed files with 1416 additions and 89 deletions
+46 -16
View File
@@ -13,6 +13,7 @@ from __future__ import annotations
import asyncio
import json
import logging
from enum import Enum, auto
from src import bg_jobs
from src.prompt_security import untrusted_context_message
@@ -26,6 +27,12 @@ POLL_INTERVAL_S = 5
_FOLLOWUP_MAX_ROUNDS = 12
class FollowupResult(Enum):
RETRYABLE_LATER = auto()
COMPLETED = auto()
TERMINAL_UNFOLLOWABLE = auto()
def _background_result_message(rec):
inject = (
f"[Background job {rec['id']} finished]\n\n"
@@ -120,22 +127,30 @@ async def _drain_agent(sess, messages, request_authority=None):
return full, tool_events
async def _run_followup(rec: dict) -> bool:
"""Re-invoke the agent in the job's session with the result. Returns True
if the follow-up completed (or there's nothing to do) — i.e. it's safe to
mark followed_up. Returns False to retry on the next tick."""
async def _run_followup(rec: dict) -> FollowupResult:
"""Continue only an exactly linked result; distinguish retry from terminal."""
from src.ai_interaction import get_session_manager
from core.models import ChatMessage
sm = get_session_manager()
if not sm:
return False # not ready yet — retry
return FollowupResult.RETRYABLE_LATER
sess = sm.get_session(rec["session_id"])
if not sess:
# Session was deleted — nothing to continue. Consider it handled so we
# don't retry forever.
logger.info("bg-followup: session %s gone for job %s — skipping", rec.get("session_id"), rec.get("id"))
return True
# The job is retired without a continuation, then pruned with its
# publication. Settle its launch effect first so it is not left RUNNING.
from src.agent_runtime.process_resources import job_from_record, validate_job
try:
resource = job_from_record(rec)
validate_job(resource)
except (ValueError, TypeError, OSError, RuntimeError):
pass # no validated linkage: nothing may be settled
else:
_settle_launch_effect(resource, rec)
return FollowupResult.TERMINAL_UNFOLLOWABLE
# Don't write into a session that's mid-stream. The followup appends to
# history + save_sessions(); a concurrent live turn does the same, and with
@@ -145,13 +160,10 @@ async def _run_followup(rec: dict) -> bool:
from src import agent_runs
if agent_runs.is_active(sess.id):
logger.info("bg-followup: session %s busy (live turn) — deferring job %s", sess.id, rec.get("id"))
return False
return FollowupResult.RETRYABLE_LATER
except Exception:
pass
context = sess.get_context_messages()
context.append(_background_result_message(rec))
from src.agent_runtime.authority import restore_background_authority
from src.settings import get_setting
authority = restore_background_authority(
@@ -165,11 +177,19 @@ async def _run_followup(rec: dict) -> bool:
_settle_launch_effect(resource, rec)
if not authority.grants or (resource.owner, resource.thread_id, resource.request_id) != (
str(getattr(sess, "owner", None) or "").strip().casefold(), sess.id, authority.request_id):
return False
return FollowupResult.TERMINAL_UNFOLLOWABLE
except (ValueError, TypeError, OSError, RuntimeError):
return False
return FollowupResult.TERMINAL_UNFOLLOWABLE
context = sess.get_context_messages()
context.append(_background_result_message(rec))
authority = authority.restrict(disabled_tools=get_setting("disabled_tools", []) or ())
full, tool_events = await _drain_agent(sess, context, request_authority=authority)
# An awaited continuation must not deliver a result after its immutable
# linkage disappears or is replaced. This check grants no new authority.
try:
validate_job(resource)
except (ValueError, TypeError, OSError, RuntimeError):
return FollowupResult.TERMINAL_UNFOLLOWABLE
# Persist ONLY the assistant continuation so it renders as a normal agent
# turn — a standard chat bubble plus `tool_events` that the frontend
@@ -188,7 +208,19 @@ async def _run_followup(rec: dict) -> bool:
sm.save_sessions()
logger.info("bg-followup: auto-continued session %s for job %s (%d chars, %d tools)",
sess.id, rec["id"], len(full), len(tool_events))
return True
return FollowupResult.COMPLETED
async def _process_followup(rec):
outcome = await _run_followup(rec)
if outcome is FollowupResult.COMPLETED:
from src.agent_runtime.process_resources import job_from_record
bg_jobs.mark_followed_up(rec["id"], expected=job_from_record(rec))
elif outcome is FollowupResult.TERMINAL_UNFOLLOWABLE:
if not bg_jobs.mark_unfollowable(rec["id"], expected_record=rec):
return FollowupResult.RETRYABLE_LATER
logger.warning("bg-followup: job %s has no valid continuation linkage; retired from pending", rec.get("id"))
return outcome
async def _loop():
@@ -196,9 +228,7 @@ async def _loop():
try:
for rec in bg_jobs.pending_followups():
try:
if await _run_followup(rec):
from src.agent_runtime.process_resources import job_from_record
bg_jobs.mark_followed_up(rec["id"], expected=job_from_record(rec))
await _process_followup(rec)
except Exception as e:
# Idempotent: leave followed_up=False so the next tick retries.
logger.warning("bg-followup failed for %s (will retry): %s", rec.get("id"), e)