Files
odysseus/src/agent_runtime/completion.py
T
Alexandre Teixeira d49071bbec fix: close Wave 1.1 completion-gate audit findings
- Headless consumers (task scheduler, background follow-up) now treat a
  completion-gate final_response as the authoritative answer instead of
  collecting deltas only. A gated replacement no longer leaves scheduled
  output empty, which used to trigger an extra, ungated grace-summary
  model call.
- The scheduler closes the agent stream with contextlib.aclosing, so the
  approval-pause break unwinds the gate's journal and teacher-takeover
  context in its own task. Chained runs no longer inherit a stale
  parent_run_id, and later finalization no longer raises ContextVar
  reset errors.
- On provider error, the completion gate applies the live answer's
  statement filter to persisted round_texts. Diagnostics and the failure
  note survive; claims rejected by the gate cannot reappear on reload.
2026-10-01 14:49:32 +01:00

367 lines
21 KiB
Python

"""One presentation gate between agent execution and externally visible prose.
Tool, progress and interaction events stay live. Answer deltas are held until
the generator unwinds so a later replacement cannot conceal an earlier false
claim. This consumes no provider calls. Cancellation closes the inner generator
under the same journal/turn authority; it never emits a successful terminal event.
"""
from __future__ import annotations
from contextlib import aclosing
from dataclasses import replace
from functools import wraps
from inspect import signature
import json
import re
from time import perf_counter
from src.agent_evidence import (
CompletionDecision, CompletionStatus, EvidenceKind, EvidenceLedger,
requirements_from_runtime_context, _execution_obligation, _unquoted_statements,
_ARTIFACT_PATH,
)
from .journal import ActionJournal, bind_journal, current_journal
_TEST_CLAIM = re.compile(
r'\b(?:(?:all\s+)?(?:tests?|checks?|verification|suite)\s+(?:have\s+|has\s+|now\s+|are\s+|is\s+)*(?:passed|passing|successful|green)|'
r'(?:passed|passing)\s+(?:all\s+)?(?:the\s+)?tests?|\d+\s+passed)\b', re.I)
_TEST_STATUS_CLAIM = re.compile(
r'\b(?:tests?|pytest|unittest|test suite|checks?|verification)\s*[:—-]?\s*'
r'(?:all\s+|have\s+|has\s+|now\s+|are\s+|is\s+|ran\s+)*'
r'(?:pass(?:ed|ing)?|succeeded|successful(?:ly)?|green)\b|'
r'\b(?:zero|no|0)\s+(?:test\s+)?failures\b', re.I)
_EXECUTION_CLAIM = re.compile(
r'\b(?:(?:I|we|I\'ve|we\'ve|and)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed)|'
rf'(?:file|artifact|command|script|service|server|{_ARTIFACT_PATH})\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started)|'
r'(?:successfully\s+)(?:ran|executed|created|updated|saved|completed))\b', re.I)
_UNATTESTED_TEST_METRIC = re.compile(
r'\b\d+\s+(?:(?:unit|integration)\s+)?tests?\s+pass(?:ed|ing)?\b|'
r'\b\d+\s+passed\b|\b\d+(?:\.\d+)?%\s+(?:test\s+)?coverage\b', re.I)
_UNBOUNDED_SUCCESS = re.compile(
r'\b(?:everything|all\s+(?:bugs|issues))\s+(?:is\s+|are\s+|has\s+been\s+)?'
r'(?:fixed|resolved|working)\b', re.I)
_MUTATION_CLAIM = re.compile(
r'\b(?:created|updated|modified|wrote|written|saved|fixed)\b', re.I)
_TEST_IDENTITY = re.compile(r'\b(?:pytest|unittest)\b', re.I)
_TEST_SUBJECT = re.compile(r'\b(?:tests?|test suite|pytest|unittest|checks?|verification)\b', re.I)
_CLAIM_PATH = re.compile(_ARTIFACT_PATH)
_BARE_SUCCESS = re.compile(r'^\s*(?:done|completed|success|all done|all set|fixed)[.!]?\s*$', re.I)
_NON_REPORT_SCOPE = re.compile(
r'^\s*(?:if|unless|suppose|imagine|hypothetically|for\s+(?:example|instance))\b|'
r'\b(?:if|when|whenever|unless|until)\b|'
r'\b(?:can|could|may|might|should|would|will|must)\b|'
r'\b(?:says?|said|states?|stated|example)\b', re.I)
def _current_run_claims(statement: str, *, execution_required: bool) -> list[tuple[str, str]]:
"""Classify asserted execution, separately from the turn's obligation.
Past actions and current result/status predicates are reports. Conditional,
modal, attributed and example clauses are scoped prose. Bare terminal
success only carries execution meaning under an execution contract.
"""
if _BARE_SUCCESS.fullmatch(statement):
return [('terminal', statement)] if execution_required else []
actions = list(_EXECUTION_CLAIM.finditer(statement))
leading = re.match(r'^\s*(?:successfully\s+)?(?:created|updated|modified|wrote|saved)\b', statement, re.I)
if leading:
actions.insert(0, leading)
candidates = [('action', match) for match in actions]
for kind, pattern in [('metric', _UNATTESTED_TEST_METRIC), ('metric', _UNBOUNDED_SUCCESS),
('test', _TEST_CLAIM), ('test', _TEST_STATUS_CLAIM)]:
candidates.extend((kind, match) for match in pattern.finditer(statement))
claims = []
for kind, match in candidates:
# Scope markers after an asserted action do not make that action
# hypothetical ("I ran pytest to see if ..."). An immediate conditional
# continuation does qualify a result ("Tests passed if ...").
if _NON_REPORT_SCOPE.search(statement[:match.start()]) or re.match(
r'\s+(?:if|when|whenever|unless|until)\b', statement[match.end():], re.I):
continue
end = next((action.start() for action in actions if action.start() > match.start()), len(statement))
scope = statement[match.start():end]
if kind == 'action':
if _MUTATION_CLAIM.search(match.group()):
kind = 'mutation'
elif _TEST_SUBJECT.search(scope):
kind = 'test'
else:
kind = 'execution'
claims.append((kind, scope))
return claims
def _supported_prose(text: str, ledger: EvidenceLedger, decision: CompletionDecision) -> tuple[str, str]:
"""Remove unsupported assertions at statement boundaries; add no notice."""
incomplete = decision.reason if not decision.can_complete and decision.status != CompletionStatus.AWAITING_USER else ''
execution_required = _execution_obligation(ledger.requirements)
kept = []
removed = ''
for statement, scoped in _unquoted_statements(text):
why = ''
for claim, scope in _current_run_claims(scoped, execution_required=execution_required):
paths = tuple(match.group().rstrip('.') for match in _CLAIM_PATH.finditer(scope))
if claim == 'metric':
why = 'test counts, coverage or exhaustive correctness were not established by execution evidence'
elif claim == 'test':
identities = tuple(match.group().lower() for match in _TEST_IDENTITY.finditer(scope))
if decision.status != CompletionStatus.VERIFIED or not ledger._supports_verifier_claim(identities, paths):
why = 'no current passing executable verification supports the claim'
elif claim == 'mutation':
if not ledger._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, paths):
why = 'no matching artifact mutation supports the execution claim'
elif claim == 'execution':
# A generic assertion cannot be tied confidently to a receipt.
why = 'no matching operation supports the execution claim'
elif claim == 'terminal' and decision.status not in {CompletionStatus.SATISFIED, CompletionStatus.VERIFIED}:
why = incomplete or 'no successful execution supports completion'
if why:
break
if why:
removed = removed or why
else:
kept.append(statement)
prose = ''.join(kept).strip() if removed else text
return prose, removed
def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDecision) -> tuple[str, str]:
"""Keep explanatory prose; remove unsupported assertions and attach facts.
Exit status proves neither test counts nor coverage. A bad assertion is
removed at statement boundaries instead of erasing an entire explanation.
The execution outcome remains separate from a discarded model assertion.
"""
incomplete = decision.reason if not decision.can_complete and decision.status != CompletionStatus.AWAITING_USER else ''
execution_required = _execution_obligation(ledger.requirements)
prose, removed = _supported_prose(text, ledger, decision)
if incomplete or (removed and execution_required and decision.status in {CompletionStatus.UNVERIFIED, CompletionStatus.AWAITING_USER}):
reason = incomplete or removed
missing = (' Missing artifacts: ' + ', '.join(decision.missing_artifacts) + '.'
if decision.missing_artifacts else '')
notice = 'The task is incomplete: ' + reason.rstrip('.') + '.' + missing
recorded = [path for path in ledger.requirements.required_artifacts
if ledger._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, (path,))]
if removed and recorded:
notice += ' Recorded artifact mutation: ' + ', '.join(recorded) + '.'
return notice + ('\n\n' + prose if prose.strip() else ''), reason
if removed and not execution_required and decision.status != CompletionStatus.VERIFIED:
notice = 'Unsupported execution claims were omitted: ' + removed.rstrip('.') + '.'
return (prose.rstrip() + '\n\n' + notice) if prose.strip() else notice, removed
if decision.can_complete and (ledger.requirements.required_artifacts or ledger.requirements.verifier_required or removed):
facts = []
if ledger.requirements.required_artifacts:
facts.append('Output available: ' + ', '.join(ledger.requirements.required_artifacts) + '.')
if decision.status == CompletionStatus.VERIFIED:
facts.append('The latest executable verification passed.')
elif any(e.kind == EvidenceKind.ARTIFACT_VALIDATION and e.authoritative and e.success for e in ledger.events):
facts.append('Artifact readback verified. No passing executable test result was recorded.')
else:
facts.append('No passing executable test result was recorded.')
summary = ' '.join(facts)
return (prose.rstrip() + '\n\n' + summary) if prose.strip() else summary, removed
return prose, removed
def _event(data: dict) -> str:
return 'data: ' + json.dumps(data) + '\n\n'
def with_completion_gate(func):
call_signature = signature(func)
@wraps(func)
async def wrapped(*args, **kwargs):
started = perf_counter()
first_answer_at = None
arguments = call_signature.bind(*args, **kwargs)
arguments.apply_defaults()
bound = arguments.arguments
messages = bound.get('messages') or []
instruction = next((m.get('content', '') for m in reversed(messages)
if m.get('role') == 'user' and isinstance(m.get('content'), str)), '')
context = bound.get('client_runtime_context') or {}
requirements = requirements_from_runtime_context(context, instruction=instruction)
from src.tool_execution import vet_workspace
# A completion declaration is not a filesystem permission. Only the
# explicit, vetted runtime workspace may be read for artifact versions.
trusted_workspace = vet_workspace(bound.get('workspace')) if bound.get('workspace') else ''
requirements = replace(requirements, workspace_root=trusted_workspace or '')
parent = current_journal()
journal = ActionJournal(
workspace=requirements.workspace_root, observed_artifacts=requirements.required_artifacts,
parent_run_id=bound.get('_parent_run_id') or (parent.run_id if parent is not None else None))
answer_events: list[dict] = []
metrics_events: list[dict] = []
answer = ''
has_final = False
done = False
awaiting = False
exhausted = False
provider_error: str | None = None
with bind_journal(journal):
async with aclosing(func(*args, **kwargs)) as stream:
async for chunk in stream:
if chunk.strip() == 'data: [DONE]':
done = True
continue
try:
data = json.loads(chunk[6:]) if chunk.startswith('data: ') else None
except (ValueError, TypeError):
data = None
if not isinstance(data, dict):
if chunk.startswith('event: error'):
# The inner stream may still emit failed-terminal
# diagnostics. Hold the original error until those
# and the buffered answer have been released.
provider_error = provider_error or chunk
continue
yield chunk
continue
kind = data.get('type')
if kind == 'completion_decision':
existing = data.get('data') or {}
awaiting |= existing.get('status') == 'awaiting_user'
exhausted |= existing.get('status') == 'exhausted'
continue
if kind in {'metrics', 'agent_terminal'}:
metrics_events.append(data)
declared = (data.get('data') or {}).get('completion_requirements')
awaiting |= bool((data.get('data') or {}).get('missing_workspace'))
if isinstance(declared, dict):
requirements = requirements_from_runtime_context({'completion_requirements': declared})
requirements = replace(requirements, workspace_root=trusted_workspace or '')
# New obligations affect future receipts only. Never
# backfill historical versions with present bytes.
journal.observed_artifacts = tuple(dict.fromkeys(
(*journal.observed_artifacts, *requirements.required_artifacts)))
continue
if kind == 'ask_user':
awaiting = True
payload = data.get('data') or {}
if isinstance(payload.get('question'), str):
current = EvidenceLedger.from_tool_events(journal.evidence_events(), requirements)
question, why = completion_answer(payload['question'], current, current.evaluate(awaiting_user=True))
if why:
data = {**data, 'data': {**payload, 'question': question}}
chunk = _event(data)
if kind == 'final_response':
if first_answer_at is None:
first_answer_at = perf_counter()
answer = str(data.get('content') or '')
has_final = True
answer_events.append(data)
continue
if 'delta' in data or isinstance(data.get('thinking'), str):
if first_answer_at is None:
first_answer_at = perf_counter()
# Boolean thinking=True marks a reasoning-only delta;
# a textual thinking companion must not hide an answer
# delta. Both shapes remain buffered until the gate.
if isinstance(data.get('thinking'), str):
answer_events.append({'delta': data['thinking'], 'thinking': True})
data = {key: value for key, value in data.items() if key != 'thinking'}
if 'delta' not in data:
continue
if data.get('thinking') is not True and 'delta' in data:
if has_final:
answer = ''
has_final = False
answer += str(data.get('delta') or '')
answer_events.append(data)
continue
yield chunk
if provider_error and not answer_events and not metrics_events:
yield provider_error
return
presentation_replaced = False
if not provider_error and not has_final and requirements.required_artifacts:
terminal_texts = next((event.get('data', {}).get('round_texts')
for event in reversed(metrics_events)
if isinstance(event.get('data', {}).get('round_texts'), list)
and all(isinstance(text, str) for text in event['data']['round_texts'])), None)
if terminal_texts is not None:
terminal_answer = '\n\n'.join(text for text in terminal_texts if text.strip())
if terminal_answer != answer:
# The loop can retract a rejected round while retaining
# its live deltas. Do not resurrect those buffered drafts
# after recovery. Terminal prose still passes this gate.
presentation_replaced = True
answer = terminal_answer
answer_events = [event for event in answer_events if event.get('thinking') is True]
ledger = EvidenceLedger.from_tool_events(journal.evidence_events(), requirements)
decision = ledger.evaluate(exhausted=exhausted, awaiting_user=awaiting)
if provider_error:
decision = replace(decision, status=CompletionStatus.FAILED,
can_complete=False, reason='Model request failed')
# Exhaustion limits execution; factual source synthesis can remain
# useful and must not be replaced merely because the budget ended.
presentation_decision = ledger.evaluate(awaiting_user=awaiting) if exhausted and not provider_error else decision
safe_answer, reason = completion_answer(answer, ledger, presentation_decision)
# Evaluate each earlier draft as well as the final replacement.
# Never replay an unsupported intermediate success claim.
draft = ''.join(str(e.get('delta') or e.get('content') or '')
+ (e['thinking'] if isinstance(e.get('thinking'), str) else '')
for e in answer_events)
_, unsafe_draft = completion_answer(draft, ledger, presentation_decision)
if not answer.strip() and unsafe_draft:
reason = reason or unsafe_draft
safe_answer, _ = completion_answer(draft, ledger, presentation_decision)
if reason and _execution_obligation(requirements) and decision.can_complete and decision.status == CompletionStatus.UNVERIFIED:
decision = CompletionDecision(CompletionStatus.UNVERIFIED, False, reason,
decision.evidence_ids, decision.missing_artifacts)
released_at = perf_counter()
if not provider_error:
yield _event({'type': 'completion_decision', 'data': decision.to_dict()})
replaced_answer = bool(presentation_replaced or reason or unsafe_draft or safe_answer != answer)
if replaced_answer:
reasoning = [event for event in answer_events if event.get('thinking') is True]
_, unsafe_reasoning = completion_answer(
''.join(str(event.get('delta') or '') for event in reasoning), ledger,
replace(presentation_decision, can_complete=True))
if not unsafe_reasoning:
for event in reasoning:
yield _event(event)
yield _event({'type': 'final_response', 'content': safe_answer})
else:
for event in answer_events:
yield _event(event)
if provider_error:
yield _event({'type': 'completion_decision', 'data': decision.to_dict()})
for event in metrics_events:
metadata = event.setdefault('data', {})
metadata.update(completion_decision=decision.to_dict(), evidence_events=ledger.to_list(),
action_receipts=journal.to_list(), completion_requirements=requirements.to_dict(),
run_id=journal.run_id, parent_run_id=journal.parent_run_id)
metadata['completion_gate'] = {
'buffer_seconds': released_at - first_answer_at if first_answer_at is not None else 0,
'first_visible_answer_seconds': released_at - started,
'additional_provider_calls': 0,
'answer_replaced': replaced_answer,
}
if replaced_answer:
if not provider_error:
metadata['round_texts'] = [safe_answer]
metadata['completion_gate_reason'] = reason or unsafe_draft or 'receipt_summary'
if provider_error and isinstance(metadata.get('round_texts'), list):
# Failed rounds stay as per-round diagnostics, but they are
# rendered again on reload. Apply the same statement filter
# as the live answer so a rejected claim cannot reappear.
metadata['round_texts'] = [
_supported_prose(text, ledger, presentation_decision)[0] if isinstance(text, str) else text
for text in metadata['round_texts']]
if isinstance(metadata.get('thinking'), str):
_, unsafe_thinking = completion_answer(metadata['thinking'], ledger,
replace(presentation_decision, can_complete=True))
if unsafe_thinking:
metadata.pop('thinking')
yield _event(event)
if provider_error:
yield provider_error
return
if done:
yield 'data: [DONE]\n\n'
return wrapped