mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 23:12:22 +02:00
- Headless consumers (task scheduler, background follow-up) now treat a completion-gate final_response as the authoritative answer instead of collecting deltas only. A gated replacement no longer leaves scheduled output empty, which used to trigger an extra, ungated grace-summary model call. - The scheduler closes the agent stream with contextlib.aclosing, so the approval-pause break unwinds the gate's journal and teacher-takeover context in its own task. Chained runs no longer inherit a stale parent_run_id, and later finalization no longer raises ContextVar reset errors. - On provider error, the completion gate applies the live answer's statement filter to persisted round_texts. Diagnostics and the failure note survive; claims rejected by the gate cannot reappear on reload.
367 lines
21 KiB
Python
367 lines
21 KiB
Python
"""One presentation gate between agent execution and externally visible prose.
|
|
|
|
Tool, progress and interaction events stay live. Answer deltas are held until
|
|
the generator unwinds so a later replacement cannot conceal an earlier false
|
|
claim. This consumes no provider calls. Cancellation closes the inner generator
|
|
under the same journal/turn authority; it never emits a successful terminal event.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from contextlib import aclosing
|
|
from dataclasses import replace
|
|
from functools import wraps
|
|
from inspect import signature
|
|
import json
|
|
import re
|
|
from time import perf_counter
|
|
|
|
from src.agent_evidence import (
|
|
CompletionDecision, CompletionStatus, EvidenceKind, EvidenceLedger,
|
|
requirements_from_runtime_context, _execution_obligation, _unquoted_statements,
|
|
_ARTIFACT_PATH,
|
|
)
|
|
from .journal import ActionJournal, bind_journal, current_journal
|
|
|
|
|
|
_TEST_CLAIM = re.compile(
|
|
r'\b(?:(?:all\s+)?(?:tests?|checks?|verification|suite)\s+(?:have\s+|has\s+|now\s+|are\s+|is\s+)*(?:passed|passing|successful|green)|'
|
|
r'(?:passed|passing)\s+(?:all\s+)?(?:the\s+)?tests?|\d+\s+passed)\b', re.I)
|
|
_TEST_STATUS_CLAIM = re.compile(
|
|
r'\b(?:tests?|pytest|unittest|test suite|checks?|verification)\s*[:—-]?\s*'
|
|
r'(?:all\s+|have\s+|has\s+|now\s+|are\s+|is\s+|ran\s+)*'
|
|
r'(?:pass(?:ed|ing)?|succeeded|successful(?:ly)?|green)\b|'
|
|
r'\b(?:zero|no|0)\s+(?:test\s+)?failures\b', re.I)
|
|
_EXECUTION_CLAIM = re.compile(
|
|
r'\b(?:(?:I|we|I\'ve|we\'ve|and)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed)|'
|
|
rf'(?:file|artifact|command|script|service|server|{_ARTIFACT_PATH})\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started)|'
|
|
r'(?:successfully\s+)(?:ran|executed|created|updated|saved|completed))\b', re.I)
|
|
_UNATTESTED_TEST_METRIC = re.compile(
|
|
r'\b\d+\s+(?:(?:unit|integration)\s+)?tests?\s+pass(?:ed|ing)?\b|'
|
|
r'\b\d+\s+passed\b|\b\d+(?:\.\d+)?%\s+(?:test\s+)?coverage\b', re.I)
|
|
_UNBOUNDED_SUCCESS = re.compile(
|
|
r'\b(?:everything|all\s+(?:bugs|issues))\s+(?:is\s+|are\s+|has\s+been\s+)?'
|
|
r'(?:fixed|resolved|working)\b', re.I)
|
|
_MUTATION_CLAIM = re.compile(
|
|
r'\b(?:created|updated|modified|wrote|written|saved|fixed)\b', re.I)
|
|
_TEST_IDENTITY = re.compile(r'\b(?:pytest|unittest)\b', re.I)
|
|
_TEST_SUBJECT = re.compile(r'\b(?:tests?|test suite|pytest|unittest|checks?|verification)\b', re.I)
|
|
_CLAIM_PATH = re.compile(_ARTIFACT_PATH)
|
|
_BARE_SUCCESS = re.compile(r'^\s*(?:done|completed|success|all done|all set|fixed)[.!]?\s*$', re.I)
|
|
_NON_REPORT_SCOPE = re.compile(
|
|
r'^\s*(?:if|unless|suppose|imagine|hypothetically|for\s+(?:example|instance))\b|'
|
|
r'\b(?:if|when|whenever|unless|until)\b|'
|
|
r'\b(?:can|could|may|might|should|would|will|must)\b|'
|
|
r'\b(?:says?|said|states?|stated|example)\b', re.I)
|
|
|
|
|
|
def _current_run_claims(statement: str, *, execution_required: bool) -> list[tuple[str, str]]:
|
|
"""Classify asserted execution, separately from the turn's obligation.
|
|
|
|
Past actions and current result/status predicates are reports. Conditional,
|
|
modal, attributed and example clauses are scoped prose. Bare terminal
|
|
success only carries execution meaning under an execution contract.
|
|
"""
|
|
if _BARE_SUCCESS.fullmatch(statement):
|
|
return [('terminal', statement)] if execution_required else []
|
|
actions = list(_EXECUTION_CLAIM.finditer(statement))
|
|
leading = re.match(r'^\s*(?:successfully\s+)?(?:created|updated|modified|wrote|saved)\b', statement, re.I)
|
|
if leading:
|
|
actions.insert(0, leading)
|
|
candidates = [('action', match) for match in actions]
|
|
for kind, pattern in [('metric', _UNATTESTED_TEST_METRIC), ('metric', _UNBOUNDED_SUCCESS),
|
|
('test', _TEST_CLAIM), ('test', _TEST_STATUS_CLAIM)]:
|
|
candidates.extend((kind, match) for match in pattern.finditer(statement))
|
|
claims = []
|
|
for kind, match in candidates:
|
|
# Scope markers after an asserted action do not make that action
|
|
# hypothetical ("I ran pytest to see if ..."). An immediate conditional
|
|
# continuation does qualify a result ("Tests passed if ...").
|
|
if _NON_REPORT_SCOPE.search(statement[:match.start()]) or re.match(
|
|
r'\s+(?:if|when|whenever|unless|until)\b', statement[match.end():], re.I):
|
|
continue
|
|
end = next((action.start() for action in actions if action.start() > match.start()), len(statement))
|
|
scope = statement[match.start():end]
|
|
if kind == 'action':
|
|
if _MUTATION_CLAIM.search(match.group()):
|
|
kind = 'mutation'
|
|
elif _TEST_SUBJECT.search(scope):
|
|
kind = 'test'
|
|
else:
|
|
kind = 'execution'
|
|
claims.append((kind, scope))
|
|
return claims
|
|
|
|
|
|
def _supported_prose(text: str, ledger: EvidenceLedger, decision: CompletionDecision) -> tuple[str, str]:
|
|
"""Remove unsupported assertions at statement boundaries; add no notice."""
|
|
incomplete = decision.reason if not decision.can_complete and decision.status != CompletionStatus.AWAITING_USER else ''
|
|
execution_required = _execution_obligation(ledger.requirements)
|
|
kept = []
|
|
removed = ''
|
|
for statement, scoped in _unquoted_statements(text):
|
|
why = ''
|
|
for claim, scope in _current_run_claims(scoped, execution_required=execution_required):
|
|
paths = tuple(match.group().rstrip('.') for match in _CLAIM_PATH.finditer(scope))
|
|
if claim == 'metric':
|
|
why = 'test counts, coverage or exhaustive correctness were not established by execution evidence'
|
|
elif claim == 'test':
|
|
identities = tuple(match.group().lower() for match in _TEST_IDENTITY.finditer(scope))
|
|
if decision.status != CompletionStatus.VERIFIED or not ledger._supports_verifier_claim(identities, paths):
|
|
why = 'no current passing executable verification supports the claim'
|
|
elif claim == 'mutation':
|
|
if not ledger._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, paths):
|
|
why = 'no matching artifact mutation supports the execution claim'
|
|
elif claim == 'execution':
|
|
# A generic assertion cannot be tied confidently to a receipt.
|
|
why = 'no matching operation supports the execution claim'
|
|
elif claim == 'terminal' and decision.status not in {CompletionStatus.SATISFIED, CompletionStatus.VERIFIED}:
|
|
why = incomplete or 'no successful execution supports completion'
|
|
if why:
|
|
break
|
|
if why:
|
|
removed = removed or why
|
|
else:
|
|
kept.append(statement)
|
|
prose = ''.join(kept).strip() if removed else text
|
|
return prose, removed
|
|
|
|
|
|
def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDecision) -> tuple[str, str]:
|
|
"""Keep explanatory prose; remove unsupported assertions and attach facts.
|
|
|
|
Exit status proves neither test counts nor coverage. A bad assertion is
|
|
removed at statement boundaries instead of erasing an entire explanation.
|
|
The execution outcome remains separate from a discarded model assertion.
|
|
"""
|
|
incomplete = decision.reason if not decision.can_complete and decision.status != CompletionStatus.AWAITING_USER else ''
|
|
execution_required = _execution_obligation(ledger.requirements)
|
|
prose, removed = _supported_prose(text, ledger, decision)
|
|
if incomplete or (removed and execution_required and decision.status in {CompletionStatus.UNVERIFIED, CompletionStatus.AWAITING_USER}):
|
|
reason = incomplete or removed
|
|
missing = (' Missing artifacts: ' + ', '.join(decision.missing_artifacts) + '.'
|
|
if decision.missing_artifacts else '')
|
|
notice = 'The task is incomplete: ' + reason.rstrip('.') + '.' + missing
|
|
recorded = [path for path in ledger.requirements.required_artifacts
|
|
if ledger._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, (path,))]
|
|
if removed and recorded:
|
|
notice += ' Recorded artifact mutation: ' + ', '.join(recorded) + '.'
|
|
return notice + ('\n\n' + prose if prose.strip() else ''), reason
|
|
if removed and not execution_required and decision.status != CompletionStatus.VERIFIED:
|
|
notice = 'Unsupported execution claims were omitted: ' + removed.rstrip('.') + '.'
|
|
return (prose.rstrip() + '\n\n' + notice) if prose.strip() else notice, removed
|
|
if decision.can_complete and (ledger.requirements.required_artifacts or ledger.requirements.verifier_required or removed):
|
|
facts = []
|
|
if ledger.requirements.required_artifacts:
|
|
facts.append('Output available: ' + ', '.join(ledger.requirements.required_artifacts) + '.')
|
|
if decision.status == CompletionStatus.VERIFIED:
|
|
facts.append('The latest executable verification passed.')
|
|
elif any(e.kind == EvidenceKind.ARTIFACT_VALIDATION and e.authoritative and e.success for e in ledger.events):
|
|
facts.append('Artifact readback verified. No passing executable test result was recorded.')
|
|
else:
|
|
facts.append('No passing executable test result was recorded.')
|
|
summary = ' '.join(facts)
|
|
return (prose.rstrip() + '\n\n' + summary) if prose.strip() else summary, removed
|
|
return prose, removed
|
|
|
|
|
|
def _event(data: dict) -> str:
|
|
return 'data: ' + json.dumps(data) + '\n\n'
|
|
|
|
|
|
def with_completion_gate(func):
|
|
call_signature = signature(func)
|
|
|
|
@wraps(func)
|
|
async def wrapped(*args, **kwargs):
|
|
started = perf_counter()
|
|
first_answer_at = None
|
|
arguments = call_signature.bind(*args, **kwargs)
|
|
arguments.apply_defaults()
|
|
bound = arguments.arguments
|
|
messages = bound.get('messages') or []
|
|
instruction = next((m.get('content', '') for m in reversed(messages)
|
|
if m.get('role') == 'user' and isinstance(m.get('content'), str)), '')
|
|
context = bound.get('client_runtime_context') or {}
|
|
requirements = requirements_from_runtime_context(context, instruction=instruction)
|
|
from src.tool_execution import vet_workspace
|
|
# A completion declaration is not a filesystem permission. Only the
|
|
# explicit, vetted runtime workspace may be read for artifact versions.
|
|
trusted_workspace = vet_workspace(bound.get('workspace')) if bound.get('workspace') else ''
|
|
requirements = replace(requirements, workspace_root=trusted_workspace or '')
|
|
parent = current_journal()
|
|
journal = ActionJournal(
|
|
workspace=requirements.workspace_root, observed_artifacts=requirements.required_artifacts,
|
|
parent_run_id=bound.get('_parent_run_id') or (parent.run_id if parent is not None else None))
|
|
answer_events: list[dict] = []
|
|
metrics_events: list[dict] = []
|
|
answer = ''
|
|
has_final = False
|
|
done = False
|
|
awaiting = False
|
|
exhausted = False
|
|
provider_error: str | None = None
|
|
with bind_journal(journal):
|
|
async with aclosing(func(*args, **kwargs)) as stream:
|
|
async for chunk in stream:
|
|
if chunk.strip() == 'data: [DONE]':
|
|
done = True
|
|
continue
|
|
try:
|
|
data = json.loads(chunk[6:]) if chunk.startswith('data: ') else None
|
|
except (ValueError, TypeError):
|
|
data = None
|
|
if not isinstance(data, dict):
|
|
if chunk.startswith('event: error'):
|
|
# The inner stream may still emit failed-terminal
|
|
# diagnostics. Hold the original error until those
|
|
# and the buffered answer have been released.
|
|
provider_error = provider_error or chunk
|
|
continue
|
|
yield chunk
|
|
continue
|
|
kind = data.get('type')
|
|
if kind == 'completion_decision':
|
|
existing = data.get('data') or {}
|
|
awaiting |= existing.get('status') == 'awaiting_user'
|
|
exhausted |= existing.get('status') == 'exhausted'
|
|
continue
|
|
if kind in {'metrics', 'agent_terminal'}:
|
|
metrics_events.append(data)
|
|
declared = (data.get('data') or {}).get('completion_requirements')
|
|
awaiting |= bool((data.get('data') or {}).get('missing_workspace'))
|
|
if isinstance(declared, dict):
|
|
requirements = requirements_from_runtime_context({'completion_requirements': declared})
|
|
requirements = replace(requirements, workspace_root=trusted_workspace or '')
|
|
# New obligations affect future receipts only. Never
|
|
# backfill historical versions with present bytes.
|
|
journal.observed_artifacts = tuple(dict.fromkeys(
|
|
(*journal.observed_artifacts, *requirements.required_artifacts)))
|
|
continue
|
|
if kind == 'ask_user':
|
|
awaiting = True
|
|
payload = data.get('data') or {}
|
|
if isinstance(payload.get('question'), str):
|
|
current = EvidenceLedger.from_tool_events(journal.evidence_events(), requirements)
|
|
question, why = completion_answer(payload['question'], current, current.evaluate(awaiting_user=True))
|
|
if why:
|
|
data = {**data, 'data': {**payload, 'question': question}}
|
|
chunk = _event(data)
|
|
if kind == 'final_response':
|
|
if first_answer_at is None:
|
|
first_answer_at = perf_counter()
|
|
answer = str(data.get('content') or '')
|
|
has_final = True
|
|
answer_events.append(data)
|
|
continue
|
|
if 'delta' in data or isinstance(data.get('thinking'), str):
|
|
if first_answer_at is None:
|
|
first_answer_at = perf_counter()
|
|
# Boolean thinking=True marks a reasoning-only delta;
|
|
# a textual thinking companion must not hide an answer
|
|
# delta. Both shapes remain buffered until the gate.
|
|
if isinstance(data.get('thinking'), str):
|
|
answer_events.append({'delta': data['thinking'], 'thinking': True})
|
|
data = {key: value for key, value in data.items() if key != 'thinking'}
|
|
if 'delta' not in data:
|
|
continue
|
|
if data.get('thinking') is not True and 'delta' in data:
|
|
if has_final:
|
|
answer = ''
|
|
has_final = False
|
|
answer += str(data.get('delta') or '')
|
|
answer_events.append(data)
|
|
continue
|
|
yield chunk
|
|
if provider_error and not answer_events and not metrics_events:
|
|
yield provider_error
|
|
return
|
|
presentation_replaced = False
|
|
if not provider_error and not has_final and requirements.required_artifacts:
|
|
terminal_texts = next((event.get('data', {}).get('round_texts')
|
|
for event in reversed(metrics_events)
|
|
if isinstance(event.get('data', {}).get('round_texts'), list)
|
|
and all(isinstance(text, str) for text in event['data']['round_texts'])), None)
|
|
if terminal_texts is not None:
|
|
terminal_answer = '\n\n'.join(text for text in terminal_texts if text.strip())
|
|
if terminal_answer != answer:
|
|
# The loop can retract a rejected round while retaining
|
|
# its live deltas. Do not resurrect those buffered drafts
|
|
# after recovery. Terminal prose still passes this gate.
|
|
presentation_replaced = True
|
|
answer = terminal_answer
|
|
answer_events = [event for event in answer_events if event.get('thinking') is True]
|
|
ledger = EvidenceLedger.from_tool_events(journal.evidence_events(), requirements)
|
|
decision = ledger.evaluate(exhausted=exhausted, awaiting_user=awaiting)
|
|
if provider_error:
|
|
decision = replace(decision, status=CompletionStatus.FAILED,
|
|
can_complete=False, reason='Model request failed')
|
|
# Exhaustion limits execution; factual source synthesis can remain
|
|
# useful and must not be replaced merely because the budget ended.
|
|
presentation_decision = ledger.evaluate(awaiting_user=awaiting) if exhausted and not provider_error else decision
|
|
safe_answer, reason = completion_answer(answer, ledger, presentation_decision)
|
|
# Evaluate each earlier draft as well as the final replacement.
|
|
# Never replay an unsupported intermediate success claim.
|
|
draft = ''.join(str(e.get('delta') or e.get('content') or '')
|
|
+ (e['thinking'] if isinstance(e.get('thinking'), str) else '')
|
|
for e in answer_events)
|
|
_, unsafe_draft = completion_answer(draft, ledger, presentation_decision)
|
|
if not answer.strip() and unsafe_draft:
|
|
reason = reason or unsafe_draft
|
|
safe_answer, _ = completion_answer(draft, ledger, presentation_decision)
|
|
if reason and _execution_obligation(requirements) and decision.can_complete and decision.status == CompletionStatus.UNVERIFIED:
|
|
decision = CompletionDecision(CompletionStatus.UNVERIFIED, False, reason,
|
|
decision.evidence_ids, decision.missing_artifacts)
|
|
released_at = perf_counter()
|
|
if not provider_error:
|
|
yield _event({'type': 'completion_decision', 'data': decision.to_dict()})
|
|
replaced_answer = bool(presentation_replaced or reason or unsafe_draft or safe_answer != answer)
|
|
if replaced_answer:
|
|
reasoning = [event for event in answer_events if event.get('thinking') is True]
|
|
_, unsafe_reasoning = completion_answer(
|
|
''.join(str(event.get('delta') or '') for event in reasoning), ledger,
|
|
replace(presentation_decision, can_complete=True))
|
|
if not unsafe_reasoning:
|
|
for event in reasoning:
|
|
yield _event(event)
|
|
yield _event({'type': 'final_response', 'content': safe_answer})
|
|
else:
|
|
for event in answer_events:
|
|
yield _event(event)
|
|
if provider_error:
|
|
yield _event({'type': 'completion_decision', 'data': decision.to_dict()})
|
|
for event in metrics_events:
|
|
metadata = event.setdefault('data', {})
|
|
metadata.update(completion_decision=decision.to_dict(), evidence_events=ledger.to_list(),
|
|
action_receipts=journal.to_list(), completion_requirements=requirements.to_dict(),
|
|
run_id=journal.run_id, parent_run_id=journal.parent_run_id)
|
|
metadata['completion_gate'] = {
|
|
'buffer_seconds': released_at - first_answer_at if first_answer_at is not None else 0,
|
|
'first_visible_answer_seconds': released_at - started,
|
|
'additional_provider_calls': 0,
|
|
'answer_replaced': replaced_answer,
|
|
}
|
|
if replaced_answer:
|
|
if not provider_error:
|
|
metadata['round_texts'] = [safe_answer]
|
|
metadata['completion_gate_reason'] = reason or unsafe_draft or 'receipt_summary'
|
|
if provider_error and isinstance(metadata.get('round_texts'), list):
|
|
# Failed rounds stay as per-round diagnostics, but they are
|
|
# rendered again on reload. Apply the same statement filter
|
|
# as the live answer so a rejected claim cannot reappear.
|
|
metadata['round_texts'] = [
|
|
_supported_prose(text, ledger, presentation_decision)[0] if isinstance(text, str) else text
|
|
for text in metadata['round_texts']]
|
|
if isinstance(metadata.get('thinking'), str):
|
|
_, unsafe_thinking = completion_answer(metadata['thinking'], ledger,
|
|
replace(presentation_decision, can_complete=True))
|
|
if unsafe_thinking:
|
|
metadata.pop('thinking')
|
|
yield _event(event)
|
|
if provider_error:
|
|
yield provider_error
|
|
return
|
|
if done:
|
|
yield 'data: [DONE]\n\n'
|
|
|
|
return wrapped
|