feat(runtime): record execution evidence and gate completion

This commit is contained in:
Alexandre Teixeira
2026-09-26 13:48:12 +01:00
parent ea847e6c0b
commit eaa5668baa
16 changed files with 1105 additions and 154 deletions
+1
View File
@@ -0,0 +1 @@
"""Run-scoped contracts behind the public agent-loop compatibility facade."""
+207
View File
@@ -0,0 +1,207 @@
"""One presentation gate between agent execution and externally visible prose.
Tool, progress and interaction events stay live. Answer deltas are held until
the generator unwinds so a later replacement cannot conceal an earlier false
claim. This consumes no provider calls. Cancellation closes the inner generator
under the same journal/turn authority; it never emits a successful terminal event.
"""
from __future__ import annotations
from contextlib import aclosing
from dataclasses import replace
from functools import wraps
from inspect import signature
import json
import re
from time import perf_counter
from src.agent_evidence import (
CompletionDecision, CompletionStatus, EvidenceKind, EvidenceLedger,
requirements_from_runtime_context,
)
from .journal import ActionJournal, bind_journal, current_journal
_TEST_CLAIM = re.compile(
r'\b(?:(?:all\s+)?(?:tests?|checks?|verification|suite)\s+(?:have\s+|has\s+|now\s+|are\s+|is\s+)*(?:passed|passing|successful|green)|'
r'(?:passed|passing)\s+(?:all\s+)?(?:the\s+)?tests?|\d+\s+passed)\b', re.I)
_TEST_STATUS_CLAIM = re.compile(
r'\b(?:tests?|pytest|unittest|test suite|checks?|verification)\s*[:—-]?\s*'
r'(?:all\s+|have\s+|has\s+|now\s+|are\s+|is\s+|ran\s+)*'
r'(?:pass(?:ed|ing)?|succeeded|successful(?:ly)?|green)\b|'
r'\b(?:zero|no|0)\s+(?:test\s+)?failures\b', re.I)
_TERMINAL_SUCCESS = re.compile(r'^\s*(?:done|completed|success|all done|all set|fixed)\b', re.I)
_EXECUTION_CLAIM = re.compile(
r'\b(?:(?:I|we|I\'ve|we\'ve)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed)|'
r'(?:file|artifact|command|script|service|server)\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started)|'
r'(?:successfully\s+)(?:ran|executed|created|updated|saved|completed))\b', re.I)
def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDecision) -> tuple[str, str]:
"""Return the answer and a reason if unsupported execution claims were removed."""
if decision.status == CompletionStatus.AWAITING_USER:
# A question may still falsely assert that preceding work passed.
unsupported = ''
elif not decision.can_complete:
unsupported = decision.reason
else:
unsupported = ''
if (_TEST_CLAIM.search(text) or _TEST_STATUS_CLAIM.search(text)) and decision.status != CompletionStatus.VERIFIED:
unsupported = unsupported or 'no current passing executable verification supports the claim'
productive = [event for event in ledger.events
if event.authoritative and event.success
and event.tool not in {'update_plan', 'todowrite', 'ask_user'}]
if (_EXECUTION_CLAIM.search(text) or _TERMINAL_SUCCESS.search(text)) and not productive:
unsupported = unsupported or 'no successful operation supports the execution claim'
if not unsupported:
# For a declared execution contract, publish facts selected from the
# receipts rather than an unconstrained model claim (test counts,
# coverage and "everything fixed" cannot be inferred from exit status).
if decision.can_complete and (ledger.requirements.required_artifacts or ledger.requirements.verifier_required):
parts = []
if ledger.requirements.required_artifacts:
parts.append('Output available: ' + ', '.join(ledger.requirements.required_artifacts) + '.')
if decision.status == CompletionStatus.VERIFIED:
parts.append('The latest executable verification passed.')
elif any(e.kind == EvidenceKind.ARTIFACT_VALIDATION and e.authoritative and e.success for e in ledger.events):
parts.append('Artifact readback verified. No passing executable test result was recorded.')
else:
parts.append('No passing executable test result was recorded.')
return ' '.join(parts), ''
return text, ''
missing = (" Missing artifacts: " + ", ".join(decision.missing_artifacts) + "."
if decision.missing_artifacts else '')
return "The task is incomplete: " + unsupported.rstrip('.') + '.' + missing, unsupported
def _event(data: dict) -> str:
return 'data: ' + json.dumps(data) + '\n\n'
def with_completion_gate(func):
call_signature = signature(func)
@wraps(func)
async def wrapped(*args, **kwargs):
started = perf_counter()
first_answer_at = None
arguments = call_signature.bind(*args, **kwargs)
arguments.apply_defaults()
bound = arguments.arguments
messages = bound.get('messages') or []
instruction = next((m.get('content', '') for m in reversed(messages)
if m.get('role') == 'user' and isinstance(m.get('content'), str)), '')
context = bound.get('client_runtime_context') or {}
requirements = requirements_from_runtime_context(context, instruction=instruction)
from src.tool_execution import vet_workspace
# A completion declaration is not a filesystem permission. Only the
# explicit, vetted runtime workspace may be read for artifact versions.
trusted_workspace = vet_workspace(bound.get('workspace')) if bound.get('workspace') else ''
requirements = replace(requirements, workspace_root=trusted_workspace or '')
parent = current_journal()
journal = parent if parent is not None and parent.workspace == requirements.workspace_root else ActionJournal(
workspace=requirements.workspace_root, observed_artifacts=requirements.required_artifacts)
answer_events: list[dict] = []
metrics_events: list[dict] = []
answer = ''
has_final = False
done = False
awaiting = False
exhausted = False
provider_error = False
with bind_journal(journal):
async with aclosing(func(*args, **kwargs)) as stream:
async for chunk in stream:
if chunk.strip() == 'data: [DONE]':
done = True
continue
try:
data = json.loads(chunk[6:]) if chunk.startswith('data: ') else None
except (ValueError, TypeError):
data = None
if not isinstance(data, dict):
if chunk.startswith('event: error'):
provider_error = True
yield chunk
continue
kind = data.get('type')
if kind == 'completion_decision':
existing = data.get('data') or {}
awaiting |= existing.get('status') == 'awaiting_user'
exhausted |= existing.get('status') == 'exhausted'
continue
if kind in {'metrics', 'agent_terminal'}:
metrics_events.append(data)
declared = (data.get('data') or {}).get('completion_requirements')
awaiting |= bool((data.get('data') or {}).get('missing_workspace'))
if isinstance(declared, dict):
requirements = requirements_from_runtime_context({'completion_requirements': declared})
requirements = replace(requirements, workspace_root=trusted_workspace or '')
continue
if kind == 'ask_user':
awaiting = True
payload = data.get('data') or {}
if isinstance(payload.get('question'), str):
current = EvidenceLedger.from_tool_events(journal.evidence_events(), requirements)
question, why = completion_answer(payload['question'], current, current.evaluate(awaiting_user=True))
if why:
data = {**data, 'data': {**payload, 'question': question}}
chunk = _event(data)
if kind == 'final_response':
if first_answer_at is None:
first_answer_at = perf_counter()
answer = str(data.get('content') or '')
has_final = True
answer_events.append(data)
continue
if 'delta' in data and not data.get('thinking'):
if first_answer_at is None:
first_answer_at = perf_counter()
if has_final:
answer = ''
has_final = False
answer += str(data.get('delta') or '')
answer_events.append(data)
continue
yield chunk
if provider_error and not answer_events and not metrics_events:
return
ledger = EvidenceLedger.from_tool_events(journal.evidence_events(), requirements)
decision = ledger.evaluate(exhausted=exhausted, awaiting_user=awaiting)
# Exhaustion limits execution; factual source synthesis can remain
# useful and must not be replaced merely because the budget ended.
presentation_decision = ledger.evaluate(awaiting_user=awaiting) if exhausted else decision
safe_answer, reason = completion_answer(answer, ledger, presentation_decision)
if reason and decision.can_complete:
decision = CompletionDecision(CompletionStatus.UNVERIFIED, False, reason,
decision.evidence_ids, decision.missing_artifacts)
released_at = perf_counter()
yield _event({'type': 'completion_decision', 'data': decision.to_dict()})
# Evaluate each earlier draft as well as the final replacement.
# Never replay an unsupported intermediate success claim.
draft = ''.join(str(e.get('delta') or e.get('content') or '') for e in answer_events)
_, unsafe_draft = completion_answer(draft, ledger, presentation_decision)
replaced_answer = bool(reason or unsafe_draft or safe_answer != answer)
if replaced_answer:
yield _event({'type': 'final_response', 'content': safe_answer})
else:
for event in answer_events:
yield _event(event)
for event in metrics_events:
metadata = event.setdefault('data', {})
metadata.update(completion_decision=decision.to_dict(), evidence_events=ledger.to_list(),
action_receipts=journal.to_list(), completion_requirements=requirements.to_dict())
metadata['completion_gate'] = {
'buffer_seconds': released_at - first_answer_at if first_answer_at is not None else 0,
'first_visible_answer_seconds': released_at - started,
'additional_provider_calls': 0,
'answer_replaced': replaced_answer,
}
if replaced_answer:
metadata['round_texts'] = [safe_answer]
metadata['completion_gate_reason'] = reason or unsafe_draft or 'receipt_summary'
yield _event(event)
if done:
yield 'data: [DONE]\n\n'
return wrapped
+146
View File
@@ -0,0 +1,146 @@
"""Canonical evidence identities; these helpers never grant filesystem access."""
from __future__ import annotations
import hashlib
import json
import os
from pathlib import Path, PurePosixPath
import re
import shlex
import stat
from .path_policy import _is_sensitive_path
def digest(value: object) -> str:
return hashlib.sha256(json.dumps(value, sort_keys=True, ensure_ascii=False,
separators=(",", ":"), default=str).encode()).hexdigest()
def artifact_version(value: str, workspace: str) -> str:
"""Content version for a confined declared output; missing/unreadable is explicit."""
identity = artifact_identity(value, workspace)
if not workspace or not identity.startswith('workspace:'):
return 'unobserved'
root = Path(workspace).resolve()
candidate = (root / identity.removeprefix('workspace:')).resolve()
if not candidate.is_relative_to(root) or _is_sensitive_path(str(candidate)):
return 'unobserved'
if not hasattr(os, 'O_NOFOLLOW') or not os.supports_dir_fd:
return 'unobserved'
directory = None
try:
directory = os.open(root, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW)
parts = candidate.relative_to(root).parts
if not parts:
return 'unobserved'
for part in parts[:-1]:
child = os.open(part, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=directory)
os.close(directory)
directory = child
descriptor = os.open(parts[-1], os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK, dir_fd=directory)
with os.fdopen(descriptor, 'rb') as stream:
info = os.fstat(stream.fileno())
if not stat.S_ISREG(info.st_mode) or info.st_size > 64 * 1024 * 1024:
return 'unobserved'
result = hashlib.sha256()
remaining = 64 * 1024 * 1024
while block := stream.read(min(1024 * 1024, remaining + 1)):
remaining -= len(block)
if remaining < 0:
return 'unobserved'
result.update(block)
return result.hexdigest()
except (OSError, ValueError):
return 'missing-or-unreadable'
finally:
if directory is not None:
os.close(directory)
def artifact_identity(value: str, workspace: str = "") -> str:
"""Unify relative, virtual and host aliases without basename matching.
Resolving symlinks is evidence bookkeeping, never a confinement check. Paths
outside the workspace retain their absolute identity and cannot satisfy a
workspace obligation with the same basename.
"""
text = str(value or "").strip()
if not text:
return ""
path = PurePosixPath(text)
root = Path(workspace or "/workspace").resolve()
if path.parts[:2] == ('/', 'workspace'):
path = PurePosixPath(*path.parts[2:])
candidate = Path(str(path))
if not candidate.is_absolute():
candidate = root / candidate
try:
resolved = candidate.resolve()
relative = resolved.relative_to(root)
return "workspace:" + relative.as_posix()
except ValueError:
return "absolute:" + str(candidate.resolve())
except (OSError, RuntimeError):
return "unresolved:" + text
def executable_words(command: str) -> tuple[str, ...]:
"""Recognize one foreground command, optionally after safe cd/set prefixes.
This is deliberately conservative evidence parsing, not shell authorization.
Pipelines, control flow, substitutions and status-masking tails are not proof
that a verifier returned the recorded shell status.
"""
text = str(command or '').strip()
if any(marker in text for marker in ('`', '$(', '${', '\n', '\r')):
return ()
try:
lexer = shlex.shlex(text, posix=True, punctuation_chars=';&|<>()')
lexer.whitespace_split = True
words = list(lexer)
except ValueError:
return ()
while '&&' in words:
index = words.index('&&')
prefix = words[:index]
if not ((len(prefix) == 2 and prefix[0] == 'cd') or prefix == ['set', '-e']):
return ()
words = words[index + 1:]
if any(word and all(c in ';&|<>()' for c in word) for word in words):
return ()
while words and re.fullmatch(r'[A-Za-z_][A-Za-z0-9_]*=[^\n]*', words[0]):
words.pop(0)
return tuple(words)
def is_test_command(command: str) -> bool:
words = executable_words(command)
if not words:
return False
if any(word in {'--help', '-h', '--version', '--collect-only', '--co'} for word in words[1:]):
return False
binary = Path(words[0]).name
if binary in {'pytest', 'py.test'}:
return True
if re.fullmatch(r'python(?:\d+(?:\.\d+)?)?', binary):
args = list(words[1:])
while args and args[0] in {'-I', '-S', '-s', '-E', '-B', '-u'}:
args.pop(0)
return len(args) >= 2 and args[:2] in (['-m', 'pytest'], ['-m', 'unittest'])
if binary in {'npm', 'pnpm', 'yarn', 'make', 'cargo', 'go'}:
args = words[1:]
return bool(args and (args[0] == 'test' or binary == 'npm' and args[:2] == ('run', 'test')))
return bool(re.match(r'^/(?:tests?|verifier)/[^/]+', words[0]))
def is_validation_command(command: str) -> bool:
words = executable_words(command)
if not words:
return False
binary = Path(words[0]).name
return (is_test_command(command)
or binary in {'cat', 'head', 'tail', 'stat', 'wc', 'jq', 'cmp', 'diff',
'coqc', 'gcc', 'g++', 'clang', 'clang++', 'javac', 'rustc'}
or binary == 'test' and len(words) > 1 and words[1] in {'-e', '-f', '-s', '-d'}
or binary in {'cargo', 'go', 'npm', 'pnpm', 'yarn'} and words[1:2] in {('build',), ('check',)}
or binary == 'npm' and words[1:3] == ('run', 'build'))
+205
View File
@@ -0,0 +1,205 @@
"""Run-owned action history. Model text cannot insert authoritative receipts."""
from __future__ import annotations
from contextlib import contextmanager
from contextvars import ContextVar
from copy import deepcopy
from dataclasses import dataclass, field, asdict
from functools import wraps
from inspect import signature
from typing import Any
from uuid import uuid4
from .identity import artifact_identity, artifact_version, digest
@dataclass
class ActionReceipt:
action_id: str
call_id: str
proposed_tool: str
proposed_arguments: str
provider_arguments: Any = None
provider_tool: str = ''
tool: str = ""
arguments: str = ""
transitions: list[dict[str, Any]] = field(default_factory=list)
execution_id: str | None = None
operation_started: bool = False
outcome: dict[str, Any] | None = None
artifact_versions: dict[str, str] = field(default_factory=dict)
artifact_changes: list[str] | None = None
def transition(self, stage: str, **details: Any) -> None:
self.transitions.append({'sequence': len(self.transitions), 'stage': stage, **details})
def normalize(self, block: Any, reason: str) -> None:
tool, arguments = str(block.tool_type), str(block.content)
if tool != self.tool or arguments != self.arguments or not any(t['stage'] == 'normalized' for t in self.transitions):
self.transition('normalized', reason=reason, tool=tool, arguments=arguments,
previous_sha256=digest((self.tool, self.arguments)))
self.tool, self.arguments = tool, arguments
def finish(self, result: dict[str, Any]) -> None:
if self.outcome is not None:
return
code = result.get('exit_code')
valid_code = isinstance(code, int) and not isinstance(code, bool)
denied = bool(result.get('blocked') or result.get('approval_required')
or str(result.get('failure_kind', '')).endswith('_denied'))
self.outcome = {
'exit_code': code if valid_code else None,
'success': valid_code and code == 0 and not result.get('error') and not denied,
'authoritative': self.execution_id is not None and valid_code and not denied,
'blocked': denied,
'output_sha256': digest(result.get('output') or result.get('error') or result.get('stdout') or ''),
}
self.transition('outcome', **self.outcome)
def to_dict(self) -> dict[str, Any]:
return asdict(self)
@dataclass
class ActionJournal:
run_id: str = field(default_factory=lambda: uuid4().hex)
actions: list[ActionReceipt] = field(default_factory=list)
workspace: str = ''
observed_artifacts: tuple[str, ...] = ()
def capture_versions(self, action: ActionReceipt) -> None:
if self.workspace:
action.artifact_versions = {
artifact_identity(path, self.workspace): artifact_version(path, self.workspace)
for path in self.observed_artifacts
}
def propose(self, block: Any, call_id: str = '', native_call: dict | None = None) -> ActionReceipt:
native = native_call or {}
function = native.get('function') or native
if not isinstance(function, dict):
function = {}
action = ActionReceipt(
action_id=f'{self.run_id}:action:{len(self.actions) + 1}', call_id=call_id,
proposed_tool=str(block.tool_type), proposed_arguments=str(block.content),
provider_arguments=deepcopy(function.get('arguments')),
provider_tool=str(function.get('name') or ''),
tool=str(block.tool_type), arguments=str(block.content),
)
action.transition('proposed')
self.actions.append(action)
return action
def to_list(self) -> list[dict[str, Any]]:
return [action.to_dict() for action in self.actions]
def evidence_events(self) -> list[dict[str, Any]]:
return [dict(tool=a.tool, command=a.arguments,
exit_code=(a.outcome or {}).get('exit_code'),
error=not (a.outcome or {}).get('success'),
execution_attempted=bool((a.outcome or {}).get('authoritative')),
blocked=(a.outcome or {}).get('blocked', False),
action_id=a.action_id, execution_id=a.execution_id,
artifact_versions=a.artifact_versions, artifact_changes=a.artifact_changes)
for a in self.actions if a.outcome is not None]
_JOURNAL: ContextVar[ActionJournal | None] = ContextVar('runtime_action_journal', default=None)
_ACTION: ContextVar[ActionReceipt | None] = ContextVar('runtime_current_action', default=None)
@contextmanager
def bind_journal(journal: ActionJournal):
token = _JOURNAL.set(journal)
try:
yield journal
finally:
_JOURNAL.reset(token)
def current_journal() -> ActionJournal | None:
return _JOURNAL.get()
def propose_action(block: Any, call_id: str = '', native_call: dict | None = None) -> ActionReceipt | None:
journal = _JOURNAL.get()
return journal.propose(block, call_id, native_call) if journal else None
def mark_authorized() -> None:
action = _ACTION.get()
if action is not None and not any(t['stage'] == 'authorized' for t in action.transitions):
action.transition('authorized', authority='existing_dispatcher_policy')
def mark_dispatch() -> None:
action = _ACTION.get()
if action is not None and action.execution_id is None:
mark_authorized()
action.execution_id = action.action_id + ':execution:1'
action.transition('dispatched', execution_id=action.execution_id)
async def dispatched(operation):
"""Record an actual backend invocation, distinct from router admission."""
mark_dispatch()
return await operation
def mark_operation_started(backend: str, **details: Any) -> None:
action = _ACTION.get()
if action is not None:
action.operation_started = True
action.transition('operation_started', backend=backend, **details)
async def execute_action(executor, action: ActionReceipt | None, block: Any, **kwargs):
"""Adapter binds the proposal across async tool-task execution and cleanup."""
if action is not None:
action.normalize(block, 'agent_loop compatibility adapters')
token = _ACTION.set(action)
try:
return await executor(block, **kwargs)
finally:
_ACTION.reset(token)
def record_action(func):
call_signature = signature(func)
@wraps(func)
async def wrapped(*args, **kwargs):
bound = call_signature.bind(*args, **kwargs)
block = bound.arguments['block']
action = _ACTION.get() or propose_action(block)
token = _ACTION.set(action)
try:
journal = current_journal()
before = {}
if action is not None:
action.normalize(block, 'dispatcher input')
if journal is not None and journal.workspace:
journal.capture_versions(action)
before = dict(action.artifact_versions)
description, result = await func(*args, **kwargs)
if action is not None:
journal = current_journal()
if journal is not None:
journal.capture_versions(action)
if journal.workspace:
action.artifact_changes = [key for key, value in action.artifact_versions.items()
if before.get(key) != value]
if 'BLOCKED' in description and action.execution_id is None:
action.transition('authorization_denied', reason=str(result.get('error', '')))
action.finish({**result, 'blocked': True})
else:
action.finish(result)
return description, result
except BaseException as exc:
if action is not None:
action.transition('interrupted', category=type(exc).__name__)
raise
finally:
_ACTION.reset(token)
return wrapped
+26
View File
@@ -0,0 +1,26 @@
"""Existing sensitive-path policy shared by tools and evidence observation.
This is a deny predicate, not an authorization grant or a workspace scope.
"""
import os
_SENSITIVE_BASENAMES: set[str] = {
".ssh", ".gnupg", ".gitconfig",
".bashrc", ".bash_profile", ".bash_logout",
".zshrc", ".zprofile", ".zshenv",
".profile", ".tcshrc", ".cshrc", ".env", ".netrc",
}
_SENSITIVE_FILE_PATTERNS: tuple[str, ...] = (
"authorized_keys", "id_rsa", "id_ed25519", "id_ecdsa",
"known_hosts", "auth.json", "app.db", "settings.json",
)
_SENSITIVE_BASENAMES_CF = frozenset(b.casefold() for b in _SENSITIVE_BASENAMES)
_SENSITIVE_FILE_PATTERNS_CF = frozenset(p.casefold() for p in _SENSITIVE_FILE_PATTERNS)
def _is_sensitive_path(resolved: str) -> bool:
# Case folding is required even on POSIX: default macOS volumes are
# case insensitive but os.path.normcase there does not fold path names.
parts = [p.casefold() for p in resolved.split(os.sep)]
filename = parts[-1] if parts else ""
return any(part in _SENSITIVE_BASENAMES_CF for part in parts) or filename in _SENSITIVE_FILE_PATTERNS_CF