fix(runtime): scope completion claims to execution obligations

This commit is contained in:
Alexandre Teixeira
2026-10-01 01:36:18 +01:00
parent 9ee9205bfc
commit 4c122de880
5 changed files with 418 additions and 27 deletions
+92 -6
View File
@@ -280,6 +280,40 @@ def _known_input_is_explicit_mutation_target(instruction: str, path: str) -> boo
)
def _unquoted_statements(text: str) -> Iterable[tuple[str, str]]:
"""Yield original statements and their reportable prose, with quotes masked.
Mask before splitting so punctuation inside an example cannot change the
scope of the surrounding sentence. Inline code identifiers stay visible.
"""
def mask(match: re.Match[str]) -> str:
value = match.group()
# Quotation marks around an artifact identify a target, rather than
# quote a report. Keep that target available for exact path matching.
if value[0] in {'"', "'"} and re.fullmatch(_ARTIFACT_PATH, value[1:-1]):
return ' ' + value[1:-1] + ' '
return re.sub(r'[^\n]', ' ', value)
masked = re.sub(
r'```[\s\S]*?```|~~~[\s\S]*?~~~|"[^"\n]*"|(?<!\w)\'[^\'\n]*\'(?!\w)',
mask, text,
)
start = 0
for boundary in re.finditer(r'(?<=[.!?;])(?=\s)|(?<=\n)', masked):
end = boundary.start()
if end > start:
yield text[start:end], masked[start:end].replace('`', '')
start = end
if start < len(text):
yield text[start:], masked[start:].replace('`', '')
def _execution_obligation(requirements: CompletionRequirements) -> bool:
"""A derived view of the existing contract, never a separate declaration."""
return bool(requirements.required_artifacts or requirements.verifier_required
or requirements.executable_verifier_available or requirements.verifier_commands)
def infer_completion_requirements(
instruction: str,
*,
@@ -289,7 +323,15 @@ def infer_completion_requirements(
) -> CompletionRequirements:
"""Infer only explicitly requested output/edit paths from an instruction."""
text = str(instruction or "")
# Explanations can contain imperative examples. Their embedded actions
# are not requests to execute those actions. Keep independent requests in
# other statements, and keep explicitly supplied verifier requirements.
explanatory_request = re.compile(
r'^\s*(?:please\s+|(?:can|could|would)\s+you\s+)?'
r'(?:explain|describe|summari[sz]e|teach|discuss|'
r'show\s+(?:me\s+)?(?:an?\s+)?example|how\b)', re.I)
text = ''.join(scoped for _, scoped in _unquoted_statements(str(instruction or ''))
if not explanatory_request.search(scoped))
paths: list[str] = []
for pattern in (
_ARTIFACT_REQUEST_RE,
@@ -353,18 +395,23 @@ def infer_completion_requirements(
for command in verifier_commands
if str(command or "").strip()
))
explicit_test_request = re.search(
r'(?:^|[.;\n]|\b(?:and|then))\s*'
r'(?:please\s+|(?:can|could|would)\s+you\s+)?'
r'(?:run|execute)\s+(?:(?:the|all|a|full)\s+)*'
r'(?:tests?\b|test\s+suite\b|pytest\b|unittest\b|npm\s+test\b)', text, re.I)
verifier_required = executable_verifier_available or bool(cleaned_verifier_commands) or bool(
re.search(
explicit_test_request or (paths and re.search(
r"\b(?:then|after(?:wards)?|and)\b[^\n]{0,100}\b(?:test|verify|check|validate)\b",
str(instruction or ""),
text,
re.IGNORECASE,
)
))
)
return CompletionRequirements(
required_artifacts=tuple(paths),
verifier_required=verifier_required,
executable_verifier_available=(
executable_verifier_available or bool(cleaned_verifier_commands)
executable_verifier_available or bool(cleaned_verifier_commands) or bool(explicit_test_request)
),
verifier_commands=cleaned_verifier_commands,
)
@@ -575,6 +622,9 @@ class EvidenceLedger:
self.events: list[EvidenceEvent] = []
self._verification_versions: dict[str, str] = {}
self._verification_versions_captured = False
# Retain receipt command identity privately for presentation matching;
# model prose and client dictionaries never populate this evidence.
self._verifier_commands: dict[str, tuple[str, ...]] = {}
@classmethod
def from_tool_events(
@@ -718,13 +768,14 @@ class EvidenceLedger:
versions = event.get('artifact_versions')
self._verification_versions = dict(versions) if isinstance(versions, Mapping) else {}
self._verification_versions_captured = isinstance(versions, Mapping)
self._append(
verifier = self._append(
kind=EvidenceKind.VERIFIER_RESULT,
success=success,
authoritative=authoritative,
source=event,
detail="executable test/verifier command",
)
self._verifier_commands[verifier.event_id] = executable_words(_command_text(command))
elif tool in {"bash", "host_shell"} and is_validation_command(command) and not mutation_paths:
for path in self.requirements.required_artifacts:
if _path_is_mentioned(command, path):
@@ -736,6 +787,41 @@ class EvidenceLedger:
artifact_path=path,
)
def _supports_verifier_claim(self, identities: Sequence[str] = (), paths: Sequence[str] = ()) -> bool:
"""Only the current passing verifier may support its named runner."""
if self.evaluate().status != CompletionStatus.VERIFIED:
return False
latest = next((event for event in reversed(self.events)
if event.kind == EvidenceKind.VERIFIER_RESULT and event.authoritative), None)
if latest is None or not latest.success:
return False
words = self._verifier_commands.get(latest.event_id, ())
names = {Path(words[0]).name} if words else set()
if words and re.fullmatch(r'python(?:\d+(?:\.\d+)*)?', Path(words[0]).name) and '-m' in words:
module_index = words.index('-m') + 1
if module_index < len(words):
names.add(words[module_index])
return (all(identity in names for identity in identities)
and all(any(_artifact_path_matches_required(word, path, self.requirements.workspace_root)
for word in words) for path in paths))
def _supports_artifact_claim(self, kind: EvidenceKind, paths: Sequence[str]) -> bool:
"""Match every claimed artifact by identity, never by basename."""
targets = tuple(paths) or self.requirements.required_artifacts
if not targets or (not paths and len(targets) != 1):
return False
for path in targets:
matching = [event for event in self.events if event.kind == kind and event.authoritative
and _artifact_path_matches_required(event.artifact_path, path, self.requirements.workspace_root)]
successful = [event for event in matching if event.success]
# Match evaluate(): atomic helper failures preserve the previous
# successful artifact; a partial shell/Python failure may not.
destructive_failure = bool(matching and not matching[-1].success
and matching[-1].tool in {'bash', 'python'})
if not successful or destructive_failure:
return False
return True
def record_media_ingress(self, metadata: Mapping[str, Any]) -> None:
for artifact in metadata.get("artifacts") or []:
if not isinstance(artifact, Mapping):
+83 -19
View File
@@ -17,7 +17,8 @@ from time import perf_counter
from src.agent_evidence import (
CompletionDecision, CompletionStatus, EvidenceKind, EvidenceLedger,
requirements_from_runtime_context,
requirements_from_runtime_context, _execution_obligation, _unquoted_statements,
_ARTIFACT_PATH,
)
from .journal import ActionJournal, bind_journal, current_journal
@@ -30,10 +31,9 @@ _TEST_STATUS_CLAIM = re.compile(
r'(?:all\s+|have\s+|has\s+|now\s+|are\s+|is\s+|ran\s+)*'
r'(?:pass(?:ed|ing)?|succeeded|successful(?:ly)?|green)\b|'
r'\b(?:zero|no|0)\s+(?:test\s+)?failures\b', re.I)
_TERMINAL_SUCCESS = re.compile(r'^\s*(?:done|completed|success|all done|all set|fixed)\b', re.I)
_EXECUTION_CLAIM = re.compile(
r'\b(?:(?:I|we|I\'ve|we\'ve)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed)|'
r'(?:file|artifact|command|script|service|server)\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started)|'
r'\b(?:(?:I|we|I\'ve|we\'ve|and)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed)|'
rf'(?:file|artifact|command|script|service|server|{_ARTIFACT_PATH})\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started)|'
r'(?:successfully\s+)(?:ran|executed|created|updated|saved|completed))\b', re.I)
_UNATTESTED_TEST_METRIC = re.compile(
r'\b\d+\s+(?:(?:unit|integration)\s+)?tests?\s+pass(?:ed|ing)?\b|'
@@ -41,6 +41,55 @@ _UNATTESTED_TEST_METRIC = re.compile(
_UNBOUNDED_SUCCESS = re.compile(
r'\b(?:everything|all\s+(?:bugs|issues))\s+(?:is\s+|are\s+|has\s+been\s+)?'
r'(?:fixed|resolved|working)\b', re.I)
_MUTATION_CLAIM = re.compile(
r'\b(?:created|updated|modified|wrote|written|saved|fixed)\b', re.I)
_TEST_IDENTITY = re.compile(r'\b(?:pytest|unittest)\b', re.I)
_TEST_SUBJECT = re.compile(r'\b(?:tests?|test suite|pytest|unittest|checks?|verification)\b', re.I)
_CLAIM_PATH = re.compile(_ARTIFACT_PATH)
_BARE_SUCCESS = re.compile(r'^\s*(?:done|completed|success|all done|all set|fixed)[.!]?\s*$', re.I)
_NON_REPORT_SCOPE = re.compile(
r'^\s*(?:if|unless|suppose|imagine|hypothetically|for\s+(?:example|instance))\b|'
r'\b(?:if|when|whenever|unless|until)\b|'
r'\b(?:can|could|may|might|should|would|will|must)\b|'
r'\b(?:says?|said|states?|stated|example)\b', re.I)
def _current_run_claims(statement: str, *, execution_required: bool) -> list[tuple[str, str]]:
"""Classify asserted execution, separately from the turn's obligation.
Past actions and current result/status predicates are reports. Conditional,
modal, attributed and example clauses are scoped prose. Bare terminal
success only carries execution meaning under an execution contract.
"""
if _BARE_SUCCESS.fullmatch(statement):
return [('terminal', statement)] if execution_required else []
actions = list(_EXECUTION_CLAIM.finditer(statement))
leading = re.match(r'^\s*(?:successfully\s+)?(?:created|updated|modified|wrote|saved)\b', statement, re.I)
if leading:
actions.insert(0, leading)
candidates = [('action', match) for match in actions]
for kind, pattern in [('metric', _UNATTESTED_TEST_METRIC), ('metric', _UNBOUNDED_SUCCESS),
('test', _TEST_CLAIM), ('test', _TEST_STATUS_CLAIM)]:
candidates.extend((kind, match) for match in pattern.finditer(statement))
claims = []
for kind, match in candidates:
# Scope markers after an asserted action do not make that action
# hypothetical ("I ran pytest to see if ..."). An immediate conditional
# continuation does qualify a result ("Tests passed if ...").
if _NON_REPORT_SCOPE.search(statement[:match.start()]) or re.match(
r'\s+(?:if|when|whenever|unless|until)\b', statement[match.end():], re.I):
continue
end = next((action.start() for action in actions if action.start() > match.start()), len(statement))
scope = statement[match.start():end]
if kind == 'action':
if _MUTATION_CLAIM.search(match.group()):
kind = 'mutation'
elif _TEST_SUBJECT.search(scope):
kind = 'test'
else:
kind = 'execution'
claims.append((kind, scope))
return claims
def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDecision) -> tuple[str, str]:
@@ -51,32 +100,47 @@ def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDec
The execution outcome remains separate from a discarded model assertion.
"""
incomplete = decision.reason if not decision.can_complete and decision.status != CompletionStatus.AWAITING_USER else ''
productive = [event for event in ledger.events
if event.authoritative and event.success
and event.tool not in {'update_plan', 'todowrite', 'ask_user'}]
execution_required = _execution_obligation(ledger.requirements)
kept = []
removed = ''
for statement in re.split(r'(?<=[.!?])(?=\s)|(?<=\n)', text):
for statement, scoped in _unquoted_statements(text):
why = ''
if _UNATTESTED_TEST_METRIC.search(statement) or _UNBOUNDED_SUCCESS.search(statement):
why = 'test counts, coverage or exhaustive correctness were not established by execution evidence'
elif (_TEST_CLAIM.search(statement) or _TEST_STATUS_CLAIM.search(statement)) and decision.status != CompletionStatus.VERIFIED:
why = 'no current passing executable verification supports the claim'
elif (_EXECUTION_CLAIM.search(statement) or _TERMINAL_SUCCESS.search(statement)) and not productive:
why = 'no successful operation supports the execution claim'
elif incomplete and _TERMINAL_SUCCESS.search(statement):
why = incomplete
for claim, scope in _current_run_claims(scoped, execution_required=execution_required):
paths = tuple(match.group().rstrip('.') for match in _CLAIM_PATH.finditer(scope))
if claim == 'metric':
why = 'test counts, coverage or exhaustive correctness were not established by execution evidence'
elif claim == 'test':
identities = tuple(match.group().lower() for match in _TEST_IDENTITY.finditer(scope))
if decision.status != CompletionStatus.VERIFIED or not ledger._supports_verifier_claim(identities, paths):
why = 'no current passing executable verification supports the claim'
elif claim == 'mutation':
if not ledger._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, paths):
why = 'no matching artifact mutation supports the execution claim'
elif claim == 'execution':
# A generic assertion cannot be tied confidently to a receipt.
why = 'no matching operation supports the execution claim'
elif claim == 'terminal' and decision.status not in {CompletionStatus.SATISFIED, CompletionStatus.VERIFIED}:
why = incomplete or 'no successful execution supports completion'
if why:
break
if why:
removed = removed or why
else:
kept.append(statement)
prose = ''.join(kept).strip() if removed else text
if incomplete or (removed and decision.status in {CompletionStatus.UNVERIFIED, CompletionStatus.AWAITING_USER}):
if incomplete or (removed and execution_required and decision.status in {CompletionStatus.UNVERIFIED, CompletionStatus.AWAITING_USER}):
reason = incomplete or removed
missing = (' Missing artifacts: ' + ', '.join(decision.missing_artifacts) + '.'
if decision.missing_artifacts else '')
notice = 'The task is incomplete: ' + reason.rstrip('.') + '.' + missing
recorded = [path for path in ledger.requirements.required_artifacts
if ledger._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, (path,))]
if removed and recorded:
notice += ' Recorded artifact mutation: ' + ', '.join(recorded) + '.'
return notice + ('\n\n' + prose if prose.strip() else ''), reason
if removed and not execution_required and decision.status != CompletionStatus.VERIFIED:
notice = 'Unsupported execution claims were omitted: ' + removed.rstrip('.') + '.'
return (prose.rstrip() + '\n\n' + notice) if prose.strip() else notice, removed
if decision.can_complete and (ledger.requirements.required_artifacts or ledger.requirements.verifier_required or removed):
facts = []
if ledger.requirements.required_artifacts:
@@ -219,8 +283,8 @@ def with_completion_gate(func):
_, unsafe_draft = completion_answer(draft, ledger, presentation_decision)
if not answer.strip() and unsafe_draft:
reason = reason or unsafe_draft
safe_answer = 'The task is incomplete: ' + reason.rstrip('.') + '.'
if reason and decision.can_complete and decision.status == CompletionStatus.UNVERIFIED:
safe_answer, _ = completion_answer(draft, ledger, presentation_decision)
if reason and _execution_obligation(requirements) and decision.can_complete and decision.status == CompletionStatus.UNVERIFIED:
decision = CompletionDecision(CompletionStatus.UNVERIFIED, False, reason,
decision.evidence_ids, decision.missing_artifacts)
released_at = perf_counter()