fix(runtime): scope completion claims to execution obligations

This commit is contained in:
Alexandre Teixeira
2026-10-01 01:36:18 +01:00
parent 9ee9205bfc
commit 4c122de880
5 changed files with 418 additions and 27 deletions
+92 -6
View File
@@ -280,6 +280,40 @@ def _known_input_is_explicit_mutation_target(instruction: str, path: str) -> boo
)
def _unquoted_statements(text: str) -> Iterable[tuple[str, str]]:
"""Yield original statements and their reportable prose, with quotes masked.
Mask before splitting so punctuation inside an example cannot change the
scope of the surrounding sentence. Inline code identifiers stay visible.
"""
def mask(match: re.Match[str]) -> str:
value = match.group()
# Quotation marks around an artifact identify a target, rather than
# quote a report. Keep that target available for exact path matching.
if value[0] in {'"', "'"} and re.fullmatch(_ARTIFACT_PATH, value[1:-1]):
return ' ' + value[1:-1] + ' '
return re.sub(r'[^\n]', ' ', value)
masked = re.sub(
r'```[\s\S]*?```|~~~[\s\S]*?~~~|"[^"\n]*"|(?<!\w)\'[^\'\n]*\'(?!\w)',
mask, text,
)
start = 0
for boundary in re.finditer(r'(?<=[.!?;])(?=\s)|(?<=\n)', masked):
end = boundary.start()
if end > start:
yield text[start:end], masked[start:end].replace('`', '')
start = end
if start < len(text):
yield text[start:], masked[start:].replace('`', '')
def _execution_obligation(requirements: CompletionRequirements) -> bool:
"""A derived view of the existing contract, never a separate declaration."""
return bool(requirements.required_artifacts or requirements.verifier_required
or requirements.executable_verifier_available or requirements.verifier_commands)
def infer_completion_requirements(
instruction: str,
*,
@@ -289,7 +323,15 @@ def infer_completion_requirements(
) -> CompletionRequirements:
"""Infer only explicitly requested output/edit paths from an instruction."""
text = str(instruction or "")
# Explanations can contain imperative examples. Their embedded actions
# are not requests to execute those actions. Keep independent requests in
# other statements, and keep explicitly supplied verifier requirements.
explanatory_request = re.compile(
r'^\s*(?:please\s+|(?:can|could|would)\s+you\s+)?'
r'(?:explain|describe|summari[sz]e|teach|discuss|'
r'show\s+(?:me\s+)?(?:an?\s+)?example|how\b)', re.I)
text = ''.join(scoped for _, scoped in _unquoted_statements(str(instruction or ''))
if not explanatory_request.search(scoped))
paths: list[str] = []
for pattern in (
_ARTIFACT_REQUEST_RE,
@@ -353,18 +395,23 @@ def infer_completion_requirements(
for command in verifier_commands
if str(command or "").strip()
))
explicit_test_request = re.search(
r'(?:^|[.;\n]|\b(?:and|then))\s*'
r'(?:please\s+|(?:can|could|would)\s+you\s+)?'
r'(?:run|execute)\s+(?:(?:the|all|a|full)\s+)*'
r'(?:tests?\b|test\s+suite\b|pytest\b|unittest\b|npm\s+test\b)', text, re.I)
verifier_required = executable_verifier_available or bool(cleaned_verifier_commands) or bool(
re.search(
explicit_test_request or (paths and re.search(
r"\b(?:then|after(?:wards)?|and)\b[^\n]{0,100}\b(?:test|verify|check|validate)\b",
str(instruction or ""),
text,
re.IGNORECASE,
)
))
)
return CompletionRequirements(
required_artifacts=tuple(paths),
verifier_required=verifier_required,
executable_verifier_available=(
executable_verifier_available or bool(cleaned_verifier_commands)
executable_verifier_available or bool(cleaned_verifier_commands) or bool(explicit_test_request)
),
verifier_commands=cleaned_verifier_commands,
)
@@ -575,6 +622,9 @@ class EvidenceLedger:
self.events: list[EvidenceEvent] = []
self._verification_versions: dict[str, str] = {}
self._verification_versions_captured = False
# Retain receipt command identity privately for presentation matching;
# model prose and client dictionaries never populate this evidence.
self._verifier_commands: dict[str, tuple[str, ...]] = {}
@classmethod
def from_tool_events(
@@ -718,13 +768,14 @@ class EvidenceLedger:
versions = event.get('artifact_versions')
self._verification_versions = dict(versions) if isinstance(versions, Mapping) else {}
self._verification_versions_captured = isinstance(versions, Mapping)
self._append(
verifier = self._append(
kind=EvidenceKind.VERIFIER_RESULT,
success=success,
authoritative=authoritative,
source=event,
detail="executable test/verifier command",
)
self._verifier_commands[verifier.event_id] = executable_words(_command_text(command))
elif tool in {"bash", "host_shell"} and is_validation_command(command) and not mutation_paths:
for path in self.requirements.required_artifacts:
if _path_is_mentioned(command, path):
@@ -736,6 +787,41 @@ class EvidenceLedger:
artifact_path=path,
)
def _supports_verifier_claim(self, identities: Sequence[str] = (), paths: Sequence[str] = ()) -> bool:
"""Only the current passing verifier may support its named runner."""
if self.evaluate().status != CompletionStatus.VERIFIED:
return False
latest = next((event for event in reversed(self.events)
if event.kind == EvidenceKind.VERIFIER_RESULT and event.authoritative), None)
if latest is None or not latest.success:
return False
words = self._verifier_commands.get(latest.event_id, ())
names = {Path(words[0]).name} if words else set()
if words and re.fullmatch(r'python(?:\d+(?:\.\d+)*)?', Path(words[0]).name) and '-m' in words:
module_index = words.index('-m') + 1
if module_index < len(words):
names.add(words[module_index])
return (all(identity in names for identity in identities)
and all(any(_artifact_path_matches_required(word, path, self.requirements.workspace_root)
for word in words) for path in paths))
def _supports_artifact_claim(self, kind: EvidenceKind, paths: Sequence[str]) -> bool:
"""Match every claimed artifact by identity, never by basename."""
targets = tuple(paths) or self.requirements.required_artifacts
if not targets or (not paths and len(targets) != 1):
return False
for path in targets:
matching = [event for event in self.events if event.kind == kind and event.authoritative
and _artifact_path_matches_required(event.artifact_path, path, self.requirements.workspace_root)]
successful = [event for event in matching if event.success]
# Match evaluate(): atomic helper failures preserve the previous
# successful artifact; a partial shell/Python failure may not.
destructive_failure = bool(matching and not matching[-1].success
and matching[-1].tool in {'bash', 'python'})
if not successful or destructive_failure:
return False
return True
def record_media_ingress(self, metadata: Mapping[str, Any]) -> None:
for artifact in metadata.get("artifacts") or []:
if not isinstance(artifact, Mapping):
+83 -19
View File
@@ -17,7 +17,8 @@ from time import perf_counter
from src.agent_evidence import (
CompletionDecision, CompletionStatus, EvidenceKind, EvidenceLedger,
requirements_from_runtime_context,
requirements_from_runtime_context, _execution_obligation, _unquoted_statements,
_ARTIFACT_PATH,
)
from .journal import ActionJournal, bind_journal, current_journal
@@ -30,10 +31,9 @@ _TEST_STATUS_CLAIM = re.compile(
r'(?:all\s+|have\s+|has\s+|now\s+|are\s+|is\s+|ran\s+)*'
r'(?:pass(?:ed|ing)?|succeeded|successful(?:ly)?|green)\b|'
r'\b(?:zero|no|0)\s+(?:test\s+)?failures\b', re.I)
_TERMINAL_SUCCESS = re.compile(r'^\s*(?:done|completed|success|all done|all set|fixed)\b', re.I)
_EXECUTION_CLAIM = re.compile(
r'\b(?:(?:I|we|I\'ve|we\'ve)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed)|'
r'(?:file|artifact|command|script|service|server)\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started)|'
r'\b(?:(?:I|we|I\'ve|we\'ve|and)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed)|'
rf'(?:file|artifact|command|script|service|server|{_ARTIFACT_PATH})\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started)|'
r'(?:successfully\s+)(?:ran|executed|created|updated|saved|completed))\b', re.I)
_UNATTESTED_TEST_METRIC = re.compile(
r'\b\d+\s+(?:(?:unit|integration)\s+)?tests?\s+pass(?:ed|ing)?\b|'
@@ -41,6 +41,55 @@ _UNATTESTED_TEST_METRIC = re.compile(
_UNBOUNDED_SUCCESS = re.compile(
r'\b(?:everything|all\s+(?:bugs|issues))\s+(?:is\s+|are\s+|has\s+been\s+)?'
r'(?:fixed|resolved|working)\b', re.I)
_MUTATION_CLAIM = re.compile(
r'\b(?:created|updated|modified|wrote|written|saved|fixed)\b', re.I)
_TEST_IDENTITY = re.compile(r'\b(?:pytest|unittest)\b', re.I)
_TEST_SUBJECT = re.compile(r'\b(?:tests?|test suite|pytest|unittest|checks?|verification)\b', re.I)
_CLAIM_PATH = re.compile(_ARTIFACT_PATH)
_BARE_SUCCESS = re.compile(r'^\s*(?:done|completed|success|all done|all set|fixed)[.!]?\s*$', re.I)
_NON_REPORT_SCOPE = re.compile(
r'^\s*(?:if|unless|suppose|imagine|hypothetically|for\s+(?:example|instance))\b|'
r'\b(?:if|when|whenever|unless|until)\b|'
r'\b(?:can|could|may|might|should|would|will|must)\b|'
r'\b(?:says?|said|states?|stated|example)\b', re.I)
def _current_run_claims(statement: str, *, execution_required: bool) -> list[tuple[str, str]]:
"""Classify asserted execution, separately from the turn's obligation.
Past actions and current result/status predicates are reports. Conditional,
modal, attributed and example clauses are scoped prose. Bare terminal
success only carries execution meaning under an execution contract.
"""
if _BARE_SUCCESS.fullmatch(statement):
return [('terminal', statement)] if execution_required else []
actions = list(_EXECUTION_CLAIM.finditer(statement))
leading = re.match(r'^\s*(?:successfully\s+)?(?:created|updated|modified|wrote|saved)\b', statement, re.I)
if leading:
actions.insert(0, leading)
candidates = [('action', match) for match in actions]
for kind, pattern in [('metric', _UNATTESTED_TEST_METRIC), ('metric', _UNBOUNDED_SUCCESS),
('test', _TEST_CLAIM), ('test', _TEST_STATUS_CLAIM)]:
candidates.extend((kind, match) for match in pattern.finditer(statement))
claims = []
for kind, match in candidates:
# Scope markers after an asserted action do not make that action
# hypothetical ("I ran pytest to see if ..."). An immediate conditional
# continuation does qualify a result ("Tests passed if ...").
if _NON_REPORT_SCOPE.search(statement[:match.start()]) or re.match(
r'\s+(?:if|when|whenever|unless|until)\b', statement[match.end():], re.I):
continue
end = next((action.start() for action in actions if action.start() > match.start()), len(statement))
scope = statement[match.start():end]
if kind == 'action':
if _MUTATION_CLAIM.search(match.group()):
kind = 'mutation'
elif _TEST_SUBJECT.search(scope):
kind = 'test'
else:
kind = 'execution'
claims.append((kind, scope))
return claims
def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDecision) -> tuple[str, str]:
@@ -51,32 +100,47 @@ def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDec
The execution outcome remains separate from a discarded model assertion.
"""
incomplete = decision.reason if not decision.can_complete and decision.status != CompletionStatus.AWAITING_USER else ''
productive = [event for event in ledger.events
if event.authoritative and event.success
and event.tool not in {'update_plan', 'todowrite', 'ask_user'}]
execution_required = _execution_obligation(ledger.requirements)
kept = []
removed = ''
for statement in re.split(r'(?<=[.!?])(?=\s)|(?<=\n)', text):
for statement, scoped in _unquoted_statements(text):
why = ''
if _UNATTESTED_TEST_METRIC.search(statement) or _UNBOUNDED_SUCCESS.search(statement):
why = 'test counts, coverage or exhaustive correctness were not established by execution evidence'
elif (_TEST_CLAIM.search(statement) or _TEST_STATUS_CLAIM.search(statement)) and decision.status != CompletionStatus.VERIFIED:
why = 'no current passing executable verification supports the claim'
elif (_EXECUTION_CLAIM.search(statement) or _TERMINAL_SUCCESS.search(statement)) and not productive:
why = 'no successful operation supports the execution claim'
elif incomplete and _TERMINAL_SUCCESS.search(statement):
why = incomplete
for claim, scope in _current_run_claims(scoped, execution_required=execution_required):
paths = tuple(match.group().rstrip('.') for match in _CLAIM_PATH.finditer(scope))
if claim == 'metric':
why = 'test counts, coverage or exhaustive correctness were not established by execution evidence'
elif claim == 'test':
identities = tuple(match.group().lower() for match in _TEST_IDENTITY.finditer(scope))
if decision.status != CompletionStatus.VERIFIED or not ledger._supports_verifier_claim(identities, paths):
why = 'no current passing executable verification supports the claim'
elif claim == 'mutation':
if not ledger._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, paths):
why = 'no matching artifact mutation supports the execution claim'
elif claim == 'execution':
# A generic assertion cannot be tied confidently to a receipt.
why = 'no matching operation supports the execution claim'
elif claim == 'terminal' and decision.status not in {CompletionStatus.SATISFIED, CompletionStatus.VERIFIED}:
why = incomplete or 'no successful execution supports completion'
if why:
break
if why:
removed = removed or why
else:
kept.append(statement)
prose = ''.join(kept).strip() if removed else text
if incomplete or (removed and decision.status in {CompletionStatus.UNVERIFIED, CompletionStatus.AWAITING_USER}):
if incomplete or (removed and execution_required and decision.status in {CompletionStatus.UNVERIFIED, CompletionStatus.AWAITING_USER}):
reason = incomplete or removed
missing = (' Missing artifacts: ' + ', '.join(decision.missing_artifacts) + '.'
if decision.missing_artifacts else '')
notice = 'The task is incomplete: ' + reason.rstrip('.') + '.' + missing
recorded = [path for path in ledger.requirements.required_artifacts
if ledger._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, (path,))]
if removed and recorded:
notice += ' Recorded artifact mutation: ' + ', '.join(recorded) + '.'
return notice + ('\n\n' + prose if prose.strip() else ''), reason
if removed and not execution_required and decision.status != CompletionStatus.VERIFIED:
notice = 'Unsupported execution claims were omitted: ' + removed.rstrip('.') + '.'
return (prose.rstrip() + '\n\n' + notice) if prose.strip() else notice, removed
if decision.can_complete and (ledger.requirements.required_artifacts or ledger.requirements.verifier_required or removed):
facts = []
if ledger.requirements.required_artifacts:
@@ -219,8 +283,8 @@ def with_completion_gate(func):
_, unsafe_draft = completion_answer(draft, ledger, presentation_decision)
if not answer.strip() and unsafe_draft:
reason = reason or unsafe_draft
safe_answer = 'The task is incomplete: ' + reason.rstrip('.') + '.'
if reason and decision.can_complete and decision.status == CompletionStatus.UNVERIFIED:
safe_answer, _ = completion_answer(draft, ledger, presentation_decision)
if reason and _execution_obligation(requirements) and decision.can_complete and decision.status == CompletionStatus.UNVERIFIED:
decision = CompletionDecision(CompletionStatus.UNVERIFIED, False, reason,
decision.evidence_ids, decision.missing_artifacts)
released_at = perf_counter()
+34
View File
@@ -135,6 +135,40 @@ def test_terminal_completion_missing_artifact_does_not_add_model_rounds(monkeypa
assert decision["missing_artifacts"] == ["answer.json"]
def test_slice2_explanatory_request_does_not_add_verification_or_model_rounds(monkeypatch):
calls = _patch_loop(monkeypatch, ['Tests pass when the command exits zero.'])
events = _run('Explain how to write code and then test it.', max_rounds=1)
assert calls() == 1
decision = next(event['data'] for event in events if event.get('type') == 'completion_decision')
assert decision['can_complete'] is True
assert 'The task is incomplete' not in json.dumps(events)
metrics = next(event['data'] for event in events if event.get('type') == 'metrics')
assert not metrics['completion_requirements']['verifier_required']
assert metrics['completion_gate']['additional_provider_calls'] == 0
def test_slice2_fabricated_execution_on_conversational_turn_does_not_add_rounds(monkeypatch):
calls = _patch_loop(monkeypatch, ['I ran pytest and all tests passed.'])
events = _run('Explain what pytest does.', max_rounds=1)
assert calls() == 1
final = next(event['content'] for event in events if event.get('type') == 'final_response')
assert 'I ran pytest' not in final
assert 'The task is incomplete' not in final
metrics = next(event['data'] for event in events if event.get('type') == 'metrics')
assert metrics['completion_gate']['additional_provider_calls'] == 0
def test_slice2_unsupported_test_report_does_not_add_model_rounds(monkeypatch):
calls = _patch_loop(monkeypatch, ['I ran pytest and all 42 tests passed.'])
events = _run('Run pytest.', max_rounds=4, relevant_tools={'bash'})
assert calls() == 1
decision = next(event['data'] for event in events if event.get('type') == 'completion_decision')
assert decision['can_complete'] is False
final = next(event['content'] for event in events if event.get('type') == 'final_response')
assert final.startswith('The task is incomplete:')
assert '42' not in final
def test_failed_trailing_tool_with_planning_prose_continues_artifact_task(monkeypatch):
calls = _patch_loop(
monkeypatch,
+131
View File
@@ -6,6 +6,8 @@ import json
import pytest
from src.agent_runtime.completion import with_completion_gate
from src.agent_runtime.completion import completion_answer
from src.agent_evidence import CompletionRequirements, EvidenceLedger, infer_completion_requirements
from src.agent_runtime.journal import current_journal
from src.tool_types import ToolBlock
from tests.runtime_evidence_helpers import authoritative_executor
@@ -225,3 +227,132 @@ async def test_cancellation_closes_inner_stream_without_releasing_completion(aft
assert _labels(chunks) == ['tool_start']
assert closed == [True]
assert current_journal() is None
@pytest.mark.parametrize('prose', [
'Tests pass when the command exits zero.',
'If all tests are passing, merge the branch.',
'Tests passed if the command exited zero.',
'The documentation says "5 passed".',
'The documentation says "Tests: FAIL" or "Tests: PASS".',
'You can run pytest to verify this.',
'A successful test run should show no failures.',
'For example, I created the file and updated config.py.',
'If I updated config.py, I would run pytest.',
'Imagine I ran the tests and all 42 passed.',
'Done is the label for a finished item.',
'```text\nI ran pytest and all 42 passed.\n```',
'Run pytest until there are no failures.',
])
def test_slice2_explanatory_prose_is_not_a_current_run_claim(prose):
ledger = EvidenceLedger()
answer, reason = completion_answer(prose, ledger, ledger.evaluate())
assert answer == prose
assert not reason
@pytest.mark.parametrize('instruction', [
'Explain how to write code and then test it.',
'Summarise this and check for typos.',
'Explain how to update config.py and then verify it.',
'Show an example of creating answer.json and checking it.',
'The documentation says "run pytest and create answer.json".',
'If you run pytest, the tests should pass.',
])
def test_slice2_explanatory_request_has_no_execution_requirements(instruction):
requirements = infer_completion_requirements(instruction)
assert requirements.required_artifacts == ()
assert not requirements.verifier_required
assert not requirements.executable_verifier_available
@pytest.mark.parametrize('instruction', [
'Run the tests.', 'Please run pytest.', 'Can you run the test suite?',
])
def test_slice2_explicit_test_execution_requires_a_verifier(instruction):
requirements = infer_completion_requirements(instruction)
assert requirements.verifier_required
assert not EvidenceLedger(requirements).evaluate().can_complete
@pytest.mark.parametrize('claim', [
'I ran the tests.', 'The tests passed.', '42 tests passed.',
'I created the file.', 'I updated config.py successfully.',
])
def test_slice2_execution_obligation_rejects_unsupported_claims(claim):
ledger = EvidenceLedger(CompletionRequirements(required_artifacts=('config.py',)))
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
assert reason
assert answer.startswith('The task is incomplete:')
assert claim not in answer
@pytest.mark.parametrize('claim', [
'I ran pytest to see if the tests passed.',
'I updated config.py as an example.',
'I ran pytest and should update config.py next.',
'config.py was updated successfully.',
])
def test_slice2_subordinate_explanation_cannot_hide_a_direct_execution_report(claim):
ledger = EvidenceLedger()
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
assert reason
assert claim not in answer
assert 'The task is incomplete' not in answer
@pytest.mark.asyncio
async def test_slice2_client_dictionary_cannot_attest_execution():
@with_completion_gate
async def stream(messages, client_runtime_context=None):
yield _event({'delta': 'I ran pytest and all tests passed.'})
yield DONE
context = {'execution_obligation': True, 'execution_verified': True,
'evidence_events': [{'tool': 'bash', 'command': 'pytest', 'exit_code': 0}]}
chunks = [chunk async for chunk in stream(
[{'role': 'user', 'content': 'Explain test output.'}], client_runtime_context=context)]
final = next(data['content'] for event, data in _frames(chunks)
if event == 'message' and data.get('type') == 'final_response')
assert 'I ran pytest' not in final
assert 'The task is incomplete' not in final
assert _decision(chunks)['can_complete'] is True
@pytest.mark.asyncio
async def test_slice2_conversational_fabrication_is_corrected_without_execution_incomplete():
invocations = []
@with_completion_gate
async def stream(messages):
invocations.append(1)
yield _event({'delta': 'The function returns a boolean. I ran pytest and all tests passed.'})
yield _event({'type': 'metrics', 'data': {}})
yield DONE
chunks = [chunk async for chunk in stream([{'role': 'user', 'content': 'Explain the function.'}])]
final = next(data['content'] for event, data in _frames(chunks)
if event == 'message' and data.get('type') == 'final_response')
assert 'The function returns a boolean.' in final
assert 'I ran pytest' not in final
assert 'The task is incomplete' not in final
assert _decision(chunks)['can_complete'] is True
assert invocations == [1]
metrics = next(data['data'] for event, data in _frames(chunks)
if event == 'message' and data.get('type') == 'metrics')
assert metrics['completion_gate']['additional_provider_calls'] == 0
@pytest.mark.asyncio
async def test_slice2_quoted_example_does_not_hide_an_unsupported_report():
@with_completion_gate
async def stream(messages):
yield _event({'delta': 'The docs say "5 passed". I ran pytest.'})
yield DONE
chunks = [chunk async for chunk in stream([{'role': 'user', 'content': 'Explain pytest output.'}])]
final = next(data['content'] for event, data in _frames(chunks)
if event == 'message' and data.get('type') == 'final_response')
assert 'The docs say "5 passed".' in final
assert 'I ran pytest' not in final
assert 'The task is incomplete' not in final
+78 -2
View File
@@ -128,7 +128,7 @@ def test_readback_does_not_substitute_for_required_executable_tests():
'Test suite ran successfully', 'No failures.', 'Done.',
'I executed the command.', 'Successfully created the file.'])
def test_no_execution_receipts_cannot_support_adversarial_success_claims(claim):
ledger = EvidenceLedger()
ledger = EvidenceLedger(CompletionRequirements(verifier_required=True))
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
assert reason
assert answer.startswith('The task is incomplete:')
@@ -178,7 +178,7 @@ async def test_mixed_thinking_delta_cannot_publish_success_before_gate(thinking)
yield 'data: {"type":"tool_start","tool":"bash"}\n\n'
yield 'data: {"type":"metrics","data":{"thinking":"All tests passed."}}\n\n'
yield 'data: [DONE]\n\n'
events = decode([chunk async for chunk in stream([])])
events = decode([chunk async for chunk in stream([{'role': 'user', 'content': 'Run the tests.'}])])
assert events[0] == {'type': 'tool_start', 'tool': 'bash'}
assert events[1]['type'] == 'completion_decision'
assert not events[1]['data']['can_complete']
@@ -398,3 +398,79 @@ async def test_unknown_tool_never_creates_dispatch_identity(monkeypatch):
await execute_tool_block(ToolBlock('unknown_nonexistent_tool', '{}'), security_context=NO_TOOL_SECURITY_CONTEXT)
assert journal.actions[0].execution_id is None
assert not journal.actions[0].outcome['authoritative']
@pytest.mark.parametrize('tool,command,claim', [
('read_file', 'README.md', 'I ran the tests.'),
('bash', 'printf observation', 'The tests passed.'),
('read_file', 'README.md', 'I updated config.py.'),
('bash', 'printf observation', 'I created the file.'),
('write_file', '{"path":"other.py"}', 'I updated config.py.'),
('write_file', '{"path":"nested/config.py"}', 'I updated config.py.'),
('bash', 'python -m unittest', 'I ran pytest and the tests passed.'),
('bash', 'pytest tests/test_other.py', 'I ran pytest tests/test_config.py.'),
('bash', 'pytest', 'I updated config.py and the tests passed.'),
('write_file', '{"path":"config.py"}', 'I updated config.py and ran pytest.'),
('bash', 'python -m unittest pytest', 'I ran pytest.'),
('write_file', '{"path":"config.py"}', 'I updated "settings.py".'),
('bash', 'pytest', 'Created config.py and ran pytest.'),
])
def test_slice2_unrelated_receipt_cannot_support_claim(tool, command, claim):
ledger = EvidenceLedger.from_tool_events([
{'tool': tool, 'command': command, 'exit_code': 0},
])
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
assert reason
assert claim not in answer
@pytest.mark.parametrize('claim', ['I ran pytest.', 'The tests passed.', 'Tests: PASS'])
def test_slice2_matching_verifier_supports_test_claim(claim):
ledger = EvidenceLedger.from_tool_events([
{'tool': 'bash', 'command': 'python3 -m pytest -q', 'exit_code': 0},
])
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
assert claim in answer
assert not reason
@pytest.mark.parametrize('claim', ['I updated config.py.', 'I updated `./config.py` successfully.',
'I updated "config.py".'])
def test_slice2_matching_mutation_supports_artifact_claim(claim):
ledger = EvidenceLedger.from_tool_events([
{'tool': 'edit_file', 'command': '{"path":"config.py"}', 'exit_code': 0},
], CompletionRequirements(required_artifacts=('config.py',)))
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
assert claim in answer
assert not reason
def test_slice2_one_matching_path_does_not_support_multiple_artifact_claims():
ledger = EvidenceLedger.from_tool_events([
{'tool': 'edit_file', 'command': '{"path":"config.py"}', 'exit_code': 0},
])
claim = 'I updated config.py and settings.py.'
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
assert reason
assert claim not in answer
def test_slice2_verifier_before_mutation_cannot_support_current_test_success():
ledger = EvidenceLedger.from_tool_events([
{'tool': 'bash', 'command': 'pytest', 'exit_code': 0},
{'tool': 'edit_file', 'command': '{"path":"config.py"}', 'exit_code': 0},
], CompletionRequirements(required_artifacts=('config.py',)))
answer, reason = completion_answer('The tests passed.', ledger, ledger.evaluate())
assert reason
assert 'The tests passed.' not in answer
def test_slice2_matching_artifact_and_verifier_support_combined_claim():
ledger = EvidenceLedger.from_tool_events([
{'tool': 'edit_file', 'command': '{"path":"config.py"}', 'exit_code': 0},
{'tool': 'bash', 'command': 'pytest tests/test_config.py', 'exit_code': 0},
], CompletionRequirements(required_artifacts=('config.py',)))
claim = 'I updated config.py and ran pytest tests/test_config.py.'
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
assert claim in answer
assert not reason