diff --git a/src/agent_evidence.py b/src/agent_evidence.py index f11baed20..674837438 100644 --- a/src/agent_evidence.py +++ b/src/agent_evidence.py @@ -280,6 +280,40 @@ def _known_input_is_explicit_mutation_target(instruction: str, path: str) -> boo ) +def _unquoted_statements(text: str) -> Iterable[tuple[str, str]]: + """Yield original statements and their reportable prose, with quotes masked. + + Mask before splitting so punctuation inside an example cannot change the + scope of the surrounding sentence. Inline code identifiers stay visible. + """ + def mask(match: re.Match[str]) -> str: + value = match.group() + # Quotation marks around an artifact identify a target, rather than + # quote a report. Keep that target available for exact path matching. + if value[0] in {'"', "'"} and re.fullmatch(_ARTIFACT_PATH, value[1:-1]): + return ' ' + value[1:-1] + ' ' + return re.sub(r'[^\n]', ' ', value) + + masked = re.sub( + r'```[\s\S]*?```|~~~[\s\S]*?~~~|"[^"\n]*"|(? start: + yield text[start:end], masked[start:end].replace('`', '') + start = end + if start < len(text): + yield text[start:], masked[start:].replace('`', '') + + +def _execution_obligation(requirements: CompletionRequirements) -> bool: + """A derived view of the existing contract, never a separate declaration.""" + return bool(requirements.required_artifacts or requirements.verifier_required + or requirements.executable_verifier_available or requirements.verifier_commands) + + def infer_completion_requirements( instruction: str, *, @@ -289,7 +323,15 @@ def infer_completion_requirements( ) -> CompletionRequirements: """Infer only explicitly requested output/edit paths from an instruction.""" - text = str(instruction or "") + # Explanations can contain imperative examples. Their embedded actions + # are not requests to execute those actions. Keep independent requests in + # other statements, and keep explicitly supplied verifier requirements. + explanatory_request = re.compile( + r'^\s*(?:please\s+|(?:can|could|would)\s+you\s+)?' + r'(?:explain|describe|summari[sz]e|teach|discuss|' + r'show\s+(?:me\s+)?(?:an?\s+)?example|how\b)', re.I) + text = ''.join(scoped for _, scoped in _unquoted_statements(str(instruction or '')) + if not explanatory_request.search(scoped)) paths: list[str] = [] for pattern in ( _ARTIFACT_REQUEST_RE, @@ -353,18 +395,23 @@ def infer_completion_requirements( for command in verifier_commands if str(command or "").strip() )) + explicit_test_request = re.search( + r'(?:^|[.;\n]|\b(?:and|then))\s*' + r'(?:please\s+|(?:can|could|would)\s+you\s+)?' + r'(?:run|execute)\s+(?:(?:the|all|a|full)\s+)*' + r'(?:tests?\b|test\s+suite\b|pytest\b|unittest\b|npm\s+test\b)', text, re.I) verifier_required = executable_verifier_available or bool(cleaned_verifier_commands) or bool( - re.search( + explicit_test_request or (paths and re.search( r"\b(?:then|after(?:wards)?|and)\b[^\n]{0,100}\b(?:test|verify|check|validate)\b", - str(instruction or ""), + text, re.IGNORECASE, - ) + )) ) return CompletionRequirements( required_artifacts=tuple(paths), verifier_required=verifier_required, executable_verifier_available=( - executable_verifier_available or bool(cleaned_verifier_commands) + executable_verifier_available or bool(cleaned_verifier_commands) or bool(explicit_test_request) ), verifier_commands=cleaned_verifier_commands, ) @@ -575,6 +622,9 @@ class EvidenceLedger: self.events: list[EvidenceEvent] = [] self._verification_versions: dict[str, str] = {} self._verification_versions_captured = False + # Retain receipt command identity privately for presentation matching; + # model prose and client dictionaries never populate this evidence. + self._verifier_commands: dict[str, tuple[str, ...]] = {} @classmethod def from_tool_events( @@ -718,13 +768,14 @@ class EvidenceLedger: versions = event.get('artifact_versions') self._verification_versions = dict(versions) if isinstance(versions, Mapping) else {} self._verification_versions_captured = isinstance(versions, Mapping) - self._append( + verifier = self._append( kind=EvidenceKind.VERIFIER_RESULT, success=success, authoritative=authoritative, source=event, detail="executable test/verifier command", ) + self._verifier_commands[verifier.event_id] = executable_words(_command_text(command)) elif tool in {"bash", "host_shell"} and is_validation_command(command) and not mutation_paths: for path in self.requirements.required_artifacts: if _path_is_mentioned(command, path): @@ -736,6 +787,41 @@ class EvidenceLedger: artifact_path=path, ) + def _supports_verifier_claim(self, identities: Sequence[str] = (), paths: Sequence[str] = ()) -> bool: + """Only the current passing verifier may support its named runner.""" + if self.evaluate().status != CompletionStatus.VERIFIED: + return False + latest = next((event for event in reversed(self.events) + if event.kind == EvidenceKind.VERIFIER_RESULT and event.authoritative), None) + if latest is None or not latest.success: + return False + words = self._verifier_commands.get(latest.event_id, ()) + names = {Path(words[0]).name} if words else set() + if words and re.fullmatch(r'python(?:\d+(?:\.\d+)*)?', Path(words[0]).name) and '-m' in words: + module_index = words.index('-m') + 1 + if module_index < len(words): + names.add(words[module_index]) + return (all(identity in names for identity in identities) + and all(any(_artifact_path_matches_required(word, path, self.requirements.workspace_root) + for word in words) for path in paths)) + + def _supports_artifact_claim(self, kind: EvidenceKind, paths: Sequence[str]) -> bool: + """Match every claimed artifact by identity, never by basename.""" + targets = tuple(paths) or self.requirements.required_artifacts + if not targets or (not paths and len(targets) != 1): + return False + for path in targets: + matching = [event for event in self.events if event.kind == kind and event.authoritative + and _artifact_path_matches_required(event.artifact_path, path, self.requirements.workspace_root)] + successful = [event for event in matching if event.success] + # Match evaluate(): atomic helper failures preserve the previous + # successful artifact; a partial shell/Python failure may not. + destructive_failure = bool(matching and not matching[-1].success + and matching[-1].tool in {'bash', 'python'}) + if not successful or destructive_failure: + return False + return True + def record_media_ingress(self, metadata: Mapping[str, Any]) -> None: for artifact in metadata.get("artifacts") or []: if not isinstance(artifact, Mapping): diff --git a/src/agent_runtime/completion.py b/src/agent_runtime/completion.py index d41a49d7a..524070861 100644 --- a/src/agent_runtime/completion.py +++ b/src/agent_runtime/completion.py @@ -17,7 +17,8 @@ from time import perf_counter from src.agent_evidence import ( CompletionDecision, CompletionStatus, EvidenceKind, EvidenceLedger, - requirements_from_runtime_context, + requirements_from_runtime_context, _execution_obligation, _unquoted_statements, + _ARTIFACT_PATH, ) from .journal import ActionJournal, bind_journal, current_journal @@ -30,10 +31,9 @@ _TEST_STATUS_CLAIM = re.compile( r'(?:all\s+|have\s+|has\s+|now\s+|are\s+|is\s+|ran\s+)*' r'(?:pass(?:ed|ing)?|succeeded|successful(?:ly)?|green)\b|' r'\b(?:zero|no|0)\s+(?:test\s+)?failures\b', re.I) -_TERMINAL_SUCCESS = re.compile(r'^\s*(?:done|completed|success|all done|all set|fixed)\b', re.I) _EXECUTION_CLAIM = re.compile( - r'\b(?:(?:I|we|I\'ve|we\'ve)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed)|' - r'(?:file|artifact|command|script|service|server)\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started)|' + r'\b(?:(?:I|we|I\'ve|we\'ve|and)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed)|' + rf'(?:file|artifact|command|script|service|server|{_ARTIFACT_PATH})\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started)|' r'(?:successfully\s+)(?:ran|executed|created|updated|saved|completed))\b', re.I) _UNATTESTED_TEST_METRIC = re.compile( r'\b\d+\s+(?:(?:unit|integration)\s+)?tests?\s+pass(?:ed|ing)?\b|' @@ -41,6 +41,55 @@ _UNATTESTED_TEST_METRIC = re.compile( _UNBOUNDED_SUCCESS = re.compile( r'\b(?:everything|all\s+(?:bugs|issues))\s+(?:is\s+|are\s+|has\s+been\s+)?' r'(?:fixed|resolved|working)\b', re.I) +_MUTATION_CLAIM = re.compile( + r'\b(?:created|updated|modified|wrote|written|saved|fixed)\b', re.I) +_TEST_IDENTITY = re.compile(r'\b(?:pytest|unittest)\b', re.I) +_TEST_SUBJECT = re.compile(r'\b(?:tests?|test suite|pytest|unittest|checks?|verification)\b', re.I) +_CLAIM_PATH = re.compile(_ARTIFACT_PATH) +_BARE_SUCCESS = re.compile(r'^\s*(?:done|completed|success|all done|all set|fixed)[.!]?\s*$', re.I) +_NON_REPORT_SCOPE = re.compile( + r'^\s*(?:if|unless|suppose|imagine|hypothetically|for\s+(?:example|instance))\b|' + r'\b(?:if|when|whenever|unless|until)\b|' + r'\b(?:can|could|may|might|should|would|will|must)\b|' + r'\b(?:says?|said|states?|stated|example)\b', re.I) + + +def _current_run_claims(statement: str, *, execution_required: bool) -> list[tuple[str, str]]: + """Classify asserted execution, separately from the turn's obligation. + + Past actions and current result/status predicates are reports. Conditional, + modal, attributed and example clauses are scoped prose. Bare terminal + success only carries execution meaning under an execution contract. + """ + if _BARE_SUCCESS.fullmatch(statement): + return [('terminal', statement)] if execution_required else [] + actions = list(_EXECUTION_CLAIM.finditer(statement)) + leading = re.match(r'^\s*(?:successfully\s+)?(?:created|updated|modified|wrote|saved)\b', statement, re.I) + if leading: + actions.insert(0, leading) + candidates = [('action', match) for match in actions] + for kind, pattern in [('metric', _UNATTESTED_TEST_METRIC), ('metric', _UNBOUNDED_SUCCESS), + ('test', _TEST_CLAIM), ('test', _TEST_STATUS_CLAIM)]: + candidates.extend((kind, match) for match in pattern.finditer(statement)) + claims = [] + for kind, match in candidates: + # Scope markers after an asserted action do not make that action + # hypothetical ("I ran pytest to see if ..."). An immediate conditional + # continuation does qualify a result ("Tests passed if ..."). + if _NON_REPORT_SCOPE.search(statement[:match.start()]) or re.match( + r'\s+(?:if|when|whenever|unless|until)\b', statement[match.end():], re.I): + continue + end = next((action.start() for action in actions if action.start() > match.start()), len(statement)) + scope = statement[match.start():end] + if kind == 'action': + if _MUTATION_CLAIM.search(match.group()): + kind = 'mutation' + elif _TEST_SUBJECT.search(scope): + kind = 'test' + else: + kind = 'execution' + claims.append((kind, scope)) + return claims def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDecision) -> tuple[str, str]: @@ -51,32 +100,47 @@ def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDec The execution outcome remains separate from a discarded model assertion. """ incomplete = decision.reason if not decision.can_complete and decision.status != CompletionStatus.AWAITING_USER else '' - productive = [event for event in ledger.events - if event.authoritative and event.success - and event.tool not in {'update_plan', 'todowrite', 'ask_user'}] + execution_required = _execution_obligation(ledger.requirements) kept = [] removed = '' - for statement in re.split(r'(?<=[.!?])(?=\s)|(?<=\n)', text): + for statement, scoped in _unquoted_statements(text): why = '' - if _UNATTESTED_TEST_METRIC.search(statement) or _UNBOUNDED_SUCCESS.search(statement): - why = 'test counts, coverage or exhaustive correctness were not established by execution evidence' - elif (_TEST_CLAIM.search(statement) or _TEST_STATUS_CLAIM.search(statement)) and decision.status != CompletionStatus.VERIFIED: - why = 'no current passing executable verification supports the claim' - elif (_EXECUTION_CLAIM.search(statement) or _TERMINAL_SUCCESS.search(statement)) and not productive: - why = 'no successful operation supports the execution claim' - elif incomplete and _TERMINAL_SUCCESS.search(statement): - why = incomplete + for claim, scope in _current_run_claims(scoped, execution_required=execution_required): + paths = tuple(match.group().rstrip('.') for match in _CLAIM_PATH.finditer(scope)) + if claim == 'metric': + why = 'test counts, coverage or exhaustive correctness were not established by execution evidence' + elif claim == 'test': + identities = tuple(match.group().lower() for match in _TEST_IDENTITY.finditer(scope)) + if decision.status != CompletionStatus.VERIFIED or not ledger._supports_verifier_claim(identities, paths): + why = 'no current passing executable verification supports the claim' + elif claim == 'mutation': + if not ledger._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, paths): + why = 'no matching artifact mutation supports the execution claim' + elif claim == 'execution': + # A generic assertion cannot be tied confidently to a receipt. + why = 'no matching operation supports the execution claim' + elif claim == 'terminal' and decision.status not in {CompletionStatus.SATISFIED, CompletionStatus.VERIFIED}: + why = incomplete or 'no successful execution supports completion' + if why: + break if why: removed = removed or why else: kept.append(statement) prose = ''.join(kept).strip() if removed else text - if incomplete or (removed and decision.status in {CompletionStatus.UNVERIFIED, CompletionStatus.AWAITING_USER}): + if incomplete or (removed and execution_required and decision.status in {CompletionStatus.UNVERIFIED, CompletionStatus.AWAITING_USER}): reason = incomplete or removed missing = (' Missing artifacts: ' + ', '.join(decision.missing_artifacts) + '.' if decision.missing_artifacts else '') notice = 'The task is incomplete: ' + reason.rstrip('.') + '.' + missing + recorded = [path for path in ledger.requirements.required_artifacts + if ledger._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, (path,))] + if removed and recorded: + notice += ' Recorded artifact mutation: ' + ', '.join(recorded) + '.' return notice + ('\n\n' + prose if prose.strip() else ''), reason + if removed and not execution_required and decision.status != CompletionStatus.VERIFIED: + notice = 'Unsupported execution claims were omitted: ' + removed.rstrip('.') + '.' + return (prose.rstrip() + '\n\n' + notice) if prose.strip() else notice, removed if decision.can_complete and (ledger.requirements.required_artifacts or ledger.requirements.verifier_required or removed): facts = [] if ledger.requirements.required_artifacts: @@ -219,8 +283,8 @@ def with_completion_gate(func): _, unsafe_draft = completion_answer(draft, ledger, presentation_decision) if not answer.strip() and unsafe_draft: reason = reason or unsafe_draft - safe_answer = 'The task is incomplete: ' + reason.rstrip('.') + '.' - if reason and decision.can_complete and decision.status == CompletionStatus.UNVERIFIED: + safe_answer, _ = completion_answer(draft, ledger, presentation_decision) + if reason and _execution_obligation(requirements) and decision.can_complete and decision.status == CompletionStatus.UNVERIFIED: decision = CompletionDecision(CompletionStatus.UNVERIFIED, False, reason, decision.evidence_ids, decision.missing_artifacts) released_at = perf_counter() diff --git a/tests/test_agent_evidence_loop.py b/tests/test_agent_evidence_loop.py index d55337815..382ae3b26 100644 --- a/tests/test_agent_evidence_loop.py +++ b/tests/test_agent_evidence_loop.py @@ -135,6 +135,40 @@ def test_terminal_completion_missing_artifact_does_not_add_model_rounds(monkeypa assert decision["missing_artifacts"] == ["answer.json"] +def test_slice2_explanatory_request_does_not_add_verification_or_model_rounds(monkeypatch): + calls = _patch_loop(monkeypatch, ['Tests pass when the command exits zero.']) + events = _run('Explain how to write code and then test it.', max_rounds=1) + assert calls() == 1 + decision = next(event['data'] for event in events if event.get('type') == 'completion_decision') + assert decision['can_complete'] is True + assert 'The task is incomplete' not in json.dumps(events) + metrics = next(event['data'] for event in events if event.get('type') == 'metrics') + assert not metrics['completion_requirements']['verifier_required'] + assert metrics['completion_gate']['additional_provider_calls'] == 0 + + +def test_slice2_fabricated_execution_on_conversational_turn_does_not_add_rounds(monkeypatch): + calls = _patch_loop(monkeypatch, ['I ran pytest and all tests passed.']) + events = _run('Explain what pytest does.', max_rounds=1) + assert calls() == 1 + final = next(event['content'] for event in events if event.get('type') == 'final_response') + assert 'I ran pytest' not in final + assert 'The task is incomplete' not in final + metrics = next(event['data'] for event in events if event.get('type') == 'metrics') + assert metrics['completion_gate']['additional_provider_calls'] == 0 + + +def test_slice2_unsupported_test_report_does_not_add_model_rounds(monkeypatch): + calls = _patch_loop(monkeypatch, ['I ran pytest and all 42 tests passed.']) + events = _run('Run pytest.', max_rounds=4, relevant_tools={'bash'}) + assert calls() == 1 + decision = next(event['data'] for event in events if event.get('type') == 'completion_decision') + assert decision['can_complete'] is False + final = next(event['content'] for event in events if event.get('type') == 'final_response') + assert final.startswith('The task is incomplete:') + assert '42' not in final + + def test_failed_trailing_tool_with_planning_prose_continues_artifact_task(monkeypatch): calls = _patch_loop( monkeypatch, diff --git a/tests/test_completion_boundary.py b/tests/test_completion_boundary.py index e7b88c8b2..473c66d28 100644 --- a/tests/test_completion_boundary.py +++ b/tests/test_completion_boundary.py @@ -6,6 +6,8 @@ import json import pytest from src.agent_runtime.completion import with_completion_gate +from src.agent_runtime.completion import completion_answer +from src.agent_evidence import CompletionRequirements, EvidenceLedger, infer_completion_requirements from src.agent_runtime.journal import current_journal from src.tool_types import ToolBlock from tests.runtime_evidence_helpers import authoritative_executor @@ -225,3 +227,132 @@ async def test_cancellation_closes_inner_stream_without_releasing_completion(aft assert _labels(chunks) == ['tool_start'] assert closed == [True] assert current_journal() is None + + +@pytest.mark.parametrize('prose', [ + 'Tests pass when the command exits zero.', + 'If all tests are passing, merge the branch.', + 'Tests passed if the command exited zero.', + 'The documentation says "5 passed".', + 'The documentation says "Tests: FAIL" or "Tests: PASS".', + 'You can run pytest to verify this.', + 'A successful test run should show no failures.', + 'For example, I created the file and updated config.py.', + 'If I updated config.py, I would run pytest.', + 'Imagine I ran the tests and all 42 passed.', + 'Done is the label for a finished item.', + '```text\nI ran pytest and all 42 passed.\n```', + 'Run pytest until there are no failures.', +]) +def test_slice2_explanatory_prose_is_not_a_current_run_claim(prose): + ledger = EvidenceLedger() + answer, reason = completion_answer(prose, ledger, ledger.evaluate()) + assert answer == prose + assert not reason + + +@pytest.mark.parametrize('instruction', [ + 'Explain how to write code and then test it.', + 'Summarise this and check for typos.', + 'Explain how to update config.py and then verify it.', + 'Show an example of creating answer.json and checking it.', + 'The documentation says "run pytest and create answer.json".', + 'If you run pytest, the tests should pass.', +]) +def test_slice2_explanatory_request_has_no_execution_requirements(instruction): + requirements = infer_completion_requirements(instruction) + assert requirements.required_artifacts == () + assert not requirements.verifier_required + assert not requirements.executable_verifier_available + + +@pytest.mark.parametrize('instruction', [ + 'Run the tests.', 'Please run pytest.', 'Can you run the test suite?', +]) +def test_slice2_explicit_test_execution_requires_a_verifier(instruction): + requirements = infer_completion_requirements(instruction) + assert requirements.verifier_required + assert not EvidenceLedger(requirements).evaluate().can_complete + + +@pytest.mark.parametrize('claim', [ + 'I ran the tests.', 'The tests passed.', '42 tests passed.', + 'I created the file.', 'I updated config.py successfully.', +]) +def test_slice2_execution_obligation_rejects_unsupported_claims(claim): + ledger = EvidenceLedger(CompletionRequirements(required_artifacts=('config.py',))) + answer, reason = completion_answer(claim, ledger, ledger.evaluate()) + assert reason + assert answer.startswith('The task is incomplete:') + assert claim not in answer + + +@pytest.mark.parametrize('claim', [ + 'I ran pytest to see if the tests passed.', + 'I updated config.py as an example.', + 'I ran pytest and should update config.py next.', + 'config.py was updated successfully.', +]) +def test_slice2_subordinate_explanation_cannot_hide_a_direct_execution_report(claim): + ledger = EvidenceLedger() + answer, reason = completion_answer(claim, ledger, ledger.evaluate()) + assert reason + assert claim not in answer + assert 'The task is incomplete' not in answer + + +@pytest.mark.asyncio +async def test_slice2_client_dictionary_cannot_attest_execution(): + @with_completion_gate + async def stream(messages, client_runtime_context=None): + yield _event({'delta': 'I ran pytest and all tests passed.'}) + yield DONE + + context = {'execution_obligation': True, 'execution_verified': True, + 'evidence_events': [{'tool': 'bash', 'command': 'pytest', 'exit_code': 0}]} + chunks = [chunk async for chunk in stream( + [{'role': 'user', 'content': 'Explain test output.'}], client_runtime_context=context)] + final = next(data['content'] for event, data in _frames(chunks) + if event == 'message' and data.get('type') == 'final_response') + assert 'I ran pytest' not in final + assert 'The task is incomplete' not in final + assert _decision(chunks)['can_complete'] is True + + +@pytest.mark.asyncio +async def test_slice2_conversational_fabrication_is_corrected_without_execution_incomplete(): + invocations = [] + + @with_completion_gate + async def stream(messages): + invocations.append(1) + yield _event({'delta': 'The function returns a boolean. I ran pytest and all tests passed.'}) + yield _event({'type': 'metrics', 'data': {}}) + yield DONE + + chunks = [chunk async for chunk in stream([{'role': 'user', 'content': 'Explain the function.'}])] + final = next(data['content'] for event, data in _frames(chunks) + if event == 'message' and data.get('type') == 'final_response') + assert 'The function returns a boolean.' in final + assert 'I ran pytest' not in final + assert 'The task is incomplete' not in final + assert _decision(chunks)['can_complete'] is True + assert invocations == [1] + metrics = next(data['data'] for event, data in _frames(chunks) + if event == 'message' and data.get('type') == 'metrics') + assert metrics['completion_gate']['additional_provider_calls'] == 0 + + +@pytest.mark.asyncio +async def test_slice2_quoted_example_does_not_hide_an_unsupported_report(): + @with_completion_gate + async def stream(messages): + yield _event({'delta': 'The docs say "5 passed". I ran pytest.'}) + yield DONE + + chunks = [chunk async for chunk in stream([{'role': 'user', 'content': 'Explain pytest output.'}])] + final = next(data['content'] for event, data in _frames(chunks) + if event == 'message' and data.get('type') == 'final_response') + assert 'The docs say "5 passed".' in final + assert 'I ran pytest' not in final + assert 'The task is incomplete' not in final diff --git a/tests/test_runtime_evidence_contract.py b/tests/test_runtime_evidence_contract.py index ae6fc084b..3238480b9 100644 --- a/tests/test_runtime_evidence_contract.py +++ b/tests/test_runtime_evidence_contract.py @@ -128,7 +128,7 @@ def test_readback_does_not_substitute_for_required_executable_tests(): 'Test suite ran successfully', 'No failures.', 'Done.', 'I executed the command.', 'Successfully created the file.']) def test_no_execution_receipts_cannot_support_adversarial_success_claims(claim): - ledger = EvidenceLedger() + ledger = EvidenceLedger(CompletionRequirements(verifier_required=True)) answer, reason = completion_answer(claim, ledger, ledger.evaluate()) assert reason assert answer.startswith('The task is incomplete:') @@ -178,7 +178,7 @@ async def test_mixed_thinking_delta_cannot_publish_success_before_gate(thinking) yield 'data: {"type":"tool_start","tool":"bash"}\n\n' yield 'data: {"type":"metrics","data":{"thinking":"All tests passed."}}\n\n' yield 'data: [DONE]\n\n' - events = decode([chunk async for chunk in stream([])]) + events = decode([chunk async for chunk in stream([{'role': 'user', 'content': 'Run the tests.'}])]) assert events[0] == {'type': 'tool_start', 'tool': 'bash'} assert events[1]['type'] == 'completion_decision' assert not events[1]['data']['can_complete'] @@ -398,3 +398,79 @@ async def test_unknown_tool_never_creates_dispatch_identity(monkeypatch): await execute_tool_block(ToolBlock('unknown_nonexistent_tool', '{}'), security_context=NO_TOOL_SECURITY_CONTEXT) assert journal.actions[0].execution_id is None assert not journal.actions[0].outcome['authoritative'] + + +@pytest.mark.parametrize('tool,command,claim', [ + ('read_file', 'README.md', 'I ran the tests.'), + ('bash', 'printf observation', 'The tests passed.'), + ('read_file', 'README.md', 'I updated config.py.'), + ('bash', 'printf observation', 'I created the file.'), + ('write_file', '{"path":"other.py"}', 'I updated config.py.'), + ('write_file', '{"path":"nested/config.py"}', 'I updated config.py.'), + ('bash', 'python -m unittest', 'I ran pytest and the tests passed.'), + ('bash', 'pytest tests/test_other.py', 'I ran pytest tests/test_config.py.'), + ('bash', 'pytest', 'I updated config.py and the tests passed.'), + ('write_file', '{"path":"config.py"}', 'I updated config.py and ran pytest.'), + ('bash', 'python -m unittest pytest', 'I ran pytest.'), + ('write_file', '{"path":"config.py"}', 'I updated "settings.py".'), + ('bash', 'pytest', 'Created config.py and ran pytest.'), +]) +def test_slice2_unrelated_receipt_cannot_support_claim(tool, command, claim): + ledger = EvidenceLedger.from_tool_events([ + {'tool': tool, 'command': command, 'exit_code': 0}, + ]) + answer, reason = completion_answer(claim, ledger, ledger.evaluate()) + assert reason + assert claim not in answer + + +@pytest.mark.parametrize('claim', ['I ran pytest.', 'The tests passed.', 'Tests: PASS']) +def test_slice2_matching_verifier_supports_test_claim(claim): + ledger = EvidenceLedger.from_tool_events([ + {'tool': 'bash', 'command': 'python3 -m pytest -q', 'exit_code': 0}, + ]) + answer, reason = completion_answer(claim, ledger, ledger.evaluate()) + assert claim in answer + assert not reason + + +@pytest.mark.parametrize('claim', ['I updated config.py.', 'I updated `./config.py` successfully.', + 'I updated "config.py".']) +def test_slice2_matching_mutation_supports_artifact_claim(claim): + ledger = EvidenceLedger.from_tool_events([ + {'tool': 'edit_file', 'command': '{"path":"config.py"}', 'exit_code': 0}, + ], CompletionRequirements(required_artifacts=('config.py',))) + answer, reason = completion_answer(claim, ledger, ledger.evaluate()) + assert claim in answer + assert not reason + + +def test_slice2_one_matching_path_does_not_support_multiple_artifact_claims(): + ledger = EvidenceLedger.from_tool_events([ + {'tool': 'edit_file', 'command': '{"path":"config.py"}', 'exit_code': 0}, + ]) + claim = 'I updated config.py and settings.py.' + answer, reason = completion_answer(claim, ledger, ledger.evaluate()) + assert reason + assert claim not in answer + + +def test_slice2_verifier_before_mutation_cannot_support_current_test_success(): + ledger = EvidenceLedger.from_tool_events([ + {'tool': 'bash', 'command': 'pytest', 'exit_code': 0}, + {'tool': 'edit_file', 'command': '{"path":"config.py"}', 'exit_code': 0}, + ], CompletionRequirements(required_artifacts=('config.py',))) + answer, reason = completion_answer('The tests passed.', ledger, ledger.evaluate()) + assert reason + assert 'The tests passed.' not in answer + + +def test_slice2_matching_artifact_and_verifier_support_combined_claim(): + ledger = EvidenceLedger.from_tool_events([ + {'tool': 'edit_file', 'command': '{"path":"config.py"}', 'exit_code': 0}, + {'tool': 'bash', 'command': 'pytest tests/test_config.py', 'exit_code': 0}, + ], CompletionRequirements(required_artifacts=('config.py',))) + claim = 'I updated config.py and ran pytest tests/test_config.py.' + answer, reason = completion_answer(claim, ledger, ledger.evaluate()) + assert claim in answer + assert not reason