mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 06:52:20 +02:00
fix(runtime): scope completion claims to execution obligations
This commit is contained in:
+92
-6
@@ -280,6 +280,40 @@ def _known_input_is_explicit_mutation_target(instruction: str, path: str) -> boo
|
||||
)
|
||||
|
||||
|
||||
def _unquoted_statements(text: str) -> Iterable[tuple[str, str]]:
|
||||
"""Yield original statements and their reportable prose, with quotes masked.
|
||||
|
||||
Mask before splitting so punctuation inside an example cannot change the
|
||||
scope of the surrounding sentence. Inline code identifiers stay visible.
|
||||
"""
|
||||
def mask(match: re.Match[str]) -> str:
|
||||
value = match.group()
|
||||
# Quotation marks around an artifact identify a target, rather than
|
||||
# quote a report. Keep that target available for exact path matching.
|
||||
if value[0] in {'"', "'"} and re.fullmatch(_ARTIFACT_PATH, value[1:-1]):
|
||||
return ' ' + value[1:-1] + ' '
|
||||
return re.sub(r'[^\n]', ' ', value)
|
||||
|
||||
masked = re.sub(
|
||||
r'```[\s\S]*?```|~~~[\s\S]*?~~~|"[^"\n]*"|(?<!\w)\'[^\'\n]*\'(?!\w)',
|
||||
mask, text,
|
||||
)
|
||||
start = 0
|
||||
for boundary in re.finditer(r'(?<=[.!?;])(?=\s)|(?<=\n)', masked):
|
||||
end = boundary.start()
|
||||
if end > start:
|
||||
yield text[start:end], masked[start:end].replace('`', '')
|
||||
start = end
|
||||
if start < len(text):
|
||||
yield text[start:], masked[start:].replace('`', '')
|
||||
|
||||
|
||||
def _execution_obligation(requirements: CompletionRequirements) -> bool:
|
||||
"""A derived view of the existing contract, never a separate declaration."""
|
||||
return bool(requirements.required_artifacts or requirements.verifier_required
|
||||
or requirements.executable_verifier_available or requirements.verifier_commands)
|
||||
|
||||
|
||||
def infer_completion_requirements(
|
||||
instruction: str,
|
||||
*,
|
||||
@@ -289,7 +323,15 @@ def infer_completion_requirements(
|
||||
) -> CompletionRequirements:
|
||||
"""Infer only explicitly requested output/edit paths from an instruction."""
|
||||
|
||||
text = str(instruction or "")
|
||||
# Explanations can contain imperative examples. Their embedded actions
|
||||
# are not requests to execute those actions. Keep independent requests in
|
||||
# other statements, and keep explicitly supplied verifier requirements.
|
||||
explanatory_request = re.compile(
|
||||
r'^\s*(?:please\s+|(?:can|could|would)\s+you\s+)?'
|
||||
r'(?:explain|describe|summari[sz]e|teach|discuss|'
|
||||
r'show\s+(?:me\s+)?(?:an?\s+)?example|how\b)', re.I)
|
||||
text = ''.join(scoped for _, scoped in _unquoted_statements(str(instruction or ''))
|
||||
if not explanatory_request.search(scoped))
|
||||
paths: list[str] = []
|
||||
for pattern in (
|
||||
_ARTIFACT_REQUEST_RE,
|
||||
@@ -353,18 +395,23 @@ def infer_completion_requirements(
|
||||
for command in verifier_commands
|
||||
if str(command or "").strip()
|
||||
))
|
||||
explicit_test_request = re.search(
|
||||
r'(?:^|[.;\n]|\b(?:and|then))\s*'
|
||||
r'(?:please\s+|(?:can|could|would)\s+you\s+)?'
|
||||
r'(?:run|execute)\s+(?:(?:the|all|a|full)\s+)*'
|
||||
r'(?:tests?\b|test\s+suite\b|pytest\b|unittest\b|npm\s+test\b)', text, re.I)
|
||||
verifier_required = executable_verifier_available or bool(cleaned_verifier_commands) or bool(
|
||||
re.search(
|
||||
explicit_test_request or (paths and re.search(
|
||||
r"\b(?:then|after(?:wards)?|and)\b[^\n]{0,100}\b(?:test|verify|check|validate)\b",
|
||||
str(instruction or ""),
|
||||
text,
|
||||
re.IGNORECASE,
|
||||
)
|
||||
))
|
||||
)
|
||||
return CompletionRequirements(
|
||||
required_artifacts=tuple(paths),
|
||||
verifier_required=verifier_required,
|
||||
executable_verifier_available=(
|
||||
executable_verifier_available or bool(cleaned_verifier_commands)
|
||||
executable_verifier_available or bool(cleaned_verifier_commands) or bool(explicit_test_request)
|
||||
),
|
||||
verifier_commands=cleaned_verifier_commands,
|
||||
)
|
||||
@@ -575,6 +622,9 @@ class EvidenceLedger:
|
||||
self.events: list[EvidenceEvent] = []
|
||||
self._verification_versions: dict[str, str] = {}
|
||||
self._verification_versions_captured = False
|
||||
# Retain receipt command identity privately for presentation matching;
|
||||
# model prose and client dictionaries never populate this evidence.
|
||||
self._verifier_commands: dict[str, tuple[str, ...]] = {}
|
||||
|
||||
@classmethod
|
||||
def from_tool_events(
|
||||
@@ -718,13 +768,14 @@ class EvidenceLedger:
|
||||
versions = event.get('artifact_versions')
|
||||
self._verification_versions = dict(versions) if isinstance(versions, Mapping) else {}
|
||||
self._verification_versions_captured = isinstance(versions, Mapping)
|
||||
self._append(
|
||||
verifier = self._append(
|
||||
kind=EvidenceKind.VERIFIER_RESULT,
|
||||
success=success,
|
||||
authoritative=authoritative,
|
||||
source=event,
|
||||
detail="executable test/verifier command",
|
||||
)
|
||||
self._verifier_commands[verifier.event_id] = executable_words(_command_text(command))
|
||||
elif tool in {"bash", "host_shell"} and is_validation_command(command) and not mutation_paths:
|
||||
for path in self.requirements.required_artifacts:
|
||||
if _path_is_mentioned(command, path):
|
||||
@@ -736,6 +787,41 @@ class EvidenceLedger:
|
||||
artifact_path=path,
|
||||
)
|
||||
|
||||
def _supports_verifier_claim(self, identities: Sequence[str] = (), paths: Sequence[str] = ()) -> bool:
|
||||
"""Only the current passing verifier may support its named runner."""
|
||||
if self.evaluate().status != CompletionStatus.VERIFIED:
|
||||
return False
|
||||
latest = next((event for event in reversed(self.events)
|
||||
if event.kind == EvidenceKind.VERIFIER_RESULT and event.authoritative), None)
|
||||
if latest is None or not latest.success:
|
||||
return False
|
||||
words = self._verifier_commands.get(latest.event_id, ())
|
||||
names = {Path(words[0]).name} if words else set()
|
||||
if words and re.fullmatch(r'python(?:\d+(?:\.\d+)*)?', Path(words[0]).name) and '-m' in words:
|
||||
module_index = words.index('-m') + 1
|
||||
if module_index < len(words):
|
||||
names.add(words[module_index])
|
||||
return (all(identity in names for identity in identities)
|
||||
and all(any(_artifact_path_matches_required(word, path, self.requirements.workspace_root)
|
||||
for word in words) for path in paths))
|
||||
|
||||
def _supports_artifact_claim(self, kind: EvidenceKind, paths: Sequence[str]) -> bool:
|
||||
"""Match every claimed artifact by identity, never by basename."""
|
||||
targets = tuple(paths) or self.requirements.required_artifacts
|
||||
if not targets or (not paths and len(targets) != 1):
|
||||
return False
|
||||
for path in targets:
|
||||
matching = [event for event in self.events if event.kind == kind and event.authoritative
|
||||
and _artifact_path_matches_required(event.artifact_path, path, self.requirements.workspace_root)]
|
||||
successful = [event for event in matching if event.success]
|
||||
# Match evaluate(): atomic helper failures preserve the previous
|
||||
# successful artifact; a partial shell/Python failure may not.
|
||||
destructive_failure = bool(matching and not matching[-1].success
|
||||
and matching[-1].tool in {'bash', 'python'})
|
||||
if not successful or destructive_failure:
|
||||
return False
|
||||
return True
|
||||
|
||||
def record_media_ingress(self, metadata: Mapping[str, Any]) -> None:
|
||||
for artifact in metadata.get("artifacts") or []:
|
||||
if not isinstance(artifact, Mapping):
|
||||
|
||||
@@ -17,7 +17,8 @@ from time import perf_counter
|
||||
|
||||
from src.agent_evidence import (
|
||||
CompletionDecision, CompletionStatus, EvidenceKind, EvidenceLedger,
|
||||
requirements_from_runtime_context,
|
||||
requirements_from_runtime_context, _execution_obligation, _unquoted_statements,
|
||||
_ARTIFACT_PATH,
|
||||
)
|
||||
from .journal import ActionJournal, bind_journal, current_journal
|
||||
|
||||
@@ -30,10 +31,9 @@ _TEST_STATUS_CLAIM = re.compile(
|
||||
r'(?:all\s+|have\s+|has\s+|now\s+|are\s+|is\s+|ran\s+)*'
|
||||
r'(?:pass(?:ed|ing)?|succeeded|successful(?:ly)?|green)\b|'
|
||||
r'\b(?:zero|no|0)\s+(?:test\s+)?failures\b', re.I)
|
||||
_TERMINAL_SUCCESS = re.compile(r'^\s*(?:done|completed|success|all done|all set|fixed)\b', re.I)
|
||||
_EXECUTION_CLAIM = re.compile(
|
||||
r'\b(?:(?:I|we|I\'ve|we\'ve)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed)|'
|
||||
r'(?:file|artifact|command|script|service|server)\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started)|'
|
||||
r'\b(?:(?:I|we|I\'ve|we\'ve|and)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed)|'
|
||||
rf'(?:file|artifact|command|script|service|server|{_ARTIFACT_PATH})\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started)|'
|
||||
r'(?:successfully\s+)(?:ran|executed|created|updated|saved|completed))\b', re.I)
|
||||
_UNATTESTED_TEST_METRIC = re.compile(
|
||||
r'\b\d+\s+(?:(?:unit|integration)\s+)?tests?\s+pass(?:ed|ing)?\b|'
|
||||
@@ -41,6 +41,55 @@ _UNATTESTED_TEST_METRIC = re.compile(
|
||||
_UNBOUNDED_SUCCESS = re.compile(
|
||||
r'\b(?:everything|all\s+(?:bugs|issues))\s+(?:is\s+|are\s+|has\s+been\s+)?'
|
||||
r'(?:fixed|resolved|working)\b', re.I)
|
||||
_MUTATION_CLAIM = re.compile(
|
||||
r'\b(?:created|updated|modified|wrote|written|saved|fixed)\b', re.I)
|
||||
_TEST_IDENTITY = re.compile(r'\b(?:pytest|unittest)\b', re.I)
|
||||
_TEST_SUBJECT = re.compile(r'\b(?:tests?|test suite|pytest|unittest|checks?|verification)\b', re.I)
|
||||
_CLAIM_PATH = re.compile(_ARTIFACT_PATH)
|
||||
_BARE_SUCCESS = re.compile(r'^\s*(?:done|completed|success|all done|all set|fixed)[.!]?\s*$', re.I)
|
||||
_NON_REPORT_SCOPE = re.compile(
|
||||
r'^\s*(?:if|unless|suppose|imagine|hypothetically|for\s+(?:example|instance))\b|'
|
||||
r'\b(?:if|when|whenever|unless|until)\b|'
|
||||
r'\b(?:can|could|may|might|should|would|will|must)\b|'
|
||||
r'\b(?:says?|said|states?|stated|example)\b', re.I)
|
||||
|
||||
|
||||
def _current_run_claims(statement: str, *, execution_required: bool) -> list[tuple[str, str]]:
|
||||
"""Classify asserted execution, separately from the turn's obligation.
|
||||
|
||||
Past actions and current result/status predicates are reports. Conditional,
|
||||
modal, attributed and example clauses are scoped prose. Bare terminal
|
||||
success only carries execution meaning under an execution contract.
|
||||
"""
|
||||
if _BARE_SUCCESS.fullmatch(statement):
|
||||
return [('terminal', statement)] if execution_required else []
|
||||
actions = list(_EXECUTION_CLAIM.finditer(statement))
|
||||
leading = re.match(r'^\s*(?:successfully\s+)?(?:created|updated|modified|wrote|saved)\b', statement, re.I)
|
||||
if leading:
|
||||
actions.insert(0, leading)
|
||||
candidates = [('action', match) for match in actions]
|
||||
for kind, pattern in [('metric', _UNATTESTED_TEST_METRIC), ('metric', _UNBOUNDED_SUCCESS),
|
||||
('test', _TEST_CLAIM), ('test', _TEST_STATUS_CLAIM)]:
|
||||
candidates.extend((kind, match) for match in pattern.finditer(statement))
|
||||
claims = []
|
||||
for kind, match in candidates:
|
||||
# Scope markers after an asserted action do not make that action
|
||||
# hypothetical ("I ran pytest to see if ..."). An immediate conditional
|
||||
# continuation does qualify a result ("Tests passed if ...").
|
||||
if _NON_REPORT_SCOPE.search(statement[:match.start()]) or re.match(
|
||||
r'\s+(?:if|when|whenever|unless|until)\b', statement[match.end():], re.I):
|
||||
continue
|
||||
end = next((action.start() for action in actions if action.start() > match.start()), len(statement))
|
||||
scope = statement[match.start():end]
|
||||
if kind == 'action':
|
||||
if _MUTATION_CLAIM.search(match.group()):
|
||||
kind = 'mutation'
|
||||
elif _TEST_SUBJECT.search(scope):
|
||||
kind = 'test'
|
||||
else:
|
||||
kind = 'execution'
|
||||
claims.append((kind, scope))
|
||||
return claims
|
||||
|
||||
|
||||
def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDecision) -> tuple[str, str]:
|
||||
@@ -51,32 +100,47 @@ def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDec
|
||||
The execution outcome remains separate from a discarded model assertion.
|
||||
"""
|
||||
incomplete = decision.reason if not decision.can_complete and decision.status != CompletionStatus.AWAITING_USER else ''
|
||||
productive = [event for event in ledger.events
|
||||
if event.authoritative and event.success
|
||||
and event.tool not in {'update_plan', 'todowrite', 'ask_user'}]
|
||||
execution_required = _execution_obligation(ledger.requirements)
|
||||
kept = []
|
||||
removed = ''
|
||||
for statement in re.split(r'(?<=[.!?])(?=\s)|(?<=\n)', text):
|
||||
for statement, scoped in _unquoted_statements(text):
|
||||
why = ''
|
||||
if _UNATTESTED_TEST_METRIC.search(statement) or _UNBOUNDED_SUCCESS.search(statement):
|
||||
why = 'test counts, coverage or exhaustive correctness were not established by execution evidence'
|
||||
elif (_TEST_CLAIM.search(statement) or _TEST_STATUS_CLAIM.search(statement)) and decision.status != CompletionStatus.VERIFIED:
|
||||
why = 'no current passing executable verification supports the claim'
|
||||
elif (_EXECUTION_CLAIM.search(statement) or _TERMINAL_SUCCESS.search(statement)) and not productive:
|
||||
why = 'no successful operation supports the execution claim'
|
||||
elif incomplete and _TERMINAL_SUCCESS.search(statement):
|
||||
why = incomplete
|
||||
for claim, scope in _current_run_claims(scoped, execution_required=execution_required):
|
||||
paths = tuple(match.group().rstrip('.') for match in _CLAIM_PATH.finditer(scope))
|
||||
if claim == 'metric':
|
||||
why = 'test counts, coverage or exhaustive correctness were not established by execution evidence'
|
||||
elif claim == 'test':
|
||||
identities = tuple(match.group().lower() for match in _TEST_IDENTITY.finditer(scope))
|
||||
if decision.status != CompletionStatus.VERIFIED or not ledger._supports_verifier_claim(identities, paths):
|
||||
why = 'no current passing executable verification supports the claim'
|
||||
elif claim == 'mutation':
|
||||
if not ledger._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, paths):
|
||||
why = 'no matching artifact mutation supports the execution claim'
|
||||
elif claim == 'execution':
|
||||
# A generic assertion cannot be tied confidently to a receipt.
|
||||
why = 'no matching operation supports the execution claim'
|
||||
elif claim == 'terminal' and decision.status not in {CompletionStatus.SATISFIED, CompletionStatus.VERIFIED}:
|
||||
why = incomplete or 'no successful execution supports completion'
|
||||
if why:
|
||||
break
|
||||
if why:
|
||||
removed = removed or why
|
||||
else:
|
||||
kept.append(statement)
|
||||
prose = ''.join(kept).strip() if removed else text
|
||||
if incomplete or (removed and decision.status in {CompletionStatus.UNVERIFIED, CompletionStatus.AWAITING_USER}):
|
||||
if incomplete or (removed and execution_required and decision.status in {CompletionStatus.UNVERIFIED, CompletionStatus.AWAITING_USER}):
|
||||
reason = incomplete or removed
|
||||
missing = (' Missing artifacts: ' + ', '.join(decision.missing_artifacts) + '.'
|
||||
if decision.missing_artifacts else '')
|
||||
notice = 'The task is incomplete: ' + reason.rstrip('.') + '.' + missing
|
||||
recorded = [path for path in ledger.requirements.required_artifacts
|
||||
if ledger._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, (path,))]
|
||||
if removed and recorded:
|
||||
notice += ' Recorded artifact mutation: ' + ', '.join(recorded) + '.'
|
||||
return notice + ('\n\n' + prose if prose.strip() else ''), reason
|
||||
if removed and not execution_required and decision.status != CompletionStatus.VERIFIED:
|
||||
notice = 'Unsupported execution claims were omitted: ' + removed.rstrip('.') + '.'
|
||||
return (prose.rstrip() + '\n\n' + notice) if prose.strip() else notice, removed
|
||||
if decision.can_complete and (ledger.requirements.required_artifacts or ledger.requirements.verifier_required or removed):
|
||||
facts = []
|
||||
if ledger.requirements.required_artifacts:
|
||||
@@ -219,8 +283,8 @@ def with_completion_gate(func):
|
||||
_, unsafe_draft = completion_answer(draft, ledger, presentation_decision)
|
||||
if not answer.strip() and unsafe_draft:
|
||||
reason = reason or unsafe_draft
|
||||
safe_answer = 'The task is incomplete: ' + reason.rstrip('.') + '.'
|
||||
if reason and decision.can_complete and decision.status == CompletionStatus.UNVERIFIED:
|
||||
safe_answer, _ = completion_answer(draft, ledger, presentation_decision)
|
||||
if reason and _execution_obligation(requirements) and decision.can_complete and decision.status == CompletionStatus.UNVERIFIED:
|
||||
decision = CompletionDecision(CompletionStatus.UNVERIFIED, False, reason,
|
||||
decision.evidence_ids, decision.missing_artifacts)
|
||||
released_at = perf_counter()
|
||||
|
||||
@@ -135,6 +135,40 @@ def test_terminal_completion_missing_artifact_does_not_add_model_rounds(monkeypa
|
||||
assert decision["missing_artifacts"] == ["answer.json"]
|
||||
|
||||
|
||||
def test_slice2_explanatory_request_does_not_add_verification_or_model_rounds(monkeypatch):
|
||||
calls = _patch_loop(monkeypatch, ['Tests pass when the command exits zero.'])
|
||||
events = _run('Explain how to write code and then test it.', max_rounds=1)
|
||||
assert calls() == 1
|
||||
decision = next(event['data'] for event in events if event.get('type') == 'completion_decision')
|
||||
assert decision['can_complete'] is True
|
||||
assert 'The task is incomplete' not in json.dumps(events)
|
||||
metrics = next(event['data'] for event in events if event.get('type') == 'metrics')
|
||||
assert not metrics['completion_requirements']['verifier_required']
|
||||
assert metrics['completion_gate']['additional_provider_calls'] == 0
|
||||
|
||||
|
||||
def test_slice2_fabricated_execution_on_conversational_turn_does_not_add_rounds(monkeypatch):
|
||||
calls = _patch_loop(monkeypatch, ['I ran pytest and all tests passed.'])
|
||||
events = _run('Explain what pytest does.', max_rounds=1)
|
||||
assert calls() == 1
|
||||
final = next(event['content'] for event in events if event.get('type') == 'final_response')
|
||||
assert 'I ran pytest' not in final
|
||||
assert 'The task is incomplete' not in final
|
||||
metrics = next(event['data'] for event in events if event.get('type') == 'metrics')
|
||||
assert metrics['completion_gate']['additional_provider_calls'] == 0
|
||||
|
||||
|
||||
def test_slice2_unsupported_test_report_does_not_add_model_rounds(monkeypatch):
|
||||
calls = _patch_loop(monkeypatch, ['I ran pytest and all 42 tests passed.'])
|
||||
events = _run('Run pytest.', max_rounds=4, relevant_tools={'bash'})
|
||||
assert calls() == 1
|
||||
decision = next(event['data'] for event in events if event.get('type') == 'completion_decision')
|
||||
assert decision['can_complete'] is False
|
||||
final = next(event['content'] for event in events if event.get('type') == 'final_response')
|
||||
assert final.startswith('The task is incomplete:')
|
||||
assert '42' not in final
|
||||
|
||||
|
||||
def test_failed_trailing_tool_with_planning_prose_continues_artifact_task(monkeypatch):
|
||||
calls = _patch_loop(
|
||||
monkeypatch,
|
||||
|
||||
@@ -6,6 +6,8 @@ import json
|
||||
import pytest
|
||||
|
||||
from src.agent_runtime.completion import with_completion_gate
|
||||
from src.agent_runtime.completion import completion_answer
|
||||
from src.agent_evidence import CompletionRequirements, EvidenceLedger, infer_completion_requirements
|
||||
from src.agent_runtime.journal import current_journal
|
||||
from src.tool_types import ToolBlock
|
||||
from tests.runtime_evidence_helpers import authoritative_executor
|
||||
@@ -225,3 +227,132 @@ async def test_cancellation_closes_inner_stream_without_releasing_completion(aft
|
||||
assert _labels(chunks) == ['tool_start']
|
||||
assert closed == [True]
|
||||
assert current_journal() is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize('prose', [
|
||||
'Tests pass when the command exits zero.',
|
||||
'If all tests are passing, merge the branch.',
|
||||
'Tests passed if the command exited zero.',
|
||||
'The documentation says "5 passed".',
|
||||
'The documentation says "Tests: FAIL" or "Tests: PASS".',
|
||||
'You can run pytest to verify this.',
|
||||
'A successful test run should show no failures.',
|
||||
'For example, I created the file and updated config.py.',
|
||||
'If I updated config.py, I would run pytest.',
|
||||
'Imagine I ran the tests and all 42 passed.',
|
||||
'Done is the label for a finished item.',
|
||||
'```text\nI ran pytest and all 42 passed.\n```',
|
||||
'Run pytest until there are no failures.',
|
||||
])
|
||||
def test_slice2_explanatory_prose_is_not_a_current_run_claim(prose):
|
||||
ledger = EvidenceLedger()
|
||||
answer, reason = completion_answer(prose, ledger, ledger.evaluate())
|
||||
assert answer == prose
|
||||
assert not reason
|
||||
|
||||
|
||||
@pytest.mark.parametrize('instruction', [
|
||||
'Explain how to write code and then test it.',
|
||||
'Summarise this and check for typos.',
|
||||
'Explain how to update config.py and then verify it.',
|
||||
'Show an example of creating answer.json and checking it.',
|
||||
'The documentation says "run pytest and create answer.json".',
|
||||
'If you run pytest, the tests should pass.',
|
||||
])
|
||||
def test_slice2_explanatory_request_has_no_execution_requirements(instruction):
|
||||
requirements = infer_completion_requirements(instruction)
|
||||
assert requirements.required_artifacts == ()
|
||||
assert not requirements.verifier_required
|
||||
assert not requirements.executable_verifier_available
|
||||
|
||||
|
||||
@pytest.mark.parametrize('instruction', [
|
||||
'Run the tests.', 'Please run pytest.', 'Can you run the test suite?',
|
||||
])
|
||||
def test_slice2_explicit_test_execution_requires_a_verifier(instruction):
|
||||
requirements = infer_completion_requirements(instruction)
|
||||
assert requirements.verifier_required
|
||||
assert not EvidenceLedger(requirements).evaluate().can_complete
|
||||
|
||||
|
||||
@pytest.mark.parametrize('claim', [
|
||||
'I ran the tests.', 'The tests passed.', '42 tests passed.',
|
||||
'I created the file.', 'I updated config.py successfully.',
|
||||
])
|
||||
def test_slice2_execution_obligation_rejects_unsupported_claims(claim):
|
||||
ledger = EvidenceLedger(CompletionRequirements(required_artifacts=('config.py',)))
|
||||
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
|
||||
assert reason
|
||||
assert answer.startswith('The task is incomplete:')
|
||||
assert claim not in answer
|
||||
|
||||
|
||||
@pytest.mark.parametrize('claim', [
|
||||
'I ran pytest to see if the tests passed.',
|
||||
'I updated config.py as an example.',
|
||||
'I ran pytest and should update config.py next.',
|
||||
'config.py was updated successfully.',
|
||||
])
|
||||
def test_slice2_subordinate_explanation_cannot_hide_a_direct_execution_report(claim):
|
||||
ledger = EvidenceLedger()
|
||||
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
|
||||
assert reason
|
||||
assert claim not in answer
|
||||
assert 'The task is incomplete' not in answer
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_slice2_client_dictionary_cannot_attest_execution():
|
||||
@with_completion_gate
|
||||
async def stream(messages, client_runtime_context=None):
|
||||
yield _event({'delta': 'I ran pytest and all tests passed.'})
|
||||
yield DONE
|
||||
|
||||
context = {'execution_obligation': True, 'execution_verified': True,
|
||||
'evidence_events': [{'tool': 'bash', 'command': 'pytest', 'exit_code': 0}]}
|
||||
chunks = [chunk async for chunk in stream(
|
||||
[{'role': 'user', 'content': 'Explain test output.'}], client_runtime_context=context)]
|
||||
final = next(data['content'] for event, data in _frames(chunks)
|
||||
if event == 'message' and data.get('type') == 'final_response')
|
||||
assert 'I ran pytest' not in final
|
||||
assert 'The task is incomplete' not in final
|
||||
assert _decision(chunks)['can_complete'] is True
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_slice2_conversational_fabrication_is_corrected_without_execution_incomplete():
|
||||
invocations = []
|
||||
|
||||
@with_completion_gate
|
||||
async def stream(messages):
|
||||
invocations.append(1)
|
||||
yield _event({'delta': 'The function returns a boolean. I ran pytest and all tests passed.'})
|
||||
yield _event({'type': 'metrics', 'data': {}})
|
||||
yield DONE
|
||||
|
||||
chunks = [chunk async for chunk in stream([{'role': 'user', 'content': 'Explain the function.'}])]
|
||||
final = next(data['content'] for event, data in _frames(chunks)
|
||||
if event == 'message' and data.get('type') == 'final_response')
|
||||
assert 'The function returns a boolean.' in final
|
||||
assert 'I ran pytest' not in final
|
||||
assert 'The task is incomplete' not in final
|
||||
assert _decision(chunks)['can_complete'] is True
|
||||
assert invocations == [1]
|
||||
metrics = next(data['data'] for event, data in _frames(chunks)
|
||||
if event == 'message' and data.get('type') == 'metrics')
|
||||
assert metrics['completion_gate']['additional_provider_calls'] == 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_slice2_quoted_example_does_not_hide_an_unsupported_report():
|
||||
@with_completion_gate
|
||||
async def stream(messages):
|
||||
yield _event({'delta': 'The docs say "5 passed". I ran pytest.'})
|
||||
yield DONE
|
||||
|
||||
chunks = [chunk async for chunk in stream([{'role': 'user', 'content': 'Explain pytest output.'}])]
|
||||
final = next(data['content'] for event, data in _frames(chunks)
|
||||
if event == 'message' and data.get('type') == 'final_response')
|
||||
assert 'The docs say "5 passed".' in final
|
||||
assert 'I ran pytest' not in final
|
||||
assert 'The task is incomplete' not in final
|
||||
|
||||
@@ -128,7 +128,7 @@ def test_readback_does_not_substitute_for_required_executable_tests():
|
||||
'Test suite ran successfully', 'No failures.', 'Done.',
|
||||
'I executed the command.', 'Successfully created the file.'])
|
||||
def test_no_execution_receipts_cannot_support_adversarial_success_claims(claim):
|
||||
ledger = EvidenceLedger()
|
||||
ledger = EvidenceLedger(CompletionRequirements(verifier_required=True))
|
||||
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
|
||||
assert reason
|
||||
assert answer.startswith('The task is incomplete:')
|
||||
@@ -178,7 +178,7 @@ async def test_mixed_thinking_delta_cannot_publish_success_before_gate(thinking)
|
||||
yield 'data: {"type":"tool_start","tool":"bash"}\n\n'
|
||||
yield 'data: {"type":"metrics","data":{"thinking":"All tests passed."}}\n\n'
|
||||
yield 'data: [DONE]\n\n'
|
||||
events = decode([chunk async for chunk in stream([])])
|
||||
events = decode([chunk async for chunk in stream([{'role': 'user', 'content': 'Run the tests.'}])])
|
||||
assert events[0] == {'type': 'tool_start', 'tool': 'bash'}
|
||||
assert events[1]['type'] == 'completion_decision'
|
||||
assert not events[1]['data']['can_complete']
|
||||
@@ -398,3 +398,79 @@ async def test_unknown_tool_never_creates_dispatch_identity(monkeypatch):
|
||||
await execute_tool_block(ToolBlock('unknown_nonexistent_tool', '{}'), security_context=NO_TOOL_SECURITY_CONTEXT)
|
||||
assert journal.actions[0].execution_id is None
|
||||
assert not journal.actions[0].outcome['authoritative']
|
||||
|
||||
|
||||
@pytest.mark.parametrize('tool,command,claim', [
|
||||
('read_file', 'README.md', 'I ran the tests.'),
|
||||
('bash', 'printf observation', 'The tests passed.'),
|
||||
('read_file', 'README.md', 'I updated config.py.'),
|
||||
('bash', 'printf observation', 'I created the file.'),
|
||||
('write_file', '{"path":"other.py"}', 'I updated config.py.'),
|
||||
('write_file', '{"path":"nested/config.py"}', 'I updated config.py.'),
|
||||
('bash', 'python -m unittest', 'I ran pytest and the tests passed.'),
|
||||
('bash', 'pytest tests/test_other.py', 'I ran pytest tests/test_config.py.'),
|
||||
('bash', 'pytest', 'I updated config.py and the tests passed.'),
|
||||
('write_file', '{"path":"config.py"}', 'I updated config.py and ran pytest.'),
|
||||
('bash', 'python -m unittest pytest', 'I ran pytest.'),
|
||||
('write_file', '{"path":"config.py"}', 'I updated "settings.py".'),
|
||||
('bash', 'pytest', 'Created config.py and ran pytest.'),
|
||||
])
|
||||
def test_slice2_unrelated_receipt_cannot_support_claim(tool, command, claim):
|
||||
ledger = EvidenceLedger.from_tool_events([
|
||||
{'tool': tool, 'command': command, 'exit_code': 0},
|
||||
])
|
||||
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
|
||||
assert reason
|
||||
assert claim not in answer
|
||||
|
||||
|
||||
@pytest.mark.parametrize('claim', ['I ran pytest.', 'The tests passed.', 'Tests: PASS'])
|
||||
def test_slice2_matching_verifier_supports_test_claim(claim):
|
||||
ledger = EvidenceLedger.from_tool_events([
|
||||
{'tool': 'bash', 'command': 'python3 -m pytest -q', 'exit_code': 0},
|
||||
])
|
||||
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
|
||||
assert claim in answer
|
||||
assert not reason
|
||||
|
||||
|
||||
@pytest.mark.parametrize('claim', ['I updated config.py.', 'I updated `./config.py` successfully.',
|
||||
'I updated "config.py".'])
|
||||
def test_slice2_matching_mutation_supports_artifact_claim(claim):
|
||||
ledger = EvidenceLedger.from_tool_events([
|
||||
{'tool': 'edit_file', 'command': '{"path":"config.py"}', 'exit_code': 0},
|
||||
], CompletionRequirements(required_artifacts=('config.py',)))
|
||||
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
|
||||
assert claim in answer
|
||||
assert not reason
|
||||
|
||||
|
||||
def test_slice2_one_matching_path_does_not_support_multiple_artifact_claims():
|
||||
ledger = EvidenceLedger.from_tool_events([
|
||||
{'tool': 'edit_file', 'command': '{"path":"config.py"}', 'exit_code': 0},
|
||||
])
|
||||
claim = 'I updated config.py and settings.py.'
|
||||
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
|
||||
assert reason
|
||||
assert claim not in answer
|
||||
|
||||
|
||||
def test_slice2_verifier_before_mutation_cannot_support_current_test_success():
|
||||
ledger = EvidenceLedger.from_tool_events([
|
||||
{'tool': 'bash', 'command': 'pytest', 'exit_code': 0},
|
||||
{'tool': 'edit_file', 'command': '{"path":"config.py"}', 'exit_code': 0},
|
||||
], CompletionRequirements(required_artifacts=('config.py',)))
|
||||
answer, reason = completion_answer('The tests passed.', ledger, ledger.evaluate())
|
||||
assert reason
|
||||
assert 'The tests passed.' not in answer
|
||||
|
||||
|
||||
def test_slice2_matching_artifact_and_verifier_support_combined_claim():
|
||||
ledger = EvidenceLedger.from_tool_events([
|
||||
{'tool': 'edit_file', 'command': '{"path":"config.py"}', 'exit_code': 0},
|
||||
{'tool': 'bash', 'command': 'pytest tests/test_config.py', 'exit_code': 0},
|
||||
], CompletionRequirements(required_artifacts=('config.py',)))
|
||||
claim = 'I updated config.py and ran pytest tests/test_config.py.'
|
||||
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
|
||||
assert claim in answer
|
||||
assert not reason
|
||||
|
||||
Reference in New Issue
Block a user