mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 15:02:20 +02:00
- Headless consumers (task scheduler, background follow-up) now treat a completion-gate final_response as the authoritative answer instead of collecting deltas only. A gated replacement no longer leaves scheduled output empty, which used to trigger an extra, ungated grace-summary model call. - The scheduler closes the agent stream with contextlib.aclosing, so the approval-pause break unwinds the gate's journal and teacher-takeover context in its own task. Chained runs no longer inherit a stale parent_run_id, and later finalization no longer raises ContextVar reset errors. - On provider error, the completion gate applies the live answer's statement filter to persisted round_texts. Diagnostics and the failure note survive; claims rejected by the gate cannot reappear on reload.
458 lines
19 KiB
Python
458 lines
19 KiB
Python
"""Provider failure is the final frame, after gated output and diagnostics."""
|
|
import asyncio
|
|
from inspect import signature
|
|
import json
|
|
|
|
import pytest
|
|
|
|
from src.agent_runtime.completion import with_completion_gate
|
|
from src.agent_runtime.completion import completion_answer
|
|
from src.agent_evidence import CompletionRequirements, EvidenceLedger, infer_completion_requirements
|
|
from src.agent_runtime.journal import current_journal
|
|
from src.tool_types import ToolBlock
|
|
from tests.runtime_evidence_helpers import authoritative_executor
|
|
|
|
|
|
ERROR = 'event: error\ndata: {"status": 504, "error": {"message": "stream timeout"}, "fallback_eligible": false}\n\n'
|
|
DONE = 'data: [DONE]\n\n'
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize('terminal_texts', [[], ['', ''], ['Recovered answer with literal [DONE] text.']])
|
|
async def test_terminal_round_retraction_does_not_resurrect_buffered_drafts(terminal_texts):
|
|
@with_completion_gate
|
|
async def stream(messages):
|
|
yield _event({'delta': 'Considering the next step.', 'thinking': True})
|
|
yield _event({'delta': 'Now I need to execute the rejected draft.'})
|
|
yield _event({'type': 'metrics', 'data': {'round_texts': terminal_texts}})
|
|
yield DONE
|
|
|
|
chunks = [chunk async for chunk in stream([{'role': 'user', 'content': 'Create answer.txt.'}])]
|
|
assert 'rejected draft' not in ''.join(chunks)
|
|
assert any('Considering the next step.' in chunk for chunk in chunks)
|
|
if terminal_texts and terminal_texts[0]:
|
|
assert any(terminal_texts[0] in chunk for chunk in chunks)
|
|
assert chunks.count(DONE) == 1
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_terminal_round_text_cannot_override_an_explicit_final_response():
|
|
@with_completion_gate
|
|
async def stream(messages):
|
|
yield _event({'type': 'final_response', 'content': 'The explicit final answer.'})
|
|
yield _event({'type': 'metrics', 'data': {'round_texts': ['Earlier draft.']}})
|
|
yield DONE
|
|
|
|
chunks = [chunk async for chunk in stream([])]
|
|
final = next(data for _, data in _frames(chunks) if isinstance(data, dict) and data.get('type') == 'final_response')
|
|
assert final['content'] == 'The explicit final answer.'
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_revised_terminal_prose_still_cannot_attest_execution():
|
|
@with_completion_gate
|
|
async def stream(messages):
|
|
yield _event({'delta': 'Earlier draft.'})
|
|
yield _event({'type': 'metrics', 'data': {'round_texts': ['All tests passed.']}})
|
|
yield DONE
|
|
|
|
chunks = [chunk async for chunk in stream([{'role': 'user', 'content': 'Create answer.txt and run the tests.'}])]
|
|
assert not _decision(chunks)['can_complete']
|
|
assert 'All tests passed.' not in ''.join(chunks)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_provider_error_preserves_partial_content_despite_empty_terminal_rounds():
|
|
@with_completion_gate
|
|
async def stream(messages):
|
|
yield _event({'type': 'tool_start', 'tool': 'read_file'})
|
|
yield _event({'delta': 'Safe partial result.'})
|
|
yield _event({'type': 'metrics', 'data': {'round_texts': []}})
|
|
yield ERROR
|
|
yield DONE
|
|
|
|
chunks = [chunk async for chunk in stream([])]
|
|
assert _labels(chunks) == ['tool_start', 'final_response', 'completion_decision', 'metrics', 'error']
|
|
assert any('Safe partial result.' in chunk for chunk in chunks)
|
|
assert chunks[-1] == ERROR
|
|
assert DONE not in chunks
|
|
|
|
|
|
def _event(payload):
|
|
return 'data: ' + json.dumps(payload) + '\n\n'
|
|
|
|
|
|
def _frames(chunks):
|
|
"""Decode network chunks without losing named error frames or [DONE]."""
|
|
pending = ''
|
|
for chunk in chunks:
|
|
pending += chunk
|
|
while '\n\n' in pending:
|
|
frame, pending = pending.split('\n\n', 1)
|
|
lines = frame.splitlines()
|
|
event = next((line[7:] for line in lines if line.startswith('event: ')), 'message')
|
|
payload = '\n'.join(line[6:] for line in lines if line.startswith('data: '))
|
|
yield event, payload if payload == '[DONE]' else json.loads(payload)
|
|
assert not pending, 'incomplete SSE frame'
|
|
|
|
|
|
def _labels(chunks):
|
|
return [event if event != 'message' else (
|
|
'done' if data == '[DONE]' else data.get('type', 'delta')
|
|
) for event, data in _frames(chunks)]
|
|
|
|
|
|
def _decision(chunks):
|
|
return next(data['data'] for event, data in _frames(chunks)
|
|
if event == 'message' and isinstance(data, dict)
|
|
and data.get('type') == 'completion_decision')
|
|
|
|
|
|
@authoritative_executor
|
|
async def _successful_tool(block):
|
|
return block.tool_type, {'exit_code': 0, 'output': 'OK'}
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_bare_error_preserves_original_frame_without_success_output():
|
|
@with_completion_gate
|
|
async def stream(messages):
|
|
yield ERROR
|
|
yield DONE
|
|
|
|
assert [chunk async for chunk in stream([])] == [ERROR]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize('partial', ['', 'The parser checks the header first.'])
|
|
async def test_provider_error_releases_partial_then_decision_terminal_and_original_error(partial):
|
|
closed = []
|
|
|
|
@with_completion_gate
|
|
async def stream(messages):
|
|
try:
|
|
yield _event({'type': 'tool_start', 'tool': 'read_file'})
|
|
if partial:
|
|
yield _event({'delta': partial})
|
|
yield ERROR
|
|
yield _event({'type': 'agent_terminal', 'data': {
|
|
'failed': True, 'failure': {'status': 504},
|
|
'round_texts': ['Earlier diagnostic', partial + '\n[Agent stopped]'],
|
|
}})
|
|
yield DONE
|
|
finally:
|
|
closed.append(current_journal() is not None)
|
|
|
|
chunks = [chunk async for chunk in stream([])]
|
|
assert _labels(chunks) == [
|
|
'tool_start', 'final_response', 'completion_decision', 'agent_terminal', 'error',
|
|
], _labels(chunks)
|
|
assert chunks[-1] == ERROR
|
|
assert DONE not in chunks
|
|
assert _decision(chunks)['can_complete'] is False
|
|
assert _decision(chunks)['status'] == 'failed'
|
|
final = next(data for event, data in _frames(chunks)
|
|
if event == 'message' and data.get('type') == 'final_response')
|
|
assert final['content'].startswith('The task is incomplete:')
|
|
assert partial in final['content']
|
|
assert closed == [True]
|
|
assert current_journal() is None
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize('successful_tool', [False, True])
|
|
@pytest.mark.parametrize('earlier_status', [None, 'awaiting_user', 'exhausted'])
|
|
async def test_provider_failure_overrides_even_successful_execution(successful_tool, earlier_status):
|
|
@with_completion_gate
|
|
async def stream(messages):
|
|
if successful_tool:
|
|
await _successful_tool(ToolBlock('bash', 'python -m unittest'))
|
|
if earlier_status:
|
|
yield _event({'type': 'completion_decision', 'data': {'status': earlier_status}})
|
|
yield _event({'delta': 'The response is partial.'})
|
|
yield ERROR
|
|
yield _event({'type': 'metrics', 'data': {}})
|
|
|
|
chunks = [chunk async for chunk in stream([])]
|
|
decision = _decision(chunks)
|
|
assert decision['can_complete'] is False, decision
|
|
assert decision['status'] == 'failed'
|
|
if successful_tool:
|
|
metrics = next(data['data'] for event, data in _frames(chunks)
|
|
if event == 'message' and data.get('type') == 'metrics')
|
|
assert any(e['authoritative'] and e['success'] for e in metrics['evidence_events'])
|
|
assert chunks[-1] == ERROR
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_error_after_final_response_does_not_add_calls_or_success_done():
|
|
invocations = []
|
|
|
|
@with_completion_gate
|
|
async def stream(messages, workspace=None, client_runtime_context=None):
|
|
invocations.append(1)
|
|
yield _event({'type': 'final_response', 'content': 'The header contains three fields.'})
|
|
yield DONE
|
|
yield ERROR
|
|
|
|
chunks = [chunk async for chunk in stream([])]
|
|
assert _labels(chunks) == ['final_response', 'completion_decision', 'error']
|
|
assert invocations == [1]
|
|
assert str(signature(stream)) == '(messages, workspace=None, client_runtime_context=None)'
|
|
assert DONE not in chunks
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize('terminal_kind', ['agent_terminal', 'metrics'])
|
|
async def test_failed_terminal_diagnostics_survive_answer_replacement(terminal_kind):
|
|
diagnostics = ['Earlier tool failure and retry', 'All tests passed.\n[Agent stopped: HTTP 504]']
|
|
|
|
@with_completion_gate
|
|
async def stream(messages):
|
|
yield _event({'delta': 'All tests passed.'})
|
|
yield ERROR
|
|
yield _event({'type': terminal_kind, 'data': {
|
|
'failed': True, 'failure': {'status': 504, 'message': 'Model request failed'},
|
|
'round_texts': diagnostics, 'round_models': ['first-model', 'failed-model'],
|
|
}})
|
|
|
|
chunks = [chunk async for chunk in stream([])]
|
|
terminal = next(data['data'] for event, data in _frames(chunks)
|
|
if event == 'message' and data.get('type') == terminal_kind)
|
|
# Diagnostics and the failure note survive; the rejected claim does not,
|
|
# because round_texts are rendered again when the turn is reloaded.
|
|
assert terminal['round_texts'] == ['Earlier tool failure and retry', '[Agent stopped: HTTP 504]']
|
|
assert terminal['round_models'] == ['first-model', 'failed-model']
|
|
assert terminal['failure'] == {'status': 504, 'message': 'Model request failed'}
|
|
assert terminal['failed'] is True
|
|
assert terminal['completion_decision'] == _decision(chunks)
|
|
assert terminal['completion_gate']['answer_replaced'] is True
|
|
assert terminal['completion_gate']['additional_provider_calls'] == 0
|
|
assert _labels(chunks).index(terminal_kind) < _labels(chunks).index('error')
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_provider_failure_round_texts_cannot_replay_removed_claim_after_reload():
|
|
claim = 'I created report.md and all tests passed.'
|
|
note = '[Agent stopped: Model request failed (HTTP 504)]'
|
|
|
|
@with_completion_gate
|
|
async def stream(messages):
|
|
yield _event({'type': 'tool_start', 'tool': 'read_file'})
|
|
yield _event({'delta': 'Inspected the layout. ' + claim})
|
|
yield ERROR
|
|
yield _event({'type': 'agent_terminal', 'data': {
|
|
'failed': True, 'failure': {'status': 504, 'message': 'Model request failed'},
|
|
'tool_events': [{'round': 1, 'tool': 'read_file'}],
|
|
'round_texts': ['Inspected the layout. ' + claim, 'Retrying the build.\n\n' + note],
|
|
}})
|
|
yield DONE
|
|
|
|
chunks = [chunk async for chunk in stream([{'role': 'user', 'content': 'create report.md and run the tests'}])]
|
|
assert _labels(chunks) == [
|
|
'tool_start', 'final_response', 'completion_decision', 'agent_terminal', 'error',
|
|
], _labels(chunks)
|
|
assert DONE not in chunks
|
|
live = next(data['content'] for event, data in _frames(chunks)
|
|
if event == 'message' and data.get('type') == 'final_response')
|
|
terminal = next(data['data'] for event, data in _frames(chunks)
|
|
if event == 'message' and data.get('type') == 'agent_terminal')
|
|
# The chat route persists this metadata and the renderer rebuilds one bubble
|
|
# per round from it, so every persisted round is presentation.
|
|
persisted = terminal['round_texts']
|
|
assert persisted == ['Inspected the layout.', 'Retrying the build.\n\n' + note]
|
|
for text in [live, *persisted]:
|
|
assert 'tests passed' not in text and 'created report.md' not in text
|
|
assert 'Inspected the layout.' in live
|
|
assert terminal['completion_decision']['status'] == 'failed'
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_error_boundary_is_independent_of_network_chunking():
|
|
@with_completion_gate
|
|
async def stream(messages):
|
|
yield _event({'delta': 'Partial explanation.'})
|
|
yield ERROR
|
|
yield _event({'type': 'agent_terminal', 'data': {'failed': True}})
|
|
|
|
chunks = [chunk async for chunk in stream([])]
|
|
wire = ''.join(chunks)
|
|
expected = list(_frames(chunks))
|
|
for delivered in [chunks, [wire], list(wire)]:
|
|
# A client stops consuming on the first error, regardless of chunking.
|
|
visible = []
|
|
for frame in _frames(delivered):
|
|
visible.append(frame)
|
|
if frame[0] == 'error':
|
|
break
|
|
assert visible == expected
|
|
assert visible[-2][1]['type'] == 'agent_terminal'
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize('after_error', [False, True])
|
|
async def test_cancellation_closes_inner_stream_without_releasing_completion(after_error):
|
|
progress_seen = asyncio.Event()
|
|
closed = []
|
|
chunks = []
|
|
|
|
@with_completion_gate
|
|
async def stream(messages):
|
|
try:
|
|
yield _event({'delta': 'Tests passed.'})
|
|
if after_error:
|
|
yield ERROR
|
|
yield _event({'type': 'tool_start', 'tool': 'bash'})
|
|
await asyncio.Event().wait()
|
|
finally:
|
|
closed.append(current_journal() is not None)
|
|
|
|
async def collect():
|
|
async for chunk in stream([]):
|
|
chunks.append(chunk)
|
|
if chunk == _event({'type': 'tool_start', 'tool': 'bash'}):
|
|
progress_seen.set()
|
|
|
|
task = asyncio.create_task(collect())
|
|
try:
|
|
await asyncio.wait_for(progress_seen.wait(), timeout=5)
|
|
task.cancel()
|
|
with pytest.raises(asyncio.CancelledError):
|
|
await task
|
|
finally:
|
|
if not task.done():
|
|
task.cancel()
|
|
await asyncio.gather(task, return_exceptions=True)
|
|
assert _labels(chunks) == ['tool_start']
|
|
assert closed == [True]
|
|
assert current_journal() is None
|
|
|
|
|
|
@pytest.mark.parametrize('prose', [
|
|
'Tests pass when the command exits zero.',
|
|
'If all tests are passing, merge the branch.',
|
|
'Tests passed if the command exited zero.',
|
|
'The documentation says "5 passed".',
|
|
'The documentation says "Tests: FAIL" or "Tests: PASS".',
|
|
'You can run pytest to verify this.',
|
|
'A successful test run should show no failures.',
|
|
'For example, I created the file and updated config.py.',
|
|
'If I updated config.py, I would run pytest.',
|
|
'Imagine I ran the tests and all 42 passed.',
|
|
'Done is the label for a finished item.',
|
|
'```text\nI ran pytest and all 42 passed.\n```',
|
|
'Run pytest until there are no failures.',
|
|
])
|
|
def test_slice2_explanatory_prose_is_not_a_current_run_claim(prose):
|
|
ledger = EvidenceLedger()
|
|
answer, reason = completion_answer(prose, ledger, ledger.evaluate())
|
|
assert answer == prose
|
|
assert not reason
|
|
|
|
|
|
@pytest.mark.parametrize('instruction', [
|
|
'Explain how to write code and then test it.',
|
|
'Summarise this and check for typos.',
|
|
'Explain how to update config.py and then verify it.',
|
|
'Show an example of creating answer.json and checking it.',
|
|
'The documentation says "run pytest and create answer.json".',
|
|
'If you run pytest, the tests should pass.',
|
|
])
|
|
def test_slice2_explanatory_request_has_no_execution_requirements(instruction):
|
|
requirements = infer_completion_requirements(instruction)
|
|
assert requirements.required_artifacts == ()
|
|
assert not requirements.verifier_required
|
|
assert not requirements.executable_verifier_available
|
|
|
|
|
|
@pytest.mark.parametrize('instruction', [
|
|
'Run the tests.', 'Please run pytest.', 'Can you run the test suite?',
|
|
])
|
|
def test_slice2_explicit_test_execution_requires_a_verifier(instruction):
|
|
requirements = infer_completion_requirements(instruction)
|
|
assert requirements.verifier_required
|
|
assert not EvidenceLedger(requirements).evaluate().can_complete
|
|
|
|
|
|
@pytest.mark.parametrize('claim', [
|
|
'I ran the tests.', 'The tests passed.', '42 tests passed.',
|
|
'I created the file.', 'I updated config.py successfully.',
|
|
])
|
|
def test_slice2_execution_obligation_rejects_unsupported_claims(claim):
|
|
ledger = EvidenceLedger(CompletionRequirements(required_artifacts=('config.py',)))
|
|
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
|
|
assert reason
|
|
assert answer.startswith('The task is incomplete:')
|
|
assert claim not in answer
|
|
|
|
|
|
@pytest.mark.parametrize('claim', [
|
|
'I ran pytest to see if the tests passed.',
|
|
'I updated config.py as an example.',
|
|
'I ran pytest and should update config.py next.',
|
|
'config.py was updated successfully.',
|
|
])
|
|
def test_slice2_subordinate_explanation_cannot_hide_a_direct_execution_report(claim):
|
|
ledger = EvidenceLedger()
|
|
answer, reason = completion_answer(claim, ledger, ledger.evaluate())
|
|
assert reason
|
|
assert claim not in answer
|
|
assert 'The task is incomplete' not in answer
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_slice2_client_dictionary_cannot_attest_execution():
|
|
@with_completion_gate
|
|
async def stream(messages, client_runtime_context=None):
|
|
yield _event({'delta': 'I ran pytest and all tests passed.'})
|
|
yield DONE
|
|
|
|
context = {'execution_obligation': True, 'execution_verified': True,
|
|
'evidence_events': [{'tool': 'bash', 'command': 'pytest', 'exit_code': 0}]}
|
|
chunks = [chunk async for chunk in stream(
|
|
[{'role': 'user', 'content': 'Explain test output.'}], client_runtime_context=context)]
|
|
final = next(data['content'] for event, data in _frames(chunks)
|
|
if event == 'message' and data.get('type') == 'final_response')
|
|
assert 'I ran pytest' not in final
|
|
assert 'The task is incomplete' not in final
|
|
assert _decision(chunks)['can_complete'] is True
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_slice2_conversational_fabrication_is_corrected_without_execution_incomplete():
|
|
invocations = []
|
|
|
|
@with_completion_gate
|
|
async def stream(messages):
|
|
invocations.append(1)
|
|
yield _event({'delta': 'The function returns a boolean. I ran pytest and all tests passed.'})
|
|
yield _event({'type': 'metrics', 'data': {}})
|
|
yield DONE
|
|
|
|
chunks = [chunk async for chunk in stream([{'role': 'user', 'content': 'Explain the function.'}])]
|
|
final = next(data['content'] for event, data in _frames(chunks)
|
|
if event == 'message' and data.get('type') == 'final_response')
|
|
assert 'The function returns a boolean.' in final
|
|
assert 'I ran pytest' not in final
|
|
assert 'The task is incomplete' not in final
|
|
assert _decision(chunks)['can_complete'] is True
|
|
assert invocations == [1]
|
|
metrics = next(data['data'] for event, data in _frames(chunks)
|
|
if event == 'message' and data.get('type') == 'metrics')
|
|
assert metrics['completion_gate']['additional_provider_calls'] == 0
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_slice2_quoted_example_does_not_hide_an_unsupported_report():
|
|
@with_completion_gate
|
|
async def stream(messages):
|
|
yield _event({'delta': 'The docs say "5 passed". I ran pytest.'})
|
|
yield DONE
|
|
|
|
chunks = [chunk async for chunk in stream([{'role': 'user', 'content': 'Explain pytest output.'}])]
|
|
final = next(data['content'] for event, data in _frames(chunks)
|
|
if event == 'message' and data.get('type') == 'final_response')
|
|
assert 'The docs say "5 passed".' in final
|
|
assert 'I ran pytest' not in final
|
|
assert 'The task is incomplete' not in final
|