diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py index cb7d2fe89..c38fcdb18 100644 --- a/src/clean_agent_preview.py +++ b/src/clean_agent_preview.py @@ -4230,7 +4230,15 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac if round_number > round_limit and not emergency_completion_round: break rounds_used = round_number - yield event({'type': 'agent_step', 'round': round_number}) + yield event({ + 'type': 'agent_step', + 'round': round_number, + 'calls_used': calls, + 'required_artifact_pending': bool( + required_artifacts and not successful_artifact_write + ), + 'artifact_write_phase': artifact_write_phase, + }) # Preserve the final model round for an actual user-facing # answer once tools have returned evidence. Previously the # model could spend the last round emitting another tool call diff --git a/tests/test_clean_agent_preview.py b/tests/test_clean_agent_preview.py index 15be56be2..6c5550dbb 100644 --- a/tests/test_clean_agent_preview.py +++ b/tests/test_clean_agent_preview.py @@ -4690,6 +4690,13 @@ async def test_native_stream_reserves_remaining_budget_for_required_artifact(mon events = [json.loads(chunk[6:]) for chunk in raw if "[DONE]" not in chunk] turn_contract = next(event for event in events if event.get("type") == "turn_contract") assert turn_contract["required_artifacts"] == ["/tmp_workspace/results"] + artifact_boundary_step = next( + event for event in events + if event.get("type") == "agent_step" + and event.get("calls_used") == module.NATIVE_ARTIFACT_RESEARCH_LIMIT + ) + assert artifact_boundary_step["required_artifact_pending"] is True + assert artifact_boundary_step["artifact_write_phase"] is False recovery = next( event for event in events if event.get("type") == "completion_recovery"