fix(agent): preserve tool evidence through final synthesis

This commit is contained in:
pewdiepie-archdaemon
2026-09-18 13:29:54 +00:00
parent 951060e4d4
commit fb9558439c
4 changed files with 202 additions and 15 deletions
+150 -7
View File
@@ -30,6 +30,7 @@ from src.tool_schemas import (
normalized_native_function_argument_error,
)
from src.tool_types import ToolBlock
from src.tool_parsing import parse_tool_blocks, strip_tool_blocks
from src.turn_contract import (
FAMILY_TOOLS, broad_web_briefing_request, required_read_operation_for_request,
targets_bound_editor_request, inline_text_transformation,
@@ -684,7 +685,21 @@ def documents_terminal_response(raw, *, user_text='', max_items=8):
def shell_listing_terminal_response(raw, *, user_text=''):
"""Return successful read-only listing evidence when prose omits the rows."""
if not re.search(r'\b(?:list|names?)\b', str(user_text or ''), re.I):
# Bare words such as "list every move" or "list the findings" describe
# the shape of the eventual answer; they do not ask for shell stdout. The
# old broad matcher made an intermediate ``ls`` (for example, after video
# frame extraction) terminate the whole agent turn as "Workspace items".
# Only let the shell own rendering when the user explicitly requested a
# filesystem/directory listing.
if not re.search(
r'\b(?:list|show|display|name)\b[^\n]{0,48}'
r'\b(?:files?|folders?|director(?:y|ies)|workspace items?|paths?|filenames?)\b'
r'|\b(?:files?|folders?|director(?:y|ies)|workspace items?|paths?|filenames?)\b'
r'[^\n]{0,48}\b(?:list|names?)\b'
r'|\blist\b[^\n]{0,32}\b(?:whats|what\'s|what is)\s+in\s+there\b',
str(user_text or ''),
re.I,
):
return ''
payload = raw
try:
@@ -718,6 +733,19 @@ def shell_output_terminal_response(raw, *, maximum=4000):
return text[:maximum] + ('\n…' if len(text) > maximum else '')
def direct_shell_output_request(user_text):
"""Whether raw stdout itself is the deliverable requested by the user."""
text = str(user_text or '')
return bool(re.search(
r'\b(?:run|execute)\b[^\n]{0,32}\b(?:this\s+)?(?:command|script)\b'
r'|\b(?:bash|shell|terminal|stdout|command output)\b'
r'|\b(?:print|show|tell me|what(?:\'s| is))\b[^\n]{0,40}'
r'\b(?:hostname|working directory|current directory|workspace path|pwd)\b',
text,
re.I,
))
def ui_panel_terminal_response(raw, *, args=None):
"""Render only UI state confirmed by a successful ui_control result."""
command = dict(args or {})
@@ -4116,6 +4144,8 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
official_source_retry_attempted = False
note_search_recovery_attempted = False
replace_streamed_draft_on_finish = False
final_synthesis_reserved = False
emergency_completion_round = False
# Stream model text immediately. A canonical final event reconciles any
# draft that completion/research checks subsequently replace.
finalize_search_answer = broad_current_web_request(direct_user_text) or requested_web_source_links(direct_user_text)
@@ -4167,9 +4197,46 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
),
limits=preview_http_limits(),
) as client:
for round_number in range(1, round_limit + 1):
# One extra iteration is available only when a provider emits raw
# tool markup during the normal final no-tools round. Ordinary
# turns still obey ``round_limit`` exactly.
for round_number in range(1, round_limit + 2):
if round_number > round_limit and not emergency_completion_round:
break
rounds_used = round_number
yield event({'type': 'agent_step', 'round': round_number})
# Preserve the final model round for an actual user-facing
# answer once tools have returned evidence. Previously the
# model could spend the last round emitting another tool call
# (or provider-native tool markup that was rendered as prose),
# leaving no opportunity to synthesize the result.
reserve_final_synthesis = bool(
round_number >= round_limit
and executions
and (not required_artifacts or successful_artifact_write)
)
if reserve_final_synthesis and not final_synthesis_reserved:
final_synthesis_reserved = True
final_instruction = (
'Final completion round: no more tools are available. Finish from the '
'evidence already returned and answer the user directly and completely now. '
'Do not emit tool-call markup, describe another planned action, or merely '
'repeat raw tool output. State any remaining uncertainty explicitly.'
)
if history and history[-1].get('_harness_control'):
history[-1]['content'] = (
str(history[-1].get('content') or '') + ' ' + final_instruction
)
else:
history.append({
'role': 'user',
'_harness_control': True,
'content': final_instruction,
})
yield event({
'type': 'completion_recovery',
'reason': 'reserved_final_synthesis_round',
})
# Enforce a known research prerequisite before asking the model
# for another response, not after streaming a premature answer.
if (
@@ -4218,7 +4285,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
request_messages = provider_request_messages(
prune_multimodal_images(history, max_images=3)
)
if prior_summary_answer or force_no_tools_next_round:
if prior_summary_answer or force_no_tools_next_round or reserve_final_synthesis:
round_offered = []
force_no_tools_next_round = False
else:
@@ -4380,6 +4447,55 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
if content and not prior_summary_answer:
yield event({'delta': content})
proposed = [pending[i] for i in sorted(pending)]
unexecutable_dsml_completion = False
if not proposed and 'DSML' in content:
offered_by_canonical = {
canonical(schema['function']['name']): schema['function']['name']
for schema in round_offered
}
parsed_dsml_blocks = parse_tool_blocks(
content,
skip_fenced=True,
additional_tool_names=offered_by_canonical.values(),
additional_tool_schemas=round_offered,
)
recovered = []
for index, block in enumerate(parsed_dsml_blocks):
actual_name = offered_by_canonical.get(canonical(block.tool_type))
if not actual_name:
continue
try:
recovered_args = json.loads(block.content or '{}')
except (TypeError, ValueError, json.JSONDecodeError):
continue
if not isinstance(recovered_args, dict):
continue
recovered.append({
'id': f'call_dsml_{round_number}_{index}',
'type': 'function',
'function': {
'name': actual_name,
'arguments': json.dumps(recovered_args, ensure_ascii=False),
},
})
if recovered:
proposed = recovered
if parsed_dsml_blocks:
content = strip_tool_blocks(
content,
skip_fenced=True,
additional_tool_names=offered_by_canonical.values(),
).strip()
replace_streamed_draft_on_finish = True
if recovered:
yield event({
'type': 'tool_markup_recovery',
'format': 'deepseek_dsml',
'round': round_number,
'calls': len(recovered),
})
else:
unexecutable_dsml_completion = True
proposed = serialize_required_email_attachment_chain(
proposed, contract_required_tools, executions,
)
@@ -4392,6 +4508,33 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
message['tool_calls'] = protocol_safe_tool_calls(proposed)
history.append(message)
if not proposed:
if (
unexecutable_dsml_completion
and (
round_number < round_limit
or (round_number == round_limit and not emergency_completion_round)
)
):
answer_recovery_attempts += 1
force_no_tools_next_round = True
if round_number == round_limit:
emergency_completion_round = True
history.pop()
history.append({
'role': 'user',
'_harness_control': True,
'content': (
'Your draft emitted tool-call markup during the no-tools completion '
'phase. That call was not executed. Do not emit DSML, XML, JSON tool '
'calls, or another action plan. Answer the original request directly '
'now from the evidence already present, and state uncertainty plainly.'
),
})
yield event({
'type': 'completion_recovery',
'reason': 'tool_markup_during_final_synthesis',
})
continue
if prior_summary_answer:
content = prior_summary_answer
history[-1]['content'] = content
@@ -5610,7 +5753,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
):
shell_terminal_response = shell_listing_terminal_response(
output, user_text=latest_user,
) or shell_output_terminal_response(output)
)
if not shell_terminal_response and direct_shell_output_request(latest_user):
shell_terminal_response = shell_output_terminal_response(output)
artifact_pending = bool(
required_artifacts and not successful_artifact_write
)
@@ -5817,9 +5962,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
break
if terminal_budget_violation:
if (
getattr(turn_contract, 'routing_experiment', '') == MODEL_CHOICE_MODE
and not native_workspace_enabled
and not budget_completion_attempted
not budget_completion_attempted
and round_number < round_limit
):
budget_completion_attempted = True
+5 -2
View File
@@ -268,8 +268,11 @@ def _normalize_dsml(text: str) -> str:
if "DSML" not in text:
return text
t = text
t = re.sub(rf"<\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*tool_calls\s*>", "<tool_call>", t, flags=re.IGNORECASE)
t = re.sub(rf"<\s*/\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*tool_calls\s*>", "</tool_call>", t, flags=re.IGNORECASE)
# Hosted DeepSeek variants use both ``tool_calls`` and the shorter
# ``calls`` wrapper. Treat them identically; otherwise the inner invoke is
# parsed but the outer DSML tags leak into the user-visible response.
t = re.sub(rf"<\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*(?:tool_calls|calls)\s*>", "<tool_call>", t, flags=re.IGNORECASE)
t = re.sub(rf"<\s*/\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*(?:tool_calls|calls)\s*>", "</tool_call>", t, flags=re.IGNORECASE)
t = re.sub(rf"<\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*invoke\s+name=", "<invoke name=", t, flags=re.IGNORECASE)
t = re.sub(rf"<\s*/\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*invoke\s*>", "</invoke>", t, flags=re.IGNORECASE)
# parameter open tag — drop any extra attrs (e.g. string="true").
+30 -6
View File
@@ -5,7 +5,7 @@ import jsonschema
import pytest
import re
from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, execution_targets_required_artifact, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, bounded_web_evidence_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, web_fetch_observation_is_boilerplate, broad_current_web_request, record_tool_execution, align_structured_tool_history, provider_request_messages, provider_compatible_tool_choice_request, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, retrieved_source_urls, serialize_required_email_attachment_chain
from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, execution_targets_required_artifact, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, direct_shell_output_request, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, bounded_web_evidence_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, web_fetch_observation_is_boilerplate, broad_current_web_request, record_tool_execution, align_structured_tool_history, provider_request_messages, provider_compatible_tool_choice_request, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, retrieved_source_urls, serialize_required_email_attachment_chain
from src.tool_capabilities import capabilities_for_tool
@@ -1468,6 +1468,21 @@ def test_shell_listing_renderer_uses_successful_stdout_rows():
assert rendered == "Workspace items (3):\n- alpha\n- beta\n- gamma"
def test_shell_listing_renderer_does_not_mistake_requested_answer_list_for_files():
assert shell_listing_terminal_response(
"f_001.png\nf_002.png\n2",
user_text="Inspect the video and list every chess move with timestamps.",
) == ""
def test_raw_shell_stdout_only_owns_explicit_shell_requests():
assert direct_shell_output_request("Run this command and return its stdout: uname -a")
assert direct_shell_output_request("What is the current working directory?")
assert not direct_shell_output_request(
"Inspect the poster, calculate the package price, and explain ambiguities."
)
def test_workspace_path_followup_reuses_prior_pwd_evidence():
history = [
{'role': 'tool', 'content': '/workspace'},
@@ -4210,7 +4225,7 @@ async def test_native_stream_explains_how_to_recover_from_timestamp_free_export(
@pytest.mark.asyncio
async def test_native_stream_terminates_on_first_post_budget_tool_call(monkeypatch):
async def test_native_stream_synthesizes_after_first_post_budget_tool_call(monkeypatch):
import src.clean_agent_preview as module
responses = iter([
{"choices": [{"delta": {"tool_calls": [{
@@ -4230,6 +4245,7 @@ async def test_native_stream_terminates_on_first_post_budget_tool_call(monkeypat
}),
},
}]}}]},
{"choices": [{"delta": {"content": "The visual evidence shows a complete result."}}]},
])
class Response:
@@ -4282,8 +4298,13 @@ async def test_native_stream_terminates_on_first_post_budget_tool_call(monkeypat
]
assert len(budget_errors) == 1
final = [event for event in events if event.get("type") == "final_response"]
assert len(final) == 1
assert "budget was exhausted" in final[0]["content"]
assert not final
assert any(
event.get("type") == "completion_recovery"
and event.get("reason") == "tool_budget_final_synthesis"
for event in events
)
assert any("The visual evidence shows a complete result." in chunk for chunk in raw)
@pytest.mark.asyncio
@@ -5808,5 +5829,8 @@ async def test_parallel_tool_results_precede_visual_evidence(monkeypatch):
messages = requests[1]["messages"]
assistant_index = max(i for i, message in enumerate(messages) if message["role"] == "assistant")
assert [message["role"] for message in messages[assistant_index + 1:]] == ["tool", "tool", "user"]
assert messages[-1]["content"][1]["type"] == "image_url"
assert [message["role"] for message in messages[assistant_index + 1:]] == [
"tool", "tool", "user", "user",
]
assert "Final completion round" in messages[-1]["content"]
assert messages[-2]["content"][1]["type"] == "image_url"
@@ -449,6 +449,23 @@ def test_skip_fenced_still_recovers_dsml_markup():
assert "latest python release" in blocks[0].content
def test_short_dsml_calls_wrapper_is_parsed_and_fully_stripped():
dsml = (
"Evidence gathered.\n"
"<||DSML|| calls>"
'<||DSML|| invoke name="extract_text">'
'<||DSML|| parameter name="path" string="true">/workspace/frame.png'
'</||DSML|| parameter>'
"</||DSML|| invoke>"
"</||DSML|| calls>"
)
blocks = parse_tool_blocks(dsml, skip_fenced=True, additional_tool_names=["extract_text"])
assert len(blocks) == 1
assert blocks[0].tool_type == "extract_text"
assert json.loads(blocks[0].content) == {"path": "/workspace/frame.png"}
assert strip_tool_blocks(dsml, skip_fenced=True) == "Evidence gathered."
def test_skip_fenced_ignores_only_the_fenced_pattern():
text = "```bash\nnpm run plan:articles\n```"
assert parse_tool_blocks(text, skip_fenced=True) == []