mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 06:52:20 +02:00
fix(agent): preserve tool evidence through final synthesis
This commit is contained in:
+150
-7
@@ -30,6 +30,7 @@ from src.tool_schemas import (
|
||||
normalized_native_function_argument_error,
|
||||
)
|
||||
from src.tool_types import ToolBlock
|
||||
from src.tool_parsing import parse_tool_blocks, strip_tool_blocks
|
||||
from src.turn_contract import (
|
||||
FAMILY_TOOLS, broad_web_briefing_request, required_read_operation_for_request,
|
||||
targets_bound_editor_request, inline_text_transformation,
|
||||
@@ -684,7 +685,21 @@ def documents_terminal_response(raw, *, user_text='', max_items=8):
|
||||
|
||||
def shell_listing_terminal_response(raw, *, user_text=''):
|
||||
"""Return successful read-only listing evidence when prose omits the rows."""
|
||||
if not re.search(r'\b(?:list|names?)\b', str(user_text or ''), re.I):
|
||||
# Bare words such as "list every move" or "list the findings" describe
|
||||
# the shape of the eventual answer; they do not ask for shell stdout. The
|
||||
# old broad matcher made an intermediate ``ls`` (for example, after video
|
||||
# frame extraction) terminate the whole agent turn as "Workspace items".
|
||||
# Only let the shell own rendering when the user explicitly requested a
|
||||
# filesystem/directory listing.
|
||||
if not re.search(
|
||||
r'\b(?:list|show|display|name)\b[^\n]{0,48}'
|
||||
r'\b(?:files?|folders?|director(?:y|ies)|workspace items?|paths?|filenames?)\b'
|
||||
r'|\b(?:files?|folders?|director(?:y|ies)|workspace items?|paths?|filenames?)\b'
|
||||
r'[^\n]{0,48}\b(?:list|names?)\b'
|
||||
r'|\blist\b[^\n]{0,32}\b(?:whats|what\'s|what is)\s+in\s+there\b',
|
||||
str(user_text or ''),
|
||||
re.I,
|
||||
):
|
||||
return ''
|
||||
payload = raw
|
||||
try:
|
||||
@@ -718,6 +733,19 @@ def shell_output_terminal_response(raw, *, maximum=4000):
|
||||
return text[:maximum] + ('\n…' if len(text) > maximum else '')
|
||||
|
||||
|
||||
def direct_shell_output_request(user_text):
|
||||
"""Whether raw stdout itself is the deliverable requested by the user."""
|
||||
text = str(user_text or '')
|
||||
return bool(re.search(
|
||||
r'\b(?:run|execute)\b[^\n]{0,32}\b(?:this\s+)?(?:command|script)\b'
|
||||
r'|\b(?:bash|shell|terminal|stdout|command output)\b'
|
||||
r'|\b(?:print|show|tell me|what(?:\'s| is))\b[^\n]{0,40}'
|
||||
r'\b(?:hostname|working directory|current directory|workspace path|pwd)\b',
|
||||
text,
|
||||
re.I,
|
||||
))
|
||||
|
||||
|
||||
def ui_panel_terminal_response(raw, *, args=None):
|
||||
"""Render only UI state confirmed by a successful ui_control result."""
|
||||
command = dict(args or {})
|
||||
@@ -4116,6 +4144,8 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
|
||||
official_source_retry_attempted = False
|
||||
note_search_recovery_attempted = False
|
||||
replace_streamed_draft_on_finish = False
|
||||
final_synthesis_reserved = False
|
||||
emergency_completion_round = False
|
||||
# Stream model text immediately. A canonical final event reconciles any
|
||||
# draft that completion/research checks subsequently replace.
|
||||
finalize_search_answer = broad_current_web_request(direct_user_text) or requested_web_source_links(direct_user_text)
|
||||
@@ -4167,9 +4197,46 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
|
||||
),
|
||||
limits=preview_http_limits(),
|
||||
) as client:
|
||||
for round_number in range(1, round_limit + 1):
|
||||
# One extra iteration is available only when a provider emits raw
|
||||
# tool markup during the normal final no-tools round. Ordinary
|
||||
# turns still obey ``round_limit`` exactly.
|
||||
for round_number in range(1, round_limit + 2):
|
||||
if round_number > round_limit and not emergency_completion_round:
|
||||
break
|
||||
rounds_used = round_number
|
||||
yield event({'type': 'agent_step', 'round': round_number})
|
||||
# Preserve the final model round for an actual user-facing
|
||||
# answer once tools have returned evidence. Previously the
|
||||
# model could spend the last round emitting another tool call
|
||||
# (or provider-native tool markup that was rendered as prose),
|
||||
# leaving no opportunity to synthesize the result.
|
||||
reserve_final_synthesis = bool(
|
||||
round_number >= round_limit
|
||||
and executions
|
||||
and (not required_artifacts or successful_artifact_write)
|
||||
)
|
||||
if reserve_final_synthesis and not final_synthesis_reserved:
|
||||
final_synthesis_reserved = True
|
||||
final_instruction = (
|
||||
'Final completion round: no more tools are available. Finish from the '
|
||||
'evidence already returned and answer the user directly and completely now. '
|
||||
'Do not emit tool-call markup, describe another planned action, or merely '
|
||||
'repeat raw tool output. State any remaining uncertainty explicitly.'
|
||||
)
|
||||
if history and history[-1].get('_harness_control'):
|
||||
history[-1]['content'] = (
|
||||
str(history[-1].get('content') or '') + ' ' + final_instruction
|
||||
)
|
||||
else:
|
||||
history.append({
|
||||
'role': 'user',
|
||||
'_harness_control': True,
|
||||
'content': final_instruction,
|
||||
})
|
||||
yield event({
|
||||
'type': 'completion_recovery',
|
||||
'reason': 'reserved_final_synthesis_round',
|
||||
})
|
||||
# Enforce a known research prerequisite before asking the model
|
||||
# for another response, not after streaming a premature answer.
|
||||
if (
|
||||
@@ -4218,7 +4285,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
|
||||
request_messages = provider_request_messages(
|
||||
prune_multimodal_images(history, max_images=3)
|
||||
)
|
||||
if prior_summary_answer or force_no_tools_next_round:
|
||||
if prior_summary_answer or force_no_tools_next_round or reserve_final_synthesis:
|
||||
round_offered = []
|
||||
force_no_tools_next_round = False
|
||||
else:
|
||||
@@ -4380,6 +4447,55 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
|
||||
if content and not prior_summary_answer:
|
||||
yield event({'delta': content})
|
||||
proposed = [pending[i] for i in sorted(pending)]
|
||||
unexecutable_dsml_completion = False
|
||||
if not proposed and 'DSML' in content:
|
||||
offered_by_canonical = {
|
||||
canonical(schema['function']['name']): schema['function']['name']
|
||||
for schema in round_offered
|
||||
}
|
||||
parsed_dsml_blocks = parse_tool_blocks(
|
||||
content,
|
||||
skip_fenced=True,
|
||||
additional_tool_names=offered_by_canonical.values(),
|
||||
additional_tool_schemas=round_offered,
|
||||
)
|
||||
recovered = []
|
||||
for index, block in enumerate(parsed_dsml_blocks):
|
||||
actual_name = offered_by_canonical.get(canonical(block.tool_type))
|
||||
if not actual_name:
|
||||
continue
|
||||
try:
|
||||
recovered_args = json.loads(block.content or '{}')
|
||||
except (TypeError, ValueError, json.JSONDecodeError):
|
||||
continue
|
||||
if not isinstance(recovered_args, dict):
|
||||
continue
|
||||
recovered.append({
|
||||
'id': f'call_dsml_{round_number}_{index}',
|
||||
'type': 'function',
|
||||
'function': {
|
||||
'name': actual_name,
|
||||
'arguments': json.dumps(recovered_args, ensure_ascii=False),
|
||||
},
|
||||
})
|
||||
if recovered:
|
||||
proposed = recovered
|
||||
if parsed_dsml_blocks:
|
||||
content = strip_tool_blocks(
|
||||
content,
|
||||
skip_fenced=True,
|
||||
additional_tool_names=offered_by_canonical.values(),
|
||||
).strip()
|
||||
replace_streamed_draft_on_finish = True
|
||||
if recovered:
|
||||
yield event({
|
||||
'type': 'tool_markup_recovery',
|
||||
'format': 'deepseek_dsml',
|
||||
'round': round_number,
|
||||
'calls': len(recovered),
|
||||
})
|
||||
else:
|
||||
unexecutable_dsml_completion = True
|
||||
proposed = serialize_required_email_attachment_chain(
|
||||
proposed, contract_required_tools, executions,
|
||||
)
|
||||
@@ -4392,6 +4508,33 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
|
||||
message['tool_calls'] = protocol_safe_tool_calls(proposed)
|
||||
history.append(message)
|
||||
if not proposed:
|
||||
if (
|
||||
unexecutable_dsml_completion
|
||||
and (
|
||||
round_number < round_limit
|
||||
or (round_number == round_limit and not emergency_completion_round)
|
||||
)
|
||||
):
|
||||
answer_recovery_attempts += 1
|
||||
force_no_tools_next_round = True
|
||||
if round_number == round_limit:
|
||||
emergency_completion_round = True
|
||||
history.pop()
|
||||
history.append({
|
||||
'role': 'user',
|
||||
'_harness_control': True,
|
||||
'content': (
|
||||
'Your draft emitted tool-call markup during the no-tools completion '
|
||||
'phase. That call was not executed. Do not emit DSML, XML, JSON tool '
|
||||
'calls, or another action plan. Answer the original request directly '
|
||||
'now from the evidence already present, and state uncertainty plainly.'
|
||||
),
|
||||
})
|
||||
yield event({
|
||||
'type': 'completion_recovery',
|
||||
'reason': 'tool_markup_during_final_synthesis',
|
||||
})
|
||||
continue
|
||||
if prior_summary_answer:
|
||||
content = prior_summary_answer
|
||||
history[-1]['content'] = content
|
||||
@@ -5610,7 +5753,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
|
||||
):
|
||||
shell_terminal_response = shell_listing_terminal_response(
|
||||
output, user_text=latest_user,
|
||||
) or shell_output_terminal_response(output)
|
||||
)
|
||||
if not shell_terminal_response and direct_shell_output_request(latest_user):
|
||||
shell_terminal_response = shell_output_terminal_response(output)
|
||||
artifact_pending = bool(
|
||||
required_artifacts and not successful_artifact_write
|
||||
)
|
||||
@@ -5817,9 +5962,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
|
||||
break
|
||||
if terminal_budget_violation:
|
||||
if (
|
||||
getattr(turn_contract, 'routing_experiment', '') == MODEL_CHOICE_MODE
|
||||
and not native_workspace_enabled
|
||||
and not budget_completion_attempted
|
||||
not budget_completion_attempted
|
||||
and round_number < round_limit
|
||||
):
|
||||
budget_completion_attempted = True
|
||||
|
||||
+5
-2
@@ -268,8 +268,11 @@ def _normalize_dsml(text: str) -> str:
|
||||
if "DSML" not in text:
|
||||
return text
|
||||
t = text
|
||||
t = re.sub(rf"<\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*tool_calls\s*>", "<tool_call>", t, flags=re.IGNORECASE)
|
||||
t = re.sub(rf"<\s*/\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*tool_calls\s*>", "</tool_call>", t, flags=re.IGNORECASE)
|
||||
# Hosted DeepSeek variants use both ``tool_calls`` and the shorter
|
||||
# ``calls`` wrapper. Treat them identically; otherwise the inner invoke is
|
||||
# parsed but the outer DSML tags leak into the user-visible response.
|
||||
t = re.sub(rf"<\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*(?:tool_calls|calls)\s*>", "<tool_call>", t, flags=re.IGNORECASE)
|
||||
t = re.sub(rf"<\s*/\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*(?:tool_calls|calls)\s*>", "</tool_call>", t, flags=re.IGNORECASE)
|
||||
t = re.sub(rf"<\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*invoke\s+name=", "<invoke name=", t, flags=re.IGNORECASE)
|
||||
t = re.sub(rf"<\s*/\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*invoke\s*>", "</invoke>", t, flags=re.IGNORECASE)
|
||||
# parameter open tag — drop any extra attrs (e.g. string="true").
|
||||
|
||||
@@ -5,7 +5,7 @@ import jsonschema
|
||||
import pytest
|
||||
import re
|
||||
|
||||
from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, execution_targets_required_artifact, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, bounded_web_evidence_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, web_fetch_observation_is_boilerplate, broad_current_web_request, record_tool_execution, align_structured_tool_history, provider_request_messages, provider_compatible_tool_choice_request, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, retrieved_source_urls, serialize_required_email_attachment_chain
|
||||
from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, execution_targets_required_artifact, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, direct_shell_output_request, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, bounded_web_evidence_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, web_fetch_observation_is_boilerplate, broad_current_web_request, record_tool_execution, align_structured_tool_history, provider_request_messages, provider_compatible_tool_choice_request, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, retrieved_source_urls, serialize_required_email_attachment_chain
|
||||
from src.tool_capabilities import capabilities_for_tool
|
||||
|
||||
|
||||
@@ -1468,6 +1468,21 @@ def test_shell_listing_renderer_uses_successful_stdout_rows():
|
||||
assert rendered == "Workspace items (3):\n- alpha\n- beta\n- gamma"
|
||||
|
||||
|
||||
def test_shell_listing_renderer_does_not_mistake_requested_answer_list_for_files():
|
||||
assert shell_listing_terminal_response(
|
||||
"f_001.png\nf_002.png\n2",
|
||||
user_text="Inspect the video and list every chess move with timestamps.",
|
||||
) == ""
|
||||
|
||||
|
||||
def test_raw_shell_stdout_only_owns_explicit_shell_requests():
|
||||
assert direct_shell_output_request("Run this command and return its stdout: uname -a")
|
||||
assert direct_shell_output_request("What is the current working directory?")
|
||||
assert not direct_shell_output_request(
|
||||
"Inspect the poster, calculate the package price, and explain ambiguities."
|
||||
)
|
||||
|
||||
|
||||
def test_workspace_path_followup_reuses_prior_pwd_evidence():
|
||||
history = [
|
||||
{'role': 'tool', 'content': '/workspace'},
|
||||
@@ -4210,7 +4225,7 @@ async def test_native_stream_explains_how_to_recover_from_timestamp_free_export(
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_native_stream_terminates_on_first_post_budget_tool_call(monkeypatch):
|
||||
async def test_native_stream_synthesizes_after_first_post_budget_tool_call(monkeypatch):
|
||||
import src.clean_agent_preview as module
|
||||
responses = iter([
|
||||
{"choices": [{"delta": {"tool_calls": [{
|
||||
@@ -4230,6 +4245,7 @@ async def test_native_stream_terminates_on_first_post_budget_tool_call(monkeypat
|
||||
}),
|
||||
},
|
||||
}]}}]},
|
||||
{"choices": [{"delta": {"content": "The visual evidence shows a complete result."}}]},
|
||||
])
|
||||
|
||||
class Response:
|
||||
@@ -4282,8 +4298,13 @@ async def test_native_stream_terminates_on_first_post_budget_tool_call(monkeypat
|
||||
]
|
||||
assert len(budget_errors) == 1
|
||||
final = [event for event in events if event.get("type") == "final_response"]
|
||||
assert len(final) == 1
|
||||
assert "budget was exhausted" in final[0]["content"]
|
||||
assert not final
|
||||
assert any(
|
||||
event.get("type") == "completion_recovery"
|
||||
and event.get("reason") == "tool_budget_final_synthesis"
|
||||
for event in events
|
||||
)
|
||||
assert any("The visual evidence shows a complete result." in chunk for chunk in raw)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@@ -5808,5 +5829,8 @@ async def test_parallel_tool_results_precede_visual_evidence(monkeypatch):
|
||||
|
||||
messages = requests[1]["messages"]
|
||||
assistant_index = max(i for i, message in enumerate(messages) if message["role"] == "assistant")
|
||||
assert [message["role"] for message in messages[assistant_index + 1:]] == ["tool", "tool", "user"]
|
||||
assert messages[-1]["content"][1]["type"] == "image_url"
|
||||
assert [message["role"] for message in messages[assistant_index + 1:]] == [
|
||||
"tool", "tool", "user", "user",
|
||||
]
|
||||
assert "Final completion round" in messages[-1]["content"]
|
||||
assert messages[-2]["content"][1]["type"] == "image_url"
|
||||
|
||||
@@ -449,6 +449,23 @@ def test_skip_fenced_still_recovers_dsml_markup():
|
||||
assert "latest python release" in blocks[0].content
|
||||
|
||||
|
||||
def test_short_dsml_calls_wrapper_is_parsed_and_fully_stripped():
|
||||
dsml = (
|
||||
"Evidence gathered.\n"
|
||||
"<||DSML|| calls>"
|
||||
'<||DSML|| invoke name="extract_text">'
|
||||
'<||DSML|| parameter name="path" string="true">/workspace/frame.png'
|
||||
'</||DSML|| parameter>'
|
||||
"</||DSML|| invoke>"
|
||||
"</||DSML|| calls>"
|
||||
)
|
||||
blocks = parse_tool_blocks(dsml, skip_fenced=True, additional_tool_names=["extract_text"])
|
||||
assert len(blocks) == 1
|
||||
assert blocks[0].tool_type == "extract_text"
|
||||
assert json.loads(blocks[0].content) == {"path": "/workspace/frame.png"}
|
||||
assert strip_tool_blocks(dsml, skip_fenced=True) == "Evidence gathered."
|
||||
|
||||
|
||||
def test_skip_fenced_ignores_only_the_fenced_pattern():
|
||||
text = "```bash\nnpm run plan:articles\n```"
|
||||
assert parse_tool_blocks(text, skip_fenced=True) == []
|
||||
|
||||
Reference in New Issue
Block a user