From 10d637f8b981bc6d1f39f6b48f5e9fad3179ea5f Mon Sep 17 00:00:00 2001 From: pewdiepie-archdaemon Date: Thu, 17 Sep 2026 10:50:20 +0000 Subject: [PATCH] ground research synthesis in retrieved source urls --- src/clean_agent_preview.py | 26 +++++++++++++++++++++++++- tests/test_clean_agent_preview.py | 16 +++++++++++++++- 2 files changed, 40 insertions(+), 2 deletions(-) diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py index 668d7b7aa..db48ff11a 100644 --- a/src/clean_agent_preview.py +++ b/src/clean_agent_preview.py @@ -1939,6 +1939,23 @@ def bounded_research_tool_policy(offered, *, searches=0, retrievals=0): return schemas, None, True +def retrieved_source_urls(arguments): + """Return explicit HTTP(S) sources actually passed to a retrieval tool.""" + if not isinstance(arguments, dict): + return [] + values = arguments.get('urls') or arguments.get('url') or arguments.get('target_url') or [] + if isinstance(values, str): + values = re.findall(r'https?://[^\s,\]\)]+', values) + if not isinstance(values, (list, tuple)): + return [] + urls = [] + for value in values: + value = str(value or '').strip() + if value.startswith(('http://', 'https://')) and value not in urls: + urls.append(value) + return urls + + def serialize_required_email_attachment_chain(proposed, required_tools, executions): """Keep speculative email attachment batches on one grounded stage.""" stages = ('search_emails', 'read_email', 'download_attachment', 'draft_email') @@ -3712,6 +3729,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac empty_web_search_attempts = 0 successful_web_searches = 0 successful_web_retrievals = 0 + retrieved_web_sources = [] browser_navigation_outcomes = {} failed_call_counts = {} semantic_attempt_counts = {} @@ -4602,12 +4620,18 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac ) if not failed and canonical(actual_tool) in {'web_fetch', 'private_browser'}: successful_web_retrievals += 1 + for source_url in retrieved_source_urls(args): + if source_url not in retrieved_web_sources: + retrieved_web_sources.append(source_url) if successful_web_searches >= 2 and not required_artifacts: force_no_tools_next_round = True round_recovery_messages.append( 'Evidence retrieval is complete. Stop using tools and deliver the ' 'complete answer now, covering every requested fact, comparison, and ' - 'caveat with the source URLs supported by the retrieved evidence.' + 'caveat with the source URLs supported by the retrieved evidence. ' + 'Retrieved source URLs: ' + + (', '.join(retrieved_web_sources) or 'none recorded') + + '.' ) if (execution_attempted and canonical(actual_tool) in contract_required_tools): diff --git a/tests/test_clean_agent_preview.py b/tests/test_clean_agent_preview.py index 8d6667a94..b4b7543c8 100644 --- a/tests/test_clean_agent_preview.py +++ b/tests/test_clean_agent_preview.py @@ -5,7 +5,7 @@ import jsonschema import pytest import re -from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, record_tool_execution, align_structured_tool_history, provider_request_messages, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, serialize_required_email_attachment_chain +from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, record_tool_execution, align_structured_tool_history, provider_request_messages, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, retrieved_source_urls, serialize_required_email_attachment_chain from src.tool_capabilities import capabilities_for_tool @@ -1067,6 +1067,16 @@ def test_bounded_research_policy_reserves_synthesis_after_retrieval(): assert active is True +def test_retrieved_source_urls_accepts_native_lists_and_serialized_arrays(): + assert retrieved_source_urls({ + 'urls': ['https://one.example/a', 'https://two.example/b'], + }) == ['https://one.example/a', 'https://two.example/b'] + assert retrieved_source_urls({ + 'urls': '[https://one.example/a, https://two.example/b]', + }) == ['https://one.example/a', 'https://two.example/b'] + assert retrieved_source_urls({'url': 'file:///tmp/not-public'}) == [] + + @pytest.mark.asyncio async def test_stream_bounds_research_to_two_searches_fetch_then_synthesis(monkeypatch): import src.clean_agent_preview as module @@ -1135,6 +1145,10 @@ async def test_stream_bounds_research_to_two_searches_fetch_then_synthesis(monke for schema in requests[2]['tools'] ) assert 'tools' not in requests[3] + assert any( + 'Retrieved source URLs: https://example.org/source' in str(message.get('content', '')) + for message in requests[3]['messages'] + ) assert any('Complete evidence-grounded answer' in chunk for chunk in raw)