ground research synthesis in retrieved source urls

This commit is contained in:
pewdiepie-archdaemon
2026-09-17 10:50:20 +00:00
parent 8ae0c31666
commit 10d637f8b9
2 changed files with 40 additions and 2 deletions
+25 -1
View File
@@ -1939,6 +1939,23 @@ def bounded_research_tool_policy(offered, *, searches=0, retrievals=0):
return schemas, None, True
def retrieved_source_urls(arguments):
"""Return explicit HTTP(S) sources actually passed to a retrieval tool."""
if not isinstance(arguments, dict):
return []
values = arguments.get('urls') or arguments.get('url') or arguments.get('target_url') or []
if isinstance(values, str):
values = re.findall(r'https?://[^\s,\]\)]+', values)
if not isinstance(values, (list, tuple)):
return []
urls = []
for value in values:
value = str(value or '').strip()
if value.startswith(('http://', 'https://')) and value not in urls:
urls.append(value)
return urls
def serialize_required_email_attachment_chain(proposed, required_tools, executions):
"""Keep speculative email attachment batches on one grounded stage."""
stages = ('search_emails', 'read_email', 'download_attachment', 'draft_email')
@@ -3712,6 +3729,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
empty_web_search_attempts = 0
successful_web_searches = 0
successful_web_retrievals = 0
retrieved_web_sources = []
browser_navigation_outcomes = {}
failed_call_counts = {}
semantic_attempt_counts = {}
@@ -4602,12 +4620,18 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
)
if not failed and canonical(actual_tool) in {'web_fetch', 'private_browser'}:
successful_web_retrievals += 1
for source_url in retrieved_source_urls(args):
if source_url not in retrieved_web_sources:
retrieved_web_sources.append(source_url)
if successful_web_searches >= 2 and not required_artifacts:
force_no_tools_next_round = True
round_recovery_messages.append(
'Evidence retrieval is complete. Stop using tools and deliver the '
'complete answer now, covering every requested fact, comparison, and '
'caveat with the source URLs supported by the retrieved evidence.'
'caveat with the source URLs supported by the retrieved evidence. '
'Retrieved source URLs: '
+ (', '.join(retrieved_web_sources) or 'none recorded')
+ '.'
)
if (execution_attempted
and canonical(actual_tool) in contract_required_tools):
+15 -1
View File
@@ -5,7 +5,7 @@ import jsonschema
import pytest
import re
from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, record_tool_execution, align_structured_tool_history, provider_request_messages, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, serialize_required_email_attachment_chain
from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, record_tool_execution, align_structured_tool_history, provider_request_messages, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, retrieved_source_urls, serialize_required_email_attachment_chain
from src.tool_capabilities import capabilities_for_tool
@@ -1067,6 +1067,16 @@ def test_bounded_research_policy_reserves_synthesis_after_retrieval():
assert active is True
def test_retrieved_source_urls_accepts_native_lists_and_serialized_arrays():
assert retrieved_source_urls({
'urls': ['https://one.example/a', 'https://two.example/b'],
}) == ['https://one.example/a', 'https://two.example/b']
assert retrieved_source_urls({
'urls': '[https://one.example/a, https://two.example/b]',
}) == ['https://one.example/a', 'https://two.example/b']
assert retrieved_source_urls({'url': 'file:///tmp/not-public'}) == []
@pytest.mark.asyncio
async def test_stream_bounds_research_to_two_searches_fetch_then_synthesis(monkeypatch):
import src.clean_agent_preview as module
@@ -1135,6 +1145,10 @@ async def test_stream_bounds_research_to_two_searches_fetch_then_synthesis(monke
for schema in requests[2]['tools']
)
assert 'tools' not in requests[3]
assert any(
'Retrieved source URLs: https://example.org/source' in str(message.get('content', ''))
for message in requests[3]['messages']
)
assert any('Complete evidence-grounded answer' in chunk for chunk in raw)