From 8ae0c31666634535884fbdc3d981ab22f9458e18 Mon Sep 17 00:00:00 2001 From: pewdiepie-archdaemon Date: Thu, 17 Sep 2026 10:41:14 +0000 Subject: [PATCH] bound web research to retrieval and synthesis --- src/clean_agent_preview.py | 61 ++++++++++++++++ tests/test_clean_agent_preview.py | 116 +++++++++++++++++++++++++++++- 2 files changed, 176 insertions(+), 1 deletion(-) diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py index d2f419dd5..668d7b7aa 100644 --- a/src/clean_agent_preview.py +++ b/src/clean_agent_preview.py @@ -1906,6 +1906,39 @@ def dependent_write_prerequisite_error(turn_contract, name, successful_required_ return None +def bounded_research_tool_policy(offered, *, searches=0, retrievals=0): + """Bound research loops after enough discovery evidence has been gathered. + + Two searches are enough to choose a source in the ordinary research flow. + The next step must retrieve source evidence, and the following step belongs + to final synthesis. This is deliberately activated by observed web calls, + so unrelated calendar, email, document, and media turns are unchanged. + """ + schemas = list(offered or ()) + if searches < 2: + return schemas, None, False + schemas = [ + schema for schema in schemas + if canonical((schema.get('function') or {}).get('name')) != 'web_search' + ] + if retrievals: + return [], 'none', True + fetch = next( + ( + (schema.get('function') or {}).get('name') + for schema in schemas + if canonical((schema.get('function') or {}).get('name')) == 'web_fetch' + ), + None, + ) + if fetch: + return schemas, { + 'type': 'function', + 'function': {'name': fetch}, + }, True + return schemas, None, True + + def serialize_required_email_attachment_chain(proposed, required_tools, executions): """Keep speculative email attachment batches on one grounded stage.""" stages = ('search_emails', 'read_email', 'download_attachment', 'draft_email') @@ -3677,6 +3710,8 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac successful_duplicate_counts = {} empty_search_intents = {} empty_web_search_attempts = 0 + successful_web_searches = 0 + successful_web_retrievals = 0 browser_navigation_outcomes = {} failed_call_counts = {} semantic_attempt_counts = {} @@ -3796,6 +3831,13 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac canonical(schema['function']['name']), 0 ) < round_number ] + research_choice = None + if not required_artifacts: + round_offered, research_choice, _ = bounded_research_tool_policy( + round_offered, + searches=successful_web_searches, + retrievals=successful_web_retrievals, + ) round_max_tokens = ( min(request_max_tokens, 4096) if artifact_body_handoff_target else request_max_tokens @@ -3856,6 +3898,8 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac 'type': 'function', 'function': {'name': suggestion_name}, } + if research_choice is not None: + request['tool_choice'] = research_choice pending, content = {}, '' async with preview_model_response(client, endpoint_url, headers, request, context_recovery) as response: response.raise_for_status() @@ -4548,6 +4592,23 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac result.get('error') or result.get('exit_code') not in (None, 0) ) + if not failed and canonical(actual_tool) == 'web_search': + successful_web_searches += 1 + if successful_web_searches == 2 and not required_artifacts: + round_recovery_messages.append( + 'Research discovery is complete after two searches. Do not search ' + 'again. Retrieve the strongest authoritative result with web_fetch, ' + 'then answer every requested fact, comparison, and caveat with source URLs.' + ) + if not failed and canonical(actual_tool) in {'web_fetch', 'private_browser'}: + successful_web_retrievals += 1 + if successful_web_searches >= 2 and not required_artifacts: + force_no_tools_next_round = True + round_recovery_messages.append( + 'Evidence retrieval is complete. Stop using tools and deliver the ' + 'complete answer now, covering every requested fact, comparison, and ' + 'caveat with the source URLs supported by the retrieved evidence.' + ) if (execution_attempted and canonical(actual_tool) in contract_required_tools): attempted_required_tools.add(canonical(actual_tool)) diff --git a/tests/test_clean_agent_preview.py b/tests/test_clean_agent_preview.py index f8ddf872a..8d6667a94 100644 --- a/tests/test_clean_agent_preview.py +++ b/tests/test_clean_agent_preview.py @@ -5,7 +5,7 @@ import jsonschema import pytest import re -from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, record_tool_execution, align_structured_tool_history, provider_request_messages, offered_tool_alias, dependent_write_prerequisite_error, serialize_required_email_attachment_chain +from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, record_tool_execution, align_structured_tool_history, provider_request_messages, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, serialize_required_email_attachment_chain from src.tool_capabilities import capabilities_for_tool @@ -1024,6 +1024,120 @@ def test_contract_item_limit_exposes_inherited_read_cap(): assert contract_item_limit(SimpleNamespace(required_read_operation=None), 20) == 20 +def test_bounded_research_policy_preserves_tools_before_second_search(): + schemas = [ + {'type': 'function', 'function': {'name': name, 'parameters': {}}} + for name in ('web_search', 'web_fetch', 'manage_calendar') + ] + offered, choice, active = bounded_research_tool_policy( + schemas, searches=1, retrievals=0, + ) + assert offered == schemas + assert choice is None + assert active is False + + +def test_bounded_research_policy_forces_fetch_after_two_searches(): + schemas = [ + {'type': 'function', 'function': {'name': name, 'parameters': {}}} + for name in ('web_search', 'web_fetch', 'manage_calendar') + ] + offered, choice, active = bounded_research_tool_policy( + schemas, searches=2, retrievals=0, + ) + assert [schema['function']['name'] for schema in offered] == [ + 'web_fetch', 'manage_calendar', + ] + assert choice == { + 'type': 'function', 'function': {'name': 'web_fetch'}, + } + assert active is True + + +def test_bounded_research_policy_reserves_synthesis_after_retrieval(): + schemas = [ + {'type': 'function', 'function': {'name': name, 'parameters': {}}} + for name in ('web_search', 'web_fetch', 'manage_calendar') + ] + offered, choice, active = bounded_research_tool_policy( + schemas, searches=2, retrievals=1, + ) + assert offered == [] + assert choice == 'none' + assert active is True + + +@pytest.mark.asyncio +async def test_stream_bounds_research_to_two_searches_fetch_then_synthesis(monkeypatch): + import src.clean_agent_preview as module + + def call(index, name, arguments): + return {'index': index, 'id': f'call-{index}', 'function': { + 'name': name, 'arguments': json.dumps(arguments), + }} + + packets = iter([ + {'choices': [{'delta': {'tool_calls': [call(0, 'web_search', {'query': 'topic overview'})]}}]}, + {'choices': [{'delta': {'tool_calls': [call(1, 'web_search', {'query': 'topic official source'})]}}]}, + {'choices': [{'delta': {'tool_calls': [call(2, 'web_fetch', {'url': 'https://example.org/source'})]}}]}, + {'choices': [{'delta': {'content': 'Complete evidence-grounded answer with https://example.org/source'}}]}, + ]) + requests = [] + executions = [] + + class Response: + def __init__(self, payload): self.payload = payload + async def __aenter__(self): return self + async def __aexit__(self, *args): pass + def raise_for_status(self): pass + async def aiter_lines(self): + yield 'data: ' + json.dumps(self.payload) + yield 'data: [DONE]' + + class Client: + def __init__(self, **kwargs): pass + async def __aenter__(self): return self + async def __aexit__(self, *args): pass + def stream(self, *args, **kwargs): + requests.append(kwargs['json']) + return Response(next(packets)) + + async def execute(block, **kwargs): + executions.append(block.tool_type) + if block.tool_type == 'web_search': + return 'web_search', { + 'output': '[1] Source\n https://example.org/source', + 'exit_code': 0, + 'evidence_status': 'available', + } + return 'web_fetch', {'output': 'Authoritative source evidence', 'exit_code': 0} + + monkeypatch.setattr(module.httpx, 'AsyncClient', Client) + monkeypatch.setattr(module, 'execute_tool_block', execute) + schemas = [ + schema for schema in FUNCTION_TOOL_SCHEMAS + if schema['function']['name'] in {'web_search', 'web_fetch'} + ] + contract = resolve_full_inventory_contract(schemas=schemas, policy=ToolPolicy()) + raw = [chunk async for chunk in stream_preview( + endpoint_url='http://test', model='test', + messages=[{'role': 'user', 'content': 'Research this topic using authoritative sources.'}], + headers={}, turn_contract=contract, session_id='test', owner='test', + disabled_tools=set(), tool_policy=ToolPolicy(), max_rounds=4, + )] + + assert executions == ['web_search', 'web_search', 'web_fetch'] + assert requests[2]['tool_choice'] == { + 'type': 'function', 'function': {'name': 'web_fetch'}, + } + assert all( + schema['function']['name'] != 'web_search' + for schema in requests[2]['tools'] + ) + assert 'tools' not in requests[3] + assert any('Complete evidence-grounded answer' in chunk for chunk in raw) + + def test_task_renderer_honors_few_and_filters_confirmed_morning_schedule(): from src.clean_agent_preview import tasks_terminal_response raw = {"response": "Found 3 tasks:\n"