bound web research to retrieval and synthesis

This commit is contained in:
pewdiepie-archdaemon
2026-09-17 10:41:14 +00:00
parent 218d762427
commit 8ae0c31666
2 changed files with 176 additions and 1 deletions
+61
View File
@@ -1906,6 +1906,39 @@ def dependent_write_prerequisite_error(turn_contract, name, successful_required_
return None
def bounded_research_tool_policy(offered, *, searches=0, retrievals=0):
"""Bound research loops after enough discovery evidence has been gathered.
Two searches are enough to choose a source in the ordinary research flow.
The next step must retrieve source evidence, and the following step belongs
to final synthesis. This is deliberately activated by observed web calls,
so unrelated calendar, email, document, and media turns are unchanged.
"""
schemas = list(offered or ())
if searches < 2:
return schemas, None, False
schemas = [
schema for schema in schemas
if canonical((schema.get('function') or {}).get('name')) != 'web_search'
]
if retrievals:
return [], 'none', True
fetch = next(
(
(schema.get('function') or {}).get('name')
for schema in schemas
if canonical((schema.get('function') or {}).get('name')) == 'web_fetch'
),
None,
)
if fetch:
return schemas, {
'type': 'function',
'function': {'name': fetch},
}, True
return schemas, None, True
def serialize_required_email_attachment_chain(proposed, required_tools, executions):
"""Keep speculative email attachment batches on one grounded stage."""
stages = ('search_emails', 'read_email', 'download_attachment', 'draft_email')
@@ -3677,6 +3710,8 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
successful_duplicate_counts = {}
empty_search_intents = {}
empty_web_search_attempts = 0
successful_web_searches = 0
successful_web_retrievals = 0
browser_navigation_outcomes = {}
failed_call_counts = {}
semantic_attempt_counts = {}
@@ -3796,6 +3831,13 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
canonical(schema['function']['name']), 0
) < round_number
]
research_choice = None
if not required_artifacts:
round_offered, research_choice, _ = bounded_research_tool_policy(
round_offered,
searches=successful_web_searches,
retrievals=successful_web_retrievals,
)
round_max_tokens = (
min(request_max_tokens, 4096)
if artifact_body_handoff_target else request_max_tokens
@@ -3856,6 +3898,8 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
'type': 'function',
'function': {'name': suggestion_name},
}
if research_choice is not None:
request['tool_choice'] = research_choice
pending, content = {}, ''
async with preview_model_response(client, endpoint_url, headers, request, context_recovery) as response:
response.raise_for_status()
@@ -4548,6 +4592,23 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
result.get('error')
or result.get('exit_code') not in (None, 0)
)
if not failed and canonical(actual_tool) == 'web_search':
successful_web_searches += 1
if successful_web_searches == 2 and not required_artifacts:
round_recovery_messages.append(
'Research discovery is complete after two searches. Do not search '
'again. Retrieve the strongest authoritative result with web_fetch, '
'then answer every requested fact, comparison, and caveat with source URLs.'
)
if not failed and canonical(actual_tool) in {'web_fetch', 'private_browser'}:
successful_web_retrievals += 1
if successful_web_searches >= 2 and not required_artifacts:
force_no_tools_next_round = True
round_recovery_messages.append(
'Evidence retrieval is complete. Stop using tools and deliver the '
'complete answer now, covering every requested fact, comparison, and '
'caveat with the source URLs supported by the retrieved evidence.'
)
if (execution_attempted
and canonical(actual_tool) in contract_required_tools):
attempted_required_tools.add(canonical(actual_tool))
+115 -1
View File
@@ -5,7 +5,7 @@ import jsonschema
import pytest
import re
from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, record_tool_execution, align_structured_tool_history, provider_request_messages, offered_tool_alias, dependent_write_prerequisite_error, serialize_required_email_attachment_chain
from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, record_tool_execution, align_structured_tool_history, provider_request_messages, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, serialize_required_email_attachment_chain
from src.tool_capabilities import capabilities_for_tool
@@ -1024,6 +1024,120 @@ def test_contract_item_limit_exposes_inherited_read_cap():
assert contract_item_limit(SimpleNamespace(required_read_operation=None), 20) == 20
def test_bounded_research_policy_preserves_tools_before_second_search():
schemas = [
{'type': 'function', 'function': {'name': name, 'parameters': {}}}
for name in ('web_search', 'web_fetch', 'manage_calendar')
]
offered, choice, active = bounded_research_tool_policy(
schemas, searches=1, retrievals=0,
)
assert offered == schemas
assert choice is None
assert active is False
def test_bounded_research_policy_forces_fetch_after_two_searches():
schemas = [
{'type': 'function', 'function': {'name': name, 'parameters': {}}}
for name in ('web_search', 'web_fetch', 'manage_calendar')
]
offered, choice, active = bounded_research_tool_policy(
schemas, searches=2, retrievals=0,
)
assert [schema['function']['name'] for schema in offered] == [
'web_fetch', 'manage_calendar',
]
assert choice == {
'type': 'function', 'function': {'name': 'web_fetch'},
}
assert active is True
def test_bounded_research_policy_reserves_synthesis_after_retrieval():
schemas = [
{'type': 'function', 'function': {'name': name, 'parameters': {}}}
for name in ('web_search', 'web_fetch', 'manage_calendar')
]
offered, choice, active = bounded_research_tool_policy(
schemas, searches=2, retrievals=1,
)
assert offered == []
assert choice == 'none'
assert active is True
@pytest.mark.asyncio
async def test_stream_bounds_research_to_two_searches_fetch_then_synthesis(monkeypatch):
import src.clean_agent_preview as module
def call(index, name, arguments):
return {'index': index, 'id': f'call-{index}', 'function': {
'name': name, 'arguments': json.dumps(arguments),
}}
packets = iter([
{'choices': [{'delta': {'tool_calls': [call(0, 'web_search', {'query': 'topic overview'})]}}]},
{'choices': [{'delta': {'tool_calls': [call(1, 'web_search', {'query': 'topic official source'})]}}]},
{'choices': [{'delta': {'tool_calls': [call(2, 'web_fetch', {'url': 'https://example.org/source'})]}}]},
{'choices': [{'delta': {'content': 'Complete evidence-grounded answer with https://example.org/source'}}]},
])
requests = []
executions = []
class Response:
def __init__(self, payload): self.payload = payload
async def __aenter__(self): return self
async def __aexit__(self, *args): pass
def raise_for_status(self): pass
async def aiter_lines(self):
yield 'data: ' + json.dumps(self.payload)
yield 'data: [DONE]'
class Client:
def __init__(self, **kwargs): pass
async def __aenter__(self): return self
async def __aexit__(self, *args): pass
def stream(self, *args, **kwargs):
requests.append(kwargs['json'])
return Response(next(packets))
async def execute(block, **kwargs):
executions.append(block.tool_type)
if block.tool_type == 'web_search':
return 'web_search', {
'output': '[1] Source\n https://example.org/source',
'exit_code': 0,
'evidence_status': 'available',
}
return 'web_fetch', {'output': 'Authoritative source evidence', 'exit_code': 0}
monkeypatch.setattr(module.httpx, 'AsyncClient', Client)
monkeypatch.setattr(module, 'execute_tool_block', execute)
schemas = [
schema for schema in FUNCTION_TOOL_SCHEMAS
if schema['function']['name'] in {'web_search', 'web_fetch'}
]
contract = resolve_full_inventory_contract(schemas=schemas, policy=ToolPolicy())
raw = [chunk async for chunk in stream_preview(
endpoint_url='http://test', model='test',
messages=[{'role': 'user', 'content': 'Research this topic using authoritative sources.'}],
headers={}, turn_contract=contract, session_id='test', owner='test',
disabled_tools=set(), tool_policy=ToolPolicy(), max_rounds=4,
)]
assert executions == ['web_search', 'web_search', 'web_fetch']
assert requests[2]['tool_choice'] == {
'type': 'function', 'function': {'name': 'web_fetch'},
}
assert all(
schema['function']['name'] != 'web_search'
for schema in requests[2]['tools']
)
assert 'tools' not in requests[3]
assert any('Complete evidence-grounded answer' in chunk for chunk in raw)
def test_task_renderer_honors_few_and_filters_confirmed_morning_schedule():
from src.clean_agent_preview import tasks_terminal_response
raw = {"response": "Found 3 tasks:\n"