From 3edf7acd21f4a6aac57860cc6584cad94cfa3a63 Mon Sep 17 00:00:00 2001 From: pewdiepie-archdaemon Date: Thu, 17 Sep 2026 20:20:40 +0000 Subject: [PATCH] Allow evidence verification after successful search and fetch --- src/clean_agent_preview.py | 25 +++++++++++++------------ tests/test_clean_agent_preview.py | 23 ++++++++++++++++------- 2 files changed, 29 insertions(+), 19 deletions(-) diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py index 3bbe92a6b..b346c04d3 100644 --- a/src/clean_agent_preview.py +++ b/src/clean_agent_preview.py @@ -2006,10 +2006,10 @@ def dependent_write_prerequisite_error(turn_contract, name, successful_required_ def bounded_research_tool_policy(offered, *, searches=0, retrievals=0, search_limit=2): """Bound research loops after enough discovery evidence has been gathered. - Two searches are enough to choose a source in the ordinary research flow. - The next step must retrieve source evidence, and the following step belongs - to final synthesis. This is deliberately activated by observed web calls, - so unrelated calendar, email, document, and media turns are unchanged. + Bound discovery without treating retrieved text as proof of sufficiency. + After discovery, source inspection and browser recovery stay available: + an obsolete page or partial excerpt may need another source. The global + turn/call budget still prevents unbounded research. """ schemas = list(offered or ()) if searches < max(1, int(search_limit)): @@ -2019,7 +2019,7 @@ def bounded_research_tool_policy(offered, *, searches=0, retrievals=0, search_li if canonical((schema.get('function') or {}).get('name')) != 'web_search' ] if retrievals: - return [], 'none', True + return schemas, None, True fetch = next( ( (schema.get('function') or {}).get('name') @@ -4058,7 +4058,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac round_offered, searches=successful_web_searches, retrievals=successful_web_retrievals, - search_limit=(2 if broad_current_web_request(direct_user_text) else 1), + search_limit=2, ) round_max_tokens = ( min(request_max_tokens, 4096) @@ -4962,8 +4962,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac successful_search_intents.append(successful_intent) if successful_web_searches == 2 and not required_artifacts: round_recovery_messages.append( - ('Search already returned readable article content. Use that ' - 'evidence to synthesize the answer now with source URLs.' + ('Search returned readable article content. Assess whether it ' + 'answers the request; inspect another source if facts are missing, ' + 'outdated, or contradictory. Otherwise answer with source URLs.' if successful_web_retrievals else 'Research discovery is complete after two searches. Do not search ' 'again. Retrieve the strongest authoritative result with web_fetch, ' @@ -4996,11 +4997,11 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac if source_url not in retrieved_web_sources: retrieved_web_sources.append(source_url) if successful_web_searches >= 2 and not required_artifacts: - force_no_tools_next_round = True round_recovery_messages.append( - 'Evidence retrieval is complete. Stop using tools and deliver the ' - 'complete answer now, covering every requested fact, comparison, and ' - 'caveat with the source URLs supported by the retrieved evidence. ' + 'Source text was retrieved, but retrieval alone does not prove the ' + 'question is answered. If evidence is sufficient, answer now. ' + 'Otherwise inspect a relevant source for the missing facts. ' + 'Cite only URLs supporting the associated claims. ' 'Retrieved source URLs: ' + (', '.join(retrieved_web_sources) or 'none recorded') + '.' diff --git a/tests/test_clean_agent_preview.py b/tests/test_clean_agent_preview.py index 51727750a..360b7632b 100644 --- a/tests/test_clean_agent_preview.py +++ b/tests/test_clean_agent_preview.py @@ -1080,7 +1080,7 @@ def test_narrow_lookup_policy_forces_retrieval_after_one_successful_search(): assert active is True -def test_bounded_research_policy_reserves_synthesis_after_retrieval(): +def test_bounded_research_policy_keeps_source_inspection_after_retrieval(): schemas = [ {'type': 'function', 'function': {'name': name, 'parameters': {}}} for name in ('web_search', 'web_fetch', 'manage_calendar') @@ -1088,8 +1088,8 @@ def test_bounded_research_policy_reserves_synthesis_after_retrieval(): offered, choice, active = bounded_research_tool_policy( schemas, searches=2, retrievals=1, ) - assert offered == [] - assert choice == 'none' + assert [schema['function']['name'] for schema in offered] == ['web_fetch', 'manage_calendar'] + assert choice is None assert active is True @@ -1116,7 +1116,14 @@ async def test_stream_bounds_research_to_two_searches_fetch_then_synthesis(monke {'choices': [{'delta': {'tool_calls': [call(0, 'web_search', {'query': 'topic overview'})]}}]}, {'choices': [{'delta': {'tool_calls': [call(1, 'web_search', {'query': 'topic official source'})]}}]}, {'choices': [{'delta': {'tool_calls': [call(2, 'web_fetch', {'url': 'https://example.org/source'})]}}]}, - {'choices': [{'delta': {'content': 'Complete evidence-grounded answer with https://example.org/source'}}]}, + {'choices': [{'delta': {'content': 'Complete evidence-grounded answer with https://example.org/source. ' + + 'The sources describe the topic and explain how the findings were obtained. ' + 'Their methods support the reported observations but do not establish every broader claim. ' + 'The official source supplies the definitions needed to interpret the comparison. ' + 'Independent coverage adds context while also noting the limitations of the available evidence. ' + 'These limitations matter when applying the findings to a different setting. ' + 'The conclusion should therefore stay within the conditions actually examined, and any ' + 'unanswered questions should be checked against additional primary evidence.'}}]}, ]) requests = [] executions = [] @@ -1159,7 +1166,7 @@ async def test_stream_bounds_research_to_two_searches_fetch_then_synthesis(monke endpoint_url='http://test', model='test', messages=[{'role': 'user', 'content': 'Research this topic using authoritative sources.'}], headers={}, turn_contract=contract, session_id='test', owner='test', - disabled_tools=set(), tool_policy=ToolPolicy(), max_rounds=4, + disabled_tools=set(), tool_policy=ToolPolicy(), max_rounds=5, )] assert executions == ['web_search', 'web_search', 'web_fetch'] @@ -1170,7 +1177,8 @@ async def test_stream_bounds_research_to_two_searches_fetch_then_synthesis(monke schema['function']['name'] != 'web_search' for schema in requests[2]['tools'] ) - assert 'tools' not in requests[3] + assert [schema['function']['name'] for schema in requests[3]['tools']] == ['web_fetch'] + assert requests[3].get('tool_choice') != 'none' assert any( 'Retrieved source URLs: https://example.org/source' in str(message.get('content', '')) for message in requests[3]['messages'] @@ -1274,7 +1282,8 @@ async def test_stream_retries_an_obviously_truncated_broad_web_answer(monkeypatc assert requests[3]['tool_choice'] == { 'type': 'function', 'function': {'name': 'web_fetch'}, } - assert 'tools' not in requests[-1] + assert any(s['function']['name'] == 'web_fetch' for s in requests[-1]['tools']) + assert requests[-1].get('tool_choice') != 'none' assert any( event.get('type') == 'completion_recovery' and event.get('reason') == 'insufficient_research_breadth'