diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py index db48ff11a..53072b609 100644 --- a/src/clean_agent_preview.py +++ b/src/clean_agent_preview.py @@ -881,6 +881,21 @@ def contentless_final_response(content): )) +def incomplete_broad_web_answer(content, user_text): + """Reject a fragmentary answer to a broad current-information request.""" + request = str(user_text or '') + if not ( + re.search(r'\b(?:latest|recent|current|today(?:\'s)?)\b', request, re.I) + and re.search(r'\b(?:info(?:rmation)?|news|nees|updates?)\b', request, re.I) + ): + return False + answer = re.sub(r'https?://\S+', ' ', str(content or '')).strip() + words = re.findall(r"[A-Za-z0-9][A-Za-z0-9'’-]*", answer) + # A broad briefing cannot be fulfilled by one headline fragment. This is + # intentionally inapplicable to narrow quick-fact searches. + return len(words) < 25 + + def progressive_thinking_for_turn(model, offered_schemas): """Use Qwen reasoning only when this turn has no Odysseus tool surface.""" @@ -3749,6 +3764,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac budget_completion_attempted = False answer_recovery_attempted = False force_no_tools_next_round = False + force_web_search_next_round = False suggestion_retry_required = False suggestion_retry_attempted = False media_detail_nudge_sent = False @@ -3918,6 +3934,20 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac } if research_choice is not None: request['tool_choice'] = research_choice + if force_web_search_next_round: + web_search_name = next( + ( + schema['function']['name'] for schema in round_offered + if canonical(schema['function']['name']) == 'web_search' + ), + None, + ) + if web_search_name: + request['tool_choice'] = { + 'type': 'function', + 'function': {'name': web_search_name}, + } + force_web_search_next_round = False pending, content = {}, '' async with preview_model_response(client, endpoint_url, headers, request, context_recovery) as response: response.raise_for_status() @@ -3992,6 +4022,31 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac }) yield event({'type': 'completion_recovery', 'reason': 'contentless_answer'}) continue + if ( + successful_web_searches + and incomplete_broad_web_answer(content, direct_user_text) + and not answer_recovery_attempted + and round_number < round_limit + ): + answer_recovery_attempted = True + force_no_tools_next_round = True + replace_streamed_draft_on_finish = True + history.pop() + history.append({ + 'role': 'user', '_harness_control': True, + 'content': ( + 'Completion check: the draft is only a fragment and does not ' + 'answer the broad current-information request. Using the Web ' + 'evidence already gathered, provide a complete useful briefing ' + 'with the main findings, source attribution, and any evidence ' + 'limitations. Do not call another tool.' + ), + }) + yield event({ + 'type': 'completion_recovery', + 'reason': 'incomplete_research_answer', + }) + continue if artifact_body_handoff_target: target = artifact_body_handoff_target artifact_body_handoff_target = '' @@ -4618,7 +4673,16 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac 'again. Retrieve the strongest authoritative result with web_fetch, ' 'then answer every requested fact, comparison, and caveat with source URLs.' ) - if not failed and canonical(actual_tool) in {'web_fetch', 'private_browser'}: + browser_access_blocked = ( + canonical(actual_tool) == 'private_browser' + and not failed + and browser_observation_access_blocked(output) + ) + if ( + not failed + and canonical(actual_tool) in {'web_fetch', 'private_browser'} + and not browser_access_blocked + ): successful_web_retrievals += 1 for source_url in retrieved_source_urls(args): if source_url not in retrieved_web_sources: @@ -4641,33 +4705,62 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac if ( canonical(actual_tool) == 'private_browser' and not failed - and browser_observation_access_blocked(output) + and browser_access_blocked and any( canonical(schema['function']['name']) == 'web_fetch' for schema in offered ) ): suppressed_tool_until_round['private_browser'] = round_number + 1 - browser_url = str( - args.get('url') or args.get('target_url') or browser_current_url or '' - ).strip().rstrip('/') - if browser_url and browser_url in static_fetch_failed_urls: - # Both independent transports have now failed for - # this exact source. Do not bounce between them. - force_no_tools_next_round = True + requested_browser_url = private_browser_open_url(args) + effective_browser_url = private_browser_effective_url(result) + browser_search_url = requested_browser_url or effective_browser_url + is_search_engine_navigation = bool(re.search( + r'https?://(?:[^/]+\.)?(?:google\.[^/]+|bing\.com|duckduckgo\.com)' + r'/(?:search|sorry|html|lite|\?)', + browser_search_url, + re.I, + )) or bool(re.search( + r'https?://(?:[^/]+\.)?google\.[^/]+/sorry/', + effective_browser_url, + re.I, + )) + has_native_search = any( + canonical(schema['function']['name']) == 'web_search' + for schema in offered + ) + if is_search_engine_navigation and has_native_search: + force_web_search_next_round = True round_recovery_messages.append( - 'Both static fetch and rendered browser access failed for this ' - 'same URL. Do not retry either path; briefly report the source ' - 'access limitation without inventing article content.' + 'The public search-engine browser page returned a CAPTCHA, not ' + 'evidence. Use the native web_search tool now with the underlying ' + 'research query; do not retry or fetch the search-engine page.' ) + yield event({ + 'type': 'tool_loop_recovery', + 'reason': 'browser_search_blocked_fallback', + }) else: - # A CAPTCHA is transport output, not article - # evidence. Try the independent static reader once. - round_recovery_messages.append( - 'The browser returned only an access block or CAPTCHA, not page ' - 'content. Retry the same known URL once with web_fetch; if that ' - 'also fails, report the limitation without inventing content.' - ) + browser_url = str( + args.get('url') or args.get('target_url') or browser_current_url or '' + ).strip().rstrip('/') + if browser_url and browser_url in static_fetch_failed_urls: + # Both independent transports have now failed for + # this exact source. Do not bounce between them. + force_no_tools_next_round = True + round_recovery_messages.append( + 'Both static fetch and rendered browser access failed for this ' + 'same URL. Do not retry either path; briefly report the source ' + 'access limitation without inventing article content.' + ) + else: + # A CAPTCHA is transport output, not article + # evidence. Try the independent static reader once. + round_recovery_messages.append( + 'The browser returned only an access block or CAPTCHA, not page ' + 'content. Retry the same known URL once with web_fetch; if that ' + 'also fails, report the limitation without inventing content.' + ) if ( canonical(actual_tool) == 'web_fetch' and failed diff --git a/src/turn_contract.py b/src/turn_contract.py index 65aeadb69..0a7703f9a 100644 --- a/src/turn_contract.py +++ b/src/turn_contract.py @@ -3962,6 +3962,14 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum # route to scheduled tasks). Keep the whole read-only web family so a # weak search can recover through fetch/browser without schema growth. return frozenset({"search_browser"}) + if ( + re.search(r"\b(?:latest|recent|current|today(?:'s)?)\b", text, re.I) + and re.search(r"\b(?:info(?:rmation)?|news|nees|updates?)\b", text, re.I) + ): + # Broad current-information requests still require live Web evidence. + # Keep the common ``nees`` typo because a missed route leaves the model + # with no way to answer and encourages it to ask unnecessary questions. + return frozenset({"search_browser"}) if ( len(concrete_urls) >= 2 and re.search(r"\b(?:open|fetch|read|retrieve|check|use)\b", text, re.I) diff --git a/tests/test_clean_agent_preview.py b/tests/test_clean_agent_preview.py index b4b7543c8..d9c3f788c 100644 --- a/tests/test_clean_agent_preview.py +++ b/tests/test_clean_agent_preview.py @@ -1152,6 +1152,79 @@ async def test_stream_bounds_research_to_two_searches_fetch_then_synthesis(monke assert any('Complete evidence-grounded answer' in chunk for chunk in raw) +@pytest.mark.asyncio +async def test_stream_retries_an_obviously_truncated_broad_web_answer(monkeypatch): + """A successful search must not end in a fragmentary headline stub.""" + import src.clean_agent_preview as module + + packets = iter([ + {'choices': [{'delta': {'tool_calls': [{ + 'index': 0, 'id': 'search-1', 'function': { + 'name': 'web_search', + 'arguments': json.dumps({'query': 'latest AI news'}), + }, + }]}}]}, + {'choices': [{'delta': {'content': 'Current AI news includes reports about U.'}}]}, + {'choices': [{'delta': {'content': ( + 'Here is a fuller evidence-based briefing covering the major current AI ' + 'developments, what each source actually reports, and the limits of the ' + 'available evidence. Source: https://example.org/ai-news' + )}}]}, + ]) + requests = [] + + class Response: + def __init__(self, payload): self.payload = payload + async def __aenter__(self): return self + async def __aexit__(self, *args): pass + def raise_for_status(self): pass + async def aiter_lines(self): + yield 'data: ' + json.dumps(self.payload) + yield 'data: [DONE]' + + class Client: + def __init__(self, **kwargs): pass + async def __aenter__(self): return self + async def __aexit__(self, *args): pass + def stream(self, *args, **kwargs): + requests.append(kwargs['json']) + return Response(next(packets)) + + async def execute(block, **kwargs): + return 'web_search', { + 'output': '[1] AI News\n https://example.org/ai-news', + 'exit_code': 0, + 'evidence_status': 'available', + } + + monkeypatch.setattr(module.httpx, 'AsyncClient', Client) + monkeypatch.setattr(module, 'execute_tool_block', execute) + schema = next( + s for s in FUNCTION_TOOL_SCHEMAS if s['function']['name'] == 'web_search' + ) + contract = resolve_full_inventory_contract(schemas=[schema], policy=ToolPolicy()) + raw = [chunk async for chunk in stream_preview( + endpoint_url='http://test', model='test', + messages=[{'role': 'user', 'content': 'Latest news in AI?'}], + headers={}, turn_contract=contract, session_id='test', owner='test', + disabled_tools=set(), tool_policy=ToolPolicy(), max_rounds=3, + )] + events = [json.loads(chunk[6:]) for chunk in raw if '[DONE]' not in chunk] + + assert len(requests) == 3 + assert 'tools' not in requests[2] + assert any( + event.get('type') == 'completion_recovery' + and event.get('reason') == 'incomplete_research_answer' + for event in events + ) + assert any( + event.get('type') == 'final_response' + and 'fuller evidence-based briefing' in event.get('content', '') + for event in events + ) + + def test_task_renderer_honors_few_and_filters_confirmed_morning_schedule(): from src.clean_agent_preview import tasks_terminal_response raw = {"response": "Found 3 tasks:\n" @@ -3886,6 +3959,103 @@ async def test_failed_static_fetch_recovers_once_through_rendered_browser(monkey assert any('blocked both access methods' in event.get('delta', '') for event in events) +@pytest.mark.asyncio +async def test_blocked_search_engine_browser_forces_native_web_search(monkeypatch): + import src.clean_agent_preview as module + + responses = iter([ + {'choices': [{'delta': {'tool_calls': [{ + 'index': 0, 'id': 'browser-1', + 'function': { + 'name': 'private_browser', + 'arguments': json.dumps({ + 'action': 'open', + 'url': 'https://www.google.com/search?q=latest+AI+news', + }), + }, + }]}}]}, + {'choices': [{'delta': {'tool_calls': [{ + 'index': 0, 'id': 'search-1', + 'function': { + 'name': 'web_search', + 'arguments': json.dumps({'query': 'latest AI news'}), + }, + }]}}]}, + {'choices': [{'delta': {'content': ( + 'Recent AI developments include new model releases and policy updates. ' + 'The search results identify the relevant primary sources and dates for each item, ' + 'including publication details, concrete findings, and links readers can inspect ' + 'for the full context behind each development.' + )}}]}, + ]) + requests = [] + + class Response: + def __init__(self, payload): self.payload = payload + async def __aenter__(self): return self + async def __aexit__(self, *args): pass + def raise_for_status(self): pass + async def aiter_lines(self): + yield 'data: ' + json.dumps(self.payload) + yield 'data: [DONE]' + + class Client: + def __init__(self, **kwargs): pass + async def __aenter__(self): return self + async def __aexit__(self, *args): pass + def stream(self, *args, **kwargs): + requests.append(kwargs['json']) + return Response(next(responses)) + + executions = [] + + async def execute(block, **kwargs): + executions.append(block.tool_type) + if block.tool_type == 'private_browser': + return 'private_browser', { + 'output': json.dumps({ + 'url': 'https://www.google.com/sorry/index?continue=search', + 'text': 'Our systems have detected unusual traffic. Complete the reCAPTCHA.', + }), + 'exit_code': 0, + } + return 'web_search', { + 'output': json.dumps({'results': [{ + 'title': 'AI update', 'url': 'https://example.com/ai', + 'snippet': 'A dated current report.', + }]}), + 'exit_code': 0, + } + + monkeypatch.setattr(module.httpx, 'AsyncClient', Client) + monkeypatch.setattr(module, 'execute_tool_block', execute) + schemas = [ + schema for schema in FUNCTION_TOOL_SCHEMAS + if schema['function']['name'] in {'web_search', 'web_fetch', 'private_browser'} + ] + contract = replace( + resolve_full_inventory_contract(schemas=schemas, policy=ToolPolicy()), + routing_experiment='recent_model_choice', + ) + raw = [chunk async for chunk in stream_preview( + endpoint_url='http://test', model='test', + messages=[{'role': 'user', 'content': 'Open browser and find the latest AI news.'}], + headers={}, turn_contract=contract, session_id='test', owner='test', + disabled_tools=set(), tool_policy=ToolPolicy(), max_rounds=4, + )] + + assert executions == ['private_browser', 'web_search'] + assert requests[1]['tool_choice'] == { + 'type': 'function', 'function': {'name': 'web_search'}, + } + events = [json.loads(chunk[6:]) for chunk in raw if '[DONE]' not in chunk] + assert any( + event.get('type') == 'tool_loop_recovery' + and event.get('reason') == 'browser_search_blocked_fallback' + for event in events + ) + + @pytest.mark.asyncio async def test_native_stream_reserves_remaining_budget_for_required_artifact(monkeypatch): import src.clean_agent_preview as module diff --git a/tests/test_turn_contract.py b/tests/test_turn_contract.py index 5c9ff9478..286e1e894 100644 --- a/tests/test_turn_contract.py +++ b/tests/test_turn_contract.py @@ -28,6 +28,12 @@ def resolve(capabilities=(), *, schemas=FUNCTION_TOOL_SCHEMAS, policy=None, requ warm_tools=warm_tools) +def test_latest_topic_info_and_common_news_typo_route_to_web_search(): + """Broad current-info prompts must not silently lose the Web surface.""" + for prompt in ("What's latest AI info?", "What's latest AI nees?"): + assert requested_capabilities(prompt) == frozenset({"search_browser"}) + + def test_successfully_used_tool_stays_offered_when_next_turn_routes_elsewhere(): contract = resolve( {"notes"},