diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py index 77ae86cf5..278dde6b6 100644 --- a/src/clean_agent_preview.py +++ b/src/clean_agent_preview.py @@ -3329,6 +3329,20 @@ def repeated_search_refinement(query, prior_intents): return False +def evidence_tool_keeps_distinct_requests_available(name): + """Whether rejecting one duplicate must not hide new evidence arguments. + + Read-only tools routinely need a new URL, page, range, or query after a + model accidentally repeats a successful call. Their exact-call guard is + already sufficient to block the duplicate; withholding the whole tool for + a round also rejects the valid corrective call. + """ + return canonical(name) in { + 'web_search', 'web_fetch', 'private_browser', 'pdf_extract', + 'read_file', 'inspect_media', 'extract_text', 'transcribe_media', + } + + def requested_web_source_links(user_text): return bool(re.search( r'\b(?:return|give|show|include|provide|cite|find)\b.{0,35}\b(?:source\s+)?links?\b' @@ -5105,7 +5119,12 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac calls += 1 duplicate_count = successful_duplicate_counts.get(call_signature, 0) + 1 successful_duplicate_counts[call_signature] = duplicate_count - if duplicate_count >= 2: + if evidence_tool_keeps_distinct_requests_available(name): + suppression = ( + 'rejected only for this exact request; the tool remains ' + 'available with different arguments' + ) + elif duplicate_count >= 2: permanently_suppressed_tools.add(canonical(name)) suppression = 'disabled for the rest of this turn' else: diff --git a/tests/test_clean_agent_preview.py b/tests/test_clean_agent_preview.py index 3d72dad66..b3f1fabf5 100644 --- a/tests/test_clean_agent_preview.py +++ b/tests/test_clean_agent_preview.py @@ -2114,6 +2114,14 @@ def test_followup_search_must_change_subject_angle_not_only_freshness(): ) +def test_evidence_tools_reject_only_exact_duplicates_not_distinct_followups(): + from src.clean_agent_preview import evidence_tool_keeps_distinct_requests_available + + assert evidence_tool_keeps_distinct_requests_available('web_fetch') + assert evidence_tool_keeps_distinct_requests_available('inspect_media') + assert not evidence_tool_keeps_distinct_requests_available('write_file') + + def test_current_search_arguments_repair_stale_year_and_add_freshness(): args = preserve_requested_web_recency( 'web_search', @@ -4451,6 +4459,70 @@ async def test_failed_static_fetch_recovers_once_through_rendered_browser(monkey assert any('blocked both access methods' in event.get('delta', '') for event in events) +@pytest.mark.asyncio +async def test_duplicate_fetch_does_not_hide_distinct_pagination_request(monkeypatch): + from dataclasses import replace + import src.clean_agent_preview as module + + def tool_call(call_id, url): + return { + 'index': 0, 'id': call_id, + 'function': {'name': 'web_fetch', 'arguments': json.dumps({'url': url})}, + } + + first = 'https://api.example.org/feed?start=0' + next_page = 'https://api.example.org/feed?start=100' + responses = iter([ + {'choices': [{'delta': {'tool_calls': [tool_call('first', first)]}}]}, + {'choices': [{'delta': {'tool_calls': [tool_call('duplicate', first)]}}]}, + {'choices': [{'delta': {'tool_calls': [tool_call('next', next_page)]}}]}, + {'choices': [{'delta': {'content': 'Read both pages.'}}]}, + ]) + requests, executions = [], [] + + class Response: + def __init__(self, payload): self.payload = payload + async def __aenter__(self): return self + async def __aexit__(self, *args): pass + def raise_for_status(self): pass + async def aiter_lines(self): + yield 'data: ' + json.dumps(self.payload) + yield 'data: [DONE]' + + class Client: + def __init__(self, **_kwargs): pass + async def __aenter__(self): return self + async def __aexit__(self, *args): pass + def stream(self, *_args, **kwargs): + requests.append(kwargs['json']) + return Response(next(responses)) + + async def execute(block, **_kwargs): + executions.append(json.loads(block.content)['url']) + return 'web_fetch', {'output': 'Readable feed page.', 'exit_code': 0} + + monkeypatch.setattr(module.httpx, 'AsyncClient', Client) + monkeypatch.setattr(module, 'execute_tool_block', execute) + schema = next(s for s in FUNCTION_TOOL_SCHEMAS if s['function']['name'] == 'web_fetch') + contract = replace( + resolve_full_inventory_contract(schemas=[schema], policy=ToolPolicy()), + routing_experiment='recent_model_choice', + ) + raw = [chunk async for chunk in stream_preview( + endpoint_url='http://test', model='test', + messages=[{'role': 'user', 'content': 'Read the public feed pages.'}], + headers={}, turn_contract=contract, session_id='test', owner='test', + disabled_tools=set(), tool_policy=ToolPolicy(), max_rounds=4, + )] + + assert executions == [first, next_page] + assert any('different arguments' in chunk for chunk in raw) + assert any( + schema['function']['name'] == 'web_fetch' + for schema in requests[2].get('tools', []) + ) + + @pytest.mark.asyncio async def test_blocked_search_engine_browser_forces_native_web_search(monkeypatch): import src.clean_agent_preview as module