From 02200b2874140b5894a314eb21a84e5568d255ed Mon Sep 17 00:00:00 2001 From: pewdiepie-archdaemon Date: Thu, 17 Sep 2026 19:57:20 +0000 Subject: [PATCH] reuse embedded search articles and reject browser error pages --- src/clean_agent_preview.py | 52 ++++++++++++++++++++++++++++++- tests/test_clean_agent_preview.py | 33 ++++++++++++++++++++ 2 files changed, 84 insertions(+), 1 deletion(-) diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py index ee2b418f1..9b0f59c80 100644 --- a/src/clean_agent_preview.py +++ b/src/clean_agent_preview.py @@ -1993,6 +1993,25 @@ def bounded_research_tool_policy(offered, *, searches=0, retrievals=0, search_li return schemas, None, True +def search_embedded_article_urls(output): + """Identify substantial article bodies actually delivered in search output.""" + text = str(output or '') + urls = [] + for match in re.finditer( + r'\[CONTENT(?: \d+)?\] From: (https?://\S+)\nTitle: [^\n]*\n-+\n' + r'([\s\S]*?)(?=\n\[CONTENT|\n(?:Key Points:|TL;DR:|Important Quotes:|Data / Statistics:|={5,})|\Z)', + text, + ): + url, body = match.groups() + if (len(body.split()) >= 80 + and not browser_observation_access_blocked(body) + and not browser_observation_page_missing(body) + and not web_fetch_observation_is_boilerplate(body)): + if url not in urls: + urls.append(url) + return urls + + def retrieved_source_urls(arguments): """Return explicit HTTP(S) sources actually passed to a retrieval tool.""" if not isinstance(arguments, dict): @@ -3492,6 +3511,16 @@ def private_browser_effective_url(result): return urls[-1] if urls else '' +def browser_observation_page_missing(raw): + """Recognize a rendered error page, rather than an article mentioning 404.""" + text = str(raw or '') + return bool(re.search( + r"(?:heading[^\n]{0,120}(?:Whoops!|Page not found|404)|" + r"This page doesn[’']t exist or can[’']t be found\.)", + text, re.I, + )) + + def browser_observation_access_blocked(raw): """Identify browser observations containing only an access gate.""" return bool(re.search( @@ -4878,6 +4907,12 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac 'reason': 'web_fetch_boilerplate_fallback', }) if not failed and canonical(actual_tool) == 'web_search': + embedded_urls = search_embedded_article_urls(output) + if embedded_urls: + successful_web_retrievals += 1 + for source_url in embedded_urls: + if source_url not in retrieved_web_sources: + retrieved_web_sources.append(source_url) if result.get('evidence_status') != 'empty': successful_web_searches += 1 for _title, source_url in web_source_links( @@ -4890,19 +4925,34 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac successful_search_intents.append(successful_intent) if successful_web_searches == 2 and not required_artifacts: round_recovery_messages.append( + ('Search already returned readable article content. Use that ' + 'evidence to synthesize the answer now with source URLs.' + if successful_web_retrievals else 'Research discovery is complete after two searches. Do not search ' 'again. Retrieve the strongest authoritative result with web_fetch, ' - 'then answer every requested fact, comparison, and caveat with source URLs.' + 'then answer every requested fact, comparison, and caveat with source URLs.') ) browser_access_blocked = ( canonical(actual_tool) == 'private_browser' and not failed and browser_observation_access_blocked(output) ) + browser_page_missing = ( + canonical(actual_tool) == 'private_browser' + and not failed + and browser_observation_page_missing(output) + ) + if browser_page_missing: + round_recovery_messages.append( + 'The browser displayed a missing-page error, not article evidence. ' + 'Use another exact source URL already returned by search. Do not ' + 'rewrite URL paths or claim this page was read successfully.' + ) if ( not failed and canonical(actual_tool) in {'web_fetch', 'private_browser'} and not browser_access_blocked + and not browser_page_missing ): successful_web_retrievals += 1 for source_url in retrieved_source_urls(args): diff --git a/tests/test_clean_agent_preview.py b/tests/test_clean_agent_preview.py index 27aa36455..01914dc24 100644 --- a/tests/test_clean_agent_preview.py +++ b/tests/test_clean_agent_preview.py @@ -1863,6 +1863,39 @@ def test_browser_access_gate_is_not_treated_as_page_evidence(): ) +def test_rendered_missing_page_is_not_article_evidence(): + from src.clean_agent_preview import browser_observation_page_missing + + assert browser_observation_page_missing( + '- heading "Whoops!" [level=1]\n' + '- paragraph: This page doesn’t exist or can’t be found.' + ) + assert browser_observation_page_missing('- heading "404 Page not found" [level=1]') + assert not browser_observation_page_missing( + '- heading "HTTP error handling" [level=1]\n' + '- paragraph: A 404 indicates a missing resource.' + ) + + +def test_search_embedded_article_requires_readable_body_not_source_metadata(): + from src.clean_agent_preview import search_embedded_article_urls + header = '[CONTENT 1] From: https://example.org/story\nTitle: Report\n------------------------------\n' + body = ( + 'The council published its transport review on Tuesday following a six month study. ' + 'Researchers counted journeys at twelve stations and interviewed residents about access. ' + 'Their findings showed that evening services were less reliable than morning departures. ' + 'Officials proposed additional buses on weekends while retaining existing train schedules. ' + 'The proposal will go through public consultation before any funding decision is made. ' + 'Several community groups welcomed the announcement but requested detailed cost estimates. ' + 'The report includes methodology, regional comparisons, and limitations of the passenger survey. ' + 'A further review is scheduled after the consultation closes next month.' + ) + assert search_embedded_article_urls(header + body) == ['https://example.org/story'] + assert search_embedded_article_urls('[1] Report\n https://example.org/story') == [] + assert search_embedded_article_urls(header + 'Short snippet.') == [] + assert search_embedded_article_urls(header + 'Verify you are human. ' + body) == [] + + def test_repeated_navigation_only_fetch_is_not_treated_as_page_evidence(): navigation = ( 'World SECTIONS Politics Tech TOP STORIES Newsletter Sign In Subscribe '