reuse embedded search articles and reject browser error pages

This commit is contained in:
pewdiepie-archdaemon
2026-09-17 19:57:20 +00:00
parent f27070c469
commit 02200b2874
2 changed files with 84 additions and 1 deletions
+51 -1
View File
@@ -1993,6 +1993,25 @@ def bounded_research_tool_policy(offered, *, searches=0, retrievals=0, search_li
return schemas, None, True
def search_embedded_article_urls(output):
"""Identify substantial article bodies actually delivered in search output."""
text = str(output or '')
urls = []
for match in re.finditer(
r'\[CONTENT(?: \d+)?\] From: (https?://\S+)\nTitle: [^\n]*\n-+\n'
r'([\s\S]*?)(?=\n\[CONTENT|\n(?:Key Points:|TL;DR:|Important Quotes:|Data / Statistics:|={5,})|\Z)',
text,
):
url, body = match.groups()
if (len(body.split()) >= 80
and not browser_observation_access_blocked(body)
and not browser_observation_page_missing(body)
and not web_fetch_observation_is_boilerplate(body)):
if url not in urls:
urls.append(url)
return urls
def retrieved_source_urls(arguments):
"""Return explicit HTTP(S) sources actually passed to a retrieval tool."""
if not isinstance(arguments, dict):
@@ -3492,6 +3511,16 @@ def private_browser_effective_url(result):
return urls[-1] if urls else ''
def browser_observation_page_missing(raw):
"""Recognize a rendered error page, rather than an article mentioning 404."""
text = str(raw or '')
return bool(re.search(
r"(?:heading[^\n]{0,120}(?:Whoops!|Page not found|404)|"
r"This page doesn[’']t exist or can[’']t be found\.)",
text, re.I,
))
def browser_observation_access_blocked(raw):
"""Identify browser observations containing only an access gate."""
return bool(re.search(
@@ -4878,6 +4907,12 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
'reason': 'web_fetch_boilerplate_fallback',
})
if not failed and canonical(actual_tool) == 'web_search':
embedded_urls = search_embedded_article_urls(output)
if embedded_urls:
successful_web_retrievals += 1
for source_url in embedded_urls:
if source_url not in retrieved_web_sources:
retrieved_web_sources.append(source_url)
if result.get('evidence_status') != 'empty':
successful_web_searches += 1
for _title, source_url in web_source_links(
@@ -4890,19 +4925,34 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
successful_search_intents.append(successful_intent)
if successful_web_searches == 2 and not required_artifacts:
round_recovery_messages.append(
('Search already returned readable article content. Use that '
'evidence to synthesize the answer now with source URLs.'
if successful_web_retrievals else
'Research discovery is complete after two searches. Do not search '
'again. Retrieve the strongest authoritative result with web_fetch, '
'then answer every requested fact, comparison, and caveat with source URLs.'
'then answer every requested fact, comparison, and caveat with source URLs.')
)
browser_access_blocked = (
canonical(actual_tool) == 'private_browser'
and not failed
and browser_observation_access_blocked(output)
)
browser_page_missing = (
canonical(actual_tool) == 'private_browser'
and not failed
and browser_observation_page_missing(output)
)
if browser_page_missing:
round_recovery_messages.append(
'The browser displayed a missing-page error, not article evidence. '
'Use another exact source URL already returned by search. Do not '
'rewrite URL paths or claim this page was read successfully.'
)
if (
not failed
and canonical(actual_tool) in {'web_fetch', 'private_browser'}
and not browser_access_blocked
and not browser_page_missing
):
successful_web_retrievals += 1
for source_url in retrieved_source_urls(args):
+33
View File
@@ -1863,6 +1863,39 @@ def test_browser_access_gate_is_not_treated_as_page_evidence():
)
def test_rendered_missing_page_is_not_article_evidence():
from src.clean_agent_preview import browser_observation_page_missing
assert browser_observation_page_missing(
'- heading "Whoops!" [level=1]\n'
'- paragraph: This page doesn’t exist or can’t be found.'
)
assert browser_observation_page_missing('- heading "404 Page not found" [level=1]')
assert not browser_observation_page_missing(
'- heading "HTTP error handling" [level=1]\n'
'- paragraph: A 404 indicates a missing resource.'
)
def test_search_embedded_article_requires_readable_body_not_source_metadata():
from src.clean_agent_preview import search_embedded_article_urls
header = '[CONTENT 1] From: https://example.org/story\nTitle: Report\n------------------------------\n'
body = (
'The council published its transport review on Tuesday following a six month study. '
'Researchers counted journeys at twelve stations and interviewed residents about access. '
'Their findings showed that evening services were less reliable than morning departures. '
'Officials proposed additional buses on weekends while retaining existing train schedules. '
'The proposal will go through public consultation before any funding decision is made. '
'Several community groups welcomed the announcement but requested detailed cost estimates. '
'The report includes methodology, regional comparisons, and limitations of the passenger survey. '
'A further review is scheduled after the consultation closes next month.'
)
assert search_embedded_article_urls(header + body) == ['https://example.org/story']
assert search_embedded_article_urls('[1] Report\n https://example.org/story') == []
assert search_embedded_article_urls(header + 'Short snippet.') == []
assert search_embedded_article_urls(header + 'Verify you are human. ' + body) == []
def test_repeated_navigation_only_fetch_is_not_treated_as_page_evidence():
navigation = (
'World SECTIONS Politics Tech TOP STORIES Newsletter Sign In Subscribe '