mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 15:02:20 +02:00
reuse embedded search articles and reject browser error pages
This commit is contained in:
@@ -1993,6 +1993,25 @@ def bounded_research_tool_policy(offered, *, searches=0, retrievals=0, search_li
|
||||
return schemas, None, True
|
||||
|
||||
|
||||
def search_embedded_article_urls(output):
|
||||
"""Identify substantial article bodies actually delivered in search output."""
|
||||
text = str(output or '')
|
||||
urls = []
|
||||
for match in re.finditer(
|
||||
r'\[CONTENT(?: \d+)?\] From: (https?://\S+)\nTitle: [^\n]*\n-+\n'
|
||||
r'([\s\S]*?)(?=\n\[CONTENT|\n(?:Key Points:|TL;DR:|Important Quotes:|Data / Statistics:|={5,})|\Z)',
|
||||
text,
|
||||
):
|
||||
url, body = match.groups()
|
||||
if (len(body.split()) >= 80
|
||||
and not browser_observation_access_blocked(body)
|
||||
and not browser_observation_page_missing(body)
|
||||
and not web_fetch_observation_is_boilerplate(body)):
|
||||
if url not in urls:
|
||||
urls.append(url)
|
||||
return urls
|
||||
|
||||
|
||||
def retrieved_source_urls(arguments):
|
||||
"""Return explicit HTTP(S) sources actually passed to a retrieval tool."""
|
||||
if not isinstance(arguments, dict):
|
||||
@@ -3492,6 +3511,16 @@ def private_browser_effective_url(result):
|
||||
return urls[-1] if urls else ''
|
||||
|
||||
|
||||
def browser_observation_page_missing(raw):
|
||||
"""Recognize a rendered error page, rather than an article mentioning 404."""
|
||||
text = str(raw or '')
|
||||
return bool(re.search(
|
||||
r"(?:heading[^\n]{0,120}(?:Whoops!|Page not found|404)|"
|
||||
r"This page doesn[’']t exist or can[’']t be found\.)",
|
||||
text, re.I,
|
||||
))
|
||||
|
||||
|
||||
def browser_observation_access_blocked(raw):
|
||||
"""Identify browser observations containing only an access gate."""
|
||||
return bool(re.search(
|
||||
@@ -4878,6 +4907,12 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
|
||||
'reason': 'web_fetch_boilerplate_fallback',
|
||||
})
|
||||
if not failed and canonical(actual_tool) == 'web_search':
|
||||
embedded_urls = search_embedded_article_urls(output)
|
||||
if embedded_urls:
|
||||
successful_web_retrievals += 1
|
||||
for source_url in embedded_urls:
|
||||
if source_url not in retrieved_web_sources:
|
||||
retrieved_web_sources.append(source_url)
|
||||
if result.get('evidence_status') != 'empty':
|
||||
successful_web_searches += 1
|
||||
for _title, source_url in web_source_links(
|
||||
@@ -4890,19 +4925,34 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
|
||||
successful_search_intents.append(successful_intent)
|
||||
if successful_web_searches == 2 and not required_artifacts:
|
||||
round_recovery_messages.append(
|
||||
('Search already returned readable article content. Use that '
|
||||
'evidence to synthesize the answer now with source URLs.'
|
||||
if successful_web_retrievals else
|
||||
'Research discovery is complete after two searches. Do not search '
|
||||
'again. Retrieve the strongest authoritative result with web_fetch, '
|
||||
'then answer every requested fact, comparison, and caveat with source URLs.'
|
||||
'then answer every requested fact, comparison, and caveat with source URLs.')
|
||||
)
|
||||
browser_access_blocked = (
|
||||
canonical(actual_tool) == 'private_browser'
|
||||
and not failed
|
||||
and browser_observation_access_blocked(output)
|
||||
)
|
||||
browser_page_missing = (
|
||||
canonical(actual_tool) == 'private_browser'
|
||||
and not failed
|
||||
and browser_observation_page_missing(output)
|
||||
)
|
||||
if browser_page_missing:
|
||||
round_recovery_messages.append(
|
||||
'The browser displayed a missing-page error, not article evidence. '
|
||||
'Use another exact source URL already returned by search. Do not '
|
||||
'rewrite URL paths or claim this page was read successfully.'
|
||||
)
|
||||
if (
|
||||
not failed
|
||||
and canonical(actual_tool) in {'web_fetch', 'private_browser'}
|
||||
and not browser_access_blocked
|
||||
and not browser_page_missing
|
||||
):
|
||||
successful_web_retrievals += 1
|
||||
for source_url in retrieved_source_urls(args):
|
||||
|
||||
@@ -1863,6 +1863,39 @@ def test_browser_access_gate_is_not_treated_as_page_evidence():
|
||||
)
|
||||
|
||||
|
||||
def test_rendered_missing_page_is_not_article_evidence():
|
||||
from src.clean_agent_preview import browser_observation_page_missing
|
||||
|
||||
assert browser_observation_page_missing(
|
||||
'- heading "Whoops!" [level=1]\n'
|
||||
'- paragraph: This page doesn’t exist or can’t be found.'
|
||||
)
|
||||
assert browser_observation_page_missing('- heading "404 Page not found" [level=1]')
|
||||
assert not browser_observation_page_missing(
|
||||
'- heading "HTTP error handling" [level=1]\n'
|
||||
'- paragraph: A 404 indicates a missing resource.'
|
||||
)
|
||||
|
||||
|
||||
def test_search_embedded_article_requires_readable_body_not_source_metadata():
|
||||
from src.clean_agent_preview import search_embedded_article_urls
|
||||
header = '[CONTENT 1] From: https://example.org/story\nTitle: Report\n------------------------------\n'
|
||||
body = (
|
||||
'The council published its transport review on Tuesday following a six month study. '
|
||||
'Researchers counted journeys at twelve stations and interviewed residents about access. '
|
||||
'Their findings showed that evening services were less reliable than morning departures. '
|
||||
'Officials proposed additional buses on weekends while retaining existing train schedules. '
|
||||
'The proposal will go through public consultation before any funding decision is made. '
|
||||
'Several community groups welcomed the announcement but requested detailed cost estimates. '
|
||||
'The report includes methodology, regional comparisons, and limitations of the passenger survey. '
|
||||
'A further review is scheduled after the consultation closes next month.'
|
||||
)
|
||||
assert search_embedded_article_urls(header + body) == ['https://example.org/story']
|
||||
assert search_embedded_article_urls('[1] Report\n https://example.org/story') == []
|
||||
assert search_embedded_article_urls(header + 'Short snippet.') == []
|
||||
assert search_embedded_article_urls(header + 'Verify you are human. ' + body) == []
|
||||
|
||||
|
||||
def test_repeated_navigation_only_fetch_is_not_treated_as_page_evidence():
|
||||
navigation = (
|
||||
'World SECTIONS Politics Tech TOP STORIES Newsletter Sign In Subscribe '
|
||||
|
||||
Reference in New Issue
Block a user