mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 06:52:20 +02:00
Separate blocked browser evidence from successful navigation and allow alternate sources
This commit is contained in:
@@ -3573,6 +3573,22 @@ def browser_observation_page_missing(raw):
|
||||
|
||||
def browser_observation_access_blocked(raw):
|
||||
"""Identify browser observations containing only an access gate."""
|
||||
try:
|
||||
decoded = json.loads(raw) if isinstance(raw, str) else raw
|
||||
except (TypeError, ValueError):
|
||||
decoded = None
|
||||
rows = decoded if isinstance(decoded, list) else [decoded]
|
||||
page_title = ''
|
||||
for row in rows:
|
||||
if not isinstance(row, dict):
|
||||
continue
|
||||
payload = row.get('result') if isinstance(row.get('result'), dict) else row
|
||||
# A new navigation supersedes the previous page title in a batch.
|
||||
if 'title' in payload:
|
||||
page_title = str(payload['title']).strip().casefold().rstrip('.!')
|
||||
if (page_title in {'client challenge', 'just a moment', 'security verification', 'verify you are human'}
|
||||
and str(payload.get('snapshot', '')).strip() == '(empty page)'):
|
||||
return True
|
||||
return bool(re.search(
|
||||
r"\b(?:captcha|access\s+(?:is\s+)?temporarily\s+restricted|access\s+denied|"
|
||||
r"verify\s+(?:that\s+)?you(?:\s+are|'re)\s+human|checking\s+your\s+browser|"
|
||||
@@ -5099,11 +5115,11 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
|
||||
if browser_url and browser_url in static_fetch_failed_urls:
|
||||
# Both independent transports have now failed for
|
||||
# this exact source. Do not bounce between them.
|
||||
force_no_tools_next_round = True
|
||||
round_recovery_messages.append(
|
||||
'Both static fetch and rendered browser access failed for this '
|
||||
'same URL. Do not retry either path; briefly report the source '
|
||||
'access limitation without inventing article content.'
|
||||
'same URL. Do not retry either path for this source. Use another '
|
||||
'relevant source already discovered if available; otherwise '
|
||||
'report the access limitation without inventing article content.'
|
||||
)
|
||||
else:
|
||||
# A CAPTCHA is transport output, not article
|
||||
|
||||
@@ -1900,6 +1900,21 @@ def test_browser_access_gate_is_not_treated_as_page_evidence():
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize('title,snapshot,blocked', [
|
||||
('Client Challenge', '(empty page)', True),
|
||||
('Just a moment...', '(empty page)', True),
|
||||
('Example documentation', '(empty page)', False),
|
||||
('Client Challenge', 'An article discussing client challenge design.', False),
|
||||
])
|
||||
def test_browser_challenge_title_with_empty_snapshot(title, snapshot, blocked):
|
||||
from src.clean_agent_preview import browser_observation_access_blocked
|
||||
raw = json.dumps([
|
||||
{'command': ['open', 'https://example.org'], 'result': {'title': title}, 'success': True},
|
||||
{'command': ['snapshot'], 'result': {'snapshot': snapshot}, 'success': True},
|
||||
])
|
||||
assert browser_observation_access_blocked(raw) is blocked
|
||||
|
||||
|
||||
def test_rendered_missing_page_is_not_article_evidence():
|
||||
from src.clean_agent_preview import browser_observation_page_missing
|
||||
|
||||
@@ -4185,7 +4200,11 @@ async def test_failed_static_fetch_recovers_once_through_rendered_browser(monkey
|
||||
'name': 'private_browser',
|
||||
'arguments': json.dumps({'action': 'open', 'url': 'https://www.reuters.com/example'}),
|
||||
}}]}}]},
|
||||
{'choices': [{'delta': {'content': 'Reuters blocked both access methods.'}}]},
|
||||
{'choices': [{'delta': {'tool_calls': [{'index': 0, 'id': 'alternative', 'function': {
|
||||
'name': 'web_fetch',
|
||||
'arguments': json.dumps({'url': 'https://example.org/report'}),
|
||||
}}]}}]},
|
||||
{'choices': [{'delta': {'content': 'Reuters blocked both access methods. The alternative report was readable.'}}]},
|
||||
])
|
||||
requests = []
|
||||
|
||||
@@ -4211,6 +4230,8 @@ async def test_failed_static_fetch_recovers_once_through_rendered_browser(monkey
|
||||
async def execute(block, **kwargs):
|
||||
executions.append(block.tool_type)
|
||||
if block.tool_type == 'web_fetch':
|
||||
if json.loads(block.content).get('url') == 'https://example.org/report':
|
||||
return 'web_fetch', {'output': 'Independent report with readable evidence.', 'exit_code': 0}
|
||||
return 'web_fetch', {'error': 'web_fetch: HTTP 401', 'exit_code': 1}
|
||||
return 'private_browser', {
|
||||
'output': 'Iframe "DataDome CAPTCHA" Access is temporarily restricted',
|
||||
@@ -4234,12 +4255,13 @@ async def test_failed_static_fetch_recovers_once_through_rendered_browser(monkey
|
||||
disabled_tools=set(), tool_policy=ToolPolicy(), max_rounds=4,
|
||||
)]
|
||||
|
||||
assert executions == ['web_fetch', 'private_browser']
|
||||
assert executions == ['web_fetch', 'private_browser', 'web_fetch']
|
||||
assert all(
|
||||
schema['function']['name'] != 'web_fetch'
|
||||
for schema in requests[1].get('tools', [])
|
||||
)
|
||||
assert 'tools' not in requests[2]
|
||||
assert any(schema['function']['name'] == 'web_fetch' for schema in requests[2]['tools'])
|
||||
assert any('Use another relevant source' in str(msg.get('content')) for msg in requests[2]['messages'])
|
||||
events = [json.loads(chunk[6:]) for chunk in raw if '[DONE]' not in chunk]
|
||||
assert any(event.get('type') == 'tool_loop_recovery' for event in events)
|
||||
assert any('blocked both access methods' in event.get('delta', '') for event in events)
|
||||
|
||||
Reference in New Issue
Block a user