ground current lookups and recover failed fetches

This commit is contained in:
pewdiepie-archdaemon
2026-09-17 19:42:38 +00:00
parent dfd4d5a80e
commit 8c7ea10520
4 changed files with 35 additions and 12 deletions
+6 -11
View File
@@ -4945,24 +4945,19 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
canonical(schema['function']['name']) == 'private_browser'
for schema in offered
)
and re.search(
r'\b(?:HTTP\s+(?:401|403|429)|access\s+(?:denied|blocked)|'
r'captcha|needs?\s+JS|login|no\s+readable\s+text|timed?\s*out)\b',
output,
re.I,
)
and retrieved_source_urls(args)
):
# Static fetchers are routinely rejected by publisher
# bot protection. That is a transport failure, not
# evidence that the source is unavailable. Offer one
# rendered-browser attempt at the same evidenced URL.
suppressed_tool_until_round['web_fetch'] = round_number + 1
fetch_url = str(args.get('url') or '').strip().rstrip('/')
if fetch_url:
static_fetch_failed_urls.add(fetch_url)
for fetch_url in retrieved_source_urls(args):
static_fetch_failed_urls.add(fetch_url.rstrip('/'))
force_private_browser_next_round = True
round_recovery_messages.append(
'The static page fetch was blocked or returned no readable content. '
'Use private_browser once to open the same known URL and inspect the '
'The static page fetch failed or returned no readable content. Use '
'private_browser once to open the strongest known URL and inspect the '
'rendered page; if that also fails, report the limitation without '
'inventing page content.'
)
+18
View File
@@ -671,6 +671,24 @@ def selected_tools_for_request(message: str) -> frozenset[str] | None:
}
if explicitly_named_web:
return frozenset(explicitly_named_web)
if (
re.search(r"\b(?:look\s*up|search|find)\b", text, re.I)
and re.search(
r"\b(?:current|latest|today(?:'s)?|right\s+now|this\s+(?:week|month|year))\b",
text,
re.I,
)
and not re.search(r"\bhttps?://", text, re.I)
and not re.match(
r"^\s*" + _REQUEST_PREFIX + r"(?:open|browse|visit|navigate|go\s+to)\b",
text,
re.I,
)
):
# Current lookups need discovery before navigation. Letting the model
# begin on an arbitrary browser page can ground an answer in stale or
# unrelated content without ever establishing a current source set.
return frozenset({"web_search"})
if re.search(
r"\buse\s+(?:the\s+)?(?:odysseus\s+)?web_search\b",
raw_text,