fix web fetch atom api recovery

This commit is contained in:
pewdiepie-archdaemon
2026-09-18 14:25:14 +00:00
parent 1d06fce37c
commit fd41ff8ce4
4 changed files with 59 additions and 3 deletions
+9 -1
View File
@@ -358,11 +358,19 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
# catches servers that mislabel text files as `application/octet-stream`.
is_html = "html" in content_type
is_json = "json" in content_type
# Atom and XML are common public API formats (for example scholarly,
# release, and government feeds). Parsing them through the HTML content
# heuristic can yield an empty body even though the response contains
# complete structured evidence. Preserve the source text so the caller
# can inspect the fields or process it with workspace tools.
is_xml = "xml" in content_type
url_path = url.lower().split("?", 1)[0].split("#", 1)[0]
looks_like_text_file = url_path.endswith(
(".md", ".markdown", ".txt", ".text", ".json", ".jsonl")
)
if not is_html and (content_type.startswith("text/") or is_json or looks_like_text_file):
if not is_html and (
content_type.startswith("text/") or is_json or is_xml or looks_like_text_file
):
text_body = (response.text or "").strip()
result = {
"url": url,
+25 -1
View File
@@ -207,6 +207,29 @@ def _looks_like_youtube_video_id(value: str) -> bool:
return bool(re.fullmatch(r"[A-Za-z0-9_-]{10,16}", str(value or "").strip()))
def _arxiv_listing_api_hint(url: str, error: str) -> str:
"""Return an explicit recovery path for a blocked arXiv date listing."""
if "406" not in str(error or ""):
return ""
parsed = urllib.parse.urlsplit(str(url or ""))
if parsed.hostname not in {"arxiv.org", "www.arxiv.org"}:
return ""
match = re.fullmatch(r"/list/([A-Za-z0-9.-]+)", parsed.path.rstrip("/"))
date = urllib.parse.parse_qs(parsed.query).get("date", [""])[0]
if not match or not re.fullmatch(r"\d{4}-\d{2}-\d{2}", date):
return ""
day = date.replace("-", "")
query = f"cat:{match.group(1)} AND submittedDate:[{day}0000 TO {day}2359]"
api_url = "https://export.arxiv.org/api/query?" + urllib.parse.urlencode({
"search_query": query, "start": "0", "max_results": "100",
})
return (
" arXiv rejected its static listing (HTTP 406). The public Atom API is "
f"available for this category/day; call web_fetch on {api_url} to read "
"the dated feed, then inspect individual paper sources for details."
)
class WebSearchTool:
async def execute(self, content: str, ctx: dict) -> dict:
from src.search import comprehensive_web_search, searxng_search_results
@@ -651,8 +674,9 @@ class WebFetchTool:
if not text:
if err:
arxiv_hint = _arxiv_listing_api_hint(url, str(err))
return {
"error": f"web_fetch: {url}: {err}",
"error": f"web_fetch: {url}: {err}{arxiv_hint}",
"exit_code": 1,
"untrusted_content": True,
}
+15 -1
View File
@@ -1,7 +1,7 @@
import asyncio
import json
from src.agent_tools.web_tools import WebFetchTool
from src.agent_tools.web_tools import WebFetchTool, _arxiv_listing_api_hint
from src.search import content as content_mod
from src.tool_schemas import FUNCTION_TOOL_SCHEMAS, function_call_to_tool_block
from src.clean_agent_preview import normalize_preview_function_args
@@ -79,6 +79,20 @@ def test_web_fetch_does_not_decode_local_html_or_binary_media(tmp_path):
assert "inspect_media" in video["error"]
def test_arxiv_list_406_offers_exact_public_api_recovery():
hint = _arxiv_listing_api_hint(
"https://arxiv.org/list/cs.CV?date=2026-02-25",
"HTTP 406: blocked",
)
assert "export.arxiv.org/api/query" in hint
assert "cat%3Acs.CV" in hint
assert "submittedDate%3A%5B202602250000" in hint
assert not _arxiv_listing_api_hint(
"https://example.com/list/cs.CV?date=2026-02-25", "HTTP 406"
)
def test_web_fetch_schema_and_native_parser_accept_urls_batch():
schema = next(
item["function"] for item in FUNCTION_TOOL_SCHEMAS
+10
View File
@@ -115,6 +115,16 @@ def test_body_under_cap_is_untouched(monkeypatch, no_cache):
assert r["fetched_bytes"] == len(b"hello world")
def test_atom_xml_api_response_is_preserved_as_readable_evidence(monkeypatch, no_cache):
body = b"<?xml version='1.0'?><feed><entry><title>Daily paper</title></entry></feed>"
_patch_stream(monkeypatch, _FakeStream(body, content_type="application/atom+xml"))
result = content_mod.fetch_webpage_content("https://example.com/api/feed")
assert result["success"] is True
assert "<title>Daily paper</title>" in result["content"]
def test_body_over_soft_cap_truncates_with_flags(monkeypatch, no_cache):
body = b"x" * (WEB_FETCH_SOFT_MAX_BYTES + 50_000)
_patch_stream(monkeypatch, _FakeStream(body, content_length=len(body)))