diff --git a/services/search/content.py b/services/search/content.py index 6b3434584..98c75543d 100644 --- a/services/search/content.py +++ b/services/search/content.py @@ -358,11 +358,19 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0, # catches servers that mislabel text files as `application/octet-stream`. is_html = "html" in content_type is_json = "json" in content_type + # Atom and XML are common public API formats (for example scholarly, + # release, and government feeds). Parsing them through the HTML content + # heuristic can yield an empty body even though the response contains + # complete structured evidence. Preserve the source text so the caller + # can inspect the fields or process it with workspace tools. + is_xml = "xml" in content_type url_path = url.lower().split("?", 1)[0].split("#", 1)[0] looks_like_text_file = url_path.endswith( (".md", ".markdown", ".txt", ".text", ".json", ".jsonl") ) - if not is_html and (content_type.startswith("text/") or is_json or looks_like_text_file): + if not is_html and ( + content_type.startswith("text/") or is_json or is_xml or looks_like_text_file + ): text_body = (response.text or "").strip() result = { "url": url, diff --git a/src/agent_tools/web_tools.py b/src/agent_tools/web_tools.py index eb8891458..4bd36d96d 100644 --- a/src/agent_tools/web_tools.py +++ b/src/agent_tools/web_tools.py @@ -207,6 +207,29 @@ def _looks_like_youtube_video_id(value: str) -> bool: return bool(re.fullmatch(r"[A-Za-z0-9_-]{10,16}", str(value or "").strip())) +def _arxiv_listing_api_hint(url: str, error: str) -> str: + """Return an explicit recovery path for a blocked arXiv date listing.""" + if "406" not in str(error or ""): + return "" + parsed = urllib.parse.urlsplit(str(url or "")) + if parsed.hostname not in {"arxiv.org", "www.arxiv.org"}: + return "" + match = re.fullmatch(r"/list/([A-Za-z0-9.-]+)", parsed.path.rstrip("/")) + date = urllib.parse.parse_qs(parsed.query).get("date", [""])[0] + if not match or not re.fullmatch(r"\d{4}-\d{2}-\d{2}", date): + return "" + day = date.replace("-", "") + query = f"cat:{match.group(1)} AND submittedDate:[{day}0000 TO {day}2359]" + api_url = "https://export.arxiv.org/api/query?" + urllib.parse.urlencode({ + "search_query": query, "start": "0", "max_results": "100", + }) + return ( + " arXiv rejected its static listing (HTTP 406). The public Atom API is " + f"available for this category/day; call web_fetch on {api_url} to read " + "the dated feed, then inspect individual paper sources for details." + ) + + class WebSearchTool: async def execute(self, content: str, ctx: dict) -> dict: from src.search import comprehensive_web_search, searxng_search_results @@ -651,8 +674,9 @@ class WebFetchTool: if not text: if err: + arxiv_hint = _arxiv_listing_api_hint(url, str(err)) return { - "error": f"web_fetch: {url}: {err}", + "error": f"web_fetch: {url}: {err}{arxiv_hint}", "exit_code": 1, "untrusted_content": True, } diff --git a/tests/test_web_fetch_batch.py b/tests/test_web_fetch_batch.py index 6b94e1c4a..bbb0575c8 100644 --- a/tests/test_web_fetch_batch.py +++ b/tests/test_web_fetch_batch.py @@ -1,7 +1,7 @@ import asyncio import json -from src.agent_tools.web_tools import WebFetchTool +from src.agent_tools.web_tools import WebFetchTool, _arxiv_listing_api_hint from src.search import content as content_mod from src.tool_schemas import FUNCTION_TOOL_SCHEMAS, function_call_to_tool_block from src.clean_agent_preview import normalize_preview_function_args @@ -79,6 +79,20 @@ def test_web_fetch_does_not_decode_local_html_or_binary_media(tmp_path): assert "inspect_media" in video["error"] +def test_arxiv_list_406_offers_exact_public_api_recovery(): + hint = _arxiv_listing_api_hint( + "https://arxiv.org/list/cs.CV?date=2026-02-25", + "HTTP 406: blocked", + ) + + assert "export.arxiv.org/api/query" in hint + assert "cat%3Acs.CV" in hint + assert "submittedDate%3A%5B202602250000" in hint + assert not _arxiv_listing_api_hint( + "https://example.com/list/cs.CV?date=2026-02-25", "HTTP 406" + ) + + def test_web_fetch_schema_and_native_parser_accept_urls_batch(): schema = next( item["function"] for item in FUNCTION_TOOL_SCHEMAS diff --git a/tests/test_web_fetch_size_caps.py b/tests/test_web_fetch_size_caps.py index 1f1ad262a..a725a082a 100644 --- a/tests/test_web_fetch_size_caps.py +++ b/tests/test_web_fetch_size_caps.py @@ -115,6 +115,16 @@ def test_body_under_cap_is_untouched(monkeypatch, no_cache): assert r["fetched_bytes"] == len(b"hello world") +def test_atom_xml_api_response_is_preserved_as_readable_evidence(monkeypatch, no_cache): + body = b"Daily paper" + _patch_stream(monkeypatch, _FakeStream(body, content_type="application/atom+xml")) + + result = content_mod.fetch_webpage_content("https://example.com/api/feed") + + assert result["success"] is True + assert "Daily paper" in result["content"] + + def test_body_over_soft_cap_truncates_with_flags(monkeypatch, no_cache): body = b"x" * (WEB_FETCH_SOFT_MAX_BYTES + 50_000) _patch_stream(monkeypatch, _FakeStream(body, content_length=len(body)))