diff --git a/services/search/content.py b/services/search/content.py
index 6b3434584..98c75543d 100644
--- a/services/search/content.py
+++ b/services/search/content.py
@@ -358,11 +358,19 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
# catches servers that mislabel text files as `application/octet-stream`.
is_html = "html" in content_type
is_json = "json" in content_type
+ # Atom and XML are common public API formats (for example scholarly,
+ # release, and government feeds). Parsing them through the HTML content
+ # heuristic can yield an empty body even though the response contains
+ # complete structured evidence. Preserve the source text so the caller
+ # can inspect the fields or process it with workspace tools.
+ is_xml = "xml" in content_type
url_path = url.lower().split("?", 1)[0].split("#", 1)[0]
looks_like_text_file = url_path.endswith(
(".md", ".markdown", ".txt", ".text", ".json", ".jsonl")
)
- if not is_html and (content_type.startswith("text/") or is_json or looks_like_text_file):
+ if not is_html and (
+ content_type.startswith("text/") or is_json or is_xml or looks_like_text_file
+ ):
text_body = (response.text or "").strip()
result = {
"url": url,
diff --git a/src/agent_tools/web_tools.py b/src/agent_tools/web_tools.py
index eb8891458..4bd36d96d 100644
--- a/src/agent_tools/web_tools.py
+++ b/src/agent_tools/web_tools.py
@@ -207,6 +207,29 @@ def _looks_like_youtube_video_id(value: str) -> bool:
return bool(re.fullmatch(r"[A-Za-z0-9_-]{10,16}", str(value or "").strip()))
+def _arxiv_listing_api_hint(url: str, error: str) -> str:
+ """Return an explicit recovery path for a blocked arXiv date listing."""
+ if "406" not in str(error or ""):
+ return ""
+ parsed = urllib.parse.urlsplit(str(url or ""))
+ if parsed.hostname not in {"arxiv.org", "www.arxiv.org"}:
+ return ""
+ match = re.fullmatch(r"/list/([A-Za-z0-9.-]+)", parsed.path.rstrip("/"))
+ date = urllib.parse.parse_qs(parsed.query).get("date", [""])[0]
+ if not match or not re.fullmatch(r"\d{4}-\d{2}-\d{2}", date):
+ return ""
+ day = date.replace("-", "")
+ query = f"cat:{match.group(1)} AND submittedDate:[{day}0000 TO {day}2359]"
+ api_url = "https://export.arxiv.org/api/query?" + urllib.parse.urlencode({
+ "search_query": query, "start": "0", "max_results": "100",
+ })
+ return (
+ " arXiv rejected its static listing (HTTP 406). The public Atom API is "
+ f"available for this category/day; call web_fetch on {api_url} to read "
+ "the dated feed, then inspect individual paper sources for details."
+ )
+
+
class WebSearchTool:
async def execute(self, content: str, ctx: dict) -> dict:
from src.search import comprehensive_web_search, searxng_search_results
@@ -651,8 +674,9 @@ class WebFetchTool:
if not text:
if err:
+ arxiv_hint = _arxiv_listing_api_hint(url, str(err))
return {
- "error": f"web_fetch: {url}: {err}",
+ "error": f"web_fetch: {url}: {err}{arxiv_hint}",
"exit_code": 1,
"untrusted_content": True,
}
diff --git a/tests/test_web_fetch_batch.py b/tests/test_web_fetch_batch.py
index 6b94e1c4a..bbb0575c8 100644
--- a/tests/test_web_fetch_batch.py
+++ b/tests/test_web_fetch_batch.py
@@ -1,7 +1,7 @@
import asyncio
import json
-from src.agent_tools.web_tools import WebFetchTool
+from src.agent_tools.web_tools import WebFetchTool, _arxiv_listing_api_hint
from src.search import content as content_mod
from src.tool_schemas import FUNCTION_TOOL_SCHEMAS, function_call_to_tool_block
from src.clean_agent_preview import normalize_preview_function_args
@@ -79,6 +79,20 @@ def test_web_fetch_does_not_decode_local_html_or_binary_media(tmp_path):
assert "inspect_media" in video["error"]
+def test_arxiv_list_406_offers_exact_public_api_recovery():
+ hint = _arxiv_listing_api_hint(
+ "https://arxiv.org/list/cs.CV?date=2026-02-25",
+ "HTTP 406: blocked",
+ )
+
+ assert "export.arxiv.org/api/query" in hint
+ assert "cat%3Acs.CV" in hint
+ assert "submittedDate%3A%5B202602250000" in hint
+ assert not _arxiv_listing_api_hint(
+ "https://example.com/list/cs.CV?date=2026-02-25", "HTTP 406"
+ )
+
+
def test_web_fetch_schema_and_native_parser_accept_urls_batch():
schema = next(
item["function"] for item in FUNCTION_TOOL_SCHEMAS
diff --git a/tests/test_web_fetch_size_caps.py b/tests/test_web_fetch_size_caps.py
index 1f1ad262a..a725a082a 100644
--- a/tests/test_web_fetch_size_caps.py
+++ b/tests/test_web_fetch_size_caps.py
@@ -115,6 +115,16 @@ def test_body_under_cap_is_untouched(monkeypatch, no_cache):
assert r["fetched_bytes"] == len(b"hello world")
+def test_atom_xml_api_response_is_preserved_as_readable_evidence(monkeypatch, no_cache):
+ body = b"