mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 15:02:20 +02:00
fix web fetch atom api recovery
This commit is contained in:
@@ -358,11 +358,19 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
|
||||
# catches servers that mislabel text files as `application/octet-stream`.
|
||||
is_html = "html" in content_type
|
||||
is_json = "json" in content_type
|
||||
# Atom and XML are common public API formats (for example scholarly,
|
||||
# release, and government feeds). Parsing them through the HTML content
|
||||
# heuristic can yield an empty body even though the response contains
|
||||
# complete structured evidence. Preserve the source text so the caller
|
||||
# can inspect the fields or process it with workspace tools.
|
||||
is_xml = "xml" in content_type
|
||||
url_path = url.lower().split("?", 1)[0].split("#", 1)[0]
|
||||
looks_like_text_file = url_path.endswith(
|
||||
(".md", ".markdown", ".txt", ".text", ".json", ".jsonl")
|
||||
)
|
||||
if not is_html and (content_type.startswith("text/") or is_json or looks_like_text_file):
|
||||
if not is_html and (
|
||||
content_type.startswith("text/") or is_json or is_xml or looks_like_text_file
|
||||
):
|
||||
text_body = (response.text or "").strip()
|
||||
result = {
|
||||
"url": url,
|
||||
|
||||
@@ -207,6 +207,29 @@ def _looks_like_youtube_video_id(value: str) -> bool:
|
||||
return bool(re.fullmatch(r"[A-Za-z0-9_-]{10,16}", str(value or "").strip()))
|
||||
|
||||
|
||||
def _arxiv_listing_api_hint(url: str, error: str) -> str:
|
||||
"""Return an explicit recovery path for a blocked arXiv date listing."""
|
||||
if "406" not in str(error or ""):
|
||||
return ""
|
||||
parsed = urllib.parse.urlsplit(str(url or ""))
|
||||
if parsed.hostname not in {"arxiv.org", "www.arxiv.org"}:
|
||||
return ""
|
||||
match = re.fullmatch(r"/list/([A-Za-z0-9.-]+)", parsed.path.rstrip("/"))
|
||||
date = urllib.parse.parse_qs(parsed.query).get("date", [""])[0]
|
||||
if not match or not re.fullmatch(r"\d{4}-\d{2}-\d{2}", date):
|
||||
return ""
|
||||
day = date.replace("-", "")
|
||||
query = f"cat:{match.group(1)} AND submittedDate:[{day}0000 TO {day}2359]"
|
||||
api_url = "https://export.arxiv.org/api/query?" + urllib.parse.urlencode({
|
||||
"search_query": query, "start": "0", "max_results": "100",
|
||||
})
|
||||
return (
|
||||
" arXiv rejected its static listing (HTTP 406). The public Atom API is "
|
||||
f"available for this category/day; call web_fetch on {api_url} to read "
|
||||
"the dated feed, then inspect individual paper sources for details."
|
||||
)
|
||||
|
||||
|
||||
class WebSearchTool:
|
||||
async def execute(self, content: str, ctx: dict) -> dict:
|
||||
from src.search import comprehensive_web_search, searxng_search_results
|
||||
@@ -651,8 +674,9 @@ class WebFetchTool:
|
||||
|
||||
if not text:
|
||||
if err:
|
||||
arxiv_hint = _arxiv_listing_api_hint(url, str(err))
|
||||
return {
|
||||
"error": f"web_fetch: {url}: {err}",
|
||||
"error": f"web_fetch: {url}: {err}{arxiv_hint}",
|
||||
"exit_code": 1,
|
||||
"untrusted_content": True,
|
||||
}
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import asyncio
|
||||
import json
|
||||
|
||||
from src.agent_tools.web_tools import WebFetchTool
|
||||
from src.agent_tools.web_tools import WebFetchTool, _arxiv_listing_api_hint
|
||||
from src.search import content as content_mod
|
||||
from src.tool_schemas import FUNCTION_TOOL_SCHEMAS, function_call_to_tool_block
|
||||
from src.clean_agent_preview import normalize_preview_function_args
|
||||
@@ -79,6 +79,20 @@ def test_web_fetch_does_not_decode_local_html_or_binary_media(tmp_path):
|
||||
assert "inspect_media" in video["error"]
|
||||
|
||||
|
||||
def test_arxiv_list_406_offers_exact_public_api_recovery():
|
||||
hint = _arxiv_listing_api_hint(
|
||||
"https://arxiv.org/list/cs.CV?date=2026-02-25",
|
||||
"HTTP 406: blocked",
|
||||
)
|
||||
|
||||
assert "export.arxiv.org/api/query" in hint
|
||||
assert "cat%3Acs.CV" in hint
|
||||
assert "submittedDate%3A%5B202602250000" in hint
|
||||
assert not _arxiv_listing_api_hint(
|
||||
"https://example.com/list/cs.CV?date=2026-02-25", "HTTP 406"
|
||||
)
|
||||
|
||||
|
||||
def test_web_fetch_schema_and_native_parser_accept_urls_batch():
|
||||
schema = next(
|
||||
item["function"] for item in FUNCTION_TOOL_SCHEMAS
|
||||
|
||||
@@ -115,6 +115,16 @@ def test_body_under_cap_is_untouched(monkeypatch, no_cache):
|
||||
assert r["fetched_bytes"] == len(b"hello world")
|
||||
|
||||
|
||||
def test_atom_xml_api_response_is_preserved_as_readable_evidence(monkeypatch, no_cache):
|
||||
body = b"<?xml version='1.0'?><feed><entry><title>Daily paper</title></entry></feed>"
|
||||
_patch_stream(monkeypatch, _FakeStream(body, content_type="application/atom+xml"))
|
||||
|
||||
result = content_mod.fetch_webpage_content("https://example.com/api/feed")
|
||||
|
||||
assert result["success"] is True
|
||||
assert "<title>Daily paper</title>" in result["content"]
|
||||
|
||||
|
||||
def test_body_over_soft_cap_truncates_with_flags(monkeypatch, no_cache):
|
||||
body = b"x" * (WEB_FETCH_SOFT_MAX_BYTES + 50_000)
|
||||
_patch_stream(monkeypatch, _FakeStream(body, content_length=len(body)))
|
||||
|
||||
Reference in New Issue
Block a user