fix web fetch atom api recovery

This commit is contained in:
pewdiepie-archdaemon
2026-09-18 14:25:14 +00:00
parent 1d06fce37c
commit fd41ff8ce4
4 changed files with 59 additions and 3 deletions
+15 -1
View File
@@ -1,7 +1,7 @@
import asyncio
import json
from src.agent_tools.web_tools import WebFetchTool
from src.agent_tools.web_tools import WebFetchTool, _arxiv_listing_api_hint
from src.search import content as content_mod
from src.tool_schemas import FUNCTION_TOOL_SCHEMAS, function_call_to_tool_block
from src.clean_agent_preview import normalize_preview_function_args
@@ -79,6 +79,20 @@ def test_web_fetch_does_not_decode_local_html_or_binary_media(tmp_path):
assert "inspect_media" in video["error"]
def test_arxiv_list_406_offers_exact_public_api_recovery():
hint = _arxiv_listing_api_hint(
"https://arxiv.org/list/cs.CV?date=2026-02-25",
"HTTP 406: blocked",
)
assert "export.arxiv.org/api/query" in hint
assert "cat%3Acs.CV" in hint
assert "submittedDate%3A%5B202602250000" in hint
assert not _arxiv_listing_api_hint(
"https://example.com/list/cs.CV?date=2026-02-25", "HTTP 406"
)
def test_web_fetch_schema_and_native_parser_accept_urls_batch():
schema = next(
item["function"] for item in FUNCTION_TOOL_SCHEMAS
+10
View File
@@ -115,6 +115,16 @@ def test_body_under_cap_is_untouched(monkeypatch, no_cache):
assert r["fetched_bytes"] == len(b"hello world")
def test_atom_xml_api_response_is_preserved_as_readable_evidence(monkeypatch, no_cache):
body = b"<?xml version='1.0'?><feed><entry><title>Daily paper</title></entry></feed>"
_patch_stream(monkeypatch, _FakeStream(body, content_type="application/atom+xml"))
result = content_mod.fetch_webpage_content("https://example.com/api/feed")
assert result["success"] is True
assert "<title>Daily paper</title>" in result["content"]
def test_body_over_soft_cap_truncates_with_flags(monkeypatch, no_cache):
body = b"x" * (WEB_FETCH_SOFT_MAX_BYTES + 50_000)
_patch_stream(monkeypatch, _FakeStream(body, content_length=len(body)))