fix: stream content fetches so the 5MB cap aborts the download
ContentExtractor._fetch used client.get(), buffering the whole body in memory before the MAX_RESPONSE_BYTES check truncated it - the cap protected Trafilatura but not memory/bandwidth (a multi-hundred-MB URL was still fully downloaded, on up to max_urls_per_batch concurrent fetches, bounded only by the read timeout). Fetches now stream via client.stream + aiter_bytes and close the connection as soon as the cap is reached; charset still comes from the Content-Type header, available before the body is read. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01QbFZyDvYksazX6nYQYZ67L
This commit is contained in:
@@ -243,6 +243,90 @@ class TestContentExtractor:
|
||||
assert result.content == "Article content here."
|
||||
|
||||
|
||||
class TestFetchStreamingCap:
|
||||
"""The 5MB cap must abort the DOWNLOAD, not just truncate after it."""
|
||||
|
||||
def _extractor_with_transport(self, handler):
|
||||
extractor = ContentExtractor(timeout=5, max_length=2000)
|
||||
extractor._http = httpx.AsyncClient(
|
||||
transport=httpx.MockTransport(handler)
|
||||
)
|
||||
return extractor
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_download_aborts_past_cap(self):
|
||||
from src.clients.content_extractor import MAX_RESPONSE_BYTES
|
||||
|
||||
chunk = b"x" * (1024 * 1024) # 1MB per chunk
|
||||
chunks_produced = []
|
||||
|
||||
async def body():
|
||||
for i in range(100): # 100MB on offer
|
||||
chunks_produced.append(i)
|
||||
yield chunk
|
||||
|
||||
def handler(request):
|
||||
return httpx.Response(
|
||||
200,
|
||||
content=body(),
|
||||
headers={"Content-Type": "text/html; charset=utf-8"},
|
||||
)
|
||||
|
||||
extractor = self._extractor_with_transport(handler)
|
||||
try:
|
||||
text = await extractor._fetch("https://example.com/huge")
|
||||
finally:
|
||||
await extractor.close()
|
||||
|
||||
assert text is not None
|
||||
assert len(text.encode()) == MAX_RESPONSE_BYTES
|
||||
# Streaming stopped at the cap instead of consuming all 100 chunks
|
||||
assert len(chunks_produced) <= (MAX_RESPONSE_BYTES // len(chunk)) + 1
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_small_response_returned_whole(self):
|
||||
def handler(request):
|
||||
return httpx.Response(
|
||||
200,
|
||||
content=TEST_HTML.encode(),
|
||||
headers={"Content-Type": "text/html; charset=utf-8"},
|
||||
)
|
||||
|
||||
extractor = self._extractor_with_transport(handler)
|
||||
try:
|
||||
text = await extractor._fetch("https://example.com/small")
|
||||
finally:
|
||||
await extractor.close()
|
||||
|
||||
assert text == TEST_HTML
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_non_200_returns_none(self):
|
||||
def handler(request):
|
||||
return httpx.Response(404, content=b"not found")
|
||||
|
||||
extractor = self._extractor_with_transport(handler)
|
||||
try:
|
||||
text = await extractor._fetch("https://example.com/missing")
|
||||
finally:
|
||||
await extractor.close()
|
||||
|
||||
assert text is None
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_empty_body_returns_none(self):
|
||||
def handler(request):
|
||||
return httpx.Response(200, content=b"")
|
||||
|
||||
extractor = self._extractor_with_transport(handler)
|
||||
try:
|
||||
text = await extractor._fetch("https://example.com/empty")
|
||||
finally:
|
||||
await extractor.close()
|
||||
|
||||
assert text is None
|
||||
|
||||
|
||||
class TestContentExtractionResult:
|
||||
"""Tests for ContentExtractionResult model."""
|
||||
|
||||
|
||||
Reference in New Issue
Block a user