Report access interstitials as fetch failures, not source evidence

This commit is contained in:
pewdiepie-archdaemon
2026-09-17 21:12:53 +00:00
parent 674bc01f3a
commit 7a11de9c7a
2 changed files with 37 additions and 1 deletions
+15 -1
View File
@@ -233,7 +233,7 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
effective_cap = min(max_bytes or WEB_FETCH_SOFT_MAX_BYTES, WEB_FETCH_HARD_MAX_BYTES)
# The cap is part of the cache identity: a truncated soft-cap fetch must
# not be served to a later full-budget request for the same URL.
cache_key = generate_cache_key(f"{url}#cap={effective_cap}#extract=semantic-v2")
cache_key = generate_cache_key(f"{url}#cap={effective_cap}#extract=semantic-v3")
cache_file = CONTENT_CACHE_DIR / f"{cache_key}.cache"
# Check cache
@@ -437,6 +437,20 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
if len(body_text) > len(main_content):
main_content = body_text
# HTTP 200 does not imply an article was retrieved. Classify only short
# interstitials with both a challenge title and corroborating body text;
# ordinary articles mentioning CAPTCHA must remain readable evidence.
challenge_title = title_text.strip().lower().rstrip('.!')
challenge_titles = {'client challenge', 'just a moment', 'security verification', 'verify you are human'}
if (challenge_title in challenge_titles and len(main_content) < 2000
and re.search(r"required part of this site|verify (?:that )?you are human|checking your browser|enable javascript|security verification|performing security", main_content, re.I)):
return {
**_empty_result(url, 'Page access challenge: article content was not retrieved. Try private_browser or another authoritative source; do not treat the challenge page as evidence.'),
'title': title_text,
'error_kind': 'access_challenge',
**_size_fields,
}
result = {
"url": url,
"title": title_text,
@@ -36,6 +36,28 @@ class _FakeErrorResponse:
)
@pytest.mark.parametrize('title,body,blocked', [
('Client Challenge', 'A required part of this site couldn’t load. Try using a different browser.', True),
('Just a moment...', 'Checking your browser. Enable JavaScript to continue.', True),
('Understanding browser challenges', 'Checking your browser is a common security message.', False),
('Client Challenge', 'An article about designing client challenges for a programming exercise.', False),
('Client Challenge', 'A required part of this site ' + 'substantive discussion ' * 150, False),
])
def test_access_interstitial_is_not_article_evidence(title, body, blocked, tmp_path, monkeypatch):
monkeypatch.setattr(service_content, 'CONTENT_CACHE_DIR', tmp_path)
html = f'<html><title>{title}</title><body><main>{body}</main></body></html>'
monkeypatch.setattr(service_content, '_get_public_url', lambda *a, **k: _FakeResponse(html))
result = service_content.fetch_webpage_content('https://example.com/challenge')
assert result['success'] is not blocked
if blocked:
assert result['content'] == ''
assert result['error_kind'] == 'access_challenge'
assert 'private_browser' in result['error']
assert not list(tmp_path.iterdir()), 'Transient challenge must not be cached as article content'
else:
assert body.strip() in result['content']
@pytest.mark.parametrize('wrapper', ['main', 'article', 'div class="content"'])
def test_extraction_does_not_repeat_nested_content_or_include_navigation(wrapper, tmp_path, monkeypatch):
closing = wrapper.split()[0]