mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 15:02:20 +02:00
Report access interstitials as fetch failures, not source evidence
This commit is contained in:
@@ -233,7 +233,7 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
|
||||
effective_cap = min(max_bytes or WEB_FETCH_SOFT_MAX_BYTES, WEB_FETCH_HARD_MAX_BYTES)
|
||||
# The cap is part of the cache identity: a truncated soft-cap fetch must
|
||||
# not be served to a later full-budget request for the same URL.
|
||||
cache_key = generate_cache_key(f"{url}#cap={effective_cap}#extract=semantic-v2")
|
||||
cache_key = generate_cache_key(f"{url}#cap={effective_cap}#extract=semantic-v3")
|
||||
cache_file = CONTENT_CACHE_DIR / f"{cache_key}.cache"
|
||||
|
||||
# Check cache
|
||||
@@ -437,6 +437,20 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
|
||||
if len(body_text) > len(main_content):
|
||||
main_content = body_text
|
||||
|
||||
# HTTP 200 does not imply an article was retrieved. Classify only short
|
||||
# interstitials with both a challenge title and corroborating body text;
|
||||
# ordinary articles mentioning CAPTCHA must remain readable evidence.
|
||||
challenge_title = title_text.strip().lower().rstrip('.!')
|
||||
challenge_titles = {'client challenge', 'just a moment', 'security verification', 'verify you are human'}
|
||||
if (challenge_title in challenge_titles and len(main_content) < 2000
|
||||
and re.search(r"required part of this site|verify (?:that )?you are human|checking your browser|enable javascript|security verification|performing security", main_content, re.I)):
|
||||
return {
|
||||
**_empty_result(url, 'Page access challenge: article content was not retrieved. Try private_browser or another authoritative source; do not treat the challenge page as evidence.'),
|
||||
'title': title_text,
|
||||
'error_kind': 'access_challenge',
|
||||
**_size_fields,
|
||||
}
|
||||
|
||||
result = {
|
||||
"url": url,
|
||||
"title": title_text,
|
||||
|
||||
@@ -36,6 +36,28 @@ class _FakeErrorResponse:
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize('title,body,blocked', [
|
||||
('Client Challenge', 'A required part of this site couldn’t load. Try using a different browser.', True),
|
||||
('Just a moment...', 'Checking your browser. Enable JavaScript to continue.', True),
|
||||
('Understanding browser challenges', 'Checking your browser is a common security message.', False),
|
||||
('Client Challenge', 'An article about designing client challenges for a programming exercise.', False),
|
||||
('Client Challenge', 'A required part of this site ' + 'substantive discussion ' * 150, False),
|
||||
])
|
||||
def test_access_interstitial_is_not_article_evidence(title, body, blocked, tmp_path, monkeypatch):
|
||||
monkeypatch.setattr(service_content, 'CONTENT_CACHE_DIR', tmp_path)
|
||||
html = f'<html><title>{title}</title><body><main>{body}</main></body></html>'
|
||||
monkeypatch.setattr(service_content, '_get_public_url', lambda *a, **k: _FakeResponse(html))
|
||||
result = service_content.fetch_webpage_content('https://example.com/challenge')
|
||||
assert result['success'] is not blocked
|
||||
if blocked:
|
||||
assert result['content'] == ''
|
||||
assert result['error_kind'] == 'access_challenge'
|
||||
assert 'private_browser' in result['error']
|
||||
assert not list(tmp_path.iterdir()), 'Transient challenge must not be cached as article content'
|
||||
else:
|
||||
assert body.strip() in result['content']
|
||||
|
||||
|
||||
@pytest.mark.parametrize('wrapper', ['main', 'article', 'div class="content"'])
|
||||
def test_extraction_does_not_repeat_nested_content_or_include_navigation(wrapper, tmp_path, monkeypatch):
|
||||
closing = wrapper.split()[0]
|
||||
|
||||
Reference in New Issue
Block a user