diff --git a/services/search/content.py b/services/search/content.py index b08476edf..c3948ba83 100644 --- a/services/search/content.py +++ b/services/search/content.py @@ -233,7 +233,7 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0, effective_cap = min(max_bytes or WEB_FETCH_SOFT_MAX_BYTES, WEB_FETCH_HARD_MAX_BYTES) # The cap is part of the cache identity: a truncated soft-cap fetch must # not be served to a later full-budget request for the same URL. - cache_key = generate_cache_key(f"{url}#cap={effective_cap}#extract=semantic-v2") + cache_key = generate_cache_key(f"{url}#cap={effective_cap}#extract=semantic-v3") cache_file = CONTENT_CACHE_DIR / f"{cache_key}.cache" # Check cache @@ -437,6 +437,20 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0, if len(body_text) > len(main_content): main_content = body_text + # HTTP 200 does not imply an article was retrieved. Classify only short + # interstitials with both a challenge title and corroborating body text; + # ordinary articles mentioning CAPTCHA must remain readable evidence. + challenge_title = title_text.strip().lower().rstrip('.!') + challenge_titles = {'client challenge', 'just a moment', 'security verification', 'verify you are human'} + if (challenge_title in challenge_titles and len(main_content) < 2000 + and re.search(r"required part of this site|verify (?:that )?you are human|checking your browser|enable javascript|security verification|performing security", main_content, re.I)): + return { + **_empty_result(url, 'Page access challenge: article content was not retrieved. Try private_browser or another authoritative source; do not treat the challenge page as evidence.'), + 'title': title_text, + 'error_kind': 'access_challenge', + **_size_fields, + } + result = { "url": url, "title": title_text, diff --git a/tests/test_search_content_extraction_parity.py b/tests/test_search_content_extraction_parity.py index a1157bd20..84a3ee405 100644 --- a/tests/test_search_content_extraction_parity.py +++ b/tests/test_search_content_extraction_parity.py @@ -36,6 +36,28 @@ class _FakeErrorResponse: ) +@pytest.mark.parametrize('title,body,blocked', [ + ('Client Challenge', 'A required part of this site couldn’t load. Try using a different browser.', True), + ('Just a moment...', 'Checking your browser. Enable JavaScript to continue.', True), + ('Understanding browser challenges', 'Checking your browser is a common security message.', False), + ('Client Challenge', 'An article about designing client challenges for a programming exercise.', False), + ('Client Challenge', 'A required part of this site ' + 'substantive discussion ' * 150, False), +]) +def test_access_interstitial_is_not_article_evidence(title, body, blocked, tmp_path, monkeypatch): + monkeypatch.setattr(service_content, 'CONTENT_CACHE_DIR', tmp_path) + html = f'{title}
{body}
' + monkeypatch.setattr(service_content, '_get_public_url', lambda *a, **k: _FakeResponse(html)) + result = service_content.fetch_webpage_content('https://example.com/challenge') + assert result['success'] is not blocked + if blocked: + assert result['content'] == '' + assert result['error_kind'] == 'access_challenge' + assert 'private_browser' in result['error'] + assert not list(tmp_path.iterdir()), 'Transient challenge must not be cached as article content' + else: + assert body.strip() in result['content'] + + @pytest.mark.parametrize('wrapper', ['main', 'article', 'div class="content"']) def test_extraction_does_not_repeat_nested_content_or_include_navigation(wrapper, tmp_path, monkeypatch): closing = wrapper.split()[0]