From 8246eb87fc674bd971e4c43ab0558527ae50e222 Mon Sep 17 00:00:00 2001 From: pewdiepie-archdaemon Date: Thu, 17 Sep 2026 21:40:19 +0000 Subject: [PATCH] Prefer substantive article boundary over surrounding main-page boilerplate --- services/search/content.py | 11 ++++++++-- .../test_search_content_extraction_parity.py | 20 +++++++++++++++++++ 2 files changed, 29 insertions(+), 2 deletions(-) diff --git a/services/search/content.py b/services/search/content.py index c3948ba83..6b3434584 100644 --- a/services/search/content.py +++ b/services/search/content.py @@ -233,7 +233,7 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0, effective_cap = min(max_bytes or WEB_FETCH_SOFT_MAX_BYTES, WEB_FETCH_HARD_MAX_BYTES) # The cap is part of the cache identity: a truncated soft-cap fetch must # not be served to a later full-budget request for the same URL. - cache_key = generate_cache_key(f"{url}#cap={effective_cap}#extract=semantic-v3") + cache_key = generate_cache_key(f"{url}#cap={effective_cap}#extract=semantic-v4") cache_file = CONTENT_CACHE_DIR / f"{cache_key}.cache" # Check cache @@ -405,7 +405,14 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0, noise.extract() main_content = "" semantic_main = text_soup.find('main') or text_soup.find(attrs={'role': 'main'}) - content_areas = [semantic_main] if semantic_main else text_soup.find_all('article') + articles = semantic_main.find_all('article') if semantic_main else text_soup.find_all('article') + # A single substantive article is a more precise content boundary than + # main, which commonly also contains tags, related links and comment forms. + # Multiple article cards usually form a listing: keep its main context. + if semantic_main and len(articles) == 1 and len(articles[0].get_text(strip=True)) >= 200: + content_areas = articles + else: + content_areas = [semantic_main] if semantic_main else articles if not content_areas: content_areas = text_soup.find_all( ["section", "div"], diff --git a/tests/test_search_content_extraction_parity.py b/tests/test_search_content_extraction_parity.py index 84a3ee405..92f89d840 100644 --- a/tests/test_search_content_extraction_parity.py +++ b/tests/test_search_content_extraction_parity.py @@ -36,6 +36,26 @@ class _FakeErrorResponse: ) +def test_single_article_inside_main_excludes_related_links_and_comment_form(tmp_path, monkeypatch): + body = 'The measured storage comparison includes uncertainty and cost limitations. ' * 8 + html = f'

Storage comparison

{body}

Related posts: home battery storage tags
Leave a comment
' + monkeypatch.setattr(service_content, 'CONTENT_CACHE_DIR', tmp_path) + monkeypatch.setattr(service_content, '_get_public_url', lambda *a, **k: _FakeResponse(html)) + result = service_content.fetch_webpage_content('https://example.org/article') + assert body.strip() in result['content'] + assert 'Related posts' not in result['content'] + assert 'Leave a comment' not in result['content'] + + +def test_multiple_article_cards_retain_main_context(tmp_path, monkeypatch): + html = '

Search results for storage

First result
Second result
' + monkeypatch.setattr(service_content, 'CONTENT_CACHE_DIR', tmp_path) + monkeypatch.setattr(service_content, '_get_public_url', lambda *a, **k: _FakeResponse(html)) + result = service_content.fetch_webpage_content('https://example.org/listing') + assert 'Search results for storage' in result['content'] + assert 'First result' in result['content'] and 'Second result' in result['content'] + + @pytest.mark.parametrize('title,body,blocked', [ ('Client Challenge', 'A required part of this site couldn’t load. Try using a different browser.', True), ('Just a moment...', 'Checking your browser. Enable JavaScript to continue.', True),