mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 15:02:20 +02:00
Prefer substantive article boundary over surrounding main-page boilerplate
This commit is contained in:
@@ -233,7 +233,7 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
|
||||
effective_cap = min(max_bytes or WEB_FETCH_SOFT_MAX_BYTES, WEB_FETCH_HARD_MAX_BYTES)
|
||||
# The cap is part of the cache identity: a truncated soft-cap fetch must
|
||||
# not be served to a later full-budget request for the same URL.
|
||||
cache_key = generate_cache_key(f"{url}#cap={effective_cap}#extract=semantic-v3")
|
||||
cache_key = generate_cache_key(f"{url}#cap={effective_cap}#extract=semantic-v4")
|
||||
cache_file = CONTENT_CACHE_DIR / f"{cache_key}.cache"
|
||||
|
||||
# Check cache
|
||||
@@ -405,7 +405,14 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
|
||||
noise.extract()
|
||||
main_content = ""
|
||||
semantic_main = text_soup.find('main') or text_soup.find(attrs={'role': 'main'})
|
||||
content_areas = [semantic_main] if semantic_main else text_soup.find_all('article')
|
||||
articles = semantic_main.find_all('article') if semantic_main else text_soup.find_all('article')
|
||||
# A single substantive article is a more precise content boundary than
|
||||
# main, which commonly also contains tags, related links and comment forms.
|
||||
# Multiple article cards usually form a listing: keep its main context.
|
||||
if semantic_main and len(articles) == 1 and len(articles[0].get_text(strip=True)) >= 200:
|
||||
content_areas = articles
|
||||
else:
|
||||
content_areas = [semantic_main] if semantic_main else articles
|
||||
if not content_areas:
|
||||
content_areas = text_soup.find_all(
|
||||
["section", "div"],
|
||||
|
||||
@@ -36,6 +36,26 @@ class _FakeErrorResponse:
|
||||
)
|
||||
|
||||
|
||||
def test_single_article_inside_main_excludes_related_links_and_comment_form(tmp_path, monkeypatch):
|
||||
body = 'The measured storage comparison includes uncertainty and cost limitations. ' * 8
|
||||
html = f'<main><article><h1>Storage comparison</h1><p>{body}</p></article><section>Related posts: home battery storage tags</section><form>Leave a comment</form></main>'
|
||||
monkeypatch.setattr(service_content, 'CONTENT_CACHE_DIR', tmp_path)
|
||||
monkeypatch.setattr(service_content, '_get_public_url', lambda *a, **k: _FakeResponse(html))
|
||||
result = service_content.fetch_webpage_content('https://example.org/article')
|
||||
assert body.strip() in result['content']
|
||||
assert 'Related posts' not in result['content']
|
||||
assert 'Leave a comment' not in result['content']
|
||||
|
||||
|
||||
def test_multiple_article_cards_retain_main_context(tmp_path, monkeypatch):
|
||||
html = '<main><h1>Search results for storage</h1><article>First result</article><article>Second result</article></main>'
|
||||
monkeypatch.setattr(service_content, 'CONTENT_CACHE_DIR', tmp_path)
|
||||
monkeypatch.setattr(service_content, '_get_public_url', lambda *a, **k: _FakeResponse(html))
|
||||
result = service_content.fetch_webpage_content('https://example.org/listing')
|
||||
assert 'Search results for storage' in result['content']
|
||||
assert 'First result' in result['content'] and 'Second result' in result['content']
|
||||
|
||||
|
||||
@pytest.mark.parametrize('title,body,blocked', [
|
||||
('Client Challenge', 'A required part of this site couldn’t load. Try using a different browser.', True),
|
||||
('Just a moment...', 'Checking your browser. Enable JavaScript to continue.', True),
|
||||
|
||||
Reference in New Issue
Block a user