Preserve news query intent and extract nonduplicated semantic page content

This commit is contained in:
pewdiepie-archdaemon
2026-09-17 20:24:00 +00:00
parent 3edf7acd21
commit cfb9315e0a
4 changed files with 52 additions and 10 deletions
@@ -36,6 +36,23 @@ class _FakeErrorResponse:
)
@pytest.mark.parametrize('wrapper', ['main', 'article', 'div class="content"'])
def test_extraction_does_not_repeat_nested_content_or_include_navigation(wrapper, tmp_path, monkeypatch):
closing = wrapper.split()[0]
html = (f'<html><body><nav>{"Navigation item " * 100}</nav><{wrapper}>'
'<header>Article title and publication date</header>'
'<div class="article-body"><div class="entry-content">'
'<p>Unique substantive evidence.</p></div></div>'
f'</{closing}><footer>Unrelated links</footer></body></html>')
monkeypatch.setattr(service_content, 'CONTENT_CACHE_DIR', tmp_path)
monkeypatch.setattr(service_content, '_get_public_url', lambda *a, **k: _FakeResponse(html))
result = service_content.fetch_webpage_content('https://example.com/nested')
assert result['content'].count('Unique substantive evidence.') == 1
assert 'Navigation item' not in result['content']
assert 'Unrelated links' not in result['content']
assert 'Article title' in result['content']
@pytest.mark.parametrize("module", [service_content])
def test_content_fetcher_extracts_og_image_and_body_fallback(module, tmp_path, monkeypatch):
html = """
+9
View File
@@ -1,4 +1,13 @@
from src.clean_agent_preview import preview_tool_result_text
from src.clean_agent_preview import preserve_requested_web_recency
def test_news_intent_survives_query_rewording_without_changing_other_fresh_queries():
result = preserve_requested_web_recency('web_search', {'query': 'artificial intelligence today'}, user_text='ai news today')
assert result['query'] == 'artificial intelligence today news'
assert result['time_filter'] == 'day'
result = preserve_requested_web_recency('web_search', {'query': 'current browser privacy features'}, user_text='compare current browser privacy features')
assert result['query'] == 'current browser privacy features'
def test_all_fetched_sources_survive_observation_budget():