mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 06:52:20 +02:00
Preserve news query intent and extract nonduplicated semantic page content
This commit is contained in:
+21
-10
@@ -233,7 +233,7 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
|
||||
effective_cap = min(max_bytes or WEB_FETCH_SOFT_MAX_BYTES, WEB_FETCH_HARD_MAX_BYTES)
|
||||
# The cap is part of the cache identity: a truncated soft-cap fetch must
|
||||
# not be served to a later full-budget request for the same URL.
|
||||
cache_key = generate_cache_key(f"{url}#cap={effective_cap}")
|
||||
cache_key = generate_cache_key(f"{url}#cap={effective_cap}#extract=semantic-v2")
|
||||
cache_file = CONTENT_CACHE_DIR / f"{cache_key}.cache"
|
||||
|
||||
# Check cache
|
||||
@@ -398,15 +398,26 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
|
||||
js_rendered = _detect_js_frameworks(soup)
|
||||
js_message = "Page appears to be rendered by a JavaScript framework; content may be incomplete." if js_rendered else ""
|
||||
|
||||
# Main textual content (heuristic): prefer semantic / "content"-classed
|
||||
# containers to skip nav/footer/boilerplate; tuned for article pages.
|
||||
# Prefer semantic containers even without CSS classes. Work on a copy so
|
||||
# lists/tables and metadata extraction below still see the original DOM.
|
||||
text_soup = copy.copy(soup)
|
||||
for noise in text_soup.select('script, style, noscript, template, nav, footer, aside, [role="navigation"], [role="banner"], [role="contentinfo"]'):
|
||||
noise.extract()
|
||||
main_content = ""
|
||||
content_areas = soup.find_all(
|
||||
["main", "article", "section", "div"],
|
||||
class_=re.compile("content|main|body|article|post|entry|text", re.I),
|
||||
)
|
||||
semantic_main = text_soup.find('main') or text_soup.find(attrs={'role': 'main'})
|
||||
content_areas = [semantic_main] if semantic_main else text_soup.find_all('article')
|
||||
if not content_areas:
|
||||
content_areas = text_soup.find_all(
|
||||
["section", "div"],
|
||||
class_=re.compile("content|main|body|article|post|entry|text", re.I),
|
||||
)
|
||||
# Ancestor and child matches contain the same text. Emit each subtree once,
|
||||
# while retaining separate sibling articles/cards.
|
||||
candidate_ids = {id(area) for area in content_areas}
|
||||
content_areas = [area for area in content_areas
|
||||
if not any(id(parent) in candidate_ids for parent in area.parents)]
|
||||
if content_areas:
|
||||
for area in content_areas[:3]:
|
||||
for area in content_areas:
|
||||
main_content += area.get_text(separator=" ", strip=True) + " "
|
||||
main_content = re.sub(r"\s+", " ", main_content).strip()
|
||||
|
||||
@@ -414,8 +425,8 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
|
||||
# obvious boilerplate stripped so UI/deep-research search results do not
|
||||
# look empty for app/landing pages.
|
||||
THIN_CONTENT_CHARS = 600
|
||||
if len(main_content) < THIN_CONTENT_CHARS:
|
||||
body = soup.find("body")
|
||||
if len(main_content) < THIN_CONTENT_CHARS and not semantic_main:
|
||||
body = text_soup.find("body")
|
||||
if body:
|
||||
body_copy = copy.copy(body)
|
||||
for noise in body_copy.find_all(
|
||||
|
||||
@@ -3458,6 +3458,11 @@ def preserve_requested_web_recency(name, args, *, user_text='', prior_search_int
|
||||
if not query:
|
||||
return args
|
||||
normalized = dict(args)
|
||||
# Keep the user's explicit news intent when a model rewrites it to a
|
||||
# subject plus "today". Freshness alone does not select the news vertical.
|
||||
if (re.search(r'\b(?:news|neews|headlines)\b', user, re.I)
|
||||
and not re.search(r'\b(?:news|headlines)\b', query, re.I)):
|
||||
query += ' news'
|
||||
current_year = datetime.now(timezone.utc).year
|
||||
current_intent = bool(re.search(
|
||||
r"\b(?:latest|recent|current|today(?:'s)?|news|updates?|"
|
||||
|
||||
@@ -36,6 +36,23 @@ class _FakeErrorResponse:
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize('wrapper', ['main', 'article', 'div class="content"'])
|
||||
def test_extraction_does_not_repeat_nested_content_or_include_navigation(wrapper, tmp_path, monkeypatch):
|
||||
closing = wrapper.split()[0]
|
||||
html = (f'<html><body><nav>{"Navigation item " * 100}</nav><{wrapper}>'
|
||||
'<header>Article title and publication date</header>'
|
||||
'<div class="article-body"><div class="entry-content">'
|
||||
'<p>Unique substantive evidence.</p></div></div>'
|
||||
f'</{closing}><footer>Unrelated links</footer></body></html>')
|
||||
monkeypatch.setattr(service_content, 'CONTENT_CACHE_DIR', tmp_path)
|
||||
monkeypatch.setattr(service_content, '_get_public_url', lambda *a, **k: _FakeResponse(html))
|
||||
result = service_content.fetch_webpage_content('https://example.com/nested')
|
||||
assert result['content'].count('Unique substantive evidence.') == 1
|
||||
assert 'Navigation item' not in result['content']
|
||||
assert 'Unrelated links' not in result['content']
|
||||
assert 'Article title' in result['content']
|
||||
|
||||
|
||||
@pytest.mark.parametrize("module", [service_content])
|
||||
def test_content_fetcher_extracts_og_image_and_body_fallback(module, tmp_path, monkeypatch):
|
||||
html = """
|
||||
|
||||
@@ -1,4 +1,13 @@
|
||||
from src.clean_agent_preview import preview_tool_result_text
|
||||
from src.clean_agent_preview import preserve_requested_web_recency
|
||||
|
||||
|
||||
def test_news_intent_survives_query_rewording_without_changing_other_fresh_queries():
|
||||
result = preserve_requested_web_recency('web_search', {'query': 'artificial intelligence today'}, user_text='ai news today')
|
||||
assert result['query'] == 'artificial intelligence today news'
|
||||
assert result['time_filter'] == 'day'
|
||||
result = preserve_requested_web_recency('web_search', {'query': 'current browser privacy features'}, user_text='compare current browser privacy features')
|
||||
assert result['query'] == 'current browser privacy features'
|
||||
|
||||
|
||||
def test_all_fetched_sources_survive_observation_budget():
|
||||
|
||||
Reference in New Issue
Block a user