mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-07 15:32:21 +02:00
Preserve query-relevant page passages across search observation limits
This commit is contained in:
@@ -8,6 +8,7 @@ from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from datetime import datetime, timedelta
|
||||
from typing import Dict, Any, Optional, List, Set
|
||||
from urllib.parse import urlparse
|
||||
from src.search_passages import search_excerpt
|
||||
|
||||
import httpx
|
||||
|
||||
@@ -1108,9 +1109,7 @@ def comprehensive_web_search(
|
||||
output_parts.append(f"Title: {content['title']}")
|
||||
output_parts.append("-" * 30)
|
||||
|
||||
text = content["content"][:3000]
|
||||
if len(content["content"]) > 3000:
|
||||
text += "... [truncated]"
|
||||
text = search_excerpt(content["content"], provider_query, 3000)
|
||||
output_parts.append(text)
|
||||
|
||||
key_points = extract_key_points(content["content"])
|
||||
|
||||
@@ -190,7 +190,8 @@ def bounded_search_observation(output, budget=8000):
|
||||
body = output[match.end():end]
|
||||
body = re.split(r'\n(?:Key Points:|TL;DR:|Important Quotes:|Data / Statistics:|={20,}|<!-- SOURCES:)', body, maxsplit=1)[0].strip()
|
||||
if len(body) > per_page:
|
||||
body = body[:per_page - 15].rstrip() + '\n[...excerpt]'
|
||||
from src.search_passages import search_excerpt
|
||||
body = search_excerpt(body, query_match.group(0) if query_match else '', per_page)
|
||||
blocks.append(headers[index] + body)
|
||||
return prefix + '\n\n' + '\n\n'.join(blocks) + suffix
|
||||
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
"""Bounded, extractive search observations; never generate source claims."""
|
||||
import math
|
||||
import re
|
||||
|
||||
_FILLER = frozenset('the and for with from this that what which how find search compare explain latest current recent official source sources documentation document please about'.split())
|
||||
|
||||
|
||||
def search_excerpt(text: str, query: str, max_chars: int) -> str:
|
||||
"""Keep the opening plus relevant, non-overlapping literal page excerpts.
|
||||
|
||||
Offsets are selected from the original text, so qualifiers and negation
|
||||
within a passage are preserved. Explicit omission markers prevent these
|
||||
disjoint excerpts from masquerading as a continuous quotation.
|
||||
"""
|
||||
if len(text) <= max_chars:
|
||||
return text
|
||||
marker = '\n[...text omitted; fetch source for full context...]\n'
|
||||
if max_chars < 200:
|
||||
return text[:max_chars]
|
||||
clean_query = re.sub(r'(?<!\S)-?(?:site|filetype):\S+', '', query, flags=re.I)
|
||||
terms = list(dict.fromkeys(
|
||||
t.casefold() for t in re.findall(r'\w+', clean_query)
|
||||
if len(t) >= 2 and t.casefold() not in _FILLER and not t.isdigit()
|
||||
))[:24]
|
||||
patterns = [re.compile(r'\b' + re.escape(term) + r'\b', re.I) for term in terms]
|
||||
hits, counts = [], []
|
||||
for pattern in patterns:
|
||||
anchors, count = [], 0
|
||||
for match in pattern.finditer(text):
|
||||
count += 1
|
||||
if len(anchors) < 24:
|
||||
anchors.append(match)
|
||||
hits.append(anchors)
|
||||
counts.append(count)
|
||||
if not any(hits):
|
||||
return text[:max_chars - len(marker)] + marker
|
||||
# Preserve page-level scope/age disclaimers rather than showing only the
|
||||
# matching section. The rest of the budget is shared by up to two spans.
|
||||
lead_end = min(240, max_chars // 5)
|
||||
boundary = text.rfind(' ', 0, lead_end)
|
||||
if boundary > 0:
|
||||
lead_end = boundary
|
||||
remaining = max_chars - lead_end - 3 * len(marker)
|
||||
width = max(1, remaining // 2)
|
||||
weights = [1 / (1 + math.log1p(count)) for count in counts]
|
||||
candidates = {}
|
||||
for matches in hits:
|
||||
for match in matches[:24]:
|
||||
start = max(lead_end, min(len(text) - width, match.start() - width // 3))
|
||||
if start > lead_end:
|
||||
boundary = text.find(' ', start, min(len(text), start + 60))
|
||||
if boundary >= 0:
|
||||
start = boundary + 1
|
||||
end = min(len(text), start + width)
|
||||
boundary = text.rfind(' ', start, end)
|
||||
if boundary > start:
|
||||
end = boundary
|
||||
if end <= start:
|
||||
continue
|
||||
passage = text[start:end]
|
||||
coverage = [bool(pattern.search(passage)) for pattern in patterns]
|
||||
score = sum(weight for weight, matched in zip(weights, coverage) if matched)
|
||||
candidates[(start, end)] = (score, {i for i, matched in enumerate(coverage) if matched})
|
||||
selected = [(0, lead_end)]
|
||||
covered = set()
|
||||
for (start, end), (score, matched) in sorted(candidates.items(), key=lambda item: (-item[1][0], item[0][0])):
|
||||
if score <= 0 or any(start < b and end > a for a, b in selected):
|
||||
continue
|
||||
if selected[1:] and not matched - covered:
|
||||
continue
|
||||
selected.append((start, end))
|
||||
covered.update(matched)
|
||||
if len(selected) == 3:
|
||||
break
|
||||
if len(selected) == 2:
|
||||
# Spend unused space on context around the best passage, rather than
|
||||
# filling a second slot with a weaker repetition of the same terms.
|
||||
start, end = selected[1]
|
||||
end = min(len(text), start + max_chars - lead_end - 2 * len(marker))
|
||||
boundary = text.rfind(' ', start, end)
|
||||
if boundary > start:
|
||||
end = boundary
|
||||
selected[1] = (start, end)
|
||||
selected.sort()
|
||||
output = marker.join(text[start:end] for start, end in selected)
|
||||
if selected[-1][1] < len(text):
|
||||
output += marker
|
||||
return output[:max_chars]
|
||||
@@ -0,0 +1,44 @@
|
||||
from src.search_passages import search_excerpt
|
||||
|
||||
|
||||
def test_late_relevant_passage_survives_budget_and_preserves_negation():
|
||||
text = ('Archived overview; check current documentation. ' + 'Browser navigation and features. ' * 180
|
||||
+ 'Privacy Guide does not block every tracker. It explains cookie settings and their limitations. '
|
||||
+ 'Other unrelated material. ' * 100)
|
||||
result = search_excerpt(text, 'browser privacy cookie settings', 1200)
|
||||
assert len(result) <= 1200
|
||||
assert result.startswith('Archived overview;')
|
||||
assert 'Privacy Guide does not block every tracker.' in result
|
||||
assert 'text omitted' in result
|
||||
|
||||
|
||||
def test_short_evidence_is_not_modified():
|
||||
text = 'A small complete source.'
|
||||
assert search_excerpt(text, 'source', 500) == text
|
||||
|
||||
|
||||
def test_no_match_uses_bounded_explicitly_incomplete_opening():
|
||||
result = search_excerpt('Other material. ' * 300, 'privacy', 500)
|
||||
assert len(result) <= 500
|
||||
assert result.startswith('Other material.')
|
||||
assert 'text omitted' in result
|
||||
|
||||
|
||||
def test_distinct_topics_retain_original_order_and_literal_text():
|
||||
text = ('Introduction. ' + 'Filler. ' * 100 + 'Alpha evidence is limited. ' + 'Filler. ' * 200
|
||||
+ 'Beta evidence is preliminary. ' + 'Filler. ' * 200)
|
||||
result = search_excerpt(text, 'alpha beta', 1000)
|
||||
assert 'Alpha evidence is limited.' in result
|
||||
assert 'Beta evidence is preliminary.' in result
|
||||
assert result.index('Alpha evidence') < result.index('Beta evidence')
|
||||
|
||||
|
||||
def test_relevant_evidence_survives_both_search_and_observation_caps():
|
||||
text = 'Page introduction. ' + 'Navigation and performance features. ' * 180
|
||||
text += 'Privacy Guide explains how to choose cookie settings. These do not prevent every form of tracking. '
|
||||
text += 'Other features and links. ' * 200
|
||||
first = search_excerpt(text, 'browser privacy cookie settings', 3000)
|
||||
final = search_excerpt(first, 'browser privacy cookie settings', 1100)
|
||||
assert len(final) <= 1100
|
||||
assert 'Privacy Guide explains how to choose cookie settings.' in final
|
||||
assert 'These do not prevent every form of tracking.' in final
|
||||
Reference in New Issue
Block a user