Preserve query-relevant page passages across search observation limits

This commit is contained in:
pewdiepie-archdaemon
2026-09-17 21:04:57 +00:00
parent 5a5a2c5195
commit 31e03c9eb7
4 changed files with 136 additions and 4 deletions
+2 -3
View File
@@ -8,6 +8,7 @@ from concurrent.futures import ThreadPoolExecutor, as_completed
from datetime import datetime, timedelta
from typing import Dict, Any, Optional, List, Set
from urllib.parse import urlparse
from src.search_passages import search_excerpt
import httpx
@@ -1108,9 +1109,7 @@ def comprehensive_web_search(
output_parts.append(f"Title: {content['title']}")
output_parts.append("-" * 30)
text = content["content"][:3000]
if len(content["content"]) > 3000:
text += "... [truncated]"
text = search_excerpt(content["content"], provider_query, 3000)
output_parts.append(text)
key_points = extract_key_points(content["content"])
+2 -1
View File
@@ -190,7 +190,8 @@ def bounded_search_observation(output, budget=8000):
body = output[match.end():end]
body = re.split(r'\n(?:Key Points:|TL;DR:|Important Quotes:|Data / Statistics:|={20,}|<!-- SOURCES:)', body, maxsplit=1)[0].strip()
if len(body) > per_page:
body = body[:per_page - 15].rstrip() + '\n[...excerpt]'
from src.search_passages import search_excerpt
body = search_excerpt(body, query_match.group(0) if query_match else '', per_page)
blocks.append(headers[index] + body)
return prefix + '\n\n' + '\n\n'.join(blocks) + suffix
+88
View File
@@ -0,0 +1,88 @@
"""Bounded, extractive search observations; never generate source claims."""
import math
import re
_FILLER = frozenset('the and for with from this that what which how find search compare explain latest current recent official source sources documentation document please about'.split())
def search_excerpt(text: str, query: str, max_chars: int) -> str:
"""Keep the opening plus relevant, non-overlapping literal page excerpts.
Offsets are selected from the original text, so qualifiers and negation
within a passage are preserved. Explicit omission markers prevent these
disjoint excerpts from masquerading as a continuous quotation.
"""
if len(text) <= max_chars:
return text
marker = '\n[...text omitted; fetch source for full context...]\n'
if max_chars < 200:
return text[:max_chars]
clean_query = re.sub(r'(?<!\S)-?(?:site|filetype):\S+', '', query, flags=re.I)
terms = list(dict.fromkeys(
t.casefold() for t in re.findall(r'\w+', clean_query)
if len(t) >= 2 and t.casefold() not in _FILLER and not t.isdigit()
))[:24]
patterns = [re.compile(r'\b' + re.escape(term) + r'\b', re.I) for term in terms]
hits, counts = [], []
for pattern in patterns:
anchors, count = [], 0
for match in pattern.finditer(text):
count += 1
if len(anchors) < 24:
anchors.append(match)
hits.append(anchors)
counts.append(count)
if not any(hits):
return text[:max_chars - len(marker)] + marker
# Preserve page-level scope/age disclaimers rather than showing only the
# matching section. The rest of the budget is shared by up to two spans.
lead_end = min(240, max_chars // 5)
boundary = text.rfind(' ', 0, lead_end)
if boundary > 0:
lead_end = boundary
remaining = max_chars - lead_end - 3 * len(marker)
width = max(1, remaining // 2)
weights = [1 / (1 + math.log1p(count)) for count in counts]
candidates = {}
for matches in hits:
for match in matches[:24]:
start = max(lead_end, min(len(text) - width, match.start() - width // 3))
if start > lead_end:
boundary = text.find(' ', start, min(len(text), start + 60))
if boundary >= 0:
start = boundary + 1
end = min(len(text), start + width)
boundary = text.rfind(' ', start, end)
if boundary > start:
end = boundary
if end <= start:
continue
passage = text[start:end]
coverage = [bool(pattern.search(passage)) for pattern in patterns]
score = sum(weight for weight, matched in zip(weights, coverage) if matched)
candidates[(start, end)] = (score, {i for i, matched in enumerate(coverage) if matched})
selected = [(0, lead_end)]
covered = set()
for (start, end), (score, matched) in sorted(candidates.items(), key=lambda item: (-item[1][0], item[0][0])):
if score <= 0 or any(start < b and end > a for a, b in selected):
continue
if selected[1:] and not matched - covered:
continue
selected.append((start, end))
covered.update(matched)
if len(selected) == 3:
break
if len(selected) == 2:
# Spend unused space on context around the best passage, rather than
# filling a second slot with a weaker repetition of the same terms.
start, end = selected[1]
end = min(len(text), start + max_chars - lead_end - 2 * len(marker))
boundary = text.rfind(' ', start, end)
if boundary > start:
end = boundary
selected[1] = (start, end)
selected.sort()
output = marker.join(text[start:end] for start, end in selected)
if selected[-1][1] < len(text):
output += marker
return output[:max_chars]
+44
View File
@@ -0,0 +1,44 @@
from src.search_passages import search_excerpt
def test_late_relevant_passage_survives_budget_and_preserves_negation():
text = ('Archived overview; check current documentation. ' + 'Browser navigation and features. ' * 180
+ 'Privacy Guide does not block every tracker. It explains cookie settings and their limitations. '
+ 'Other unrelated material. ' * 100)
result = search_excerpt(text, 'browser privacy cookie settings', 1200)
assert len(result) <= 1200
assert result.startswith('Archived overview;')
assert 'Privacy Guide does not block every tracker.' in result
assert 'text omitted' in result
def test_short_evidence_is_not_modified():
text = 'A small complete source.'
assert search_excerpt(text, 'source', 500) == text
def test_no_match_uses_bounded_explicitly_incomplete_opening():
result = search_excerpt('Other material. ' * 300, 'privacy', 500)
assert len(result) <= 500
assert result.startswith('Other material.')
assert 'text omitted' in result
def test_distinct_topics_retain_original_order_and_literal_text():
text = ('Introduction. ' + 'Filler. ' * 100 + 'Alpha evidence is limited. ' + 'Filler. ' * 200
+ 'Beta evidence is preliminary. ' + 'Filler. ' * 200)
result = search_excerpt(text, 'alpha beta', 1000)
assert 'Alpha evidence is limited.' in result
assert 'Beta evidence is preliminary.' in result
assert result.index('Alpha evidence') < result.index('Beta evidence')
def test_relevant_evidence_survives_both_search_and_observation_caps():
text = 'Page introduction. ' + 'Navigation and performance features. ' * 180
text += 'Privacy Guide explains how to choose cookie settings. These do not prevent every form of tracking. '
text += 'Other features and links. ' * 200
first = search_excerpt(text, 'browser privacy cookie settings', 3000)
final = search_excerpt(first, 'browser privacy cookie settings', 1100)
assert len(final) <= 1100
assert 'Privacy Guide explains how to choose cookie settings.' in final
assert 'These do not prevent every form of tracking.' in final