mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-09 00:12:21 +02:00
Preserve all fetched search sources before transport truncation
This commit is contained in:
@@ -304,6 +304,10 @@ class WebSearchTool:
|
|||||||
"elapsed_s": 30,
|
"elapsed_s": 30,
|
||||||
"tail": "Search completed; preparing sources.",
|
"tail": "Search completed; preparing sources.",
|
||||||
})
|
})
|
||||||
|
# Compact the complete report before the transport cap. Otherwise
|
||||||
|
# repeated summaries from early sources permanently erase later pages.
|
||||||
|
from src.search_passages import bounded_search_observation
|
||||||
|
text = bounded_search_observation(text, MAX_OUTPUT_CHARS)
|
||||||
output = text[:MAX_OUTPUT_CHARS] if len(text) > MAX_OUTPUT_CHARS else text
|
output = text[:MAX_OUTPUT_CHARS] if len(text) > MAX_OUTPUT_CHARS else text
|
||||||
if sources:
|
if sources:
|
||||||
output += "\n\n<!-- SOURCES:" + json.dumps(sources) + " -->"
|
output += "\n\n<!-- SOURCES:" + json.dumps(sources) + " -->"
|
||||||
|
|||||||
@@ -182,37 +182,8 @@ def search_tool_choice_request(request):
|
|||||||
|
|
||||||
|
|
||||||
def bounded_search_observation(output, budget=8000):
|
def bounded_search_observation(output, budget=8000):
|
||||||
"""Share the observation budget across fetched sources, not prefix order.
|
from src.search_passages import bounded_search_observation as compact
|
||||||
|
return compact(output, budget)
|
||||||
Search's full report repeats bodies in summaries/quotes/statistics. Retain
|
|
||||||
source attribution and an excerpt of every fetched page before truncating.
|
|
||||||
This is evidence selection, never answer generation.
|
|
||||||
"""
|
|
||||||
if len(output) <= budget:
|
|
||||||
return output
|
|
||||||
pattern = re.compile(r'\n(\[CONTENT(?: \d+)?\] From: [^\n]+\nTitle: [^\n]*\n-+\n)')
|
|
||||||
matches = list(pattern.finditer(output))
|
|
||||||
if not matches:
|
|
||||||
return output
|
|
||||||
source_match = re.search(r'```sources\n.*?```', output, re.DOTALL)
|
|
||||||
query_match = re.search(r'^Query: .*$', output, re.MULTILINE)
|
|
||||||
prefix = '\n'.join(match.group(0) for match in (source_match, query_match) if match)
|
|
||||||
suffix = '\n[Excerpts shortened across sources; use web_fetch on a source URL for full details.]'
|
|
||||||
headers = [match.group(1) for match in matches]
|
|
||||||
room = budget - len(prefix) - len(suffix) - sum(len(h) + 2 for h in headers)
|
|
||||||
if room < 100 * len(matches):
|
|
||||||
return output # Unusually large metadata: preserve existing hard cap.
|
|
||||||
per_page = room // len(matches)
|
|
||||||
blocks = []
|
|
||||||
for index, match in enumerate(matches):
|
|
||||||
end = matches[index + 1].start() if index + 1 < len(matches) else len(output)
|
|
||||||
body = output[match.end():end]
|
|
||||||
body = re.split(r'\n(?:Key Points:|TL;DR:|Important Quotes:|Data / Statistics:|={20,}|<!-- SOURCES:)', body, maxsplit=1)[0].strip()
|
|
||||||
if len(body) > per_page:
|
|
||||||
from src.search_passages import search_excerpt
|
|
||||||
body = search_excerpt(body, query_match.group(0) if query_match else '', per_page)
|
|
||||||
blocks.append(headers[index] + body)
|
|
||||||
return prefix + '\n\n' + '\n\n'.join(blocks) + suffix
|
|
||||||
|
|
||||||
|
|
||||||
def preview_tool_result_text(result, tool, args):
|
def preview_tool_result_text(result, tool, args):
|
||||||
|
|||||||
@@ -5,6 +5,34 @@ import re
|
|||||||
_FILLER = frozenset('the and for with from this that what which how find search compare explain latest current recent official source sources documentation document please about'.split())
|
_FILLER = frozenset('the and for with from this that what which how find search compare explain latest current recent official source sources documentation document please about'.split())
|
||||||
|
|
||||||
|
|
||||||
|
def bounded_search_observation(output, budget=8000):
|
||||||
|
"""Budget all fetched sources before any transport-level prefix truncation."""
|
||||||
|
if len(output) <= budget:
|
||||||
|
return output
|
||||||
|
pattern = re.compile(r'\n(\[CONTENT(?: \d+)?\] From: [^\n]+\nTitle: [^\n]*\n-+\n)')
|
||||||
|
matches = list(pattern.finditer(output))
|
||||||
|
if not matches:
|
||||||
|
return output
|
||||||
|
source_match = re.search(r'```sources\n.*?```', output, re.DOTALL)
|
||||||
|
query_match = re.search(r'^Query: .*$', output, re.MULTILINE)
|
||||||
|
prefix = '\n'.join(match.group(0) for match in (source_match, query_match) if match)
|
||||||
|
suffix = '\n[Excerpts shortened across sources; use web_fetch on a source URL for full details.]'
|
||||||
|
headers = [match.group(1) for match in matches]
|
||||||
|
room = budget - len(prefix) - len(suffix) - sum(len(h) + 2 for h in headers) - 2
|
||||||
|
if room < 100 * len(matches):
|
||||||
|
return output # Caller retains its hard cap for exceptional metadata.
|
||||||
|
per_page = room // len(matches)
|
||||||
|
blocks = []
|
||||||
|
for index, match in enumerate(matches):
|
||||||
|
end = matches[index + 1].start() if index + 1 < len(matches) else len(output)
|
||||||
|
body = output[match.end():end]
|
||||||
|
body = re.split(r'\n(?:Key Points:|TL;DR:|Important Quotes:|Data / Statistics:|={20,}|<!-- SOURCES:)', body, maxsplit=1)[0].strip()
|
||||||
|
if len(body) > per_page:
|
||||||
|
body = search_excerpt(body, query_match.group(0) if query_match else '', per_page)
|
||||||
|
blocks.append(headers[index] + body)
|
||||||
|
return prefix + '\n\n' + '\n\n'.join(blocks) + suffix
|
||||||
|
|
||||||
|
|
||||||
def search_excerpt(text: str, query: str, max_chars: int) -> str:
|
def search_excerpt(text: str, query: str, max_chars: int) -> str:
|
||||||
"""Keep the opening plus relevant, non-overlapping literal page excerpts.
|
"""Keep the opening plus relevant, non-overlapping literal page excerpts.
|
||||||
|
|
||||||
|
|||||||
@@ -10,6 +10,27 @@ from src.agent_tools.web_tools import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_long_report_preserves_late_source_through_tool_and_runtime(monkeypatch):
|
||||||
|
from src.clean_agent_preview import preview_tool_result_text
|
||||||
|
sources = [{'title': f'Page {i}', 'url': f'https://example.org/{i}'} for i in range(1, 6)]
|
||||||
|
report = '```sources\n' + '\n'.join(
|
||||||
|
f'[{i}] {s["title"]}\n {s["url"]}' for i, s in enumerate(sources, 1)
|
||||||
|
) + '\n```\nQuery: measured battery evidence\n'
|
||||||
|
for i, source in enumerate(sources, 1):
|
||||||
|
report += (f'\n[CONTENT {i}] From: {source["url"]}\nTitle: {source["title"]}\n------------------------------\n'
|
||||||
|
+ f'Unique evidence for page {i}. ' * 100
|
||||||
|
+ '\nTL;DR:\n' + 'Repeated body summary. ' * 200)
|
||||||
|
monkeypatch.setattr(search, 'comprehensive_web_search', lambda *a, **kw: (report, sources))
|
||||||
|
result = asyncio.run(WebSearchTool().execute(json.dumps({'query': 'measured battery evidence'}), {}))
|
||||||
|
observation = preview_tool_result_text(result, 'web_search', {})
|
||||||
|
assert len(observation) <= 8000
|
||||||
|
for i in range(1, 6):
|
||||||
|
assert f'[CONTENT {i}] From:' in observation
|
||||||
|
assert f'Unique evidence for page {i}.' in observation
|
||||||
|
assert 'Repeated body summary' not in observation
|
||||||
|
assert json.loads(result['output'].split('<!-- SOURCES:')[1].split(' -->')[0]) == sources
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.parametrize('sources,status', [([], 'empty'),
|
@pytest.mark.parametrize('sources,status', [([], 'empty'),
|
||||||
([{'title': 'Example', 'url': 'https://example.org'}], 'available')])
|
([{'title': 'Example', 'url': 'https://example.org'}], 'available')])
|
||||||
def test_search_reports_evidence_availability_independently_of_execution(monkeypatch, sources, status):
|
def test_search_reports_evidence_availability_independently_of_execution(monkeypatch, sources, status):
|
||||||
|
|||||||
Reference in New Issue
Block a user