Preserve every fetched search source within observation budget

This commit is contained in:
pewdiepie-archdaemon
2026-09-17 20:17:03 +00:00
parent a48ab46f7d
commit 7257319004
2 changed files with 72 additions and 0 deletions
+35
View File
@@ -162,6 +162,39 @@ SAFE_ACTIONS = {
}
def bounded_search_observation(output, budget=8000):
"""Share the observation budget across fetched sources, not prefix order.
Search's full report repeats bodies in summaries/quotes/statistics. Retain
source attribution and an excerpt of every fetched page before truncating.
This is evidence selection, never answer generation.
"""
if len(output) <= budget:
return output
pattern = re.compile(r'\n(\[CONTENT(?: \d+)?\] From: [^\n]+\nTitle: [^\n]*\n-+\n)')
matches = list(pattern.finditer(output))
if not matches:
return output
source_match = re.search(r'```sources\n.*?```', output, re.DOTALL)
query_match = re.search(r'^Query: .*$', output, re.MULTILINE)
prefix = '\n'.join(match.group(0) for match in (source_match, query_match) if match)
suffix = '\n[Excerpts shortened across sources; use web_fetch on a source URL for full details.]'
headers = [match.group(1) for match in matches]
room = budget - len(prefix) - len(suffix) - sum(len(h) + 2 for h in headers)
if room < 100 * len(matches):
return output # Unusually large metadata: preserve existing hard cap.
per_page = room // len(matches)
blocks = []
for index, match in enumerate(matches):
end = matches[index + 1].start() if index + 1 < len(matches) else len(output)
body = output[match.end():end]
body = re.split(r'\n(?:Key Points:|TL;DR:|Important Quotes:|Data / Statistics:|={20,}|<!-- SOURCES:)', body, maxsplit=1)[0].strip()
if len(body) > per_page:
body = body[:per_page - 15].rstrip() + '\n[...excerpt]'
blocks.append(headers[index] + body)
return prefix + '\n\n' + '\n\n'.join(blocks) + suffix
def preview_tool_result_text(result, tool, args):
"""Preserve failure evidence before applying the observation budget."""
output = result.get('output') or result.get('error') or result
@@ -185,6 +218,8 @@ def preview_tool_result_text(result, tool, args):
# leaving neither the model nor canonical renderer usable evidence.
output = result['results']
output = output if isinstance(output, str) else json.dumps(output, ensure_ascii=False)
if canonical(tool) == 'web_search' and not result.get('error') and result.get('exit_code') in (None, 0):
output = bounded_search_observation(output)
if len(output) > 8000:
output = output[:8000] + '\n[Tool result truncated at 8000 characters.]'
return output
+37
View File
@@ -0,0 +1,37 @@
from src.clean_agent_preview import preview_tool_result_text
def test_all_fetched_sources_survive_observation_budget():
sources = '```sources\n' + '\n'.join(
f'[{i}] Page {i}\nhttps://example.org/{i}' for i in range(1, 6)
) + '\n```\nQuery: example research\n'
report = sources
for i in range(1, 6):
report += (f'\n[CONTENT {i}] From: https://example.org/{i}\n'
f'Title: Page {i}\n------------------------------\n'
+ f'Evidence from page {i}. ' * 200
+ '\nTL;DR:\n' + 'Repeated summary. ' * 200)
output = preview_tool_result_text({'output': report, 'exit_code': 0}, 'web_search', {})
assert len(output) <= 8000
for i in range(1, 6):
assert f'[CONTENT {i}] From: https://example.org/{i}' in output
assert f'Evidence from page {i}.' in output
assert 'Repeated summary.' not in output
assert 'full details' in output
def test_short_search_results_are_unchanged():
text = 'No matching sources were found.'
assert preview_tool_result_text({'output': text}, 'web_search', {}) == text
def test_failure_status_is_not_lost_to_search_compaction():
result = {'output': 'Partial evidence. ' * 1000, 'error': 'fetch failed', 'exit_code': 1}
output = preview_tool_result_text(result, 'web_search', {})
assert 'fetch failed' in output[:100]
def test_other_tool_observations_keep_existing_budget():
output = preview_tool_result_text({'output': 'x' * 9000}, 'bash', {})
assert output.startswith('x' * 8000)
assert 'truncated at 8000' in output