From 7257319004f26bbef3a9ef08ec300ca41d1912e1 Mon Sep 17 00:00:00 2001 From: pewdiepie-archdaemon Date: Thu, 17 Sep 2026 20:17:03 +0000 Subject: [PATCH] Preserve every fetched search source within observation budget --- src/clean_agent_preview.py | 35 +++++++++++++++++++++++ tests/test_search_observation_budget.py | 37 +++++++++++++++++++++++++ 2 files changed, 72 insertions(+) create mode 100644 tests/test_search_observation_budget.py diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py index d44f0f385..3bbe92a6b 100644 --- a/src/clean_agent_preview.py +++ b/src/clean_agent_preview.py @@ -162,6 +162,39 @@ SAFE_ACTIONS = { } +def bounded_search_observation(output, budget=8000): + """Share the observation budget across fetched sources, not prefix order. + + Search's full report repeats bodies in summaries/quotes/statistics. Retain + source attribution and an excerpt of every fetched page before truncating. + This is evidence selection, never answer generation. + """ + if len(output) <= budget: + return output + pattern = re.compile(r'\n(\[CONTENT(?: \d+)?\] From: [^\n]+\nTitle: [^\n]*\n-+\n)') + matches = list(pattern.finditer(output)) + if not matches: + return output + source_match = re.search(r'```sources\n.*?```', output, re.DOTALL) + query_match = re.search(r'^Query: .*$', output, re.MULTILINE) + prefix = '\n'.join(match.group(0) for match in (source_match, query_match) if match) + suffix = '\n[Excerpts shortened across sources; use web_fetch on a source URL for full details.]' + headers = [match.group(1) for match in matches] + room = budget - len(prefix) - len(suffix) - sum(len(h) + 2 for h in headers) + if room < 100 * len(matches): + return output # Unusually large metadata: preserve existing hard cap. + per_page = room // len(matches) + blocks = [] + for index, match in enumerate(matches): + end = matches[index + 1].start() if index + 1 < len(matches) else len(output) + body = output[match.end():end] + body = re.split(r'\n(?:Key Points:|TL;DR:|Important Quotes:|Data / Statistics:|={20,}|