diff --git a/src/agent_tools/web_tools.py b/src/agent_tools/web_tools.py index d46681f0f..eb8891458 100644 --- a/src/agent_tools/web_tools.py +++ b/src/agent_tools/web_tools.py @@ -304,6 +304,10 @@ class WebSearchTool: "elapsed_s": 30, "tail": "Search completed; preparing sources.", }) + # Compact the complete report before the transport cap. Otherwise + # repeated summaries from early sources permanently erase later pages. + from src.search_passages import bounded_search_observation + text = bounded_search_observation(text, MAX_OUTPUT_CHARS) output = text[:MAX_OUTPUT_CHARS] if len(text) > MAX_OUTPUT_CHARS else text if sources: output += "\n\n" diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py index d168657c1..41cdb5088 100644 --- a/src/clean_agent_preview.py +++ b/src/clean_agent_preview.py @@ -182,37 +182,8 @@ def search_tool_choice_request(request): def bounded_search_observation(output, budget=8000): - """Share the observation budget across fetched sources, not prefix order. - - Search's full report repeats bodies in summaries/quotes/statistics. Retain - source attribution and an excerpt of every fetched page before truncating. - This is evidence selection, never answer generation. - """ - if len(output) <= budget: - return output - pattern = re.compile(r'\n(\[CONTENT(?: \d+)?\] From: [^\n]+\nTitle: [^\n]*\n-+\n)') - matches = list(pattern.finditer(output)) - if not matches: - return output - source_match = re.search(r'```sources\n.*?```', output, re.DOTALL) - query_match = re.search(r'^Query: .*$', output, re.MULTILINE) - prefix = '\n'.join(match.group(0) for match in (source_match, query_match) if match) - suffix = '\n[Excerpts shortened across sources; use web_fetch on a source URL for full details.]' - headers = [match.group(1) for match in matches] - room = budget - len(prefix) - len(suffix) - sum(len(h) + 2 for h in headers) - if room < 100 * len(matches): - return output # Unusually large metadata: preserve existing hard cap. - per_page = room // len(matches) - blocks = [] - for index, match in enumerate(matches): - end = matches[index + 1].start() if index + 1 < len(matches) else len(output) - body = output[match.end():end] - body = re.split(r'\n(?:Key Points:|TL;DR:|Important Quotes:|Data / Statistics:|={20,}|')[0]) == sources + + @pytest.mark.parametrize('sources,status', [([], 'empty'), ([{'title': 'Example', 'url': 'https://example.org'}], 'available')]) def test_search_reports_evidence_availability_independently_of_execution(monkeypatch, sources, status):