diff --git a/services/search/core.py b/services/search/core.py index cafe3a7a2..4a8d58a7e 100644 --- a/services/search/core.py +++ b/services/search/core.py @@ -720,7 +720,7 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None) # first instead of spending the full tool deadline retrying generic search # providers; the returned official URL lets the agent proceed to PDF tools. scholarly_title = _scholarly_title_from_query(provider_query) - if scholarly_title: + if scholarly_title and not time_filter: direct_results = [result for result in _direct_scholarly_title_results(scholarly_title, count) if _result_matches_site_scope(provider_query, result)] if direct_results: diff --git a/services/search/providers.py b/services/search/providers.py index e94c3f731..95525ec7d 100644 --- a/services/search/providers.py +++ b/services/search/providers.py @@ -220,14 +220,12 @@ def searxng_search_api(query: str, count: Optional[int] = None, categories: str if is_news and categories == "general": params["categories"] = "news" if time_filter in ("day", "week", "month", "year"): - # 'day' is too sparse on most SearXNG news engines — widen to a week - # so there's enough volume; the news category already biases recent. - params["time_range"] = "week" if time_filter in ("day", "week") else time_filter + params["time_range"] = time_filter else: params["categories"] = categories # Freshness and source category are independent: current manuals, # comparisons and documentation still belong in general search. - if not is_software_release_query and time_filter in ("day", "week", "month", "year"): + if time_filter in ("day", "week", "month", "year"): params["time_range"] = time_filter # Route general queries to engines that aren't blocked (default general # set returns 0 on this instance — see _GENERAL_ENGINES). diff --git a/src/agent_tools/web_tools.py b/src/agent_tools/web_tools.py index 7c191d07c..d46681f0f 100644 --- a/src/agent_tools/web_tools.py +++ b/src/agent_tools/web_tools.py @@ -231,18 +231,8 @@ class WebSearchTool: if not query: query = raw.split("\n")[0].strip() if time_filter is None: - q_lc = query.lower() - if any(kw in q_lc for kw in ( - "today", "latest", "breaking", "this morning", "right now", - "currently", "current events", "what's happening", "what is happening", - )): - time_filter = "day" - elif any(kw in q_lc for kw in ("this week", "past week", "recent news", "last few days")): - time_filter = "week" - elif any(kw in q_lc for kw in ("this month", "past month")): - time_filter = "month" - elif " news" in q_lc or q_lc.startswith("news ") or q_lc.endswith(" news"): - time_filter = "week" + from src.search_intent import inferred_search_publication_window + time_filter = inferred_search_publication_window(query) loop = asyncio.get_running_loop() if progress_cb: await progress_cb({ @@ -254,7 +244,7 @@ class WebSearchTool: results = await asyncio.wait_for( loop.run_in_executor( None, - lambda: searxng_search_results(query, max_pages), + lambda: searxng_search_results(query, max_pages, **({'time_filter': time_filter} if time_filter else {})), ), timeout=30, ) @@ -283,7 +273,7 @@ class WebSearchTool: results = await asyncio.wait_for( loop.run_in_executor( None, - lambda: searxng_search_results(query, max_pages), + lambda: searxng_search_results(query, max_pages, **({'time_filter': time_filter} if time_filter else {})), ), timeout=12, ) diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py index b651d8bb8..01ec94713 100644 --- a/src/clean_agent_preview.py +++ b/src/clean_agent_preview.py @@ -3490,15 +3490,19 @@ def preserve_requested_web_recency(name, args, *, user_text='', prior_search_int if (current_intent and not re.search(r'\b(?:latest|recent|current|today|news|updates?|20\d{2})\b', query, re.I)): query = f'{query} latest {current_year}' - if current_intent and not normalized.get('time_filter'): - if re.search(r'\b(?:version|release|driver)\b', user, re.I): - normalized['time_filter'] = 'year' - elif re.search(r"\btoday(?:'s)?\b", user, re.I): - normalized['time_filter'] = 'day' - elif re.search(r'\brecent\b', user, re.I): - normalized['time_filter'] = 'month' - else: - normalized['time_filter'] = 'week' + from src.search_intent import inferred_search_publication_window, reference_lookup_without_date_window, requested_search_publication_window + requested_window = requested_search_publication_window(user) + if requested_window: + normalized['time_filter'] = requested_window + elif reference_lookup_without_date_window(user): + # A model-generated publication cutoff must not hide still-current + # reference pages when the user did not ask for recent publications. + normalized.pop('time_filter', None) + normalized.pop('freshness', None) + elif not normalized.get('time_filter'): + window = inferred_search_publication_window(user) + if window: + normalized['time_filter'] = window normalized['query'] = query return normalized diff --git a/src/search_intent.py b/src/search_intent.py new file mode 100644 index 000000000..e4b63d2b8 --- /dev/null +++ b/src/search_intent.py @@ -0,0 +1,42 @@ +"""Shared publication-recency semantics for query repair and search execution.""" +import re + + +def reference_lookup_without_date_window(text: str) -> bool: + """Current reference information need not have been published recently.""" + reference = re.search( + r'\b(?:documentation|docs|manuals?|guides?|reference|installation|configuration)\b' + r'|\bprivacy\s+(?:features|settings|protections)\b', text, re.I, + ) + publication = re.search( + r'\b(?:published|publication|announced|released|news|headlines|recent|today|yesterday)\b' + r'|\b(?:this|last|past)\s+(?:\d+\s+)?(?:days?|weeks?|months?|years?)\b' + r'|\b(?:since|after|before|between|during)\b|\b20\d{2}\b', text, re.I, + ) + return bool(reference and not publication) + + +def requested_search_publication_window(text: str) -> str | None: + """Recognize explicit named publication windows.""" + if re.search(r'\b(?:today|yesterday|this morning|right now)\b', text, re.I): + return 'day' + for unit, value in [('week', 'week'), ('month', 'month'), ('year', 'year')]: + if re.search(rf'\b(?:this|past|last)\s+{unit}\b', text, re.I): + return value + return None + + +def inferred_search_publication_window(text: str) -> str | None: + """Infer publication recency, not freshness of every requested fact.""" + requested = requested_search_publication_window(text) + if requested: + return requested + if reference_lookup_without_date_window(text): + return None + if re.search(r"\b(?:current events|what(?:'s| is) happening)\b", text, re.I): + return 'day' + if re.search(r'\b(?:news|neews|headlines|breaking|latest developments)\b', text, re.I): + return 'week' + if re.search(r'\brecent\b', text, re.I): + return 'month' + return None diff --git a/src/tool_schemas.py b/src/tool_schemas.py index 5f64ddcfc..88278d4d6 100644 --- a/src/tool_schemas.py +++ b/src/tool_schemas.py @@ -337,7 +337,7 @@ FUNCTION_TOOL_SCHEMAS = [ "properties": { "query": {"type": "string", "description": "Search query"}, "command": {"type": "string", "description": "Search query in text command form"}, - "time_filter": {"type": "string", "enum": ["day", "week", "month", "year"], "description": "Optional freshness filter for news/latest/today queries"} + "time_filter": {"type": "string", "enum": ["day", "week", "month", "year"], "description": "Optional publication-date window for recent articles/news. Omit for current documentation, manuals, or features unless the user specifies a publication window."} }, "required": [] } diff --git a/tests/test_clean_agent_preview.py b/tests/test_clean_agent_preview.py index 360b7632b..56fcf5ce1 100644 --- a/tests/test_clean_agent_preview.py +++ b/tests/test_clean_agent_preview.py @@ -1990,7 +1990,7 @@ def test_current_search_arguments_repair_stale_year_and_add_freshness(): assert '2025' not in args['query'] assert str(__import__('datetime').datetime.now(__import__('datetime').timezone.utc).year) in args['query'] - assert args['time_filter'] == 'week' + assert args['time_filter'] == 'day' def test_missing_refinement_query_is_grounded_in_user_request(): diff --git a/tests/test_search_publication_intent.py b/tests/test_search_publication_intent.py new file mode 100644 index 000000000..519c99e31 --- /dev/null +++ b/tests/test_search_publication_intent.py @@ -0,0 +1,84 @@ +import asyncio +import json +import pytest +from src.search_intent import inferred_search_publication_window, reference_lookup_without_date_window +from src.clean_agent_preview import preserve_requested_web_recency + + +@pytest.mark.parametrize('query,expected', [ + ('latest Python version', None), + ('current Firefox privacy features', None), + ('latest printer installation guide', None), + ('browser documentation published this month', 'month'), + ('AI news today', 'day'), + ('news this week', 'week'), + ('latest AI news', 'week'), + ('current events in Japan', 'day'), +]) +def test_publication_window_is_not_synonymous_with_current_information(query, expected): + assert inferred_search_publication_window(query) == expected + + +@pytest.mark.parametrize('prompt', [ + 'compare current Firefox and Chrome privacy features', + 'find the latest official printer manual', + 'current installation documentation', +]) +def test_reference_queries_do_not_inherit_model_invented_publication_cutoffs(prompt): + assert reference_lookup_without_date_window(prompt) + args = preserve_requested_web_recency('web_search', {'query': prompt, 'time_filter': 'month'}, user_text=prompt) + assert 'time_filter' not in args + + +def test_explicit_publication_constraints_are_preserved(): + request = 'Find privacy guides published this month' + args = preserve_requested_web_recency('web_search', {'query': request, 'time_filter': 'month'}, user_text=request) + assert args['time_filter'] == 'month' + args = preserve_requested_web_recency('web_search', {'query': 'privacy guides', 'time_filter': 'year'}, user_text=request) + assert args['time_filter'] == 'month' + + +def test_provider_does_not_silently_widen_news_window(monkeypatch): + from services.search import providers + seen = {} + class Response: + def raise_for_status(self): pass + def json(self): return {'results': [{'title': 'AI news', 'url': 'https://example.org', 'content': 'AI news today'}]} + def get(url, **kwargs): + seen.update(kwargs['params']) + return Response() + monkeypatch.setattr(providers, '_get_search_instance', lambda: 'http://searx.test') + monkeypatch.setattr(providers, '_get_search_settings', lambda: {}) + monkeypatch.setattr(providers.httpx, 'get', get) + providers.searxng_search_api('AI news today', time_filter='day') + assert seen['categories'] == 'news' + assert seen['time_range'] == 'day' + + +@pytest.mark.parametrize('arguments,expected', [ + ({'query': 'latest browser documentation'}, None), + ({'query': 'browser documentation', 'time_filter': 'month'}, 'month'), + ({'query': 'AI news today'}, 'day'), +]) +def test_search_execution_honors_explicit_filter_but_does_not_invent_one(monkeypatch, arguments, expected): + import src.search as search + from src.agent_tools.web_tools import WebSearchTool + seen = {} + def execute(query, **kwargs): + seen.update(kwargs) + return 'Evidence', [{'title': 'Source', 'url': 'https://example.org'}] + monkeypatch.setattr(search, 'comprehensive_web_search', execute) + asyncio.run(WebSearchTool().execute(json.dumps(arguments), {})) + assert seen['time_filter'] == expected + + +def test_metadata_search_keeps_explicit_publication_window(monkeypatch): + import src.search as search + from src.agent_tools.web_tools import WebSearchTool + seen = {} + def execute(query, count, **kwargs): + seen.update(kwargs) + return [{'title': 'Official source', 'url': 'https://example.org', 'snippet': 'Reference'}] + monkeypatch.setattr(search, 'searxng_search_results', execute) + asyncio.run(WebSearchTool().execute(json.dumps({'query': 'official Python website', 'time_filter': 'month'}), {})) + assert seen['time_filter'] == 'month' diff --git a/tests/test_service_search_provider_guards.py b/tests/test_service_search_provider_guards.py index 50a5090cc..524572a87 100644 --- a/tests/test_service_search_provider_guards.py +++ b/tests/test_service_search_provider_guards.py @@ -76,7 +76,7 @@ def test_service_searxng_json_sends_safesearch(monkeypatch): @pytest.mark.parametrize('query,expected_time', [ - ('latest ollama release version github', None), + ('latest ollama release version github', 'day'), ('current Firefox Chrome privacy features comparison', 'day'), ('Sony headphone manual', 'day'), ])