Separate current reference lookups from publication date restrictions

This commit is contained in:
pewdiepie-archdaemon
2026-09-17 20:47:06 +00:00
parent edcd94a312
commit 4d34ae1799
9 changed files with 149 additions and 31 deletions
+1 -1
View File
@@ -720,7 +720,7 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None)
# first instead of spending the full tool deadline retrying generic search
# providers; the returned official URL lets the agent proceed to PDF tools.
scholarly_title = _scholarly_title_from_query(provider_query)
if scholarly_title:
if scholarly_title and not time_filter:
direct_results = [result for result in _direct_scholarly_title_results(scholarly_title, count)
if _result_matches_site_scope(provider_query, result)]
if direct_results:
+2 -4
View File
@@ -220,14 +220,12 @@ def searxng_search_api(query: str, count: Optional[int] = None, categories: str
if is_news and categories == "general":
params["categories"] = "news"
if time_filter in ("day", "week", "month", "year"):
# 'day' is too sparse on most SearXNG news engines — widen to a week
# so there's enough volume; the news category already biases recent.
params["time_range"] = "week" if time_filter in ("day", "week") else time_filter
params["time_range"] = time_filter
else:
params["categories"] = categories
# Freshness and source category are independent: current manuals,
# comparisons and documentation still belong in general search.
if not is_software_release_query and time_filter in ("day", "week", "month", "year"):
if time_filter in ("day", "week", "month", "year"):
params["time_range"] = time_filter
# Route general queries to engines that aren't blocked (default general
# set returns 0 on this instance — see _GENERAL_ENGINES).
+4 -14
View File
@@ -231,18 +231,8 @@ class WebSearchTool:
if not query:
query = raw.split("\n")[0].strip()
if time_filter is None:
q_lc = query.lower()
if any(kw in q_lc for kw in (
"today", "latest", "breaking", "this morning", "right now",
"currently", "current events", "what's happening", "what is happening",
)):
time_filter = "day"
elif any(kw in q_lc for kw in ("this week", "past week", "recent news", "last few days")):
time_filter = "week"
elif any(kw in q_lc for kw in ("this month", "past month")):
time_filter = "month"
elif " news" in q_lc or q_lc.startswith("news ") or q_lc.endswith(" news"):
time_filter = "week"
from src.search_intent import inferred_search_publication_window
time_filter = inferred_search_publication_window(query)
loop = asyncio.get_running_loop()
if progress_cb:
await progress_cb({
@@ -254,7 +244,7 @@ class WebSearchTool:
results = await asyncio.wait_for(
loop.run_in_executor(
None,
lambda: searxng_search_results(query, max_pages),
lambda: searxng_search_results(query, max_pages, **({'time_filter': time_filter} if time_filter else {})),
),
timeout=30,
)
@@ -283,7 +273,7 @@ class WebSearchTool:
results = await asyncio.wait_for(
loop.run_in_executor(
None,
lambda: searxng_search_results(query, max_pages),
lambda: searxng_search_results(query, max_pages, **({'time_filter': time_filter} if time_filter else {})),
),
timeout=12,
)
+13 -9
View File
@@ -3490,15 +3490,19 @@ def preserve_requested_web_recency(name, args, *, user_text='', prior_search_int
if (current_intent
and not re.search(r'\b(?:latest|recent|current|today|news|updates?|20\d{2})\b', query, re.I)):
query = f'{query} latest {current_year}'
if current_intent and not normalized.get('time_filter'):
if re.search(r'\b(?:version|release|driver)\b', user, re.I):
normalized['time_filter'] = 'year'
elif re.search(r"\btoday(?:'s)?\b", user, re.I):
normalized['time_filter'] = 'day'
elif re.search(r'\brecent\b', user, re.I):
normalized['time_filter'] = 'month'
else:
normalized['time_filter'] = 'week'
from src.search_intent import inferred_search_publication_window, reference_lookup_without_date_window, requested_search_publication_window
requested_window = requested_search_publication_window(user)
if requested_window:
normalized['time_filter'] = requested_window
elif reference_lookup_without_date_window(user):
# A model-generated publication cutoff must not hide still-current
# reference pages when the user did not ask for recent publications.
normalized.pop('time_filter', None)
normalized.pop('freshness', None)
elif not normalized.get('time_filter'):
window = inferred_search_publication_window(user)
if window:
normalized['time_filter'] = window
normalized['query'] = query
return normalized
+42
View File
@@ -0,0 +1,42 @@
"""Shared publication-recency semantics for query repair and search execution."""
import re
def reference_lookup_without_date_window(text: str) -> bool:
"""Current reference information need not have been published recently."""
reference = re.search(
r'\b(?:documentation|docs|manuals?|guides?|reference|installation|configuration)\b'
r'|\bprivacy\s+(?:features|settings|protections)\b', text, re.I,
)
publication = re.search(
r'\b(?:published|publication|announced|released|news|headlines|recent|today|yesterday)\b'
r'|\b(?:this|last|past)\s+(?:\d+\s+)?(?:days?|weeks?|months?|years?)\b'
r'|\b(?:since|after|before|between|during)\b|\b20\d{2}\b', text, re.I,
)
return bool(reference and not publication)
def requested_search_publication_window(text: str) -> str | None:
"""Recognize explicit named publication windows."""
if re.search(r'\b(?:today|yesterday|this morning|right now)\b', text, re.I):
return 'day'
for unit, value in [('week', 'week'), ('month', 'month'), ('year', 'year')]:
if re.search(rf'\b(?:this|past|last)\s+{unit}\b', text, re.I):
return value
return None
def inferred_search_publication_window(text: str) -> str | None:
"""Infer publication recency, not freshness of every requested fact."""
requested = requested_search_publication_window(text)
if requested:
return requested
if reference_lookup_without_date_window(text):
return None
if re.search(r"\b(?:current events|what(?:'s| is) happening)\b", text, re.I):
return 'day'
if re.search(r'\b(?:news|neews|headlines|breaking|latest developments)\b', text, re.I):
return 'week'
if re.search(r'\brecent\b', text, re.I):
return 'month'
return None
+1 -1
View File
@@ -337,7 +337,7 @@ FUNCTION_TOOL_SCHEMAS = [
"properties": {
"query": {"type": "string", "description": "Search query"},
"command": {"type": "string", "description": "Search query in text command form"},
"time_filter": {"type": "string", "enum": ["day", "week", "month", "year"], "description": "Optional freshness filter for news/latest/today queries"}
"time_filter": {"type": "string", "enum": ["day", "week", "month", "year"], "description": "Optional publication-date window for recent articles/news. Omit for current documentation, manuals, or features unless the user specifies a publication window."}
},
"required": []
}
+1 -1
View File
@@ -1990,7 +1990,7 @@ def test_current_search_arguments_repair_stale_year_and_add_freshness():
assert '2025' not in args['query']
assert str(__import__('datetime').datetime.now(__import__('datetime').timezone.utc).year) in args['query']
assert args['time_filter'] == 'week'
assert args['time_filter'] == 'day'
def test_missing_refinement_query_is_grounded_in_user_request():
+84
View File
@@ -0,0 +1,84 @@
import asyncio
import json
import pytest
from src.search_intent import inferred_search_publication_window, reference_lookup_without_date_window
from src.clean_agent_preview import preserve_requested_web_recency
@pytest.mark.parametrize('query,expected', [
('latest Python version', None),
('current Firefox privacy features', None),
('latest printer installation guide', None),
('browser documentation published this month', 'month'),
('AI news today', 'day'),
('news this week', 'week'),
('latest AI news', 'week'),
('current events in Japan', 'day'),
])
def test_publication_window_is_not_synonymous_with_current_information(query, expected):
assert inferred_search_publication_window(query) == expected
@pytest.mark.parametrize('prompt', [
'compare current Firefox and Chrome privacy features',
'find the latest official printer manual',
'current installation documentation',
])
def test_reference_queries_do_not_inherit_model_invented_publication_cutoffs(prompt):
assert reference_lookup_without_date_window(prompt)
args = preserve_requested_web_recency('web_search', {'query': prompt, 'time_filter': 'month'}, user_text=prompt)
assert 'time_filter' not in args
def test_explicit_publication_constraints_are_preserved():
request = 'Find privacy guides published this month'
args = preserve_requested_web_recency('web_search', {'query': request, 'time_filter': 'month'}, user_text=request)
assert args['time_filter'] == 'month'
args = preserve_requested_web_recency('web_search', {'query': 'privacy guides', 'time_filter': 'year'}, user_text=request)
assert args['time_filter'] == 'month'
def test_provider_does_not_silently_widen_news_window(monkeypatch):
from services.search import providers
seen = {}
class Response:
def raise_for_status(self): pass
def json(self): return {'results': [{'title': 'AI news', 'url': 'https://example.org', 'content': 'AI news today'}]}
def get(url, **kwargs):
seen.update(kwargs['params'])
return Response()
monkeypatch.setattr(providers, '_get_search_instance', lambda: 'http://searx.test')
monkeypatch.setattr(providers, '_get_search_settings', lambda: {})
monkeypatch.setattr(providers.httpx, 'get', get)
providers.searxng_search_api('AI news today', time_filter='day')
assert seen['categories'] == 'news'
assert seen['time_range'] == 'day'
@pytest.mark.parametrize('arguments,expected', [
({'query': 'latest browser documentation'}, None),
({'query': 'browser documentation', 'time_filter': 'month'}, 'month'),
({'query': 'AI news today'}, 'day'),
])
def test_search_execution_honors_explicit_filter_but_does_not_invent_one(monkeypatch, arguments, expected):
import src.search as search
from src.agent_tools.web_tools import WebSearchTool
seen = {}
def execute(query, **kwargs):
seen.update(kwargs)
return 'Evidence', [{'title': 'Source', 'url': 'https://example.org'}]
monkeypatch.setattr(search, 'comprehensive_web_search', execute)
asyncio.run(WebSearchTool().execute(json.dumps(arguments), {}))
assert seen['time_filter'] == expected
def test_metadata_search_keeps_explicit_publication_window(monkeypatch):
import src.search as search
from src.agent_tools.web_tools import WebSearchTool
seen = {}
def execute(query, count, **kwargs):
seen.update(kwargs)
return [{'title': 'Official source', 'url': 'https://example.org', 'snippet': 'Reference'}]
monkeypatch.setattr(search, 'searxng_search_results', execute)
asyncio.run(WebSearchTool().execute(json.dumps({'query': 'official Python website', 'time_filter': 'month'}), {}))
assert seen['time_filter'] == 'month'
+1 -1
View File
@@ -76,7 +76,7 @@ def test_service_searxng_json_sends_safesearch(monkeypatch):
@pytest.mark.parametrize('query,expected_time', [
('latest ollama release version github', None),
('latest ollama release version github', 'day'),
('current Firefox Chrome privacy features comparison', 'day'),
('Sony headphone manual', 'day'),
])