ground fresh searches and verify official documents

This commit is contained in:
pewdiepie-archdaemon
2026-09-17 19:49:28 +00:00
parent 994413c435
commit ff4d01a55e
2 changed files with 101 additions and 6 deletions
+46 -6
View File
@@ -3224,7 +3224,17 @@ def web_source_links(raw, *, max_items=1, prefer_official=False, query=''):
host == domain or host.endswith('.' + domain)
for domain in official_domains
)
if query_host_match or not official_domains:
direct_document = bool(re.search(
r'\.(?:pdf|docx?|xlsx?|pptx?)(?:$|[?#])', row[1], re.I,
))
query_tokens_in_url = any(
token in row[1].casefold()
for token in query_tokens
if len(token) >= 4
)
if query_host_match or (
not official_domains and direct_document and query_tokens_in_url
):
primary.append(row)
rows = primary
links = []
@@ -3383,23 +3393,51 @@ def requested_web_link_limit(user_text):
return max(1, min(5, int(token) if token.isdigit() else words[token]))
def preserve_requested_web_recency(name, args, *, user_text=''):
"""Keep explicit source/freshness intent when the model shortens a query."""
def preserve_requested_web_recency(name, args, *, user_text='', prior_search_intents=()):
"""Ground omitted/stale search arguments in the user's current request."""
if canonical(name) != 'web_search' or not isinstance(args, dict):
return args
user = str(user_text or '')
query = str(args.get('query') or '').strip()
if not query:
return args
query = re.sub(r'\s+', ' ', user).strip().rstrip('?.!')
if prior_search_intents:
query += ' corroborating analysis authoritative sources'
if not query:
return args
normalized = dict(args)
current_year = datetime.now(timezone.utc).year
current_intent = bool(re.search(
r"\b(?:latest|recent|current|today(?:'s)?|news|updates?|"
r"what(?:'s|\s+is)\s+(?:new|happening))\b",
user,
re.I,
))
if current_intent and not re.search(r'\b20\d{2}\b', user):
query = re.sub(r'\b20(?:0\d|1\d|2[0-5])\b', str(current_year), query)
if re.search(r'\bofficial\b', user, re.I) and not re.search(r'\bofficial\b', query, re.I):
query = f'{query} official source'
domains = official_domains_for_text(user + ' ' + query)
if re.search(r'\bofficial\b', user, re.I) and domains and 'site:' not in query:
query = f'{query} site:{domains[0]}'
if (re.search(r'\b(?:latest|recent|current|today|news|updates?)\b', user, re.I)
if (
re.search(r'\bofficial\b', user, re.I)
and re.search(r'\b(?:manual|handbook|pdf)\b', user, re.I)
and not re.search(r'(?:filetype:pdf|\.pdf)\b', query, re.I)
):
query = f'{query} filetype:pdf'
if (current_intent
and not re.search(r'\b(?:latest|recent|current|today|news|updates?|20\d{2})\b', query, re.I)):
query = f'{query} latest {datetime.now(timezone.utc).year}'
query = f'{query} latest {current_year}'
if current_intent and not normalized.get('time_filter'):
if re.search(r'\b(?:version|release|driver)\b', user, re.I):
normalized['time_filter'] = 'year'
elif re.search(r"\btoday(?:'s)?\b", user, re.I):
normalized['time_filter'] = 'day'
elif re.search(r'\brecent\b', user, re.I):
normalized['time_filter'] = 'month'
else:
normalized['time_filter'] = 'week'
normalized['query'] = query
return normalized
@@ -4445,6 +4483,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
)
args = preserve_requested_web_recency(
name, args, user_text=direct_user_text,
prior_search_intents=successful_search_intents,
)
args = preserve_requested_email_account(
name, args, user_text=direct_user_text,
@@ -5096,6 +5135,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
and re.search(r'\bofficial\b', direct_user_text, re.I)):
if not official_source_retry_attempted and round_number < round_limit:
official_source_retry_attempted = True
force_web_search_next_round = True
round_recovery_messages.append(
'No verifiable official-domain source was returned. Search once more '
'with the official organization or domain made explicit; do not cite '
+55
View File
@@ -1916,6 +1916,61 @@ def test_followup_search_must_change_subject_angle_not_only_freshness():
)
def test_current_search_arguments_repair_stale_year_and_add_freshness():
args = preserve_requested_web_recency(
'web_search',
{'query': 'Norway current events updated 2025'},
user_text="What's happening in Norway?",
)
assert '2025' not in args['query']
assert str(__import__('datetime').datetime.now(__import__('datetime').timezone.utc).year) in args['query']
assert args['time_filter'] == 'week'
def test_missing_refinement_query_is_grounded_in_user_request():
args = preserve_requested_web_recency(
'web_search',
{'time_filter': 'day'},
user_text='What are the latest important AI developments?',
prior_search_intents=['important ai developments'],
)
assert 'AI developments' in args['query']
assert 'corroborating analysis' in args['query']
assert args['time_filter'] == 'day'
def test_official_manual_search_requests_direct_pdf_results():
args = preserve_requested_web_recency(
'web_search',
{'query': 'WIKING Miro stove manual official source'},
user_text='Find the official English WIKING Miro stove manual online.',
)
assert args['query'].endswith('filetype:pdf')
def test_unknown_official_domain_rejects_reseller_but_accepts_direct_document():
raw = '''
[1] WIKING Miro 4 Wood Burning Stove
https://scottishstovecentre.co.uk/product/wiking-miro-4/
[2] WIKING Miro Installation and User Manual
https://www.hwam.com/pub/media/wiking/53-0756_Miro_EN.pdf
'''
links = web_source_links(
raw, max_items=2, prefer_official=True,
query='WIKING Miro stove manual official source filetype:pdf',
)
assert links == [(
'https://www.hwam.com/pub/media/wiking/53-0756_Miro_EN.pdf',
'[Source: WIKING Miro Installation and User Manual]'
'(https://www.hwam.com/pub/media/wiking/53-0756_Miro_EN.pdf)',
)]
def test_web_fetch_collapses_single_and_batch_url_fields_without_losing_targets():
tool, args = normalize_preview_function_args('web_fetch', {
'url': 'https://example.org/a',