broaden empty searches through resilient providers

This commit is contained in:
pewdiepie-archdaemon
2026-09-17 19:12:28 +00:00
parent e22cf4b500
commit e48ea98d2f
3 changed files with 162 additions and 4 deletions
+86 -4
View File
@@ -137,10 +137,11 @@ def _build_provider_chain(primary: str) -> List[str]:
configured = [provider for provider in chain if provider_configured(provider)]
for provider in set(chain) - set(configured):
logger.warning("Skipping unconfigured search provider: %s", provider)
if primary == "searxng" and configured == ["searxng"]:
# No usable configured fallback: try a separate engine on the same
# private metasearch instance before reporting retrieval failure.
configured.append("searxng_yep")
if primary == "searxng" and "searxng_yep" not in configured:
# The no-key DuckDuckGo fallback can be configured yet unavailable or
# CAPTCHA-limited. Always retain a distinct engine on the private
# metasearch instance before reporting retrieval failure.
configured.insert(1, "searxng_yep")
return configured
@@ -164,6 +165,41 @@ _SEARCH_QUERY_FILLER = {
_SHORT_QUERY_SUBJECTS = {"ai", "ar", "eu", "uk", "us", "vr"}
_EMPTY_RESULT_RELAXATION_TERMS = {
"find", "search", "lookup", "look", "online", "official", "source",
"sources", "english", "download", "please", "latest", "current",
}
def _relaxed_query_after_empty(query: str) -> str:
"""Remove request scaffolding once an exact provider query returns nothing."""
tokens = re.findall(r"[A-Za-z0-9][A-Za-z0-9_.+-]*", str(query or ""))
retained = [
token for token in tokens
if token.casefold() not in _EMPTY_RESULT_RELAXATION_TERMS
]
relaxed = " ".join(retained).strip()
return relaxed if len(retained) >= 2 and relaxed.casefold() != str(query or "").strip().casefold() else ""
def _empty_result_query_relaxations(query: str) -> list[str]:
"""Return bounded increasingly broad discovery queries for an empty SERP."""
first = _relaxed_query_after_empty(query)
candidates = [first] if first else []
if first:
document_terms = {
"manual", "manuals", "guide", "guides", "instructions", "instruction",
"operator", "owners", "owner", "pdf", "documentation", "docs",
}
entity_tokens = [
token for token in first.split()
if token.casefold() not in document_terms
]
entity_query = " ".join(entity_tokens).strip()
if len(entity_tokens) >= 2 and entity_query.casefold() != first.casefold():
candidates.append(entity_query)
return list(dict.fromkeys(candidate for candidate in candidates if candidate))
_WEATHER_QUERY_HINTS = {
"weather", "forecast", "forecasts", "temperature", "temperatures",
"rain", "raining", "precipitation", "humid", "humidity", "wind",
@@ -698,6 +734,25 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None)
if results:
break
if not results:
for relaxed_query in _empty_result_query_relaxations(provider_query):
logger.info(
"Exact search returned no evidence for %r; retrying broadened query %r",
provider_query, relaxed_query,
)
for provider_name in provider_chain:
try:
results = _call_provider(provider_name, relaxed_query, count, time_filter)
results = _filter_low_relevance_results(relaxed_query, results)
except Exception as exc:
error_logger.error("Relaxed %s search failed: %s", provider_name, exc)
results = []
if results:
break
if results:
provider_query = relaxed_query
break
results = _augment_scholarly_results(provider_query, results, count)
success = bool(results)
@@ -815,6 +870,33 @@ def comprehensive_web_search(
elif empty:
provider_attempts[provider_name] = "empty"
if not search_results:
for relaxed_query in _empty_result_query_relaxations(provider_query):
logger.info(
"Comprehensive search empty for %r; retrying broadened query %r",
provider_query, relaxed_query,
)
for provider_name in provider_chain:
try:
search_results = _call_provider(
provider_name, relaxed_query, fetch_count, time_filter,
)
search_results = _filter_low_relevance_results(
relaxed_query, search_results,
)
except Exception as exc:
provider_attempts[f"{provider_name}:relaxed"] = f"error: {exc}"
search_results = []
if search_results:
provider_attempts[f"{provider_name}:relaxed"] = (
f"ok ({len(search_results)})"
)
provider_query = relaxed_query
break
provider_attempts[f"{provider_name}:relaxed"] = "empty"
if search_results:
break
search_results = _augment_scholarly_results(
provider_query,
search_results,
+14
View File
@@ -3796,6 +3796,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
successful_web_searches = 0
successful_web_retrievals = 0
retrieved_web_sources = []
discovered_web_sources = []
browser_navigation_outcomes = {}
failed_call_counts = {}
semantic_attempt_counts = {}
@@ -4294,6 +4295,14 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
# it or ask the model for another round just for formatting.
missing_links = [link for target, link in entity_result_links.items()
if f']({target})' not in content]
if (
broad_current_web_request(direct_user_text)
and not re.search(r'https?://\S+', content or '')
):
missing_links.extend(
link for link in discovered_web_sources[:5]
if link not in missing_links
)
if missing_links:
suffix = ('\n\n' if content else '') + '\n'.join(missing_links)
content += suffix
@@ -4796,6 +4805,11 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
if not failed and canonical(actual_tool) == 'web_search':
if result.get('evidence_status') != 'empty':
successful_web_searches += 1
for _title, source_url in web_source_links(
output, max_items=5, query=args.get('query', ''),
):
if source_url not in discovered_web_sources:
discovered_web_sources.append(source_url)
successful_intent = normalized_search_intent(args.get('query'))
if successful_intent and result.get('evidence_status') != 'empty':
successful_search_intents.append(successful_intent)
@@ -0,0 +1,62 @@
from services.search import core
def test_searxng_chain_always_keeps_distinct_private_engine_fallback(monkeypatch):
import services.search.providers as providers
monkeypatch.setattr(providers, 'provider_configured', lambda name: True)
monkeypatch.setattr(core, '_get_search_settings', lambda: {
'search_fallback_chain': ['duckduckgo'],
})
assert core._build_provider_chain('searxng') == [
'searxng', 'searxng_yep', 'duckduckgo',
]
def test_empty_document_search_relaxes_scaffolding_then_entity():
assert core._empty_result_query_relaxations(
'Find WIKING Miro 3 English manual official source online'
) == [
'WIKING Miro 3 manual',
'WIKING Miro 3',
]
def test_search_uses_entity_relaxation_only_after_exact_queries_are_empty(
monkeypatch, tmp_path,
):
calls = []
def provider(name, query, count, time_filter=None):
calls.append((name, query))
if name == 'searxng_yep' and query == 'WIKING Miro 3':
return [{
'title': 'WIKING Miro 3+ black with lower door - HWAM',
'url': 'https://www.hwam.com/miro3-side-glass-lower-door',
'snippet': 'Official WIKING Miro 3 product page.',
}]
return []
monkeypatch.setattr(core, 'SEARCH_CACHE_DIR', tmp_path)
monkeypatch.setattr(core, 'search_cache_index', {})
monkeypatch.setattr(core, '_get_search_settings', lambda: {
'search_provider': 'searxng',
'search_fallback_chain': ['duckduckgo'],
})
monkeypatch.setattr(core, '_build_provider_chain', lambda primary: [
'searxng', 'searxng_yep', 'duckduckgo',
])
monkeypatch.setattr(core, '_call_provider', provider)
monkeypatch.setattr(core, '_record_query', lambda *args, **kwargs: None)
monkeypatch.setattr(core, 'cleanup_cache', lambda *args, **kwargs: None)
results = core.searxng_search_results(
'WIKING Miro 3 English manual official source', count=5,
)
assert results[0]['url'] == 'https://www.hwam.com/miro3-side-glass-lower-door'
assert ('searxng_yep', 'WIKING Miro 3') in calls
assert calls.index(('searxng_yep', 'WIKING Miro 3')) > calls.index(
('searxng_yep', 'WIKING Miro 3 manual')
)