"""Regression tests for the canonical services.search provider implementation.
The old src.search provider path aliases this module; these tests pin the
behavior at the single implementation point.
"""
import sys
import pytest
from services.search import core
from services.search import providers
@pytest.mark.parametrize('query,category', [
('AI developments this week', 'news'),
('recent developments in battery manufacturing', 'news'),
('latest developments in quantum computing', 'news'),
('web development tutorial this week', 'general'),
('current Firefox privacy documentation', 'general'),
('historical developments in mathematics', 'general'),
('latest Python version developments', 'general'),
])
def test_temporal_developments_select_news_not_reference_search(monkeypatch, query, category):
seen = []
class Response:
def raise_for_status(self): pass
def json(self):
return {'results': [{'title': 'Result', 'url': 'https://example.org/report'}]}
def get(*args, **kwargs):
seen.append(kwargs['params'])
return Response()
monkeypatch.setattr(providers, '_get_search_instance', lambda: 'http://searx.test')
monkeypatch.setattr(providers, '_get_search_settings', lambda: {})
monkeypatch.setattr(providers, '_get_provider_key', lambda name: '')
monkeypatch.setattr(providers.httpx, 'get', get)
providers.searxng_search_api(query, time_filter='week')
assert seen[0]['categories'] == category
assert seen[0]['time_range'] == 'week'
def test_dead_credentialed_fallback_is_skipped_for_same_instance_engine(monkeypatch):
monkeypatch.setattr(core, "_get_search_settings", lambda: {"search_fallback_chain": ["google_pse"]})
monkeypatch.setattr(providers, "_get_search_settings", lambda: {})
monkeypatch.setattr(providers, "_get_provider_key", lambda name: "")
assert core._build_provider_chain("searxng") == ["searxng", "searxng_yep"]
def test_valid_configured_fallback_is_preserved(monkeypatch):
monkeypatch.setattr(core, "_get_search_settings", lambda: {"search_fallback_chain": ["google_pse"]})
monkeypatch.setattr(providers, "_get_search_settings", lambda: {"google_pse_cx": "test-cx"})
monkeypatch.setattr(providers, "_get_provider_key", lambda name: "test-key")
assert core._build_provider_chain("searxng") == [
"searxng", "searxng_yep", "google_pse",
]
def test_service_safesearch_values_match_provider_contract(monkeypatch):
monkeypatch.setattr(providers, "_get_search_settings", lambda: {"search_safesearch": "strict"})
assert providers._safesearch_for("searxng") == "2"
assert providers._safesearch_for("brave") == "strict"
assert providers._safesearch_for("duckduckgo_lib") == "on"
assert providers._safesearch_for("duckduckgo_html") == "1"
assert providers._safesearch_for("google_pse") == "active"
assert providers._safesearch_for("serper") == "active"
monkeypatch.setattr(providers, "_get_search_settings", lambda: {"search_safesearch": "off"})
assert providers._safesearch_for("searxng") == "0"
assert providers._safesearch_for("brave") == "off"
assert providers._safesearch_for("duckduckgo_lib") == "off"
assert providers._safesearch_for("duckduckgo_html") == "-2"
assert providers._safesearch_for("google_pse") is None
assert providers._safesearch_for("serper") is None
def test_service_searxng_json_sends_safesearch(monkeypatch):
seen = {}
class _Response:
def raise_for_status(self):
return None
def json(self):
return {
"results": [
{"title": "Result", "url": "https://example.com", "content": "Snippet"}
]
}
def fake_get(url, **kwargs):
seen["url"] = url
seen["params"] = kwargs["params"]
return _Response()
monkeypatch.setattr(providers, "_get_search_instance", lambda: "http://searx.test")
monkeypatch.setattr(providers, "_get_search_settings", lambda: {"search_safesearch": "moderate"})
monkeypatch.setattr(providers.httpx, "get", fake_get)
results = providers.searxng_search_api("odysseus", count=1)
assert results
assert seen["url"] == "http://searx.test/search"
assert seen["params"]["safesearch"] == "1"
@pytest.mark.parametrize('query,expected_time', [
('latest ollama release version github', 'day'),
('current Firefox Chrome privacy features comparison', 'day'),
('Sony headphone manual', 'day'),
])
def test_service_searxng_latest_release_uses_general_search(monkeypatch, query, expected_time):
seen = {}
class _Response:
def raise_for_status(self):
return None
def json(self):
return {
"results": [
{
"title": "ollama/ollama releases",
"url": "https://github.com/ollama/ollama/releases",
"content": "Latest release v0.33.0",
}
]
}
def fake_get(url, **kwargs):
seen["params"] = kwargs["params"]
return _Response()
monkeypatch.setattr(providers, "_get_search_instance", lambda: "http://searx.test")
monkeypatch.setattr(providers, "_get_search_settings", lambda: {"search_safesearch": "strict"})
monkeypatch.setattr(providers.httpx, "get", fake_get)
results = providers.searxng_search_api(
query,
count=1,
time_filter="day",
)
assert results
assert seen["params"]["categories"] == "general"
assert seen["params"]["engines"] == providers._GENERAL_ENGINES
assert seen['params'].get('time_range') == expected_time
def test_service_searxng_specific_current_event_uses_news_search(monkeypatch):
seen = {}
class _Response:
def raise_for_status(self):
return None
def json(self):
return {
"results": [
{
"title": "British widow faces deportation from Sweden",
"url": "https://example.com/news/story",
"content": "The 78-year-old has lived in Sweden for 22 years.",
}
]
}
def fake_get(url, **kwargs):
seen["params"] = kwargs["params"]
return _Response()
monkeypatch.setattr(providers, "_get_search_instance", lambda: "http://sear.test")
monkeypatch.setattr(providers, "_get_search_settings", lambda: {"search_safesearch": "moderate"})
monkeypatch.setattr(providers.httpx, "get", fake_get)
results = providers.searxng_search_api(
"Sweden 78 year old British woman deportation Brexit residence application",
count=5,
)
assert results
assert seen["params"]["categories"] == "news"
assert "engines" not in seen["params"]
def test_low_relevance_filter_ignores_generic_freshness_terms():
results = core._filter_low_relevance_results(
"latest ollama release version github",
[
{
"title": "Fox News - Breaking News Updates",
"url": "https://www.foxnews.com/",
"snippet": "Latest Current News: U.S., World, Entertainment.",
},
{
"title": "Releases · ollama/ollama - GitHub",
"url": "https://github.com/ollama/ollama/releases",
"snippet": "Get up and running with Kimi, DeepSeek, Qwen and other models.",
},
],
)
assert [r["url"] for r in results] == ["https://github.com/ollama/ollama/releases"]
def test_low_relevance_filter_rejects_one_broad_match_for_specific_query():
results = core._filter_low_relevance_results(
"Sweden 78 year old British woman deportation Brexit residence application",
[
{
"title": "Geography of Sweden",
"url": "https://en.wikipedia.org/wiki/Geography_of_Sweden",
"snippet": "Sweden is a country in Northern Europe.",
},
{
"title": "British widow faces deportation from Sweden after 22 years",
"url": "https://example.com/news/british-widow-sweden",
"snippet": "A 78-year-old woman missed a Brexit residence application.",
},
],
)
assert [r["url"] for r in results] == [
"https://example.com/news/british-widow-sweden"
]
def test_low_relevance_filter_treats_ai_as_subject_not_news_as_subject():
results = core._filter_low_relevance_results(
"Latest news in AI",
[
{
"title": "Anthropic researcher quits over AI risks",
"url": "https://example.com/technology/anthropic-ai",
"snippet": "The departure highlights concern inside AI labs.",
},
{
"title": "Latest world news and headlines",
"url": "https://example.com/world",
"snippet": "Breaking updates from around the world.",
},
],
)
assert [r["url"] for r in results] == [
"https://example.com/technology/anthropic-ai"
]
def test_provider_query_removes_generic_interrogative_shell():
assert core._provider_friendly_query(
"What year did Ethiopia become independent"
) == "Ethiopia become independent year"
def test_provider_query_separates_conversational_freshness_shell_from_subject():
assert core._provider_friendly_query(
"Any latest info on quantum physics"
) == "quantum physics"
assert core._meaningful_query_terms(
"Any latest info on quantum physics"
) == ["quantum", "physics"]
def test_relevance_matching_accepts_conservative_word_family_variants():
results = core._filter_low_relevance_results(
"Ethiopia become independent year",
[
{
"title": "History of Ethiopia",
"url": "https://example.com/ethiopia-history",
"snippet": "The country's independence and periods of occupation.",
},
{
"title": "Calendar year",
"url": "https://example.com/calendar-year",
"snippet": "A year has twelve months.",
},
],
)
assert [r["url"] for r in results] == ["https://example.com/ethiopia-history"]
def test_low_relevance_filter_weather_queries_require_location_terms():
results = core._filter_low_relevance_results(
"Kyoto weather forecast tomorrow August 27 2026",
[
{
"title": "Weather Tomorrow for Kyoto-shi, Kyoto, Japan",
"url": "https://www.accuweather.com/en/jp/kyoto-shi/224436/weather-tomorrow/224436",
"snippet": "Detailed forecast including temperature and rain.",
},
{
"title": "Seattle, WA Weather Forecast",
"url": "https://www.accuweather.com/en/us/seattle/98104/weather-forecast/351409",
"snippet": "Seattle weather forecast with current conditions.",
},
{
"title": "Kyoto - Wikipedia",
"url": "https://en.wikipedia.org/wiki/Kyoto",
"snippet": "Kyoto is a city in Japan.",
},
{
"title": "Kyoto Travel | Kyoto City Official Guide",
"url": "https://kyoto.travel/en/",
"snippet": "Kyoto tourism tips, itineraries, and things to do.",
},
],
)
assert [r["url"] for r in results] == [
"https://www.accuweather.com/en/jp/kyoto-shi/224436/weather-tomorrow/224436"
]
def test_weather_query_rewrite_moves_location_first():
assert (
core._subject_first_weather_query("What is the weather in Kyoto tomorrow?")
== "Kyoto weather forecast tomorrow"
)
def test_scholarly_title_extraction_prefers_probable_paper_title():
assert core._scholarly_title_from_query(
'"Attention Is All You Need" "Table 2" "Training Cost" FLOPS'
) == "Attention Is All You Need"
assert core._scholarly_title_from_query(
'Find the paper "LLaVA-OneVision: Easy Visual Task Transfer" Table 5'
) == "LLaVA-OneVision: Easy Visual Task Transfer"
assert core._scholarly_title_from_query('search for "ordinary quoted phrase"') == ""
def test_scholarly_arxiv_fallback_prepends_only_strong_title_match(monkeypatch):
seen = {}
atom = """\