Enforce search source restrictions and expand live prompt coverage

This commit is contained in:
pewdiepie-archdaemon
2026-09-17 20:14:32 +00:00
parent 7fa6c93fb3
commit a48ab46f7d
3 changed files with 114 additions and 6 deletions
+18 -4
View File
@@ -20,7 +20,7 @@ const token = Object.entries(sessions).find(([, value]) => value?.username === o
if (!token) throw Error(`No active ${owner} session`);
const marker = `ody-search-${crypto.randomUUID()}`;
const report = { run, owner, marker, model, endpointUrl, status: 'running', scenarios: [], turns: [], privacy: 'Public synthetic queries only; fetched bodies and private data are not retained.' };
const report = { run, owner, marker, model, endpointUrl, status: 'running', scenarios: [], turns: [], privacy: 'Public test queries and bounded public tool evidence; no private account data.' };
const save = () => fs.writeFileSync(reportPath, JSON.stringify(report, null, 2) + '\n');
fs.mkdirSync(path.dirname(reportPath), { recursive: true }); save();
const canonical = value => String(value || '').replace(/^mcp__email__/, '');
@@ -68,6 +68,10 @@ async function send(page, prompt) {
rounds: metrics?.agent_rounds ?? null,
actual_model: metrics?.model ?? null,
tools: starts, outputs, final,
evidence: events.filter(event => event.type === 'tool_output').map(event => ({
tool: canonical(event.tool), arguments: event.command,
output: String(event.output || '').slice(0, 10000), error: Boolean(event.error),
})),
runtime_error: events.some(event => event.type === 'error') || /v3 test encountered an error/i.test(final),
};
report.turns.push(observation); save();
@@ -90,6 +94,14 @@ try {
['research', ['How do sodium ion batteries compare with lithium ion for home storage? Find evidence and explain tradeoffs.'], true],
['no-web-typo', ['whats 12 tims 7'], false],
['no-web-greeting', ['helo'], false],
['news-short', ['ai news today'], true],
['news-natural', ['Catch me up on the biggest AI developments this week. Explain why they matter and link your sources.'], true],
['research-typo', ['reserch sodium ion vs lithium batterys for home stroage. whats the tradeof? sources pls'], true],
['official-domain', ['latest Python stable release? use only python.org sources'], true],
['context-refinement', ['Find current Firefox privacy documentation from Mozilla.', 'How does that compare with Chrome? Find official sources for that too.'], true],
['evidence-reuse', ['Find the official Python release page.', 'Explain what you found in plain English. Do not search again.'], [true, false]],
['no-web-rewrite', ['fix spelling: i recieved the calender invte'], false],
['no-web-compound', ['helo can u explain what a web browser is? no search needed'], false],
];
async function runCase([name, prompts, needsWeb]) {
const scenario = { name, status: 'running', turns: [] };
@@ -98,7 +110,8 @@ try {
try {
id = await createSession(context, `[search-variety] ${name} ${marker}`);
page = await preparePage(context, id);
for (const prompt of prompts) {
for (const [turnIndex, prompt] of prompts.entries()) {
const turnNeedsWeb = Array.isArray(needsWeb) ? needsWeb[turnIndex] : needsWeb;
const turn = await send(page, prompt);
const tools = turn.starts.map(x => x.tool);
const checks = {
@@ -106,7 +119,7 @@ try {
model_matches: turn.actual_model === model,
nonempty_answer: turn.final.trim().length > 0,
no_reasoning_leak: noLeak(turn.final),
expected_web_use: needsWeb ? tools.some(x => ['web_search', 'web_fetch', 'private_browser'].includes(x)) : tools.length === 0,
expected_web_use: turnNeedsWeb ? tools.some(x => ['web_search', 'web_fetch', 'private_browser'].includes(x)) : tools.length === 0,
};
scenario.turns.push({ prompt, checks, seconds: turn.seconds, tools,
status: Object.values(checks).every(Boolean) ? 'mechanics_passed' : 'failed',
@@ -121,7 +134,8 @@ try {
}
}
// Two simultaneous conversations keep endpoint contention bounded.
const queue = [...cases];
const selected = new Set((process.env.CASES || '').split(',').filter(Boolean));
const queue = cases.filter(([name]) => !selected.size || selected.has(name));
await Promise.all([0, 1].map(async () => { while (queue.length) await runCase(queue.shift()); }));
} else {
+35 -2
View File
@@ -173,6 +173,10 @@ _EMPTY_RESULT_RELAXATION_TERMS = {
def _relaxed_query_after_empty(query: str) -> str:
"""Remove request scaffolding once an exact provider query returns nothing."""
# Token-based relaxation cannot preserve search operators, quoted phrases,
# or exclusions. Do not silently broaden an explicit source constraint.
if re.search(r'\b\w+:|["\u201c\u201d]|(?:^|\s)-\S', str(query or "")):
return ""
tokens = re.findall(r"[A-Za-z0-9][A-Za-z0-9_.+-]*", str(query or ""))
retained = [
token for token in tokens
@@ -281,12 +285,40 @@ def _result_has_query_overlap(query: str, result: dict) -> bool:
def _filter_low_relevance_results(query: str, results: list[dict]) -> list[dict]:
if not results:
return []
relevant = [result for result in results if _result_has_query_overlap(query, result)]
relevant = [result for result in results
if _result_matches_site_scope(query, result)
and _result_has_query_overlap(query, result)]
# Only reject a provider when it returned a fully off-topic page set. Mixed
# result pages are common; ranking can handle those.
return relevant if relevant else []
def _result_matches_site_scope(query: str, result: dict) -> bool:
"""Enforce explicit site constraints even when a provider ignores them."""
scopes = re.findall(r'(?<!\S)(-?)site:([^\s()]+)', query, re.IGNORECASE)
if not scopes:
return True
try:
target = urlparse(str(result.get("url") or ""))
if target.scheme not in {"https", "http"} or not target.hostname:
return False
host = target.hostname.lower().rstrip(".")
included = []
for excluded, scope in scopes:
parsed = urlparse(scope if "://" in scope else "https://" + scope)
domain = (parsed.hostname or "").lower().rstrip(".")
matches = bool(domain) and (host == domain or host.endswith("." + domain))
if parsed.path and parsed.path != "/":
matches = matches and target.path.startswith(parsed.path)
if excluded and matches:
return False
if not excluded:
included.append(matches)
return any(included) if included else True
except ValueError:
return False
_SCHOLARLY_QUERY_CUE_RE = re.compile(
r"\b(?:paper|preprint|arxiv|proceedings|table\s+\d+|figure\s+\d+|"
r"appendix\s+[a-z0-9]+|benchmark(?:s)?)\b",
@@ -689,7 +721,8 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None)
# providers; the returned official URL lets the agent proceed to PDF tools.
scholarly_title = _scholarly_title_from_query(provider_query)
if scholarly_title:
direct_results = _direct_scholarly_title_results(scholarly_title, count)
direct_results = [result for result in _direct_scholarly_title_results(scholarly_title, count)
if _result_matches_site_scope(provider_query, result)]
if direct_results:
_record_query(provider_query, True, cache_hit=False)
return direct_results[:count]
@@ -1,4 +1,65 @@
from services.search import core
import pytest
@pytest.mark.parametrize('url,accepted', [
('https://python.org/downloads/', True),
('https://docs.python.org/3/', True),
('https://python.org.evil.example/downloads/', False),
('https://evil.example/python.org', False),
('https://evil.example/?site=python.org', False),
('https://python.org@evil.example/', False),
('https://en.wikipedia.org/wiki/Microsoft_campus', False),
])
def test_provider_results_must_obey_site_scope(url, accepted):
result = {'url': url, 'title': 'Python official release source python.org'}
assert bool(core._filter_low_relevance_results(
'latest Python release site:python.org', [result],
)) is accepted
def test_site_scope_exclusions_and_unrestricted_queries():
result = {'url': 'https://docs.python.org/3/'}
assert core._result_matches_site_scope('Python documentation', result)
assert not core._result_matches_site_scope('Python -site:python.org', result)
assert core._result_matches_site_scope('site:example.org OR site:python.org', result)
assert not core._result_matches_site_scope('site:python.org/downloads/', result)
@pytest.mark.parametrize('query', [
'latest Python official site:python.org',
'Python manual -site:example.org',
'official manual filetype:pdf',
'official "exact phrase" manual',
])
def test_relaxation_never_drops_explicit_query_constraints(query):
assert core._empty_result_query_relaxations(query) == []
@pytest.mark.parametrize('comprehensive', [False, True])
def test_out_of_scope_provider_results_never_become_evidence(monkeypatch, tmp_path, comprehensive):
queries = []
def provider(name, query, count, time_filter=None):
queries.append(query)
return [{'title': 'Python release official python.org',
'url': 'https://evil.example/python.org', 'snippet': 'Python release'}]
monkeypatch.setattr(core, 'SEARCH_CACHE_DIR', tmp_path)
monkeypatch.setattr(core, 'search_cache_index', {})
monkeypatch.setattr(core, '_get_search_settings', lambda: {'search_provider': 'searxng'})
monkeypatch.setattr(core, '_build_provider_chain', lambda primary: ['searxng'])
monkeypatch.setattr(core, '_call_provider', provider)
monkeypatch.setattr(core, '_record_query', lambda *a, **k: None)
monkeypatch.setattr(core, 'cleanup_cache', lambda *a, **k: None)
def forbidden_fetch(*a, **k):
pytest.fail('Out-of-scope results must not be fetched')
monkeypatch.setattr(core, 'fetch_webpage_content', forbidden_fetch)
query = 'latest Python release site:python.org'
if comprehensive:
_, sources = core.comprehensive_web_search(query, return_sources=True)
else:
sources = core.searxng_search_results(query)
assert sources == []
assert queries and set(queries) == {query}
def test_searxng_chain_always_keeps_distinct_private_engine_fallback(monkeypatch):