mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 06:52:20 +02:00
Enforce search source restrictions and expand live prompt coverage
This commit is contained in:
@@ -20,7 +20,7 @@ const token = Object.entries(sessions).find(([, value]) => value?.username === o
|
||||
if (!token) throw Error(`No active ${owner} session`);
|
||||
|
||||
const marker = `ody-search-${crypto.randomUUID()}`;
|
||||
const report = { run, owner, marker, model, endpointUrl, status: 'running', scenarios: [], turns: [], privacy: 'Public synthetic queries only; fetched bodies and private data are not retained.' };
|
||||
const report = { run, owner, marker, model, endpointUrl, status: 'running', scenarios: [], turns: [], privacy: 'Public test queries and bounded public tool evidence; no private account data.' };
|
||||
const save = () => fs.writeFileSync(reportPath, JSON.stringify(report, null, 2) + '\n');
|
||||
fs.mkdirSync(path.dirname(reportPath), { recursive: true }); save();
|
||||
const canonical = value => String(value || '').replace(/^mcp__email__/, '');
|
||||
@@ -68,6 +68,10 @@ async function send(page, prompt) {
|
||||
rounds: metrics?.agent_rounds ?? null,
|
||||
actual_model: metrics?.model ?? null,
|
||||
tools: starts, outputs, final,
|
||||
evidence: events.filter(event => event.type === 'tool_output').map(event => ({
|
||||
tool: canonical(event.tool), arguments: event.command,
|
||||
output: String(event.output || '').slice(0, 10000), error: Boolean(event.error),
|
||||
})),
|
||||
runtime_error: events.some(event => event.type === 'error') || /v3 test encountered an error/i.test(final),
|
||||
};
|
||||
report.turns.push(observation); save();
|
||||
@@ -90,6 +94,14 @@ try {
|
||||
['research', ['How do sodium ion batteries compare with lithium ion for home storage? Find evidence and explain tradeoffs.'], true],
|
||||
['no-web-typo', ['whats 12 tims 7'], false],
|
||||
['no-web-greeting', ['helo'], false],
|
||||
['news-short', ['ai news today'], true],
|
||||
['news-natural', ['Catch me up on the biggest AI developments this week. Explain why they matter and link your sources.'], true],
|
||||
['research-typo', ['reserch sodium ion vs lithium batterys for home stroage. whats the tradeof? sources pls'], true],
|
||||
['official-domain', ['latest Python stable release? use only python.org sources'], true],
|
||||
['context-refinement', ['Find current Firefox privacy documentation from Mozilla.', 'How does that compare with Chrome? Find official sources for that too.'], true],
|
||||
['evidence-reuse', ['Find the official Python release page.', 'Explain what you found in plain English. Do not search again.'], [true, false]],
|
||||
['no-web-rewrite', ['fix spelling: i recieved the calender invte'], false],
|
||||
['no-web-compound', ['helo can u explain what a web browser is? no search needed'], false],
|
||||
];
|
||||
async function runCase([name, prompts, needsWeb]) {
|
||||
const scenario = { name, status: 'running', turns: [] };
|
||||
@@ -98,7 +110,8 @@ try {
|
||||
try {
|
||||
id = await createSession(context, `[search-variety] ${name} ${marker}`);
|
||||
page = await preparePage(context, id);
|
||||
for (const prompt of prompts) {
|
||||
for (const [turnIndex, prompt] of prompts.entries()) {
|
||||
const turnNeedsWeb = Array.isArray(needsWeb) ? needsWeb[turnIndex] : needsWeb;
|
||||
const turn = await send(page, prompt);
|
||||
const tools = turn.starts.map(x => x.tool);
|
||||
const checks = {
|
||||
@@ -106,7 +119,7 @@ try {
|
||||
model_matches: turn.actual_model === model,
|
||||
nonempty_answer: turn.final.trim().length > 0,
|
||||
no_reasoning_leak: noLeak(turn.final),
|
||||
expected_web_use: needsWeb ? tools.some(x => ['web_search', 'web_fetch', 'private_browser'].includes(x)) : tools.length === 0,
|
||||
expected_web_use: turnNeedsWeb ? tools.some(x => ['web_search', 'web_fetch', 'private_browser'].includes(x)) : tools.length === 0,
|
||||
};
|
||||
scenario.turns.push({ prompt, checks, seconds: turn.seconds, tools,
|
||||
status: Object.values(checks).every(Boolean) ? 'mechanics_passed' : 'failed',
|
||||
@@ -121,7 +134,8 @@ try {
|
||||
}
|
||||
}
|
||||
// Two simultaneous conversations keep endpoint contention bounded.
|
||||
const queue = [...cases];
|
||||
const selected = new Set((process.env.CASES || '').split(',').filter(Boolean));
|
||||
const queue = cases.filter(([name]) => !selected.size || selected.has(name));
|
||||
await Promise.all([0, 1].map(async () => { while (queue.length) await runCase(queue.shift()); }));
|
||||
} else {
|
||||
|
||||
|
||||
+35
-2
@@ -173,6 +173,10 @@ _EMPTY_RESULT_RELAXATION_TERMS = {
|
||||
|
||||
def _relaxed_query_after_empty(query: str) -> str:
|
||||
"""Remove request scaffolding once an exact provider query returns nothing."""
|
||||
# Token-based relaxation cannot preserve search operators, quoted phrases,
|
||||
# or exclusions. Do not silently broaden an explicit source constraint.
|
||||
if re.search(r'\b\w+:|["\u201c\u201d]|(?:^|\s)-\S', str(query or "")):
|
||||
return ""
|
||||
tokens = re.findall(r"[A-Za-z0-9][A-Za-z0-9_.+-]*", str(query or ""))
|
||||
retained = [
|
||||
token for token in tokens
|
||||
@@ -281,12 +285,40 @@ def _result_has_query_overlap(query: str, result: dict) -> bool:
|
||||
def _filter_low_relevance_results(query: str, results: list[dict]) -> list[dict]:
|
||||
if not results:
|
||||
return []
|
||||
relevant = [result for result in results if _result_has_query_overlap(query, result)]
|
||||
relevant = [result for result in results
|
||||
if _result_matches_site_scope(query, result)
|
||||
and _result_has_query_overlap(query, result)]
|
||||
# Only reject a provider when it returned a fully off-topic page set. Mixed
|
||||
# result pages are common; ranking can handle those.
|
||||
return relevant if relevant else []
|
||||
|
||||
|
||||
def _result_matches_site_scope(query: str, result: dict) -> bool:
|
||||
"""Enforce explicit site constraints even when a provider ignores them."""
|
||||
scopes = re.findall(r'(?<!\S)(-?)site:([^\s()]+)', query, re.IGNORECASE)
|
||||
if not scopes:
|
||||
return True
|
||||
try:
|
||||
target = urlparse(str(result.get("url") or ""))
|
||||
if target.scheme not in {"https", "http"} or not target.hostname:
|
||||
return False
|
||||
host = target.hostname.lower().rstrip(".")
|
||||
included = []
|
||||
for excluded, scope in scopes:
|
||||
parsed = urlparse(scope if "://" in scope else "https://" + scope)
|
||||
domain = (parsed.hostname or "").lower().rstrip(".")
|
||||
matches = bool(domain) and (host == domain or host.endswith("." + domain))
|
||||
if parsed.path and parsed.path != "/":
|
||||
matches = matches and target.path.startswith(parsed.path)
|
||||
if excluded and matches:
|
||||
return False
|
||||
if not excluded:
|
||||
included.append(matches)
|
||||
return any(included) if included else True
|
||||
except ValueError:
|
||||
return False
|
||||
|
||||
|
||||
_SCHOLARLY_QUERY_CUE_RE = re.compile(
|
||||
r"\b(?:paper|preprint|arxiv|proceedings|table\s+\d+|figure\s+\d+|"
|
||||
r"appendix\s+[a-z0-9]+|benchmark(?:s)?)\b",
|
||||
@@ -689,7 +721,8 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None)
|
||||
# providers; the returned official URL lets the agent proceed to PDF tools.
|
||||
scholarly_title = _scholarly_title_from_query(provider_query)
|
||||
if scholarly_title:
|
||||
direct_results = _direct_scholarly_title_results(scholarly_title, count)
|
||||
direct_results = [result for result in _direct_scholarly_title_results(scholarly_title, count)
|
||||
if _result_matches_site_scope(provider_query, result)]
|
||||
if direct_results:
|
||||
_record_query(provider_query, True, cache_hit=False)
|
||||
return direct_results[:count]
|
||||
|
||||
@@ -1,4 +1,65 @@
|
||||
from services.search import core
|
||||
import pytest
|
||||
|
||||
|
||||
@pytest.mark.parametrize('url,accepted', [
|
||||
('https://python.org/downloads/', True),
|
||||
('https://docs.python.org/3/', True),
|
||||
('https://python.org.evil.example/downloads/', False),
|
||||
('https://evil.example/python.org', False),
|
||||
('https://evil.example/?site=python.org', False),
|
||||
('https://python.org@evil.example/', False),
|
||||
('https://en.wikipedia.org/wiki/Microsoft_campus', False),
|
||||
])
|
||||
def test_provider_results_must_obey_site_scope(url, accepted):
|
||||
result = {'url': url, 'title': 'Python official release source python.org'}
|
||||
assert bool(core._filter_low_relevance_results(
|
||||
'latest Python release site:python.org', [result],
|
||||
)) is accepted
|
||||
|
||||
|
||||
def test_site_scope_exclusions_and_unrestricted_queries():
|
||||
result = {'url': 'https://docs.python.org/3/'}
|
||||
assert core._result_matches_site_scope('Python documentation', result)
|
||||
assert not core._result_matches_site_scope('Python -site:python.org', result)
|
||||
assert core._result_matches_site_scope('site:example.org OR site:python.org', result)
|
||||
assert not core._result_matches_site_scope('site:python.org/downloads/', result)
|
||||
|
||||
|
||||
@pytest.mark.parametrize('query', [
|
||||
'latest Python official site:python.org',
|
||||
'Python manual -site:example.org',
|
||||
'official manual filetype:pdf',
|
||||
'official "exact phrase" manual',
|
||||
])
|
||||
def test_relaxation_never_drops_explicit_query_constraints(query):
|
||||
assert core._empty_result_query_relaxations(query) == []
|
||||
|
||||
|
||||
@pytest.mark.parametrize('comprehensive', [False, True])
|
||||
def test_out_of_scope_provider_results_never_become_evidence(monkeypatch, tmp_path, comprehensive):
|
||||
queries = []
|
||||
def provider(name, query, count, time_filter=None):
|
||||
queries.append(query)
|
||||
return [{'title': 'Python release official python.org',
|
||||
'url': 'https://evil.example/python.org', 'snippet': 'Python release'}]
|
||||
monkeypatch.setattr(core, 'SEARCH_CACHE_DIR', tmp_path)
|
||||
monkeypatch.setattr(core, 'search_cache_index', {})
|
||||
monkeypatch.setattr(core, '_get_search_settings', lambda: {'search_provider': 'searxng'})
|
||||
monkeypatch.setattr(core, '_build_provider_chain', lambda primary: ['searxng'])
|
||||
monkeypatch.setattr(core, '_call_provider', provider)
|
||||
monkeypatch.setattr(core, '_record_query', lambda *a, **k: None)
|
||||
monkeypatch.setattr(core, 'cleanup_cache', lambda *a, **k: None)
|
||||
def forbidden_fetch(*a, **k):
|
||||
pytest.fail('Out-of-scope results must not be fetched')
|
||||
monkeypatch.setattr(core, 'fetch_webpage_content', forbidden_fetch)
|
||||
query = 'latest Python release site:python.org'
|
||||
if comprehensive:
|
||||
_, sources = core.comprehensive_web_search(query, return_sources=True)
|
||||
else:
|
||||
sources = core.searxng_search_results(query)
|
||||
assert sources == []
|
||||
assert queries and set(queries) == {query}
|
||||
|
||||
|
||||
def test_searxng_chain_always_keeps_distinct_private_engine_fallback(monkeypatch):
|
||||
|
||||
Reference in New Issue
Block a user