mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-10 08:52:21 +02:00
Squash Odysseus development history
This commit is contained in:
+37
-12
@@ -1,18 +1,43 @@
|
||||
# services/__init__.py
|
||||
"""
|
||||
Service layer — plug-in capabilities for the chat core.
|
||||
"""Service-layer exports with lazy loading.
|
||||
|
||||
Each service:
|
||||
- Does one thing well
|
||||
- Exposes a clean async interface
|
||||
- Can run in-process or as a standalone HTTP service
|
||||
Importing one service, such as ``services.hwfit``, must not initialize every
|
||||
other service. The eager exports previously imported search, document,
|
||||
research, memory, and shell stacks during any ``services.*`` import, making
|
||||
Cookbook hardware/model discovery needlessly slow on a cold process.
|
||||
"""
|
||||
|
||||
from .search import SearchService, SearchResult, SearchResponse
|
||||
from .docs import DocsService, DocChunk, IndexResult
|
||||
from .research import ResearchService, ResearchResult, ResearchSource
|
||||
from .memory import MemoryService, Memory, MemorySearchResult
|
||||
from .shell import ShellService, ShellResult
|
||||
from importlib import import_module
|
||||
|
||||
_LAZY_EXPORTS = {
|
||||
"SearchService": ("search", "SearchService"),
|
||||
"SearchResult": ("search", "SearchResult"),
|
||||
"SearchResponse": ("search", "SearchResponse"),
|
||||
"DocsService": ("docs", "DocsService"),
|
||||
"DocChunk": ("docs", "DocChunk"),
|
||||
"IndexResult": ("docs", "IndexResult"),
|
||||
"ResearchService": ("research", "ResearchService"),
|
||||
"ResearchResult": ("research", "ResearchResult"),
|
||||
"ResearchSource": ("research", "ResearchSource"),
|
||||
"MemoryService": ("memory", "MemoryService"),
|
||||
"Memory": ("memory", "Memory"),
|
||||
"MemorySearchResult": ("memory", "MemorySearchResult"),
|
||||
"ShellService": ("shell", "ShellService"),
|
||||
"ShellResult": ("shell", "ShellResult"),
|
||||
}
|
||||
|
||||
|
||||
def __getattr__(name):
|
||||
target = _LAZY_EXPORTS.get(name)
|
||||
if target is None:
|
||||
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
||||
module_name, attribute = target
|
||||
value = getattr(import_module(f"{__name__}.{module_name}"), attribute)
|
||||
globals()[name] = value
|
||||
return value
|
||||
|
||||
|
||||
def __dir__():
|
||||
return sorted(set(globals()) | set(_LAZY_EXPORTS))
|
||||
|
||||
__all__ = [
|
||||
# Search
|
||||
|
||||
+66954
-19475
File diff suppressed because it is too large
Load Diff
@@ -748,6 +748,7 @@ def rank_models(system, use_case=None, limit=50, search=None, sort="score", quan
|
||||
"is_image_gen": True,
|
||||
"capabilities": im.get("capabilities", []),
|
||||
"description": im.get("description", ""),
|
||||
"dependency_package": im.get("dependency_package", ""),
|
||||
})
|
||||
if use_case == "image_gen":
|
||||
sort_fn = SORT_KEYS.get(sort, SORT_KEYS["score"])
|
||||
@@ -839,7 +840,6 @@ def rank_models(system, use_case=None, limit=50, search=None, sort="score", quan
|
||||
# native AWQ rows only on accelerator servers that can serve them.
|
||||
if (
|
||||
quant == "Q4_K_M"
|
||||
and system.get("gpu_count", 1) >= 2
|
||||
and not (apple_silicon or consumer_amd or is_windows)
|
||||
and native_q == "AWQ-4bit"
|
||||
):
|
||||
|
||||
@@ -7,8 +7,11 @@ import re
|
||||
import time
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from src.constants import DATA_DIR
|
||||
|
||||
# Image models are discovered from HuggingFace collections/search and local cache.
|
||||
# Keep this empty: source-coded repo IDs become hidden recommendations.
|
||||
IMAGE_MODEL_REGISTRY: list[dict[str, Any]] = []
|
||||
@@ -31,6 +34,8 @@ HF_IMAGE_REPO_SEEDS: list[str] = []
|
||||
|
||||
_HF_COLLECTION_CACHE = {"ts": 0.0, "models": []}
|
||||
_HF_COLLECTION_TTL = 30 * 60
|
||||
_IMAGE_COLLECTION_DISK_CACHE = Path(DATA_DIR) / "hwfit" / "image_collection_models.json"
|
||||
_IMAGE_COLLECTION_DISK_TTL = 24 * 3600
|
||||
_HF_VARIANT_CACHE: dict[str, dict[str, str]] = {}
|
||||
_HF_SEARCH_DISABLED_UNTIL = 0.0
|
||||
|
||||
@@ -161,6 +166,12 @@ def _collection_item_to_model(item: dict[str, Any], collection_title: str = "",
|
||||
"speed": est["speed"],
|
||||
"released": "",
|
||||
}
|
||||
# Optional catalog metadata may identify a non-default runtime package.
|
||||
# Keep this data-driven: the fitter must not infer private/model-specific
|
||||
# dependencies from repository names.
|
||||
dependency_package = item.get("dependency_package") or item.get("runtime_dependency")
|
||||
if isinstance(dependency_package, str) and dependency_package.strip():
|
||||
out["dependency_package"] = dependency_package.strip()
|
||||
if mlx_only:
|
||||
out["mlx_only"] = True
|
||||
out["description"] = (out["description"] + " Apple Silicon / MLX only.").strip()
|
||||
@@ -171,6 +182,21 @@ def _fetch_hf_image_collection_models() -> list[dict[str, Any]]:
|
||||
now = time.time()
|
||||
if now - float(_HF_COLLECTION_CACHE.get("ts") or 0) < _HF_COLLECTION_TTL:
|
||||
return list(_HF_COLLECTION_CACHE.get("models") or [])
|
||||
# Reuse the last successful discovery across process restarts. A stale
|
||||
# catalog is preferable to blocking the first image-tab render on several
|
||||
# sequential Hugging Face requests; a later refresh replaces it.
|
||||
if not _HF_COLLECTION_CACHE.get("models"):
|
||||
try:
|
||||
cached = json.loads(_IMAGE_COLLECTION_DISK_CACHE.read_text(encoding="utf-8"))
|
||||
cached_models = cached.get("models") if isinstance(cached, dict) else None
|
||||
cached_ts = float(cached.get("fetched_at") or 0) if isinstance(cached, dict) else 0
|
||||
if isinstance(cached_models, list) and cached_models:
|
||||
_HF_COLLECTION_CACHE["ts"] = cached_ts
|
||||
_HF_COLLECTION_CACHE["models"] = cached_models
|
||||
if now - cached_ts < _IMAGE_COLLECTION_DISK_TTL:
|
||||
return list(cached_models)
|
||||
except (OSError, ValueError, TypeError):
|
||||
pass
|
||||
models: list[dict[str, Any]] = []
|
||||
for slug, mlx_only in [(slug, False) for slug in HF_IMAGE_COLLECTIONS] + [(slug, True) for slug in HF_MLX_IMAGE_COLLECTIONS]:
|
||||
url = f"https://huggingface.co/api/collections/{slug}"
|
||||
@@ -186,9 +212,24 @@ def _fetch_hf_image_collection_models() -> list[dict[str, Any]]:
|
||||
model = _collection_item_to_model(item, title, mlx_only=mlx_only)
|
||||
if model:
|
||||
models.append(model)
|
||||
if models:
|
||||
_HF_COLLECTION_CACHE["ts"] = now
|
||||
_HF_COLLECTION_CACHE["models"] = models
|
||||
try:
|
||||
_IMAGE_COLLECTION_DISK_CACHE.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = _IMAGE_COLLECTION_DISK_CACHE.with_suffix(".tmp")
|
||||
tmp.write_text(json.dumps({"fetched_at": now, "models": models}), encoding="utf-8")
|
||||
tmp.replace(_IMAGE_COLLECTION_DISK_CACHE)
|
||||
except OSError:
|
||||
pass
|
||||
return list(models)
|
||||
# Preserve stale results if the network is unavailable. The in-memory
|
||||
# timestamp prevents every subsequent ranking request from retrying it.
|
||||
if _HF_COLLECTION_CACHE.get("models"):
|
||||
_HF_COLLECTION_CACHE["ts"] = now
|
||||
return list(_HF_COLLECTION_CACHE["models"])
|
||||
_HF_COLLECTION_CACHE["ts"] = now
|
||||
_HF_COLLECTION_CACHE["models"] = models
|
||||
return list(models)
|
||||
return []
|
||||
|
||||
|
||||
def _hf_model_search(query: str, limit: int = 10) -> list[dict[str, Any]]:
|
||||
@@ -420,6 +461,7 @@ def rank_image_models(system, search=None, sort="fit"):
|
||||
"capabilities": model["capabilities"],
|
||||
"description": model["description"],
|
||||
"released": model.get("released", ""),
|
||||
"dependency_package": model.get("dependency_package", ""),
|
||||
})
|
||||
|
||||
# Sort
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
"""Install tracked built-in skills into the shared immutable skill catalog."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Iterable
|
||||
|
||||
from .skill_format import Skill
|
||||
from .skills import SkillsManager
|
||||
|
||||
|
||||
_BUILTIN_ROOT = Path(__file__).resolve().parents[2] / "resources" / "skills"
|
||||
_SYNC_FIELDS = (
|
||||
"name",
|
||||
"description",
|
||||
"version",
|
||||
"category",
|
||||
"tags",
|
||||
"status",
|
||||
"confidence",
|
||||
"source",
|
||||
"owner",
|
||||
"when_to_use",
|
||||
"procedure",
|
||||
"pitfalls",
|
||||
"verification",
|
||||
"platforms",
|
||||
"requires_toolsets",
|
||||
"fallback_for_toolsets",
|
||||
"body_extra",
|
||||
)
|
||||
|
||||
|
||||
def install_builtin_skills(manager: SkillsManager, owners: Iterable[str]) -> int:
|
||||
"""Copy missing built-in skills into the ownerless shared catalog.
|
||||
|
||||
Built-ins are explicitly marked and remain ownerless because the on-disk
|
||||
skill path is not owner-qualified. ``SkillsManager.load(owner=...)``
|
||||
exposes only these immutable built-ins in addition to that owner's files.
|
||||
Installation is safe before first-user setup because no owner identity is
|
||||
assigned and unauthenticated requests still cannot access skill routes.
|
||||
"""
|
||||
existing = {row.get("name") for row in manager.load_all()}
|
||||
installed = 0
|
||||
paths = sorted(_BUILTIN_ROOT.rglob("SKILL.md")) if _BUILTIN_ROOT.is_dir() else []
|
||||
for path in paths:
|
||||
try:
|
||||
skill = Skill.from_markdown(path.read_text(encoding="utf-8"))
|
||||
except Exception:
|
||||
continue
|
||||
# Tracked procedures ship as trusted application behavior. They are
|
||||
# available immediately and never enter the user's audit queue.
|
||||
skill.status = "published"
|
||||
skill.confidence = 1.0
|
||||
existing_rows = [row for row in manager.load_all() if row.get("name") == skill.name]
|
||||
if existing_rows:
|
||||
row = existing_rows[0]
|
||||
# Built-ins are immutable tracked assets. Synchronize updated
|
||||
# versions/procedures on startup while leaving usage counters in
|
||||
# their sidecar untouched. Older startup code could also stamp the
|
||||
# first admin onto one; normalize that migration at the same time.
|
||||
if row.get("source") == "builtin":
|
||||
skill.owner = ""
|
||||
skill.source = "builtin"
|
||||
desired = skill.to_dict()
|
||||
if any(row.get(field) != desired.get(field) for field in _SYNC_FIELDS):
|
||||
manager._write_skill(skill)
|
||||
continue
|
||||
skill.owner = ""
|
||||
skill.source = "builtin"
|
||||
manager._write_skill(skill)
|
||||
existing.add(skill.name)
|
||||
installed += 1
|
||||
return installed
|
||||
@@ -90,6 +90,29 @@ EXTRACT_SYSTEM_PROMPT = (
|
||||
# How many recent messages to include for extraction
|
||||
CONTEXT_WINDOW = 6
|
||||
|
||||
PERSONA_MEMORY_SYSTEM_PROMPT = (
|
||||
"You maintain concise continuity notes for one active chat persona. "
|
||||
"Update the existing notes using only durable details established in the transcript. "
|
||||
"Keep details that help the same persona stay consistent in future conversations: "
|
||||
"relationship context, names, preferences, recurring story details, boundaries, and unresolved threads. "
|
||||
"Do not store generic chat events, temporary wording, assistant reasoning, or one-off requests. "
|
||||
"Never invent details. Return only the updated notes as short bullet points, max 12 bullets. "
|
||||
"If there is nothing worth keeping, return the existing notes unchanged or an empty string."
|
||||
)
|
||||
|
||||
HEALTH_PERSONA_MEMORY_SYSTEM_PROMPT = (
|
||||
"You maintain a cautious health-record brief for a medical reasoning persona. "
|
||||
"Update the existing brief using only medically durable information from the transcript. "
|
||||
"Keep facts that may matter in future health conversations: confirmed diagnoses, chronic conditions, "
|
||||
"surgeries/procedures, allergies, regular medications/supplements, important test results, clinicians/hospitals, "
|
||||
"ongoing symptoms or care plans, and the user's preferences for medical explanations. "
|
||||
"Use uncertainty labels when needed: 'reported', 'possible', 'asked about', 'unclear'. "
|
||||
"Do not turn guesses into diagnoses. Do not store casual one-off symptoms unless they are recurring, severe, "
|
||||
"or tied to an ongoing episode. Never invent facts. Return only the updated brief with these headings when useful: "
|
||||
"Medical profile, Medications/allergies, Episodes/open questions, Preferences. Max 16 concise bullets total. "
|
||||
"If nothing medically durable changed, return the existing brief unchanged or an empty string."
|
||||
)
|
||||
|
||||
AUDIT_SYSTEM_PROMPT = (
|
||||
"You are a memory database curator. Be CONSERVATIVE: remove only TRUE "
|
||||
"duplicates and clearly useless entries. Every distinct fact must survive. "
|
||||
@@ -112,6 +135,20 @@ AUDIT_SYSTEM_PROMPT = (
|
||||
)
|
||||
|
||||
AUDIT_INTERVAL = 5 # audit every N new memories added
|
||||
AUTO_PINNED_IDENTITY_LIMIT = 5
|
||||
|
||||
|
||||
def _is_owner_memory(entry, owner):
|
||||
if owner:
|
||||
return entry.get("owner") == owner or entry.get("owner") is None
|
||||
return True
|
||||
|
||||
|
||||
def _is_auto_pinned_identity(entry):
|
||||
return (
|
||||
bool(entry.get("pinned"))
|
||||
and (entry.get("category") or "").lower() in {"identity", "contact"}
|
||||
)
|
||||
_extractions_since_audit = 0
|
||||
|
||||
|
||||
@@ -397,6 +434,10 @@ async def extract_and_store(
|
||||
logger.error("Skipping auto memory extraction, store unreadable: %s", e)
|
||||
return
|
||||
added = 0
|
||||
auto_pinned_identity_count = sum(
|
||||
1 for entry in existing
|
||||
if _is_owner_memory(entry, _owner) and _is_auto_pinned_identity(entry)
|
||||
)
|
||||
|
||||
for fact in facts:
|
||||
if isinstance(fact, str):
|
||||
@@ -404,7 +445,7 @@ async def extract_and_store(
|
||||
category = "fact"
|
||||
elif isinstance(fact, dict):
|
||||
fact_text = fact.get("text", "").strip()
|
||||
category = fact.get("category", "fact")
|
||||
category = str(fact.get("category", "fact") or "fact")
|
||||
else:
|
||||
continue
|
||||
|
||||
@@ -446,9 +487,15 @@ async def extract_and_store(
|
||||
continue
|
||||
|
||||
entry = memory_manager.add_entry(fact_text, source="auto", category=category, owner=_owner)
|
||||
# Auto-pin identity facts (name, job, location) — core context
|
||||
if category == "identity":
|
||||
# Auto-pin only the first few identity/contact facts. Extra identity
|
||||
# memories are still saved, but they must be recalled by relevance
|
||||
# instead of riding along in every prompt forever.
|
||||
if (
|
||||
category.lower() in {"identity", "contact"}
|
||||
and auto_pinned_identity_count < AUTO_PINNED_IDENTITY_LIMIT
|
||||
):
|
||||
entry["pinned"] = True
|
||||
auto_pinned_identity_count += 1
|
||||
if hasattr(session, "session_id"):
|
||||
entry["session_id"] = session.session_id
|
||||
elif hasattr(session, "name"):
|
||||
@@ -492,6 +539,88 @@ async def extract_and_store(
|
||||
logger.error(f"Memory extraction failed: {e}")
|
||||
|
||||
|
||||
async def update_persona_memory(
|
||||
session,
|
||||
preset_manager,
|
||||
character_name: str,
|
||||
endpoint_url: str,
|
||||
model: str,
|
||||
headers: Optional[dict] = None,
|
||||
schema: str = "general",
|
||||
):
|
||||
"""Update the active persona's continuity notes from recent conversation.
|
||||
|
||||
Persona memory is stored with the persona/template data, not in the global
|
||||
memory DB, so deleting a saved persona also deletes its notes.
|
||||
"""
|
||||
character_name = (character_name or "").strip()
|
||||
if not character_name or not endpoint_url or not model or preset_manager is None:
|
||||
return
|
||||
|
||||
try:
|
||||
from src.llm_core import llm_call_async
|
||||
from src.text_helpers import strip_think
|
||||
|
||||
custom = {}
|
||||
try:
|
||||
custom = preset_manager.presets.get("custom", {}) if isinstance(preset_manager.presets, dict) else {}
|
||||
except Exception:
|
||||
custom = {}
|
||||
existing_memory = ""
|
||||
if isinstance(custom, dict) and custom.get("character_name") == character_name:
|
||||
existing_memory = custom.get("persona_memory", "") or ""
|
||||
|
||||
messages = session.get_context_messages()
|
||||
recent = messages[-CONTEXT_WINDOW:] if len(messages) > CONTEXT_WINDOW else messages
|
||||
if len(recent) < 2:
|
||||
return
|
||||
|
||||
lines = []
|
||||
for msg in recent:
|
||||
role = msg.get("role")
|
||||
content = msg.get("content", "")
|
||||
if isinstance(content, list):
|
||||
content = " ".join(
|
||||
b.get("text", "") for b in content
|
||||
if isinstance(b, dict) and b.get("type") == "text"
|
||||
)
|
||||
content = str(content or "").strip()
|
||||
if content:
|
||||
lines.append(f"{role}: {content}")
|
||||
if not lines:
|
||||
return
|
||||
|
||||
system_prompt = HEALTH_PERSONA_MEMORY_SYSTEM_PROMPT if schema == "health" else PERSONA_MEMORY_SYSTEM_PROMPT
|
||||
raw = await llm_call_async(
|
||||
endpoint_url,
|
||||
model,
|
||||
[
|
||||
{"role": "system", "content": system_prompt},
|
||||
{"role": "user", "content": (
|
||||
f"Persona name: {character_name}\n\n"
|
||||
f"Existing continuity notes:\n{existing_memory or '(none)'}\n\n"
|
||||
"Recent transcript:\n"
|
||||
+ "\n\n".join(lines)
|
||||
+ "\n\nReturn only the updated continuity notes."
|
||||
)},
|
||||
],
|
||||
temperature=0.1,
|
||||
max_tokens=1200,
|
||||
headers=headers,
|
||||
)
|
||||
|
||||
updated = strip_think(str(raw or ""), prose=True, prompt_echo=True).strip()
|
||||
updated = re.sub(r"^```(?:text|markdown)?\s*|\s*```$", "", updated, flags=re.I | re.S).strip()
|
||||
if len(updated) > 6000:
|
||||
updated = updated[:6000].rstrip()
|
||||
if updated == existing_memory:
|
||||
return
|
||||
if preset_manager.update_persona_memory(character_name, updated):
|
||||
logger.info("Updated persona memory for %s", character_name)
|
||||
except Exception as e:
|
||||
logger.warning("Persona memory update failed: %s", e)
|
||||
|
||||
|
||||
async def audit_memories(
|
||||
memory_manager,
|
||||
memory_vector,
|
||||
|
||||
@@ -28,6 +28,10 @@ SKILL_EXTRACT_PROMPT = (
|
||||
"(personal errands, a specific person/place/date, casual conversation).\n"
|
||||
"- A pure question/answer or explanation with no transferable method.\n"
|
||||
"- The agent failed, gave up, or the approach is not worth repeating.\n\n"
|
||||
"- Routine use of an existing tool, or a generic checklist with no new discovery.\n"
|
||||
"Prefer a specific successful workaround, an unexpected pitfall, or a verified "
|
||||
"sequence that would save rediscovery. Preserve exact useful commands and "
|
||||
"verification steps, but replace private identifiers and credentials with placeholders.\n\n"
|
||||
"When (and only when) a genuine reusable procedure exists, return a JSON "
|
||||
"object with:\n"
|
||||
'- "title": short name (under 10 words)\n'
|
||||
@@ -259,19 +263,9 @@ async def maybe_extract_skill(
|
||||
logger.debug("[skill-extract] '%s' already exists — dropped as duplicate", title)
|
||||
return None
|
||||
|
||||
# Auto-publish gate: if the user has `auto_approve_skills` on, the
|
||||
# newly-extracted skill is created `published` immediately rather
|
||||
# than waiting for the next audit batch. The audit still runs later
|
||||
# and can demote it back to `draft` (or delete) on failure. Default
|
||||
# ON matches the UI label "Auto-approve skills".
|
||||
# Automatic approval happens only after the audit has passed. A new
|
||||
# extraction begins as a draft so it cannot enter chat context early.
|
||||
_initial_status = "draft"
|
||||
try:
|
||||
from routes.prefs_routes import _load_for_user as _load_prefs
|
||||
_prefs = _load_prefs(owner) or {}
|
||||
if _prefs.get("auto_approve_skills", True):
|
||||
_initial_status = "published"
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
entry = skills_manager.add_skill(
|
||||
title=title,
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
"""Bounded automatic review queue for user-owned procedural memory."""
|
||||
import time
|
||||
|
||||
|
||||
def automatic_audit_candidates(skills, limit=8, now=None):
|
||||
"""Retry transient checks daily and failed repairs weekly, oldest first."""
|
||||
now = time.time() if now is None else now
|
||||
pending = []
|
||||
for skill in skills:
|
||||
if not skill.get("name") or skill.get("source") == "builtin" or skill.get("status") == "binned":
|
||||
continue
|
||||
verdict = skill.get("audit_verdict")
|
||||
if verdict in {"pass", "skipped"}:
|
||||
continue
|
||||
checked = float(skill.get("audited_at") or 0)
|
||||
delay = 7 * 86400 if verdict in {"fail", "needs_work"} else 86400
|
||||
if not verdict or now - checked >= delay:
|
||||
pending.append(skill)
|
||||
pending.sort(key=lambda skill: float(skill.get("audited_at") or 0))
|
||||
return pending[:max(1, limit)]
|
||||
+157
-44
@@ -54,6 +54,25 @@ def _to_float(x, default: float = 0.0) -> float:
|
||||
return default
|
||||
|
||||
|
||||
def _approval_policy(owner: Optional[str]) -> tuple[bool, float]:
|
||||
"""Read the user's automatic skill-approval gate without breaking retrieval."""
|
||||
try:
|
||||
from routes.prefs_routes import _load_for_user
|
||||
prefs = _load_for_user(owner) or {}
|
||||
except Exception:
|
||||
prefs = {}
|
||||
try:
|
||||
from src.settings import get_setting
|
||||
default_minimum = float(get_setting("skill_autosave_min_confidence", 0.85))
|
||||
except Exception:
|
||||
default_minimum = 0.85
|
||||
try:
|
||||
minimum = float(prefs.get("skill_min_confidence", default_minimum))
|
||||
except (TypeError, ValueError):
|
||||
minimum = default_minimum
|
||||
return bool(prefs.get("auto_approve_skills", True)), max(0.0, min(1.0, minimum))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# SkillsManager
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -120,7 +139,11 @@ class SkillsManager:
|
||||
|
||||
def set_audit(self, name: str, verdict: str, by_teacher: bool = False,
|
||||
worker_model: str = "", teacher_model: str = "",
|
||||
owner: Optional[str] = None) -> None:
|
||||
owner: Optional[str] = None, saved_turns: Optional[int] = None,
|
||||
saved_tool_calls: Optional[int] = None,
|
||||
baseline_verdict: Optional[str] = None,
|
||||
usefulness: Optional[float] = None,
|
||||
audit_summary: Optional[str] = None) -> None:
|
||||
"""Record the last test/audit result for a skill in the usage sidecar
|
||||
(so it surfaces in load() without touching SKILL.md). Drives the
|
||||
'verified' check + teacher mark on the card."""
|
||||
@@ -129,11 +152,34 @@ class SkillsManager:
|
||||
key = self._usage_key(name, owner)
|
||||
e = usage.setdefault(key, {"uses": 0, "last_used": None})
|
||||
e["audit_verdict"] = verdict
|
||||
# Replace, rather than retain, the explanation from a previous run.
|
||||
e["audit_summary"] = str(audit_summary or "")[:2000]
|
||||
# Version 2 fixes audit-arm isolation and separates functional success
|
||||
# from baseline utility. Legacy inconclusive results are not evidence
|
||||
# under that protocol and should be eligible for a clean re-audit.
|
||||
e["audit_version"] = 2
|
||||
e["audit_by_teacher"] = bool(by_teacher)
|
||||
if worker_model:
|
||||
e["audit_worker_model"] = worker_model
|
||||
if teacher_model:
|
||||
e["audit_teacher_model"] = teacher_model
|
||||
if saved_turns is not None:
|
||||
try:
|
||||
e["saved_turns"] = int(saved_turns)
|
||||
except (TypeError, ValueError):
|
||||
e.pop("saved_turns", None)
|
||||
if saved_tool_calls is not None:
|
||||
try:
|
||||
e["saved_tool_calls"] = int(saved_tool_calls)
|
||||
except (TypeError, ValueError):
|
||||
e.pop("saved_tool_calls", None)
|
||||
if baseline_verdict is not None:
|
||||
e["baseline_verdict"] = str(baseline_verdict or "unknown")
|
||||
if usefulness is not None:
|
||||
try:
|
||||
e["usefulness"] = float(usefulness)
|
||||
except (TypeError, ValueError):
|
||||
e.pop("usefulness", None)
|
||||
e["audited_at"] = _t.time()
|
||||
self._save_usage(usage)
|
||||
|
||||
@@ -197,6 +243,8 @@ class SkillsManager:
|
||||
sk = self._read_skill(path)
|
||||
if not sk:
|
||||
continue
|
||||
if sk.source == "builtin":
|
||||
continue
|
||||
owner = (sk.owner or "").strip()
|
||||
if owner == primary_owner:
|
||||
continue
|
||||
@@ -227,11 +275,24 @@ class SkillsManager:
|
||||
u = self._usage_entry(usage, sk.name, sk.owner)
|
||||
d["uses"] = int(u.get("uses", 0))
|
||||
d["last_used"] = u.get("last_used")
|
||||
d["audit_verdict"] = u.get("audit_verdict")
|
||||
audit_verdict = u.get("audit_verdict")
|
||||
try:
|
||||
audit_version = int(u.get("audit_version") or 0)
|
||||
except (TypeError, ValueError):
|
||||
audit_version = 0
|
||||
if audit_verdict == "inconclusive" and audit_version < 2:
|
||||
audit_verdict = None
|
||||
d["audit_verdict"] = audit_verdict
|
||||
d["audit_summary"] = u.get("audit_summary", "") if audit_verdict else ""
|
||||
d["audit_version"] = audit_version
|
||||
d["audit_by_teacher"] = bool(u.get("audit_by_teacher"))
|
||||
d["audit_worker_model"] = u.get("audit_worker_model")
|
||||
d["audit_teacher_model"] = u.get("audit_teacher_model")
|
||||
d["audited_at"] = u.get("audited_at")
|
||||
d["audited_at"] = u.get("audited_at") if audit_verdict else None
|
||||
d["saved_turns"] = u.get("saved_turns")
|
||||
d["saved_tool_calls"] = u.get("saved_tool_calls")
|
||||
d["baseline_verdict"] = u.get("baseline_verdict")
|
||||
d["usefulness"] = u.get("usefulness")
|
||||
d["necessity"] = u.get("necessity")
|
||||
out.append(d)
|
||||
seen_names.add(sk.name)
|
||||
@@ -284,7 +345,11 @@ class SkillsManager:
|
||||
# leaked legacy / un-stamped skills to every authenticated user.
|
||||
# Hide them now; the owner needs to be backfilled on disk if those
|
||||
# skills should be visible to a specific user.
|
||||
return [s for s in entries if s.get("owner") == owner]
|
||||
return [
|
||||
s for s in entries
|
||||
if s.get("owner") == owner
|
||||
or (s.get("source") == "builtin" and not s.get("owner"))
|
||||
]
|
||||
|
||||
# ----------------------------------------------------------------------
|
||||
# CRUD — disk-backed
|
||||
@@ -546,7 +611,15 @@ class SkillsManager:
|
||||
sk = self._read_skill(path)
|
||||
if not sk or sk.name != name:
|
||||
continue
|
||||
if (sk.owner or "") != (owner or ""):
|
||||
# Built-in skills are shared, ownerless procedures. ``load``
|
||||
# exposes them to every owner, so direct progressive-disclosure
|
||||
# reads must apply the same visibility rule as the index/list
|
||||
# path. Previously a built-in appeared in `list` but `view`
|
||||
# returned not-found for authenticated users.
|
||||
if not (
|
||||
(sk.owner or "") == (owner or "")
|
||||
or (sk.source == "builtin" and not (sk.owner or ""))
|
||||
):
|
||||
continue
|
||||
try:
|
||||
with open(path, encoding="utf-8") as f:
|
||||
@@ -562,7 +635,10 @@ class SkillsManager:
|
||||
sk = self._read_skill(path)
|
||||
if not sk or sk.name != name:
|
||||
continue
|
||||
if (sk.owner or "") != (owner or ""):
|
||||
if not (
|
||||
(sk.owner or "") == (owner or "")
|
||||
or (sk.source == "builtin" and not (sk.owner or ""))
|
||||
):
|
||||
continue
|
||||
base = os.path.realpath(os.path.dirname(path))
|
||||
target = os.path.realpath(os.path.join(base, ref_path))
|
||||
@@ -591,18 +667,12 @@ class SkillsManager:
|
||||
"""Return the `[{name, description, category, status}]` list the
|
||||
agent sees in its system prompt.
|
||||
|
||||
Includes:
|
||||
- All published skills.
|
||||
- Drafts written by the teacher-escalation loop
|
||||
(`source == "teacher-escalation"`). The whole point of
|
||||
the teacher loop is for the student to find the new
|
||||
procedure on the very next turn — waiting for a manual
|
||||
publish click defeats the loop.
|
||||
|
||||
Excludes user-created drafts (status=draft, source != teacher-
|
||||
escalation) — those are work-in-progress and pollute the
|
||||
prompt with half-finished procedures.
|
||||
Includes built-ins plus user skills that have passed their audit and
|
||||
meet the owner's current automatic-approval threshold. A persistent
|
||||
``published`` flag is not sufficient: a changed threshold or a legacy
|
||||
record must not make an unaudited skill eligible for prompt injection.
|
||||
"""
|
||||
auto_approve, min_confidence = _approval_policy(owner)
|
||||
out = []
|
||||
for s in self.load(owner=owner):
|
||||
status = s.get("status")
|
||||
@@ -613,6 +683,19 @@ class SkillsManager:
|
||||
pass # let it through
|
||||
else:
|
||||
continue
|
||||
# A stale published record must not remain injectable after an
|
||||
# audit has recorded a failure. Inconclusive is not a failure.
|
||||
audit_verdict = str(s.get("audit_verdict") or "").lower()
|
||||
if audit_verdict in {"needs_work", "fail"}:
|
||||
continue
|
||||
if s.get("source") != "builtin" and auto_approve:
|
||||
if status != "published" or audit_verdict != "pass":
|
||||
continue
|
||||
if _to_float(s.get("confidence"), 0.0) < min_confidence:
|
||||
continue
|
||||
necessity = s.get("necessity") or {}
|
||||
if isinstance(necessity, dict) and necessity.get("necessary") is False:
|
||||
continue
|
||||
# Platform gating
|
||||
if platform and s.get("platforms") and platform not in s["platforms"]:
|
||||
continue
|
||||
@@ -649,6 +732,8 @@ class SkillsManager:
|
||||
threshold: float = 0.3,
|
||||
max_items: int = 5,
|
||||
min_confidence: float = 0.0,
|
||||
available_toolsets: Optional[Iterable[str]] = None,
|
||||
platform: Optional[str] = None,
|
||||
) -> List[Dict]:
|
||||
if skills is None:
|
||||
skills = self.load_all()
|
||||
@@ -660,37 +745,62 @@ class SkillsManager:
|
||||
# without a manual publish click. The UI flags teacher-written
|
||||
# entries with a 🎓 badge so users can demote / delete bad
|
||||
# ones when they spot them.
|
||||
skills = [s for s in skills if s.get("status") in ("published", "draft")]
|
||||
# Confidence gate (used by prompt-injection, NOT by search): a DRAFT
|
||||
# skill must clear the bar to be injected. Published skills are already
|
||||
# vetted, so they always qualify. Missing confidence = treat as 1.0
|
||||
# (legacy skills shouldn't silently vanish). 0 disables the gate.
|
||||
skills = [
|
||||
s for s in skills
|
||||
if s.get("status") in ("published", "draft")
|
||||
and str(s.get("audit_verdict") or "").lower()
|
||||
not in {"needs_work", "fail", "skipped"}
|
||||
]
|
||||
available = set(available_toolsets) if available_toolsets is not None else None
|
||||
if available is not None:
|
||||
skills = [
|
||||
skill for skill in skills
|
||||
if all(tool in available for tool in (skill.get("requires_toolsets") or []))
|
||||
and not any(tool in available for tool in (skill.get("fallback_for_toolsets") or []))
|
||||
]
|
||||
if platform:
|
||||
skills = [
|
||||
skill for skill in skills
|
||||
if not skill.get("platforms") or platform in skill.get("platforms", [])
|
||||
]
|
||||
# Prompt injection is fail-closed for user skills. Built-ins are
|
||||
# shipped procedures; every other skill needs a passing audit and a
|
||||
# confidence score at the user's current threshold.
|
||||
if min_confidence > 0:
|
||||
def _passes(s):
|
||||
if s.get("status") == "published":
|
||||
if s.get("source") == "builtin":
|
||||
return True
|
||||
# Teacher-escalation drafts are auto-written from a (possibly
|
||||
# untrusted) trace and injected as authoritative guidance, so they
|
||||
# must EARN injection with an explicit, parseable confidence that
|
||||
# clears the bar — fail closed on a missing/garbage value instead
|
||||
# of treating it as 1.0. Hand-authored legacy drafts keep the
|
||||
# lenient "unset → keep" behavior so they don't silently vanish.
|
||||
if s.get("source") == "teacher-escalation":
|
||||
c = s.get("confidence")
|
||||
if c is None:
|
||||
return False
|
||||
return _to_float(c, 0.0) >= min_confidence # unparseable → fail closed
|
||||
c = s.get("confidence")
|
||||
if c is None:
|
||||
return True # unset → don't filter (legacy)
|
||||
return _to_float(c, 1.0) >= min_confidence # unparseable → pass
|
||||
return (
|
||||
s.get("status") == "published"
|
||||
and str(s.get("audit_verdict") or "").lower() == "pass"
|
||||
and _to_float(s.get("confidence"), 0.0) >= min_confidence
|
||||
)
|
||||
skills = [s for s in skills if _passes(s)]
|
||||
if not skills:
|
||||
return []
|
||||
|
||||
query_tokens = _tokenize(query)
|
||||
semantic_scores: Dict[int, float] = {}
|
||||
semantic_enabled = str(
|
||||
os.environ.get("ODYSSEUS_SKILL_SEMANTIC_RETRIEVAL", "1")
|
||||
).strip().lower() not in {"0", "false", "no", "off"}
|
||||
if semantic_enabled:
|
||||
try:
|
||||
from src.skill_index import semantic_skill_scores
|
||||
|
||||
semantic_scores = semantic_skill_scores(query, skills)
|
||||
except Exception as exc:
|
||||
logger.debug("Semantic skill retrieval unavailable: %s", exc)
|
||||
try:
|
||||
semantic_threshold = float(
|
||||
os.environ.get("ODYSSEUS_SKILL_SEMANTIC_THRESHOLD", "0.4")
|
||||
)
|
||||
except (TypeError, ValueError):
|
||||
semantic_threshold = 0.4
|
||||
semantic_threshold = max(-1.0, min(1.0, semantic_threshold))
|
||||
|
||||
scored = []
|
||||
for sk in skills:
|
||||
for position, sk in enumerate(skills):
|
||||
text = " ".join([
|
||||
sk.get("name", ""),
|
||||
sk.get("description", ""),
|
||||
@@ -698,19 +808,22 @@ class SkillsManager:
|
||||
" ".join(sk.get("tags", []) or []),
|
||||
" ".join(sk.get("procedure", []) or []),
|
||||
])
|
||||
score = _jaccard(query_tokens, _tokenize(text))
|
||||
lexical_score = _jaccard(query_tokens, _tokenize(text))
|
||||
for tag in sk.get("tags", []) or []:
|
||||
# Match tags as whole tokens, not substrings: `tag in query`
|
||||
# boosted e.g. a "ai" tag for any query containing "email".
|
||||
tag_tokens = _tokenize(tag)
|
||||
if tag_tokens and tag_tokens <= query_tokens:
|
||||
score = max(score, 0.3) * 1.3
|
||||
lexical_score = max(lexical_score, 0.3) * 1.3
|
||||
if query.lower() in (sk.get("description") or "").lower():
|
||||
score = max(score, 0.6)
|
||||
lexical_score = max(lexical_score, 0.6)
|
||||
semantic_score = semantic_scores.get(position, -1.0)
|
||||
if lexical_score < threshold and semantic_score < semantic_threshold:
|
||||
continue
|
||||
score = max(lexical_score, semantic_score)
|
||||
score *= 1.0 + _to_float(sk.get("confidence"), 0.5) * 0.1
|
||||
if sk.get("uses", 0) > 0:
|
||||
score *= 1.05
|
||||
if score >= threshold:
|
||||
scored.append((score, sk))
|
||||
scored.append((score, sk))
|
||||
scored.sort(key=lambda x: x[0], reverse=True)
|
||||
return [sk for _, sk in scored[:max_items]]
|
||||
|
||||
+75
-18
@@ -65,6 +65,49 @@ try:
|
||||
except ImportError:
|
||||
pdf_extract_text = None # type: ignore
|
||||
|
||||
try:
|
||||
from pypdf import PdfReader
|
||||
except ImportError:
|
||||
PdfReader = None # type: ignore
|
||||
|
||||
|
||||
def _extract_pdf_text(pdf_bytes: bytes, url: str = "") -> str:
|
||||
"""Extract PDF text with available permissive dependencies."""
|
||||
# Prefer pypdf's layout mode. Plain text extraction and pdfminer often
|
||||
# collapse table columns into an ambiguous number stream, which makes a
|
||||
# correct source passage easy for the model to misread.
|
||||
if PdfReader is not None:
|
||||
try:
|
||||
reader = PdfReader(io.BytesIO(pdf_bytes))
|
||||
pages: List[str] = []
|
||||
for idx, page in enumerate(reader.pages):
|
||||
try:
|
||||
try:
|
||||
page_text = page.extract_text(extraction_mode="layout") or ""
|
||||
except TypeError:
|
||||
page_text = page.extract_text() or ""
|
||||
except Exception as e:
|
||||
logger.warning(f"pypdf extraction failed for {url} page {idx + 1}: {e}")
|
||||
page_text = ""
|
||||
if page_text.strip():
|
||||
pages.append(f"[Page {idx + 1}]\n{page_text.strip()}")
|
||||
if pages:
|
||||
return "\n\n".join(pages)
|
||||
except Exception as e:
|
||||
logger.warning(f"pypdf extraction failed for {url}: {e}")
|
||||
|
||||
if pdf_extract_text is not None:
|
||||
try:
|
||||
text = pdf_extract_text(io.BytesIO(pdf_bytes)) or ""
|
||||
if text.strip():
|
||||
return text
|
||||
except Exception as e:
|
||||
logger.warning(f"pdfminer extraction failed for {url}: {e}")
|
||||
|
||||
if PdfReader is None and pdf_extract_text is None:
|
||||
logger.error("No PDF text extractor installed; install pdfminer.six or pypdf.")
|
||||
return ""
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------
|
||||
# HTML extraction helpers
|
||||
@@ -216,9 +259,6 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
|
||||
"User-Agent": WEB_FETCH_USER_AGENT,
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "en-US,en;q=0.5",
|
||||
# identity so the streamed size cap in _get_public_url stays honest
|
||||
# (a compressed body can decode to far more than Content-Length).
|
||||
"Accept-Encoding": "identity",
|
||||
"Connection": "keep-alive",
|
||||
}
|
||||
response = _get_public_url(url, headers=headers, timeout=timeout,
|
||||
@@ -252,26 +292,43 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
|
||||
# PDF handling
|
||||
content_type = response.headers.get("Content-Type", "").lower()
|
||||
if "application/pdf" in content_type or url.lower().endswith(".pdf"):
|
||||
if (
|
||||
_size_fields["truncated"]
|
||||
and effective_cap < WEB_FETCH_HARD_MAX_BYTES
|
||||
and (
|
||||
_size_fields["total_bytes"] is None
|
||||
or _size_fields["total_bytes"] <= WEB_FETCH_HARD_MAX_BYTES
|
||||
)
|
||||
):
|
||||
try:
|
||||
response = _get_public_url(
|
||||
url,
|
||||
headers=headers,
|
||||
timeout=timeout,
|
||||
max_bytes=WEB_FETCH_HARD_MAX_BYTES,
|
||||
)
|
||||
_size_fields = {
|
||||
"truncated": getattr(response, "truncated", False),
|
||||
"fetched_bytes": len(response.content),
|
||||
"total_bytes": getattr(response, "declared_bytes", None),
|
||||
}
|
||||
effective_cap = WEB_FETCH_HARD_MAX_BYTES
|
||||
except BodyTooLargeError as e:
|
||||
error_logger.warning(f"Refused oversized PDF body for {url}: {e}")
|
||||
return _empty_result(url, f"TooLarge: {e}")
|
||||
except Exception as e:
|
||||
logger.warning(f"Full-budget PDF retry failed for {url}: {e}")
|
||||
if _size_fields["truncated"]:
|
||||
# A PDF cut mid-stream is not parseable; unlike text there is no
|
||||
# useful partial result, so report the budget problem instead.
|
||||
_declared = _size_fields["total_bytes"]
|
||||
return _empty_result(
|
||||
url,
|
||||
f"TooLarge: PDF exceeds the {effective_cap:,}-byte fetch budget"
|
||||
+ (f" (size {_declared:,} bytes)" if _declared else "")
|
||||
+ "; retry with a larger budget if it fits under the hard cap",
|
||||
error = (
|
||||
f"TooLarge: PDF decoded body exceeded the {effective_cap:,}-byte fetch budget"
|
||||
+ (f" (declared compressed size {_declared:,} bytes)" if _declared else "")
|
||||
+ "; retry with a larger budget if it fits under the hard cap"
|
||||
)
|
||||
if pdf_extract_text is None:
|
||||
logger.error("pdfminer.six is not installed; cannot extract PDF text.")
|
||||
pdf_text = ""
|
||||
else:
|
||||
try:
|
||||
pdf_bytes = io.BytesIO(response.content)
|
||||
pdf_text = pdf_extract_text(pdf_bytes)
|
||||
except Exception as e:
|
||||
logger.warning(f"PDF extraction failed for {url}: {e}")
|
||||
pdf_text = ""
|
||||
return {**_empty_result(url, error), **_size_fields}
|
||||
pdf_text = _extract_pdf_text(response.content, url)
|
||||
result = {
|
||||
"url": url,
|
||||
"title": os.path.basename(url),
|
||||
|
||||
+538
-13
@@ -2,11 +2,15 @@
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import xml.etree.ElementTree as ET
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from datetime import datetime, timedelta
|
||||
from typing import Dict, Any, Optional, List, Set
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import httpx
|
||||
|
||||
from .analytics import (
|
||||
NetworkError,
|
||||
ParseError,
|
||||
@@ -97,6 +101,8 @@ def _call_provider(provider_name: str, query: str, count: int, time_filter: str
|
||||
"""Call a search provider by name. Returns list of results or empty list."""
|
||||
if provider_name == "searxng":
|
||||
return searxng_search_api(query, count, time_filter=time_filter)
|
||||
elif provider_name == "searxng_yep":
|
||||
return searxng_search_api(query, count, time_filter=time_filter, engines="yep")
|
||||
elif provider_name == "brave":
|
||||
return brave_search(query, count, time_filter)
|
||||
elif provider_name == "duckduckgo":
|
||||
@@ -127,7 +133,484 @@ def _build_provider_chain(primary: str) -> List[str]:
|
||||
for fb in fallbacks:
|
||||
if fb and fb != primary and fb not in chain and fb != "disabled":
|
||||
chain.append(fb)
|
||||
return chain
|
||||
from .providers import provider_configured
|
||||
configured = [provider for provider in chain if provider_configured(provider)]
|
||||
for provider in set(chain) - set(configured):
|
||||
logger.warning("Skipping unconfigured search provider: %s", provider)
|
||||
if primary == "searxng" and configured == ["searxng"]:
|
||||
# No usable configured fallback: try a separate engine on the same
|
||||
# private metasearch instance before reporting retrieval failure.
|
||||
configured.append("searxng_yep")
|
||||
return configured
|
||||
|
||||
|
||||
_SEARCH_QUERY_FILLER = {
|
||||
"what", "whats", "what's", "which", "when", "where", "year", "from",
|
||||
"any", "info", "information", "details", "update", "updates",
|
||||
"with", "this", "that", "search", "lookup", "look", "find", "tell",
|
||||
"about", "quick", "please", "pls", "official", "links", "source",
|
||||
"sources", "news", "headlines", "breaking", "latest", "current",
|
||||
"newest", "recent", "today", "now",
|
||||
"release", "releases", "version", "versions", "changelog", "github",
|
||||
"gitlab", "weather", "forecast", "forecasts", "tomorrow", "hourly",
|
||||
"daily", "temperature", "temperatures", "conditions", "rain", "raining",
|
||||
"chance", "precipitation",
|
||||
"january", "february", "march", "april", "may", "june", "july",
|
||||
"august", "september", "october", "november", "december",
|
||||
"the", "and", "or", "but", "are", "was", "were", "does", "did",
|
||||
"can", "could", "should", "would", "will", "has", "have", "had",
|
||||
"for", "into", "onto", "near", "over", "under",
|
||||
}
|
||||
|
||||
_SHORT_QUERY_SUBJECTS = {"ai", "ar", "eu", "uk", "us", "vr"}
|
||||
|
||||
_WEATHER_QUERY_HINTS = {
|
||||
"weather", "forecast", "forecasts", "temperature", "temperatures",
|
||||
"rain", "raining", "precipitation", "humid", "humidity", "wind",
|
||||
}
|
||||
_WEATHER_RESULT_HINTS = {
|
||||
"weather", "forecast", "temperature", "temperatures", "rain",
|
||||
"precipitation", "humidity", "wind", "accuweather", "meteoblue",
|
||||
"weather-atlas", "weather25", "weather365", "easeweather",
|
||||
}
|
||||
|
||||
|
||||
def _meaningful_query_terms(query: str) -> list[str]:
|
||||
return [
|
||||
term
|
||||
for term in re.findall(r"[a-z0-9]+", str(query or "").lower())
|
||||
if (len(term) > 2 or term in _SHORT_QUERY_SUBJECTS)
|
||||
and not term.isdigit()
|
||||
and term not in _SEARCH_QUERY_FILLER
|
||||
]
|
||||
|
||||
|
||||
def _result_has_query_overlap(query: str, result: dict) -> bool:
|
||||
terms = _meaningful_query_terms(query)
|
||||
if not terms:
|
||||
return True
|
||||
text = " ".join(
|
||||
str(result.get(key) or "").lower()
|
||||
for key in ("title", "snippet", "url")
|
||||
)
|
||||
query_tokens = set(re.findall(r"[a-z0-9]+", str(query or "").lower()))
|
||||
if query_tokens & _WEATHER_QUERY_HINTS:
|
||||
return (
|
||||
any(re.search(rf"\b{re.escape(term)}\b", text) for term in terms)
|
||||
and any(marker in text for marker in _WEATHER_RESULT_HINTS)
|
||||
)
|
||||
result_tokens = set(re.findall(r"[a-z0-9]+", text))
|
||||
|
||||
def lexical_root(word: str) -> str:
|
||||
for suffix in ("ation", "ition", "ence", "ance", "ment", "ents", "ent", "ant", "ing", "ed", "es", "s"):
|
||||
if word.endswith(suffix) and len(word) - len(suffix) >= 6:
|
||||
return word[:-len(suffix)]
|
||||
return word
|
||||
|
||||
result_roots = {lexical_root(token) for token in result_tokens}
|
||||
matched_terms = {
|
||||
term for term in terms
|
||||
if term in result_tokens or lexical_root(term) in result_roots
|
||||
}
|
||||
# A single broad token is not enough evidence for a detailed entity/event
|
||||
# query. For example, SearXNG may answer "Sweden 78 year old British woman
|
||||
# deportation Brexit ..." with generic Sweden tourism pages. Treat that as
|
||||
# an empty provider result so the configured fallback gets a chance.
|
||||
minimum_matches = 2 if len(set(terms)) >= 4 else 1
|
||||
return len(matched_terms) >= minimum_matches
|
||||
|
||||
|
||||
def _filter_low_relevance_results(query: str, results: list[dict]) -> list[dict]:
|
||||
if not results:
|
||||
return []
|
||||
relevant = [result for result in results if _result_has_query_overlap(query, result)]
|
||||
# Only reject a provider when it returned a fully off-topic page set. Mixed
|
||||
# result pages are common; ranking can handle those.
|
||||
return relevant if relevant else []
|
||||
|
||||
|
||||
_SCHOLARLY_QUERY_CUE_RE = re.compile(
|
||||
r"\b(?:paper|preprint|arxiv|proceedings|table\s+\d+|figure\s+\d+|"
|
||||
r"appendix\s+[a-z0-9]+|benchmark(?:s)?)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_SCHOLARLY_TITLE_FILLER = _SEARCH_QUERY_FILLER | {
|
||||
"paper", "preprint", "arxiv", "proceedings", "table", "figure",
|
||||
"appendix", "authors", "author", "extract", "locate", "read",
|
||||
}
|
||||
_ARXIV_IDENTIFIER_RE = re.compile(
|
||||
r"(?i)(?:\barxiv\s*:\s*|\barxiv\.org/(?:abs|pdf|html)/)?"
|
||||
r"(?P<identifier>\d{4}\.\d{4,5}(?:v\d+)?)\b"
|
||||
)
|
||||
_FORMAL_PUBLICATION_CUE_RE = re.compile(
|
||||
r"\b(?:publish(?:ed|ing|cation)?|venue|conference|journal|proceedings|doi)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def _exact_arxiv_identifier_results(query: str) -> list[dict]:
|
||||
"""Return deterministic official landing pages for explicit arXiv IDs."""
|
||||
seen: set[str] = set()
|
||||
results: list[dict] = []
|
||||
for match in _ARXIV_IDENTIFIER_RE.finditer(str(query or "")):
|
||||
identifier = match.group("identifier")
|
||||
canonical = re.sub(r"v\d+$", "", identifier, flags=re.IGNORECASE)
|
||||
if canonical in seen:
|
||||
continue
|
||||
seen.add(canonical)
|
||||
results.append({
|
||||
"title": f"arXiv:{canonical} — exact identifier match",
|
||||
"url": f"https://arxiv.org/abs/{canonical}",
|
||||
"snippet": (
|
||||
"Official arXiv landing page resolved directly from the exact "
|
||||
"identifier in the query."
|
||||
),
|
||||
"source": "arxiv",
|
||||
})
|
||||
return results
|
||||
|
||||
|
||||
def _title_before_explicit_arxiv_identifier(query: str) -> str:
|
||||
"""Extract a probable title that precedes an explicit arXiv identifier."""
|
||||
|
||||
text = re.sub(r"\s+", " ", str(query or "")).strip()
|
||||
match = _ARXIV_IDENTIFIER_RE.search(text)
|
||||
if not match or not _FORMAL_PUBLICATION_CUE_RE.search(text):
|
||||
return ""
|
||||
candidate = text[:match.start()].strip(" \t,;:-'\"")
|
||||
candidate = re.sub(
|
||||
r"\barxiv(?:\.org)?(?:\s*:\s*|\s+(?:abs|pdf|html)\s*[/ :]*)?$",
|
||||
"",
|
||||
candidate,
|
||||
flags=re.IGNORECASE,
|
||||
).strip(" \t,;:-'\"")
|
||||
candidate = re.sub(
|
||||
r"^(?:(?:please\s+)?(?:find|locate|search\s+for|look\s+up|verify|check)\s+)"
|
||||
r"(?:(?:the|this)\s+)?(?:paper\s+)?",
|
||||
"",
|
||||
candidate,
|
||||
flags=re.IGNORECASE,
|
||||
).strip(" \t,;:-'\"")
|
||||
return candidate if len(_normalized_title_terms(candidate)) >= 2 else ""
|
||||
|
||||
|
||||
def _normalized_title_terms(value: str) -> list[str]:
|
||||
return [
|
||||
token
|
||||
for token in re.findall(r"[a-z0-9]+", str(value or "").lower())
|
||||
if len(token) > 1 and token not in _SCHOLARLY_TITLE_FILLER
|
||||
]
|
||||
|
||||
|
||||
def _is_distinctive_short_scholarly_title(value: str) -> bool:
|
||||
"""Recognize compact model/report names without accepting generic phrases."""
|
||||
|
||||
terms = _normalized_title_terms(value)
|
||||
if not 1 <= len(terms) <= 2:
|
||||
return False
|
||||
text = str(value or "").strip()
|
||||
return bool(
|
||||
re.search(r"\d", text)
|
||||
or re.search(r"\b[A-Z][A-Za-z0-9]*-[A-Z][A-Za-z0-9]*\b", text)
|
||||
)
|
||||
|
||||
|
||||
def _scholarly_title_from_query(query: str) -> str:
|
||||
"""Extract a probable paper title only from clearly scholarly searches."""
|
||||
|
||||
text = re.sub(r"\s+", " ", str(query or "")).strip()
|
||||
if not text or not _SCHOLARLY_QUERY_CUE_RE.search(text):
|
||||
return ""
|
||||
|
||||
quoted = [
|
||||
candidate.strip()
|
||||
for candidate in re.findall(r'["“”]([^"“”]{4,180})["“”]', text)
|
||||
if len(_normalized_title_terms(candidate)) >= 3
|
||||
or _is_distinctive_short_scholarly_title(candidate)
|
||||
]
|
||||
if quoted:
|
||||
return max(quoted, key=lambda candidate: len(_normalized_title_terms(candidate)))
|
||||
|
||||
before_paper = re.search(
|
||||
r"(?:^|\b(?:find|locate|read|from|about)\s+)(.{4,160}?)\s+"
|
||||
r"(?:paper|preprint)\b",
|
||||
text,
|
||||
re.IGNORECASE,
|
||||
)
|
||||
if before_paper:
|
||||
candidate = before_paper.group(1).strip(" ,:;-'")
|
||||
if (
|
||||
len(_normalized_title_terms(candidate)) >= 3
|
||||
or _is_distinctive_short_scholarly_title(candidate)
|
||||
):
|
||||
return candidate
|
||||
|
||||
before_locator = re.match(
|
||||
r"(.{2,80}?)\s+(?:table|figure)\s+\d+\b",
|
||||
text,
|
||||
re.IGNORECASE,
|
||||
)
|
||||
if before_locator:
|
||||
candidate = before_locator.group(1).strip(" ,:;-'\"")
|
||||
if _is_distinctive_short_scholarly_title(candidate):
|
||||
return candidate
|
||||
return ""
|
||||
|
||||
|
||||
def _result_strongly_matches_title(title: str, result: dict) -> bool:
|
||||
wanted = set(_normalized_title_terms(title))
|
||||
found = set(_normalized_title_terms(str(result.get("title") or "")))
|
||||
if len(wanted) < 2 or not found:
|
||||
return False
|
||||
overlap = len(wanted & found) / len(wanted)
|
||||
return overlap >= (1.0 if len(wanted) == 2 else 0.8)
|
||||
|
||||
|
||||
def _arxiv_title_results(title: str, count: int = 3) -> list[dict]:
|
||||
"""Resolve a paper title through arXiv's public Atom API."""
|
||||
|
||||
try:
|
||||
response = httpx.get(
|
||||
"https://export.arxiv.org/api/query",
|
||||
params={
|
||||
"search_query": f'ti:"{title}"',
|
||||
"start": 0,
|
||||
"max_results": max(1, min(int(count), 5)),
|
||||
},
|
||||
headers={"User-Agent": "Odysseus/0.20 scholarly-title-resolver"},
|
||||
timeout=12.0,
|
||||
follow_redirects=True,
|
||||
)
|
||||
response.raise_for_status()
|
||||
root = ET.fromstring(response.text)
|
||||
except Exception as exc:
|
||||
logger.info("arXiv title lookup failed for %r: %s", title, exc)
|
||||
return []
|
||||
|
||||
namespace = {"atom": "http://www.w3.org/2005/Atom"}
|
||||
matches: list[dict] = []
|
||||
for entry in root.findall("atom:entry", namespace):
|
||||
result_title = " ".join(
|
||||
(entry.findtext("atom:title", default="", namespaces=namespace) or "").split()
|
||||
)
|
||||
if not _result_strongly_matches_title(title, {"title": result_title}):
|
||||
continue
|
||||
entry_id = (entry.findtext("atom:id", default="", namespaces=namespace) or "").strip()
|
||||
arxiv_id = entry_id.rstrip("/").rsplit("/", 1)[-1]
|
||||
if not arxiv_id:
|
||||
continue
|
||||
summary = " ".join(
|
||||
(entry.findtext("atom:summary", default="", namespaces=namespace) or "").split()
|
||||
)
|
||||
matches.append({
|
||||
"title": result_title,
|
||||
"url": f"https://arxiv.org/abs/{arxiv_id}",
|
||||
"snippet": summary,
|
||||
"source": "arxiv",
|
||||
})
|
||||
return matches
|
||||
|
||||
|
||||
def _openalex_title_results(title: str, count: int = 3) -> list[dict]:
|
||||
"""Resolve an exact scholarly title through OpenAlex metadata."""
|
||||
|
||||
try:
|
||||
# OpenAlex treats a literal question mark as query syntax and returns
|
||||
# HTTP 400 for otherwise valid titles such as "How Far ... GPT-4V?".
|
||||
search_title = re.sub(r"[?]+", " ", str(title or "")).strip()
|
||||
response = httpx.get(
|
||||
"https://api.openalex.org/works",
|
||||
params={
|
||||
"search": search_title,
|
||||
"per-page": max(1, min(int(count), 5)),
|
||||
"select": (
|
||||
"display_name,doi,primary_location,publication_year,type"
|
||||
),
|
||||
},
|
||||
headers={"User-Agent": "Odysseus/0.20 scholarly-title-resolver"},
|
||||
timeout=12.0,
|
||||
follow_redirects=True,
|
||||
)
|
||||
response.raise_for_status()
|
||||
payload = response.json()
|
||||
except Exception as exc:
|
||||
logger.info("OpenAlex title lookup failed for %r: %s", title, exc)
|
||||
return []
|
||||
|
||||
matches: list[dict] = []
|
||||
for item in payload.get("results", []):
|
||||
result_title = str(item.get("display_name") or "").strip()
|
||||
if not _result_strongly_matches_title(title, {"title": result_title}):
|
||||
continue
|
||||
location = item.get("primary_location") or {}
|
||||
url = str(location.get("landing_page_url") or item.get("doi") or "").strip()
|
||||
if url.startswith("http://arxiv.org/"):
|
||||
url = "https://" + url[len("http://"):]
|
||||
if not url:
|
||||
continue
|
||||
snippet = "Exact scholarly-title match from OpenAlex metadata."
|
||||
venue = str(location.get("raw_source_name") or "").strip()
|
||||
year = item.get("publication_year")
|
||||
publication_type = str(item.get("type") or "").strip()
|
||||
version = str(location.get("version") or "").strip()
|
||||
formal_parts: list[str] = []
|
||||
if venue:
|
||||
formal_parts.append(f"{venue}, {year}" if year else venue)
|
||||
elif year:
|
||||
formal_parts.append(str(year))
|
||||
if publication_type:
|
||||
formal_parts.append(f"type: {publication_type}")
|
||||
if version:
|
||||
formal_parts.append(f"version: {version}")
|
||||
if formal_parts:
|
||||
snippet += f" Formal publication: {'; '.join(formal_parts)}."
|
||||
matches.append({
|
||||
"title": result_title,
|
||||
"url": url,
|
||||
"snippet": snippet,
|
||||
"source": "openalex",
|
||||
})
|
||||
return matches
|
||||
|
||||
|
||||
def _scholarly_title_results(title: str, count: int = 3) -> list[dict]:
|
||||
"""Retry a noisy scholarly query as a bare title, then use arXiv API."""
|
||||
|
||||
try:
|
||||
simplified = searxng_search_api(title, count=max(3, count))
|
||||
except Exception as exc:
|
||||
logger.info("Simplified scholarly search failed for %r: %s", title, exc)
|
||||
simplified = []
|
||||
exact = [
|
||||
result for result in simplified
|
||||
if _result_strongly_matches_title(title, result)
|
||||
]
|
||||
if exact:
|
||||
return exact[:count]
|
||||
openalex = _openalex_title_results(title, count)
|
||||
if openalex:
|
||||
return openalex
|
||||
return _arxiv_title_results(title, count)
|
||||
|
||||
|
||||
def _direct_scholarly_title_results(title: str, count: int = 3) -> list[dict]:
|
||||
"""Resolve a clear paper title without waiting on generic search providers."""
|
||||
|
||||
# OpenAlex typically resolves titles in under a second and often returns
|
||||
# the official arXiv landing page. The arXiv API remains the fallback.
|
||||
openalex = _openalex_title_results(title, count)
|
||||
if openalex:
|
||||
return openalex
|
||||
return _arxiv_title_results(title, count)
|
||||
|
||||
|
||||
def _augment_scholarly_results(query: str, results: list[dict], count: int) -> list[dict]:
|
||||
"""Prepend an exact arXiv match when a scholarly SERP missed its title."""
|
||||
|
||||
current = list(results or [])
|
||||
identifier_results = _exact_arxiv_identifier_results(query)
|
||||
if identifier_results:
|
||||
title = _title_before_explicit_arxiv_identifier(query)
|
||||
formal_results: list[dict] = []
|
||||
if title:
|
||||
formal_results = [
|
||||
item
|
||||
for item in _openalex_title_results(title, min(count, 3))
|
||||
if "arxiv.org/" not in str(item.get("url") or "").lower()
|
||||
]
|
||||
exact_urls = {str(item["url"]) for item in identifier_results}
|
||||
formal_urls = {str(item.get("url") or "") for item in formal_results}
|
||||
return (
|
||||
formal_results
|
||||
+ identifier_results
|
||||
+ [
|
||||
item for item in current
|
||||
if str(item.get("url") or "") not in exact_urls | formal_urls
|
||||
]
|
||||
)[:count]
|
||||
title = _scholarly_title_from_query(query)
|
||||
if not title:
|
||||
return current
|
||||
exact_current = [
|
||||
item for item in current
|
||||
if _result_strongly_matches_title(title, item)
|
||||
]
|
||||
if exact_current:
|
||||
exact_ids = {id(item) for item in exact_current}
|
||||
return (exact_current + [item for item in current if id(item) not in exact_ids])[:count]
|
||||
arxiv_results = _scholarly_title_results(title, min(count, 3))
|
||||
if not arxiv_results:
|
||||
return current
|
||||
seen = {str(item.get("url") or "") for item in arxiv_results}
|
||||
return (arxiv_results + [item for item in current if str(item.get("url") or "") not in seen])[:count]
|
||||
|
||||
|
||||
def _subject_first_weather_query(query: str) -> str:
|
||||
"""Rewrite natural weather questions into the shape SearXNG handles best."""
|
||||
text = re.sub(r"\s+", " ", str(query or "")).strip(" ?")
|
||||
if not text:
|
||||
return text
|
||||
if not (set(re.findall(r"[a-z0-9]+", text.lower())) & _WEATHER_QUERY_HINTS):
|
||||
return text
|
||||
loc_match = re.search(
|
||||
r"\b(?:weather|forecast)\s+(?:in|for|at)\s+(.+)$",
|
||||
text,
|
||||
re.IGNORECASE,
|
||||
)
|
||||
if not loc_match:
|
||||
loc_match = re.search(
|
||||
r"\b(?:weather|forecast)\b.*?\b(?:in|for|at)\s+(.+)$",
|
||||
text,
|
||||
re.IGNORECASE,
|
||||
)
|
||||
if not loc_match:
|
||||
return text
|
||||
location = loc_match.group(1).strip(" ?.,")
|
||||
timing = ""
|
||||
timing_match = re.search(
|
||||
r"\b(today|tomorrow|tonight|this\s+week|next\s+week|now|current)\b",
|
||||
location,
|
||||
re.IGNORECASE,
|
||||
)
|
||||
if timing_match:
|
||||
timing = timing_match.group(1).lower()
|
||||
location = (
|
||||
location[: timing_match.start()] + location[timing_match.end():]
|
||||
).strip(" ?.,")
|
||||
if not location:
|
||||
return text
|
||||
return re.sub(r"\s+", " ", f"{location} weather forecast {timing}").strip()
|
||||
|
||||
|
||||
def _provider_friendly_query(query: str) -> str:
|
||||
"""Convert generic question grammar to keyword order without changing its topic."""
|
||||
text = _subject_first_weather_query(query)
|
||||
match = re.fullmatch(
|
||||
r"(?:what|which)\s+(year|date|time)\s+(?:did|does|do|was|were|is|are)\s+(.+)",
|
||||
text,
|
||||
re.IGNORECASE,
|
||||
)
|
||||
if match:
|
||||
return f"{match.group(2).strip()} {match.group(1).lower()}"
|
||||
# Search providers already receive recency separately. Remove a leading
|
||||
# conversational request shell so ranking is driven by the subject rather
|
||||
# than words such as "any", "latest", and "information".
|
||||
cleaned = re.sub(
|
||||
r"^(?:can|could|would)\s+you\s+(?:find|search|look\s+up)\s+",
|
||||
"",
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
cleaned = re.sub(
|
||||
r"^(?:any\s+)?(?:latest|current|recent)?\s*"
|
||||
r"(?:news|info(?:rmation)?|updates?|details?)\s+(?:on|about)\s+",
|
||||
"",
|
||||
cleaned,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
if cleaned.strip():
|
||||
return cleaned.strip()
|
||||
return text
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------
|
||||
@@ -135,6 +618,7 @@ def _build_provider_chain(primary: str) -> List[str]:
|
||||
# ----------------------------------------------------------------------
|
||||
def searxng_search_results(query: str, count: int = 10, time_filter: str = None) -> list[dict]:
|
||||
"""Perform a web search using configured provider with caching and retry."""
|
||||
provider_query = _provider_friendly_query(query)
|
||||
settings = _get_search_settings()
|
||||
search_provider = settings.get("search_provider", "searxng")
|
||||
result_count = _get_result_count()
|
||||
@@ -142,7 +626,17 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None)
|
||||
if count == 10:
|
||||
count = result_count
|
||||
|
||||
cache_key = generate_cache_key(f"{query}|{count}|{time_filter}")
|
||||
# A named scholarly work has a deterministic metadata path. Resolve that
|
||||
# first instead of spending the full tool deadline retrying generic search
|
||||
# providers; the returned official URL lets the agent proceed to PDF tools.
|
||||
scholarly_title = _scholarly_title_from_query(provider_query)
|
||||
if scholarly_title:
|
||||
direct_results = _direct_scholarly_title_results(scholarly_title, count)
|
||||
if direct_results:
|
||||
_record_query(provider_query, True, cache_hit=False)
|
||||
return direct_results[:count]
|
||||
|
||||
cache_key = generate_cache_key(f"{provider_query}|{count}|{time_filter}")
|
||||
cache_file = SEARCH_CACHE_DIR / f"{cache_key}.cache"
|
||||
|
||||
# Check cache
|
||||
@@ -155,8 +649,22 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None)
|
||||
if expiry and datetime.now() < expiry:
|
||||
logger.debug(f"Search cache hit for query: {query}")
|
||||
results = cached_data["data"]
|
||||
_record_query(query, bool(results), cache_hit=True)
|
||||
return results
|
||||
# Ranking/relevance logic evolves independently from provider
|
||||
# results. Re-apply it on cache hits so stale cached ordering
|
||||
# does not preserve bad SERP choices after a harness fix.
|
||||
results = _filter_low_relevance_results(provider_query, results)
|
||||
if results:
|
||||
results = rank_search_results(provider_query, results)
|
||||
results = _augment_scholarly_results(provider_query, results, count)
|
||||
if results:
|
||||
_record_query(query, True, cache_hit=True)
|
||||
return results
|
||||
logger.info(
|
||||
"Search cache hit for %r became empty after relevance filtering; refetching",
|
||||
provider_query,
|
||||
)
|
||||
cache_file.unlink(missing_ok=True)
|
||||
search_cache_index.pop(cache_key, None)
|
||||
else:
|
||||
cache_file.unlink(missing_ok=True)
|
||||
search_cache_index.pop(cache_key, None)
|
||||
@@ -178,7 +686,8 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None)
|
||||
for attempt in range(2):
|
||||
try:
|
||||
logger.info(f"Attempting {provider_name} search (attempt {attempt + 1})")
|
||||
results = _call_provider(provider_name, query, count, time_filter)
|
||||
results = _call_provider(provider_name, provider_query, count, time_filter)
|
||||
results = _filter_low_relevance_results(provider_query, results)
|
||||
if results:
|
||||
logger.info(f"{provider_name} search succeeded with {len(results)} results")
|
||||
break
|
||||
@@ -189,11 +698,14 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None)
|
||||
if results:
|
||||
break
|
||||
|
||||
results = _augment_scholarly_results(provider_query, results, count)
|
||||
|
||||
success = bool(results)
|
||||
_record_query(query, success, cache_hit=False)
|
||||
_record_query(provider_query, success, cache_hit=False)
|
||||
|
||||
if success:
|
||||
results = rank_search_results(query, results)
|
||||
results = rank_search_results(provider_query, results)
|
||||
results = _augment_scholarly_results(provider_query, results, count)
|
||||
try:
|
||||
expiry = datetime.now() + _cache_duration_for_query(query)
|
||||
cache_data = {
|
||||
@@ -206,10 +718,10 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None)
|
||||
search_cache_index[cache_key] = datetime.now()
|
||||
cleanup_cache(SEARCH_CACHE_DIR, search_cache_index, timedelta(hours=1))
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to write search cache for {query}: {e}")
|
||||
logger.warning(f"Failed to write search cache for {provider_query}: {e}")
|
||||
|
||||
if not success:
|
||||
logger.error(f"All search providers failed for query: {query}")
|
||||
logger.error(f"All search providers failed for query: {provider_query}")
|
||||
|
||||
return results
|
||||
|
||||
@@ -260,7 +772,8 @@ def comprehensive_web_search(
|
||||
return_sources: bool = False,
|
||||
):
|
||||
"""Perform comprehensive web search with content fetching and advanced filtering."""
|
||||
logger.info(f"Starting comprehensive search for: {query}")
|
||||
provider_query = _provider_friendly_query(query)
|
||||
logger.info(f"Starting comprehensive search for: {provider_query}")
|
||||
if time_filter:
|
||||
logger.info(f"Applying time filter: {time_filter}")
|
||||
|
||||
@@ -285,7 +798,8 @@ def comprehensive_web_search(
|
||||
empty = False
|
||||
for attempt in range(2):
|
||||
try:
|
||||
search_results = _call_provider(provider_name, query, fetch_count, time_filter)
|
||||
search_results = _call_provider(provider_name, provider_query, fetch_count, time_filter)
|
||||
search_results = _filter_low_relevance_results(provider_query, search_results)
|
||||
if search_results:
|
||||
provider_attempts[provider_name] = f"ok ({len(search_results)})"
|
||||
logger.info(f"Comprehensive search: {provider_name} returned {len(search_results)} results")
|
||||
@@ -301,6 +815,12 @@ def comprehensive_web_search(
|
||||
elif empty:
|
||||
provider_attempts[provider_name] = "empty"
|
||||
|
||||
search_results = _augment_scholarly_results(
|
||||
provider_query,
|
||||
search_results,
|
||||
fetch_count,
|
||||
)
|
||||
|
||||
if not search_results:
|
||||
tally = ", ".join(f"{p}:{r}" for p, r in provider_attempts.items()) or "no providers configured"
|
||||
any_errors = any(r.startswith("error") for r in provider_attempts.values())
|
||||
@@ -315,7 +835,12 @@ def comprehensive_web_search(
|
||||
logger.warning(msg)
|
||||
return (msg, []) if return_sources else msg
|
||||
|
||||
search_results = rank_search_results(query, search_results)
|
||||
search_results = rank_search_results(provider_query, search_results)
|
||||
search_results = _augment_scholarly_results(
|
||||
provider_query,
|
||||
search_results,
|
||||
fetch_count,
|
||||
)
|
||||
|
||||
# URL filter helper
|
||||
def url_passes_filters(url: str) -> bool:
|
||||
@@ -399,7 +924,7 @@ def comprehensive_web_search(
|
||||
|
||||
output_parts.append("=" * 70)
|
||||
output_parts.append("WEB SEARCH RESULTS AND FETCHED CONTENT")
|
||||
output_parts.append(f"Query: {query}")
|
||||
output_parts.append(f"Query: {provider_query}")
|
||||
output_parts.append(f"Searched {len(search_results)} results, fetched {len(fetched_content)} pages")
|
||||
output_parts.append("=" * 70)
|
||||
output_parts.append("")
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
from typing import List, Optional
|
||||
from urllib.parse import urljoin, urlparse, parse_qs
|
||||
|
||||
@@ -33,9 +34,16 @@ def _get_search_settings() -> dict:
|
||||
"""Return search settings from admin config, falling back to env defaults."""
|
||||
try:
|
||||
from src.settings import load_settings
|
||||
return load_settings()
|
||||
settings = dict(load_settings())
|
||||
except Exception:
|
||||
return {}
|
||||
settings = {}
|
||||
# Headless/native deployments do not necessarily have an admin settings
|
||||
# database. Require an explicit Odysseus-prefixed override so ordinary UI
|
||||
# configuration remains authoritative by default.
|
||||
env_provider = os.environ.get("ODYSSEUS_SEARCH_PROVIDER", "").strip().lower()
|
||||
if env_provider:
|
||||
settings["search_provider"] = env_provider
|
||||
return settings
|
||||
|
||||
|
||||
def _get_search_instance() -> str:
|
||||
@@ -66,13 +74,18 @@ def _get_provider_key(provider: str) -> str:
|
||||
if legacy:
|
||||
return legacy
|
||||
env_map = {
|
||||
"brave": "DATA_BRAVE_API_KEY",
|
||||
"google_pse": "GOOGLE_API_KEY",
|
||||
"tavily": "TAVILY_API_KEY",
|
||||
"serper": "SERPER_API_KEY",
|
||||
# DATA_BRAVE_API_KEY is the historical Odysseus name; BRAVE_API_KEY is
|
||||
# the standard name used by headless runners and the Brave SDK.
|
||||
"brave": ("DATA_BRAVE_API_KEY", "BRAVE_API_KEY"),
|
||||
"google_pse": ("GOOGLE_API_KEY",),
|
||||
"tavily": ("TAVILY_API_KEY",),
|
||||
"serper": ("SERPER_API_KEY",),
|
||||
}
|
||||
env_name = env_map.get(provider, "")
|
||||
return (os.environ.get(env_name) or "").strip() if env_name else ""
|
||||
for env_name in env_map.get(provider, ()):
|
||||
value = (os.environ.get(env_name) or "").strip()
|
||||
if value:
|
||||
return value
|
||||
return ""
|
||||
|
||||
|
||||
def _get_result_count() -> int:
|
||||
@@ -84,6 +97,19 @@ def _get_result_count() -> int:
|
||||
return 5
|
||||
|
||||
|
||||
def provider_configured(provider: str) -> bool:
|
||||
"""Configuration readiness only; a configured engine can still fail upstream."""
|
||||
if provider in {"searxng", "searxng_yep", "duckduckgo"}:
|
||||
return True
|
||||
if provider not in {"brave", "google_pse", "tavily", "serper"}:
|
||||
return False
|
||||
if not _get_provider_key(provider):
|
||||
return False
|
||||
if provider == "google_pse":
|
||||
return bool(_get_search_settings().get("google_pse_cx") or os.environ.get("GOOGLE_PSE_CX"))
|
||||
return True
|
||||
|
||||
|
||||
# Canonical SafeSearch levels: "strict" (default), "moderate", "off".
|
||||
# Each provider has its own knob name and value space -- see _safesearch_for(...).
|
||||
_SAFESEARCH_LEVELS = ("strict", "moderate", "off")
|
||||
@@ -124,6 +150,24 @@ def _safesearch_for(provider: str) -> Optional[str]:
|
||||
# ── SearXNG ──
|
||||
|
||||
_NEWS_HINTS = ("news", "nyheter", "headlines", "breaking", "latest", "today", "idag")
|
||||
_NEWS_EVENT_HINT_RE = re.compile(
|
||||
r"\b(?:deport(?:ation|ed|ing)?|arrest(?:ed|s)?|election(?:s)?|"
|
||||
r"evacuat(?:e|ed|ion)|flood(?:ing|s|ed)?|sanction(?:s|ed)?)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_SOFTWARE_RELEASE_HINTS = (
|
||||
"github",
|
||||
"gitlab",
|
||||
"release",
|
||||
"releases",
|
||||
"version",
|
||||
"versions",
|
||||
"changelog",
|
||||
"change log",
|
||||
"pypi",
|
||||
"npm",
|
||||
"package",
|
||||
)
|
||||
|
||||
# Default general engines (google/duckduckgo/brave/startpage/wikipedia) are
|
||||
# routinely rate-limited / CAPTCHA-blocked on this instance and return nothing.
|
||||
@@ -133,7 +177,7 @@ _GENERAL_ENGINES = os.environ.get("SEARXNG_GENERAL_ENGINES", "bing,mojeek,presea
|
||||
|
||||
|
||||
def searxng_search_api(query: str, count: Optional[int] = None, categories: str = "general",
|
||||
time_filter: Optional[str] = None) -> List[dict]:
|
||||
time_filter: Optional[str] = None, *, engines: Optional[str] = None) -> List[dict]:
|
||||
"""Search using SearXNG JSON API. Returns list of {title, url, snippet}."""
|
||||
count = count if count is not None else _get_result_count()
|
||||
instance = _get_search_instance()
|
||||
@@ -158,7 +202,19 @@ def searxng_search_api(query: str, count: Optional[int] = None, categories: str
|
||||
"safesearch": _safesearch_for("searxng"),
|
||||
}
|
||||
q_lc = query.lower()
|
||||
is_news = time_filter is not None or any(h in q_lc for h in _NEWS_HINTS)
|
||||
# Fresh software-version queries are usually better served by general
|
||||
# search or canonical project pages than by the news vertical. For example
|
||||
# "latest ollama release version github" can return a sparse news result
|
||||
# that gets filtered as irrelevant, while general engines find GitHub.
|
||||
is_software_release_query = any(h in q_lc for h in _SOFTWARE_RELEASE_HINTS)
|
||||
is_news = (
|
||||
not is_software_release_query
|
||||
and (
|
||||
time_filter is not None
|
||||
or any(h in q_lc for h in _NEWS_HINTS)
|
||||
or bool(_NEWS_EVENT_HINT_RE.search(query))
|
||||
)
|
||||
)
|
||||
if is_news and categories == "general":
|
||||
params["categories"] = "news"
|
||||
if time_filter in ("day", "week", "month", "year"):
|
||||
@@ -171,6 +227,9 @@ def searxng_search_api(query: str, count: Optional[int] = None, categories: str
|
||||
# set returns 0 on this instance — see _GENERAL_ENGINES).
|
||||
if categories == "general" and _GENERAL_ENGINES:
|
||||
params["engines"] = _GENERAL_ENGINES
|
||||
if engines:
|
||||
params["categories"] = "general"
|
||||
params["engines"] = engines
|
||||
try:
|
||||
def _parse_results(results):
|
||||
return [
|
||||
@@ -178,6 +237,10 @@ def searxng_search_api(query: str, count: Optional[int] = None, categories: str
|
||||
"title": r.get("title", ""),
|
||||
"url": r.get("url", ""),
|
||||
"snippet": r.get("content", ""),
|
||||
"provider": "searxng",
|
||||
"engines": r.get("engines", []),
|
||||
"published_date": r.get("publishedDate"),
|
||||
"query": query,
|
||||
}
|
||||
for r in results[:count]
|
||||
if r.get("url")
|
||||
|
||||
@@ -67,6 +67,22 @@ _TRUSTED_NEWS_DOMAINS = {
|
||||
"www.theguardian.com", "euronews.com", "www.euronews.com",
|
||||
"dw.com", "www.dw.com", "government.se", "www.government.se",
|
||||
}
|
||||
_SOFTWARE_RELEASE_HINTS = {
|
||||
"github", "gitlab", "release", "releases", "version", "versions",
|
||||
"changelog", "package", "pypi", "npm",
|
||||
}
|
||||
_PRODUCT_SPEC_HINTS = {
|
||||
"product", "hardware", "device", "phone", "laptop", "desktop", "computer",
|
||||
"chip", "cpu", "gpu", "mac", "iphone", "ipad", "android", "camera",
|
||||
"console", "kindle", "tesla", "car", "model", "price", "pricing", "cost",
|
||||
"buy", "shop", "order", "preorder", "pre-order", "spec", "specs",
|
||||
"specifications", "available", "availability", "ship", "shipping",
|
||||
"released", "launch", "launched", "vram", "memory", "ram", "storage",
|
||||
}
|
||||
_COMMERCE_OR_SPEC_PATH_HINTS = (
|
||||
"/shop", "/buy", "/store", "/product", "/products", "/spec", "/specs",
|
||||
"/support", "/tech-specs", "/technical-specifications",
|
||||
)
|
||||
|
||||
|
||||
def _domain(url: str) -> str:
|
||||
@@ -95,6 +111,8 @@ def rank_search_results(query: str, results: List[dict]) -> List[dict]:
|
||||
query_lc = query.lower()
|
||||
is_news_query = any(term in _NEWS_HINTS for term in query_terms)
|
||||
is_sports_query = bool(_SPORTS_HINT_RE.search(query_lc))
|
||||
is_software_release_query = any(term in _SOFTWARE_RELEASE_HINTS for term in query_terms)
|
||||
is_product_spec_query = any(term in _PRODUCT_SPEC_HINTS for term in query_terms)
|
||||
|
||||
def title_score(title: str) -> float:
|
||||
if not title:
|
||||
@@ -144,6 +162,41 @@ def rank_search_results(query: str, results: List[dict]) -> List[dict]:
|
||||
adjustment -= 1.0
|
||||
return adjustment
|
||||
|
||||
def software_release_adjustment(title: str, snippet: str, url: str) -> float:
|
||||
if not is_software_release_query:
|
||||
return 0.0
|
||||
netloc = _domain(url)
|
||||
path = urlparse(url).path.lower()
|
||||
text = f"{title} {snippet} {netloc} {path}".lower()
|
||||
adjustment = 0.0
|
||||
if netloc in {"github.com", "www.github.com", "gitlab.com", "www.gitlab.com"}:
|
||||
adjustment += 1.6
|
||||
if "/releases" in path or "/tags" in path:
|
||||
adjustment += 1.2
|
||||
if any(_has_word(text, term) for term in ("release", "releases", "changelog", "version")):
|
||||
adjustment += 0.4
|
||||
if netloc in {"releasealert.dev", "releases.sh", "releasebot.io"}:
|
||||
adjustment -= 0.8
|
||||
return adjustment
|
||||
|
||||
def product_spec_adjustment(title: str, snippet: str, url: str) -> float:
|
||||
if not is_product_spec_query:
|
||||
return 0.0
|
||||
parsed = urlparse(url)
|
||||
netloc = parsed.netloc.lower()
|
||||
path = parsed.path.lower()
|
||||
text = f"{title} {snippet} {netloc} {path}".lower()
|
||||
adjustment = 0.0
|
||||
if any(hint in path for hint in _COMMERCE_OR_SPEC_PATH_HINTS):
|
||||
adjustment += 1.1
|
||||
if re.search(r"\b(?:official|specs?|specifications|tech specs|buy|shop|store|price|pricing|available|ships?)\b", text):
|
||||
adjustment += 0.5
|
||||
if netloc.endswith(".com") and any(_has_word(netloc, term) for term in query_terms if len(term) >= 4):
|
||||
adjustment += 0.4
|
||||
if re.search(r"\b(?:rumor|rumour|leak|may|could|expected|reportedly|unannounced)\b", text):
|
||||
adjustment -= 0.8
|
||||
return adjustment
|
||||
|
||||
ranked = []
|
||||
for result in results:
|
||||
title = result.get("title", "")
|
||||
@@ -157,6 +210,8 @@ def rank_search_results(query: str, results: List[dict]) -> List[dict]:
|
||||
+ 1.5 * domain_score(url)
|
||||
+ 1.0 * recency_score(age)
|
||||
+ news_quality_adjustment(title, snippet, url)
|
||||
+ software_release_adjustment(title, snippet, url)
|
||||
+ product_spec_adjustment(title, snippet, url)
|
||||
)
|
||||
ranked.append((score, result))
|
||||
|
||||
|
||||
Reference in New Issue
Block a user