Squash Odysseus development history

This commit is contained in:
pewdiepie-archdaemon
2026-09-11 06:04:19 +00:00
parent c9dd68d890
commit 84aa9a91de
871 changed files with 265870 additions and 27854 deletions
+37 -12
View File
@@ -1,18 +1,43 @@
# services/__init__.py
"""
Service layer — plug-in capabilities for the chat core.
"""Service-layer exports with lazy loading.
Each service:
- Does one thing well
- Exposes a clean async interface
- Can run in-process or as a standalone HTTP service
Importing one service, such as ``services.hwfit``, must not initialize every
other service. The eager exports previously imported search, document,
research, memory, and shell stacks during any ``services.*`` import, making
Cookbook hardware/model discovery needlessly slow on a cold process.
"""
from .search import SearchService, SearchResult, SearchResponse
from .docs import DocsService, DocChunk, IndexResult
from .research import ResearchService, ResearchResult, ResearchSource
from .memory import MemoryService, Memory, MemorySearchResult
from .shell import ShellService, ShellResult
from importlib import import_module
_LAZY_EXPORTS = {
"SearchService": ("search", "SearchService"),
"SearchResult": ("search", "SearchResult"),
"SearchResponse": ("search", "SearchResponse"),
"DocsService": ("docs", "DocsService"),
"DocChunk": ("docs", "DocChunk"),
"IndexResult": ("docs", "IndexResult"),
"ResearchService": ("research", "ResearchService"),
"ResearchResult": ("research", "ResearchResult"),
"ResearchSource": ("research", "ResearchSource"),
"MemoryService": ("memory", "MemoryService"),
"Memory": ("memory", "Memory"),
"MemorySearchResult": ("memory", "MemorySearchResult"),
"ShellService": ("shell", "ShellService"),
"ShellResult": ("shell", "ShellResult"),
}
def __getattr__(name):
target = _LAZY_EXPORTS.get(name)
if target is None:
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
module_name, attribute = target
value = getattr(import_module(f"{__name__}.{module_name}"), attribute)
globals()[name] = value
return value
def __dir__():
return sorted(set(globals()) | set(_LAZY_EXPORTS))
__all__ = [
# Search
File diff suppressed because it is too large Load Diff
+1 -1
View File
@@ -748,6 +748,7 @@ def rank_models(system, use_case=None, limit=50, search=None, sort="score", quan
"is_image_gen": True,
"capabilities": im.get("capabilities", []),
"description": im.get("description", ""),
"dependency_package": im.get("dependency_package", ""),
})
if use_case == "image_gen":
sort_fn = SORT_KEYS.get(sort, SORT_KEYS["score"])
@@ -839,7 +840,6 @@ def rank_models(system, use_case=None, limit=50, search=None, sort="score", quan
# native AWQ rows only on accelerator servers that can serve them.
if (
quant == "Q4_K_M"
and system.get("gpu_count", 1) >= 2
and not (apple_silicon or consumer_amd or is_windows)
and native_q == "AWQ-4bit"
):
+44 -2
View File
@@ -7,8 +7,11 @@ import re
import time
import urllib.parse
import urllib.request
from pathlib import Path
from typing import Any
from src.constants import DATA_DIR
# Image models are discovered from HuggingFace collections/search and local cache.
# Keep this empty: source-coded repo IDs become hidden recommendations.
IMAGE_MODEL_REGISTRY: list[dict[str, Any]] = []
@@ -31,6 +34,8 @@ HF_IMAGE_REPO_SEEDS: list[str] = []
_HF_COLLECTION_CACHE = {"ts": 0.0, "models": []}
_HF_COLLECTION_TTL = 30 * 60
_IMAGE_COLLECTION_DISK_CACHE = Path(DATA_DIR) / "hwfit" / "image_collection_models.json"
_IMAGE_COLLECTION_DISK_TTL = 24 * 3600
_HF_VARIANT_CACHE: dict[str, dict[str, str]] = {}
_HF_SEARCH_DISABLED_UNTIL = 0.0
@@ -161,6 +166,12 @@ def _collection_item_to_model(item: dict[str, Any], collection_title: str = "",
"speed": est["speed"],
"released": "",
}
# Optional catalog metadata may identify a non-default runtime package.
# Keep this data-driven: the fitter must not infer private/model-specific
# dependencies from repository names.
dependency_package = item.get("dependency_package") or item.get("runtime_dependency")
if isinstance(dependency_package, str) and dependency_package.strip():
out["dependency_package"] = dependency_package.strip()
if mlx_only:
out["mlx_only"] = True
out["description"] = (out["description"] + " Apple Silicon / MLX only.").strip()
@@ -171,6 +182,21 @@ def _fetch_hf_image_collection_models() -> list[dict[str, Any]]:
now = time.time()
if now - float(_HF_COLLECTION_CACHE.get("ts") or 0) < _HF_COLLECTION_TTL:
return list(_HF_COLLECTION_CACHE.get("models") or [])
# Reuse the last successful discovery across process restarts. A stale
# catalog is preferable to blocking the first image-tab render on several
# sequential Hugging Face requests; a later refresh replaces it.
if not _HF_COLLECTION_CACHE.get("models"):
try:
cached = json.loads(_IMAGE_COLLECTION_DISK_CACHE.read_text(encoding="utf-8"))
cached_models = cached.get("models") if isinstance(cached, dict) else None
cached_ts = float(cached.get("fetched_at") or 0) if isinstance(cached, dict) else 0
if isinstance(cached_models, list) and cached_models:
_HF_COLLECTION_CACHE["ts"] = cached_ts
_HF_COLLECTION_CACHE["models"] = cached_models
if now - cached_ts < _IMAGE_COLLECTION_DISK_TTL:
return list(cached_models)
except (OSError, ValueError, TypeError):
pass
models: list[dict[str, Any]] = []
for slug, mlx_only in [(slug, False) for slug in HF_IMAGE_COLLECTIONS] + [(slug, True) for slug in HF_MLX_IMAGE_COLLECTIONS]:
url = f"https://huggingface.co/api/collections/{slug}"
@@ -186,9 +212,24 @@ def _fetch_hf_image_collection_models() -> list[dict[str, Any]]:
model = _collection_item_to_model(item, title, mlx_only=mlx_only)
if model:
models.append(model)
if models:
_HF_COLLECTION_CACHE["ts"] = now
_HF_COLLECTION_CACHE["models"] = models
try:
_IMAGE_COLLECTION_DISK_CACHE.parent.mkdir(parents=True, exist_ok=True)
tmp = _IMAGE_COLLECTION_DISK_CACHE.with_suffix(".tmp")
tmp.write_text(json.dumps({"fetched_at": now, "models": models}), encoding="utf-8")
tmp.replace(_IMAGE_COLLECTION_DISK_CACHE)
except OSError:
pass
return list(models)
# Preserve stale results if the network is unavailable. The in-memory
# timestamp prevents every subsequent ranking request from retrying it.
if _HF_COLLECTION_CACHE.get("models"):
_HF_COLLECTION_CACHE["ts"] = now
return list(_HF_COLLECTION_CACHE["models"])
_HF_COLLECTION_CACHE["ts"] = now
_HF_COLLECTION_CACHE["models"] = models
return list(models)
return []
def _hf_model_search(query: str, limit: int = 10) -> list[dict[str, Any]]:
@@ -420,6 +461,7 @@ def rank_image_models(system, search=None, sort="fit"):
"capabilities": model["capabilities"],
"description": model["description"],
"released": model.get("released", ""),
"dependency_package": model.get("dependency_package", ""),
})
# Sort
+74
View File
@@ -0,0 +1,74 @@
"""Install tracked built-in skills into the shared immutable skill catalog."""
from __future__ import annotations
from pathlib import Path
from typing import Iterable
from .skill_format import Skill
from .skills import SkillsManager
_BUILTIN_ROOT = Path(__file__).resolve().parents[2] / "resources" / "skills"
_SYNC_FIELDS = (
"name",
"description",
"version",
"category",
"tags",
"status",
"confidence",
"source",
"owner",
"when_to_use",
"procedure",
"pitfalls",
"verification",
"platforms",
"requires_toolsets",
"fallback_for_toolsets",
"body_extra",
)
def install_builtin_skills(manager: SkillsManager, owners: Iterable[str]) -> int:
"""Copy missing built-in skills into the ownerless shared catalog.
Built-ins are explicitly marked and remain ownerless because the on-disk
skill path is not owner-qualified. ``SkillsManager.load(owner=...)``
exposes only these immutable built-ins in addition to that owner's files.
Installation is safe before first-user setup because no owner identity is
assigned and unauthenticated requests still cannot access skill routes.
"""
existing = {row.get("name") for row in manager.load_all()}
installed = 0
paths = sorted(_BUILTIN_ROOT.rglob("SKILL.md")) if _BUILTIN_ROOT.is_dir() else []
for path in paths:
try:
skill = Skill.from_markdown(path.read_text(encoding="utf-8"))
except Exception:
continue
# Tracked procedures ship as trusted application behavior. They are
# available immediately and never enter the user's audit queue.
skill.status = "published"
skill.confidence = 1.0
existing_rows = [row for row in manager.load_all() if row.get("name") == skill.name]
if existing_rows:
row = existing_rows[0]
# Built-ins are immutable tracked assets. Synchronize updated
# versions/procedures on startup while leaving usage counters in
# their sidecar untouched. Older startup code could also stamp the
# first admin onto one; normalize that migration at the same time.
if row.get("source") == "builtin":
skill.owner = ""
skill.source = "builtin"
desired = skill.to_dict()
if any(row.get(field) != desired.get(field) for field in _SYNC_FIELDS):
manager._write_skill(skill)
continue
skill.owner = ""
skill.source = "builtin"
manager._write_skill(skill)
existing.add(skill.name)
installed += 1
return installed
+132 -3
View File
@@ -90,6 +90,29 @@ EXTRACT_SYSTEM_PROMPT = (
# How many recent messages to include for extraction
CONTEXT_WINDOW = 6
PERSONA_MEMORY_SYSTEM_PROMPT = (
"You maintain concise continuity notes for one active chat persona. "
"Update the existing notes using only durable details established in the transcript. "
"Keep details that help the same persona stay consistent in future conversations: "
"relationship context, names, preferences, recurring story details, boundaries, and unresolved threads. "
"Do not store generic chat events, temporary wording, assistant reasoning, or one-off requests. "
"Never invent details. Return only the updated notes as short bullet points, max 12 bullets. "
"If there is nothing worth keeping, return the existing notes unchanged or an empty string."
)
HEALTH_PERSONA_MEMORY_SYSTEM_PROMPT = (
"You maintain a cautious health-record brief for a medical reasoning persona. "
"Update the existing brief using only medically durable information from the transcript. "
"Keep facts that may matter in future health conversations: confirmed diagnoses, chronic conditions, "
"surgeries/procedures, allergies, regular medications/supplements, important test results, clinicians/hospitals, "
"ongoing symptoms or care plans, and the user's preferences for medical explanations. "
"Use uncertainty labels when needed: 'reported', 'possible', 'asked about', 'unclear'. "
"Do not turn guesses into diagnoses. Do not store casual one-off symptoms unless they are recurring, severe, "
"or tied to an ongoing episode. Never invent facts. Return only the updated brief with these headings when useful: "
"Medical profile, Medications/allergies, Episodes/open questions, Preferences. Max 16 concise bullets total. "
"If nothing medically durable changed, return the existing brief unchanged or an empty string."
)
AUDIT_SYSTEM_PROMPT = (
"You are a memory database curator. Be CONSERVATIVE: remove only TRUE "
"duplicates and clearly useless entries. Every distinct fact must survive. "
@@ -112,6 +135,20 @@ AUDIT_SYSTEM_PROMPT = (
)
AUDIT_INTERVAL = 5 # audit every N new memories added
AUTO_PINNED_IDENTITY_LIMIT = 5
def _is_owner_memory(entry, owner):
if owner:
return entry.get("owner") == owner or entry.get("owner") is None
return True
def _is_auto_pinned_identity(entry):
return (
bool(entry.get("pinned"))
and (entry.get("category") or "").lower() in {"identity", "contact"}
)
_extractions_since_audit = 0
@@ -397,6 +434,10 @@ async def extract_and_store(
logger.error("Skipping auto memory extraction, store unreadable: %s", e)
return
added = 0
auto_pinned_identity_count = sum(
1 for entry in existing
if _is_owner_memory(entry, _owner) and _is_auto_pinned_identity(entry)
)
for fact in facts:
if isinstance(fact, str):
@@ -404,7 +445,7 @@ async def extract_and_store(
category = "fact"
elif isinstance(fact, dict):
fact_text = fact.get("text", "").strip()
category = fact.get("category", "fact")
category = str(fact.get("category", "fact") or "fact")
else:
continue
@@ -446,9 +487,15 @@ async def extract_and_store(
continue
entry = memory_manager.add_entry(fact_text, source="auto", category=category, owner=_owner)
# Auto-pin identity facts (name, job, location) — core context
if category == "identity":
# Auto-pin only the first few identity/contact facts. Extra identity
# memories are still saved, but they must be recalled by relevance
# instead of riding along in every prompt forever.
if (
category.lower() in {"identity", "contact"}
and auto_pinned_identity_count < AUTO_PINNED_IDENTITY_LIMIT
):
entry["pinned"] = True
auto_pinned_identity_count += 1
if hasattr(session, "session_id"):
entry["session_id"] = session.session_id
elif hasattr(session, "name"):
@@ -492,6 +539,88 @@ async def extract_and_store(
logger.error(f"Memory extraction failed: {e}")
async def update_persona_memory(
session,
preset_manager,
character_name: str,
endpoint_url: str,
model: str,
headers: Optional[dict] = None,
schema: str = "general",
):
"""Update the active persona's continuity notes from recent conversation.
Persona memory is stored with the persona/template data, not in the global
memory DB, so deleting a saved persona also deletes its notes.
"""
character_name = (character_name or "").strip()
if not character_name or not endpoint_url or not model or preset_manager is None:
return
try:
from src.llm_core import llm_call_async
from src.text_helpers import strip_think
custom = {}
try:
custom = preset_manager.presets.get("custom", {}) if isinstance(preset_manager.presets, dict) else {}
except Exception:
custom = {}
existing_memory = ""
if isinstance(custom, dict) and custom.get("character_name") == character_name:
existing_memory = custom.get("persona_memory", "") or ""
messages = session.get_context_messages()
recent = messages[-CONTEXT_WINDOW:] if len(messages) > CONTEXT_WINDOW else messages
if len(recent) < 2:
return
lines = []
for msg in recent:
role = msg.get("role")
content = msg.get("content", "")
if isinstance(content, list):
content = " ".join(
b.get("text", "") for b in content
if isinstance(b, dict) and b.get("type") == "text"
)
content = str(content or "").strip()
if content:
lines.append(f"{role}: {content}")
if not lines:
return
system_prompt = HEALTH_PERSONA_MEMORY_SYSTEM_PROMPT if schema == "health" else PERSONA_MEMORY_SYSTEM_PROMPT
raw = await llm_call_async(
endpoint_url,
model,
[
{"role": "system", "content": system_prompt},
{"role": "user", "content": (
f"Persona name: {character_name}\n\n"
f"Existing continuity notes:\n{existing_memory or '(none)'}\n\n"
"Recent transcript:\n"
+ "\n\n".join(lines)
+ "\n\nReturn only the updated continuity notes."
)},
],
temperature=0.1,
max_tokens=1200,
headers=headers,
)
updated = strip_think(str(raw or ""), prose=True, prompt_echo=True).strip()
updated = re.sub(r"^```(?:text|markdown)?\s*|\s*```$", "", updated, flags=re.I | re.S).strip()
if len(updated) > 6000:
updated = updated[:6000].rstrip()
if updated == existing_memory:
return
if preset_manager.update_persona_memory(character_name, updated):
logger.info("Updated persona memory for %s", character_name)
except Exception as e:
logger.warning("Persona memory update failed: %s", e)
async def audit_memories(
memory_manager,
memory_vector,
+6 -12
View File
@@ -28,6 +28,10 @@ SKILL_EXTRACT_PROMPT = (
"(personal errands, a specific person/place/date, casual conversation).\n"
"- A pure question/answer or explanation with no transferable method.\n"
"- The agent failed, gave up, or the approach is not worth repeating.\n\n"
"- Routine use of an existing tool, or a generic checklist with no new discovery.\n"
"Prefer a specific successful workaround, an unexpected pitfall, or a verified "
"sequence that would save rediscovery. Preserve exact useful commands and "
"verification steps, but replace private identifiers and credentials with placeholders.\n\n"
"When (and only when) a genuine reusable procedure exists, return a JSON "
"object with:\n"
'- "title": short name (under 10 words)\n'
@@ -259,19 +263,9 @@ async def maybe_extract_skill(
logger.debug("[skill-extract] '%s' already exists — dropped as duplicate", title)
return None
# Auto-publish gate: if the user has `auto_approve_skills` on, the
# newly-extracted skill is created `published` immediately rather
# than waiting for the next audit batch. The audit still runs later
# and can demote it back to `draft` (or delete) on failure. Default
# ON matches the UI label "Auto-approve skills".
# Automatic approval happens only after the audit has passed. A new
# extraction begins as a draft so it cannot enter chat context early.
_initial_status = "draft"
try:
from routes.prefs_routes import _load_for_user as _load_prefs
_prefs = _load_prefs(owner) or {}
if _prefs.get("auto_approve_skills", True):
_initial_status = "published"
except Exception:
pass
entry = skills_manager.add_skill(
title=title,
+20
View File
@@ -0,0 +1,20 @@
"""Bounded automatic review queue for user-owned procedural memory."""
import time
def automatic_audit_candidates(skills, limit=8, now=None):
"""Retry transient checks daily and failed repairs weekly, oldest first."""
now = time.time() if now is None else now
pending = []
for skill in skills:
if not skill.get("name") or skill.get("source") == "builtin" or skill.get("status") == "binned":
continue
verdict = skill.get("audit_verdict")
if verdict in {"pass", "skipped"}:
continue
checked = float(skill.get("audited_at") or 0)
delay = 7 * 86400 if verdict in {"fail", "needs_work"} else 86400
if not verdict or now - checked >= delay:
pending.append(skill)
pending.sort(key=lambda skill: float(skill.get("audited_at") or 0))
return pending[:max(1, limit)]
+157 -44
View File
@@ -54,6 +54,25 @@ def _to_float(x, default: float = 0.0) -> float:
return default
def _approval_policy(owner: Optional[str]) -> tuple[bool, float]:
"""Read the user's automatic skill-approval gate without breaking retrieval."""
try:
from routes.prefs_routes import _load_for_user
prefs = _load_for_user(owner) or {}
except Exception:
prefs = {}
try:
from src.settings import get_setting
default_minimum = float(get_setting("skill_autosave_min_confidence", 0.85))
except Exception:
default_minimum = 0.85
try:
minimum = float(prefs.get("skill_min_confidence", default_minimum))
except (TypeError, ValueError):
minimum = default_minimum
return bool(prefs.get("auto_approve_skills", True)), max(0.0, min(1.0, minimum))
# ---------------------------------------------------------------------------
# SkillsManager
# ---------------------------------------------------------------------------
@@ -120,7 +139,11 @@ class SkillsManager:
def set_audit(self, name: str, verdict: str, by_teacher: bool = False,
worker_model: str = "", teacher_model: str = "",
owner: Optional[str] = None) -> None:
owner: Optional[str] = None, saved_turns: Optional[int] = None,
saved_tool_calls: Optional[int] = None,
baseline_verdict: Optional[str] = None,
usefulness: Optional[float] = None,
audit_summary: Optional[str] = None) -> None:
"""Record the last test/audit result for a skill in the usage sidecar
(so it surfaces in load() without touching SKILL.md). Drives the
'verified' check + teacher mark on the card."""
@@ -129,11 +152,34 @@ class SkillsManager:
key = self._usage_key(name, owner)
e = usage.setdefault(key, {"uses": 0, "last_used": None})
e["audit_verdict"] = verdict
# Replace, rather than retain, the explanation from a previous run.
e["audit_summary"] = str(audit_summary or "")[:2000]
# Version 2 fixes audit-arm isolation and separates functional success
# from baseline utility. Legacy inconclusive results are not evidence
# under that protocol and should be eligible for a clean re-audit.
e["audit_version"] = 2
e["audit_by_teacher"] = bool(by_teacher)
if worker_model:
e["audit_worker_model"] = worker_model
if teacher_model:
e["audit_teacher_model"] = teacher_model
if saved_turns is not None:
try:
e["saved_turns"] = int(saved_turns)
except (TypeError, ValueError):
e.pop("saved_turns", None)
if saved_tool_calls is not None:
try:
e["saved_tool_calls"] = int(saved_tool_calls)
except (TypeError, ValueError):
e.pop("saved_tool_calls", None)
if baseline_verdict is not None:
e["baseline_verdict"] = str(baseline_verdict or "unknown")
if usefulness is not None:
try:
e["usefulness"] = float(usefulness)
except (TypeError, ValueError):
e.pop("usefulness", None)
e["audited_at"] = _t.time()
self._save_usage(usage)
@@ -197,6 +243,8 @@ class SkillsManager:
sk = self._read_skill(path)
if not sk:
continue
if sk.source == "builtin":
continue
owner = (sk.owner or "").strip()
if owner == primary_owner:
continue
@@ -227,11 +275,24 @@ class SkillsManager:
u = self._usage_entry(usage, sk.name, sk.owner)
d["uses"] = int(u.get("uses", 0))
d["last_used"] = u.get("last_used")
d["audit_verdict"] = u.get("audit_verdict")
audit_verdict = u.get("audit_verdict")
try:
audit_version = int(u.get("audit_version") or 0)
except (TypeError, ValueError):
audit_version = 0
if audit_verdict == "inconclusive" and audit_version < 2:
audit_verdict = None
d["audit_verdict"] = audit_verdict
d["audit_summary"] = u.get("audit_summary", "") if audit_verdict else ""
d["audit_version"] = audit_version
d["audit_by_teacher"] = bool(u.get("audit_by_teacher"))
d["audit_worker_model"] = u.get("audit_worker_model")
d["audit_teacher_model"] = u.get("audit_teacher_model")
d["audited_at"] = u.get("audited_at")
d["audited_at"] = u.get("audited_at") if audit_verdict else None
d["saved_turns"] = u.get("saved_turns")
d["saved_tool_calls"] = u.get("saved_tool_calls")
d["baseline_verdict"] = u.get("baseline_verdict")
d["usefulness"] = u.get("usefulness")
d["necessity"] = u.get("necessity")
out.append(d)
seen_names.add(sk.name)
@@ -284,7 +345,11 @@ class SkillsManager:
# leaked legacy / un-stamped skills to every authenticated user.
# Hide them now; the owner needs to be backfilled on disk if those
# skills should be visible to a specific user.
return [s for s in entries if s.get("owner") == owner]
return [
s for s in entries
if s.get("owner") == owner
or (s.get("source") == "builtin" and not s.get("owner"))
]
# ----------------------------------------------------------------------
# CRUD — disk-backed
@@ -546,7 +611,15 @@ class SkillsManager:
sk = self._read_skill(path)
if not sk or sk.name != name:
continue
if (sk.owner or "") != (owner or ""):
# Built-in skills are shared, ownerless procedures. ``load``
# exposes them to every owner, so direct progressive-disclosure
# reads must apply the same visibility rule as the index/list
# path. Previously a built-in appeared in `list` but `view`
# returned not-found for authenticated users.
if not (
(sk.owner or "") == (owner or "")
or (sk.source == "builtin" and not (sk.owner or ""))
):
continue
try:
with open(path, encoding="utf-8") as f:
@@ -562,7 +635,10 @@ class SkillsManager:
sk = self._read_skill(path)
if not sk or sk.name != name:
continue
if (sk.owner or "") != (owner or ""):
if not (
(sk.owner or "") == (owner or "")
or (sk.source == "builtin" and not (sk.owner or ""))
):
continue
base = os.path.realpath(os.path.dirname(path))
target = os.path.realpath(os.path.join(base, ref_path))
@@ -591,18 +667,12 @@ class SkillsManager:
"""Return the `[{name, description, category, status}]` list the
agent sees in its system prompt.
Includes:
- All published skills.
- Drafts written by the teacher-escalation loop
(`source == "teacher-escalation"`). The whole point of
the teacher loop is for the student to find the new
procedure on the very next turn — waiting for a manual
publish click defeats the loop.
Excludes user-created drafts (status=draft, source != teacher-
escalation) — those are work-in-progress and pollute the
prompt with half-finished procedures.
Includes built-ins plus user skills that have passed their audit and
meet the owner's current automatic-approval threshold. A persistent
``published`` flag is not sufficient: a changed threshold or a legacy
record must not make an unaudited skill eligible for prompt injection.
"""
auto_approve, min_confidence = _approval_policy(owner)
out = []
for s in self.load(owner=owner):
status = s.get("status")
@@ -613,6 +683,19 @@ class SkillsManager:
pass # let it through
else:
continue
# A stale published record must not remain injectable after an
# audit has recorded a failure. Inconclusive is not a failure.
audit_verdict = str(s.get("audit_verdict") or "").lower()
if audit_verdict in {"needs_work", "fail"}:
continue
if s.get("source") != "builtin" and auto_approve:
if status != "published" or audit_verdict != "pass":
continue
if _to_float(s.get("confidence"), 0.0) < min_confidence:
continue
necessity = s.get("necessity") or {}
if isinstance(necessity, dict) and necessity.get("necessary") is False:
continue
# Platform gating
if platform and s.get("platforms") and platform not in s["platforms"]:
continue
@@ -649,6 +732,8 @@ class SkillsManager:
threshold: float = 0.3,
max_items: int = 5,
min_confidence: float = 0.0,
available_toolsets: Optional[Iterable[str]] = None,
platform: Optional[str] = None,
) -> List[Dict]:
if skills is None:
skills = self.load_all()
@@ -660,37 +745,62 @@ class SkillsManager:
# without a manual publish click. The UI flags teacher-written
# entries with a 🎓 badge so users can demote / delete bad
# ones when they spot them.
skills = [s for s in skills if s.get("status") in ("published", "draft")]
# Confidence gate (used by prompt-injection, NOT by search): a DRAFT
# skill must clear the bar to be injected. Published skills are already
# vetted, so they always qualify. Missing confidence = treat as 1.0
# (legacy skills shouldn't silently vanish). 0 disables the gate.
skills = [
s for s in skills
if s.get("status") in ("published", "draft")
and str(s.get("audit_verdict") or "").lower()
not in {"needs_work", "fail", "skipped"}
]
available = set(available_toolsets) if available_toolsets is not None else None
if available is not None:
skills = [
skill for skill in skills
if all(tool in available for tool in (skill.get("requires_toolsets") or []))
and not any(tool in available for tool in (skill.get("fallback_for_toolsets") or []))
]
if platform:
skills = [
skill for skill in skills
if not skill.get("platforms") or platform in skill.get("platforms", [])
]
# Prompt injection is fail-closed for user skills. Built-ins are
# shipped procedures; every other skill needs a passing audit and a
# confidence score at the user's current threshold.
if min_confidence > 0:
def _passes(s):
if s.get("status") == "published":
if s.get("source") == "builtin":
return True
# Teacher-escalation drafts are auto-written from a (possibly
# untrusted) trace and injected as authoritative guidance, so they
# must EARN injection with an explicit, parseable confidence that
# clears the bar — fail closed on a missing/garbage value instead
# of treating it as 1.0. Hand-authored legacy drafts keep the
# lenient "unset → keep" behavior so they don't silently vanish.
if s.get("source") == "teacher-escalation":
c = s.get("confidence")
if c is None:
return False
return _to_float(c, 0.0) >= min_confidence # unparseable → fail closed
c = s.get("confidence")
if c is None:
return True # unset → don't filter (legacy)
return _to_float(c, 1.0) >= min_confidence # unparseable → pass
return (
s.get("status") == "published"
and str(s.get("audit_verdict") or "").lower() == "pass"
and _to_float(s.get("confidence"), 0.0) >= min_confidence
)
skills = [s for s in skills if _passes(s)]
if not skills:
return []
query_tokens = _tokenize(query)
semantic_scores: Dict[int, float] = {}
semantic_enabled = str(
os.environ.get("ODYSSEUS_SKILL_SEMANTIC_RETRIEVAL", "1")
).strip().lower() not in {"0", "false", "no", "off"}
if semantic_enabled:
try:
from src.skill_index import semantic_skill_scores
semantic_scores = semantic_skill_scores(query, skills)
except Exception as exc:
logger.debug("Semantic skill retrieval unavailable: %s", exc)
try:
semantic_threshold = float(
os.environ.get("ODYSSEUS_SKILL_SEMANTIC_THRESHOLD", "0.4")
)
except (TypeError, ValueError):
semantic_threshold = 0.4
semantic_threshold = max(-1.0, min(1.0, semantic_threshold))
scored = []
for sk in skills:
for position, sk in enumerate(skills):
text = " ".join([
sk.get("name", ""),
sk.get("description", ""),
@@ -698,19 +808,22 @@ class SkillsManager:
" ".join(sk.get("tags", []) or []),
" ".join(sk.get("procedure", []) or []),
])
score = _jaccard(query_tokens, _tokenize(text))
lexical_score = _jaccard(query_tokens, _tokenize(text))
for tag in sk.get("tags", []) or []:
# Match tags as whole tokens, not substrings: `tag in query`
# boosted e.g. a "ai" tag for any query containing "email".
tag_tokens = _tokenize(tag)
if tag_tokens and tag_tokens <= query_tokens:
score = max(score, 0.3) * 1.3
lexical_score = max(lexical_score, 0.3) * 1.3
if query.lower() in (sk.get("description") or "").lower():
score = max(score, 0.6)
lexical_score = max(lexical_score, 0.6)
semantic_score = semantic_scores.get(position, -1.0)
if lexical_score < threshold and semantic_score < semantic_threshold:
continue
score = max(lexical_score, semantic_score)
score *= 1.0 + _to_float(sk.get("confidence"), 0.5) * 0.1
if sk.get("uses", 0) > 0:
score *= 1.05
if score >= threshold:
scored.append((score, sk))
scored.append((score, sk))
scored.sort(key=lambda x: x[0], reverse=True)
return [sk for _, sk in scored[:max_items]]
+75 -18
View File
@@ -65,6 +65,49 @@ try:
except ImportError:
pdf_extract_text = None # type: ignore
try:
from pypdf import PdfReader
except ImportError:
PdfReader = None # type: ignore
def _extract_pdf_text(pdf_bytes: bytes, url: str = "") -> str:
"""Extract PDF text with available permissive dependencies."""
# Prefer pypdf's layout mode. Plain text extraction and pdfminer often
# collapse table columns into an ambiguous number stream, which makes a
# correct source passage easy for the model to misread.
if PdfReader is not None:
try:
reader = PdfReader(io.BytesIO(pdf_bytes))
pages: List[str] = []
for idx, page in enumerate(reader.pages):
try:
try:
page_text = page.extract_text(extraction_mode="layout") or ""
except TypeError:
page_text = page.extract_text() or ""
except Exception as e:
logger.warning(f"pypdf extraction failed for {url} page {idx + 1}: {e}")
page_text = ""
if page_text.strip():
pages.append(f"[Page {idx + 1}]\n{page_text.strip()}")
if pages:
return "\n\n".join(pages)
except Exception as e:
logger.warning(f"pypdf extraction failed for {url}: {e}")
if pdf_extract_text is not None:
try:
text = pdf_extract_text(io.BytesIO(pdf_bytes)) or ""
if text.strip():
return text
except Exception as e:
logger.warning(f"pdfminer extraction failed for {url}: {e}")
if PdfReader is None and pdf_extract_text is None:
logger.error("No PDF text extractor installed; install pdfminer.six or pypdf.")
return ""
# ----------------------------------------------------------------------
# HTML extraction helpers
@@ -216,9 +259,6 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
"User-Agent": WEB_FETCH_USER_AGENT,
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
"Accept-Language": "en-US,en;q=0.5",
# identity so the streamed size cap in _get_public_url stays honest
# (a compressed body can decode to far more than Content-Length).
"Accept-Encoding": "identity",
"Connection": "keep-alive",
}
response = _get_public_url(url, headers=headers, timeout=timeout,
@@ -252,26 +292,43 @@ def fetch_webpage_content(url: str, timeout: int = 5, retry_attempt: int = 0,
# PDF handling
content_type = response.headers.get("Content-Type", "").lower()
if "application/pdf" in content_type or url.lower().endswith(".pdf"):
if (
_size_fields["truncated"]
and effective_cap < WEB_FETCH_HARD_MAX_BYTES
and (
_size_fields["total_bytes"] is None
or _size_fields["total_bytes"] <= WEB_FETCH_HARD_MAX_BYTES
)
):
try:
response = _get_public_url(
url,
headers=headers,
timeout=timeout,
max_bytes=WEB_FETCH_HARD_MAX_BYTES,
)
_size_fields = {
"truncated": getattr(response, "truncated", False),
"fetched_bytes": len(response.content),
"total_bytes": getattr(response, "declared_bytes", None),
}
effective_cap = WEB_FETCH_HARD_MAX_BYTES
except BodyTooLargeError as e:
error_logger.warning(f"Refused oversized PDF body for {url}: {e}")
return _empty_result(url, f"TooLarge: {e}")
except Exception as e:
logger.warning(f"Full-budget PDF retry failed for {url}: {e}")
if _size_fields["truncated"]:
# A PDF cut mid-stream is not parseable; unlike text there is no
# useful partial result, so report the budget problem instead.
_declared = _size_fields["total_bytes"]
return _empty_result(
url,
f"TooLarge: PDF exceeds the {effective_cap:,}-byte fetch budget"
+ (f" (size {_declared:,} bytes)" if _declared else "")
+ "; retry with a larger budget if it fits under the hard cap",
error = (
f"TooLarge: PDF decoded body exceeded the {effective_cap:,}-byte fetch budget"
+ (f" (declared compressed size {_declared:,} bytes)" if _declared else "")
+ "; retry with a larger budget if it fits under the hard cap"
)
if pdf_extract_text is None:
logger.error("pdfminer.six is not installed; cannot extract PDF text.")
pdf_text = ""
else:
try:
pdf_bytes = io.BytesIO(response.content)
pdf_text = pdf_extract_text(pdf_bytes)
except Exception as e:
logger.warning(f"PDF extraction failed for {url}: {e}")
pdf_text = ""
return {**_empty_result(url, error), **_size_fields}
pdf_text = _extract_pdf_text(response.content, url)
result = {
"url": url,
"title": os.path.basename(url),
+538 -13
View File
@@ -2,11 +2,15 @@
import json
import logging
import re
import xml.etree.ElementTree as ET
from concurrent.futures import ThreadPoolExecutor, as_completed
from datetime import datetime, timedelta
from typing import Dict, Any, Optional, List, Set
from urllib.parse import urlparse
import httpx
from .analytics import (
NetworkError,
ParseError,
@@ -97,6 +101,8 @@ def _call_provider(provider_name: str, query: str, count: int, time_filter: str
"""Call a search provider by name. Returns list of results or empty list."""
if provider_name == "searxng":
return searxng_search_api(query, count, time_filter=time_filter)
elif provider_name == "searxng_yep":
return searxng_search_api(query, count, time_filter=time_filter, engines="yep")
elif provider_name == "brave":
return brave_search(query, count, time_filter)
elif provider_name == "duckduckgo":
@@ -127,7 +133,484 @@ def _build_provider_chain(primary: str) -> List[str]:
for fb in fallbacks:
if fb and fb != primary and fb not in chain and fb != "disabled":
chain.append(fb)
return chain
from .providers import provider_configured
configured = [provider for provider in chain if provider_configured(provider)]
for provider in set(chain) - set(configured):
logger.warning("Skipping unconfigured search provider: %s", provider)
if primary == "searxng" and configured == ["searxng"]:
# No usable configured fallback: try a separate engine on the same
# private metasearch instance before reporting retrieval failure.
configured.append("searxng_yep")
return configured
_SEARCH_QUERY_FILLER = {
"what", "whats", "what's", "which", "when", "where", "year", "from",
"any", "info", "information", "details", "update", "updates",
"with", "this", "that", "search", "lookup", "look", "find", "tell",
"about", "quick", "please", "pls", "official", "links", "source",
"sources", "news", "headlines", "breaking", "latest", "current",
"newest", "recent", "today", "now",
"release", "releases", "version", "versions", "changelog", "github",
"gitlab", "weather", "forecast", "forecasts", "tomorrow", "hourly",
"daily", "temperature", "temperatures", "conditions", "rain", "raining",
"chance", "precipitation",
"january", "february", "march", "april", "may", "june", "july",
"august", "september", "october", "november", "december",
"the", "and", "or", "but", "are", "was", "were", "does", "did",
"can", "could", "should", "would", "will", "has", "have", "had",
"for", "into", "onto", "near", "over", "under",
}
_SHORT_QUERY_SUBJECTS = {"ai", "ar", "eu", "uk", "us", "vr"}
_WEATHER_QUERY_HINTS = {
"weather", "forecast", "forecasts", "temperature", "temperatures",
"rain", "raining", "precipitation", "humid", "humidity", "wind",
}
_WEATHER_RESULT_HINTS = {
"weather", "forecast", "temperature", "temperatures", "rain",
"precipitation", "humidity", "wind", "accuweather", "meteoblue",
"weather-atlas", "weather25", "weather365", "easeweather",
}
def _meaningful_query_terms(query: str) -> list[str]:
return [
term
for term in re.findall(r"[a-z0-9]+", str(query or "").lower())
if (len(term) > 2 or term in _SHORT_QUERY_SUBJECTS)
and not term.isdigit()
and term not in _SEARCH_QUERY_FILLER
]
def _result_has_query_overlap(query: str, result: dict) -> bool:
terms = _meaningful_query_terms(query)
if not terms:
return True
text = " ".join(
str(result.get(key) or "").lower()
for key in ("title", "snippet", "url")
)
query_tokens = set(re.findall(r"[a-z0-9]+", str(query or "").lower()))
if query_tokens & _WEATHER_QUERY_HINTS:
return (
any(re.search(rf"\b{re.escape(term)}\b", text) for term in terms)
and any(marker in text for marker in _WEATHER_RESULT_HINTS)
)
result_tokens = set(re.findall(r"[a-z0-9]+", text))
def lexical_root(word: str) -> str:
for suffix in ("ation", "ition", "ence", "ance", "ment", "ents", "ent", "ant", "ing", "ed", "es", "s"):
if word.endswith(suffix) and len(word) - len(suffix) >= 6:
return word[:-len(suffix)]
return word
result_roots = {lexical_root(token) for token in result_tokens}
matched_terms = {
term for term in terms
if term in result_tokens or lexical_root(term) in result_roots
}
# A single broad token is not enough evidence for a detailed entity/event
# query. For example, SearXNG may answer "Sweden 78 year old British woman
# deportation Brexit ..." with generic Sweden tourism pages. Treat that as
# an empty provider result so the configured fallback gets a chance.
minimum_matches = 2 if len(set(terms)) >= 4 else 1
return len(matched_terms) >= minimum_matches
def _filter_low_relevance_results(query: str, results: list[dict]) -> list[dict]:
if not results:
return []
relevant = [result for result in results if _result_has_query_overlap(query, result)]
# Only reject a provider when it returned a fully off-topic page set. Mixed
# result pages are common; ranking can handle those.
return relevant if relevant else []
_SCHOLARLY_QUERY_CUE_RE = re.compile(
r"\b(?:paper|preprint|arxiv|proceedings|table\s+\d+|figure\s+\d+|"
r"appendix\s+[a-z0-9]+|benchmark(?:s)?)\b",
re.IGNORECASE,
)
_SCHOLARLY_TITLE_FILLER = _SEARCH_QUERY_FILLER | {
"paper", "preprint", "arxiv", "proceedings", "table", "figure",
"appendix", "authors", "author", "extract", "locate", "read",
}
_ARXIV_IDENTIFIER_RE = re.compile(
r"(?i)(?:\barxiv\s*:\s*|\barxiv\.org/(?:abs|pdf|html)/)?"
r"(?P<identifier>\d{4}\.\d{4,5}(?:v\d+)?)\b"
)
_FORMAL_PUBLICATION_CUE_RE = re.compile(
r"\b(?:publish(?:ed|ing|cation)?|venue|conference|journal|proceedings|doi)\b",
re.IGNORECASE,
)
def _exact_arxiv_identifier_results(query: str) -> list[dict]:
"""Return deterministic official landing pages for explicit arXiv IDs."""
seen: set[str] = set()
results: list[dict] = []
for match in _ARXIV_IDENTIFIER_RE.finditer(str(query or "")):
identifier = match.group("identifier")
canonical = re.sub(r"v\d+$", "", identifier, flags=re.IGNORECASE)
if canonical in seen:
continue
seen.add(canonical)
results.append({
"title": f"arXiv:{canonical} — exact identifier match",
"url": f"https://arxiv.org/abs/{canonical}",
"snippet": (
"Official arXiv landing page resolved directly from the exact "
"identifier in the query."
),
"source": "arxiv",
})
return results
def _title_before_explicit_arxiv_identifier(query: str) -> str:
"""Extract a probable title that precedes an explicit arXiv identifier."""
text = re.sub(r"\s+", " ", str(query or "")).strip()
match = _ARXIV_IDENTIFIER_RE.search(text)
if not match or not _FORMAL_PUBLICATION_CUE_RE.search(text):
return ""
candidate = text[:match.start()].strip(" \t,;:-'\"")
candidate = re.sub(
r"\barxiv(?:\.org)?(?:\s*:\s*|\s+(?:abs|pdf|html)\s*[/ :]*)?$",
"",
candidate,
flags=re.IGNORECASE,
).strip(" \t,;:-'\"")
candidate = re.sub(
r"^(?:(?:please\s+)?(?:find|locate|search\s+for|look\s+up|verify|check)\s+)"
r"(?:(?:the|this)\s+)?(?:paper\s+)?",
"",
candidate,
flags=re.IGNORECASE,
).strip(" \t,;:-'\"")
return candidate if len(_normalized_title_terms(candidate)) >= 2 else ""
def _normalized_title_terms(value: str) -> list[str]:
return [
token
for token in re.findall(r"[a-z0-9]+", str(value or "").lower())
if len(token) > 1 and token not in _SCHOLARLY_TITLE_FILLER
]
def _is_distinctive_short_scholarly_title(value: str) -> bool:
"""Recognize compact model/report names without accepting generic phrases."""
terms = _normalized_title_terms(value)
if not 1 <= len(terms) <= 2:
return False
text = str(value or "").strip()
return bool(
re.search(r"\d", text)
or re.search(r"\b[A-Z][A-Za-z0-9]*-[A-Z][A-Za-z0-9]*\b", text)
)
def _scholarly_title_from_query(query: str) -> str:
"""Extract a probable paper title only from clearly scholarly searches."""
text = re.sub(r"\s+", " ", str(query or "")).strip()
if not text or not _SCHOLARLY_QUERY_CUE_RE.search(text):
return ""
quoted = [
candidate.strip()
for candidate in re.findall(r'["“”]([^"“”]{4,180})["“”]', text)
if len(_normalized_title_terms(candidate)) >= 3
or _is_distinctive_short_scholarly_title(candidate)
]
if quoted:
return max(quoted, key=lambda candidate: len(_normalized_title_terms(candidate)))
before_paper = re.search(
r"(?:^|\b(?:find|locate|read|from|about)\s+)(.{4,160}?)\s+"
r"(?:paper|preprint)\b",
text,
re.IGNORECASE,
)
if before_paper:
candidate = before_paper.group(1).strip(" ,:;-'")
if (
len(_normalized_title_terms(candidate)) >= 3
or _is_distinctive_short_scholarly_title(candidate)
):
return candidate
before_locator = re.match(
r"(.{2,80}?)\s+(?:table|figure)\s+\d+\b",
text,
re.IGNORECASE,
)
if before_locator:
candidate = before_locator.group(1).strip(" ,:;-'\"")
if _is_distinctive_short_scholarly_title(candidate):
return candidate
return ""
def _result_strongly_matches_title(title: str, result: dict) -> bool:
wanted = set(_normalized_title_terms(title))
found = set(_normalized_title_terms(str(result.get("title") or "")))
if len(wanted) < 2 or not found:
return False
overlap = len(wanted & found) / len(wanted)
return overlap >= (1.0 if len(wanted) == 2 else 0.8)
def _arxiv_title_results(title: str, count: int = 3) -> list[dict]:
"""Resolve a paper title through arXiv's public Atom API."""
try:
response = httpx.get(
"https://export.arxiv.org/api/query",
params={
"search_query": f'ti:"{title}"',
"start": 0,
"max_results": max(1, min(int(count), 5)),
},
headers={"User-Agent": "Odysseus/0.20 scholarly-title-resolver"},
timeout=12.0,
follow_redirects=True,
)
response.raise_for_status()
root = ET.fromstring(response.text)
except Exception as exc:
logger.info("arXiv title lookup failed for %r: %s", title, exc)
return []
namespace = {"atom": "http://www.w3.org/2005/Atom"}
matches: list[dict] = []
for entry in root.findall("atom:entry", namespace):
result_title = " ".join(
(entry.findtext("atom:title", default="", namespaces=namespace) or "").split()
)
if not _result_strongly_matches_title(title, {"title": result_title}):
continue
entry_id = (entry.findtext("atom:id", default="", namespaces=namespace) or "").strip()
arxiv_id = entry_id.rstrip("/").rsplit("/", 1)[-1]
if not arxiv_id:
continue
summary = " ".join(
(entry.findtext("atom:summary", default="", namespaces=namespace) or "").split()
)
matches.append({
"title": result_title,
"url": f"https://arxiv.org/abs/{arxiv_id}",
"snippet": summary,
"source": "arxiv",
})
return matches
def _openalex_title_results(title: str, count: int = 3) -> list[dict]:
"""Resolve an exact scholarly title through OpenAlex metadata."""
try:
# OpenAlex treats a literal question mark as query syntax and returns
# HTTP 400 for otherwise valid titles such as "How Far ... GPT-4V?".
search_title = re.sub(r"[?]+", " ", str(title or "")).strip()
response = httpx.get(
"https://api.openalex.org/works",
params={
"search": search_title,
"per-page": max(1, min(int(count), 5)),
"select": (
"display_name,doi,primary_location,publication_year,type"
),
},
headers={"User-Agent": "Odysseus/0.20 scholarly-title-resolver"},
timeout=12.0,
follow_redirects=True,
)
response.raise_for_status()
payload = response.json()
except Exception as exc:
logger.info("OpenAlex title lookup failed for %r: %s", title, exc)
return []
matches: list[dict] = []
for item in payload.get("results", []):
result_title = str(item.get("display_name") or "").strip()
if not _result_strongly_matches_title(title, {"title": result_title}):
continue
location = item.get("primary_location") or {}
url = str(location.get("landing_page_url") or item.get("doi") or "").strip()
if url.startswith("http://arxiv.org/"):
url = "https://" + url[len("http://"):]
if not url:
continue
snippet = "Exact scholarly-title match from OpenAlex metadata."
venue = str(location.get("raw_source_name") or "").strip()
year = item.get("publication_year")
publication_type = str(item.get("type") or "").strip()
version = str(location.get("version") or "").strip()
formal_parts: list[str] = []
if venue:
formal_parts.append(f"{venue}, {year}" if year else venue)
elif year:
formal_parts.append(str(year))
if publication_type:
formal_parts.append(f"type: {publication_type}")
if version:
formal_parts.append(f"version: {version}")
if formal_parts:
snippet += f" Formal publication: {'; '.join(formal_parts)}."
matches.append({
"title": result_title,
"url": url,
"snippet": snippet,
"source": "openalex",
})
return matches
def _scholarly_title_results(title: str, count: int = 3) -> list[dict]:
"""Retry a noisy scholarly query as a bare title, then use arXiv API."""
try:
simplified = searxng_search_api(title, count=max(3, count))
except Exception as exc:
logger.info("Simplified scholarly search failed for %r: %s", title, exc)
simplified = []
exact = [
result for result in simplified
if _result_strongly_matches_title(title, result)
]
if exact:
return exact[:count]
openalex = _openalex_title_results(title, count)
if openalex:
return openalex
return _arxiv_title_results(title, count)
def _direct_scholarly_title_results(title: str, count: int = 3) -> list[dict]:
"""Resolve a clear paper title without waiting on generic search providers."""
# OpenAlex typically resolves titles in under a second and often returns
# the official arXiv landing page. The arXiv API remains the fallback.
openalex = _openalex_title_results(title, count)
if openalex:
return openalex
return _arxiv_title_results(title, count)
def _augment_scholarly_results(query: str, results: list[dict], count: int) -> list[dict]:
"""Prepend an exact arXiv match when a scholarly SERP missed its title."""
current = list(results or [])
identifier_results = _exact_arxiv_identifier_results(query)
if identifier_results:
title = _title_before_explicit_arxiv_identifier(query)
formal_results: list[dict] = []
if title:
formal_results = [
item
for item in _openalex_title_results(title, min(count, 3))
if "arxiv.org/" not in str(item.get("url") or "").lower()
]
exact_urls = {str(item["url"]) for item in identifier_results}
formal_urls = {str(item.get("url") or "") for item in formal_results}
return (
formal_results
+ identifier_results
+ [
item for item in current
if str(item.get("url") or "") not in exact_urls | formal_urls
]
)[:count]
title = _scholarly_title_from_query(query)
if not title:
return current
exact_current = [
item for item in current
if _result_strongly_matches_title(title, item)
]
if exact_current:
exact_ids = {id(item) for item in exact_current}
return (exact_current + [item for item in current if id(item) not in exact_ids])[:count]
arxiv_results = _scholarly_title_results(title, min(count, 3))
if not arxiv_results:
return current
seen = {str(item.get("url") or "") for item in arxiv_results}
return (arxiv_results + [item for item in current if str(item.get("url") or "") not in seen])[:count]
def _subject_first_weather_query(query: str) -> str:
"""Rewrite natural weather questions into the shape SearXNG handles best."""
text = re.sub(r"\s+", " ", str(query or "")).strip(" ?")
if not text:
return text
if not (set(re.findall(r"[a-z0-9]+", text.lower())) & _WEATHER_QUERY_HINTS):
return text
loc_match = re.search(
r"\b(?:weather|forecast)\s+(?:in|for|at)\s+(.+)$",
text,
re.IGNORECASE,
)
if not loc_match:
loc_match = re.search(
r"\b(?:weather|forecast)\b.*?\b(?:in|for|at)\s+(.+)$",
text,
re.IGNORECASE,
)
if not loc_match:
return text
location = loc_match.group(1).strip(" ?.,")
timing = ""
timing_match = re.search(
r"\b(today|tomorrow|tonight|this\s+week|next\s+week|now|current)\b",
location,
re.IGNORECASE,
)
if timing_match:
timing = timing_match.group(1).lower()
location = (
location[: timing_match.start()] + location[timing_match.end():]
).strip(" ?.,")
if not location:
return text
return re.sub(r"\s+", " ", f"{location} weather forecast {timing}").strip()
def _provider_friendly_query(query: str) -> str:
"""Convert generic question grammar to keyword order without changing its topic."""
text = _subject_first_weather_query(query)
match = re.fullmatch(
r"(?:what|which)\s+(year|date|time)\s+(?:did|does|do|was|were|is|are)\s+(.+)",
text,
re.IGNORECASE,
)
if match:
return f"{match.group(2).strip()} {match.group(1).lower()}"
# Search providers already receive recency separately. Remove a leading
# conversational request shell so ranking is driven by the subject rather
# than words such as "any", "latest", and "information".
cleaned = re.sub(
r"^(?:can|could|would)\s+you\s+(?:find|search|look\s+up)\s+",
"",
text,
flags=re.IGNORECASE,
)
cleaned = re.sub(
r"^(?:any\s+)?(?:latest|current|recent)?\s*"
r"(?:news|info(?:rmation)?|updates?|details?)\s+(?:on|about)\s+",
"",
cleaned,
flags=re.IGNORECASE,
)
if cleaned.strip():
return cleaned.strip()
return text
# ----------------------------------------------------------------------
@@ -135,6 +618,7 @@ def _build_provider_chain(primary: str) -> List[str]:
# ----------------------------------------------------------------------
def searxng_search_results(query: str, count: int = 10, time_filter: str = None) -> list[dict]:
"""Perform a web search using configured provider with caching and retry."""
provider_query = _provider_friendly_query(query)
settings = _get_search_settings()
search_provider = settings.get("search_provider", "searxng")
result_count = _get_result_count()
@@ -142,7 +626,17 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None)
if count == 10:
count = result_count
cache_key = generate_cache_key(f"{query}|{count}|{time_filter}")
# A named scholarly work has a deterministic metadata path. Resolve that
# first instead of spending the full tool deadline retrying generic search
# providers; the returned official URL lets the agent proceed to PDF tools.
scholarly_title = _scholarly_title_from_query(provider_query)
if scholarly_title:
direct_results = _direct_scholarly_title_results(scholarly_title, count)
if direct_results:
_record_query(provider_query, True, cache_hit=False)
return direct_results[:count]
cache_key = generate_cache_key(f"{provider_query}|{count}|{time_filter}")
cache_file = SEARCH_CACHE_DIR / f"{cache_key}.cache"
# Check cache
@@ -155,8 +649,22 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None)
if expiry and datetime.now() < expiry:
logger.debug(f"Search cache hit for query: {query}")
results = cached_data["data"]
_record_query(query, bool(results), cache_hit=True)
return results
# Ranking/relevance logic evolves independently from provider
# results. Re-apply it on cache hits so stale cached ordering
# does not preserve bad SERP choices after a harness fix.
results = _filter_low_relevance_results(provider_query, results)
if results:
results = rank_search_results(provider_query, results)
results = _augment_scholarly_results(provider_query, results, count)
if results:
_record_query(query, True, cache_hit=True)
return results
logger.info(
"Search cache hit for %r became empty after relevance filtering; refetching",
provider_query,
)
cache_file.unlink(missing_ok=True)
search_cache_index.pop(cache_key, None)
else:
cache_file.unlink(missing_ok=True)
search_cache_index.pop(cache_key, None)
@@ -178,7 +686,8 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None)
for attempt in range(2):
try:
logger.info(f"Attempting {provider_name} search (attempt {attempt + 1})")
results = _call_provider(provider_name, query, count, time_filter)
results = _call_provider(provider_name, provider_query, count, time_filter)
results = _filter_low_relevance_results(provider_query, results)
if results:
logger.info(f"{provider_name} search succeeded with {len(results)} results")
break
@@ -189,11 +698,14 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None)
if results:
break
results = _augment_scholarly_results(provider_query, results, count)
success = bool(results)
_record_query(query, success, cache_hit=False)
_record_query(provider_query, success, cache_hit=False)
if success:
results = rank_search_results(query, results)
results = rank_search_results(provider_query, results)
results = _augment_scholarly_results(provider_query, results, count)
try:
expiry = datetime.now() + _cache_duration_for_query(query)
cache_data = {
@@ -206,10 +718,10 @@ def searxng_search_results(query: str, count: int = 10, time_filter: str = None)
search_cache_index[cache_key] = datetime.now()
cleanup_cache(SEARCH_CACHE_DIR, search_cache_index, timedelta(hours=1))
except Exception as e:
logger.warning(f"Failed to write search cache for {query}: {e}")
logger.warning(f"Failed to write search cache for {provider_query}: {e}")
if not success:
logger.error(f"All search providers failed for query: {query}")
logger.error(f"All search providers failed for query: {provider_query}")
return results
@@ -260,7 +772,8 @@ def comprehensive_web_search(
return_sources: bool = False,
):
"""Perform comprehensive web search with content fetching and advanced filtering."""
logger.info(f"Starting comprehensive search for: {query}")
provider_query = _provider_friendly_query(query)
logger.info(f"Starting comprehensive search for: {provider_query}")
if time_filter:
logger.info(f"Applying time filter: {time_filter}")
@@ -285,7 +798,8 @@ def comprehensive_web_search(
empty = False
for attempt in range(2):
try:
search_results = _call_provider(provider_name, query, fetch_count, time_filter)
search_results = _call_provider(provider_name, provider_query, fetch_count, time_filter)
search_results = _filter_low_relevance_results(provider_query, search_results)
if search_results:
provider_attempts[provider_name] = f"ok ({len(search_results)})"
logger.info(f"Comprehensive search: {provider_name} returned {len(search_results)} results")
@@ -301,6 +815,12 @@ def comprehensive_web_search(
elif empty:
provider_attempts[provider_name] = "empty"
search_results = _augment_scholarly_results(
provider_query,
search_results,
fetch_count,
)
if not search_results:
tally = ", ".join(f"{p}:{r}" for p, r in provider_attempts.items()) or "no providers configured"
any_errors = any(r.startswith("error") for r in provider_attempts.values())
@@ -315,7 +835,12 @@ def comprehensive_web_search(
logger.warning(msg)
return (msg, []) if return_sources else msg
search_results = rank_search_results(query, search_results)
search_results = rank_search_results(provider_query, search_results)
search_results = _augment_scholarly_results(
provider_query,
search_results,
fetch_count,
)
# URL filter helper
def url_passes_filters(url: str) -> bool:
@@ -399,7 +924,7 @@ def comprehensive_web_search(
output_parts.append("=" * 70)
output_parts.append("WEB SEARCH RESULTS AND FETCHED CONTENT")
output_parts.append(f"Query: {query}")
output_parts.append(f"Query: {provider_query}")
output_parts.append(f"Searched {len(search_results)} results, fetched {len(fetched_content)} pages")
output_parts.append("=" * 70)
output_parts.append("")
+73 -10
View File
@@ -3,6 +3,7 @@
import json
import logging
import os
import re
from typing import List, Optional
from urllib.parse import urljoin, urlparse, parse_qs
@@ -33,9 +34,16 @@ def _get_search_settings() -> dict:
"""Return search settings from admin config, falling back to env defaults."""
try:
from src.settings import load_settings
return load_settings()
settings = dict(load_settings())
except Exception:
return {}
settings = {}
# Headless/native deployments do not necessarily have an admin settings
# database. Require an explicit Odysseus-prefixed override so ordinary UI
# configuration remains authoritative by default.
env_provider = os.environ.get("ODYSSEUS_SEARCH_PROVIDER", "").strip().lower()
if env_provider:
settings["search_provider"] = env_provider
return settings
def _get_search_instance() -> str:
@@ -66,13 +74,18 @@ def _get_provider_key(provider: str) -> str:
if legacy:
return legacy
env_map = {
"brave": "DATA_BRAVE_API_KEY",
"google_pse": "GOOGLE_API_KEY",
"tavily": "TAVILY_API_KEY",
"serper": "SERPER_API_KEY",
# DATA_BRAVE_API_KEY is the historical Odysseus name; BRAVE_API_KEY is
# the standard name used by headless runners and the Brave SDK.
"brave": ("DATA_BRAVE_API_KEY", "BRAVE_API_KEY"),
"google_pse": ("GOOGLE_API_KEY",),
"tavily": ("TAVILY_API_KEY",),
"serper": ("SERPER_API_KEY",),
}
env_name = env_map.get(provider, "")
return (os.environ.get(env_name) or "").strip() if env_name else ""
for env_name in env_map.get(provider, ()):
value = (os.environ.get(env_name) or "").strip()
if value:
return value
return ""
def _get_result_count() -> int:
@@ -84,6 +97,19 @@ def _get_result_count() -> int:
return 5
def provider_configured(provider: str) -> bool:
"""Configuration readiness only; a configured engine can still fail upstream."""
if provider in {"searxng", "searxng_yep", "duckduckgo"}:
return True
if provider not in {"brave", "google_pse", "tavily", "serper"}:
return False
if not _get_provider_key(provider):
return False
if provider == "google_pse":
return bool(_get_search_settings().get("google_pse_cx") or os.environ.get("GOOGLE_PSE_CX"))
return True
# Canonical SafeSearch levels: "strict" (default), "moderate", "off".
# Each provider has its own knob name and value space -- see _safesearch_for(...).
_SAFESEARCH_LEVELS = ("strict", "moderate", "off")
@@ -124,6 +150,24 @@ def _safesearch_for(provider: str) -> Optional[str]:
# ── SearXNG ──
_NEWS_HINTS = ("news", "nyheter", "headlines", "breaking", "latest", "today", "idag")
_NEWS_EVENT_HINT_RE = re.compile(
r"\b(?:deport(?:ation|ed|ing)?|arrest(?:ed|s)?|election(?:s)?|"
r"evacuat(?:e|ed|ion)|flood(?:ing|s|ed)?|sanction(?:s|ed)?)\b",
re.IGNORECASE,
)
_SOFTWARE_RELEASE_HINTS = (
"github",
"gitlab",
"release",
"releases",
"version",
"versions",
"changelog",
"change log",
"pypi",
"npm",
"package",
)
# Default general engines (google/duckduckgo/brave/startpage/wikipedia) are
# routinely rate-limited / CAPTCHA-blocked on this instance and return nothing.
@@ -133,7 +177,7 @@ _GENERAL_ENGINES = os.environ.get("SEARXNG_GENERAL_ENGINES", "bing,mojeek,presea
def searxng_search_api(query: str, count: Optional[int] = None, categories: str = "general",
time_filter: Optional[str] = None) -> List[dict]:
time_filter: Optional[str] = None, *, engines: Optional[str] = None) -> List[dict]:
"""Search using SearXNG JSON API. Returns list of {title, url, snippet}."""
count = count if count is not None else _get_result_count()
instance = _get_search_instance()
@@ -158,7 +202,19 @@ def searxng_search_api(query: str, count: Optional[int] = None, categories: str
"safesearch": _safesearch_for("searxng"),
}
q_lc = query.lower()
is_news = time_filter is not None or any(h in q_lc for h in _NEWS_HINTS)
# Fresh software-version queries are usually better served by general
# search or canonical project pages than by the news vertical. For example
# "latest ollama release version github" can return a sparse news result
# that gets filtered as irrelevant, while general engines find GitHub.
is_software_release_query = any(h in q_lc for h in _SOFTWARE_RELEASE_HINTS)
is_news = (
not is_software_release_query
and (
time_filter is not None
or any(h in q_lc for h in _NEWS_HINTS)
or bool(_NEWS_EVENT_HINT_RE.search(query))
)
)
if is_news and categories == "general":
params["categories"] = "news"
if time_filter in ("day", "week", "month", "year"):
@@ -171,6 +227,9 @@ def searxng_search_api(query: str, count: Optional[int] = None, categories: str
# set returns 0 on this instance — see _GENERAL_ENGINES).
if categories == "general" and _GENERAL_ENGINES:
params["engines"] = _GENERAL_ENGINES
if engines:
params["categories"] = "general"
params["engines"] = engines
try:
def _parse_results(results):
return [
@@ -178,6 +237,10 @@ def searxng_search_api(query: str, count: Optional[int] = None, categories: str
"title": r.get("title", ""),
"url": r.get("url", ""),
"snippet": r.get("content", ""),
"provider": "searxng",
"engines": r.get("engines", []),
"published_date": r.get("publishedDate"),
"query": query,
}
for r in results[:count]
if r.get("url")
+55
View File
@@ -67,6 +67,22 @@ _TRUSTED_NEWS_DOMAINS = {
"www.theguardian.com", "euronews.com", "www.euronews.com",
"dw.com", "www.dw.com", "government.se", "www.government.se",
}
_SOFTWARE_RELEASE_HINTS = {
"github", "gitlab", "release", "releases", "version", "versions",
"changelog", "package", "pypi", "npm",
}
_PRODUCT_SPEC_HINTS = {
"product", "hardware", "device", "phone", "laptop", "desktop", "computer",
"chip", "cpu", "gpu", "mac", "iphone", "ipad", "android", "camera",
"console", "kindle", "tesla", "car", "model", "price", "pricing", "cost",
"buy", "shop", "order", "preorder", "pre-order", "spec", "specs",
"specifications", "available", "availability", "ship", "shipping",
"released", "launch", "launched", "vram", "memory", "ram", "storage",
}
_COMMERCE_OR_SPEC_PATH_HINTS = (
"/shop", "/buy", "/store", "/product", "/products", "/spec", "/specs",
"/support", "/tech-specs", "/technical-specifications",
)
def _domain(url: str) -> str:
@@ -95,6 +111,8 @@ def rank_search_results(query: str, results: List[dict]) -> List[dict]:
query_lc = query.lower()
is_news_query = any(term in _NEWS_HINTS for term in query_terms)
is_sports_query = bool(_SPORTS_HINT_RE.search(query_lc))
is_software_release_query = any(term in _SOFTWARE_RELEASE_HINTS for term in query_terms)
is_product_spec_query = any(term in _PRODUCT_SPEC_HINTS for term in query_terms)
def title_score(title: str) -> float:
if not title:
@@ -144,6 +162,41 @@ def rank_search_results(query: str, results: List[dict]) -> List[dict]:
adjustment -= 1.0
return adjustment
def software_release_adjustment(title: str, snippet: str, url: str) -> float:
if not is_software_release_query:
return 0.0
netloc = _domain(url)
path = urlparse(url).path.lower()
text = f"{title} {snippet} {netloc} {path}".lower()
adjustment = 0.0
if netloc in {"github.com", "www.github.com", "gitlab.com", "www.gitlab.com"}:
adjustment += 1.6
if "/releases" in path or "/tags" in path:
adjustment += 1.2
if any(_has_word(text, term) for term in ("release", "releases", "changelog", "version")):
adjustment += 0.4
if netloc in {"releasealert.dev", "releases.sh", "releasebot.io"}:
adjustment -= 0.8
return adjustment
def product_spec_adjustment(title: str, snippet: str, url: str) -> float:
if not is_product_spec_query:
return 0.0
parsed = urlparse(url)
netloc = parsed.netloc.lower()
path = parsed.path.lower()
text = f"{title} {snippet} {netloc} {path}".lower()
adjustment = 0.0
if any(hint in path for hint in _COMMERCE_OR_SPEC_PATH_HINTS):
adjustment += 1.1
if re.search(r"\b(?:official|specs?|specifications|tech specs|buy|shop|store|price|pricing|available|ships?)\b", text):
adjustment += 0.5
if netloc.endswith(".com") and any(_has_word(netloc, term) for term in query_terms if len(term) >= 4):
adjustment += 0.4
if re.search(r"\b(?:rumor|rumour|leak|may|could|expected|reportedly|unannounced)\b", text):
adjustment -= 0.8
return adjustment
ranked = []
for result in results:
title = result.get("title", "")
@@ -157,6 +210,8 @@ def rank_search_results(query: str, results: List[dict]) -> List[dict]:
+ 1.5 * domain_score(url)
+ 1.0 * recency_score(age)
+ news_quality_adjustment(title, snippet, url)
+ software_release_adjustment(title, snippet, url)
+ product_spec_adjustment(title, snippet, url)
)
ranked.append((score, result))