mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-09 08:22:19 +02:00
Squash Odysseus development history
This commit is contained in:
@@ -0,0 +1,74 @@
|
||||
"""Install tracked built-in skills into the shared immutable skill catalog."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Iterable
|
||||
|
||||
from .skill_format import Skill
|
||||
from .skills import SkillsManager
|
||||
|
||||
|
||||
_BUILTIN_ROOT = Path(__file__).resolve().parents[2] / "resources" / "skills"
|
||||
_SYNC_FIELDS = (
|
||||
"name",
|
||||
"description",
|
||||
"version",
|
||||
"category",
|
||||
"tags",
|
||||
"status",
|
||||
"confidence",
|
||||
"source",
|
||||
"owner",
|
||||
"when_to_use",
|
||||
"procedure",
|
||||
"pitfalls",
|
||||
"verification",
|
||||
"platforms",
|
||||
"requires_toolsets",
|
||||
"fallback_for_toolsets",
|
||||
"body_extra",
|
||||
)
|
||||
|
||||
|
||||
def install_builtin_skills(manager: SkillsManager, owners: Iterable[str]) -> int:
|
||||
"""Copy missing built-in skills into the ownerless shared catalog.
|
||||
|
||||
Built-ins are explicitly marked and remain ownerless because the on-disk
|
||||
skill path is not owner-qualified. ``SkillsManager.load(owner=...)``
|
||||
exposes only these immutable built-ins in addition to that owner's files.
|
||||
Installation is safe before first-user setup because no owner identity is
|
||||
assigned and unauthenticated requests still cannot access skill routes.
|
||||
"""
|
||||
existing = {row.get("name") for row in manager.load_all()}
|
||||
installed = 0
|
||||
paths = sorted(_BUILTIN_ROOT.rglob("SKILL.md")) if _BUILTIN_ROOT.is_dir() else []
|
||||
for path in paths:
|
||||
try:
|
||||
skill = Skill.from_markdown(path.read_text(encoding="utf-8"))
|
||||
except Exception:
|
||||
continue
|
||||
# Tracked procedures ship as trusted application behavior. They are
|
||||
# available immediately and never enter the user's audit queue.
|
||||
skill.status = "published"
|
||||
skill.confidence = 1.0
|
||||
existing_rows = [row for row in manager.load_all() if row.get("name") == skill.name]
|
||||
if existing_rows:
|
||||
row = existing_rows[0]
|
||||
# Built-ins are immutable tracked assets. Synchronize updated
|
||||
# versions/procedures on startup while leaving usage counters in
|
||||
# their sidecar untouched. Older startup code could also stamp the
|
||||
# first admin onto one; normalize that migration at the same time.
|
||||
if row.get("source") == "builtin":
|
||||
skill.owner = ""
|
||||
skill.source = "builtin"
|
||||
desired = skill.to_dict()
|
||||
if any(row.get(field) != desired.get(field) for field in _SYNC_FIELDS):
|
||||
manager._write_skill(skill)
|
||||
continue
|
||||
skill.owner = ""
|
||||
skill.source = "builtin"
|
||||
manager._write_skill(skill)
|
||||
existing.add(skill.name)
|
||||
installed += 1
|
||||
return installed
|
||||
@@ -90,6 +90,29 @@ EXTRACT_SYSTEM_PROMPT = (
|
||||
# How many recent messages to include for extraction
|
||||
CONTEXT_WINDOW = 6
|
||||
|
||||
PERSONA_MEMORY_SYSTEM_PROMPT = (
|
||||
"You maintain concise continuity notes for one active chat persona. "
|
||||
"Update the existing notes using only durable details established in the transcript. "
|
||||
"Keep details that help the same persona stay consistent in future conversations: "
|
||||
"relationship context, names, preferences, recurring story details, boundaries, and unresolved threads. "
|
||||
"Do not store generic chat events, temporary wording, assistant reasoning, or one-off requests. "
|
||||
"Never invent details. Return only the updated notes as short bullet points, max 12 bullets. "
|
||||
"If there is nothing worth keeping, return the existing notes unchanged or an empty string."
|
||||
)
|
||||
|
||||
HEALTH_PERSONA_MEMORY_SYSTEM_PROMPT = (
|
||||
"You maintain a cautious health-record brief for a medical reasoning persona. "
|
||||
"Update the existing brief using only medically durable information from the transcript. "
|
||||
"Keep facts that may matter in future health conversations: confirmed diagnoses, chronic conditions, "
|
||||
"surgeries/procedures, allergies, regular medications/supplements, important test results, clinicians/hospitals, "
|
||||
"ongoing symptoms or care plans, and the user's preferences for medical explanations. "
|
||||
"Use uncertainty labels when needed: 'reported', 'possible', 'asked about', 'unclear'. "
|
||||
"Do not turn guesses into diagnoses. Do not store casual one-off symptoms unless they are recurring, severe, "
|
||||
"or tied to an ongoing episode. Never invent facts. Return only the updated brief with these headings when useful: "
|
||||
"Medical profile, Medications/allergies, Episodes/open questions, Preferences. Max 16 concise bullets total. "
|
||||
"If nothing medically durable changed, return the existing brief unchanged or an empty string."
|
||||
)
|
||||
|
||||
AUDIT_SYSTEM_PROMPT = (
|
||||
"You are a memory database curator. Be CONSERVATIVE: remove only TRUE "
|
||||
"duplicates and clearly useless entries. Every distinct fact must survive. "
|
||||
@@ -112,6 +135,20 @@ AUDIT_SYSTEM_PROMPT = (
|
||||
)
|
||||
|
||||
AUDIT_INTERVAL = 5 # audit every N new memories added
|
||||
AUTO_PINNED_IDENTITY_LIMIT = 5
|
||||
|
||||
|
||||
def _is_owner_memory(entry, owner):
|
||||
if owner:
|
||||
return entry.get("owner") == owner or entry.get("owner") is None
|
||||
return True
|
||||
|
||||
|
||||
def _is_auto_pinned_identity(entry):
|
||||
return (
|
||||
bool(entry.get("pinned"))
|
||||
and (entry.get("category") or "").lower() in {"identity", "contact"}
|
||||
)
|
||||
_extractions_since_audit = 0
|
||||
|
||||
|
||||
@@ -397,6 +434,10 @@ async def extract_and_store(
|
||||
logger.error("Skipping auto memory extraction, store unreadable: %s", e)
|
||||
return
|
||||
added = 0
|
||||
auto_pinned_identity_count = sum(
|
||||
1 for entry in existing
|
||||
if _is_owner_memory(entry, _owner) and _is_auto_pinned_identity(entry)
|
||||
)
|
||||
|
||||
for fact in facts:
|
||||
if isinstance(fact, str):
|
||||
@@ -404,7 +445,7 @@ async def extract_and_store(
|
||||
category = "fact"
|
||||
elif isinstance(fact, dict):
|
||||
fact_text = fact.get("text", "").strip()
|
||||
category = fact.get("category", "fact")
|
||||
category = str(fact.get("category", "fact") or "fact")
|
||||
else:
|
||||
continue
|
||||
|
||||
@@ -446,9 +487,15 @@ async def extract_and_store(
|
||||
continue
|
||||
|
||||
entry = memory_manager.add_entry(fact_text, source="auto", category=category, owner=_owner)
|
||||
# Auto-pin identity facts (name, job, location) — core context
|
||||
if category == "identity":
|
||||
# Auto-pin only the first few identity/contact facts. Extra identity
|
||||
# memories are still saved, but they must be recalled by relevance
|
||||
# instead of riding along in every prompt forever.
|
||||
if (
|
||||
category.lower() in {"identity", "contact"}
|
||||
and auto_pinned_identity_count < AUTO_PINNED_IDENTITY_LIMIT
|
||||
):
|
||||
entry["pinned"] = True
|
||||
auto_pinned_identity_count += 1
|
||||
if hasattr(session, "session_id"):
|
||||
entry["session_id"] = session.session_id
|
||||
elif hasattr(session, "name"):
|
||||
@@ -492,6 +539,88 @@ async def extract_and_store(
|
||||
logger.error(f"Memory extraction failed: {e}")
|
||||
|
||||
|
||||
async def update_persona_memory(
|
||||
session,
|
||||
preset_manager,
|
||||
character_name: str,
|
||||
endpoint_url: str,
|
||||
model: str,
|
||||
headers: Optional[dict] = None,
|
||||
schema: str = "general",
|
||||
):
|
||||
"""Update the active persona's continuity notes from recent conversation.
|
||||
|
||||
Persona memory is stored with the persona/template data, not in the global
|
||||
memory DB, so deleting a saved persona also deletes its notes.
|
||||
"""
|
||||
character_name = (character_name or "").strip()
|
||||
if not character_name or not endpoint_url or not model or preset_manager is None:
|
||||
return
|
||||
|
||||
try:
|
||||
from src.llm_core import llm_call_async
|
||||
from src.text_helpers import strip_think
|
||||
|
||||
custom = {}
|
||||
try:
|
||||
custom = preset_manager.presets.get("custom", {}) if isinstance(preset_manager.presets, dict) else {}
|
||||
except Exception:
|
||||
custom = {}
|
||||
existing_memory = ""
|
||||
if isinstance(custom, dict) and custom.get("character_name") == character_name:
|
||||
existing_memory = custom.get("persona_memory", "") or ""
|
||||
|
||||
messages = session.get_context_messages()
|
||||
recent = messages[-CONTEXT_WINDOW:] if len(messages) > CONTEXT_WINDOW else messages
|
||||
if len(recent) < 2:
|
||||
return
|
||||
|
||||
lines = []
|
||||
for msg in recent:
|
||||
role = msg.get("role")
|
||||
content = msg.get("content", "")
|
||||
if isinstance(content, list):
|
||||
content = " ".join(
|
||||
b.get("text", "") for b in content
|
||||
if isinstance(b, dict) and b.get("type") == "text"
|
||||
)
|
||||
content = str(content or "").strip()
|
||||
if content:
|
||||
lines.append(f"{role}: {content}")
|
||||
if not lines:
|
||||
return
|
||||
|
||||
system_prompt = HEALTH_PERSONA_MEMORY_SYSTEM_PROMPT if schema == "health" else PERSONA_MEMORY_SYSTEM_PROMPT
|
||||
raw = await llm_call_async(
|
||||
endpoint_url,
|
||||
model,
|
||||
[
|
||||
{"role": "system", "content": system_prompt},
|
||||
{"role": "user", "content": (
|
||||
f"Persona name: {character_name}\n\n"
|
||||
f"Existing continuity notes:\n{existing_memory or '(none)'}\n\n"
|
||||
"Recent transcript:\n"
|
||||
+ "\n\n".join(lines)
|
||||
+ "\n\nReturn only the updated continuity notes."
|
||||
)},
|
||||
],
|
||||
temperature=0.1,
|
||||
max_tokens=1200,
|
||||
headers=headers,
|
||||
)
|
||||
|
||||
updated = strip_think(str(raw or ""), prose=True, prompt_echo=True).strip()
|
||||
updated = re.sub(r"^```(?:text|markdown)?\s*|\s*```$", "", updated, flags=re.I | re.S).strip()
|
||||
if len(updated) > 6000:
|
||||
updated = updated[:6000].rstrip()
|
||||
if updated == existing_memory:
|
||||
return
|
||||
if preset_manager.update_persona_memory(character_name, updated):
|
||||
logger.info("Updated persona memory for %s", character_name)
|
||||
except Exception as e:
|
||||
logger.warning("Persona memory update failed: %s", e)
|
||||
|
||||
|
||||
async def audit_memories(
|
||||
memory_manager,
|
||||
memory_vector,
|
||||
|
||||
@@ -28,6 +28,10 @@ SKILL_EXTRACT_PROMPT = (
|
||||
"(personal errands, a specific person/place/date, casual conversation).\n"
|
||||
"- A pure question/answer or explanation with no transferable method.\n"
|
||||
"- The agent failed, gave up, or the approach is not worth repeating.\n\n"
|
||||
"- Routine use of an existing tool, or a generic checklist with no new discovery.\n"
|
||||
"Prefer a specific successful workaround, an unexpected pitfall, or a verified "
|
||||
"sequence that would save rediscovery. Preserve exact useful commands and "
|
||||
"verification steps, but replace private identifiers and credentials with placeholders.\n\n"
|
||||
"When (and only when) a genuine reusable procedure exists, return a JSON "
|
||||
"object with:\n"
|
||||
'- "title": short name (under 10 words)\n'
|
||||
@@ -259,19 +263,9 @@ async def maybe_extract_skill(
|
||||
logger.debug("[skill-extract] '%s' already exists — dropped as duplicate", title)
|
||||
return None
|
||||
|
||||
# Auto-publish gate: if the user has `auto_approve_skills` on, the
|
||||
# newly-extracted skill is created `published` immediately rather
|
||||
# than waiting for the next audit batch. The audit still runs later
|
||||
# and can demote it back to `draft` (or delete) on failure. Default
|
||||
# ON matches the UI label "Auto-approve skills".
|
||||
# Automatic approval happens only after the audit has passed. A new
|
||||
# extraction begins as a draft so it cannot enter chat context early.
|
||||
_initial_status = "draft"
|
||||
try:
|
||||
from routes.prefs_routes import _load_for_user as _load_prefs
|
||||
_prefs = _load_prefs(owner) or {}
|
||||
if _prefs.get("auto_approve_skills", True):
|
||||
_initial_status = "published"
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
entry = skills_manager.add_skill(
|
||||
title=title,
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
"""Bounded automatic review queue for user-owned procedural memory."""
|
||||
import time
|
||||
|
||||
|
||||
def automatic_audit_candidates(skills, limit=8, now=None):
|
||||
"""Retry transient checks daily and failed repairs weekly, oldest first."""
|
||||
now = time.time() if now is None else now
|
||||
pending = []
|
||||
for skill in skills:
|
||||
if not skill.get("name") or skill.get("source") == "builtin" or skill.get("status") == "binned":
|
||||
continue
|
||||
verdict = skill.get("audit_verdict")
|
||||
if verdict in {"pass", "skipped"}:
|
||||
continue
|
||||
checked = float(skill.get("audited_at") or 0)
|
||||
delay = 7 * 86400 if verdict in {"fail", "needs_work"} else 86400
|
||||
if not verdict or now - checked >= delay:
|
||||
pending.append(skill)
|
||||
pending.sort(key=lambda skill: float(skill.get("audited_at") or 0))
|
||||
return pending[:max(1, limit)]
|
||||
+157
-44
@@ -54,6 +54,25 @@ def _to_float(x, default: float = 0.0) -> float:
|
||||
return default
|
||||
|
||||
|
||||
def _approval_policy(owner: Optional[str]) -> tuple[bool, float]:
|
||||
"""Read the user's automatic skill-approval gate without breaking retrieval."""
|
||||
try:
|
||||
from routes.prefs_routes import _load_for_user
|
||||
prefs = _load_for_user(owner) or {}
|
||||
except Exception:
|
||||
prefs = {}
|
||||
try:
|
||||
from src.settings import get_setting
|
||||
default_minimum = float(get_setting("skill_autosave_min_confidence", 0.85))
|
||||
except Exception:
|
||||
default_minimum = 0.85
|
||||
try:
|
||||
minimum = float(prefs.get("skill_min_confidence", default_minimum))
|
||||
except (TypeError, ValueError):
|
||||
minimum = default_minimum
|
||||
return bool(prefs.get("auto_approve_skills", True)), max(0.0, min(1.0, minimum))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# SkillsManager
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -120,7 +139,11 @@ class SkillsManager:
|
||||
|
||||
def set_audit(self, name: str, verdict: str, by_teacher: bool = False,
|
||||
worker_model: str = "", teacher_model: str = "",
|
||||
owner: Optional[str] = None) -> None:
|
||||
owner: Optional[str] = None, saved_turns: Optional[int] = None,
|
||||
saved_tool_calls: Optional[int] = None,
|
||||
baseline_verdict: Optional[str] = None,
|
||||
usefulness: Optional[float] = None,
|
||||
audit_summary: Optional[str] = None) -> None:
|
||||
"""Record the last test/audit result for a skill in the usage sidecar
|
||||
(so it surfaces in load() without touching SKILL.md). Drives the
|
||||
'verified' check + teacher mark on the card."""
|
||||
@@ -129,11 +152,34 @@ class SkillsManager:
|
||||
key = self._usage_key(name, owner)
|
||||
e = usage.setdefault(key, {"uses": 0, "last_used": None})
|
||||
e["audit_verdict"] = verdict
|
||||
# Replace, rather than retain, the explanation from a previous run.
|
||||
e["audit_summary"] = str(audit_summary or "")[:2000]
|
||||
# Version 2 fixes audit-arm isolation and separates functional success
|
||||
# from baseline utility. Legacy inconclusive results are not evidence
|
||||
# under that protocol and should be eligible for a clean re-audit.
|
||||
e["audit_version"] = 2
|
||||
e["audit_by_teacher"] = bool(by_teacher)
|
||||
if worker_model:
|
||||
e["audit_worker_model"] = worker_model
|
||||
if teacher_model:
|
||||
e["audit_teacher_model"] = teacher_model
|
||||
if saved_turns is not None:
|
||||
try:
|
||||
e["saved_turns"] = int(saved_turns)
|
||||
except (TypeError, ValueError):
|
||||
e.pop("saved_turns", None)
|
||||
if saved_tool_calls is not None:
|
||||
try:
|
||||
e["saved_tool_calls"] = int(saved_tool_calls)
|
||||
except (TypeError, ValueError):
|
||||
e.pop("saved_tool_calls", None)
|
||||
if baseline_verdict is not None:
|
||||
e["baseline_verdict"] = str(baseline_verdict or "unknown")
|
||||
if usefulness is not None:
|
||||
try:
|
||||
e["usefulness"] = float(usefulness)
|
||||
except (TypeError, ValueError):
|
||||
e.pop("usefulness", None)
|
||||
e["audited_at"] = _t.time()
|
||||
self._save_usage(usage)
|
||||
|
||||
@@ -197,6 +243,8 @@ class SkillsManager:
|
||||
sk = self._read_skill(path)
|
||||
if not sk:
|
||||
continue
|
||||
if sk.source == "builtin":
|
||||
continue
|
||||
owner = (sk.owner or "").strip()
|
||||
if owner == primary_owner:
|
||||
continue
|
||||
@@ -227,11 +275,24 @@ class SkillsManager:
|
||||
u = self._usage_entry(usage, sk.name, sk.owner)
|
||||
d["uses"] = int(u.get("uses", 0))
|
||||
d["last_used"] = u.get("last_used")
|
||||
d["audit_verdict"] = u.get("audit_verdict")
|
||||
audit_verdict = u.get("audit_verdict")
|
||||
try:
|
||||
audit_version = int(u.get("audit_version") or 0)
|
||||
except (TypeError, ValueError):
|
||||
audit_version = 0
|
||||
if audit_verdict == "inconclusive" and audit_version < 2:
|
||||
audit_verdict = None
|
||||
d["audit_verdict"] = audit_verdict
|
||||
d["audit_summary"] = u.get("audit_summary", "") if audit_verdict else ""
|
||||
d["audit_version"] = audit_version
|
||||
d["audit_by_teacher"] = bool(u.get("audit_by_teacher"))
|
||||
d["audit_worker_model"] = u.get("audit_worker_model")
|
||||
d["audit_teacher_model"] = u.get("audit_teacher_model")
|
||||
d["audited_at"] = u.get("audited_at")
|
||||
d["audited_at"] = u.get("audited_at") if audit_verdict else None
|
||||
d["saved_turns"] = u.get("saved_turns")
|
||||
d["saved_tool_calls"] = u.get("saved_tool_calls")
|
||||
d["baseline_verdict"] = u.get("baseline_verdict")
|
||||
d["usefulness"] = u.get("usefulness")
|
||||
d["necessity"] = u.get("necessity")
|
||||
out.append(d)
|
||||
seen_names.add(sk.name)
|
||||
@@ -284,7 +345,11 @@ class SkillsManager:
|
||||
# leaked legacy / un-stamped skills to every authenticated user.
|
||||
# Hide them now; the owner needs to be backfilled on disk if those
|
||||
# skills should be visible to a specific user.
|
||||
return [s for s in entries if s.get("owner") == owner]
|
||||
return [
|
||||
s for s in entries
|
||||
if s.get("owner") == owner
|
||||
or (s.get("source") == "builtin" and not s.get("owner"))
|
||||
]
|
||||
|
||||
# ----------------------------------------------------------------------
|
||||
# CRUD — disk-backed
|
||||
@@ -546,7 +611,15 @@ class SkillsManager:
|
||||
sk = self._read_skill(path)
|
||||
if not sk or sk.name != name:
|
||||
continue
|
||||
if (sk.owner or "") != (owner or ""):
|
||||
# Built-in skills are shared, ownerless procedures. ``load``
|
||||
# exposes them to every owner, so direct progressive-disclosure
|
||||
# reads must apply the same visibility rule as the index/list
|
||||
# path. Previously a built-in appeared in `list` but `view`
|
||||
# returned not-found for authenticated users.
|
||||
if not (
|
||||
(sk.owner or "") == (owner or "")
|
||||
or (sk.source == "builtin" and not (sk.owner or ""))
|
||||
):
|
||||
continue
|
||||
try:
|
||||
with open(path, encoding="utf-8") as f:
|
||||
@@ -562,7 +635,10 @@ class SkillsManager:
|
||||
sk = self._read_skill(path)
|
||||
if not sk or sk.name != name:
|
||||
continue
|
||||
if (sk.owner or "") != (owner or ""):
|
||||
if not (
|
||||
(sk.owner or "") == (owner or "")
|
||||
or (sk.source == "builtin" and not (sk.owner or ""))
|
||||
):
|
||||
continue
|
||||
base = os.path.realpath(os.path.dirname(path))
|
||||
target = os.path.realpath(os.path.join(base, ref_path))
|
||||
@@ -591,18 +667,12 @@ class SkillsManager:
|
||||
"""Return the `[{name, description, category, status}]` list the
|
||||
agent sees in its system prompt.
|
||||
|
||||
Includes:
|
||||
- All published skills.
|
||||
- Drafts written by the teacher-escalation loop
|
||||
(`source == "teacher-escalation"`). The whole point of
|
||||
the teacher loop is for the student to find the new
|
||||
procedure on the very next turn — waiting for a manual
|
||||
publish click defeats the loop.
|
||||
|
||||
Excludes user-created drafts (status=draft, source != teacher-
|
||||
escalation) — those are work-in-progress and pollute the
|
||||
prompt with half-finished procedures.
|
||||
Includes built-ins plus user skills that have passed their audit and
|
||||
meet the owner's current automatic-approval threshold. A persistent
|
||||
``published`` flag is not sufficient: a changed threshold or a legacy
|
||||
record must not make an unaudited skill eligible for prompt injection.
|
||||
"""
|
||||
auto_approve, min_confidence = _approval_policy(owner)
|
||||
out = []
|
||||
for s in self.load(owner=owner):
|
||||
status = s.get("status")
|
||||
@@ -613,6 +683,19 @@ class SkillsManager:
|
||||
pass # let it through
|
||||
else:
|
||||
continue
|
||||
# A stale published record must not remain injectable after an
|
||||
# audit has recorded a failure. Inconclusive is not a failure.
|
||||
audit_verdict = str(s.get("audit_verdict") or "").lower()
|
||||
if audit_verdict in {"needs_work", "fail"}:
|
||||
continue
|
||||
if s.get("source") != "builtin" and auto_approve:
|
||||
if status != "published" or audit_verdict != "pass":
|
||||
continue
|
||||
if _to_float(s.get("confidence"), 0.0) < min_confidence:
|
||||
continue
|
||||
necessity = s.get("necessity") or {}
|
||||
if isinstance(necessity, dict) and necessity.get("necessary") is False:
|
||||
continue
|
||||
# Platform gating
|
||||
if platform and s.get("platforms") and platform not in s["platforms"]:
|
||||
continue
|
||||
@@ -649,6 +732,8 @@ class SkillsManager:
|
||||
threshold: float = 0.3,
|
||||
max_items: int = 5,
|
||||
min_confidence: float = 0.0,
|
||||
available_toolsets: Optional[Iterable[str]] = None,
|
||||
platform: Optional[str] = None,
|
||||
) -> List[Dict]:
|
||||
if skills is None:
|
||||
skills = self.load_all()
|
||||
@@ -660,37 +745,62 @@ class SkillsManager:
|
||||
# without a manual publish click. The UI flags teacher-written
|
||||
# entries with a 🎓 badge so users can demote / delete bad
|
||||
# ones when they spot them.
|
||||
skills = [s for s in skills if s.get("status") in ("published", "draft")]
|
||||
# Confidence gate (used by prompt-injection, NOT by search): a DRAFT
|
||||
# skill must clear the bar to be injected. Published skills are already
|
||||
# vetted, so they always qualify. Missing confidence = treat as 1.0
|
||||
# (legacy skills shouldn't silently vanish). 0 disables the gate.
|
||||
skills = [
|
||||
s for s in skills
|
||||
if s.get("status") in ("published", "draft")
|
||||
and str(s.get("audit_verdict") or "").lower()
|
||||
not in {"needs_work", "fail", "skipped"}
|
||||
]
|
||||
available = set(available_toolsets) if available_toolsets is not None else None
|
||||
if available is not None:
|
||||
skills = [
|
||||
skill for skill in skills
|
||||
if all(tool in available for tool in (skill.get("requires_toolsets") or []))
|
||||
and not any(tool in available for tool in (skill.get("fallback_for_toolsets") or []))
|
||||
]
|
||||
if platform:
|
||||
skills = [
|
||||
skill for skill in skills
|
||||
if not skill.get("platforms") or platform in skill.get("platforms", [])
|
||||
]
|
||||
# Prompt injection is fail-closed for user skills. Built-ins are
|
||||
# shipped procedures; every other skill needs a passing audit and a
|
||||
# confidence score at the user's current threshold.
|
||||
if min_confidence > 0:
|
||||
def _passes(s):
|
||||
if s.get("status") == "published":
|
||||
if s.get("source") == "builtin":
|
||||
return True
|
||||
# Teacher-escalation drafts are auto-written from a (possibly
|
||||
# untrusted) trace and injected as authoritative guidance, so they
|
||||
# must EARN injection with an explicit, parseable confidence that
|
||||
# clears the bar — fail closed on a missing/garbage value instead
|
||||
# of treating it as 1.0. Hand-authored legacy drafts keep the
|
||||
# lenient "unset → keep" behavior so they don't silently vanish.
|
||||
if s.get("source") == "teacher-escalation":
|
||||
c = s.get("confidence")
|
||||
if c is None:
|
||||
return False
|
||||
return _to_float(c, 0.0) >= min_confidence # unparseable → fail closed
|
||||
c = s.get("confidence")
|
||||
if c is None:
|
||||
return True # unset → don't filter (legacy)
|
||||
return _to_float(c, 1.0) >= min_confidence # unparseable → pass
|
||||
return (
|
||||
s.get("status") == "published"
|
||||
and str(s.get("audit_verdict") or "").lower() == "pass"
|
||||
and _to_float(s.get("confidence"), 0.0) >= min_confidence
|
||||
)
|
||||
skills = [s for s in skills if _passes(s)]
|
||||
if not skills:
|
||||
return []
|
||||
|
||||
query_tokens = _tokenize(query)
|
||||
semantic_scores: Dict[int, float] = {}
|
||||
semantic_enabled = str(
|
||||
os.environ.get("ODYSSEUS_SKILL_SEMANTIC_RETRIEVAL", "1")
|
||||
).strip().lower() not in {"0", "false", "no", "off"}
|
||||
if semantic_enabled:
|
||||
try:
|
||||
from src.skill_index import semantic_skill_scores
|
||||
|
||||
semantic_scores = semantic_skill_scores(query, skills)
|
||||
except Exception as exc:
|
||||
logger.debug("Semantic skill retrieval unavailable: %s", exc)
|
||||
try:
|
||||
semantic_threshold = float(
|
||||
os.environ.get("ODYSSEUS_SKILL_SEMANTIC_THRESHOLD", "0.4")
|
||||
)
|
||||
except (TypeError, ValueError):
|
||||
semantic_threshold = 0.4
|
||||
semantic_threshold = max(-1.0, min(1.0, semantic_threshold))
|
||||
|
||||
scored = []
|
||||
for sk in skills:
|
||||
for position, sk in enumerate(skills):
|
||||
text = " ".join([
|
||||
sk.get("name", ""),
|
||||
sk.get("description", ""),
|
||||
@@ -698,19 +808,22 @@ class SkillsManager:
|
||||
" ".join(sk.get("tags", []) or []),
|
||||
" ".join(sk.get("procedure", []) or []),
|
||||
])
|
||||
score = _jaccard(query_tokens, _tokenize(text))
|
||||
lexical_score = _jaccard(query_tokens, _tokenize(text))
|
||||
for tag in sk.get("tags", []) or []:
|
||||
# Match tags as whole tokens, not substrings: `tag in query`
|
||||
# boosted e.g. a "ai" tag for any query containing "email".
|
||||
tag_tokens = _tokenize(tag)
|
||||
if tag_tokens and tag_tokens <= query_tokens:
|
||||
score = max(score, 0.3) * 1.3
|
||||
lexical_score = max(lexical_score, 0.3) * 1.3
|
||||
if query.lower() in (sk.get("description") or "").lower():
|
||||
score = max(score, 0.6)
|
||||
lexical_score = max(lexical_score, 0.6)
|
||||
semantic_score = semantic_scores.get(position, -1.0)
|
||||
if lexical_score < threshold and semantic_score < semantic_threshold:
|
||||
continue
|
||||
score = max(lexical_score, semantic_score)
|
||||
score *= 1.0 + _to_float(sk.get("confidence"), 0.5) * 0.1
|
||||
if sk.get("uses", 0) > 0:
|
||||
score *= 1.05
|
||||
if score >= threshold:
|
||||
scored.append((score, sk))
|
||||
scored.append((score, sk))
|
||||
scored.sort(key=lambda x: x[0], reverse=True)
|
||||
return [sk for _, sk in scored[:max_items]]
|
||||
|
||||
Reference in New Issue
Block a user