Consolidate Odysseus agent harness and tool contracts

This commit is contained in:
pewdiepie-archdaemon
2026-09-17 10:07:40 +00:00
parent 84aa9a91de
commit 218d762427
229 changed files with 28899 additions and 1551 deletions
+333
View File
@@ -0,0 +1,333 @@
#!/usr/bin/env python3
"""Build an accountable harness/SFT seed corpus from historical SFT sessions."""
from __future__ import annotations
import argparse
import json
import re
import sqlite3
from collections import Counter
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
EXCLUDED_PREFIXES = ("[harness-qa]",)
FAMILY_ALIASES = {
"cookbook": "cookbook_admin",
"shell_files": "shell_files",
"search": "search_browser",
"search_ai": "search_browser",
}
CANONICAL_FAMILIES = {
"calendar", "notes", "email", "memory", "documents", "tasks", "skills",
"search_browser", "cookbook_admin", "shell_files", "research", "ui", "switching",
}
def case_name(session_name: str) -> str:
return session_name.split("]", 1)[-1].strip()
def infer_family(name: str) -> str:
value = case_name(name).casefold()
value = re.sub(r"^(?:typo|ambiguous|related)[-_]", "", value)
if "_to_" in value or value.startswith("greeting_to_"):
return "switching"
if value.startswith(("browser", "news_followup", "search_")):
return "search_browser"
stem = re.split(r"[-_]\d", value, maxsplit=1)[0]
if stem in CANONICAL_FAMILIES:
return stem
for alias, family in FAMILY_ALIASES.items():
if stem == alias or value.startswith(alias + "-"):
return family
return "unknown"
def infer_text_family(text: str) -> str:
value = re.sub(r"\s+", " ", text).casefold()
groups = (
("calendar", ("calendar", "event", "schedule", "appointment", "meeting")),
("notes", ("note", "checklist")),
("email", ("email", "inbox", "sender", "unsubscribe", "spam")),
("memory", ("memory", "remember", "forget")),
("documents", ("document", "write reply", "write this", "editor")),
("tasks", ("task", "scheduled job", "cron")),
("skills", ("skill",)),
("research", ("research",)),
("cookbook_admin", ("model server", "endpoint", "runpod", "served model", "cookbook")),
("shell_files", ("workspace", "file", "folder", "directory", "bash", "python", "ssh")),
("ui", ("open gallery", "open panel", "theme")),
("search_browser", ("http://", "https://", "search", "look up", "browse", "website", "latest", "weather", "news")),
)
matched = [family for family, words in groups if any(word in value for word in words)]
if len(set(matched)) > 1:
return "switching"
return matched[0] if matched else "general"
def infer_turn_family(session_family: str, turn: dict[str, Any]) -> str:
"""Prefer observed tool/contract evidence over unreliable session titles."""
metadata = turn.get("metadata") or {}
names = {
str(event.get("tool") or "")
for event in (metadata.get("tool_events") or [])
if isinstance(event, dict)
}
contract = metadata.get("turn_contract") or {}
capabilities = contract.get("capabilities") or metadata.get("capabilities") or []
hints = " ".join(sorted(names | {str(value) for value in capabilities})).casefold()
mappings = (
(("calendar", "manage_calendar"), "calendar"),
(("notes", "manage_notes"), "notes"),
(("email", "inbox", "draft_email"), "email"),
(("memory", "manage_memory"), "memory"),
(("document", "manage_documents"), "documents"),
(("task", "manage_tasks"), "tasks"),
(("skill", "manage_skills"), "skills"),
(("research", "trigger_research"), "research"),
(("browser", "web_search", "web_fetch", "youtube"), "search_browser"),
(("cookbook", "served_model", "cached_model", "endpoint"), "cookbook_admin"),
(("shell", "bash", "read_file", "write_file", "\bls\b"), "shell_files"),
(("ui_control",), "ui"),
)
matched = [family for needles, family in mappings if any(needle in hints for needle in needles)]
if len(set(matched)) > 1:
return "switching"
if matched:
return matched[0]
if session_family != "unknown":
return session_family
return infer_text_family(str(turn.get("user") or ""))
def normalized_flow_key(turns: list[dict[str, Any]]) -> str:
texts = []
for turn in turns:
text = re.sub(r"\s+", " ", str(turn.get("user") or "")).strip().casefold()
texts.append(text)
return "\n".join(texts)
def event_failed(event: dict[str, Any]) -> bool:
return bool(event.get("error") or event.get("exit_code") not in (None, 0))
def classify(turns: list[dict[str, Any]]) -> tuple[str, list[str]]:
"""Conservative historical triage; replay resolves everything uncertain."""
reasons: list[str] = []
backend = False
harness = False
model_sft = False
successful_tool = False
for index, turn in enumerate(turns):
assistant = str(turn.get("assistant") or "")
metadata = turn.get("metadata") or {}
events = metadata.get("tool_events") or []
successful_tool |= any(not event_failed(event) for event in events)
combined_errors = "\n".join(
str(event.get("error") or "") + "\n" + str(event.get("output") or "")
for event in events if event_failed(event)
)
if re.search(r"connection refused|timed? out|backend unavailable|service unavailable", combined_errors, re.I):
backend = True
reasons.append(f"turn {index + 1}: tool/backend transport failed")
denied = any(
isinstance(decision, dict) and decision.get("allowed") is False
for decision in (metadata.get("policy_decisions") or [])
)
if metadata.get("required_operation_succeeded") is False or denied:
harness = True
reasons.append(f"turn {index + 1}: harness policy or required operation blocked execution")
if index and re.search(r"no preceding (?:answer|message)|not in this conversation", assistant, re.I):
harness = True
reasons.append(f"turn {index + 1}: prior conversation state was lost")
if successful_tool and re.search(
r"(?:cannot|can't|unable to) (?:access|view|open|read|use).{0,40}(?:notes?|calendar|emails?|tasks?|documents?)",
assistant,
re.I,
):
model_sft = True
reasons.append(f"turn {index + 1}: response contradicted successful tool evidence")
if any(event_failed(event) and re.search(
r"placeholder|not returned by|invalid arguments?|validation|must be an exact",
str(event.get("error") or "") + str(event.get("output") or ""), re.I,
) for event in events):
model_sft = True
reasons.append(f"turn {index + 1}: model proposed invalid or ungrounded arguments")
if backend:
return "backend", sorted(set(reasons))
if harness:
return "harness", sorted(set(reasons))
if model_sft:
return "model_sft", sorted(set(reasons))
return "replay_first", ["historical result is not sufficient for a reliable owner classification"]
def load_sessions(db_path: Path, owner: str) -> list[dict[str, Any]]:
db = sqlite3.connect(db_path)
db.row_factory = sqlite3.Row
sessions = db.execute(
"SELECT id, name, created_at FROM sessions WHERE owner=? ORDER BY created_at DESC",
(owner,),
).fetchall()
output = []
for session in sessions:
if any(str(session["name"] or "").startswith(prefix) for prefix in EXCLUDED_PREFIXES):
continue
rows = db.execute(
"SELECT role, content, metadata FROM chat_messages WHERE session_id=? ORDER BY timestamp, rowid",
(session["id"],),
).fetchall()
turns = []
pending = None
for row in rows:
if row["role"] == "user":
pending = {"user": row["content"], "assistant": "", "metadata": {}}
turns.append(pending)
elif row["role"] == "assistant" and pending is not None:
pending["assistant"] = row["content"]
try:
pending["metadata"] = json.loads(row["metadata"] or "{}")
except (TypeError, ValueError, json.JSONDecodeError):
pending["metadata"] = {}
pending = None
if turns:
session_family = infer_family(session["name"])
for turn in turns:
turn["family"] = infer_turn_family(session_family, turn)
output.append({
"source_session_id": session["id"],
"source_name": session["name"],
"created_at": session["created_at"],
"family": session_family,
"turns": turns,
})
db.close()
return output
def build_seeds(sessions: list[dict[str, Any]], context_turns: int = 3) -> list[dict[str, Any]]:
"""Create exactly one teacher seed for every historical user turn.
A seed retains preceding user context so ambiguous follow-ups remain
ambiguous in the same useful way. Repeated source runs are intentionally
retained; they measure stability instead of disappearing via deduplication.
"""
seeds: list[dict[str, Any]] = []
for session in sessions:
turns = session["turns"]
for index, turn in enumerate(turns):
start = max(0, index - context_turns)
context = [
{"user": item["user"]}
for item in turns[start:index + 1]
]
seeds.append({
"seed_id": f"{session['source_session_id']}:{index + 1}",
"source_session_id": session["source_session_id"],
"source_name": session["source_name"],
"source_turn": index + 1,
"family": turn.get("family") or session["family"],
"context": context,
"target_user": turn["user"],
})
return seeds
def build_queue(sessions: list[dict[str, Any]]) -> dict[str, Any]:
seeds = build_seeds(sessions)
unique: dict[str, dict[str, Any]] = {}
duplicate_counts = Counter()
for session in sessions:
key = normalized_flow_key(session["turns"])
duplicate_counts[key] += 1
if key not in unique: # sessions arrive newest first
unique[key] = session
workstreams = {name: [] for name in ("harness", "model_sft", "backend", "replay_first")}
replay_flows = []
for number, (key, session) in enumerate(unique.items(), 1):
bucket, reasons = classify(session["turns"])
row = {
"id": f"historical-{number:04d}",
"family": session["family"],
"case": case_name(session["source_name"]),
"source_session_id": session["source_session_id"],
"duplicate_runs": duplicate_counts[key],
"reasons": reasons,
"turns": [
{
"user": turn["user"],
"assistant": turn["assistant"],
"tools": [event.get("tool") for event in (turn["metadata"].get("tool_events") or [])],
}
for turn in session["turns"]
],
}
workstreams[bucket].append(row)
replay_flows.append({
"id": row["id"],
"family": row["family"],
"purpose": f"Replay historical contract case {row['case']}",
"turns": [{
"user": turn["user"],
"expect": "Honor the request and conversation context; use the correct tool only when needed and rely on successful tool evidence.",
} for turn in session["turns"]],
})
return {
"created_at": datetime.now(timezone.utc).isoformat(),
"source_sessions": len(sessions),
"source_user_turns": sum(len(session["turns"]) for session in sessions),
"seed_count": len(seeds),
"unique_flows": len(unique),
"counts": {name: len(rows) for name, rows in workstreams.items()},
"families": dict(sorted(Counter(row["family"] for row in unique.values()).items())),
"seed_families": dict(sorted(Counter(row["family"] for row in seeds).items())),
"workstreams": workstreams,
"flows": replay_flows,
"seeds": seeds,
}
def render_summary(queue: dict[str, Any]) -> str:
lines = [
"# Historical Odysseus QA Queue", "",
f"- Source sessions: {queue['source_sessions']}",
f"- Source user turns / teacher seeds: {queue['seed_count']}",
f"- Unique conversation flows: {queue['unique_flows']}",
"- Historical labels are conservative; `replay_first` must be replayed before assigning ownership.",
"", "## Workstreams", "",
]
for name, count in queue["counts"].items():
lines.append(f"- `{name}`: {count}")
lines.extend(["", "## Families", ""])
for family, count in queue["seed_families"].items():
lines.append(f"- `{family}`: {count}")
lines.extend([
"", "## Workflow", "",
"1. Cook one fresh conversation from every seed using the complete tool catalog.",
"2. Replay safe cooked cases on the current 7011 Agent runtime.",
"3. Judge, classify ownership, and patch recurring behavior classes.",
"4. Retain duplicate source runs as stability evidence; account for quarantined cases explicitly.",
])
return "\n".join(lines) + "\n"
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--db", type=Path, required=True)
parser.add_argument("--owner", default="sft_alex_creator")
parser.add_argument("--output", type=Path, required=True)
parser.add_argument("--summary", type=Path, required=True)
args = parser.parse_args()
queue = build_queue(load_sessions(args.db, args.owner))
args.output.parent.mkdir(parents=True, exist_ok=True)
args.summary.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(json.dumps(queue, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
args.summary.write_text(render_summary(queue), encoding="utf-8")
print(json.dumps({key: queue[key] for key in ("source_sessions", "unique_flows", "counts", "families")}, indent=2))
if __name__ == "__main__":
main()
@@ -0,0 +1,230 @@
#!/usr/bin/env python3
"""Build a reproducible model-only repair pool from conversation QA runs."""
from __future__ import annotations
import argparse
import hashlib
import json
import re
from collections import Counter
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
SFT_WEBUI_POLICY_DISABLED_TOOLS = frozenset({
"python", "read_file", "write_file", "edit_file", "apply_patch",
})
def source_seed_id(row: dict[str, Any]) -> str:
return str(row.get("source_seed_id") or row.get("id") or "").strip()
def behavior_category(value: str) -> str:
text = str(value or "").casefold()
rules = (
("response_constraint_adherence", (
"limit", "constraint", "instruction_noncompliance", "instruction_following",
"counting_error",
)),
("required_tool_execution", (
"missing_tool", "missing_required_tool", "missing_required_action",
"false_refusal", "refusal",
)),
("tool_action_selection", (
"wrong_action", "wrong_tool", "incorrect_tool", "malformed_tool",
"command_selection",
)),
("required_argument_grounding", ("argument", "identifier", "filter")),
("tool_error_recovery", (
"no_retry", "error_recovery", "false_empty", "empty_result",
"unrecovered", "missing_fallback", "stale_id_loop",
)),
("result_rendering", (
"render", "empty_answer", "missing_requested_content", "missing_note_titles",
"missing_progress_link", "non_answer", "uninformative_answer",
)),
("followup_evidence_use", ("followup", "follow_up", "continuity", "unanswered", "incomplete")),
("evidence_grounding", (
"hallucin", "wrong_answer", "unsupported", "grounding", "false_success",
"unfaithful", "content_mismatch",
)),
)
for category, needles in rules:
if any(needle in text for needle in needles):
return category
return "other_model_behavior"
def has_transport_failure(row: dict[str, Any]) -> bool:
needles = (
"connection refused", "connecterror", "remoteprotocolerror",
"replay_transport_unavailable", "session_start_failed", "readtimeout",
)
return any(needle in json.dumps(row, ensure_ascii=False).casefold() for needle in needles)
def eligible_failed_turns(row: dict[str, Any]) -> tuple[list[int], list[int]]:
observed = row.get("observed") or []
failed = [value for value in (row.get("judge") or {}).get("failed_turns") or []
if isinstance(value, int) and 1 <= value <= len(observed)]
if not failed:
failed = list(range(1, len(observed) + 1))
eligible, absent_surface = [], []
for number in failed:
turn = observed[number - 1]
contract = turn.get("contract") or {}
if not (contract.get("offered") or []) and not (turn.get("tool_calls") or []):
absent_surface.append(number)
else:
eligible.append(number)
return eligible, absent_surface
def requires_native_workspace_tool(row: dict[str, Any]) -> bool:
expected = "\n".join(
str(turn.get("expect") or "")
for turn in (row.get("turns") or [])
if isinstance(turn, dict)
)
return any(
re.search(rf"(?<!\w){re.escape(tool)}(?!\w)", expected, re.I)
for tool in SFT_WEBUI_POLICY_DISABLED_TOOLS
) or bool(re.search(
r"\b(?:run|use|execute)\s+(?:a\s+)?(?:local\s+)?(?:shell|bash)\b|"
r"\b(?:shell|bash)\s+(?:version\s+)?check\b",
expected,
re.I,
))
def build_manifest(paths: list[Path], excluded_seeds: set[str],
routing_experiment: str | None = None,
resolved_seeds: set[str] | None = None) -> dict[str, Any]:
"""Retain each seed's latest confirmed model-owned failure.
A later stochastic pass does not prove a repair and must not silently erase
a useful failure example. Operators can explicitly resolve or exclude a
seed after a verified fix or after discovering a defective expectation.
"""
resolved_seeds = resolved_seeds or set()
latest_failure: dict[str, tuple[int, dict[str, Any], Path]] = {}
inputs = []
ignored_nonbehavioral_rows = 0
ignored_runtime_inputs = 0
for order, path in enumerate(paths):
raw = path.read_bytes()
payload = json.loads(raw)
runtime = payload.get("routing_experiment", "baseline")
inputs.append({
"path": str(path), "sha256": hashlib.sha256(raw).hexdigest(),
"routing_experiment": runtime,
})
if routing_experiment is not None and runtime != routing_experiment:
ignored_runtime_inputs += 1
continue
for row in payload.get("results") or []:
seed = source_seed_id(row)
judge = row.get("judge") or {}
# An unavailable judge or broken replay does not supersede older
# valid behavioral evidence for the same seed.
if not seed or judge.get("verdict") not in {"pass", "fail"} or has_transport_failure(row):
ignored_nonbehavioral_rows += 1
continue
if judge.get("verdict") == "fail" and judge.get("owner") == "model_sft":
latest_failure[seed] = (order, row, path)
candidates, exclusions = [], []
for seed, (_, row, path) in sorted(latest_failure.items()):
judge = row.get("judge") or {}
reason = None
if seed in excluded_seeds:
reason = "explicit_ambiguous_or_defective_seed"
elif seed in resolved_seeds:
reason = "explicitly_resolved_after_verified_fix"
elif requires_native_workspace_tool(row):
reason = "requires_native_workspace_tool_on_webui_surface"
elif has_transport_failure(row):
reason = "transport_contaminated"
eligible, absent_surface = eligible_failed_turns(row)
if reason is None and not eligible:
reason = "no_failed_turn_with_executable_tool_surface"
if reason:
exclusions.append({"source_seed_id": seed, "reason": reason})
continue
candidates.append({
"source_seed_id": seed,
"family": row.get("family"),
"purpose": row.get("purpose"),
"behavior_category": behavior_category(judge.get("failure_category", "")),
"eligible_failed_turns": eligible,
"excluded_absent_surface_turns": absent_surface,
"judge": judge,
"turns": row.get("turns") or [],
"observed": row.get("observed") or [],
"session_id": row.get("session_id"),
"url": row.get("url"),
"latest_run": str(path),
})
return {
"created_at": datetime.now(timezone.utc).isoformat(),
"policy": {
"precedence": "latest confirmed model_sft failure wins per source_seed_id; later stochastic passes do not erase it",
"include": "latest model_sft fail verdict with executable tool surface",
"exclude": [
"pass/uncertain", "non-model owners", "transport contamination",
"failed turns with absent tool surface", "explicit ambiguous/defective seeds",
"native-workspace-only expectations on the WebUI surface", "explicitly resolved seeds",
],
},
"routing_experiment": routing_experiment,
"inputs": inputs,
"ignored_runtime_inputs": ignored_runtime_inputs,
"ignored_nonbehavioral_rows": ignored_nonbehavioral_rows,
"candidate_count": len(candidates),
"counts_by_family": dict(sorted(Counter(row["family"] for row in candidates).items())),
"counts_by_behavior": dict(sorted(Counter(row["behavior_category"] for row in candidates).items())),
"candidates": candidates,
"exclusion_count": len(exclusions),
"exclusions": exclusions,
}
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--run", type=Path, action="append", required=True,
help="QA run in chronological order; repeat for later replays")
parser.add_argument("--exclude-seed", action="append", default=[],
help="Explicitly exclude an ambiguous or defective generated seed")
parser.add_argument("--resolved-seed", action="append", default=[],
help="Drop a model failure only after a verified repair replay")
parser.add_argument(
"--routing-experiment", default="recent_model_choice",
help="Include only runs from this exact routing runtime",
)
parser.add_argument("--output", type=Path, required=True)
return parser.parse_args()
def main() -> int:
args = parse_args()
manifest = build_manifest(
args.run, set(args.exclude_seed), args.routing_experiment,
set(args.resolved_seed),
)
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(json.dumps(manifest, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(json.dumps({
"output": str(args.output),
"candidates": manifest["candidate_count"],
"by_family": manifest["counts_by_family"],
"by_behavior": manifest["counts_by_behavior"],
"excluded": manifest["exclusion_count"],
}, indent=2))
return 0
if __name__ == "__main__":
raise SystemExit(main())
+5 -1
View File
@@ -10,11 +10,15 @@ from collections import Counter
from pathlib import Path
from typing import Any
from dotenv import load_dotenv
ROOT = Path(__file__).resolve().parents[1]
load_dotenv(ROOT / ".env")
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from core.database import CalendarCal, CalendarEvent, Document, Memory, Note, ScheduledTask, Session, SessionLocal, UserTool # noqa: E402
from src.constants import DATA_DIR # noqa: E402
from scripts.sft_email_overseer import PROFILES # noqa: E402
OWNERS = ["sft_maya_ops", "sft_jules_research", "sft_nora_design", "sft_omar_finance"]
@@ -25,7 +29,7 @@ def clip(value: Any, limit: int = 180) -> str:
def email_inventory() -> dict[str, list[dict[str, Any]]]:
payload = json.loads((ROOT / "data/fixture_email_messages.json").read_text(encoding="utf-8"))
payload = json.loads((Path(DATA_DIR) / "fixture_email_messages.json").read_text(encoding="utf-8"))
rows = payload.get("messages") if isinstance(payload, dict) else payload
out = {owner: [] for owner in OWNERS}
for row in rows or []:
+298
View File
@@ -0,0 +1,298 @@
#!/usr/bin/env python3
"""Cook every historical SFT Alex user turn into a fresh tool conversation."""
from __future__ import annotations
import argparse
import concurrent.futures
import fcntl
import json
import re
import sys
import threading
from collections import Counter
from pathlib import Path
from typing import Any
SCRIPT_DIR = Path(__file__).resolve().parent
if str(SCRIPT_DIR) not in sys.path:
sys.path.insert(0, str(SCRIPT_DIR))
from odysseus_conversation_qa import (
DEFAULT_DATA,
DEFAULT_JUDGE_ENDPOINT,
DEFAULT_JUDGE_MODEL,
FAMILY_SEEDS,
compact_tool_catalog,
endpoint_from_db,
teacher_json,
)
ROOT = Path(__file__).resolve().parents[1]
DEFAULT_SEEDS = ROOT / "tmp/odysseus-conversation-qa/sft-alex-all-seeds.json"
DEFAULT_OUTPUT = ROOT / "tmp/odysseus-conversation-qa/sft-alex-cooked.jsonl"
LOCK = threading.Lock()
_CREATE_RE = re.compile(
r"\b(?:create|make|start|write|add|save|draft|new)\b", re.IGNORECASE
)
_LOOKUP_RE = re.compile(
r"\b(?:open|find|show|read|list|search|retrieve|look\s+up|already\s+have|saved)\b",
re.IGNORECASE,
)
_NEW_TOPIC_RE = re.compile(
r"\b(?:about|on)\s+(.+?)(?=\s+(?:and|then|with|using)\b|[.!?]|$)",
re.IGNORECASE,
)
_ENTITY_PATTERNS = (
re.compile(r"([`\"])([^`\"\r\n]{3,120})\1"),
re.compile(r"https?://[^\s<>]+", re.IGNORECASE),
re.compile(r"\b[\w.-]+\.(?:md|txt|csv|json|pdf|html|docx?|xlsx?)\b", re.IGNORECASE),
re.compile(
r"\b(?:titled|called|named)\s+(.+?)(?=\s+(?:with|in|so|and|for|from|that)\b|[.!?,;]|$)",
re.IGNORECASE,
),
)
def explicit_entities(text: str) -> set[str]:
"""Extract source-grounded names that a cooked flow must not replace."""
entities: set[str] = set()
for pattern in _ENTITY_PATTERNS:
for match in pattern.finditer(str(text or "")):
if pattern is _ENTITY_PATTERNS[0]:
value = match.group(2).strip()
else:
value = (match.group(1) if match.lastindex else match.group(0)).strip()
if len(value) >= 3:
entities.add(value.casefold())
return entities
def grounding_issues(seed: dict[str, Any], flow: dict[str, Any]) -> list[str]:
"""Reject synthetic flows whose private-object state contradicts the seed."""
source_turns = [str(item.get("user") or "") for item in seed.get("context") or []]
generated_turns = [str(item.get("user") or "") for item in flow.get("turns") or []]
source_text = "\n".join(source_turns)
generated_text = "\n".join(generated_turns)
issues: list[str] = []
for entity in sorted(explicit_entities(source_text)):
if entity not in generated_text.casefold():
issues.append(f"missing_source_entity:{entity}")
# Standalone flows must recreate source-created private state before use.
for index, source_turn in enumerate(source_turns[:-1]):
if not _CREATE_RE.search(source_turn):
continue
entities = explicit_entities(source_turn)
later_source = "\n".join(source_turns[index + 1:]).casefold()
for entity in entities:
if entity not in later_source:
continue
mentions = [turn for turn in generated_turns if entity in turn.casefold()]
if mentions and not _CREATE_RE.search(mentions[0]):
issues.append(f"unestablished_private_entity:{entity}")
# A source topic introduced by create/start cannot become pre-existing state.
target = str(seed.get("target_user") or "")
if _CREATE_RE.search(target):
target_entities = explicit_entities(target)
target_entities.update(
match.group(1).strip().casefold()
for match in _NEW_TOPIC_RE.finditer(target)
if len(match.group(1).strip()) >= 3
)
for entity in target_entities:
for turn in generated_turns:
if entity not in turn.casefold():
continue
if _CREATE_RE.search(turn):
break
if _LOOKUP_RE.search(turn):
issues.append(f"lookup_before_creation:{entity}")
break
return sorted(set(issues))
def redact(text: str) -> str:
"""Remove likely credentials while retaining natural request structure."""
value = str(text or "")
value = re.sub(r"hf_[A-Za-z0-9]{20,}", "[REDACTED_HF_TOKEN]", value)
value = re.sub(r"(?i)(api[_ -]?key|token|password)\s*[:=]\s*\S+", r"\1=[REDACTED]", value)
value = re.sub(r"\b(?:\d{1,3}\.){3}\d{1,3}\b", "[REDACTED_IP]", value)
return value[:1200]
def load_seeds(path: Path) -> list[dict[str, Any]]:
payload = json.loads(path.read_text(encoding="utf-8"))
seeds = payload.get("seeds") if isinstance(payload, dict) else None
if not isinstance(seeds, list):
raise RuntimeError("seed file must contain a top-level seeds array")
output = []
for seed in seeds:
if not isinstance(seed, dict) or not seed.get("seed_id"):
continue
row = dict(seed)
row["context"] = [
{"user": redact(item.get("user", ""))}
for item in (seed.get("context") or []) if isinstance(item, dict)
]
row["target_user"] = redact(seed.get("target_user", ""))
output.append(row)
return output
def completed_ids(path: Path) -> set[str]:
if not path.exists():
return set()
ids = set()
for line in path.read_text(encoding="utf-8").splitlines():
try:
row = json.loads(line)
except json.JSONDecodeError:
continue
if isinstance(row, dict) and row.get("source_seed_id"):
ids.add(str(row["source_seed_id"]))
return ids
def chunks(rows: list[dict[str, Any]], size: int) -> list[list[dict[str, Any]]]:
return [rows[index:index + size] for index in range(0, len(rows), size)]
def validate_flows(
result: Any,
wanted: set[str],
seeds: dict[str, dict[str, Any]] | None = None,
) -> dict[str, dict[str, Any]]:
rows = result.get("flows") if isinstance(result, dict) else None
valid: dict[str, dict[str, Any]] = {}
if not isinstance(rows, list):
return valid
for row in rows:
if not isinstance(row, dict):
continue
seed_id = str(row.get("source_seed_id") or "")
turns = row.get("turns")
if seed_id not in wanted or seed_id in valid:
continue
if row.get("family") not in FAMILY_SEEDS or not isinstance(turns, list) or not 2 <= len(turns) <= 4:
continue
if any(not isinstance(turn, dict) or not str(turn.get("user") or "").strip() for turn in turns):
continue
if seeds and seed_id in seeds and grounding_issues(seeds[seed_id], row):
continue
row["id"] = "sft-alex-" + re.sub(r"[^A-Za-z0-9_-]", "-", seed_id)[:72]
row["source_seed_id"] = seed_id
valid[seed_id] = row
return valid
def cook_batch(endpoint: Any, batch: list[dict[str, Any]]) -> list[dict[str, Any]]:
pending = {str(seed["seed_id"]): seed for seed in batch}
cooked: dict[str, dict[str, Any]] = {}
for _ in range(3):
if not pending:
break
result = teacher_json(endpoint, {
"task": "Turn every supplied historical seed into one fresh realistic multi-turn conversation that tests Odysseus tool use.",
"rules": [
"Return exactly one flow for every source_seed_id; never merge, omit, or duplicate seeds.",
"Preserve the seed's behavioral intent, but do not copy its wording mechanically.",
"Preserve exact names, titles, filenames, URLs, contacts, and named research topics from the source seed; never replace them with invented private objects.",
"Every generated flow is replayed independently against a clean fixture. If a later action depends on an object created earlier in the source context, include that creation before using the object.",
"Never find, open, or read an invented private object. A new note, document, task, event, skill, email, or research report must be created earlier in that generated flow.",
"Each flow has 2-4 user turns and at least one context-dependent follow-up.",
"The conversation must naturally require at least one Odysseus tool; for a general question, add an adjacent save, verify, open, or retrieve request.",
"Use natural short wording and occasional realistic misspelling, not regex-like substitutions.",
"Do not include record IDs, credentials, real email addresses, destructive shell operations, email sending, purchases, or irreversible actions.",
"Expected behavior is semantic and names the appropriate action/tool family without prescribing exact prose.",
"Choose exactly one canonical family from the supplied family list; use switching when the conversation crosses families.",
],
"schema": {"flows": [{
"source_seed_id": "exact supplied ID", "id": "short ID",
"family": "canonical family", "purpose": "behavior under test",
"turns": [{"user": "message", "expect": "semantic expected behavior"}],
}]},
"canonical_families": sorted(FAMILY_SEEDS),
"complete_odysseus_tool_catalog": compact_tool_catalog(),
"seeds": list(pending.values()),
}, max_tokens=7500, temperature=0.65)
accepted = validate_flows(result, set(pending), pending)
cooked.update(accepted)
for seed_id in accepted:
pending.pop(seed_id, None)
if pending:
raise RuntimeError(f"teacher omitted {len(pending)} seeds: {sorted(pending)[:3]}")
return [cooked[str(seed["seed_id"])] for seed in batch]
def append_rows(path: Path, rows: list[dict[str, Any]]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
with LOCK, path.open("a", encoding="utf-8") as handle:
for row in rows:
handle.write(json.dumps(row, ensure_ascii=False) + "\n")
handle.flush()
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--seeds", type=Path, default=DEFAULT_SEEDS)
parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT)
parser.add_argument("--data-dir", type=Path, default=DEFAULT_DATA)
parser.add_argument("--endpoint-id", default=DEFAULT_JUDGE_ENDPOINT)
parser.add_argument("--model", default=DEFAULT_JUDGE_MODEL)
parser.add_argument("--batch-size", type=int, default=12)
parser.add_argument("--workers", type=int, default=8)
parser.add_argument("--limit", type=int)
return parser.parse_args()
def main() -> int:
args = parse_args()
args.output.parent.mkdir(parents=True, exist_ok=True)
lock_path = args.output.with_suffix(args.output.suffix + ".lock")
lock_handle = lock_path.open("w", encoding="utf-8")
try:
fcntl.flock(lock_handle, fcntl.LOCK_EX | fcntl.LOCK_NB)
except BlockingIOError:
raise SystemExit(f"another cooker already owns {lock_path}")
endpoint = endpoint_from_db(args.data_dir, args.endpoint_id, args.model)
seeds = load_seeds(args.seeds)
done = completed_ids(args.output)
pending = [seed for seed in seeds if str(seed["seed_id"]) not in done]
if args.limit is not None:
pending = pending[:args.limit]
batches = chunks(pending, args.batch_size)
failures: list[str] = []
cooked_count = 0
with concurrent.futures.ThreadPoolExecutor(max_workers=args.workers) as pool:
future_map = {pool.submit(cook_batch, endpoint, batch): batch for batch in batches}
for future in concurrent.futures.as_completed(future_map):
batch = future_map[future]
try:
rows = future.result()
append_rows(args.output, rows)
cooked_count += len(rows)
print(json.dumps({"cooked": len(done) + cooked_count, "total": len(seeds)}), flush=True)
except Exception as exc:
failures.extend(str(seed["seed_id"]) for seed in batch)
print(json.dumps({"batch_failed": len(batch), "error": repr(exc)}), flush=True)
counts = Counter()
if args.output.exists():
for line in args.output.read_text(encoding="utf-8").splitlines():
try:
counts[json.loads(line).get("family", "unknown")] += 1
except (json.JSONDecodeError, AttributeError):
pass
print(json.dumps({
"source_seeds": len(seeds), "already_done": len(done),
"cooked_now": cooked_count, "failed": len(failures),
"remaining": len(seeds) - len(done) - cooked_count,
"families": dict(sorted(counts.items())),
}, indent=2))
return 2 if failures else 0
if __name__ == "__main__":
raise SystemExit(main())
+35 -3
View File
@@ -21,6 +21,36 @@ if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from scripts.repair_sft_corpus_with_kimi import endpoint, parse_json # noqa: E402
from src.tool_schemas import FUNCTION_TOOL_SCHEMAS # noqa: E402
ALL_TOOL_NAMES = frozenset(
str(schema.get("function", {}).get("name") or "")
for schema in FUNCTION_TOOL_SCHEMAS
if schema.get("function", {}).get("name") and schema.get("function", {}).get("name") != "host_shell"
)
def compact_tool_catalog() -> list[dict[str, Any]]:
"""Expose the complete product tool vocabulary to the scenario author."""
catalog = []
for schema in FUNCTION_TOOL_SCHEMAS:
function = schema.get("function") or {}
name = str(function.get("name") or "")
if not name or name == "host_shell":
continue
parameters = function.get("parameters") or {}
properties = parameters.get("properties") or {}
entry: dict[str, Any] = {
"name": name,
"purpose": str(function.get("description") or "")[:700],
"required": list(parameters.get("required") or []),
}
action = properties.get("action") if isinstance(properties, dict) else None
if isinstance(action, dict) and isinstance(action.get("enum"), list):
entry["actions"] = action["enum"]
catalog.append(entry)
return catalog
OWNERS = ["sft_maya_ops", "sft_jules_research", "sft_nora_design", "sft_omar_finance"]
EFFECTFUL_WITHOUT_DRY_RUN = {
@@ -107,7 +137,8 @@ For each case return:
- cleanup: fixture types that must be restored or removed
Rules:
- The source is a behavioral seed, not text to paraphrase. Preserve its useful tool strategy and outcome while changing scenario, entities, wording, and follow-up style.
- The source is behavioral evidence, not text to paraphrase and not an allowlist. Use the complete tool catalog to independently identify the best intended tool for each new turn. Preserve the useful outcome while changing scenario, entities, wording, and follow-up style.
- Distinguish tools with overlapping names by their documented purpose and required arguments. If the source used a less suitable tool, choose the catalog tool that actually fulfills the new prompt.
- Make the turns one coherent conversation. Later turns should naturally build on earlier tool results.
- Use exact IDs/titles/UIDs from the target inventory for read/update/delete workflows, or create a marker-scoped object first. Never invent an existing object.
- Give temporary objects ordinary, project-specific names that a real user might choose. Keep them distinct from supplied inventory names, but never expose run IDs, markers, fixtures, tests, audits, or cleanup mechanics to the user.
@@ -127,7 +158,7 @@ Rules:
"""
if STYLE_CONTRACT.exists():
system += "\nApply this speaking-style contract to every generated conversation:\n\n" + STYLE_CONTRACT.read_text(encoding="utf-8")
allowed_tools = sorted({tool for tool in seed["tools"]} | {"ask_user", "ui_control"})
allowed_tools = sorted(ALL_TOOL_NAMES | set(seed["tools"]))
payload = {
"model": ep["model"],
"messages": [
@@ -136,6 +167,7 @@ Rules:
"seed": compact_seed(seed),
"current_date": date.today().isoformat(),
"allowed_tools": allowed_tools,
"tool_catalog": compact_tool_catalog(),
"targets": [compact_environment(target) for target in targets],
}, ensure_ascii=False)},
],
@@ -176,7 +208,7 @@ def validate_case(
turns = raw.get("turns")
if not isinstance(turns, list) or not 3 <= len(turns) <= 4:
raise ValueError("case must contain 3-4 turns")
allowed = set(seed["tools"]) | {"ask_user", "ui_control"}
allowed = set(ALL_TOOL_NAMES) | set(seed["tools"])
clean_turns = []
normalized = set()
for index, turn in enumerate(turns, 1):
+194
View File
@@ -0,0 +1,194 @@
#!/usr/bin/env python3
"""Independently classify seeded live-replay failures with a full tool catalog."""
from __future__ import annotations
import argparse
import concurrent.futures
import json
import sys
import uuid
import urllib.request
from pathlib import Path
from typing import Any
from dotenv import load_dotenv
ROOT = Path(__file__).resolve().parents[1]
load_dotenv(ROOT / ".env")
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from scripts.generate_sft_environment_expansion import compact_tool_catalog # noqa: E402
from scripts.repair_sft_corpus_with_kimi import endpoint, parse_json # noqa: E402
def compact_evidence(result: dict[str, Any]) -> dict[str, Any]:
turns = []
for turn in result.get("turns") or []:
contract = next(
(event for event in turn.get("evidence") or [] if event.get("type") == "turn_contract"),
{},
)
outputs = [
str(event.get("output") or "")[:1200]
for event in turn.get("evidence") or []
if event.get("type") == "tool_output"
]
errors = [
event for event in turn.get("evidence") or []
if event.get("type") in {"error", "parse_error"}
]
turns.append({
"id": turn.get("id"),
"prompt": turn.get("prompt"),
"expected_tools": turn.get("expected_tools"),
"observed_tools": turn.get("observed_tools"),
"answer": str(turn.get("answer") or "")[:1800],
"deterministic_failures": turn.get("failures"),
"upstream_failed": bool(turn.get("upstream_failed", False)),
"contract": {
"capabilities": contract.get("capabilities") or [],
"required": contract.get("required") or [],
"offered": contract.get("offered") or [],
"unavailable": contract.get("unavailable") or [],
"selection_mode": contract.get("selection_mode"),
"schema_mode": contract.get("schema_mode"),
},
"tool_outputs": outputs,
"stream_errors": errors,
})
return {
"case_id": result.get("case_id"),
"seed_family_id": result.get("seed_family_id"),
"owner": result.get("owner"),
"deterministic_pass": result.get("pass"),
"deterministic_failures": result.get("failures"),
"turns": turns,
}
def judge_once(ep: dict[str, str], case: dict[str, Any], result: dict[str, Any], timeout: float) -> dict[str, Any]:
system = """You audit a real tool-agent replay. Return strict JSON only:
{"case_id":"...","case_valid":true,"overall_class":"pass|bad_generated_case|harness_routing|harness_execution|model_sft|tool_backend|mixed","confidence":0.0,"summary":"...","turns":[{"id":"...","valid_expectation":true,"best_tools":["..."],"classification":"pass|bad_generated_case|harness_routing|harness_execution|model_sft|tool_backend","reason":"...","generic_repair":"..."}]}
Use the COMPLETE tool catalog, the generated conversation, and the observed immutable turn contract.
- First decide whether the prompt and supplied environment actually support the expected tool. Reject ambiguous or invented expectations.
- harness_routing: the correct family/tool was absent, the wrong family was required, or the contract offered zero/wrong tools.
- harness_execution: the contract selected the correct deterministic operation but failed to execute/render it independently of model choice.
- model_sft: the correct tools were offered and executable, but the model chose the wrong tool/action, malformed arguments, leaked reasoning, or falsely answered.
- tool_backend: a correct call failed in the underlying service.
- Do not propose phrase-specific rules. Generic repairs must describe a semantic boundary or contract invariant.
- A prior turn's successful result can establish references for a follow-up. An active document fixture means deictic editing prompts may validly target document tools.
- Judge the complete 3-4 turn trajectory. If an earlier failed operation removed the object or evidence needed later, mark later failures as causal fallout in the reason instead of inventing another root cause.
- Recommend a harness patch only for a semantic category that should generalize across varied wording and entities. Never recommend a literal prompt/entity/domain-name rule. A single case can justify only a clear contract, authorization, or security invariant; otherwise request more variants.
- Do not reveal or reconstruct hidden benchmark answers. Judge only the supplied synthetic replay.
"""
payload = {
"model": ep["model"],
"messages": [
{"role": "system", "content": system},
{"role": "user", "content": json.dumps({
"tool_catalog": compact_tool_catalog(),
"generated_case": case,
"live_result": compact_evidence(result),
}, ensure_ascii=False)},
],
"temperature": 0,
"max_tokens": 5000,
"response_format": {"type": "json_object"},
}
request = urllib.request.Request(
ep["base_url"].rstrip("/") + "/chat/completions",
data=json.dumps(payload).encode(),
headers={"Content-Type": "application/json", "Authorization": f"Bearer {ep['api_key']}"},
method="POST",
)
with urllib.request.urlopen(request, timeout=timeout) as response:
body = json.loads(response.read().decode())
message = body["choices"][0]["message"]
verdict = parse_json(str(message.get("content") or message.get("reasoning_content") or ""))
if str(verdict.get("case_id") or "") != str(result.get("case_id") or ""):
raise ValueError("judge returned the wrong case_id")
return verdict
def judge(
ep: dict[str, str],
case: dict[str, Any],
result: dict[str, Any],
timeout: float,
retries: int,
) -> dict[str, Any]:
"""Retry provider/JSON failures without changing the case being judged."""
last_error: Exception | None = None
for _attempt in range(max(0, retries) + 1):
try:
return judge_once(ep, case, result, timeout)
except Exception as exc:
last_error = exc
assert last_error is not None
raise last_error
def atomic_write(path: Path, payload: dict[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp")
temporary.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
temporary.replace(path)
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--cases", type=Path, required=True)
parser.add_argument("--results", type=Path, required=True)
parser.add_argument("--out", type=Path, required=True)
parser.add_argument("--endpoint-id", default="e17d4b33")
parser.add_argument("--model", default="deepseek-v4-pro")
parser.add_argument("--workers", type=int, default=4)
parser.add_argument("--timeout", type=float, default=180)
parser.add_argument("--retries", type=int, default=2)
parser.add_argument("--case-id", action="append", help="Judge only the named case; repeatable")
args = parser.parse_args()
cases = {row["case_id"]: row for row in json.loads(args.cases.read_text(encoding="utf-8"))["cases"]}
results = json.loads(args.results.read_text(encoding="utf-8"))["results"]
if args.case_id:
wanted = set(args.case_id)
results = [row for row in results if row["case_id"] in wanted]
ep = endpoint(args.endpoint_id, args.model)
verdicts: dict[str, dict[str, Any]] = {}
errors: list[dict[str, str]] = []
with concurrent.futures.ThreadPoolExecutor(max_workers=max(1, args.workers)) as pool:
futures = {
pool.submit(
judge,
ep,
cases[result["case_id"]],
result,
args.timeout,
args.retries,
): result
for result in results
}
for future in concurrent.futures.as_completed(futures):
result = futures[future]
case_id = str(result["case_id"])
try:
verdicts[case_id] = future.result()
print(f"judged {case_id}: {verdicts[case_id].get('overall_class')}", flush=True)
except Exception as exc:
errors.append({"case_id": case_id, "error": repr(exc)})
print(f"failed {case_id}: {exc!r}", flush=True)
atomic_write(args.out, {"verdicts": list(verdicts.values()), "errors": errors})
ordered = [verdicts[row["case_id"]] for row in results if row["case_id"] in verdicts]
atomic_write(args.out, {"verdicts": ordered, "errors": errors})
counts: dict[str, int] = {}
for row in ordered:
key = str(row.get("overall_class") or "unknown")
counts[key] = counts.get(key, 0) + 1
print(json.dumps({"judged": len(ordered), "errors": len(errors), "classes": counts}, indent=2))
if __name__ == "__main__":
main()
+282
View File
@@ -0,0 +1,282 @@
#!/usr/bin/env python3
"""Snapshot or restore durable state for one Odysseus SFT fixture owner.
Sessions and chat messages are intentionally excluded so replay evidence keeps
working. Only owner-scoped tool data and its dependent rows are managed.
"""
from __future__ import annotations
import argparse
import base64
import json
import re
import shutil
import sqlite3
import time
from pathlib import Path
from typing import Any
DIRECT_TABLES = (
"notes",
"memories",
"scheduled_tasks",
"documents",
"calendars",
"editor_drafts",
"notification_logs",
"caldav_deleted_events",
)
CHILD_TABLES = {
"document_versions": ("documents", "document_id", "id"),
"task_runs": ("scheduled_tasks", "task_id", "id"),
"calendar_events": ("calendars", "calendar_id", "id"),
}
def _read_json(path: Path, default: Any) -> Any:
try:
return json.loads(path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError):
return default
def _atomic_json(path: Path, payload: Any) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
temporary = path.with_name(f".{path.name}.fixture-state.tmp")
temporary.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
temporary.replace(path)
def _skill_owner(path: Path) -> str:
try:
text = path.read_text(encoding="utf-8")
except (OSError, UnicodeDecodeError):
return ""
match = re.search(r'^owner:\s*["\']?([^"\'\n#]+)', text, re.M)
return match.group(1).strip() if match else ""
def _snapshot_external(data_dir: Path, owner: str) -> dict[str, Any]:
prefs = _read_json(data_dir / "user_prefs.json", {"_users": {}})
blocked = _read_json(data_dir / "email_blocked_senders.json", {"owners": {}})
email_payload = _read_json(data_dir / "fixture_email_messages.json", {"messages": []})
email_rows = email_payload.get("messages", []) if isinstance(email_payload, dict) else email_payload
skills_root = data_dir / "skills"
skill_files: list[dict[str, str]] = []
skill_dirs: list[str] = []
if skills_root.exists():
for skill_md in skills_root.rglob("SKILL.md"):
if _skill_owner(skill_md) != owner:
continue
directory = skill_md.parent
skill_dirs.append(str(directory.relative_to(skills_root)))
for path in directory.rglob("*"):
if path.is_file():
skill_files.append({
"path": str(path.relative_to(skills_root)),
"base64": base64.b64encode(path.read_bytes()).decode("ascii"),
})
usage = _read_json(skills_root / "_usage.json", {})
return {
"prefs_present": owner in ((prefs.get("_users") or {}) if isinstance(prefs, dict) else {}),
"prefs": ((prefs.get("_users") or {}).get(owner) if isinstance(prefs, dict) else None),
"blocked_present": owner in ((blocked.get("owners") or {}) if isinstance(blocked, dict) else {}),
"blocked_senders": ((blocked.get("owners") or {}).get(owner) if isinstance(blocked, dict) else None),
"email_rows": [
row for row in (email_rows if isinstance(email_rows, list) else [])
if isinstance(row, dict) and str(row.get("owner") or "") == owner
],
"skill_dirs": sorted(set(skill_dirs)),
"skill_files": skill_files,
"skill_usage": {
key: value for key, value in (usage.items() if isinstance(usage, dict) else [])
if str(key).startswith(f"{owner}::")
},
}
def _table_exists(db: sqlite3.Connection, table: str) -> bool:
return db.execute(
"SELECT 1 FROM sqlite_master WHERE type='table' AND name=?", (table,),
).fetchone() is not None
def _columns(db: sqlite3.Connection, table: str) -> list[str]:
return [str(row[1]) for row in db.execute(f'PRAGMA table_info("{table}")')]
def _rows(db: sqlite3.Connection, table: str, where: str, values: tuple[Any, ...]) -> list[dict[str, Any]]:
db.row_factory = sqlite3.Row
return [dict(row) for row in db.execute(f'SELECT * FROM "{table}" WHERE {where}', values)]
def snapshot_owner(db_path: Path, owner: str, data_dir: Path | None = None) -> dict[str, Any]:
db = sqlite3.connect(db_path)
try:
tables: dict[str, list[dict[str, Any]]] = {}
for table in DIRECT_TABLES:
if _table_exists(db, table) and "owner" in _columns(db, table):
tables[table] = _rows(db, table, '"owner"=?', (owner,))
for table, (parent, foreign_key, parent_key) in CHILD_TABLES.items():
if not _table_exists(db, table):
continue
parent_ids = [row[parent_key] for row in tables.get(parent, [])]
if not parent_ids:
tables[table] = []
continue
placeholders = ",".join("?" for _ in parent_ids)
tables[table] = _rows(
db, table, f'"{foreign_key}" IN ({placeholders})', tuple(parent_ids),
)
return {
"format": "odysseus-owner-fixture-v1",
"created_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
"source_db": str(db_path),
"owner": owner,
"tables": tables,
"counts": {table: len(rows) for table, rows in tables.items()},
"external": _snapshot_external(data_dir or db_path.parent, owner),
}
finally:
db.close()
def _delete_owner_rows(db: sqlite3.Connection, owner: str) -> None:
for table, (parent, foreign_key, parent_key) in CHILD_TABLES.items():
if not (_table_exists(db, table) and _table_exists(db, parent)):
continue
db.execute(
f'DELETE FROM "{table}" WHERE "{foreign_key}" IN '
f'(SELECT "{parent_key}" FROM "{parent}" WHERE "owner"=?)',
(owner,),
)
for table in DIRECT_TABLES:
if _table_exists(db, table) and "owner" in _columns(db, table):
db.execute(f'DELETE FROM "{table}" WHERE "owner"=?', (owner,))
def _restore_external(data_dir: Path, external: dict[str, Any], owner: str) -> None:
prefs_path = data_dir / "user_prefs.json"
prefs = _read_json(prefs_path, {"_users": {}})
users = prefs.setdefault("_users", {})
if external.get("prefs_present"):
users[owner] = external.get("prefs")
else:
users.pop(owner, None)
_atomic_json(prefs_path, prefs)
blocked_path = data_dir / "email_blocked_senders.json"
blocked = _read_json(blocked_path, {"owners": {}})
blocked_owners = blocked.setdefault("owners", {})
if external.get("blocked_present"):
blocked_owners[owner] = external.get("blocked_senders")
else:
blocked_owners.pop(owner, None)
_atomic_json(blocked_path, blocked)
email_path = data_dir / "fixture_email_messages.json"
email_payload = _read_json(email_path, {"messages": []})
email_rows = email_payload.get("messages", []) if isinstance(email_payload, dict) else email_payload
retained = [
row for row in (email_rows if isinstance(email_rows, list) else [])
if not (isinstance(row, dict) and str(row.get("owner") or "") == owner)
]
restored_rows = retained + list(external.get("email_rows") or [])
if isinstance(email_payload, dict):
email_payload["messages"] = restored_rows
else:
email_payload = restored_rows
_atomic_json(email_path, email_payload)
skills_root = data_dir / "skills"
if skills_root.exists():
for skill_md in list(skills_root.rglob("SKILL.md")):
if _skill_owner(skill_md) == owner:
shutil.rmtree(skill_md.parent, ignore_errors=True)
for entry in external.get("skill_files") or []:
relative = Path(str(entry.get("path") or ""))
if not relative.parts or relative.is_absolute() or ".." in relative.parts:
raise ValueError("unsafe skill path in fixture snapshot")
destination = skills_root / relative
destination.parent.mkdir(parents=True, exist_ok=True)
destination.write_bytes(base64.b64decode(entry.get("base64") or ""))
usage_path = skills_root / "_usage.json"
usage = _read_json(usage_path, {})
usage = usage if isinstance(usage, dict) else {}
usage = {key: value for key, value in usage.items() if not str(key).startswith(f"{owner}::")}
usage.update(external.get("skill_usage") or {})
_atomic_json(usage_path, usage)
def restore_owner(target_db: Path, snapshot: dict[str, Any], owner: str,
data_dir: Path | None = None) -> None:
if snapshot.get("format") != "odysseus-owner-fixture-v1":
raise ValueError("unsupported fixture snapshot format")
if str(snapshot.get("owner") or "") != owner:
raise ValueError("snapshot owner does not match requested owner")
tables = snapshot.get("tables")
if not isinstance(tables, dict):
raise ValueError("snapshot has no tables")
db = sqlite3.connect(target_db, timeout=60)
try:
db.execute("BEGIN IMMEDIATE")
_delete_owner_rows(db, owner)
insertion_order = (*DIRECT_TABLES, *CHILD_TABLES)
for table in insertion_order:
rows = tables.get(table) or []
if not rows or not _table_exists(db, table):
continue
target_columns = set(_columns(db, table))
columns = [column for column in rows[0] if column in target_columns]
quoted = ",".join(f'"{column}"' for column in columns)
placeholders = ",".join("?" for _ in columns)
db.executemany(
f'INSERT INTO "{table}" ({quoted}) VALUES ({placeholders})',
[[row.get(column) for column in columns] for row in rows],
)
db.commit()
except Exception:
db.rollback()
raise
finally:
db.close()
external = snapshot.get("external")
if isinstance(external, dict):
_restore_external(data_dir or target_db.parent, external, owner)
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--db", type=Path, required=True, help="Live target app.db")
parser.add_argument("--data-dir", type=Path, help="External fixture state directory; defaults to DB parent")
parser.add_argument("--owner", default="sft_alex_creator")
parser.add_argument("--snapshot-out", type=Path)
parser.add_argument("--restore-json", type=Path)
parser.add_argument("--restore-from-db", type=Path)
args = parser.parse_args()
operations = sum(bool(value) for value in (
args.snapshot_out, args.restore_json, args.restore_from_db,
))
if operations != 1:
parser.error("choose exactly one of --snapshot-out, --restore-json, or --restore-from-db")
if args.snapshot_out:
payload = snapshot_owner(args.db, args.owner, args.data_dir)
args.snapshot_out.parent.mkdir(parents=True, exist_ok=True)
args.snapshot_out.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
print(json.dumps({"snapshot": str(args.snapshot_out), "counts": payload["counts"]}, indent=2))
return
if args.restore_json:
payload = json.loads(args.restore_json.read_text(encoding="utf-8"))
else:
payload = snapshot_owner(args.restore_from_db, args.owner, args.data_dir)
restore_owner(args.db, payload, args.owner, args.data_dir)
print(json.dumps({"restored_owner": args.owner, "counts": payload["counts"]}, indent=2))
if __name__ == "__main__":
main()
File diff suppressed because it is too large Load Diff
+13 -2
View File
@@ -14,8 +14,10 @@ from pathlib import Path
from typing import Any
from cryptography.fernet import Fernet
from dotenv import load_dotenv
ROOT = Path(__file__).resolve().parents[1]
load_dotenv(ROOT / ".env")
def decrypt(value: str) -> str:
@@ -26,7 +28,13 @@ def decrypt(value: str) -> str:
def endpoint(endpoint_id: str, model: str) -> dict[str, str]:
con = sqlite3.connect(ROOT / "data" / "app.db")
# Honor the same configured data directory as the live Odysseus service.
# Eval worktrees commonly keep only source under ROOT while 7011 points at
# the canonical shared database via ODYSSEUS_DATA_DIR.
from src.constants import DATA_DIR
data_dir = Path(DATA_DIR)
con = sqlite3.connect(data_dir / "app.db")
con.row_factory = sqlite3.Row
row = con.execute(
"SELECT base_url,api_key FROM model_endpoints WHERE id=? AND is_enabled=1",
@@ -34,7 +42,10 @@ def endpoint(endpoint_id: str, model: str) -> dict[str, str]:
).fetchone()
if row is None:
raise RuntimeError(f"Enabled endpoint not found: {endpoint_id}")
return {"base_url": row["base_url"], "api_key": decrypt(row["api_key"]), "model": model}
value = str(row["api_key"] or "")
if value.startswith("enc:"):
value = Fernet((data_dir / ".app_key").read_bytes()).decrypt(value[4:].encode()).decode()
return {"base_url": row["base_url"], "api_key": value, "model": model}
def parse_json(text: str) -> dict[str, Any]:
+266 -10
View File
@@ -2,22 +2,25 @@
"""Execute generated SFT workflows through Odysseus with rollback and gating."""
from __future__ import annotations
import os
import argparse
import contextlib
import json
import os
import re
import shutil
import signal
import time
import uuid
from datetime import datetime, timedelta
from pathlib import Path
from typing import Any
import httpx
from dotenv import load_dotenv
ROOT = Path(__file__).resolve().parents[1]
load_dotenv(ROOT / ".env")
if str(ROOT) not in __import__("sys").path:
__import__("sys").path.insert(0, str(ROOT))
@@ -37,12 +40,15 @@ from scripts.eval_odysseus_tool_use import ( # noqa: E402
_visible_event_text,
)
DATA_DIR = ROOT / "data"
from src.constants import DATA_DIR as CONFIGURED_DATA_DIR # noqa: E402
DATA_DIR = Path(CONFIGURED_DATA_DIR)
BAD_ANSWER_RE = re.compile(
r"\b(?:can't|cannot|don't have|do not have|not available|no .*tool|enable .*integration|"
r"invalid credentials|not authenticated|i can only|i'm unable)\b",
re.I,
)
_COOKIE_CACHE: dict[str, str] = {}
TOOL_FAILURE_RE = re.compile(r"(?:tool (?:failed|error)|exit_code[^\d]*[1-9]|permission denied|not found)", re.I)
INTERNAL_NARRATION_RE = re.compile(
r"(?:^|\n)(?:The user (?:asks|asked|wants)|I (?:should|need to|can see)|Let me (?:call|use|retry|try))\b",
@@ -65,7 +71,158 @@ def atomic_json(path: Path, payload: Any) -> None:
temp.replace(path)
def install_fixture_environments(path: Path) -> dict[str, int]:
"""Materialize an inventory snapshot for an isolated replay app."""
payload = json.loads(path.read_text(encoding="utf-8"))
environments = payload.get("environments", []) if isinstance(payload, dict) else []
messages: list[dict[str, Any]] = []
counts = {
"emails": 0, "notes": 0, "memories": 0, "documents": 0,
"tasks": 0, "calendars": 0, "events": 0,
}
db = SessionLocal()
owners = [
str(row.get("owner") or "").strip()
for row in environments if isinstance(row, dict)
]
try:
for owner in filter(None, owners):
document_ids = [
value[0] for value in db.query(Document.id).filter(Document.owner == owner).all()
]
if document_ids:
db.query(DocumentVersion).filter(
DocumentVersion.document_id.in_(document_ids)
).delete(synchronize_session=False)
calendar_ids = [
value[0] for value in db.query(CalendarCal.id).filter(CalendarCal.owner == owner).all()
]
if calendar_ids:
db.query(CalendarEvent).filter(
CalendarEvent.calendar_id.in_(calendar_ids)
).delete(synchronize_session=False)
db.query(Document).filter(Document.owner == owner).delete(synchronize_session=False)
db.query(Note).filter(Note.owner == owner).delete(synchronize_session=False)
db.query(Memory).filter(Memory.owner == owner).delete(synchronize_session=False)
db.query(ScheduledTask).filter(ScheduledTask.owner == owner).delete(synchronize_session=False)
db.query(CalendarCal).filter(CalendarCal.owner == owner).delete(synchronize_session=False)
for environment in environments:
if not isinstance(environment, dict):
continue
owner = str(environment.get("owner") or "").strip()
for row in environment.get("notes") or []:
db.add(Note(
id=str(row.get("id") or uuid.uuid4()), owner=owner,
title=str(row.get("title") or ""), content=str(row.get("content") or ""),
note_type=str(row.get("type") or "note"), label=row.get("label"),
archived=False, source="user",
))
counts["notes"] += 1
for row in environment.get("memories") or []:
db.add(Memory(
id=str(row.get("id") or uuid.uuid4()), owner=owner,
text=str(row.get("text") or ""),
category=str(row.get("category") or "fact"), source="user",
))
counts["memories"] += 1
for row in environment.get("documents") or []:
document_id = str(row.get("id") or uuid.uuid4())
content = str(row.get("content") or "")
db.add(Document(
id=document_id, owner=owner, title=str(row.get("title") or "Untitled"),
language=str(row.get("language") or "text"), current_content=content,
version_count=1, is_active=True, archived=False,
))
db.add(DocumentVersion(
id=str(uuid.uuid4()), document_id=document_id, version_number=1,
content=content, summary="Isolated replay fixture", source="user",
))
counts["documents"] += 1
for row in environment.get("tasks") or []:
db.add(ScheduledTask(
id=str(row.get("id") or uuid.uuid4()), owner=owner,
name=str(row.get("name") or "Untitled Task"),
status=str(row.get("status") or "active"),
schedule=row.get("schedule"), task_type="llm",
))
counts["tasks"] += 1
calendar_map: dict[str, str] = {}
for row in environment.get("calendars") or []:
calendar_id = str(row.get("id") or uuid.uuid4())
calendar_map[calendar_id] = calendar_id
db.add(CalendarCal(
id=calendar_id, owner=owner, name=str(row.get("name") or "Personal"),
source=str(row.get("source") or "local"),
))
counts["calendars"] += 1
default_calendar = next(iter(calendar_map), None)
for row in environment.get("events") or []:
if default_calendar is None:
default_calendar = str(uuid.uuid4())
db.add(CalendarCal(
id=default_calendar, owner=owner, name="Personal", source="local",
))
counts["calendars"] += 1
start = datetime.fromisoformat(str(row.get("start") or "").replace("Z", "+00:00"))
db.add(CalendarEvent(
uid=str(row.get("uid") or uuid.uuid4()), calendar_id=default_calendar,
summary=str(row.get("summary") or ""), dtstart=start,
dtend=start + timedelta(hours=1), all_day=bool(row.get("all_day")),
))
counts["events"] += 1
db.commit()
except Exception:
db.rollback()
raise
finally:
db.close()
for environment in environments:
if not isinstance(environment, dict):
continue
owner = str(environment.get("owner") or "").strip()
profile = environment.get("profile") if isinstance(environment.get("profile"), dict) else {}
primary_name = str(profile.get("primary_account") or "Primary Inbox")
secondary_name = str(profile.get("secondary_account") or "Secondary Inbox")
for source in environment.get("emails") or []:
if not isinstance(source, dict):
continue
row = dict(source)
account = str(row.get("account") or primary_name)
secondary = account == secondary_name
row.update({
"owner": owner,
"account": account,
"account_email": str(profile.get("secondary" if secondary else "primary") or owner),
"account_id": "secondary-inbox" if secondary else "primary-inbox",
"folder": str(row.get("folder") or "INBOX"),
"body": str(row.get("body") or (
f"Fixture message for: {row.get('subject') or '(no subject)'}. "
"Please review the referenced materials and reply with the next step."
)),
})
messages.append(row)
atomic_json(DATA_DIR / "fixture_email_messages.json", {"messages": messages})
counts["emails"] = len(messages)
return counts
def login(client: httpx.Client, base_url: str, owner: str, password: str) -> None:
token = _COOKIE_CACHE.get(owner)
if not token:
sessions_path = DATA_DIR / "sessions.json"
if sessions_path.exists():
with contextlib.suppress(Exception):
sessions = json.loads(sessions_path.read_text(encoding="utf-8"))
token = next(
key for key, value in reversed(list(sessions.items()))
if isinstance(value, dict) and value.get("username") == owner
)
if token:
_COOKIE_CACHE[owner] = token
client.cookies.set("odysseus_session", token)
return
response = client.post(
base_url.rstrip("/") + "/api/auth/login",
json={"username": owner, "password": password, "remember": True},
@@ -74,6 +231,9 @@ def login(client: httpx.Client, base_url: str, owner: str, password: str) -> Non
_raise_for_status_with_body(response)
if not response.json().get("ok"):
raise RuntimeError(f"login failed for {owner}")
token = client.cookies.get("odysseus_session")
if token:
_COOKIE_CACHE[owner] = token
def create_session(client: httpx.Client, args: argparse.Namespace, case: dict[str, Any]) -> str:
@@ -94,7 +254,8 @@ def create_session(client: httpx.Client, args: argparse.Namespace, case: dict[st
def stream_turn(
client: httpx.Client, args: argparse.Namespace, session_id: str, prompt: str
client: httpx.Client, args: argparse.Namespace, session_id: str, prompt: str,
*, active_doc_id: str = "",
) -> tuple[list[dict[str, Any]], str]:
events: list[dict[str, Any]] = []
text: list[str] = []
@@ -106,10 +267,15 @@ def stream_turn(
"selected_endpoint_id": args.endpoint_id,
"selected_endpoint_url": args.endpoint,
"selected_model": args.model,
"thinking_mode": args.thinking_mode,
"client_runtime_context": json.dumps(
{"timezone": args.timezone, "tz_offset_min": args.tz_offset_min}, separators=(",", ":")
),
}
if active_doc_id:
form["active_doc_id"] = active_doc_id
if getattr(args, "allow_web_search", False):
form["allow_web_search"] = "true"
with client.stream(
"POST",
args.base_url.rstrip("/") + "/api/chat_stream",
@@ -154,6 +320,46 @@ def tool_outputs(events: list[dict[str, Any]]) -> str:
return "\n".join(str(e.get("output") or "") for e in events if e.get("type") == "tool_output")
def has_unrecovered_tool_failure(events: list[dict[str, Any]]) -> bool:
"""Count a tool failure only when that tool never subsequently succeeds."""
pending: set[str] = set()
for event in events:
if event.get("type") != "tool_output":
continue
name = normalized_tool(str(event.get("tool") or "unknown"))
output = str(event.get("output") or "")
failed = bool(
event.get("error")
or event.get("exit_code") not in (None, 0)
or TOOL_FAILURE_RE.search(output)
)
if failed:
pending.add(name)
else:
pending.discard(name)
return bool(pending)
def compact_evidence(events: list[dict[str, Any]]) -> list[dict[str, Any]]:
"""Keep routing and execution evidence without bloating the replay report."""
retained = {
"turn_contract", "tool_start", "tool_output", "tool_resolution_audit",
"error", "parse_error", "metrics", "final_response",
}
rows = []
for event in events:
if event.get("type") not in retained:
continue
row = dict(event)
for key in ("output", "text", "delta"):
if isinstance(row.get(key), str) and len(row[key]) > 3000:
row[key] = row[key][:3000] + "..."
if row.get("type") == "turn_contract":
row.pop("executable", None)
rows.append(row)
return rows
def tool_actions(events: list[dict[str, Any]], tool_name: str) -> set[str]:
actions: set[str] = set()
for event in events:
@@ -200,6 +406,11 @@ def score_turn(turn: dict[str, Any], events: list[dict[str, Any]], answer: str)
failures: list[str] = []
names = tool_names(events)
expected = {normalized_tool(str(name)) for name in turn.get("expected_tools") or []}
# Both document writers satisfy a requested active-draft mutation. Which
# one is most efficient depends on how much of the draft the model changes;
# exact-name imitation is not a functional correctness requirement.
if expected & {"edit_document", "update_document"}:
expected.update({"edit_document", "update_document"})
if expected and not expected.intersection(names):
failures.append(f"missing_acceptable_tool expected={sorted(expected)} got={names}")
for tool_name, expected_actions in inferred_expected_actions(turn).items():
@@ -213,8 +424,7 @@ def score_turn(turn: dict[str, Any], events: list[dict[str, Any]], answer: str)
failures.append("stream_error")
if BAD_ANSWER_RE.search(answer):
failures.append("tool_unavailable_answer")
output = tool_outputs(events)
if TOOL_FAILURE_RE.search(output):
if has_unrecovered_tool_failure(events):
failures.append("tool_output_failure")
if INTERNAL_NARRATION_RE.search(answer):
failures.append("internal_narration_leaked")
@@ -375,10 +585,11 @@ def marker_fields(value: Any, marker: str) -> Any:
return value
def apply_fixture_plan(case: dict[str, Any], owner: str, session_id: str, marker: str) -> None:
def apply_fixture_plan(case: dict[str, Any], owner: str, session_id: str, marker: str) -> dict[str, str]:
"""Create only owner-scoped local fixtures required before the first turn."""
first_tools = set((case.get("turns") or [{}])[0].get("expected_tools") or [])
db = SessionLocal()
context: dict[str, str] = {}
try:
for fixture in case.get("fixture_plan") or []:
if not isinstance(fixture, dict):
@@ -407,6 +618,7 @@ def apply_fixture_plan(case: dict[str, Any], owner: str, session_id: str, marker
summary="Expansion fixture",
source="user",
))
context["active_doc_id"] = document_id
elif fixture_type == "note":
db.add(Note(
id=str(uuid.uuid4()),
@@ -426,6 +638,7 @@ def apply_fixture_plan(case: dict[str, Any], owner: str, session_id: str, marker
raise
finally:
db.close()
return context
def delete_session(client: httpx.Client, base_url: str, session_id: str) -> None:
@@ -482,10 +695,14 @@ def run_case(args: argparse.Namespace, case: dict[str, Any]) -> dict[str, Any]:
snapshot.capture()
login(client, args.base_url, owner, args.password)
session_id = create_session(client, args, case)
apply_fixture_plan(case, owner, session_id, marker)
fixture_context = apply_fixture_plan(case, owner, session_id, marker)
upstream_failed = False
for turn in case["turns"]:
prompt = str(turn["prompt"]).replace("{marker}", marker)
events, answer = stream_turn(client, args, session_id, prompt)
events, answer = stream_turn(
client, args, session_id, prompt,
active_doc_id=fixture_context.get("active_doc_id", ""),
)
turn_failures = score_turn(turn, events, answer)
turns_out.append({
"id": turn["id"],
@@ -494,10 +711,14 @@ def run_case(args: argparse.Namespace, case: dict[str, Any]) -> dict[str, Any]:
"observed_tools": tool_names(events),
"answer": answer,
"failures": turn_failures,
"evidence": compact_evidence(events),
"upstream_failed": upstream_failed,
})
failures.extend(f"{turn['id']}:{failure}" for failure in turn_failures)
if turn_failures:
break
# Keep executing the full 3-4 turn trajectory. Later misses may be
# causal fallout from an earlier failed create/read, so the judge
# receives this marker and can separate root causes from cascades.
upstream_failed = upstream_failed or bool(turn_failures)
except Exception as exc:
failures.append(f"exception:{exc!r}")
finally:
@@ -538,6 +759,20 @@ def main() -> None:
parser.add_argument("--endpoint-id", default="f3904562")
parser.add_argument("--endpoint", default="https://openrouter.ai/api/v1/chat/completions")
parser.add_argument("--model", default="moonshotai/kimi-k3")
parser.add_argument("--thinking-mode", choices=("on", "off"), default="off")
parser.add_argument(
"--allow-web-search",
action="store_true",
help="Enable Odysseus public web_search/web_fetch for this replay.",
)
parser.add_argument(
"--fixture-environments",
type=Path,
help=(
"Install owner-scoped synthetic inventory rows for replay. "
"Use only against an isolated app with ODYSSEUS_EMAIL_FIXTURE=1."
),
)
parser.add_argument("--turn-timeout", type=float, default=180)
parser.add_argument("--case-timeout", type=float, default=600)
parser.add_argument("--timezone", default="Asia/Tokyo")
@@ -545,13 +780,34 @@ def main() -> None:
parser.add_argument("--limit", type=int)
parser.add_argument("--owner", action="append")
parser.add_argument("--case-id", action="append")
parser.add_argument(
"--one-per-seed",
action="store_true",
help="Run the first validated environment variant for each source seed family",
)
args = parser.parse_args()
if args.fixture_environments:
if os.environ.get("ODYSSEUS_EMAIL_FIXTURE") != "1":
parser.error("--fixture-environments requires ODYSSEUS_EMAIL_FIXTURE=1")
installed = install_fixture_environments(args.fixture_environments)
print(f"installed isolated fixture inventory: {installed}", flush=True)
cases = json.loads(args.cases.read_text(encoding="utf-8"))["cases"]
if args.owner:
cases = [case for case in cases if case["owner"] in set(args.owner)]
if args.case_id:
cases = [case for case in cases if case["case_id"] in set(args.case_id)]
if args.one_per_seed:
seen_seeds: set[str] = set()
first_cases = []
for case in cases:
seed_id = str(case.get("seed_family_id") or "")
if seed_id in seen_seeds:
continue
seen_seeds.add(seed_id)
first_cases.append(case)
cases = first_cases
if args.limit:
cases = cases[: args.limit]
existing = {row["case_id"]: row for row in json.loads(args.out.read_text(encoding="utf-8")).get("results", [])} if args.out.exists() else {}
+37 -3
View File
@@ -60,7 +60,7 @@ try {
const events = parseSSE(await response.text());
const contract = events.find(x => x.type === 'turn_contract') || {};
const tools = events.filter(x => x.type === 'tool_start').map(x => bare(x.tool));
const outputs = events.filter(x => x.type === 'tool_output').map(x => ({ tool: bare(x.tool), exit_code: x.exit_code ?? null, error: Boolean(x.error) }));
const outputs = events.filter(x => x.type === 'tool_output').map(x => ({ tool: bare(x.tool), command: x.command || '', exit_code: x.exit_code ?? null, error: Boolean(x.error) }));
const final = events.filter(x => x.type === 'final_response').map(x => x.content || '').join('') || events.filter(x => typeof x.delta === 'string').map(x => x.delta).join('');
return { response, contract, tools, outputs, final };
};
@@ -68,11 +68,18 @@ try {
const browserSession = await makeSession('deliberate');
await openSession(browserSession);
const opened = await send('Browse https://example.com and take a snapshot. Report the rendered page heading.');
await page.locator('.private-browser-preview-img[src^="data:image/"]').last().waitFor({ state: 'visible', timeout: 10000 });
const screenshotState = await page.locator('.private-browser-preview-img[src^="data:image/"]').last().evaluate(img => ({
complete: img.complete,
naturalWidth: img.naturalWidth,
sourceLength: img.getAttribute('src')?.length || 0,
}));
const openChecks = {
http_ok: opened.response.ok(), clean_route: opened.contract.selection_mode === 'clean_compact_v3_preview',
offered_private_browser: (opened.contract.offered || []).some(x => bare(x) === 'private_browser'),
browser_only: opened.tools.length >= 1 && opened.tools.every(x => x === 'private_browser'),
tool_success: opened.outputs.some(x => x.tool === 'private_browser' && !x.error && (x.exit_code == null || x.exit_code === 0)),
screenshot_visible: screenshotState.complete && screenshotState.naturalWidth > 0 && screenshotState.sourceLength > 100,
grounded: /example domain/i.test(opened.final), no_reasoning_leak: noLeak(opened.final),
};
report.turns.push({ kind: 'domain-browse-snapshot-web-off', tools: opened.tools, outputs: opened.outputs, offered_private_browser: openChecks.offered_private_browser, checks: openChecks, status: Object.values(openChecks).every(Boolean) ? 'passed' : 'failed' }); save();
@@ -97,6 +104,33 @@ try {
};
report.turns.push({ kind: 'typed-evidence-follow-up-web-off', tools: follow.tools, outputs: follow.outputs, offered_private_browser: followChecks.warm_private_browser, offered: (follow.contract.offered || []).map(bare), unavailable: follow.contract.unavailable || [], active_capabilities: follow.contract.active_capabilities || [], checks: followChecks, status: Object.values(followChecks).every(Boolean) ? 'passed' : 'failed' }); save();
const mapsSession = await makeSession('plain-open-preview');
await openSession(mapsSession);
const maps = await send('Browse Google Maps and find the closest coffee shop to Todoroki Station.');
await page.locator('.private-browser-preview-img[src^="data:image/"]').last().waitFor({ state: 'visible', timeout: 10000 });
const mapsScreenshot = await page.locator('.private-browser-preview-img[src^="data:image/"]').last().evaluate(img => ({
complete: img.complete,
naturalWidth: img.naturalWidth,
sourceLength: img.getAttribute('src')?.length || 0,
}));
const mapsChecks = {
http_ok: maps.response.ok(),
browser_used: maps.tools.includes('private_browser'),
screenshot_visible: mapsScreenshot.complete && mapsScreenshot.naturalWidth > 0 && mapsScreenshot.sourceLength > 100,
no_reasoning_leak: noLeak(maps.final),
};
report.turns.push({ kind: 'plain-open-renders-screenshot', tools: maps.tools, checks: mapsChecks, status: Object.values(mapsChecks).every(Boolean) ? 'passed' : 'failed' }); save();
const menu = await send('Which one has a grilled cheese sandwich on the menu?');
const menuCommands = menu.outputs.map(item => String(item.command || '').toLowerCase());
const menuChecks = {
http_ok: menu.response.ok(),
web_followup_used: menu.tools.some(tool => ['private_browser', 'web_search', 'web_fetch'].includes(tool)),
prior_subject_retained: menuCommands.some(command => /todoroki|coffee shop|peak by swell|yeti roastery|toe coffee/.test(command)),
no_reasoning_leak: noLeak(menu.final),
};
report.turns.push({ kind: 'maps-result-property-followup', tools: menu.tools, commands: menuCommands, checks: menuChecks, status: Object.values(menuChecks).every(Boolean) ? 'passed' : 'failed' }); save();
const searchSession = await makeSession('ordinary-web');
await openSession(searchSession);
await page.locator('#web-toggle-btn').click();
@@ -120,8 +154,8 @@ try {
}
if (browser) await browser.close();
}
report.status = report.turns.length === 3 && report.turns.every(x => x.status === 'passed') && report.cleanup.length === sessions.length && report.cleanup.every(x => x.removed) ? 'passed' : 'failed';
report.summary = { passed: report.turns.filter(x => x.status === 'passed').length, total: 3 };
report.status = report.turns.length === 5 && report.turns.every(x => x.status === 'passed') && report.cleanup.length === sessions.length && report.cleanup.every(x => x.removed) ? 'passed' : 'failed';
report.summary = { passed: report.turns.filter(x => x.status === 'passed').length, total: 5 };
save();
console.log(JSON.stringify({ report: path.relative(root, reportPath), status: report.status, summary: report.summary }));
if (report.status !== 'passed') process.exitCode = 1;
+41 -8
View File
@@ -12,6 +12,8 @@ const endpointUrl = process.env.ENDPOINT_URL || (() => { throw new Error("ENDPOI
const model = process.env.MODEL || 'odysseus-qwen3.5-tools-pre-heretic';
const owner = process.env.OWNER || 'sft_alex_creator';
const routingMode = process.env.ROUTING_MODE || 'baseline';
const expectCleanRoute = process.env.EXPECT_CLEAN_ROUTE !== 'false';
const expectRoutingMetadata = process.env.EXPECT_ROUTING_METADATA !== 'false';
if (!['baseline', 'recent', 'all', 'default'].includes(routingMode)) throw Error('Invalid routing mode');
const expectedMode = routingMode === 'default' ? 'recent_model_choice' : routingMode;
const run = new Date().toISOString().replace(/[:.]/g, '-');
@@ -287,6 +289,7 @@ try {
const failure_category = ok ? null
: /not found|no such|unknown (?:uid|id)|does not exist/i.test(detail) ? 'not_found'
: /invalid|missing|required|argument|json|parse/i.test(detail) ? 'invalid_arguments'
: /covered by|obscured by|blocking (?:dialog|overlay)|dismiss or interact with the covering/i.test(detail) ? 'interaction_blocked'
: /connection|unavailable|timeout|refused/i.test(detail) ? 'backend_unavailable'
: /permission|not offered|not permitted|denied/i.test(detail) ? 'permission_denied'
: 'other';
@@ -299,6 +302,30 @@ try {
previousEmailUids = [...detail.matchAll(/^\s*UID:\s*(\S+)/gmi)].map(match => match[1]);
}
const final = events.filter(x => x.type === 'final_response').map(x => x.content || '').join('') || events.filter(x => typeof x.delta === 'string').map(x => x.delta).join('');
const recoveredBrowserInteraction = events.some((event, eventIndex) => {
if (event.type !== 'tool_output' || bare(event.tool) !== 'private_browser') return false;
const detail = String(event.output || event.error_message || '');
const blocked = /covered by|obscured by|blocking (?:dialog|overlay)|dismiss or interact with the covering/i.test(detail);
if (!blocked) return false;
return events.slice(eventIndex + 1).some(later => (
later.type === 'tool_output'
&& bare(later.tool) === 'private_browser'
&& !later.error
&& (later.exit_code == null || later.exit_code === 0)
));
});
const prefetchedWebSources = events
.filter(x => x.type === 'web_sources')
.flatMap(x => Array.isArray(x.data) ? x.data : [])
.filter(source => source?.acquisition === 'automatic_url_fetch');
const prefetchedYoutubeSources = events
.filter(x => x.type === 'web_sources')
.flatMap(x => Array.isArray(x.data) ? x.data : [])
.filter(source => source?.acquisition === 'automatic_youtube_context');
const exactUrlPrefetched = expected.includes('web_fetch')
&& prefetchedWebSources.length > 0;
const youtubePrefetched = expected.includes('youtube_tool')
&& prefetchedYoutubeSources.length > 0;
if (spec.name === 'skills-cookbook-skills' && index === 0) {
// Compare in memory only: never retain private skill names/content.
previousSkillRows = events.filter(x => x.type === 'tool_output' && bare(x.tool) === 'manage_skills')
@@ -340,24 +367,25 @@ try {
nodes.slice(-8).map(node => String(node.className || node.tagName || '').slice(0, 120))
) : [];
const checks = {
experiment_selected: contract.routing_experiment === expectedMode,
http_ok: response.ok(), terminal: response.ok() && !events.some(x => x.type === 'invalid_sse'), clean_route: contract.selection_mode === 'clean_compact_v3_preview',
capability: Boolean(spec.deniedTools?.[index]) || capabilityAvailable(contract, capability, expected)
experiment_selected: !expectRoutingMetadata || contract.routing_experiment === expectedMode,
http_ok: response.ok(), terminal: response.ok() && !events.some(x => x.type === 'invalid_sse'),
clean_route: !expectCleanRoute || contract.selection_mode === 'clean_compact_v3_preview',
capability: !expectRoutingMetadata || Boolean(spec.deniedTools?.[index]) || capabilityAvailable(contract, capability, expected)
|| (spec.noToolTurns?.includes(index) && starts.length === 0)
// An intentionally ambiguous continuation can use the retained
// family without the classifier guessing a fresh active topic.
|| (!expected.length && index > 0 && routingMode !== 'baseline'
&& priorCapability === capability && priorFamilyTools.some(name => offered.includes(name))),
expected_offered: !expected.length || expected.some(name => offered.includes(name)), expected_called: reusedSkillSummary || reusedSkillDetail || !expected.length || expected.some(name => starts.includes(name)),
expected_offered: !expectRoutingMetadata || !expected.length || expected.some(name => offered.includes(name)), expected_called: reusedSkillSummary || reusedSkillDetail || exactUrlPrefetched || youtubePrefetched || !expected.length || expected.some(name => starts.includes(name)),
expected_execution_outcome: spec.expectedExitCodes?.[index] !== undefined
? events.filter(e => e.type === 'tool_output' && expected.includes(bare(e.tool))).length === 1
&& events.some(e => e.type === 'tool_output' && expected.includes(bare(e.tool)) && e.exit_code === spec.expectedExitCodes[index])
: reusedSkillSummary || reusedSkillDetail || !expected.length || outputs.some(x => expected.includes(x.tool) && x.ok),
: reusedSkillSummary || reusedSkillDetail || exactUrlPrefetched || youtubePrefetched || !expected.length || outputs.some(x => expected.includes(x.tool) && x.ok),
requested_execution_count: spec.name !== 'shell-failure-recovery' || starts.length === (index === 1 ? 0 : 1),
failed_execution_provenance: !(spec.expectedExitCodes?.[index] > 0)
|| events.some(e => e.type === 'tool_output' && expected.includes(bare(e.tool))
&& e.exit_code === spec.expectedExitCodes[index] && e.execution_attempted === true && e.blocked === false),
saved_failure_status: !(spec.expectedExitCodes?.[index] > 0)
&& e.exit_code === spec.expectedExitCodes[index] && e.execution_attempted === true && e.blocked !== true),
saved_failure_status: !expectCleanRoute || !(spec.expectedExitCodes?.[index] > 0)
|| (metrics.data?.clean_v3_turn || metrics.clean_v3_turn || []).some(m => {
if (m.role !== 'tool') return false;
try { return JSON.parse(m.content).exit_code === spec.expectedExitCodes[index]; } catch { return false; }
@@ -365,7 +393,7 @@ try {
exact_skill_detail_reference: spec.name !== 'skills-cookbook-skills' || index !== 3
|| reusedSkillDetail || calls.some(call => call.tool === 'manage_skills' && call.skill_action === 'view' && call.skill_matches_second),
skill_detail_answer_evidence: spec.name !== 'skills-cookbook-skills' || index !== 3 || detailEvidence.covered,
no_prior_family_leak: routingMode !== 'baseline' || index === 0 || priorCapability === capability
no_prior_family_leak: !expectRoutingMetadata || routingMode !== 'baseline' || index === 0 || priorCapability === capability
|| offered.every(name => !priorFamilyTools.includes(name) || expected.includes(name)),
one_user_turn: afterUsers === beforeUsers + 1,
visible_answer: final.trim().length > 0, no_reasoning_leak: noLeak(final),
@@ -373,6 +401,9 @@ try {
no_tool_errors: outputs.every(item => item.ok || (
spec.deniedTools?.[index]?.includes(item.tool)
&& item.failure_category === 'permission_denied' && starts.length === 0)
|| (item.tool === 'private_browser'
&& item.failure_category === 'interaction_blocked'
&& recoveredBrowserInteraction)
|| (spec.expectedExitCodes?.[index] > 0 && expected.includes(item.tool)
&& events.some(e => e.type === 'tool_output' && bare(e.tool) === item.tool && e.exit_code === spec.expectedExitCodes[index]))),
expected_answer_evidence: !spec.expectedAnswers
@@ -398,6 +429,8 @@ try {
metrics: Object.fromEntries(['input_tokens', 'output_tokens', 'injected_tokens',
'time_to_first_token', 'response_time'].map(key => [key, metrics[key] ?? metrics.data?.[key] ?? null])),
user_count_before: beforeUsers, user_count_after: afterUsers,
prefetched_web_sources: prefetchedWebSources.length,
prefetched_youtube_sources: prefetchedYoutubeSources.length,
dom_classes_on_user_mismatch: domClasses,
page_errors: pageErrors.splice(0),
unavailable: contract.unavailable || [], checks, status: Object.values(checks).every(Boolean) ? 'passed' : 'failed' };
@@ -10,7 +10,10 @@ const endpointId = process.env.ENDPOINT_ID || '1d1022ef';
const endpointUrl = process.env.ENDPOINT_URL || (() => { throw new Error("ENDPOINT_URL is required"); })();
const model = process.env.MODEL || 'odysseus-qwen3.5-tools-pre-heretic';
const owner = 'sft_alex_creator';
const routingMode = 'recent_model_choice';
const routingMode = process.env.ROUTING_MODE || 'recent';
const expectedRoutingMode = routingMode === 'recent' ? 'recent_model_choice' : routingMode;
const expectCleanRoute = process.env.EXPECT_CLEAN_ROUTE !== 'false';
const expectExactRouting = process.env.EXPECT_EXACT_ROUTING !== 'false';
const run = new Date().toISOString().replace(/[:.]/g, '-');
const reportPath = path.resolve(process.env.REPORT_PATH || path.join(root, `reports/mobile-active-editor-followups-${run}.json`));
if (!reportPath.startsWith(path.join(root, 'reports') + path.sep) || fs.existsSync(reportPath)) throw Error('Report path must be new and under reports/');
@@ -115,8 +118,9 @@ try {
const fetched = await context.request.get(`${base}/api/document/${encodeURIComponent(docId)}`);
const current = fetched.ok() ? String((await fetched.json()).current_content || '') : '';
const checks = {
http_ok: response.ok(), clean_route: contract.selection_mode === 'clean_compact_v3_preview',
exact_runtime: contract.routing_experiment === routingMode,
http_ok: response.ok(),
clean_route: !expectCleanRoute || contract.selection_mode === 'clean_compact_v3_preview',
exact_runtime: !expectExactRouting || contract.routing_experiment === expectedRoutingMode,
request_has_fixture_editor: response.request().postData()?.includes(docId) || false,
documents_capability: (contract.active_capabilities || contract.capabilities || []).includes('documents'),
same_open_editor: await page.evaluate(id => window.documentModule?.getChatDocumentId?.() === id, docId),
@@ -8,6 +8,7 @@ import {AMBIGUOUS_CASES,expectedNoteTitles,compareNoteState} from './note_test_o
const root = path.resolve(new URL('..', import.meta.url).pathname);
const base = process.env.BASE_URL || 'http://127.0.0.1:7011';
const model = process.env.MODEL || 'odysseus-qwen3.5-tools-pre-heretic';
const routingMode = process.env.ROUTING_MODE || 'baseline';
const followupCase = process.env.FOLLOWUP_CASE || 'original';
const plainTitles = process.env.TITLE_STYLE === 'plain';
@@ -73,7 +74,7 @@ try {
} });
await context.addCookies([{ name: 'odysseus_session', value: token, url: base }]);
const created = await context.request.post(`${base}/api/session`, { multipart: {
name: `[multi-note-followup] ${marker}`, model: 'odysseus-qwen3.5-tools-pre-heretic',
name: `[multi-note-followup] ${marker}`, model,
endpoint_id: process.env.ENDPOINT_ID || '1d1022ef',
endpoint_url: process.env.ENDPOINT_URL || (() => { throw new Error("ENDPOINT_URL is required"); })(),
skip_validation: 'true', rag: 'false',
+6 -3
View File
@@ -10,6 +10,8 @@ const endpointId = process.env.ENDPOINT_ID || '1d1022ef';
const endpointUrl = process.env.ENDPOINT_URL || (() => { throw new Error("ENDPOINT_URL is required"); })();
const model = process.env.MODEL || 'odysseus-qwen3.5-tools-pre-heretic';
const owner = 'sft_alex_creator';
const expectCleanRoute = process.env.EXPECT_CLEAN_ROUTE !== 'false';
const expectNativeContractMetadata = process.env.EXPECT_NATIVE_CONTRACT_METADATA !== 'false';
const seconds = value => {
if (typeof value === 'number') return value;
const text = String(value ?? '').trim();
@@ -112,9 +114,10 @@ try {
const expected = starts.filter(event => event.tool === spec.tool);
const args = parseArgs(expected[0]);
const checks = {
http_ok: response.ok(), clean_route: contract.selection_mode === 'clean_compact_v3_preview',
native_workspace: contract.native_workspace === true,
expected_offered: (contract.offered || []).includes(spec.tool),
http_ok: response.ok(),
clean_route: !expectCleanRoute || contract.selection_mode === 'clean_compact_v3_preview',
native_workspace: !expectNativeContractMetadata || contract.native_workspace === true,
expected_offered: !expectNativeContractMetadata || (contract.offered || []).includes(spec.tool),
exactly_one_expected_call: starts.length === 1 && expected.length === 1,
argument_contract: expected.length === 1 && spec.validate(index, args),
exactly_one_successful_output: successfulOutputs.length === 1,
+21 -1
View File
@@ -106,6 +106,26 @@ try {
const removed = await send(`Delete the second ${family === 'tasks' ? 'task' : 'event'} from that list.`);
const deleteOutputs = removed.events.filter(event => event.type === 'tool_output');
const deleteStarts = removed.events.filter(event => event.type === 'tool_start');
const mutationActions = new Set(['delete', 'delete_event', 'remove', 'cancel']);
const pendingStarts = new Map();
const successfulDeleteOutputs = [];
for (const event of removed.events) {
if (event.type === 'tool_start') {
const queue = pendingStarts.get(event.tool) || [];
queue.push(event);
pendingStarts.set(event.tool, queue);
continue;
}
if (event.type !== 'tool_output') continue;
const start = (pendingStarts.get(event.tool) || []).shift();
if (!start || event.error || (event.exit_code != null && event.exit_code !== 0)) continue;
try {
const command = typeof start.command === 'string' ? JSON.parse(start.command) : start.command;
if (mutationActions.has(String(command?.action || '').toLowerCase())) {
successfulDeleteOutputs.push(event);
}
} catch (_) {}
}
const remaining = [];
for (const id of seeded) {
const response = await context.request.get(`${base}${family === 'tasks' ? '/api/tasks/' : '/api/calendar/events/'}${encodeURIComponent(id)}`);
@@ -114,7 +134,7 @@ try {
item.turns.push({ name: 'delete-second', target_id: target, tools: deleteStarts.map(event => event.tool), tool_events: removed.events.filter(event => ['tool_start', 'tool_output'].includes(event.type)).map(event => ({ type: event.type, tool: event.tool, command: event.command, output: event.output, exit_code: event.exit_code, error: event.error })), checks: {
target_resolved: expectedSet.has(target), http_ok: removed.response.ok(),
correct_capability: (removed.contract.active_capabilities || []).includes(family),
one_successful_delete: deleteOutputs.filter(event => !event.error && (event.exit_code == null || event.exit_code === 0)).length === 1,
one_successful_delete: successfulDeleteOutputs.length === 1,
second_item_deleted: !!target && !remaining.includes(target),
other_item_preserved: seeded.filter(id => id !== target).every(id => remaining.includes(id)),
no_stream_error: !removed.events.some(event => ['error', 'invalid_sse'].includes(event.type)),