mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-07 15:32:21 +02:00
Consolidate Odysseus agent harness and tool contracts
This commit is contained in:
@@ -0,0 +1,333 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Build an accountable harness/SFT seed corpus from historical SFT sessions."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import sqlite3
|
||||
from collections import Counter
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
EXCLUDED_PREFIXES = ("[harness-qa]",)
|
||||
FAMILY_ALIASES = {
|
||||
"cookbook": "cookbook_admin",
|
||||
"shell_files": "shell_files",
|
||||
"search": "search_browser",
|
||||
"search_ai": "search_browser",
|
||||
}
|
||||
CANONICAL_FAMILIES = {
|
||||
"calendar", "notes", "email", "memory", "documents", "tasks", "skills",
|
||||
"search_browser", "cookbook_admin", "shell_files", "research", "ui", "switching",
|
||||
}
|
||||
|
||||
|
||||
def case_name(session_name: str) -> str:
|
||||
return session_name.split("]", 1)[-1].strip()
|
||||
|
||||
|
||||
def infer_family(name: str) -> str:
|
||||
value = case_name(name).casefold()
|
||||
value = re.sub(r"^(?:typo|ambiguous|related)[-_]", "", value)
|
||||
if "_to_" in value or value.startswith("greeting_to_"):
|
||||
return "switching"
|
||||
if value.startswith(("browser", "news_followup", "search_")):
|
||||
return "search_browser"
|
||||
stem = re.split(r"[-_]\d", value, maxsplit=1)[0]
|
||||
if stem in CANONICAL_FAMILIES:
|
||||
return stem
|
||||
for alias, family in FAMILY_ALIASES.items():
|
||||
if stem == alias or value.startswith(alias + "-"):
|
||||
return family
|
||||
return "unknown"
|
||||
|
||||
|
||||
def infer_text_family(text: str) -> str:
|
||||
value = re.sub(r"\s+", " ", text).casefold()
|
||||
groups = (
|
||||
("calendar", ("calendar", "event", "schedule", "appointment", "meeting")),
|
||||
("notes", ("note", "checklist")),
|
||||
("email", ("email", "inbox", "sender", "unsubscribe", "spam")),
|
||||
("memory", ("memory", "remember", "forget")),
|
||||
("documents", ("document", "write reply", "write this", "editor")),
|
||||
("tasks", ("task", "scheduled job", "cron")),
|
||||
("skills", ("skill",)),
|
||||
("research", ("research",)),
|
||||
("cookbook_admin", ("model server", "endpoint", "runpod", "served model", "cookbook")),
|
||||
("shell_files", ("workspace", "file", "folder", "directory", "bash", "python", "ssh")),
|
||||
("ui", ("open gallery", "open panel", "theme")),
|
||||
("search_browser", ("http://", "https://", "search", "look up", "browse", "website", "latest", "weather", "news")),
|
||||
)
|
||||
matched = [family for family, words in groups if any(word in value for word in words)]
|
||||
if len(set(matched)) > 1:
|
||||
return "switching"
|
||||
return matched[0] if matched else "general"
|
||||
|
||||
|
||||
def infer_turn_family(session_family: str, turn: dict[str, Any]) -> str:
|
||||
"""Prefer observed tool/contract evidence over unreliable session titles."""
|
||||
metadata = turn.get("metadata") or {}
|
||||
names = {
|
||||
str(event.get("tool") or "")
|
||||
for event in (metadata.get("tool_events") or [])
|
||||
if isinstance(event, dict)
|
||||
}
|
||||
contract = metadata.get("turn_contract") or {}
|
||||
capabilities = contract.get("capabilities") or metadata.get("capabilities") or []
|
||||
hints = " ".join(sorted(names | {str(value) for value in capabilities})).casefold()
|
||||
mappings = (
|
||||
(("calendar", "manage_calendar"), "calendar"),
|
||||
(("notes", "manage_notes"), "notes"),
|
||||
(("email", "inbox", "draft_email"), "email"),
|
||||
(("memory", "manage_memory"), "memory"),
|
||||
(("document", "manage_documents"), "documents"),
|
||||
(("task", "manage_tasks"), "tasks"),
|
||||
(("skill", "manage_skills"), "skills"),
|
||||
(("research", "trigger_research"), "research"),
|
||||
(("browser", "web_search", "web_fetch", "youtube"), "search_browser"),
|
||||
(("cookbook", "served_model", "cached_model", "endpoint"), "cookbook_admin"),
|
||||
(("shell", "bash", "read_file", "write_file", "\bls\b"), "shell_files"),
|
||||
(("ui_control",), "ui"),
|
||||
)
|
||||
matched = [family for needles, family in mappings if any(needle in hints for needle in needles)]
|
||||
if len(set(matched)) > 1:
|
||||
return "switching"
|
||||
if matched:
|
||||
return matched[0]
|
||||
if session_family != "unknown":
|
||||
return session_family
|
||||
return infer_text_family(str(turn.get("user") or ""))
|
||||
|
||||
|
||||
def normalized_flow_key(turns: list[dict[str, Any]]) -> str:
|
||||
texts = []
|
||||
for turn in turns:
|
||||
text = re.sub(r"\s+", " ", str(turn.get("user") or "")).strip().casefold()
|
||||
texts.append(text)
|
||||
return "\n".join(texts)
|
||||
|
||||
|
||||
def event_failed(event: dict[str, Any]) -> bool:
|
||||
return bool(event.get("error") or event.get("exit_code") not in (None, 0))
|
||||
|
||||
|
||||
def classify(turns: list[dict[str, Any]]) -> tuple[str, list[str]]:
|
||||
"""Conservative historical triage; replay resolves everything uncertain."""
|
||||
reasons: list[str] = []
|
||||
backend = False
|
||||
harness = False
|
||||
model_sft = False
|
||||
successful_tool = False
|
||||
for index, turn in enumerate(turns):
|
||||
assistant = str(turn.get("assistant") or "")
|
||||
metadata = turn.get("metadata") or {}
|
||||
events = metadata.get("tool_events") or []
|
||||
successful_tool |= any(not event_failed(event) for event in events)
|
||||
combined_errors = "\n".join(
|
||||
str(event.get("error") or "") + "\n" + str(event.get("output") or "")
|
||||
for event in events if event_failed(event)
|
||||
)
|
||||
if re.search(r"connection refused|timed? out|backend unavailable|service unavailable", combined_errors, re.I):
|
||||
backend = True
|
||||
reasons.append(f"turn {index + 1}: tool/backend transport failed")
|
||||
denied = any(
|
||||
isinstance(decision, dict) and decision.get("allowed") is False
|
||||
for decision in (metadata.get("policy_decisions") or [])
|
||||
)
|
||||
if metadata.get("required_operation_succeeded") is False or denied:
|
||||
harness = True
|
||||
reasons.append(f"turn {index + 1}: harness policy or required operation blocked execution")
|
||||
if index and re.search(r"no preceding (?:answer|message)|not in this conversation", assistant, re.I):
|
||||
harness = True
|
||||
reasons.append(f"turn {index + 1}: prior conversation state was lost")
|
||||
if successful_tool and re.search(
|
||||
r"(?:cannot|can't|unable to) (?:access|view|open|read|use).{0,40}(?:notes?|calendar|emails?|tasks?|documents?)",
|
||||
assistant,
|
||||
re.I,
|
||||
):
|
||||
model_sft = True
|
||||
reasons.append(f"turn {index + 1}: response contradicted successful tool evidence")
|
||||
if any(event_failed(event) and re.search(
|
||||
r"placeholder|not returned by|invalid arguments?|validation|must be an exact",
|
||||
str(event.get("error") or "") + str(event.get("output") or ""), re.I,
|
||||
) for event in events):
|
||||
model_sft = True
|
||||
reasons.append(f"turn {index + 1}: model proposed invalid or ungrounded arguments")
|
||||
if backend:
|
||||
return "backend", sorted(set(reasons))
|
||||
if harness:
|
||||
return "harness", sorted(set(reasons))
|
||||
if model_sft:
|
||||
return "model_sft", sorted(set(reasons))
|
||||
return "replay_first", ["historical result is not sufficient for a reliable owner classification"]
|
||||
|
||||
|
||||
def load_sessions(db_path: Path, owner: str) -> list[dict[str, Any]]:
|
||||
db = sqlite3.connect(db_path)
|
||||
db.row_factory = sqlite3.Row
|
||||
sessions = db.execute(
|
||||
"SELECT id, name, created_at FROM sessions WHERE owner=? ORDER BY created_at DESC",
|
||||
(owner,),
|
||||
).fetchall()
|
||||
output = []
|
||||
for session in sessions:
|
||||
if any(str(session["name"] or "").startswith(prefix) for prefix in EXCLUDED_PREFIXES):
|
||||
continue
|
||||
rows = db.execute(
|
||||
"SELECT role, content, metadata FROM chat_messages WHERE session_id=? ORDER BY timestamp, rowid",
|
||||
(session["id"],),
|
||||
).fetchall()
|
||||
turns = []
|
||||
pending = None
|
||||
for row in rows:
|
||||
if row["role"] == "user":
|
||||
pending = {"user": row["content"], "assistant": "", "metadata": {}}
|
||||
turns.append(pending)
|
||||
elif row["role"] == "assistant" and pending is not None:
|
||||
pending["assistant"] = row["content"]
|
||||
try:
|
||||
pending["metadata"] = json.loads(row["metadata"] or "{}")
|
||||
except (TypeError, ValueError, json.JSONDecodeError):
|
||||
pending["metadata"] = {}
|
||||
pending = None
|
||||
if turns:
|
||||
session_family = infer_family(session["name"])
|
||||
for turn in turns:
|
||||
turn["family"] = infer_turn_family(session_family, turn)
|
||||
output.append({
|
||||
"source_session_id": session["id"],
|
||||
"source_name": session["name"],
|
||||
"created_at": session["created_at"],
|
||||
"family": session_family,
|
||||
"turns": turns,
|
||||
})
|
||||
db.close()
|
||||
return output
|
||||
|
||||
|
||||
def build_seeds(sessions: list[dict[str, Any]], context_turns: int = 3) -> list[dict[str, Any]]:
|
||||
"""Create exactly one teacher seed for every historical user turn.
|
||||
|
||||
A seed retains preceding user context so ambiguous follow-ups remain
|
||||
ambiguous in the same useful way. Repeated source runs are intentionally
|
||||
retained; they measure stability instead of disappearing via deduplication.
|
||||
"""
|
||||
seeds: list[dict[str, Any]] = []
|
||||
for session in sessions:
|
||||
turns = session["turns"]
|
||||
for index, turn in enumerate(turns):
|
||||
start = max(0, index - context_turns)
|
||||
context = [
|
||||
{"user": item["user"]}
|
||||
for item in turns[start:index + 1]
|
||||
]
|
||||
seeds.append({
|
||||
"seed_id": f"{session['source_session_id']}:{index + 1}",
|
||||
"source_session_id": session["source_session_id"],
|
||||
"source_name": session["source_name"],
|
||||
"source_turn": index + 1,
|
||||
"family": turn.get("family") or session["family"],
|
||||
"context": context,
|
||||
"target_user": turn["user"],
|
||||
})
|
||||
return seeds
|
||||
|
||||
|
||||
def build_queue(sessions: list[dict[str, Any]]) -> dict[str, Any]:
|
||||
seeds = build_seeds(sessions)
|
||||
unique: dict[str, dict[str, Any]] = {}
|
||||
duplicate_counts = Counter()
|
||||
for session in sessions:
|
||||
key = normalized_flow_key(session["turns"])
|
||||
duplicate_counts[key] += 1
|
||||
if key not in unique: # sessions arrive newest first
|
||||
unique[key] = session
|
||||
workstreams = {name: [] for name in ("harness", "model_sft", "backend", "replay_first")}
|
||||
replay_flows = []
|
||||
for number, (key, session) in enumerate(unique.items(), 1):
|
||||
bucket, reasons = classify(session["turns"])
|
||||
row = {
|
||||
"id": f"historical-{number:04d}",
|
||||
"family": session["family"],
|
||||
"case": case_name(session["source_name"]),
|
||||
"source_session_id": session["source_session_id"],
|
||||
"duplicate_runs": duplicate_counts[key],
|
||||
"reasons": reasons,
|
||||
"turns": [
|
||||
{
|
||||
"user": turn["user"],
|
||||
"assistant": turn["assistant"],
|
||||
"tools": [event.get("tool") for event in (turn["metadata"].get("tool_events") or [])],
|
||||
}
|
||||
for turn in session["turns"]
|
||||
],
|
||||
}
|
||||
workstreams[bucket].append(row)
|
||||
replay_flows.append({
|
||||
"id": row["id"],
|
||||
"family": row["family"],
|
||||
"purpose": f"Replay historical contract case {row['case']}",
|
||||
"turns": [{
|
||||
"user": turn["user"],
|
||||
"expect": "Honor the request and conversation context; use the correct tool only when needed and rely on successful tool evidence.",
|
||||
} for turn in session["turns"]],
|
||||
})
|
||||
return {
|
||||
"created_at": datetime.now(timezone.utc).isoformat(),
|
||||
"source_sessions": len(sessions),
|
||||
"source_user_turns": sum(len(session["turns"]) for session in sessions),
|
||||
"seed_count": len(seeds),
|
||||
"unique_flows": len(unique),
|
||||
"counts": {name: len(rows) for name, rows in workstreams.items()},
|
||||
"families": dict(sorted(Counter(row["family"] for row in unique.values()).items())),
|
||||
"seed_families": dict(sorted(Counter(row["family"] for row in seeds).items())),
|
||||
"workstreams": workstreams,
|
||||
"flows": replay_flows,
|
||||
"seeds": seeds,
|
||||
}
|
||||
|
||||
|
||||
def render_summary(queue: dict[str, Any]) -> str:
|
||||
lines = [
|
||||
"# Historical Odysseus QA Queue", "",
|
||||
f"- Source sessions: {queue['source_sessions']}",
|
||||
f"- Source user turns / teacher seeds: {queue['seed_count']}",
|
||||
f"- Unique conversation flows: {queue['unique_flows']}",
|
||||
"- Historical labels are conservative; `replay_first` must be replayed before assigning ownership.",
|
||||
"", "## Workstreams", "",
|
||||
]
|
||||
for name, count in queue["counts"].items():
|
||||
lines.append(f"- `{name}`: {count}")
|
||||
lines.extend(["", "## Families", ""])
|
||||
for family, count in queue["seed_families"].items():
|
||||
lines.append(f"- `{family}`: {count}")
|
||||
lines.extend([
|
||||
"", "## Workflow", "",
|
||||
"1. Cook one fresh conversation from every seed using the complete tool catalog.",
|
||||
"2. Replay safe cooked cases on the current 7011 Agent runtime.",
|
||||
"3. Judge, classify ownership, and patch recurring behavior classes.",
|
||||
"4. Retain duplicate source runs as stability evidence; account for quarantined cases explicitly.",
|
||||
])
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--db", type=Path, required=True)
|
||||
parser.add_argument("--owner", default="sft_alex_creator")
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
parser.add_argument("--summary", type=Path, required=True)
|
||||
args = parser.parse_args()
|
||||
queue = build_queue(load_sessions(args.db, args.owner))
|
||||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.summary.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.output.write_text(json.dumps(queue, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
|
||||
args.summary.write_text(render_summary(queue), encoding="utf-8")
|
||||
print(json.dumps({key: queue[key] for key in ("source_sessions", "unique_flows", "counts", "families")}, indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,230 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Build a reproducible model-only repair pool from conversation QA runs."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import re
|
||||
from collections import Counter
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
SFT_WEBUI_POLICY_DISABLED_TOOLS = frozenset({
|
||||
"python", "read_file", "write_file", "edit_file", "apply_patch",
|
||||
})
|
||||
|
||||
|
||||
def source_seed_id(row: dict[str, Any]) -> str:
|
||||
return str(row.get("source_seed_id") or row.get("id") or "").strip()
|
||||
|
||||
|
||||
def behavior_category(value: str) -> str:
|
||||
text = str(value or "").casefold()
|
||||
rules = (
|
||||
("response_constraint_adherence", (
|
||||
"limit", "constraint", "instruction_noncompliance", "instruction_following",
|
||||
"counting_error",
|
||||
)),
|
||||
("required_tool_execution", (
|
||||
"missing_tool", "missing_required_tool", "missing_required_action",
|
||||
"false_refusal", "refusal",
|
||||
)),
|
||||
("tool_action_selection", (
|
||||
"wrong_action", "wrong_tool", "incorrect_tool", "malformed_tool",
|
||||
"command_selection",
|
||||
)),
|
||||
("required_argument_grounding", ("argument", "identifier", "filter")),
|
||||
("tool_error_recovery", (
|
||||
"no_retry", "error_recovery", "false_empty", "empty_result",
|
||||
"unrecovered", "missing_fallback", "stale_id_loop",
|
||||
)),
|
||||
("result_rendering", (
|
||||
"render", "empty_answer", "missing_requested_content", "missing_note_titles",
|
||||
"missing_progress_link", "non_answer", "uninformative_answer",
|
||||
)),
|
||||
("followup_evidence_use", ("followup", "follow_up", "continuity", "unanswered", "incomplete")),
|
||||
("evidence_grounding", (
|
||||
"hallucin", "wrong_answer", "unsupported", "grounding", "false_success",
|
||||
"unfaithful", "content_mismatch",
|
||||
)),
|
||||
)
|
||||
for category, needles in rules:
|
||||
if any(needle in text for needle in needles):
|
||||
return category
|
||||
return "other_model_behavior"
|
||||
|
||||
|
||||
def has_transport_failure(row: dict[str, Any]) -> bool:
|
||||
needles = (
|
||||
"connection refused", "connecterror", "remoteprotocolerror",
|
||||
"replay_transport_unavailable", "session_start_failed", "readtimeout",
|
||||
)
|
||||
return any(needle in json.dumps(row, ensure_ascii=False).casefold() for needle in needles)
|
||||
|
||||
|
||||
def eligible_failed_turns(row: dict[str, Any]) -> tuple[list[int], list[int]]:
|
||||
observed = row.get("observed") or []
|
||||
failed = [value for value in (row.get("judge") or {}).get("failed_turns") or []
|
||||
if isinstance(value, int) and 1 <= value <= len(observed)]
|
||||
if not failed:
|
||||
failed = list(range(1, len(observed) + 1))
|
||||
eligible, absent_surface = [], []
|
||||
for number in failed:
|
||||
turn = observed[number - 1]
|
||||
contract = turn.get("contract") or {}
|
||||
if not (contract.get("offered") or []) and not (turn.get("tool_calls") or []):
|
||||
absent_surface.append(number)
|
||||
else:
|
||||
eligible.append(number)
|
||||
return eligible, absent_surface
|
||||
|
||||
|
||||
def requires_native_workspace_tool(row: dict[str, Any]) -> bool:
|
||||
expected = "\n".join(
|
||||
str(turn.get("expect") or "")
|
||||
for turn in (row.get("turns") or [])
|
||||
if isinstance(turn, dict)
|
||||
)
|
||||
return any(
|
||||
re.search(rf"(?<!\w){re.escape(tool)}(?!\w)", expected, re.I)
|
||||
for tool in SFT_WEBUI_POLICY_DISABLED_TOOLS
|
||||
) or bool(re.search(
|
||||
r"\b(?:run|use|execute)\s+(?:a\s+)?(?:local\s+)?(?:shell|bash)\b|"
|
||||
r"\b(?:shell|bash)\s+(?:version\s+)?check\b",
|
||||
expected,
|
||||
re.I,
|
||||
))
|
||||
|
||||
|
||||
def build_manifest(paths: list[Path], excluded_seeds: set[str],
|
||||
routing_experiment: str | None = None,
|
||||
resolved_seeds: set[str] | None = None) -> dict[str, Any]:
|
||||
"""Retain each seed's latest confirmed model-owned failure.
|
||||
|
||||
A later stochastic pass does not prove a repair and must not silently erase
|
||||
a useful failure example. Operators can explicitly resolve or exclude a
|
||||
seed after a verified fix or after discovering a defective expectation.
|
||||
"""
|
||||
resolved_seeds = resolved_seeds or set()
|
||||
latest_failure: dict[str, tuple[int, dict[str, Any], Path]] = {}
|
||||
inputs = []
|
||||
ignored_nonbehavioral_rows = 0
|
||||
ignored_runtime_inputs = 0
|
||||
for order, path in enumerate(paths):
|
||||
raw = path.read_bytes()
|
||||
payload = json.loads(raw)
|
||||
runtime = payload.get("routing_experiment", "baseline")
|
||||
inputs.append({
|
||||
"path": str(path), "sha256": hashlib.sha256(raw).hexdigest(),
|
||||
"routing_experiment": runtime,
|
||||
})
|
||||
if routing_experiment is not None and runtime != routing_experiment:
|
||||
ignored_runtime_inputs += 1
|
||||
continue
|
||||
for row in payload.get("results") or []:
|
||||
seed = source_seed_id(row)
|
||||
judge = row.get("judge") or {}
|
||||
# An unavailable judge or broken replay does not supersede older
|
||||
# valid behavioral evidence for the same seed.
|
||||
if not seed or judge.get("verdict") not in {"pass", "fail"} or has_transport_failure(row):
|
||||
ignored_nonbehavioral_rows += 1
|
||||
continue
|
||||
if judge.get("verdict") == "fail" and judge.get("owner") == "model_sft":
|
||||
latest_failure[seed] = (order, row, path)
|
||||
|
||||
candidates, exclusions = [], []
|
||||
for seed, (_, row, path) in sorted(latest_failure.items()):
|
||||
judge = row.get("judge") or {}
|
||||
reason = None
|
||||
if seed in excluded_seeds:
|
||||
reason = "explicit_ambiguous_or_defective_seed"
|
||||
elif seed in resolved_seeds:
|
||||
reason = "explicitly_resolved_after_verified_fix"
|
||||
elif requires_native_workspace_tool(row):
|
||||
reason = "requires_native_workspace_tool_on_webui_surface"
|
||||
elif has_transport_failure(row):
|
||||
reason = "transport_contaminated"
|
||||
eligible, absent_surface = eligible_failed_turns(row)
|
||||
if reason is None and not eligible:
|
||||
reason = "no_failed_turn_with_executable_tool_surface"
|
||||
if reason:
|
||||
exclusions.append({"source_seed_id": seed, "reason": reason})
|
||||
continue
|
||||
candidates.append({
|
||||
"source_seed_id": seed,
|
||||
"family": row.get("family"),
|
||||
"purpose": row.get("purpose"),
|
||||
"behavior_category": behavior_category(judge.get("failure_category", "")),
|
||||
"eligible_failed_turns": eligible,
|
||||
"excluded_absent_surface_turns": absent_surface,
|
||||
"judge": judge,
|
||||
"turns": row.get("turns") or [],
|
||||
"observed": row.get("observed") or [],
|
||||
"session_id": row.get("session_id"),
|
||||
"url": row.get("url"),
|
||||
"latest_run": str(path),
|
||||
})
|
||||
return {
|
||||
"created_at": datetime.now(timezone.utc).isoformat(),
|
||||
"policy": {
|
||||
"precedence": "latest confirmed model_sft failure wins per source_seed_id; later stochastic passes do not erase it",
|
||||
"include": "latest model_sft fail verdict with executable tool surface",
|
||||
"exclude": [
|
||||
"pass/uncertain", "non-model owners", "transport contamination",
|
||||
"failed turns with absent tool surface", "explicit ambiguous/defective seeds",
|
||||
"native-workspace-only expectations on the WebUI surface", "explicitly resolved seeds",
|
||||
],
|
||||
},
|
||||
"routing_experiment": routing_experiment,
|
||||
"inputs": inputs,
|
||||
"ignored_runtime_inputs": ignored_runtime_inputs,
|
||||
"ignored_nonbehavioral_rows": ignored_nonbehavioral_rows,
|
||||
"candidate_count": len(candidates),
|
||||
"counts_by_family": dict(sorted(Counter(row["family"] for row in candidates).items())),
|
||||
"counts_by_behavior": dict(sorted(Counter(row["behavior_category"] for row in candidates).items())),
|
||||
"candidates": candidates,
|
||||
"exclusion_count": len(exclusions),
|
||||
"exclusions": exclusions,
|
||||
}
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--run", type=Path, action="append", required=True,
|
||||
help="QA run in chronological order; repeat for later replays")
|
||||
parser.add_argument("--exclude-seed", action="append", default=[],
|
||||
help="Explicitly exclude an ambiguous or defective generated seed")
|
||||
parser.add_argument("--resolved-seed", action="append", default=[],
|
||||
help="Drop a model failure only after a verified repair replay")
|
||||
parser.add_argument(
|
||||
"--routing-experiment", default="recent_model_choice",
|
||||
help="Include only runs from this exact routing runtime",
|
||||
)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
manifest = build_manifest(
|
||||
args.run, set(args.exclude_seed), args.routing_experiment,
|
||||
set(args.resolved_seed),
|
||||
)
|
||||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.output.write_text(json.dumps(manifest, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
print(json.dumps({
|
||||
"output": str(args.output),
|
||||
"candidates": manifest["candidate_count"],
|
||||
"by_family": manifest["counts_by_family"],
|
||||
"by_behavior": manifest["counts_by_behavior"],
|
||||
"excluded": manifest["exclusion_count"],
|
||||
}, indent=2))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -10,11 +10,15 @@ from collections import Counter
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from dotenv import load_dotenv
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
load_dotenv(ROOT / ".env")
|
||||
if str(ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
|
||||
from core.database import CalendarCal, CalendarEvent, Document, Memory, Note, ScheduledTask, Session, SessionLocal, UserTool # noqa: E402
|
||||
from src.constants import DATA_DIR # noqa: E402
|
||||
from scripts.sft_email_overseer import PROFILES # noqa: E402
|
||||
OWNERS = ["sft_maya_ops", "sft_jules_research", "sft_nora_design", "sft_omar_finance"]
|
||||
|
||||
@@ -25,7 +29,7 @@ def clip(value: Any, limit: int = 180) -> str:
|
||||
|
||||
|
||||
def email_inventory() -> dict[str, list[dict[str, Any]]]:
|
||||
payload = json.loads((ROOT / "data/fixture_email_messages.json").read_text(encoding="utf-8"))
|
||||
payload = json.loads((Path(DATA_DIR) / "fixture_email_messages.json").read_text(encoding="utf-8"))
|
||||
rows = payload.get("messages") if isinstance(payload, dict) else payload
|
||||
out = {owner: [] for owner in OWNERS}
|
||||
for row in rows or []:
|
||||
|
||||
@@ -0,0 +1,298 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Cook every historical SFT Alex user turn into a fresh tool conversation."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import concurrent.futures
|
||||
import fcntl
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
import threading
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||||
if str(SCRIPT_DIR) not in sys.path:
|
||||
sys.path.insert(0, str(SCRIPT_DIR))
|
||||
|
||||
from odysseus_conversation_qa import (
|
||||
DEFAULT_DATA,
|
||||
DEFAULT_JUDGE_ENDPOINT,
|
||||
DEFAULT_JUDGE_MODEL,
|
||||
FAMILY_SEEDS,
|
||||
compact_tool_catalog,
|
||||
endpoint_from_db,
|
||||
teacher_json,
|
||||
)
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
DEFAULT_SEEDS = ROOT / "tmp/odysseus-conversation-qa/sft-alex-all-seeds.json"
|
||||
DEFAULT_OUTPUT = ROOT / "tmp/odysseus-conversation-qa/sft-alex-cooked.jsonl"
|
||||
LOCK = threading.Lock()
|
||||
|
||||
_CREATE_RE = re.compile(
|
||||
r"\b(?:create|make|start|write|add|save|draft|new)\b", re.IGNORECASE
|
||||
)
|
||||
_LOOKUP_RE = re.compile(
|
||||
r"\b(?:open|find|show|read|list|search|retrieve|look\s+up|already\s+have|saved)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_NEW_TOPIC_RE = re.compile(
|
||||
r"\b(?:about|on)\s+(.+?)(?=\s+(?:and|then|with|using)\b|[.!?]|$)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_ENTITY_PATTERNS = (
|
||||
re.compile(r"([`\"])([^`\"\r\n]{3,120})\1"),
|
||||
re.compile(r"https?://[^\s<>]+", re.IGNORECASE),
|
||||
re.compile(r"\b[\w.-]+\.(?:md|txt|csv|json|pdf|html|docx?|xlsx?)\b", re.IGNORECASE),
|
||||
re.compile(
|
||||
r"\b(?:titled|called|named)\s+(.+?)(?=\s+(?:with|in|so|and|for|from|that)\b|[.!?,;]|$)",
|
||||
re.IGNORECASE,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def explicit_entities(text: str) -> set[str]:
|
||||
"""Extract source-grounded names that a cooked flow must not replace."""
|
||||
entities: set[str] = set()
|
||||
for pattern in _ENTITY_PATTERNS:
|
||||
for match in pattern.finditer(str(text or "")):
|
||||
if pattern is _ENTITY_PATTERNS[0]:
|
||||
value = match.group(2).strip()
|
||||
else:
|
||||
value = (match.group(1) if match.lastindex else match.group(0)).strip()
|
||||
if len(value) >= 3:
|
||||
entities.add(value.casefold())
|
||||
return entities
|
||||
|
||||
|
||||
def grounding_issues(seed: dict[str, Any], flow: dict[str, Any]) -> list[str]:
|
||||
"""Reject synthetic flows whose private-object state contradicts the seed."""
|
||||
source_turns = [str(item.get("user") or "") for item in seed.get("context") or []]
|
||||
generated_turns = [str(item.get("user") or "") for item in flow.get("turns") or []]
|
||||
source_text = "\n".join(source_turns)
|
||||
generated_text = "\n".join(generated_turns)
|
||||
issues: list[str] = []
|
||||
|
||||
for entity in sorted(explicit_entities(source_text)):
|
||||
if entity not in generated_text.casefold():
|
||||
issues.append(f"missing_source_entity:{entity}")
|
||||
|
||||
# Standalone flows must recreate source-created private state before use.
|
||||
for index, source_turn in enumerate(source_turns[:-1]):
|
||||
if not _CREATE_RE.search(source_turn):
|
||||
continue
|
||||
entities = explicit_entities(source_turn)
|
||||
later_source = "\n".join(source_turns[index + 1:]).casefold()
|
||||
for entity in entities:
|
||||
if entity not in later_source:
|
||||
continue
|
||||
mentions = [turn for turn in generated_turns if entity in turn.casefold()]
|
||||
if mentions and not _CREATE_RE.search(mentions[0]):
|
||||
issues.append(f"unestablished_private_entity:{entity}")
|
||||
|
||||
# A source topic introduced by create/start cannot become pre-existing state.
|
||||
target = str(seed.get("target_user") or "")
|
||||
if _CREATE_RE.search(target):
|
||||
target_entities = explicit_entities(target)
|
||||
target_entities.update(
|
||||
match.group(1).strip().casefold()
|
||||
for match in _NEW_TOPIC_RE.finditer(target)
|
||||
if len(match.group(1).strip()) >= 3
|
||||
)
|
||||
for entity in target_entities:
|
||||
for turn in generated_turns:
|
||||
if entity not in turn.casefold():
|
||||
continue
|
||||
if _CREATE_RE.search(turn):
|
||||
break
|
||||
if _LOOKUP_RE.search(turn):
|
||||
issues.append(f"lookup_before_creation:{entity}")
|
||||
break
|
||||
return sorted(set(issues))
|
||||
|
||||
|
||||
def redact(text: str) -> str:
|
||||
"""Remove likely credentials while retaining natural request structure."""
|
||||
value = str(text or "")
|
||||
value = re.sub(r"hf_[A-Za-z0-9]{20,}", "[REDACTED_HF_TOKEN]", value)
|
||||
value = re.sub(r"(?i)(api[_ -]?key|token|password)\s*[:=]\s*\S+", r"\1=[REDACTED]", value)
|
||||
value = re.sub(r"\b(?:\d{1,3}\.){3}\d{1,3}\b", "[REDACTED_IP]", value)
|
||||
return value[:1200]
|
||||
|
||||
|
||||
def load_seeds(path: Path) -> list[dict[str, Any]]:
|
||||
payload = json.loads(path.read_text(encoding="utf-8"))
|
||||
seeds = payload.get("seeds") if isinstance(payload, dict) else None
|
||||
if not isinstance(seeds, list):
|
||||
raise RuntimeError("seed file must contain a top-level seeds array")
|
||||
output = []
|
||||
for seed in seeds:
|
||||
if not isinstance(seed, dict) or not seed.get("seed_id"):
|
||||
continue
|
||||
row = dict(seed)
|
||||
row["context"] = [
|
||||
{"user": redact(item.get("user", ""))}
|
||||
for item in (seed.get("context") or []) if isinstance(item, dict)
|
||||
]
|
||||
row["target_user"] = redact(seed.get("target_user", ""))
|
||||
output.append(row)
|
||||
return output
|
||||
|
||||
|
||||
def completed_ids(path: Path) -> set[str]:
|
||||
if not path.exists():
|
||||
return set()
|
||||
ids = set()
|
||||
for line in path.read_text(encoding="utf-8").splitlines():
|
||||
try:
|
||||
row = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if isinstance(row, dict) and row.get("source_seed_id"):
|
||||
ids.add(str(row["source_seed_id"]))
|
||||
return ids
|
||||
|
||||
|
||||
def chunks(rows: list[dict[str, Any]], size: int) -> list[list[dict[str, Any]]]:
|
||||
return [rows[index:index + size] for index in range(0, len(rows), size)]
|
||||
|
||||
|
||||
def validate_flows(
|
||||
result: Any,
|
||||
wanted: set[str],
|
||||
seeds: dict[str, dict[str, Any]] | None = None,
|
||||
) -> dict[str, dict[str, Any]]:
|
||||
rows = result.get("flows") if isinstance(result, dict) else None
|
||||
valid: dict[str, dict[str, Any]] = {}
|
||||
if not isinstance(rows, list):
|
||||
return valid
|
||||
for row in rows:
|
||||
if not isinstance(row, dict):
|
||||
continue
|
||||
seed_id = str(row.get("source_seed_id") or "")
|
||||
turns = row.get("turns")
|
||||
if seed_id not in wanted or seed_id in valid:
|
||||
continue
|
||||
if row.get("family") not in FAMILY_SEEDS or not isinstance(turns, list) or not 2 <= len(turns) <= 4:
|
||||
continue
|
||||
if any(not isinstance(turn, dict) or not str(turn.get("user") or "").strip() for turn in turns):
|
||||
continue
|
||||
if seeds and seed_id in seeds and grounding_issues(seeds[seed_id], row):
|
||||
continue
|
||||
row["id"] = "sft-alex-" + re.sub(r"[^A-Za-z0-9_-]", "-", seed_id)[:72]
|
||||
row["source_seed_id"] = seed_id
|
||||
valid[seed_id] = row
|
||||
return valid
|
||||
|
||||
|
||||
def cook_batch(endpoint: Any, batch: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||
pending = {str(seed["seed_id"]): seed for seed in batch}
|
||||
cooked: dict[str, dict[str, Any]] = {}
|
||||
for _ in range(3):
|
||||
if not pending:
|
||||
break
|
||||
result = teacher_json(endpoint, {
|
||||
"task": "Turn every supplied historical seed into one fresh realistic multi-turn conversation that tests Odysseus tool use.",
|
||||
"rules": [
|
||||
"Return exactly one flow for every source_seed_id; never merge, omit, or duplicate seeds.",
|
||||
"Preserve the seed's behavioral intent, but do not copy its wording mechanically.",
|
||||
"Preserve exact names, titles, filenames, URLs, contacts, and named research topics from the source seed; never replace them with invented private objects.",
|
||||
"Every generated flow is replayed independently against a clean fixture. If a later action depends on an object created earlier in the source context, include that creation before using the object.",
|
||||
"Never find, open, or read an invented private object. A new note, document, task, event, skill, email, or research report must be created earlier in that generated flow.",
|
||||
"Each flow has 2-4 user turns and at least one context-dependent follow-up.",
|
||||
"The conversation must naturally require at least one Odysseus tool; for a general question, add an adjacent save, verify, open, or retrieve request.",
|
||||
"Use natural short wording and occasional realistic misspelling, not regex-like substitutions.",
|
||||
"Do not include record IDs, credentials, real email addresses, destructive shell operations, email sending, purchases, or irreversible actions.",
|
||||
"Expected behavior is semantic and names the appropriate action/tool family without prescribing exact prose.",
|
||||
"Choose exactly one canonical family from the supplied family list; use switching when the conversation crosses families.",
|
||||
],
|
||||
"schema": {"flows": [{
|
||||
"source_seed_id": "exact supplied ID", "id": "short ID",
|
||||
"family": "canonical family", "purpose": "behavior under test",
|
||||
"turns": [{"user": "message", "expect": "semantic expected behavior"}],
|
||||
}]},
|
||||
"canonical_families": sorted(FAMILY_SEEDS),
|
||||
"complete_odysseus_tool_catalog": compact_tool_catalog(),
|
||||
"seeds": list(pending.values()),
|
||||
}, max_tokens=7500, temperature=0.65)
|
||||
accepted = validate_flows(result, set(pending), pending)
|
||||
cooked.update(accepted)
|
||||
for seed_id in accepted:
|
||||
pending.pop(seed_id, None)
|
||||
if pending:
|
||||
raise RuntimeError(f"teacher omitted {len(pending)} seeds: {sorted(pending)[:3]}")
|
||||
return [cooked[str(seed["seed_id"])] for seed in batch]
|
||||
|
||||
|
||||
def append_rows(path: Path, rows: list[dict[str, Any]]) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with LOCK, path.open("a", encoding="utf-8") as handle:
|
||||
for row in rows:
|
||||
handle.write(json.dumps(row, ensure_ascii=False) + "\n")
|
||||
handle.flush()
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--seeds", type=Path, default=DEFAULT_SEEDS)
|
||||
parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT)
|
||||
parser.add_argument("--data-dir", type=Path, default=DEFAULT_DATA)
|
||||
parser.add_argument("--endpoint-id", default=DEFAULT_JUDGE_ENDPOINT)
|
||||
parser.add_argument("--model", default=DEFAULT_JUDGE_MODEL)
|
||||
parser.add_argument("--batch-size", type=int, default=12)
|
||||
parser.add_argument("--workers", type=int, default=8)
|
||||
parser.add_argument("--limit", type=int)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||
lock_path = args.output.with_suffix(args.output.suffix + ".lock")
|
||||
lock_handle = lock_path.open("w", encoding="utf-8")
|
||||
try:
|
||||
fcntl.flock(lock_handle, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
||||
except BlockingIOError:
|
||||
raise SystemExit(f"another cooker already owns {lock_path}")
|
||||
endpoint = endpoint_from_db(args.data_dir, args.endpoint_id, args.model)
|
||||
seeds = load_seeds(args.seeds)
|
||||
done = completed_ids(args.output)
|
||||
pending = [seed for seed in seeds if str(seed["seed_id"]) not in done]
|
||||
if args.limit is not None:
|
||||
pending = pending[:args.limit]
|
||||
batches = chunks(pending, args.batch_size)
|
||||
failures: list[str] = []
|
||||
cooked_count = 0
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=args.workers) as pool:
|
||||
future_map = {pool.submit(cook_batch, endpoint, batch): batch for batch in batches}
|
||||
for future in concurrent.futures.as_completed(future_map):
|
||||
batch = future_map[future]
|
||||
try:
|
||||
rows = future.result()
|
||||
append_rows(args.output, rows)
|
||||
cooked_count += len(rows)
|
||||
print(json.dumps({"cooked": len(done) + cooked_count, "total": len(seeds)}), flush=True)
|
||||
except Exception as exc:
|
||||
failures.extend(str(seed["seed_id"]) for seed in batch)
|
||||
print(json.dumps({"batch_failed": len(batch), "error": repr(exc)}), flush=True)
|
||||
counts = Counter()
|
||||
if args.output.exists():
|
||||
for line in args.output.read_text(encoding="utf-8").splitlines():
|
||||
try:
|
||||
counts[json.loads(line).get("family", "unknown")] += 1
|
||||
except (json.JSONDecodeError, AttributeError):
|
||||
pass
|
||||
print(json.dumps({
|
||||
"source_seeds": len(seeds), "already_done": len(done),
|
||||
"cooked_now": cooked_count, "failed": len(failures),
|
||||
"remaining": len(seeds) - len(done) - cooked_count,
|
||||
"families": dict(sorted(counts.items())),
|
||||
}, indent=2))
|
||||
return 2 if failures else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -21,6 +21,36 @@ if str(ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
|
||||
from scripts.repair_sft_corpus_with_kimi import endpoint, parse_json # noqa: E402
|
||||
from src.tool_schemas import FUNCTION_TOOL_SCHEMAS # noqa: E402
|
||||
|
||||
|
||||
ALL_TOOL_NAMES = frozenset(
|
||||
str(schema.get("function", {}).get("name") or "")
|
||||
for schema in FUNCTION_TOOL_SCHEMAS
|
||||
if schema.get("function", {}).get("name") and schema.get("function", {}).get("name") != "host_shell"
|
||||
)
|
||||
|
||||
|
||||
def compact_tool_catalog() -> list[dict[str, Any]]:
|
||||
"""Expose the complete product tool vocabulary to the scenario author."""
|
||||
catalog = []
|
||||
for schema in FUNCTION_TOOL_SCHEMAS:
|
||||
function = schema.get("function") or {}
|
||||
name = str(function.get("name") or "")
|
||||
if not name or name == "host_shell":
|
||||
continue
|
||||
parameters = function.get("parameters") or {}
|
||||
properties = parameters.get("properties") or {}
|
||||
entry: dict[str, Any] = {
|
||||
"name": name,
|
||||
"purpose": str(function.get("description") or "")[:700],
|
||||
"required": list(parameters.get("required") or []),
|
||||
}
|
||||
action = properties.get("action") if isinstance(properties, dict) else None
|
||||
if isinstance(action, dict) and isinstance(action.get("enum"), list):
|
||||
entry["actions"] = action["enum"]
|
||||
catalog.append(entry)
|
||||
return catalog
|
||||
|
||||
OWNERS = ["sft_maya_ops", "sft_jules_research", "sft_nora_design", "sft_omar_finance"]
|
||||
EFFECTFUL_WITHOUT_DRY_RUN = {
|
||||
@@ -107,7 +137,8 @@ For each case return:
|
||||
- cleanup: fixture types that must be restored or removed
|
||||
|
||||
Rules:
|
||||
- The source is a behavioral seed, not text to paraphrase. Preserve its useful tool strategy and outcome while changing scenario, entities, wording, and follow-up style.
|
||||
- The source is behavioral evidence, not text to paraphrase and not an allowlist. Use the complete tool catalog to independently identify the best intended tool for each new turn. Preserve the useful outcome while changing scenario, entities, wording, and follow-up style.
|
||||
- Distinguish tools with overlapping names by their documented purpose and required arguments. If the source used a less suitable tool, choose the catalog tool that actually fulfills the new prompt.
|
||||
- Make the turns one coherent conversation. Later turns should naturally build on earlier tool results.
|
||||
- Use exact IDs/titles/UIDs from the target inventory for read/update/delete workflows, or create a marker-scoped object first. Never invent an existing object.
|
||||
- Give temporary objects ordinary, project-specific names that a real user might choose. Keep them distinct from supplied inventory names, but never expose run IDs, markers, fixtures, tests, audits, or cleanup mechanics to the user.
|
||||
@@ -127,7 +158,7 @@ Rules:
|
||||
"""
|
||||
if STYLE_CONTRACT.exists():
|
||||
system += "\nApply this speaking-style contract to every generated conversation:\n\n" + STYLE_CONTRACT.read_text(encoding="utf-8")
|
||||
allowed_tools = sorted({tool for tool in seed["tools"]} | {"ask_user", "ui_control"})
|
||||
allowed_tools = sorted(ALL_TOOL_NAMES | set(seed["tools"]))
|
||||
payload = {
|
||||
"model": ep["model"],
|
||||
"messages": [
|
||||
@@ -136,6 +167,7 @@ Rules:
|
||||
"seed": compact_seed(seed),
|
||||
"current_date": date.today().isoformat(),
|
||||
"allowed_tools": allowed_tools,
|
||||
"tool_catalog": compact_tool_catalog(),
|
||||
"targets": [compact_environment(target) for target in targets],
|
||||
}, ensure_ascii=False)},
|
||||
],
|
||||
@@ -176,7 +208,7 @@ def validate_case(
|
||||
turns = raw.get("turns")
|
||||
if not isinstance(turns, list) or not 3 <= len(turns) <= 4:
|
||||
raise ValueError("case must contain 3-4 turns")
|
||||
allowed = set(seed["tools"]) | {"ask_user", "ui_control"}
|
||||
allowed = set(ALL_TOOL_NAMES) | set(seed["tools"])
|
||||
clean_turns = []
|
||||
normalized = set()
|
||||
for index, turn in enumerate(turns, 1):
|
||||
|
||||
@@ -0,0 +1,194 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Independently classify seeded live-replay failures with a full tool catalog."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import concurrent.futures
|
||||
import json
|
||||
import sys
|
||||
import uuid
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from dotenv import load_dotenv
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
load_dotenv(ROOT / ".env")
|
||||
if str(ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
|
||||
from scripts.generate_sft_environment_expansion import compact_tool_catalog # noqa: E402
|
||||
from scripts.repair_sft_corpus_with_kimi import endpoint, parse_json # noqa: E402
|
||||
|
||||
|
||||
def compact_evidence(result: dict[str, Any]) -> dict[str, Any]:
|
||||
turns = []
|
||||
for turn in result.get("turns") or []:
|
||||
contract = next(
|
||||
(event for event in turn.get("evidence") or [] if event.get("type") == "turn_contract"),
|
||||
{},
|
||||
)
|
||||
outputs = [
|
||||
str(event.get("output") or "")[:1200]
|
||||
for event in turn.get("evidence") or []
|
||||
if event.get("type") == "tool_output"
|
||||
]
|
||||
errors = [
|
||||
event for event in turn.get("evidence") or []
|
||||
if event.get("type") in {"error", "parse_error"}
|
||||
]
|
||||
turns.append({
|
||||
"id": turn.get("id"),
|
||||
"prompt": turn.get("prompt"),
|
||||
"expected_tools": turn.get("expected_tools"),
|
||||
"observed_tools": turn.get("observed_tools"),
|
||||
"answer": str(turn.get("answer") or "")[:1800],
|
||||
"deterministic_failures": turn.get("failures"),
|
||||
"upstream_failed": bool(turn.get("upstream_failed", False)),
|
||||
"contract": {
|
||||
"capabilities": contract.get("capabilities") or [],
|
||||
"required": contract.get("required") or [],
|
||||
"offered": contract.get("offered") or [],
|
||||
"unavailable": contract.get("unavailable") or [],
|
||||
"selection_mode": contract.get("selection_mode"),
|
||||
"schema_mode": contract.get("schema_mode"),
|
||||
},
|
||||
"tool_outputs": outputs,
|
||||
"stream_errors": errors,
|
||||
})
|
||||
return {
|
||||
"case_id": result.get("case_id"),
|
||||
"seed_family_id": result.get("seed_family_id"),
|
||||
"owner": result.get("owner"),
|
||||
"deterministic_pass": result.get("pass"),
|
||||
"deterministic_failures": result.get("failures"),
|
||||
"turns": turns,
|
||||
}
|
||||
|
||||
|
||||
def judge_once(ep: dict[str, str], case: dict[str, Any], result: dict[str, Any], timeout: float) -> dict[str, Any]:
|
||||
system = """You audit a real tool-agent replay. Return strict JSON only:
|
||||
{"case_id":"...","case_valid":true,"overall_class":"pass|bad_generated_case|harness_routing|harness_execution|model_sft|tool_backend|mixed","confidence":0.0,"summary":"...","turns":[{"id":"...","valid_expectation":true,"best_tools":["..."],"classification":"pass|bad_generated_case|harness_routing|harness_execution|model_sft|tool_backend","reason":"...","generic_repair":"..."}]}
|
||||
|
||||
Use the COMPLETE tool catalog, the generated conversation, and the observed immutable turn contract.
|
||||
- First decide whether the prompt and supplied environment actually support the expected tool. Reject ambiguous or invented expectations.
|
||||
- harness_routing: the correct family/tool was absent, the wrong family was required, or the contract offered zero/wrong tools.
|
||||
- harness_execution: the contract selected the correct deterministic operation but failed to execute/render it independently of model choice.
|
||||
- model_sft: the correct tools were offered and executable, but the model chose the wrong tool/action, malformed arguments, leaked reasoning, or falsely answered.
|
||||
- tool_backend: a correct call failed in the underlying service.
|
||||
- Do not propose phrase-specific rules. Generic repairs must describe a semantic boundary or contract invariant.
|
||||
- A prior turn's successful result can establish references for a follow-up. An active document fixture means deictic editing prompts may validly target document tools.
|
||||
- Judge the complete 3-4 turn trajectory. If an earlier failed operation removed the object or evidence needed later, mark later failures as causal fallout in the reason instead of inventing another root cause.
|
||||
- Recommend a harness patch only for a semantic category that should generalize across varied wording and entities. Never recommend a literal prompt/entity/domain-name rule. A single case can justify only a clear contract, authorization, or security invariant; otherwise request more variants.
|
||||
- Do not reveal or reconstruct hidden benchmark answers. Judge only the supplied synthetic replay.
|
||||
"""
|
||||
payload = {
|
||||
"model": ep["model"],
|
||||
"messages": [
|
||||
{"role": "system", "content": system},
|
||||
{"role": "user", "content": json.dumps({
|
||||
"tool_catalog": compact_tool_catalog(),
|
||||
"generated_case": case,
|
||||
"live_result": compact_evidence(result),
|
||||
}, ensure_ascii=False)},
|
||||
],
|
||||
"temperature": 0,
|
||||
"max_tokens": 5000,
|
||||
"response_format": {"type": "json_object"},
|
||||
}
|
||||
request = urllib.request.Request(
|
||||
ep["base_url"].rstrip("/") + "/chat/completions",
|
||||
data=json.dumps(payload).encode(),
|
||||
headers={"Content-Type": "application/json", "Authorization": f"Bearer {ep['api_key']}"},
|
||||
method="POST",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
body = json.loads(response.read().decode())
|
||||
message = body["choices"][0]["message"]
|
||||
verdict = parse_json(str(message.get("content") or message.get("reasoning_content") or ""))
|
||||
if str(verdict.get("case_id") or "") != str(result.get("case_id") or ""):
|
||||
raise ValueError("judge returned the wrong case_id")
|
||||
return verdict
|
||||
|
||||
|
||||
def judge(
|
||||
ep: dict[str, str],
|
||||
case: dict[str, Any],
|
||||
result: dict[str, Any],
|
||||
timeout: float,
|
||||
retries: int,
|
||||
) -> dict[str, Any]:
|
||||
"""Retry provider/JSON failures without changing the case being judged."""
|
||||
last_error: Exception | None = None
|
||||
for _attempt in range(max(0, retries) + 1):
|
||||
try:
|
||||
return judge_once(ep, case, result, timeout)
|
||||
except Exception as exc:
|
||||
last_error = exc
|
||||
assert last_error is not None
|
||||
raise last_error
|
||||
|
||||
|
||||
def atomic_write(path: Path, payload: dict[str, Any]) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp")
|
||||
temporary.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
temporary.replace(path)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--cases", type=Path, required=True)
|
||||
parser.add_argument("--results", type=Path, required=True)
|
||||
parser.add_argument("--out", type=Path, required=True)
|
||||
parser.add_argument("--endpoint-id", default="e17d4b33")
|
||||
parser.add_argument("--model", default="deepseek-v4-pro")
|
||||
parser.add_argument("--workers", type=int, default=4)
|
||||
parser.add_argument("--timeout", type=float, default=180)
|
||||
parser.add_argument("--retries", type=int, default=2)
|
||||
parser.add_argument("--case-id", action="append", help="Judge only the named case; repeatable")
|
||||
args = parser.parse_args()
|
||||
|
||||
cases = {row["case_id"]: row for row in json.loads(args.cases.read_text(encoding="utf-8"))["cases"]}
|
||||
results = json.loads(args.results.read_text(encoding="utf-8"))["results"]
|
||||
if args.case_id:
|
||||
wanted = set(args.case_id)
|
||||
results = [row for row in results if row["case_id"] in wanted]
|
||||
ep = endpoint(args.endpoint_id, args.model)
|
||||
verdicts: dict[str, dict[str, Any]] = {}
|
||||
errors: list[dict[str, str]] = []
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=max(1, args.workers)) as pool:
|
||||
futures = {
|
||||
pool.submit(
|
||||
judge,
|
||||
ep,
|
||||
cases[result["case_id"]],
|
||||
result,
|
||||
args.timeout,
|
||||
args.retries,
|
||||
): result
|
||||
for result in results
|
||||
}
|
||||
for future in concurrent.futures.as_completed(futures):
|
||||
result = futures[future]
|
||||
case_id = str(result["case_id"])
|
||||
try:
|
||||
verdicts[case_id] = future.result()
|
||||
print(f"judged {case_id}: {verdicts[case_id].get('overall_class')}", flush=True)
|
||||
except Exception as exc:
|
||||
errors.append({"case_id": case_id, "error": repr(exc)})
|
||||
print(f"failed {case_id}: {exc!r}", flush=True)
|
||||
atomic_write(args.out, {"verdicts": list(verdicts.values()), "errors": errors})
|
||||
ordered = [verdicts[row["case_id"]] for row in results if row["case_id"] in verdicts]
|
||||
atomic_write(args.out, {"verdicts": ordered, "errors": errors})
|
||||
counts: dict[str, int] = {}
|
||||
for row in ordered:
|
||||
key = str(row.get("overall_class") or "unknown")
|
||||
counts[key] = counts.get(key, 0) + 1
|
||||
print(json.dumps({"judged": len(ordered), "errors": len(errors), "classes": counts}, indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,282 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Snapshot or restore durable state for one Odysseus SFT fixture owner.
|
||||
|
||||
Sessions and chat messages are intentionally excluded so replay evidence keeps
|
||||
working. Only owner-scoped tool data and its dependent rows are managed.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import base64
|
||||
import json
|
||||
import re
|
||||
import shutil
|
||||
import sqlite3
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
DIRECT_TABLES = (
|
||||
"notes",
|
||||
"memories",
|
||||
"scheduled_tasks",
|
||||
"documents",
|
||||
"calendars",
|
||||
"editor_drafts",
|
||||
"notification_logs",
|
||||
"caldav_deleted_events",
|
||||
)
|
||||
CHILD_TABLES = {
|
||||
"document_versions": ("documents", "document_id", "id"),
|
||||
"task_runs": ("scheduled_tasks", "task_id", "id"),
|
||||
"calendar_events": ("calendars", "calendar_id", "id"),
|
||||
}
|
||||
|
||||
|
||||
def _read_json(path: Path, default: Any) -> Any:
|
||||
try:
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError):
|
||||
return default
|
||||
|
||||
|
||||
def _atomic_json(path: Path, payload: Any) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary = path.with_name(f".{path.name}.fixture-state.tmp")
|
||||
temporary.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
temporary.replace(path)
|
||||
|
||||
|
||||
def _skill_owner(path: Path) -> str:
|
||||
try:
|
||||
text = path.read_text(encoding="utf-8")
|
||||
except (OSError, UnicodeDecodeError):
|
||||
return ""
|
||||
match = re.search(r'^owner:\s*["\']?([^"\'\n#]+)', text, re.M)
|
||||
return match.group(1).strip() if match else ""
|
||||
|
||||
|
||||
def _snapshot_external(data_dir: Path, owner: str) -> dict[str, Any]:
|
||||
prefs = _read_json(data_dir / "user_prefs.json", {"_users": {}})
|
||||
blocked = _read_json(data_dir / "email_blocked_senders.json", {"owners": {}})
|
||||
email_payload = _read_json(data_dir / "fixture_email_messages.json", {"messages": []})
|
||||
email_rows = email_payload.get("messages", []) if isinstance(email_payload, dict) else email_payload
|
||||
skills_root = data_dir / "skills"
|
||||
skill_files: list[dict[str, str]] = []
|
||||
skill_dirs: list[str] = []
|
||||
if skills_root.exists():
|
||||
for skill_md in skills_root.rglob("SKILL.md"):
|
||||
if _skill_owner(skill_md) != owner:
|
||||
continue
|
||||
directory = skill_md.parent
|
||||
skill_dirs.append(str(directory.relative_to(skills_root)))
|
||||
for path in directory.rglob("*"):
|
||||
if path.is_file():
|
||||
skill_files.append({
|
||||
"path": str(path.relative_to(skills_root)),
|
||||
"base64": base64.b64encode(path.read_bytes()).decode("ascii"),
|
||||
})
|
||||
usage = _read_json(skills_root / "_usage.json", {})
|
||||
return {
|
||||
"prefs_present": owner in ((prefs.get("_users") or {}) if isinstance(prefs, dict) else {}),
|
||||
"prefs": ((prefs.get("_users") or {}).get(owner) if isinstance(prefs, dict) else None),
|
||||
"blocked_present": owner in ((blocked.get("owners") or {}) if isinstance(blocked, dict) else {}),
|
||||
"blocked_senders": ((blocked.get("owners") or {}).get(owner) if isinstance(blocked, dict) else None),
|
||||
"email_rows": [
|
||||
row for row in (email_rows if isinstance(email_rows, list) else [])
|
||||
if isinstance(row, dict) and str(row.get("owner") or "") == owner
|
||||
],
|
||||
"skill_dirs": sorted(set(skill_dirs)),
|
||||
"skill_files": skill_files,
|
||||
"skill_usage": {
|
||||
key: value for key, value in (usage.items() if isinstance(usage, dict) else [])
|
||||
if str(key).startswith(f"{owner}::")
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _table_exists(db: sqlite3.Connection, table: str) -> bool:
|
||||
return db.execute(
|
||||
"SELECT 1 FROM sqlite_master WHERE type='table' AND name=?", (table,),
|
||||
).fetchone() is not None
|
||||
|
||||
|
||||
def _columns(db: sqlite3.Connection, table: str) -> list[str]:
|
||||
return [str(row[1]) for row in db.execute(f'PRAGMA table_info("{table}")')]
|
||||
|
||||
|
||||
def _rows(db: sqlite3.Connection, table: str, where: str, values: tuple[Any, ...]) -> list[dict[str, Any]]:
|
||||
db.row_factory = sqlite3.Row
|
||||
return [dict(row) for row in db.execute(f'SELECT * FROM "{table}" WHERE {where}', values)]
|
||||
|
||||
|
||||
def snapshot_owner(db_path: Path, owner: str, data_dir: Path | None = None) -> dict[str, Any]:
|
||||
db = sqlite3.connect(db_path)
|
||||
try:
|
||||
tables: dict[str, list[dict[str, Any]]] = {}
|
||||
for table in DIRECT_TABLES:
|
||||
if _table_exists(db, table) and "owner" in _columns(db, table):
|
||||
tables[table] = _rows(db, table, '"owner"=?', (owner,))
|
||||
for table, (parent, foreign_key, parent_key) in CHILD_TABLES.items():
|
||||
if not _table_exists(db, table):
|
||||
continue
|
||||
parent_ids = [row[parent_key] for row in tables.get(parent, [])]
|
||||
if not parent_ids:
|
||||
tables[table] = []
|
||||
continue
|
||||
placeholders = ",".join("?" for _ in parent_ids)
|
||||
tables[table] = _rows(
|
||||
db, table, f'"{foreign_key}" IN ({placeholders})', tuple(parent_ids),
|
||||
)
|
||||
return {
|
||||
"format": "odysseus-owner-fixture-v1",
|
||||
"created_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
||||
"source_db": str(db_path),
|
||||
"owner": owner,
|
||||
"tables": tables,
|
||||
"counts": {table: len(rows) for table, rows in tables.items()},
|
||||
"external": _snapshot_external(data_dir or db_path.parent, owner),
|
||||
}
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
def _delete_owner_rows(db: sqlite3.Connection, owner: str) -> None:
|
||||
for table, (parent, foreign_key, parent_key) in CHILD_TABLES.items():
|
||||
if not (_table_exists(db, table) and _table_exists(db, parent)):
|
||||
continue
|
||||
db.execute(
|
||||
f'DELETE FROM "{table}" WHERE "{foreign_key}" IN '
|
||||
f'(SELECT "{parent_key}" FROM "{parent}" WHERE "owner"=?)',
|
||||
(owner,),
|
||||
)
|
||||
for table in DIRECT_TABLES:
|
||||
if _table_exists(db, table) and "owner" in _columns(db, table):
|
||||
db.execute(f'DELETE FROM "{table}" WHERE "owner"=?', (owner,))
|
||||
|
||||
|
||||
def _restore_external(data_dir: Path, external: dict[str, Any], owner: str) -> None:
|
||||
prefs_path = data_dir / "user_prefs.json"
|
||||
prefs = _read_json(prefs_path, {"_users": {}})
|
||||
users = prefs.setdefault("_users", {})
|
||||
if external.get("prefs_present"):
|
||||
users[owner] = external.get("prefs")
|
||||
else:
|
||||
users.pop(owner, None)
|
||||
_atomic_json(prefs_path, prefs)
|
||||
|
||||
blocked_path = data_dir / "email_blocked_senders.json"
|
||||
blocked = _read_json(blocked_path, {"owners": {}})
|
||||
blocked_owners = blocked.setdefault("owners", {})
|
||||
if external.get("blocked_present"):
|
||||
blocked_owners[owner] = external.get("blocked_senders")
|
||||
else:
|
||||
blocked_owners.pop(owner, None)
|
||||
_atomic_json(blocked_path, blocked)
|
||||
|
||||
email_path = data_dir / "fixture_email_messages.json"
|
||||
email_payload = _read_json(email_path, {"messages": []})
|
||||
email_rows = email_payload.get("messages", []) if isinstance(email_payload, dict) else email_payload
|
||||
retained = [
|
||||
row for row in (email_rows if isinstance(email_rows, list) else [])
|
||||
if not (isinstance(row, dict) and str(row.get("owner") or "") == owner)
|
||||
]
|
||||
restored_rows = retained + list(external.get("email_rows") or [])
|
||||
if isinstance(email_payload, dict):
|
||||
email_payload["messages"] = restored_rows
|
||||
else:
|
||||
email_payload = restored_rows
|
||||
_atomic_json(email_path, email_payload)
|
||||
|
||||
skills_root = data_dir / "skills"
|
||||
if skills_root.exists():
|
||||
for skill_md in list(skills_root.rglob("SKILL.md")):
|
||||
if _skill_owner(skill_md) == owner:
|
||||
shutil.rmtree(skill_md.parent, ignore_errors=True)
|
||||
for entry in external.get("skill_files") or []:
|
||||
relative = Path(str(entry.get("path") or ""))
|
||||
if not relative.parts or relative.is_absolute() or ".." in relative.parts:
|
||||
raise ValueError("unsafe skill path in fixture snapshot")
|
||||
destination = skills_root / relative
|
||||
destination.parent.mkdir(parents=True, exist_ok=True)
|
||||
destination.write_bytes(base64.b64decode(entry.get("base64") or ""))
|
||||
usage_path = skills_root / "_usage.json"
|
||||
usage = _read_json(usage_path, {})
|
||||
usage = usage if isinstance(usage, dict) else {}
|
||||
usage = {key: value for key, value in usage.items() if not str(key).startswith(f"{owner}::")}
|
||||
usage.update(external.get("skill_usage") or {})
|
||||
_atomic_json(usage_path, usage)
|
||||
|
||||
|
||||
def restore_owner(target_db: Path, snapshot: dict[str, Any], owner: str,
|
||||
data_dir: Path | None = None) -> None:
|
||||
if snapshot.get("format") != "odysseus-owner-fixture-v1":
|
||||
raise ValueError("unsupported fixture snapshot format")
|
||||
if str(snapshot.get("owner") or "") != owner:
|
||||
raise ValueError("snapshot owner does not match requested owner")
|
||||
tables = snapshot.get("tables")
|
||||
if not isinstance(tables, dict):
|
||||
raise ValueError("snapshot has no tables")
|
||||
|
||||
db = sqlite3.connect(target_db, timeout=60)
|
||||
try:
|
||||
db.execute("BEGIN IMMEDIATE")
|
||||
_delete_owner_rows(db, owner)
|
||||
insertion_order = (*DIRECT_TABLES, *CHILD_TABLES)
|
||||
for table in insertion_order:
|
||||
rows = tables.get(table) or []
|
||||
if not rows or not _table_exists(db, table):
|
||||
continue
|
||||
target_columns = set(_columns(db, table))
|
||||
columns = [column for column in rows[0] if column in target_columns]
|
||||
quoted = ",".join(f'"{column}"' for column in columns)
|
||||
placeholders = ",".join("?" for _ in columns)
|
||||
db.executemany(
|
||||
f'INSERT INTO "{table}" ({quoted}) VALUES ({placeholders})',
|
||||
[[row.get(column) for column in columns] for row in rows],
|
||||
)
|
||||
db.commit()
|
||||
except Exception:
|
||||
db.rollback()
|
||||
raise
|
||||
finally:
|
||||
db.close()
|
||||
external = snapshot.get("external")
|
||||
if isinstance(external, dict):
|
||||
_restore_external(data_dir or target_db.parent, external, owner)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--db", type=Path, required=True, help="Live target app.db")
|
||||
parser.add_argument("--data-dir", type=Path, help="External fixture state directory; defaults to DB parent")
|
||||
parser.add_argument("--owner", default="sft_alex_creator")
|
||||
parser.add_argument("--snapshot-out", type=Path)
|
||||
parser.add_argument("--restore-json", type=Path)
|
||||
parser.add_argument("--restore-from-db", type=Path)
|
||||
args = parser.parse_args()
|
||||
operations = sum(bool(value) for value in (
|
||||
args.snapshot_out, args.restore_json, args.restore_from_db,
|
||||
))
|
||||
if operations != 1:
|
||||
parser.error("choose exactly one of --snapshot-out, --restore-json, or --restore-from-db")
|
||||
|
||||
if args.snapshot_out:
|
||||
payload = snapshot_owner(args.db, args.owner, args.data_dir)
|
||||
args.snapshot_out.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.snapshot_out.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
print(json.dumps({"snapshot": str(args.snapshot_out), "counts": payload["counts"]}, indent=2))
|
||||
return
|
||||
|
||||
if args.restore_json:
|
||||
payload = json.loads(args.restore_json.read_text(encoding="utf-8"))
|
||||
else:
|
||||
payload = snapshot_owner(args.restore_from_db, args.owner, args.data_dir)
|
||||
restore_owner(args.db, payload, args.owner, args.data_dir)
|
||||
print(json.dumps({"restored_owner": args.owner, "counts": payload["counts"]}, indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
File diff suppressed because it is too large
Load Diff
@@ -14,8 +14,10 @@ from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from cryptography.fernet import Fernet
|
||||
from dotenv import load_dotenv
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
load_dotenv(ROOT / ".env")
|
||||
|
||||
|
||||
def decrypt(value: str) -> str:
|
||||
@@ -26,7 +28,13 @@ def decrypt(value: str) -> str:
|
||||
|
||||
|
||||
def endpoint(endpoint_id: str, model: str) -> dict[str, str]:
|
||||
con = sqlite3.connect(ROOT / "data" / "app.db")
|
||||
# Honor the same configured data directory as the live Odysseus service.
|
||||
# Eval worktrees commonly keep only source under ROOT while 7011 points at
|
||||
# the canonical shared database via ODYSSEUS_DATA_DIR.
|
||||
from src.constants import DATA_DIR
|
||||
|
||||
data_dir = Path(DATA_DIR)
|
||||
con = sqlite3.connect(data_dir / "app.db")
|
||||
con.row_factory = sqlite3.Row
|
||||
row = con.execute(
|
||||
"SELECT base_url,api_key FROM model_endpoints WHERE id=? AND is_enabled=1",
|
||||
@@ -34,7 +42,10 @@ def endpoint(endpoint_id: str, model: str) -> dict[str, str]:
|
||||
).fetchone()
|
||||
if row is None:
|
||||
raise RuntimeError(f"Enabled endpoint not found: {endpoint_id}")
|
||||
return {"base_url": row["base_url"], "api_key": decrypt(row["api_key"]), "model": model}
|
||||
value = str(row["api_key"] or "")
|
||||
if value.startswith("enc:"):
|
||||
value = Fernet((data_dir / ".app_key").read_bytes()).decrypt(value[4:].encode()).decode()
|
||||
return {"base_url": row["base_url"], "api_key": value, "model": model}
|
||||
|
||||
|
||||
def parse_json(text: str) -> dict[str, Any]:
|
||||
|
||||
@@ -2,22 +2,25 @@
|
||||
"""Execute generated SFT workflows through Odysseus with rollback and gating."""
|
||||
|
||||
from __future__ import annotations
|
||||
import os
|
||||
|
||||
import argparse
|
||||
import contextlib
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import signal
|
||||
import time
|
||||
import uuid
|
||||
from datetime import datetime, timedelta
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
from dotenv import load_dotenv
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
load_dotenv(ROOT / ".env")
|
||||
if str(ROOT) not in __import__("sys").path:
|
||||
__import__("sys").path.insert(0, str(ROOT))
|
||||
|
||||
@@ -37,12 +40,15 @@ from scripts.eval_odysseus_tool_use import ( # noqa: E402
|
||||
_visible_event_text,
|
||||
)
|
||||
|
||||
DATA_DIR = ROOT / "data"
|
||||
from src.constants import DATA_DIR as CONFIGURED_DATA_DIR # noqa: E402
|
||||
|
||||
DATA_DIR = Path(CONFIGURED_DATA_DIR)
|
||||
BAD_ANSWER_RE = re.compile(
|
||||
r"\b(?:can't|cannot|don't have|do not have|not available|no .*tool|enable .*integration|"
|
||||
r"invalid credentials|not authenticated|i can only|i'm unable)\b",
|
||||
re.I,
|
||||
)
|
||||
_COOKIE_CACHE: dict[str, str] = {}
|
||||
TOOL_FAILURE_RE = re.compile(r"(?:tool (?:failed|error)|exit_code[^\d]*[1-9]|permission denied|not found)", re.I)
|
||||
INTERNAL_NARRATION_RE = re.compile(
|
||||
r"(?:^|\n)(?:The user (?:asks|asked|wants)|I (?:should|need to|can see)|Let me (?:call|use|retry|try))\b",
|
||||
@@ -65,7 +71,158 @@ def atomic_json(path: Path, payload: Any) -> None:
|
||||
temp.replace(path)
|
||||
|
||||
|
||||
def install_fixture_environments(path: Path) -> dict[str, int]:
|
||||
"""Materialize an inventory snapshot for an isolated replay app."""
|
||||
payload = json.loads(path.read_text(encoding="utf-8"))
|
||||
environments = payload.get("environments", []) if isinstance(payload, dict) else []
|
||||
messages: list[dict[str, Any]] = []
|
||||
counts = {
|
||||
"emails": 0, "notes": 0, "memories": 0, "documents": 0,
|
||||
"tasks": 0, "calendars": 0, "events": 0,
|
||||
}
|
||||
db = SessionLocal()
|
||||
owners = [
|
||||
str(row.get("owner") or "").strip()
|
||||
for row in environments if isinstance(row, dict)
|
||||
]
|
||||
try:
|
||||
for owner in filter(None, owners):
|
||||
document_ids = [
|
||||
value[0] for value in db.query(Document.id).filter(Document.owner == owner).all()
|
||||
]
|
||||
if document_ids:
|
||||
db.query(DocumentVersion).filter(
|
||||
DocumentVersion.document_id.in_(document_ids)
|
||||
).delete(synchronize_session=False)
|
||||
calendar_ids = [
|
||||
value[0] for value in db.query(CalendarCal.id).filter(CalendarCal.owner == owner).all()
|
||||
]
|
||||
if calendar_ids:
|
||||
db.query(CalendarEvent).filter(
|
||||
CalendarEvent.calendar_id.in_(calendar_ids)
|
||||
).delete(synchronize_session=False)
|
||||
db.query(Document).filter(Document.owner == owner).delete(synchronize_session=False)
|
||||
db.query(Note).filter(Note.owner == owner).delete(synchronize_session=False)
|
||||
db.query(Memory).filter(Memory.owner == owner).delete(synchronize_session=False)
|
||||
db.query(ScheduledTask).filter(ScheduledTask.owner == owner).delete(synchronize_session=False)
|
||||
db.query(CalendarCal).filter(CalendarCal.owner == owner).delete(synchronize_session=False)
|
||||
|
||||
for environment in environments:
|
||||
if not isinstance(environment, dict):
|
||||
continue
|
||||
owner = str(environment.get("owner") or "").strip()
|
||||
for row in environment.get("notes") or []:
|
||||
db.add(Note(
|
||||
id=str(row.get("id") or uuid.uuid4()), owner=owner,
|
||||
title=str(row.get("title") or ""), content=str(row.get("content") or ""),
|
||||
note_type=str(row.get("type") or "note"), label=row.get("label"),
|
||||
archived=False, source="user",
|
||||
))
|
||||
counts["notes"] += 1
|
||||
for row in environment.get("memories") or []:
|
||||
db.add(Memory(
|
||||
id=str(row.get("id") or uuid.uuid4()), owner=owner,
|
||||
text=str(row.get("text") or ""),
|
||||
category=str(row.get("category") or "fact"), source="user",
|
||||
))
|
||||
counts["memories"] += 1
|
||||
for row in environment.get("documents") or []:
|
||||
document_id = str(row.get("id") or uuid.uuid4())
|
||||
content = str(row.get("content") or "")
|
||||
db.add(Document(
|
||||
id=document_id, owner=owner, title=str(row.get("title") or "Untitled"),
|
||||
language=str(row.get("language") or "text"), current_content=content,
|
||||
version_count=1, is_active=True, archived=False,
|
||||
))
|
||||
db.add(DocumentVersion(
|
||||
id=str(uuid.uuid4()), document_id=document_id, version_number=1,
|
||||
content=content, summary="Isolated replay fixture", source="user",
|
||||
))
|
||||
counts["documents"] += 1
|
||||
for row in environment.get("tasks") or []:
|
||||
db.add(ScheduledTask(
|
||||
id=str(row.get("id") or uuid.uuid4()), owner=owner,
|
||||
name=str(row.get("name") or "Untitled Task"),
|
||||
status=str(row.get("status") or "active"),
|
||||
schedule=row.get("schedule"), task_type="llm",
|
||||
))
|
||||
counts["tasks"] += 1
|
||||
calendar_map: dict[str, str] = {}
|
||||
for row in environment.get("calendars") or []:
|
||||
calendar_id = str(row.get("id") or uuid.uuid4())
|
||||
calendar_map[calendar_id] = calendar_id
|
||||
db.add(CalendarCal(
|
||||
id=calendar_id, owner=owner, name=str(row.get("name") or "Personal"),
|
||||
source=str(row.get("source") or "local"),
|
||||
))
|
||||
counts["calendars"] += 1
|
||||
default_calendar = next(iter(calendar_map), None)
|
||||
for row in environment.get("events") or []:
|
||||
if default_calendar is None:
|
||||
default_calendar = str(uuid.uuid4())
|
||||
db.add(CalendarCal(
|
||||
id=default_calendar, owner=owner, name="Personal", source="local",
|
||||
))
|
||||
counts["calendars"] += 1
|
||||
start = datetime.fromisoformat(str(row.get("start") or "").replace("Z", "+00:00"))
|
||||
db.add(CalendarEvent(
|
||||
uid=str(row.get("uid") or uuid.uuid4()), calendar_id=default_calendar,
|
||||
summary=str(row.get("summary") or ""), dtstart=start,
|
||||
dtend=start + timedelta(hours=1), all_day=bool(row.get("all_day")),
|
||||
))
|
||||
counts["events"] += 1
|
||||
db.commit()
|
||||
except Exception:
|
||||
db.rollback()
|
||||
raise
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
for environment in environments:
|
||||
if not isinstance(environment, dict):
|
||||
continue
|
||||
owner = str(environment.get("owner") or "").strip()
|
||||
profile = environment.get("profile") if isinstance(environment.get("profile"), dict) else {}
|
||||
primary_name = str(profile.get("primary_account") or "Primary Inbox")
|
||||
secondary_name = str(profile.get("secondary_account") or "Secondary Inbox")
|
||||
for source in environment.get("emails") or []:
|
||||
if not isinstance(source, dict):
|
||||
continue
|
||||
row = dict(source)
|
||||
account = str(row.get("account") or primary_name)
|
||||
secondary = account == secondary_name
|
||||
row.update({
|
||||
"owner": owner,
|
||||
"account": account,
|
||||
"account_email": str(profile.get("secondary" if secondary else "primary") or owner),
|
||||
"account_id": "secondary-inbox" if secondary else "primary-inbox",
|
||||
"folder": str(row.get("folder") or "INBOX"),
|
||||
"body": str(row.get("body") or (
|
||||
f"Fixture message for: {row.get('subject') or '(no subject)'}. "
|
||||
"Please review the referenced materials and reply with the next step."
|
||||
)),
|
||||
})
|
||||
messages.append(row)
|
||||
atomic_json(DATA_DIR / "fixture_email_messages.json", {"messages": messages})
|
||||
counts["emails"] = len(messages)
|
||||
return counts
|
||||
|
||||
|
||||
def login(client: httpx.Client, base_url: str, owner: str, password: str) -> None:
|
||||
token = _COOKIE_CACHE.get(owner)
|
||||
if not token:
|
||||
sessions_path = DATA_DIR / "sessions.json"
|
||||
if sessions_path.exists():
|
||||
with contextlib.suppress(Exception):
|
||||
sessions = json.loads(sessions_path.read_text(encoding="utf-8"))
|
||||
token = next(
|
||||
key for key, value in reversed(list(sessions.items()))
|
||||
if isinstance(value, dict) and value.get("username") == owner
|
||||
)
|
||||
if token:
|
||||
_COOKIE_CACHE[owner] = token
|
||||
client.cookies.set("odysseus_session", token)
|
||||
return
|
||||
response = client.post(
|
||||
base_url.rstrip("/") + "/api/auth/login",
|
||||
json={"username": owner, "password": password, "remember": True},
|
||||
@@ -74,6 +231,9 @@ def login(client: httpx.Client, base_url: str, owner: str, password: str) -> Non
|
||||
_raise_for_status_with_body(response)
|
||||
if not response.json().get("ok"):
|
||||
raise RuntimeError(f"login failed for {owner}")
|
||||
token = client.cookies.get("odysseus_session")
|
||||
if token:
|
||||
_COOKIE_CACHE[owner] = token
|
||||
|
||||
|
||||
def create_session(client: httpx.Client, args: argparse.Namespace, case: dict[str, Any]) -> str:
|
||||
@@ -94,7 +254,8 @@ def create_session(client: httpx.Client, args: argparse.Namespace, case: dict[st
|
||||
|
||||
|
||||
def stream_turn(
|
||||
client: httpx.Client, args: argparse.Namespace, session_id: str, prompt: str
|
||||
client: httpx.Client, args: argparse.Namespace, session_id: str, prompt: str,
|
||||
*, active_doc_id: str = "",
|
||||
) -> tuple[list[dict[str, Any]], str]:
|
||||
events: list[dict[str, Any]] = []
|
||||
text: list[str] = []
|
||||
@@ -106,10 +267,15 @@ def stream_turn(
|
||||
"selected_endpoint_id": args.endpoint_id,
|
||||
"selected_endpoint_url": args.endpoint,
|
||||
"selected_model": args.model,
|
||||
"thinking_mode": args.thinking_mode,
|
||||
"client_runtime_context": json.dumps(
|
||||
{"timezone": args.timezone, "tz_offset_min": args.tz_offset_min}, separators=(",", ":")
|
||||
),
|
||||
}
|
||||
if active_doc_id:
|
||||
form["active_doc_id"] = active_doc_id
|
||||
if getattr(args, "allow_web_search", False):
|
||||
form["allow_web_search"] = "true"
|
||||
with client.stream(
|
||||
"POST",
|
||||
args.base_url.rstrip("/") + "/api/chat_stream",
|
||||
@@ -154,6 +320,46 @@ def tool_outputs(events: list[dict[str, Any]]) -> str:
|
||||
return "\n".join(str(e.get("output") or "") for e in events if e.get("type") == "tool_output")
|
||||
|
||||
|
||||
def has_unrecovered_tool_failure(events: list[dict[str, Any]]) -> bool:
|
||||
"""Count a tool failure only when that tool never subsequently succeeds."""
|
||||
pending: set[str] = set()
|
||||
for event in events:
|
||||
if event.get("type") != "tool_output":
|
||||
continue
|
||||
name = normalized_tool(str(event.get("tool") or "unknown"))
|
||||
output = str(event.get("output") or "")
|
||||
failed = bool(
|
||||
event.get("error")
|
||||
or event.get("exit_code") not in (None, 0)
|
||||
or TOOL_FAILURE_RE.search(output)
|
||||
)
|
||||
if failed:
|
||||
pending.add(name)
|
||||
else:
|
||||
pending.discard(name)
|
||||
return bool(pending)
|
||||
|
||||
|
||||
def compact_evidence(events: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||
"""Keep routing and execution evidence without bloating the replay report."""
|
||||
retained = {
|
||||
"turn_contract", "tool_start", "tool_output", "tool_resolution_audit",
|
||||
"error", "parse_error", "metrics", "final_response",
|
||||
}
|
||||
rows = []
|
||||
for event in events:
|
||||
if event.get("type") not in retained:
|
||||
continue
|
||||
row = dict(event)
|
||||
for key in ("output", "text", "delta"):
|
||||
if isinstance(row.get(key), str) and len(row[key]) > 3000:
|
||||
row[key] = row[key][:3000] + "..."
|
||||
if row.get("type") == "turn_contract":
|
||||
row.pop("executable", None)
|
||||
rows.append(row)
|
||||
return rows
|
||||
|
||||
|
||||
def tool_actions(events: list[dict[str, Any]], tool_name: str) -> set[str]:
|
||||
actions: set[str] = set()
|
||||
for event in events:
|
||||
@@ -200,6 +406,11 @@ def score_turn(turn: dict[str, Any], events: list[dict[str, Any]], answer: str)
|
||||
failures: list[str] = []
|
||||
names = tool_names(events)
|
||||
expected = {normalized_tool(str(name)) for name in turn.get("expected_tools") or []}
|
||||
# Both document writers satisfy a requested active-draft mutation. Which
|
||||
# one is most efficient depends on how much of the draft the model changes;
|
||||
# exact-name imitation is not a functional correctness requirement.
|
||||
if expected & {"edit_document", "update_document"}:
|
||||
expected.update({"edit_document", "update_document"})
|
||||
if expected and not expected.intersection(names):
|
||||
failures.append(f"missing_acceptable_tool expected={sorted(expected)} got={names}")
|
||||
for tool_name, expected_actions in inferred_expected_actions(turn).items():
|
||||
@@ -213,8 +424,7 @@ def score_turn(turn: dict[str, Any], events: list[dict[str, Any]], answer: str)
|
||||
failures.append("stream_error")
|
||||
if BAD_ANSWER_RE.search(answer):
|
||||
failures.append("tool_unavailable_answer")
|
||||
output = tool_outputs(events)
|
||||
if TOOL_FAILURE_RE.search(output):
|
||||
if has_unrecovered_tool_failure(events):
|
||||
failures.append("tool_output_failure")
|
||||
if INTERNAL_NARRATION_RE.search(answer):
|
||||
failures.append("internal_narration_leaked")
|
||||
@@ -375,10 +585,11 @@ def marker_fields(value: Any, marker: str) -> Any:
|
||||
return value
|
||||
|
||||
|
||||
def apply_fixture_plan(case: dict[str, Any], owner: str, session_id: str, marker: str) -> None:
|
||||
def apply_fixture_plan(case: dict[str, Any], owner: str, session_id: str, marker: str) -> dict[str, str]:
|
||||
"""Create only owner-scoped local fixtures required before the first turn."""
|
||||
first_tools = set((case.get("turns") or [{}])[0].get("expected_tools") or [])
|
||||
db = SessionLocal()
|
||||
context: dict[str, str] = {}
|
||||
try:
|
||||
for fixture in case.get("fixture_plan") or []:
|
||||
if not isinstance(fixture, dict):
|
||||
@@ -407,6 +618,7 @@ def apply_fixture_plan(case: dict[str, Any], owner: str, session_id: str, marker
|
||||
summary="Expansion fixture",
|
||||
source="user",
|
||||
))
|
||||
context["active_doc_id"] = document_id
|
||||
elif fixture_type == "note":
|
||||
db.add(Note(
|
||||
id=str(uuid.uuid4()),
|
||||
@@ -426,6 +638,7 @@ def apply_fixture_plan(case: dict[str, Any], owner: str, session_id: str, marker
|
||||
raise
|
||||
finally:
|
||||
db.close()
|
||||
return context
|
||||
|
||||
|
||||
def delete_session(client: httpx.Client, base_url: str, session_id: str) -> None:
|
||||
@@ -482,10 +695,14 @@ def run_case(args: argparse.Namespace, case: dict[str, Any]) -> dict[str, Any]:
|
||||
snapshot.capture()
|
||||
login(client, args.base_url, owner, args.password)
|
||||
session_id = create_session(client, args, case)
|
||||
apply_fixture_plan(case, owner, session_id, marker)
|
||||
fixture_context = apply_fixture_plan(case, owner, session_id, marker)
|
||||
upstream_failed = False
|
||||
for turn in case["turns"]:
|
||||
prompt = str(turn["prompt"]).replace("{marker}", marker)
|
||||
events, answer = stream_turn(client, args, session_id, prompt)
|
||||
events, answer = stream_turn(
|
||||
client, args, session_id, prompt,
|
||||
active_doc_id=fixture_context.get("active_doc_id", ""),
|
||||
)
|
||||
turn_failures = score_turn(turn, events, answer)
|
||||
turns_out.append({
|
||||
"id": turn["id"],
|
||||
@@ -494,10 +711,14 @@ def run_case(args: argparse.Namespace, case: dict[str, Any]) -> dict[str, Any]:
|
||||
"observed_tools": tool_names(events),
|
||||
"answer": answer,
|
||||
"failures": turn_failures,
|
||||
"evidence": compact_evidence(events),
|
||||
"upstream_failed": upstream_failed,
|
||||
})
|
||||
failures.extend(f"{turn['id']}:{failure}" for failure in turn_failures)
|
||||
if turn_failures:
|
||||
break
|
||||
# Keep executing the full 3-4 turn trajectory. Later misses may be
|
||||
# causal fallout from an earlier failed create/read, so the judge
|
||||
# receives this marker and can separate root causes from cascades.
|
||||
upstream_failed = upstream_failed or bool(turn_failures)
|
||||
except Exception as exc:
|
||||
failures.append(f"exception:{exc!r}")
|
||||
finally:
|
||||
@@ -538,6 +759,20 @@ def main() -> None:
|
||||
parser.add_argument("--endpoint-id", default="f3904562")
|
||||
parser.add_argument("--endpoint", default="https://openrouter.ai/api/v1/chat/completions")
|
||||
parser.add_argument("--model", default="moonshotai/kimi-k3")
|
||||
parser.add_argument("--thinking-mode", choices=("on", "off"), default="off")
|
||||
parser.add_argument(
|
||||
"--allow-web-search",
|
||||
action="store_true",
|
||||
help="Enable Odysseus public web_search/web_fetch for this replay.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--fixture-environments",
|
||||
type=Path,
|
||||
help=(
|
||||
"Install owner-scoped synthetic inventory rows for replay. "
|
||||
"Use only against an isolated app with ODYSSEUS_EMAIL_FIXTURE=1."
|
||||
),
|
||||
)
|
||||
parser.add_argument("--turn-timeout", type=float, default=180)
|
||||
parser.add_argument("--case-timeout", type=float, default=600)
|
||||
parser.add_argument("--timezone", default="Asia/Tokyo")
|
||||
@@ -545,13 +780,34 @@ def main() -> None:
|
||||
parser.add_argument("--limit", type=int)
|
||||
parser.add_argument("--owner", action="append")
|
||||
parser.add_argument("--case-id", action="append")
|
||||
parser.add_argument(
|
||||
"--one-per-seed",
|
||||
action="store_true",
|
||||
help="Run the first validated environment variant for each source seed family",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.fixture_environments:
|
||||
if os.environ.get("ODYSSEUS_EMAIL_FIXTURE") != "1":
|
||||
parser.error("--fixture-environments requires ODYSSEUS_EMAIL_FIXTURE=1")
|
||||
installed = install_fixture_environments(args.fixture_environments)
|
||||
print(f"installed isolated fixture inventory: {installed}", flush=True)
|
||||
|
||||
cases = json.loads(args.cases.read_text(encoding="utf-8"))["cases"]
|
||||
if args.owner:
|
||||
cases = [case for case in cases if case["owner"] in set(args.owner)]
|
||||
if args.case_id:
|
||||
cases = [case for case in cases if case["case_id"] in set(args.case_id)]
|
||||
if args.one_per_seed:
|
||||
seen_seeds: set[str] = set()
|
||||
first_cases = []
|
||||
for case in cases:
|
||||
seed_id = str(case.get("seed_family_id") or "")
|
||||
if seed_id in seen_seeds:
|
||||
continue
|
||||
seen_seeds.add(seed_id)
|
||||
first_cases.append(case)
|
||||
cases = first_cases
|
||||
if args.limit:
|
||||
cases = cases[: args.limit]
|
||||
existing = {row["case_id"]: row for row in json.loads(args.out.read_text(encoding="utf-8")).get("results", [])} if args.out.exists() else {}
|
||||
|
||||
@@ -60,7 +60,7 @@ try {
|
||||
const events = parseSSE(await response.text());
|
||||
const contract = events.find(x => x.type === 'turn_contract') || {};
|
||||
const tools = events.filter(x => x.type === 'tool_start').map(x => bare(x.tool));
|
||||
const outputs = events.filter(x => x.type === 'tool_output').map(x => ({ tool: bare(x.tool), exit_code: x.exit_code ?? null, error: Boolean(x.error) }));
|
||||
const outputs = events.filter(x => x.type === 'tool_output').map(x => ({ tool: bare(x.tool), command: x.command || '', exit_code: x.exit_code ?? null, error: Boolean(x.error) }));
|
||||
const final = events.filter(x => x.type === 'final_response').map(x => x.content || '').join('') || events.filter(x => typeof x.delta === 'string').map(x => x.delta).join('');
|
||||
return { response, contract, tools, outputs, final };
|
||||
};
|
||||
@@ -68,11 +68,18 @@ try {
|
||||
const browserSession = await makeSession('deliberate');
|
||||
await openSession(browserSession);
|
||||
const opened = await send('Browse https://example.com and take a snapshot. Report the rendered page heading.');
|
||||
await page.locator('.private-browser-preview-img[src^="data:image/"]').last().waitFor({ state: 'visible', timeout: 10000 });
|
||||
const screenshotState = await page.locator('.private-browser-preview-img[src^="data:image/"]').last().evaluate(img => ({
|
||||
complete: img.complete,
|
||||
naturalWidth: img.naturalWidth,
|
||||
sourceLength: img.getAttribute('src')?.length || 0,
|
||||
}));
|
||||
const openChecks = {
|
||||
http_ok: opened.response.ok(), clean_route: opened.contract.selection_mode === 'clean_compact_v3_preview',
|
||||
offered_private_browser: (opened.contract.offered || []).some(x => bare(x) === 'private_browser'),
|
||||
browser_only: opened.tools.length >= 1 && opened.tools.every(x => x === 'private_browser'),
|
||||
tool_success: opened.outputs.some(x => x.tool === 'private_browser' && !x.error && (x.exit_code == null || x.exit_code === 0)),
|
||||
screenshot_visible: screenshotState.complete && screenshotState.naturalWidth > 0 && screenshotState.sourceLength > 100,
|
||||
grounded: /example domain/i.test(opened.final), no_reasoning_leak: noLeak(opened.final),
|
||||
};
|
||||
report.turns.push({ kind: 'domain-browse-snapshot-web-off', tools: opened.tools, outputs: opened.outputs, offered_private_browser: openChecks.offered_private_browser, checks: openChecks, status: Object.values(openChecks).every(Boolean) ? 'passed' : 'failed' }); save();
|
||||
@@ -97,6 +104,33 @@ try {
|
||||
};
|
||||
report.turns.push({ kind: 'typed-evidence-follow-up-web-off', tools: follow.tools, outputs: follow.outputs, offered_private_browser: followChecks.warm_private_browser, offered: (follow.contract.offered || []).map(bare), unavailable: follow.contract.unavailable || [], active_capabilities: follow.contract.active_capabilities || [], checks: followChecks, status: Object.values(followChecks).every(Boolean) ? 'passed' : 'failed' }); save();
|
||||
|
||||
const mapsSession = await makeSession('plain-open-preview');
|
||||
await openSession(mapsSession);
|
||||
const maps = await send('Browse Google Maps and find the closest coffee shop to Todoroki Station.');
|
||||
await page.locator('.private-browser-preview-img[src^="data:image/"]').last().waitFor({ state: 'visible', timeout: 10000 });
|
||||
const mapsScreenshot = await page.locator('.private-browser-preview-img[src^="data:image/"]').last().evaluate(img => ({
|
||||
complete: img.complete,
|
||||
naturalWidth: img.naturalWidth,
|
||||
sourceLength: img.getAttribute('src')?.length || 0,
|
||||
}));
|
||||
const mapsChecks = {
|
||||
http_ok: maps.response.ok(),
|
||||
browser_used: maps.tools.includes('private_browser'),
|
||||
screenshot_visible: mapsScreenshot.complete && mapsScreenshot.naturalWidth > 0 && mapsScreenshot.sourceLength > 100,
|
||||
no_reasoning_leak: noLeak(maps.final),
|
||||
};
|
||||
report.turns.push({ kind: 'plain-open-renders-screenshot', tools: maps.tools, checks: mapsChecks, status: Object.values(mapsChecks).every(Boolean) ? 'passed' : 'failed' }); save();
|
||||
|
||||
const menu = await send('Which one has a grilled cheese sandwich on the menu?');
|
||||
const menuCommands = menu.outputs.map(item => String(item.command || '').toLowerCase());
|
||||
const menuChecks = {
|
||||
http_ok: menu.response.ok(),
|
||||
web_followup_used: menu.tools.some(tool => ['private_browser', 'web_search', 'web_fetch'].includes(tool)),
|
||||
prior_subject_retained: menuCommands.some(command => /todoroki|coffee shop|peak by swell|yeti roastery|toe coffee/.test(command)),
|
||||
no_reasoning_leak: noLeak(menu.final),
|
||||
};
|
||||
report.turns.push({ kind: 'maps-result-property-followup', tools: menu.tools, commands: menuCommands, checks: menuChecks, status: Object.values(menuChecks).every(Boolean) ? 'passed' : 'failed' }); save();
|
||||
|
||||
const searchSession = await makeSession('ordinary-web');
|
||||
await openSession(searchSession);
|
||||
await page.locator('#web-toggle-btn').click();
|
||||
@@ -120,8 +154,8 @@ try {
|
||||
}
|
||||
if (browser) await browser.close();
|
||||
}
|
||||
report.status = report.turns.length === 3 && report.turns.every(x => x.status === 'passed') && report.cleanup.length === sessions.length && report.cleanup.every(x => x.removed) ? 'passed' : 'failed';
|
||||
report.summary = { passed: report.turns.filter(x => x.status === 'passed').length, total: 3 };
|
||||
report.status = report.turns.length === 5 && report.turns.every(x => x.status === 'passed') && report.cleanup.length === sessions.length && report.cleanup.every(x => x.removed) ? 'passed' : 'failed';
|
||||
report.summary = { passed: report.turns.filter(x => x.status === 'passed').length, total: 5 };
|
||||
save();
|
||||
console.log(JSON.stringify({ report: path.relative(root, reportPath), status: report.status, summary: report.summary }));
|
||||
if (report.status !== 'passed') process.exitCode = 1;
|
||||
|
||||
@@ -12,6 +12,8 @@ const endpointUrl = process.env.ENDPOINT_URL || (() => { throw new Error("ENDPOI
|
||||
const model = process.env.MODEL || 'odysseus-qwen3.5-tools-pre-heretic';
|
||||
const owner = process.env.OWNER || 'sft_alex_creator';
|
||||
const routingMode = process.env.ROUTING_MODE || 'baseline';
|
||||
const expectCleanRoute = process.env.EXPECT_CLEAN_ROUTE !== 'false';
|
||||
const expectRoutingMetadata = process.env.EXPECT_ROUTING_METADATA !== 'false';
|
||||
if (!['baseline', 'recent', 'all', 'default'].includes(routingMode)) throw Error('Invalid routing mode');
|
||||
const expectedMode = routingMode === 'default' ? 'recent_model_choice' : routingMode;
|
||||
const run = new Date().toISOString().replace(/[:.]/g, '-');
|
||||
@@ -287,6 +289,7 @@ try {
|
||||
const failure_category = ok ? null
|
||||
: /not found|no such|unknown (?:uid|id)|does not exist/i.test(detail) ? 'not_found'
|
||||
: /invalid|missing|required|argument|json|parse/i.test(detail) ? 'invalid_arguments'
|
||||
: /covered by|obscured by|blocking (?:dialog|overlay)|dismiss or interact with the covering/i.test(detail) ? 'interaction_blocked'
|
||||
: /connection|unavailable|timeout|refused/i.test(detail) ? 'backend_unavailable'
|
||||
: /permission|not offered|not permitted|denied/i.test(detail) ? 'permission_denied'
|
||||
: 'other';
|
||||
@@ -299,6 +302,30 @@ try {
|
||||
previousEmailUids = [...detail.matchAll(/^\s*UID:\s*(\S+)/gmi)].map(match => match[1]);
|
||||
}
|
||||
const final = events.filter(x => x.type === 'final_response').map(x => x.content || '').join('') || events.filter(x => typeof x.delta === 'string').map(x => x.delta).join('');
|
||||
const recoveredBrowserInteraction = events.some((event, eventIndex) => {
|
||||
if (event.type !== 'tool_output' || bare(event.tool) !== 'private_browser') return false;
|
||||
const detail = String(event.output || event.error_message || '');
|
||||
const blocked = /covered by|obscured by|blocking (?:dialog|overlay)|dismiss or interact with the covering/i.test(detail);
|
||||
if (!blocked) return false;
|
||||
return events.slice(eventIndex + 1).some(later => (
|
||||
later.type === 'tool_output'
|
||||
&& bare(later.tool) === 'private_browser'
|
||||
&& !later.error
|
||||
&& (later.exit_code == null || later.exit_code === 0)
|
||||
));
|
||||
});
|
||||
const prefetchedWebSources = events
|
||||
.filter(x => x.type === 'web_sources')
|
||||
.flatMap(x => Array.isArray(x.data) ? x.data : [])
|
||||
.filter(source => source?.acquisition === 'automatic_url_fetch');
|
||||
const prefetchedYoutubeSources = events
|
||||
.filter(x => x.type === 'web_sources')
|
||||
.flatMap(x => Array.isArray(x.data) ? x.data : [])
|
||||
.filter(source => source?.acquisition === 'automatic_youtube_context');
|
||||
const exactUrlPrefetched = expected.includes('web_fetch')
|
||||
&& prefetchedWebSources.length > 0;
|
||||
const youtubePrefetched = expected.includes('youtube_tool')
|
||||
&& prefetchedYoutubeSources.length > 0;
|
||||
if (spec.name === 'skills-cookbook-skills' && index === 0) {
|
||||
// Compare in memory only: never retain private skill names/content.
|
||||
previousSkillRows = events.filter(x => x.type === 'tool_output' && bare(x.tool) === 'manage_skills')
|
||||
@@ -340,24 +367,25 @@ try {
|
||||
nodes.slice(-8).map(node => String(node.className || node.tagName || '').slice(0, 120))
|
||||
) : [];
|
||||
const checks = {
|
||||
experiment_selected: contract.routing_experiment === expectedMode,
|
||||
http_ok: response.ok(), terminal: response.ok() && !events.some(x => x.type === 'invalid_sse'), clean_route: contract.selection_mode === 'clean_compact_v3_preview',
|
||||
capability: Boolean(spec.deniedTools?.[index]) || capabilityAvailable(contract, capability, expected)
|
||||
experiment_selected: !expectRoutingMetadata || contract.routing_experiment === expectedMode,
|
||||
http_ok: response.ok(), terminal: response.ok() && !events.some(x => x.type === 'invalid_sse'),
|
||||
clean_route: !expectCleanRoute || contract.selection_mode === 'clean_compact_v3_preview',
|
||||
capability: !expectRoutingMetadata || Boolean(spec.deniedTools?.[index]) || capabilityAvailable(contract, capability, expected)
|
||||
|| (spec.noToolTurns?.includes(index) && starts.length === 0)
|
||||
// An intentionally ambiguous continuation can use the retained
|
||||
// family without the classifier guessing a fresh active topic.
|
||||
|| (!expected.length && index > 0 && routingMode !== 'baseline'
|
||||
&& priorCapability === capability && priorFamilyTools.some(name => offered.includes(name))),
|
||||
expected_offered: !expected.length || expected.some(name => offered.includes(name)), expected_called: reusedSkillSummary || reusedSkillDetail || !expected.length || expected.some(name => starts.includes(name)),
|
||||
expected_offered: !expectRoutingMetadata || !expected.length || expected.some(name => offered.includes(name)), expected_called: reusedSkillSummary || reusedSkillDetail || exactUrlPrefetched || youtubePrefetched || !expected.length || expected.some(name => starts.includes(name)),
|
||||
expected_execution_outcome: spec.expectedExitCodes?.[index] !== undefined
|
||||
? events.filter(e => e.type === 'tool_output' && expected.includes(bare(e.tool))).length === 1
|
||||
&& events.some(e => e.type === 'tool_output' && expected.includes(bare(e.tool)) && e.exit_code === spec.expectedExitCodes[index])
|
||||
: reusedSkillSummary || reusedSkillDetail || !expected.length || outputs.some(x => expected.includes(x.tool) && x.ok),
|
||||
: reusedSkillSummary || reusedSkillDetail || exactUrlPrefetched || youtubePrefetched || !expected.length || outputs.some(x => expected.includes(x.tool) && x.ok),
|
||||
requested_execution_count: spec.name !== 'shell-failure-recovery' || starts.length === (index === 1 ? 0 : 1),
|
||||
failed_execution_provenance: !(spec.expectedExitCodes?.[index] > 0)
|
||||
|| events.some(e => e.type === 'tool_output' && expected.includes(bare(e.tool))
|
||||
&& e.exit_code === spec.expectedExitCodes[index] && e.execution_attempted === true && e.blocked === false),
|
||||
saved_failure_status: !(spec.expectedExitCodes?.[index] > 0)
|
||||
&& e.exit_code === spec.expectedExitCodes[index] && e.execution_attempted === true && e.blocked !== true),
|
||||
saved_failure_status: !expectCleanRoute || !(spec.expectedExitCodes?.[index] > 0)
|
||||
|| (metrics.data?.clean_v3_turn || metrics.clean_v3_turn || []).some(m => {
|
||||
if (m.role !== 'tool') return false;
|
||||
try { return JSON.parse(m.content).exit_code === spec.expectedExitCodes[index]; } catch { return false; }
|
||||
@@ -365,7 +393,7 @@ try {
|
||||
exact_skill_detail_reference: spec.name !== 'skills-cookbook-skills' || index !== 3
|
||||
|| reusedSkillDetail || calls.some(call => call.tool === 'manage_skills' && call.skill_action === 'view' && call.skill_matches_second),
|
||||
skill_detail_answer_evidence: spec.name !== 'skills-cookbook-skills' || index !== 3 || detailEvidence.covered,
|
||||
no_prior_family_leak: routingMode !== 'baseline' || index === 0 || priorCapability === capability
|
||||
no_prior_family_leak: !expectRoutingMetadata || routingMode !== 'baseline' || index === 0 || priorCapability === capability
|
||||
|| offered.every(name => !priorFamilyTools.includes(name) || expected.includes(name)),
|
||||
one_user_turn: afterUsers === beforeUsers + 1,
|
||||
visible_answer: final.trim().length > 0, no_reasoning_leak: noLeak(final),
|
||||
@@ -373,6 +401,9 @@ try {
|
||||
no_tool_errors: outputs.every(item => item.ok || (
|
||||
spec.deniedTools?.[index]?.includes(item.tool)
|
||||
&& item.failure_category === 'permission_denied' && starts.length === 0)
|
||||
|| (item.tool === 'private_browser'
|
||||
&& item.failure_category === 'interaction_blocked'
|
||||
&& recoveredBrowserInteraction)
|
||||
|| (spec.expectedExitCodes?.[index] > 0 && expected.includes(item.tool)
|
||||
&& events.some(e => e.type === 'tool_output' && bare(e.tool) === item.tool && e.exit_code === spec.expectedExitCodes[index]))),
|
||||
expected_answer_evidence: !spec.expectedAnswers
|
||||
@@ -398,6 +429,8 @@ try {
|
||||
metrics: Object.fromEntries(['input_tokens', 'output_tokens', 'injected_tokens',
|
||||
'time_to_first_token', 'response_time'].map(key => [key, metrics[key] ?? metrics.data?.[key] ?? null])),
|
||||
user_count_before: beforeUsers, user_count_after: afterUsers,
|
||||
prefetched_web_sources: prefetchedWebSources.length,
|
||||
prefetched_youtube_sources: prefetchedYoutubeSources.length,
|
||||
dom_classes_on_user_mismatch: domClasses,
|
||||
page_errors: pageErrors.splice(0),
|
||||
unavailable: contract.unavailable || [], checks, status: Object.values(checks).every(Boolean) ? 'passed' : 'failed' };
|
||||
|
||||
@@ -10,7 +10,10 @@ const endpointId = process.env.ENDPOINT_ID || '1d1022ef';
|
||||
const endpointUrl = process.env.ENDPOINT_URL || (() => { throw new Error("ENDPOINT_URL is required"); })();
|
||||
const model = process.env.MODEL || 'odysseus-qwen3.5-tools-pre-heretic';
|
||||
const owner = 'sft_alex_creator';
|
||||
const routingMode = 'recent_model_choice';
|
||||
const routingMode = process.env.ROUTING_MODE || 'recent';
|
||||
const expectedRoutingMode = routingMode === 'recent' ? 'recent_model_choice' : routingMode;
|
||||
const expectCleanRoute = process.env.EXPECT_CLEAN_ROUTE !== 'false';
|
||||
const expectExactRouting = process.env.EXPECT_EXACT_ROUTING !== 'false';
|
||||
const run = new Date().toISOString().replace(/[:.]/g, '-');
|
||||
const reportPath = path.resolve(process.env.REPORT_PATH || path.join(root, `reports/mobile-active-editor-followups-${run}.json`));
|
||||
if (!reportPath.startsWith(path.join(root, 'reports') + path.sep) || fs.existsSync(reportPath)) throw Error('Report path must be new and under reports/');
|
||||
@@ -115,8 +118,9 @@ try {
|
||||
const fetched = await context.request.get(`${base}/api/document/${encodeURIComponent(docId)}`);
|
||||
const current = fetched.ok() ? String((await fetched.json()).current_content || '') : '';
|
||||
const checks = {
|
||||
http_ok: response.ok(), clean_route: contract.selection_mode === 'clean_compact_v3_preview',
|
||||
exact_runtime: contract.routing_experiment === routingMode,
|
||||
http_ok: response.ok(),
|
||||
clean_route: !expectCleanRoute || contract.selection_mode === 'clean_compact_v3_preview',
|
||||
exact_runtime: !expectExactRouting || contract.routing_experiment === expectedRoutingMode,
|
||||
request_has_fixture_editor: response.request().postData()?.includes(docId) || false,
|
||||
documents_capability: (contract.active_capabilities || contract.capabilities || []).includes('documents'),
|
||||
same_open_editor: await page.evaluate(id => window.documentModule?.getChatDocumentId?.() === id, docId),
|
||||
|
||||
@@ -8,6 +8,7 @@ import {AMBIGUOUS_CASES,expectedNoteTitles,compareNoteState} from './note_test_o
|
||||
|
||||
const root = path.resolve(new URL('..', import.meta.url).pathname);
|
||||
const base = process.env.BASE_URL || 'http://127.0.0.1:7011';
|
||||
const model = process.env.MODEL || 'odysseus-qwen3.5-tools-pre-heretic';
|
||||
const routingMode = process.env.ROUTING_MODE || 'baseline';
|
||||
const followupCase = process.env.FOLLOWUP_CASE || 'original';
|
||||
const plainTitles = process.env.TITLE_STYLE === 'plain';
|
||||
@@ -73,7 +74,7 @@ try {
|
||||
} });
|
||||
await context.addCookies([{ name: 'odysseus_session', value: token, url: base }]);
|
||||
const created = await context.request.post(`${base}/api/session`, { multipart: {
|
||||
name: `[multi-note-followup] ${marker}`, model: 'odysseus-qwen3.5-tools-pre-heretic',
|
||||
name: `[multi-note-followup] ${marker}`, model,
|
||||
endpoint_id: process.env.ENDPOINT_ID || '1d1022ef',
|
||||
endpoint_url: process.env.ENDPOINT_URL || (() => { throw new Error("ENDPOINT_URL is required"); })(),
|
||||
skip_validation: 'true', rag: 'false',
|
||||
|
||||
@@ -10,6 +10,8 @@ const endpointId = process.env.ENDPOINT_ID || '1d1022ef';
|
||||
const endpointUrl = process.env.ENDPOINT_URL || (() => { throw new Error("ENDPOINT_URL is required"); })();
|
||||
const model = process.env.MODEL || 'odysseus-qwen3.5-tools-pre-heretic';
|
||||
const owner = 'sft_alex_creator';
|
||||
const expectCleanRoute = process.env.EXPECT_CLEAN_ROUTE !== 'false';
|
||||
const expectNativeContractMetadata = process.env.EXPECT_NATIVE_CONTRACT_METADATA !== 'false';
|
||||
const seconds = value => {
|
||||
if (typeof value === 'number') return value;
|
||||
const text = String(value ?? '').trim();
|
||||
@@ -112,9 +114,10 @@ try {
|
||||
const expected = starts.filter(event => event.tool === spec.tool);
|
||||
const args = parseArgs(expected[0]);
|
||||
const checks = {
|
||||
http_ok: response.ok(), clean_route: contract.selection_mode === 'clean_compact_v3_preview',
|
||||
native_workspace: contract.native_workspace === true,
|
||||
expected_offered: (contract.offered || []).includes(spec.tool),
|
||||
http_ok: response.ok(),
|
||||
clean_route: !expectCleanRoute || contract.selection_mode === 'clean_compact_v3_preview',
|
||||
native_workspace: !expectNativeContractMetadata || contract.native_workspace === true,
|
||||
expected_offered: !expectNativeContractMetadata || (contract.offered || []).includes(spec.tool),
|
||||
exactly_one_expected_call: starts.length === 1 && expected.length === 1,
|
||||
argument_contract: expected.length === 1 && spec.validate(index, args),
|
||||
exactly_one_successful_output: successfulOutputs.length === 1,
|
||||
|
||||
@@ -106,6 +106,26 @@ try {
|
||||
const removed = await send(`Delete the second ${family === 'tasks' ? 'task' : 'event'} from that list.`);
|
||||
const deleteOutputs = removed.events.filter(event => event.type === 'tool_output');
|
||||
const deleteStarts = removed.events.filter(event => event.type === 'tool_start');
|
||||
const mutationActions = new Set(['delete', 'delete_event', 'remove', 'cancel']);
|
||||
const pendingStarts = new Map();
|
||||
const successfulDeleteOutputs = [];
|
||||
for (const event of removed.events) {
|
||||
if (event.type === 'tool_start') {
|
||||
const queue = pendingStarts.get(event.tool) || [];
|
||||
queue.push(event);
|
||||
pendingStarts.set(event.tool, queue);
|
||||
continue;
|
||||
}
|
||||
if (event.type !== 'tool_output') continue;
|
||||
const start = (pendingStarts.get(event.tool) || []).shift();
|
||||
if (!start || event.error || (event.exit_code != null && event.exit_code !== 0)) continue;
|
||||
try {
|
||||
const command = typeof start.command === 'string' ? JSON.parse(start.command) : start.command;
|
||||
if (mutationActions.has(String(command?.action || '').toLowerCase())) {
|
||||
successfulDeleteOutputs.push(event);
|
||||
}
|
||||
} catch (_) {}
|
||||
}
|
||||
const remaining = [];
|
||||
for (const id of seeded) {
|
||||
const response = await context.request.get(`${base}${family === 'tasks' ? '/api/tasks/' : '/api/calendar/events/'}${encodeURIComponent(id)}`);
|
||||
@@ -114,7 +134,7 @@ try {
|
||||
item.turns.push({ name: 'delete-second', target_id: target, tools: deleteStarts.map(event => event.tool), tool_events: removed.events.filter(event => ['tool_start', 'tool_output'].includes(event.type)).map(event => ({ type: event.type, tool: event.tool, command: event.command, output: event.output, exit_code: event.exit_code, error: event.error })), checks: {
|
||||
target_resolved: expectedSet.has(target), http_ok: removed.response.ok(),
|
||||
correct_capability: (removed.contract.active_capabilities || []).includes(family),
|
||||
one_successful_delete: deleteOutputs.filter(event => !event.error && (event.exit_code == null || event.exit_code === 0)).length === 1,
|
||||
one_successful_delete: successfulDeleteOutputs.length === 1,
|
||||
second_item_deleted: !!target && !remaining.includes(target),
|
||||
other_item_preserved: seeded.filter(id => id !== target).every(id => remaining.includes(id)),
|
||||
no_stream_error: !removed.events.some(event => ['error', 'invalid_sse'].includes(event.type)),
|
||||
|
||||
Reference in New Issue
Block a user