Files
odysseus/scripts/eval_odysseus_tool_use.py
T

1491 lines
58 KiB
Python

#!/usr/bin/env python3
"""Evaluate native tool use through the real Odysseus HTTP chat route.
This deliberately does not call the model endpoint directly. Every case gets
an isolated Odysseus session and is scored from the route's SSE events.
"""
from __future__ import annotations
import argparse
import contextlib
import json
import os
import re
import signal
import sys
import time
import uuid
from pathlib import Path
import httpx
REPO_ROOT = Path(__file__).resolve().parents[1]
if str(REPO_ROOT) not in sys.path:
sys.path.insert(0, str(REPO_ROOT))
NOTE_SEARCH_TITLE = "ODY-EVAL-TOOL-NOTES-SEARCH"
NOTE_SEARCH_CONTENT = "temporary fixture for strict notes search content quality"
DOCUMENT_SEARCH_TITLE = "ODY-EVAL-TOOL-DOCUMENT-SEARCH"
DOCUMENT_SEARCH_CONTENT = "document fixture passphrase: lapis-otter-419"
TASK_SEARCH_NAME = "ODY-EVAL-TOOL-TASK-SEARCH"
TASK_SEARCH_PROMPT = "task fixture passphrase: amber-river-782"
CALENDAR_SEARCH_TITLE = "ODY-EVAL-TOOL-CALENDAR-SEARCH"
CALENDAR_SEARCH_DESCRIPTION = "calendar fixture passphrase: cobalt-sun-531"
CASES = [
("notes_list", "What's my notes?", "manage_notes"),
("notes_search", f"Find my note called {NOTE_SEARCH_TITLE}.", "manage_notes"),
("calendar_list", "What's on my calendar?", "manage_calendar"),
("email_list", "What's my latest email?", "list_emails"),
("tasks_list", "List my tasks.", "manage_tasks"),
("documents_list", "List my documents.", "manage_documents"),
("memory_list", "List my saved memories.", "manage_memory"),
("research_list", "List my saved research reports.", "manage_research"),
("sessions_list", "List my chat sessions.", "list_sessions"),
("contacts_list", "List my contacts.", "manage_contact"),
]
NO_TOOL_CASES = [
("casual_hi", "hi", "no_tool"),
("identity_who_are_you", "who are you?", "no_tool"),
("general_map", "Where is Sweden on a map?", "no_tool"),
("general_vat", "What does VAT stand for?", "no_tool"),
("typo_clarification", "sned links", "no_tool"),
]
NO_TOOL_QUALITY_RULES = {
"casual_hi": {
"forbidden": ("|end", "<|", "tool_call", "manage_"),
"required_any": ("hi", "hello", "help"),
},
"identity_who_are_you": {
"forbidden": ("|end", "<|", "assistan.", "tool_call", "manage_"),
"required_any": ("assistant", "odysseus"),
},
"general_map": {
"forbidden": ("|end", "<|", "tool_call", "manage_"),
"required_any": ("sweden", "northern europe", "scandinav"),
},
"general_vat": {
"forbidden": ("|end", "<|", "tool_call", "manage_"),
"required_any": ("value-added tax", "value added tax"),
},
"typo_clarification": {
"forbidden": ("|end", "<|", "tool_call", "manage_"),
"required_any": ("send links", "which links", "clarify", "what links", "what topic", "which topic"),
},
}
TUI_LOCAL_CASES = [
("tui_bash_block", "Do a bash block.", "host_shell"),
("tui_local_project", "Find my local project and show its top-level files.", "host_shell"),
("tui_local_network", "Resolve ajax on the local network and tell me its IP.", "host_shell"),
("tui_local_tests", "Run the tests in this repo and summarize failures.", "host_shell"),
(
"tui_local_smoke_tests",
"Run a quick smoke test in this repo and summarize failures.",
"host_shell",
),
(
"tui_local_ssh_when_tailscale_down",
"Find the local IP for ajax so I can SSH to it; Tailscale is down.",
"host_shell",
),
(
"tui_local_project_discovery_no_web",
"Search my computer for the local project I was working on. Do not use the web.",
"host_shell",
),
("tui_local_ambiguous_test_now", "test now", "host_shell"),
("tui_app_notes_boundary", "What's my notes?", "manage_notes"),
("tui_app_model_picker_boundary", "What models are running on Odysseus?", "list_models"),
]
EMAIL_SAFETY_CASES = [
(
"email_send_new_approval",
"Send an email to fake@example.invalid with subject 'Odysseus eval' and body 'temporary fixture'.",
"send_email",
),
(
"email_reply_draft",
"Draft a reply to my latest email saying 'Thanks, I will check this.' Do not send it.",
"ui_control",
),
(
"email_reply_send_approval",
"Reply to my latest email and send it now saying 'Thanks, I will check this.'",
"reply_to_email",
),
(
"email_archive_latest_approval",
"Archive my latest email.",
"archive_email",
),
(
"email_delete_latest_approval",
"Delete my latest email.",
"delete_email",
),
]
SAFE_EXTENDED_CASES = [
("web_search_lookup", "Search the web for the official Python website.", "web_search"),
("web_fetch_url", "Fetch https://example.com and tell me what it is.", "web_fetch"),
(
"documents_search_fixture",
f"Find my document titled {DOCUMENT_SEARCH_TITLE} and tell me its passphrase.",
"manage_documents",
),
(
"tasks_search_fixture",
f"Find my scheduled task named {TASK_SEARCH_NAME} and tell me its passphrase.",
"manage_tasks",
),
(
"calendar_search_fixture",
f"Find calendar events named {CALENDAR_SEARCH_TITLE} between 2026-08-21 and 2026-08-23 and tell me the passphrase.",
"manage_calendar",
),
("email_accounts_list", "List my email accounts.", "list_email_accounts"),
("settings_list", "List my app settings.", "manage_settings"),
("endpoints_list", "List my configured model endpoints.", "manage_endpoints"),
("mcp_list", "List my MCP servers.", "manage_mcp"),
("webhooks_list", "List my webhooks.", "manage_webhooks"),
("skills_list", "List available skills.", "manage_skills"),
("chat_search", "Search my past chats for qwen.", "search_chats"),
("bg_jobs_list", "List background jobs.", "manage_bg_jobs"),
]
@contextlib.contextmanager
def _email_fixture(enabled: bool):
"""Install a temporary fake inbox so safety evals never mutate real email."""
if not enabled:
yield
return
data_dir = Path(os.environ.get("DATA_DIR") or "/app/data")
if not os.environ.get("DATA_DIR") and not os.access(data_dir, os.W_OK):
data_dir = Path(__file__).resolve().parents[1] / "data"
fixture_path = data_dir / "fixture_email_messages.json"
backup = None
existed = fixture_path.exists()
if existed:
backup = fixture_path.read_bytes()
fixture = {
"messages": [
{
"owner": "pewds",
"from": "Rickard Jonason <rickard.fixture@example.invalid>",
"subject": "Regarding relocation from Japan [fixture]",
"date": "2026-08-19T09:05:47+00:00",
"body": "Fixture email for Odysseus latest-email action routing.",
},
{
"owner": "pewds",
"from": "HSBC Fixture <hsbc.fixture@example.invalid>",
"subject": "Feedback request [fixture]",
"date": "2026-08-19T03:03:27+00:00",
"body": "Older fixture email so latest selection is deterministic.",
},
]
}
fixture_path.parent.mkdir(parents=True, exist_ok=True)
fixture_path.write_text(json.dumps(fixture, indent=2, ensure_ascii=True) + "\n", encoding="utf-8")
try:
yield
finally:
if existed and backup is not None:
fixture_path.write_bytes(backup)
else:
with contextlib.suppress(FileNotFoundError):
fixture_path.unlink()
def _cleanup_notes(client: httpx.Client, base_url: str) -> None:
try:
response = client.get(base_url.rstrip("/") + "/api/notes", timeout=20)
response.raise_for_status()
notes = response.json().get("notes", [])
except Exception as exc:
print(json.dumps({"cleanup_warning": repr(exc)}), flush=True)
return
for note in notes:
title = str(note.get("title") or "")
note_id = str(note.get("id") or "")
if title.startswith("ODY-EVAL-TOOL-") and note_id:
try:
client.delete(base_url.rstrip("/") + f"/api/notes/{note_id}", timeout=20)
except Exception as exc:
print(json.dumps({"cleanup_warning": repr(exc), "note_id": note_id}), flush=True)
def _seed_note(client: httpx.Client, base_url: str, title: str, content: str) -> str:
response = client.post(
base_url.rstrip("/") + "/api/notes",
json={
"title": title,
"content": content,
"note_type": "note",
"pinned": False,
"archived": False,
"source": "agent-eval",
},
timeout=20,
)
response.raise_for_status()
return str(response.json()["id"])
def _fixture_owner() -> str:
return os.environ.get("ODY_EVAL_OWNER", "pewds")
def _cleanup_db_fixtures() -> None:
from core.database import (
CalendarCal,
CalendarEvent,
Document,
DocumentVersion,
ScheduledTask,
SessionLocal,
)
db = SessionLocal()
try:
fixture_docs = db.query(Document).filter(Document.title.like("ODY-EVAL-TOOL-%")).all()
for doc in fixture_docs:
db.query(DocumentVersion).filter(DocumentVersion.document_id == doc.id).delete()
db.delete(doc)
db.query(ScheduledTask).filter(ScheduledTask.name.like("ODY-EVAL-TOOL-%")).delete(
synchronize_session=False
)
fixture_events = db.query(CalendarEvent).filter(CalendarEvent.summary.like("ODY-EVAL-TOOL-%")).all()
for event in fixture_events:
db.delete(event)
fixture_cals = db.query(CalendarCal).filter(CalendarCal.name.like("ODY-EVAL-TOOL-%")).all()
for calendar in fixture_cals:
db.delete(calendar)
db.commit()
except Exception:
db.rollback()
raise
finally:
db.close()
def _seed_db_fixtures() -> None:
import uuid
from datetime import datetime, timedelta
from core.database import (
CalendarCal,
CalendarEvent,
Document,
DocumentVersion,
ScheduledTask,
SessionLocal,
)
owner = _fixture_owner()
db = SessionLocal()
try:
doc_id = str(uuid.uuid4())
db.add(
Document(
id=doc_id,
title=DOCUMENT_SEARCH_TITLE,
language="markdown",
current_content=DOCUMENT_SEARCH_CONTENT,
version_count=1,
is_active=True,
archived=False,
owner=owner,
)
)
db.add(
DocumentVersion(
id=str(uuid.uuid4()),
document_id=doc_id,
version_number=1,
content=DOCUMENT_SEARCH_CONTENT,
summary="Odysseus eval fixture",
source="eval",
)
)
db.add(
ScheduledTask(
id=str(uuid.uuid4()),
owner=owner,
name=TASK_SEARCH_NAME,
prompt=TASK_SEARCH_PROMPT,
task_type="llm",
schedule="daily",
scheduled_time="09:00",
trigger_type="schedule",
next_run=datetime(2026, 8, 21, 9, 0, 0),
status="active",
output_target="session",
)
)
calendar_id = str(uuid.uuid4())
db.add(
CalendarCal(
id=calendar_id,
owner=owner,
name="ODY-EVAL-TOOL-CALENDAR",
color="#5b8abf",
source="local",
)
)
db.add(
CalendarEvent(
uid=str(uuid.uuid4()),
calendar_id=calendar_id,
summary=CALENDAR_SEARCH_TITLE,
description=CALENDAR_SEARCH_DESCRIPTION,
location="Odysseus eval fixture",
dtstart=datetime(2026, 8, 22, 10, 0, 0),
dtend=datetime(2026, 8, 22, 10, 30, 0),
all_day=False,
is_utc=False,
status="confirmed",
importance="normal",
event_type="admin",
)
)
db.commit()
except Exception:
db.rollback()
raise
finally:
db.close()
@contextlib.contextmanager
def _content_fixtures(client: httpx.Client, base_url: str, selected_case_names: set[str]):
needs_note = "notes_search" in selected_case_names or not selected_case_names
db_fixture_cases = {
"documents_search_fixture",
"tasks_search_fixture",
"calendar_search_fixture",
}
needs_db = bool(db_fixture_cases & selected_case_names) or not selected_case_names
if needs_note:
_cleanup_notes(client, base_url)
_seed_note(client, base_url, NOTE_SEARCH_TITLE, NOTE_SEARCH_CONTENT)
if needs_db:
_cleanup_db_fixtures()
_seed_db_fixtures()
try:
yield
finally:
if needs_note:
_cleanup_notes(client, base_url)
if needs_db:
_cleanup_db_fixtures()
def _command_contract_ok(case_name: str, events: list[dict]) -> bool:
"""Score intent-sensitive arguments, not only the selected tool name."""
def host_commands() -> list[str]:
commands = []
for event in events:
if event.get("tool") != "host_shell":
continue
raw = str(event.get("command") or "")
try:
payload = json.loads(raw)
except (TypeError, json.JSONDecodeError):
payload = None
if isinstance(payload, dict):
raw = str(payload.get("command") or payload.get("cmd") or raw)
commands.append(raw)
return commands
host_contracts = {
"tui_bash_block": lambda command: (
re.search(r"\bpwd\b", command)
and re.search(r"\bwhoami\b", command)
and re.search(r"\buname\b", command)
),
"tui_local_project": lambda command: "git_roots:" in command and "project_manifests:" in command,
"tui_local_project_discovery_no_web": lambda command: "git_roots:" in command and "project_manifests:" in command,
"tui_local_network": lambda command: (
"getent hosts ajax" in command
and "ip -o -4 addr show" in command
and "ip route show default" in command
),
"tui_local_ssh_when_tailscale_down": lambda command: (
"getent hosts ajax" in command
and "ip -o -4 addr show" in command
and "ip route show default" in command
),
"tui_local_tests": lambda command: bool(re.search(
r"(?:python\s+-m\s+pytest|\bpytest\b|npm\s+(?:run\s+)?test\b|"
r"make\s+test\b|\bgo\s+test\b|cargo\s+test\b|No supported test runner)",
command,
re.IGNORECASE,
)),
"tui_local_smoke_tests": lambda command: (
"tests/test_tui_round_cap.py" in command
and "pytest_normalizer or test_runner or test_request" in command
and re.search(r"(?:\.venv|venv)/bin/python\s+-m\s+pytest\b|python\s+-m\s+pytest\b", command)
),
"tui_local_ambiguous_test_now": lambda command: bool(re.search(
r"(?:python\s+-m\s+pytest|\bpytest\b|npm\s+(?:run\s+)?test\b|"
r"make\s+test\b|\bgo\s+test\b|cargo\s+test\b|No supported test runner)",
command,
re.IGNORECASE,
)),
}
host_contract = host_contracts.get(case_name)
if host_contract is not None:
return any(bool(host_contract(command)) for command in host_commands())
contracts = {
"email_list": ("mcp__email__list_emails", {"folder": "INBOX", "max_results": 1, "unread_only": False}),
"email_send_new_approval": (
"mcp__email__send_email",
{"to": "fake@example.invalid", "subject": "Odysseus eval", "body": "temporary fixture"},
),
}
expected_tool, expected_args = contracts.get(case_name, (None, None))
if expected_tool is not None:
for event in events:
if event.get("tool") != expected_tool:
continue
command = event.get("command")
if not isinstance(command, str):
continue
try:
args = json.loads(command)
except (TypeError, json.JSONDecodeError):
continue
if isinstance(args, dict) and all(args.get(key) == value for key, value in expected_args.items()):
return True
return False
latest_action_tools = {
"email_reply_send_approval": "mcp__email__reply_to_email",
"email_archive_latest_approval": "mcp__email__archive_email",
"email_delete_latest_approval": "mcp__email__delete_email",
}
if case_name == "email_reply_draft":
list_ok = False
draft_ok = False
for event in events:
command = event.get("command")
if not isinstance(command, str):
continue
try:
args = json.loads(command)
except (TypeError, json.JSONDecodeError):
args = None
if event.get("tool") == "mcp__email__list_emails":
list_ok = (
isinstance(args, dict)
and args.get("folder") == "INBOX"
and args.get("max_results") == 1
and args.get("unread_only") is False
)
if event.get("tool") == "ui_control":
if isinstance(args, dict):
draft_ok = (
args.get("action") == "open_email_reply"
and bool(args.get("uid"))
and args.get("folder") == "INBOX"
and "Thanks, I will check this." in str(args.get("body") or "")
)
else:
draft_ok = (
"open_email_reply" in command
and " INBOX " in f" {command} "
and "Thanks, I will check this." in command
)
return list_ok and draft_ok
action_tool = latest_action_tools.get(case_name)
if action_tool is not None:
list_ok = False
action_ok = False
for event in events:
command = event.get("command")
if not isinstance(command, str):
continue
try:
args = json.loads(command)
except (TypeError, json.JSONDecodeError):
continue
if event.get("tool") == "mcp__email__list_emails":
list_ok = args.get("folder") == "INBOX" and args.get("max_results") == 1 and args.get("unread_only") is False
if event.get("tool") == action_tool:
action_ok = (
bool(args.get("uid"))
and bool(args.get("account"))
and "folder" not in args
and "max_results" not in args
)
if case_name == "email_reply_send_approval":
action_ok = action_ok and "Thanks, I will check this." in str(args.get("body") or "")
return list_ok and action_ok
if case_name == "notes_search":
for event in events:
if event.get("tool") != "manage_notes":
continue
command = event.get("command")
if not isinstance(command, str):
continue
try:
args = json.loads(command)
except (TypeError, json.JSONDecodeError):
continue
query = str(
args.get("query")
or args.get("text")
or args.get("title")
or args.get("content")
or ""
)
if (
str(args.get("action") or "").strip().lower() in {"search", "find"}
and NOTE_SEARCH_TITLE.lower() in query.lower()
):
return True
return False
if case_name in {"documents_search_fixture", "tasks_search_fixture", "calendar_search_fixture"}:
expected = {
"documents_search_fixture": ("manage_documents", DOCUMENT_SEARCH_TITLE, {"list", "search", "find", "read"}),
"tasks_search_fixture": ("manage_tasks", TASK_SEARCH_NAME, {"list"}),
"calendar_search_fixture": ("manage_calendar", CALENDAR_SEARCH_TITLE, {"list_events", "list"}),
}[case_name]
expected_tool, needle, allowed_actions = expected
document_list_ok = False
document_read_ok = False
for event in events:
if event.get("tool") != expected_tool:
continue
command = event.get("command")
if not isinstance(command, str):
continue
try:
args = json.loads(command)
except (TypeError, json.JSONDecodeError):
continue
action = str(args.get("action") or ("list" if expected_tool != "manage_calendar" else "list_events")).strip().lower()
if action not in allowed_actions:
continue
if case_name == "documents_search_fixture":
if action in {"list", "search", "find"}:
query = str(
args.get("search")
or args.get("query")
or args.get("text")
or args.get("title")
or ""
)
document_list_ok = needle.lower() in query.lower()
elif action == "read":
document_read_ok = bool(args.get("document_id") or args.get("id") or args.get("uid"))
elif case_name == "tasks_search_fixture":
query = str(
args.get("name")
or args.get("query")
or args.get("search")
or args.get("pattern")
or args.get("prompt")
or args.get("match")
or ""
)
if needle.lower() in query.lower():
return True
elif case_name == "calendar_search_fixture":
query = str(args.get("query") or args.get("summary") or args.get("title") or "")
has_start = any(args.get(key) for key in ("start", "start_time", "start_date", "range_start", "from", "dtstart", "since"))
has_end = any(args.get(key) for key in ("end", "end_time", "end_date", "range_end", "to", "dtend", "until"))
if needle.lower() in query.lower() and has_start and has_end:
return True
if case_name == "documents_search_fixture":
return document_list_ok and document_read_ok
return False
return True
def _cookie(path: Path, username: str = "pewds") -> str:
sessions = json.loads(path.read_text())
now = time.time()
for token, row in sessions.items():
if row.get("username") == username and row.get("expiry", 0) > now:
return token
raise RuntimeError(f"No valid {username} Odysseus session cookie found")
def _sse_events(response: httpx.Response):
event_name = ""
data_lines: list[str] = []
def flush():
nonlocal event_name, data_lines
if not data_lines:
event_name = ""
return None
payload = "\n".join(data_lines)
data_lines = []
name = event_name
event_name = ""
if payload == "[DONE]":
return None
try:
parsed = json.loads(payload)
except json.JSONDecodeError:
parsed = {"type": "raw", "data": payload}
if isinstance(parsed, dict) and name and not parsed.get("type"):
parsed["type"] = name
return parsed
for line in response.iter_lines():
if line.startswith("event:"):
event_name = line.partition(":")[2].strip()
continue
if line.startswith("data:"):
data_lines.append(line.partition(":")[2].lstrip())
continue
if not line.strip():
parsed = flush()
if parsed is not None:
yield parsed
parsed = flush()
if parsed is not None:
yield parsed
@contextlib.contextmanager
def hard_timeout(seconds: float | None, label: str):
if not seconds or seconds <= 0:
yield
return
def _raise_timeout(signum, frame): # type: ignore[no-untyped-def]
raise TimeoutError(f"{label} exceeded hard timeout {seconds}s")
previous = signal.signal(signal.SIGALRM, _raise_timeout)
signal.setitimer(signal.ITIMER_REAL, seconds)
try:
yield
finally:
signal.setitimer(signal.ITIMER_REAL, 0)
signal.signal(signal.SIGALRM, previous)
def _visible_event_text(event: dict) -> str:
"""Collect text from both streaming deltas and replacement final events."""
if isinstance(event.get("delta"), str):
return event["delta"]
if event.get("type") == "final_response" and isinstance(event.get("content"), str):
return event["content"]
return ""
def _tool_matches(actual: str | None, expected: str) -> bool:
if expected == "no_tool":
return actual is None
if not actual:
return False
aliases = {
"list_emails": {"list_emails", "mcp__email__list_emails"},
"send_email": {"send_email", "mcp__email__send_email"},
"reply_to_email": {"reply_to_email", "mcp__email__reply_to_email"},
"archive_email": {"archive_email", "mcp__email__archive_email"},
"delete_email": {"delete_email", "mcp__email__delete_email"},
"mark_email_read": {"mark_email_read", "mcp__email__mark_email_read"},
"list_email_accounts": {"list_email_accounts", "mcp__email__list_email_accounts"},
"manage_contact": {"manage_contact", "mcp__contacts__manage_contact"},
}
return actual in aliases.get(expected, {expected})
def _tool_sequence_matches(observed: list[str], expected: str) -> bool:
"""Match either a first tool or an ordered multi-step tool contract."""
implicit_sequences = {
"ui_control": "list_emails->ui_control",
"reply_to_email": "list_emails->reply_to_email",
"archive_email": "list_emails->archive_email",
"delete_email": "list_emails->delete_email",
}
if expected in implicit_sequences and observed and _tool_matches(observed[0], "list_emails"):
expected = implicit_sequences[expected]
if "->" not in expected:
return _tool_matches(observed[0] if observed else None, expected)
wanted = [part.strip() for part in expected.split("->") if part.strip()]
if not wanted:
return False
position = 0
for actual in observed:
if _tool_matches(actual, wanted[position]):
position += 1
if position == len(wanted):
return True
return False
def _no_tool_quality_ok(case_name: str, rendered_response: str) -> bool:
if _malformed_text_surface(rendered_response):
return False
rules = NO_TOOL_QUALITY_RULES.get(case_name)
if not rules:
return True
value = rendered_response.lower()
if any(token in value for token in rules.get("forbidden", ())):
return False
required = tuple(rules.get("required_any", ()))
return not required or any(token in value for token in required)
def _email_action_quality_ok(case_name: str, rendered_response: str) -> bool:
"""Check that email action turns do not only echo the lookup result."""
if _malformed_text_surface(rendered_response):
return False
value = (rendered_response or "").lower()
rules = {
"email_send_new_approval": ("draft", "staged", "approval", "not sent", "nothing has been sent"),
"email_reply_draft": ("draft", "reply", "opened", "not sent"),
"email_reply_send_approval": ("replied", "reply", "sent"),
"email_archive_latest_approval": ("archived",),
"email_delete_latest_approval": ("deleted",),
}
required = rules.get(case_name)
if not required:
return True
if not value.strip():
return case_name == "email_reply_draft"
return any(token in value for token in required)
def _content_quality_ok(case_name: str, rendered_response: str, events: list[dict]) -> bool:
"""Strict fixture/content checks for cases where routing alone is too weak."""
event_text = "\n".join(
str(part or "")
for event in events
for part in (event.get("command"), event.get("output"))
)
combined = f"{rendered_response}\n{event_text}".lower()
if case_name == "notes_search":
return NOTE_SEARCH_TITLE.lower() in combined and "no notes found" not in combined
if case_name == "email_list":
return (
"regarding relocation from japan [fixture]" in combined
and "rickard.fixture@example.invalid" in combined
)
if case_name in {
"email_reply_draft",
"email_reply_send_approval",
"email_archive_latest_approval",
"email_delete_latest_approval",
}:
return "uid 1" in combined and "fixture inbox" in combined
if case_name == "web_search_lookup":
return "python.org" in combined and (
"official home of the python" in combined
or "welcome to python.org" in combined
or "https://www.python.org" in combined
)
if case_name == "web_fetch_url":
return "example domain" in combined and "https://example.com" in combined
response_lower = (rendered_response or "").lower()
if case_name == "documents_search_fixture":
return DOCUMENT_SEARCH_TITLE.lower() in combined and "lapis-otter-419" in response_lower
if case_name == "tasks_search_fixture":
return TASK_SEARCH_NAME.lower() in combined and "amber-river-782" in response_lower
if case_name == "calendar_search_fixture":
return CALENDAR_SEARCH_TITLE.lower() in combined and "cobalt-sun-531" in response_lower
if case_name == "chat_search":
return "qwen" in combined and ("found" in combined or "session" in combined)
return True
def _malformed_text_surface(rendered_response: str) -> bool:
value = (rendered_response or "").lower()
if any(
marker in value
for marker in (
"<function",
"<parameter",
"function=",
"parameter=",
"</parameter",
'"function"',
"tool_call",
"|end|",
)
):
return True
return any(
re.search(pattern, rendered_response or "")
for pattern in (
r"\bIamOdysseus\b",
r"\bSwedenisin\b",
r"\bVATstands\b",
r"\bpublic\s+domain\s+ar\b(?!t)",
r"\bthe\s+me\s+open\s+access\b",
r"\bar\s+institute\b(?!t)",
r"\bpublicdomainar\b",
r"\bTheMeOpenAccess\b",
r"\bCanyouclarifywhich\b",
r"\bwan\s+me\b",
r"\bwan me\s+link\b",
r"\bwhat you wan me\b",
r"\bwan links\b",
r"\bget/se\b",
r"\bdefaul account\b",
r"\bdraf\b",
r"\bye\b",
r"\bagen limits\b",
r"\bthis cha\b",
)
)
def _raise_for_status_with_body(response: httpx.Response) -> None:
"""Keep API validation details in live-eval output instead of hiding them."""
try:
response.raise_for_status()
except httpx.HTTPStatusError as exc:
# ``client.stream`` has not buffered the body yet. Read it explicitly
# before accessing ``text`` or a parser error can hide the real API
# validation failure behind ``ResponseNotRead``.
if not response.is_closed:
response.read()
detail = response.text.strip().replace("\n", " ")[:500]
if detail:
raise RuntimeError(f"{exc}; response={detail}") from exc
raise
def _hard_turn_timeout(args) -> float:
"""Read the shared turn timeout across evaluator argument namespaces.
The extended evaluator reuses ``run_case`` but names its outer watchdog
``hard_case_timeout``. Keep the shared runner compatible with both entry
points instead of failing before the HTTP request starts.
"""
return float(
getattr(
args,
"hard_turn_timeout",
getattr(args, "hard_case_timeout", 0) or 0,
)
or 0
)
def _reported_model(args) -> str:
"""Name the model that actually receives the evaluated request."""
return str(
getattr(args, "selected_model", "")
or getattr(args, "model", "")
or ""
)
def _summary_exit_code(records: list[dict]) -> int:
"""Fail the CLI when any selected case did not actually complete."""
if not records:
return 2
return 0 if all(
bool(record.get("execution_ok"))
and bool(record.get("response_quality_ok"))
and not bool(record.get("duplicate_textual_call"))
for record in records
) else 1
def _is_infra_failure_error(error: dict) -> bool:
"""Classify transport/provider outages separately from model behavior."""
if not isinstance(error, dict):
return False
status = error.get("status")
text = " ".join(
str(error.get(key) or "")
for key in ("error", "message", "detail", "type")
).lower()
if status in {502, 503, 504, 520, 521, 522, 523, 524}:
return True
return bool(
"cannot reach" in text
or "connection refused" in text
or "connection reset" in text
or "connect timeout" in text
or "read timeout" in text
or "unreachable" in text
or "cooldown active" in text
or "upstream protocol error" in text
or "upstream" in text and "failed" in text
)
def _exception_record(name: str, message: str, expected: str, exc: Exception) -> dict:
error = repr(exc)
return {
"case": name,
"message": message,
"expected_tool": expected,
"first_tool": None,
"native_call_ok": False,
"command_contract_ok": False,
"tool_count": 0,
"clean_execution_ok": False,
"failed_tool_events": [],
"tool_invocation_ok": False,
"command_outcome_ok": False,
"infra_failure": True,
"model_evaluable": False,
"execution_ok": False,
"duplicate_textual_call": False,
"repetitive_tool_call": False,
"stream_errors": [{"type": "case_exception", "error": error}],
"stream_exception": error,
"tool_outputs": [],
"approval_tool_events": [],
"metrics": None,
"model_request_snapshots": [],
"elapsed_seconds": 0,
"response": "",
"content_quality_ok": False,
"response_quality_ok": False,
"approval_turns": 0,
}
def _is_infra_failure_tool_output(event: dict) -> bool:
"""Classify tool-runner outages separately from model behavior.
TUI/local cases are only meaningful when the browser/TUI advertises a host
bridge. The model can correctly route to host_shell while the HTTP eval
container still cannot execute it; count that as infrastructure so it does
not look like a failed tool-routing train.
"""
if not isinstance(event, dict):
return False
text = " ".join(
str(event.get(key) or "")
for key in ("output", "error", "message", "detail")
).lower()
return bool(
"no tui host bridge advertised" in text
or "missing tui host bridge" in text
or "host bridge unavailable" in text
)
def _stream_exception_if_empty(
events: list[dict], response_text: list[str], stream_exception: str | None
) -> str | None:
"""Return a diagnostic when a supposedly successful stream had no data."""
if not events and not response_text and not stream_exception:
return "empty SSE stream"
return stream_exception
def _tool_approval_from_event(event: dict) -> dict | None:
"""Return an approval payload regardless of which SSE wrapper carried it."""
candidates = [event, event.get("data"), event.get("ask_user")]
for candidate in candidates:
if not isinstance(candidate, dict):
continue
approval = candidate.get("ask_user") if isinstance(candidate.get("ask_user"), dict) else candidate
if (
isinstance(approval, dict)
and approval.get("kind") == "tool_approval"
and approval.get("approval_id")
):
return approval
return None
def run_case(client: httpx.Client, args, name: str, message: str, expected: str):
# The route reconciles the selected endpoint on the chat request. Create
# the disposable session with that same route so the evaluator cannot
# accidentally validate one model and execute another.
session_endpoint = args.selected_endpoint_url or args.endpoint
session_model = args.selected_model or args.model
create = client.post(
args.base_url.rstrip("/") + "/api/session",
data={
"name": "[eval] " + name,
"endpoint_url": session_endpoint,
**({"endpoint_id": args.endpoint_id} if args.endpoint_id else {}),
"model": session_model,
"skip_validation": "true",
"rag": "false",
},
timeout=30,
)
_raise_for_status_with_body(create)
session_id = create.json()["id"]
started = time.monotonic()
events = []
response_text = []
stream_exception = None
approval_turns = 0
try:
try:
turn_data = {
"message": message,
"session": session_id,
"mode": "agent",
"agent_prompt_mode": args.prompt_mode,
**({"selected_endpoint_id": args.endpoint_id} if args.endpoint_id else {}),
**({"selected_endpoint_url": args.selected_endpoint_url} if args.selected_endpoint_url else {}),
**({"selected_model": args.selected_model} if args.selected_model else {}),
}
runtime_context = getattr(args, "client_runtime_context", None)
if runtime_context:
turn_data["client_runtime_context"] = json.dumps(
runtime_context,
separators=(",", ":"),
sort_keys=True,
)
# The TUI sends the active cwd through both the form fields
# and runtime JSON. Keep live evaluations on that same
# contract; runtime JSON alone is not enough for the backend
# workspace guard.
session_cwd = str(
runtime_context.get("session_cwd")
or runtime_context.get("sessionCwd")
or runtime_context.get("cwd")
or ""
).strip()
if session_cwd:
turn_data["cwd"] = session_cwd
turn_data["workspace"] = session_cwd
with hard_timeout(_hard_turn_timeout(args), name):
while True:
approval = None
with client.stream(
"POST",
args.base_url.rstrip("/") + "/api/chat_stream",
data=turn_data,
headers={"Accept": "text/event-stream"},
timeout=args.timeout,
) as response:
_raise_for_status_with_body(response)
for event in _sse_events(response):
events.append(event)
visible_text = _visible_event_text(event)
if visible_text:
if event.get("type") == "final_response":
# Approval continuations replace the pending
# draft in the TUI. Do the same in the live
# response metric instead of reporting the
# old approval question concatenated with the
# final result.
response_text[:] = [visible_text]
else:
response_text.append(visible_text)
approval = approval or _tool_approval_from_event(event)
if (
not getattr(args, "auto_approve", True)
or not approval
or approval_turns >= 3
):
break
approval_turns += 1
turn_data = {
**turn_data,
"tool_approval_id": approval["approval_id"],
"tool_approval_decision": "approve",
}
except Exception as exc:
stream_exception = repr(exc)
finally:
# The session is disposable. Failure to delete must not hide the test
# result, and deletion is intentionally best-effort.
try:
client.delete(args.base_url.rstrip("/") + f"/api/session/{session_id}", timeout=15)
except Exception:
pass
stream_exception = _stream_exception_if_empty(
events, response_text, stream_exception
)
starts = [e for e in events if e.get("type") == "tool_start"]
outputs = [e for e in events if e.get("type") == "tool_output"]
errors = [e for e in events if e.get("type") == "error"]
if stream_exception:
errors.append({"type": "client_exception", "error": stream_exception})
infra_failure = any(_is_infra_failure_error(error) for error in errors)
metrics = [e.get("data") for e in events if e.get("type") == "metrics" and isinstance(e.get("data"), dict)]
model_request_snapshots = [
e for e in events if e.get("type") == "model_request_snapshot"
]
aggregate_metrics = dict(metrics[-1]) if metrics else None
if aggregate_metrics is not None:
aggregate_metrics["tool_events"] = [
tool_event
for metric in metrics
for tool_event in (metric.get("tool_events") or [])
]
aggregate_metrics["round_texts"] = [
str(round_text)
for metric in metrics
for round_text in (metric.get("round_texts") or [])
]
rendered_response = "".join(response_text).strip()
if not rendered_response and aggregate_metrics:
round_texts = aggregate_metrics.get("round_texts") or []
rendered_response = next(
(str(item).strip() for item in reversed(round_texts) if str(item).strip()),
"",
)
first_tool = starts[0].get("tool") if starts else None
response_blob = "".join(response_text).lower()
duplicate_text = any(
token in response_blob
for token in (
"manage_notes(",
'"function"',
"<tool_call",
"<function=",
"<parameter=",
)
) or bool(re.search(r"\bmcp__email__list_emails\s*(?:\(|\{)", response_blob))
def _output_ok(event):
ask_user = event.get("ask_user")
if isinstance(ask_user, dict) and ask_user.get("kind") == "tool_approval":
return False
if str(event.get("output") or "").lstrip().lower().startswith("waiting for an exact user approval"):
return False
if event.get("exit_code") not in (0, None):
return False
output = event.get("output")
return not isinstance(output, str) or not output.lstrip().lower().startswith("error")
def _invocation_ok(event):
ask_user = event.get("ask_user")
if isinstance(ask_user, dict) and ask_user.get("kind") == "tool_approval":
return False
if str(event.get("output") or "").lstrip().lower().startswith("waiting for an exact user approval"):
return False
output = event.get("output")
return not isinstance(output, str) or not output.lstrip().lower().startswith("error")
# Approval continuations emit their executed result in the continuation's
# metrics tool_events. Include those alongside raw SSE tool_output events;
# otherwise the evaluator sees only the initial "waiting for approval"
# placeholder and reports a successful approved action as a failure.
metric_tool_events = [
event
for metric in metrics
for event in (metric.get("tool_events") or [])
if isinstance(event, dict)
]
infra_failure = infra_failure or any(
_is_infra_failure_tool_output(event)
for event in [*outputs, *metric_tool_events]
)
executed_outputs = [
event for event in [*outputs, *metric_tool_events]
if _output_ok(event)
]
invoked_outputs = [
event for event in [*outputs, *metric_tool_events]
if _invocation_ok(event)
]
failed_tool_events = [
{
"tool": event.get("tool"),
"command": event.get("command"),
"output": str(event.get("output") or "")[:500],
"exit_code": event.get("exit_code"),
}
for event in metric_tool_events or outputs
if event.get("exit_code") not in (0, None)
or str(event.get("output") or "").lstrip().lower().startswith("error")
]
# A safe agent turn may stop before tool_start because exact approval is
# required. The sealed action is still authoritative evidence of what the
# agent proposed; score that separately from execution.
approval_tool_events = []
for event in outputs:
approval = event.get("ask_user")
action = approval.get("action") if isinstance(approval, dict) else None
if not isinstance(action, dict):
continue
approval_tool_events.append(
{
"tool": action.get("tool") or event.get("tool"),
"command": action.get("content"),
"approval_required": True,
}
)
observed_tools = starts or approval_tool_events
observed_first_tool = observed_tools[0].get("tool") if observed_tools else None
observed_tool_names = [str(event.get("tool") or "") for event in observed_tools]
observed_tool_calls = [
(
str(event.get("tool") or ""),
str(event.get("command") or ""),
)
for event in observed_tools
]
repetitive_tool_call = any(
observed_tool_calls.count(call) > 1 for call in set(observed_tool_calls)
)
native_call_ok = _tool_sequence_matches(observed_tool_names, expected)
command_contract_ok = expected == "no_tool" or _command_contract_ok(name, [*starts, *approval_tool_events, *metric_tool_events])
response_quality_ok = bool(rendered_response) and not _malformed_text_surface(rendered_response) and not any(
marker in rendered_response.lower()
for marker in (
"the model returned an empty response",
"allow this exact action once?allow this exact action once?",
"i gathered some search results but couldn't pull a clean answer together",
)
)
# A host-local TUI case must never succeed by touching the web route. This
# is intentionally a response/behavior quality gate in addition to the
# first-tool score, so a later fallback cannot hide a bad initial route.
if expected == "host_shell" and "web_search" in observed_tool_names:
response_quality_ok = False
if expected == "no_tool" and not _no_tool_quality_ok(name, rendered_response):
response_quality_ok = False
if name.startswith("email_") and not _email_action_quality_ok(name, rendered_response):
response_quality_ok = False
content_quality_ok = _content_quality_ok(name, rendered_response, [*outputs, *metric_tool_events])
if not content_quality_ok:
response_quality_ok = False
if not command_contract_ok:
response_quality_ok = False
if repetitive_tool_call:
response_quality_ok = False
if infra_failure:
response_quality_ok = False
tool_invocation_ok = (
bool(rendered_response)
if expected == "no_tool"
else native_call_ok and bool(invoked_outputs)
) and not errors
command_outcome_ok = (
bool(rendered_response)
if expected == "no_tool"
else native_call_ok and bool(executed_outputs)
) and not errors
return {
"case": name,
"message": message,
"expected_tool": expected,
"first_tool": observed_first_tool,
"native_call_ok": native_call_ok,
"command_contract_ok": command_contract_ok,
"tool_count": len(observed_tools),
"clean_execution_ok": not failed_tool_events and not errors,
"failed_tool_events": failed_tool_events,
# tool_invocation_ok: the right tool actually ran and produced a
# usable result event, regardless of the command/program exit code.
# command_outcome_ok: the invoked command/tool also completed with a
# successful outcome. Keep both so model-routing regressions are not
# conflated with legitimate test/build failures from the environment.
"tool_invocation_ok": tool_invocation_ok,
"command_outcome_ok": command_outcome_ok,
"infra_failure": infra_failure,
"model_evaluable": not infra_failure,
# Some registry-backed read tools intentionally omit exit_code. An
# output without an error is still a successful execution.
# A partial tool result followed by a stream timeout is not a
# successful agent turn. Keep the raw outputs for diagnosis, but fail
# the execution score whenever the client observed a stream error.
"execution_ok": command_outcome_ok,
"duplicate_textual_call": duplicate_text,
"repetitive_tool_call": repetitive_tool_call,
"stream_errors": errors,
"stream_exception": stream_exception,
"tool_outputs": [
{"tool": e.get("tool"), "exit_code": e.get("exit_code")}
for e in outputs
],
"approval_tool_events": approval_tool_events,
"metrics": aggregate_metrics,
"model_request_snapshots": model_request_snapshots,
"elapsed_seconds": round(time.monotonic() - started, 3),
"response": rendered_response[:2000],
"content_quality_ok": content_quality_ok,
"response_quality_ok": response_quality_ok,
"approval_turns": approval_turns,
}
def main():
parser = argparse.ArgumentParser()
parser.add_argument("--base-url", default="http://127.0.0.1:7011")
parser.add_argument("--endpoint", default="http://192.168.1.21:8065/v1/chat/completions")
parser.add_argument("--model", default="/Users/pewds/models/qwen36-27b-mlx-8bit")
parser.add_argument("--endpoint-id", default="")
parser.add_argument("--selected-endpoint-url", default="")
parser.add_argument("--selected-model", default="")
parser.add_argument(
"--client-runtime-context",
default="",
help="JSON object passed as the TUI client_runtime_context form field.",
)
parser.add_argument("--cookie-file", default="data/sessions.json")
parser.add_argument("--output", required=True)
parser.add_argument("--prompt-mode", default="auto")
parser.add_argument("--timeout", type=float, default=180)
parser.add_argument("--hard-turn-timeout", type=float, default=0)
parser.add_argument(
"--no-auto-approve",
dest="auto_approve",
action="store_false",
help="Stop at the first exact approval instead of continuing the sealed action.",
)
parser.add_argument(
"--cases",
default="",
help="Comma-separated case names to run. Default: all cases.",
)
parser.add_argument(
"--include-no-tool",
action="store_true",
help="Include regular chat/general knowledge cases that should not call tools.",
)
parser.add_argument(
"--include-tui-local",
action="store_true",
help="Include host-workspace/network prompts; pass --client-runtime-context too.",
)
parser.add_argument(
"--include-email-safety",
action="store_true",
help="Include explicit email send/reply/archive/delete cases against a temporary fake inbox.",
)
parser.add_argument(
"--include-safe-extended",
action="store_true",
help="Include read-only/list/search coverage for lower-frequency Odysseus tools.",
)
parser.add_argument(
"--no-email-fixture",
action="store_true",
help="Disable the temporary fake inbox for email-safety cases. Dangerous outside disposable fixtures.",
)
args = parser.parse_args()
if args.client_runtime_context:
try:
args.client_runtime_context = json.loads(args.client_runtime_context)
except json.JSONDecodeError as exc:
raise SystemExit(f"--client-runtime-context must be valid JSON: {exc}") from exc
if not isinstance(args.client_runtime_context, dict):
raise SystemExit("--client-runtime-context must decode to a JSON object")
else:
args.client_runtime_context = None
if args.include_tui_local:
if not args.client_runtime_context:
raise SystemExit("--include-tui-local requires --client-runtime-context JSON")
surface = str(args.client_runtime_context.get("surface") or "").strip()
if surface != "odysseus-tui":
raise SystemExit(
"--include-tui-local requires client_runtime_context.surface='odysseus-tui'; "
f"got {surface!r}. Other surface values are dropped by the live chat route."
)
output = Path(args.output)
output.parent.mkdir(parents=True, exist_ok=True)
client = httpx.Client(
cookies={"odysseus_session": _cookie(Path(args.cookie_file))},
follow_redirects=False,
)
records = []
try:
requested = {
item.strip()
for item in args.cases.split(",")
if item.strip()
}
available_cases = (
CASES
+ (NO_TOOL_CASES if args.include_no_tool else [])
+ (TUI_LOCAL_CASES if args.include_tui_local else [])
+ (EMAIL_SAFETY_CASES if args.include_email_safety else [])
+ (SAFE_EXTENDED_CASES if args.include_safe_extended else [])
)
selected_cases = [
case for case in available_cases
if not requested or case[0] in requested
]
unknown = requested - {case[0] for case in available_cases}
if unknown:
raise SystemExit(f"Unknown case(s): {', '.join(sorted(unknown))}")
selected_case_names = {case[0] for case in selected_cases}
use_email_fixture = (
not args.no_email_fixture
and any(name.startswith("email_") for name in selected_case_names)
)
with _email_fixture(use_email_fixture):
with _content_fixtures(client, args.base_url, selected_case_names):
for name, message, expected in selected_cases:
try:
record = run_case(client, args, name, message, expected)
except Exception as exc:
record = _exception_record(name, message, expected, exc)
records.append(record)
print(json.dumps(record, ensure_ascii=True), flush=True)
break
records.append(record)
print(json.dumps(record, ensure_ascii=True), flush=True)
finally:
client.close()
evaluable_records = [
record for record in records
if not bool(record.get("infra_failure"))
]
summary = {
"model": _reported_model(args),
"cases": len(records),
"infra_failures": sum(bool(r.get("infra_failure")) for r in records),
"evaluable_cases": len(evaluable_records),
"native_success": sum(r["native_call_ok"] for r in records),
"native_success_evaluable": sum(r["native_call_ok"] for r in evaluable_records),
"command_contract_success": sum(r["command_contract_ok"] for r in records),
"command_contract_success_evaluable": sum(r["command_contract_ok"] for r in evaluable_records),
"tool_invocation_success": sum(r.get("tool_invocation_ok", r["execution_ok"]) for r in records),
"tool_invocation_success_evaluable": sum(
r.get("tool_invocation_ok", r["execution_ok"]) for r in evaluable_records
),
"command_outcome_success": sum(r.get("command_outcome_ok", r["execution_ok"]) for r in records),
"command_outcome_success_evaluable": sum(
r.get("command_outcome_ok", r["execution_ok"]) for r in evaluable_records
),
"execution_success": sum(r["execution_ok"] for r in records),
"execution_success_evaluable": sum(r["execution_ok"] for r in evaluable_records),
"response_quality_success": sum(r["response_quality_ok"] for r in records),
"response_quality_success_evaluable": sum(r["response_quality_ok"] for r in evaluable_records),
"content_quality_success": sum(r.get("content_quality_ok", r["response_quality_ok"]) for r in records),
"content_quality_success_evaluable": sum(
r.get("content_quality_ok", r["response_quality_ok"]) for r in evaluable_records
),
"clean_execution_success": sum(r.get("clean_execution_ok", r["execution_ok"]) for r in records),
"clean_execution_success_evaluable": sum(
r.get("clean_execution_ok", r["execution_ok"]) for r in evaluable_records
),
"failed_tool_event_cases": sum(bool(r.get("failed_tool_events")) for r in records),
"duplicate_textual_calls": sum(r["duplicate_textual_call"] for r in records),
"repetitive_tool_calls": sum(r.get("repetitive_tool_call", False) for r in records),
"stream_errors": sum(bool(r["stream_errors"]) for r in records),
"records": records,
}
output.write_text(json.dumps(summary, indent=2, ensure_ascii=True) + "\n")
print("SUMMARY", json.dumps({k: summary[k] for k in summary if k != "records"}))
return _summary_exit_code(records)
if __name__ == "__main__":
raise SystemExit(main())