test(bench): add labelled routing fixtures and router benchmark
Measures Steward routing against model and thinking settings by talking
to Ollama directly. No server, no agents, nothing executed — the
mutating fixtures only ever produce a routing decision — so the run is
cheap, repeatable and isolates routing from everything downstream. The
request body mirrors StewardAgent._call_ollama, so the `unset` cell is
exactly what production sends today.
Three thinking settings rather than two. `unset` is production, and it
is not neutral: gemma4 reasons by default and returns no `thinking`
field, so those tokens are generated and discarded.
Scoring is asymmetric on purpose. Each fixture carries `forbid` as well
as `expect`, because over-routing is the predicted failure when thinking
is off and it is the expensive one — a spurious librarian is a real web
call on a query that asked for arithmetic.
The adversarial group is regression coverage for the extraction fix in
a905363: those queries invite the vocabulary that used to select agents
by substring, so they now assert that routing follows what the Steward
decided rather than the words it used while explaining.
Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,221 @@
|
||||
"""
|
||||
Benchmark Steward routing quality against model and thinking settings.
|
||||
|
||||
Talks to Ollama directly. No Tatlock server, no agents, no tools, nothing is
|
||||
executed — the mutating fixtures ("turn on the lights", "update the wiki") only
|
||||
ever produce a routing decision. That makes this cheap and repeatable, and it
|
||||
isolates the question: does the Steward still pick the right capabilities when
|
||||
the model reasons less?
|
||||
|
||||
The request body is byte-identical to StewardAgent._call_ollama, plus the
|
||||
`think` flag under test, so a cell labelled `unset` is exactly what production
|
||||
sends today.
|
||||
|
||||
Three thinking settings, because "on vs off" hides the interesting case:
|
||||
|
||||
unset what production sends now. gemma4 reasons by default, and the
|
||||
response carries no `thinking` field, so those tokens are generated
|
||||
and discarded.
|
||||
true reasoning requested explicitly and returned in `thinking`.
|
||||
false reasoning suppressed.
|
||||
|
||||
Scoring is deliberately asymmetric. A missing capability under-routes and the
|
||||
Butler answers without a tool it needed; a spurious one over-routes, and that is
|
||||
a real agent call — a stray librarian is a multi-second web search on a query
|
||||
that asked for arithmetic. Over-routing is the predicted failure when thinking
|
||||
is off, so `forbid` violations are reported separately rather than folded into
|
||||
one accuracy number.
|
||||
|
||||
Usage:
|
||||
.venv/bin/python scripts/benchmark_routing.py
|
||||
.venv/bin/python scripts/benchmark_routing.py --models gemma4:e2b
|
||||
.venv/bin/python scripts/benchmark_routing.py --think false --repeats 3
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import statistics
|
||||
import sys
|
||||
import time
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
PROJECT_ROOT = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(PROJECT_ROOT))
|
||||
|
||||
from scripts.fixtures.routing_fixtures import FIXTURES # noqa: E402
|
||||
from src.agents.steward.agent import build_steward_prompt # noqa: E402
|
||||
from src.agents.steward.service import _DELEGATE_LINE_RE, _extract_capabilities # noqa: E402
|
||||
from src.core.startup import register_household_members # noqa: E402
|
||||
|
||||
OLLAMA_URL = "http://localhost:11434"
|
||||
DEFAULT_MODELS = ["gemma4:e2b", "gemma4:e4b"]
|
||||
DEFAULT_THINK = ["unset", "true", "false"]
|
||||
RESULTS_DIR = PROJECT_ROOT / "logs"
|
||||
|
||||
|
||||
def build_body(model: str, prompt: str, think: str) -> dict[str, Any]:
|
||||
"""Mirror StewardAgent._call_ollama exactly, then add the flag under test."""
|
||||
body: dict[str, Any] = {
|
||||
"model": model,
|
||||
"prompt": prompt,
|
||||
"stream": False,
|
||||
"options": {
|
||||
"temperature": 0.3, # Lower = more consistent
|
||||
"top_p": 0.9,
|
||||
},
|
||||
}
|
||||
if think != "unset":
|
||||
body["think"] = think == "true"
|
||||
return body
|
||||
|
||||
|
||||
def call(client: httpx.Client, body: dict[str, Any]) -> dict[str, Any] | None:
|
||||
try:
|
||||
response = client.post(f"{OLLAMA_URL}/api/generate", json=body)
|
||||
response.raise_for_status()
|
||||
return response.json()
|
||||
except Exception as exc: # noqa: BLE001 - a failed cell must not abort the run
|
||||
print(f" ! {exc}", file=sys.stderr)
|
||||
return None
|
||||
|
||||
|
||||
def score(fixture: dict, found: list[str]) -> dict[str, Any]:
|
||||
expected = set(fixture["expect"])
|
||||
forbidden = set(fixture["forbid"])
|
||||
got = set(found)
|
||||
missing = sorted(expected - got)
|
||||
spurious = sorted(got & forbidden)
|
||||
return {
|
||||
"found": found,
|
||||
"missing": missing,
|
||||
"spurious": spurious,
|
||||
# Exact only when everything expected arrived and nothing forbidden did.
|
||||
"exact": not missing and not spurious,
|
||||
"under_routed": bool(missing),
|
||||
"over_routed": bool(spurious),
|
||||
}
|
||||
|
||||
|
||||
def run_cell(client: httpx.Client, model: str, think: str, repeats: int) -> list[dict[str, Any]]:
|
||||
rows: list[dict[str, Any]] = []
|
||||
for fixture in FIXTURES:
|
||||
prompt = build_steward_prompt(fixture["query"], [])
|
||||
body = build_body(model, prompt, think)
|
||||
for rep in range(repeats):
|
||||
started = time.perf_counter()
|
||||
data = call(client, body)
|
||||
elapsed_ms = (time.perf_counter() - started) * 1000
|
||||
if data is None:
|
||||
rows.append({
|
||||
"id": fixture["id"], "group": fixture["group"], "rep": rep,
|
||||
"error": True, "exact": False, "under_routed": False, "over_routed": False,
|
||||
})
|
||||
continue
|
||||
|
||||
text = data.get("response", "") or ""
|
||||
found = _extract_capabilities(text)
|
||||
rows.append({
|
||||
"id": fixture["id"],
|
||||
"group": fixture["group"],
|
||||
"rep": rep,
|
||||
"error": False,
|
||||
"latency_ms": round(elapsed_ms, 1),
|
||||
"eval_tokens": data.get("eval_count"),
|
||||
"prompt_tokens": data.get("prompt_eval_count"),
|
||||
# Did the model obey the documented output shape at all?
|
||||
"has_delegate_line": bool(_DELEGATE_LINE_RE.search(text)),
|
||||
# Whether reasoning came back, as opposed to being generated and dropped.
|
||||
"thinking_returned": bool(data.get("thinking")),
|
||||
"response_chars": len(text),
|
||||
**score(fixture, found),
|
||||
})
|
||||
return rows
|
||||
|
||||
|
||||
def summarise(rows: list[dict[str, Any]]) -> dict[str, Any]:
|
||||
ok = [r for r in rows if not r["error"]]
|
||||
if not ok:
|
||||
return {"n": 0, "errors": len(rows)}
|
||||
latencies = [r["latency_ms"] for r in ok]
|
||||
tokens = [r["eval_tokens"] for r in ok if r["eval_tokens"] is not None]
|
||||
return {
|
||||
"n": len(ok),
|
||||
"errors": len(rows) - len(ok),
|
||||
"exact_pct": round(100 * sum(r["exact"] for r in ok) / len(ok), 1),
|
||||
"under_routed_pct": round(100 * sum(r["under_routed"] for r in ok) / len(ok), 1),
|
||||
"over_routed_pct": round(100 * sum(r["over_routed"] for r in ok) / len(ok), 1),
|
||||
"format_ok_pct": round(100 * sum(r["has_delegate_line"] for r in ok) / len(ok), 1),
|
||||
"thinking_returned_pct": round(100 * sum(r["thinking_returned"] for r in ok) / len(ok), 1),
|
||||
"latency_ms_median": round(statistics.median(latencies), 1),
|
||||
"latency_ms_mean": round(statistics.fmean(latencies), 1),
|
||||
"eval_tokens_median": round(statistics.median(tokens), 1) if tokens else None,
|
||||
"eval_tokens_total": sum(tokens) if tokens else None,
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--models", default=",".join(DEFAULT_MODELS))
|
||||
parser.add_argument("--think", default=",".join(DEFAULT_THINK),
|
||||
help="comma-separated subset of unset,true,false")
|
||||
parser.add_argument("--repeats", type=int, default=1)
|
||||
parser.add_argument("--timeout", type=float, default=180.0)
|
||||
args = parser.parse_args()
|
||||
|
||||
models = [m.strip() for m in args.models.split(",") if m.strip()]
|
||||
think_modes = [t.strip() for t in args.think.split(",") if t.strip()]
|
||||
|
||||
# build_steward_prompt reads the registry, and the registry is populated at
|
||||
# application startup. Without this the prompt lists no capabilities and every
|
||||
# cell scores zero for reasons that have nothing to do with the model.
|
||||
register_household_members()
|
||||
|
||||
print(f"{len(FIXTURES)} fixtures x {len(models)} models x {len(think_modes)} think "
|
||||
f"x {args.repeats} repeats = {len(FIXTURES) * len(models) * len(think_modes) * args.repeats} calls\n")
|
||||
|
||||
cells: dict[str, Any] = {}
|
||||
with httpx.Client(timeout=args.timeout) as client:
|
||||
for model in models:
|
||||
# Absorb the cold load (~36s) outside the measurements.
|
||||
print(f"warming {model} ...", flush=True)
|
||||
call(client, build_body(model, "hi", "false"))
|
||||
for think in think_modes:
|
||||
key = f"{model}|think={think}"
|
||||
print(f" {key} ...", end=" ", flush=True)
|
||||
started = time.perf_counter()
|
||||
rows = run_cell(client, model, think, args.repeats)
|
||||
summary = summarise(rows)
|
||||
cells[key] = {"summary": summary, "rows": rows}
|
||||
print(f"exact={summary.get('exact_pct')}% "
|
||||
f"over={summary.get('over_routed_pct')}% "
|
||||
f"median={summary.get('latency_ms_median')}ms "
|
||||
f"({time.perf_counter() - started:.0f}s)")
|
||||
|
||||
RESULTS_DIR.mkdir(parents=True, exist_ok=True)
|
||||
stamp = datetime.now(UTC).strftime("%Y%m%dT%H%M%SZ")
|
||||
out = RESULTS_DIR / f"routing-bench-{stamp}.json"
|
||||
out.write_text(json.dumps({
|
||||
"generated_at": datetime.now(UTC).isoformat(),
|
||||
"fixtures": len(FIXTURES),
|
||||
"repeats": args.repeats,
|
||||
"cells": cells,
|
||||
}, indent=2))
|
||||
|
||||
print(f"\n{'cell':28} {'exact':>7} {'under':>7} {'over':>7} {'fmt':>6} {'tok':>7} {'ms':>8}")
|
||||
print("-" * 76)
|
||||
for key, cell in cells.items():
|
||||
s = cell["summary"]
|
||||
print(f"{key:28} {s.get('exact_pct'):>6}% {s.get('under_routed_pct'):>6}% "
|
||||
f"{s.get('over_routed_pct'):>6}% {s.get('format_ok_pct'):>5}% "
|
||||
f"{str(s.get('eval_tokens_median')):>7} {s.get('latency_ms_median'):>8}")
|
||||
print(f"\nwritten to {out}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,159 @@
|
||||
"""
|
||||
Labelled queries for the Steward routing benchmark.
|
||||
|
||||
Each fixture carries both `expect` and `forbid`:
|
||||
|
||||
expect capabilities that must appear. Missing one is under-routing — the
|
||||
Butler answers without a tool it needed.
|
||||
forbid capabilities that must not appear. Over-routing is not cosmetic: a
|
||||
spurious librarian is a real multi-second web call, and a spurious
|
||||
housekeeper can actuate hardware.
|
||||
|
||||
`forbid` matters more than `expect` here, because over-recommendation is the
|
||||
predicted failure when model thinking is disabled and the Steward has less room
|
||||
to discriminate.
|
||||
|
||||
The `adversarial` group deserves explanation. Until 2026-08-08 the extractor
|
||||
substring-matched capability *domains* across the Steward's whole response, so
|
||||
ordinary English in its REASON line selected agents: "description" contains the
|
||||
housekeeper domain "script", "acknowledge" contains "knowledge" and "know",
|
||||
"economy" contains the biographer domain "my". Those queries invite exactly that
|
||||
vocabulary. They now serve as an end-to-end regression: routing must depend on
|
||||
what the Steward *decided*, not on the words it happened to use while explaining.
|
||||
|
||||
Expectations follow the routing rules stated in the Steward prompt itself
|
||||
(src/agents/steward/agent.py), not on what a capability could plausibly cover.
|
||||
"""
|
||||
|
||||
CORE = "tatlock_core"
|
||||
LIB = "librarian"
|
||||
BIO = "biographer"
|
||||
HOUSE = "housekeeper"
|
||||
ALL = [CORE, LIB, BIO, HOUSE]
|
||||
|
||||
|
||||
def _others(*keep: str) -> list[str]:
|
||||
return [c for c in ALL if c not in keep]
|
||||
|
||||
|
||||
FIXTURES: list[dict] = [
|
||||
# --- arithmetic and computation -> tatlock_core --------------------------
|
||||
{"id": "math_add", "group": "math", "query": "What is 61 plus 12?",
|
||||
"expect": [CORE], "forbid": _others(CORE)},
|
||||
{"id": "math_percent", "group": "math", "query": "What is 15% of 240?",
|
||||
"expect": [CORE], "forbid": _others(CORE)},
|
||||
{"id": "math_compound", "group": "math", "query": "If I save 200 a month for 3 years, how much is that?",
|
||||
"expect": [CORE], "forbid": _others(CORE)},
|
||||
{"id": "math_sqrt", "group": "math", "query": "What is the square root of 1764?",
|
||||
"expect": [CORE], "forbid": _others(CORE)},
|
||||
|
||||
# --- date and time -> tatlock_core ---------------------------------------
|
||||
{"id": "time_now", "group": "datetime", "query": "What time is it?",
|
||||
"expect": [CORE], "forbid": _others(CORE)},
|
||||
{"id": "time_date", "group": "datetime", "query": "What is today's date?",
|
||||
"expect": [CORE], "forbid": _others(CORE)},
|
||||
{"id": "time_delta", "group": "datetime", "query": "How many days until Christmas?",
|
||||
"expect": [CORE], "forbid": _others(CORE)},
|
||||
|
||||
# --- personal memory -> biographer ---------------------------------------
|
||||
{"id": "bio_location", "group": "biographer", "query": "Where do I live?",
|
||||
"expect": [BIO], "forbid": [LIB, HOUSE]},
|
||||
{"id": "bio_name", "group": "biographer", "query": "What's my name?",
|
||||
"expect": [BIO], "forbid": [LIB, HOUSE]},
|
||||
{"id": "bio_car", "group": "biographer", "query": "What car do I drive?",
|
||||
"expect": [BIO], "forbid": [LIB, HOUSE]},
|
||||
{"id": "bio_store", "group": "biographer", "query": "Remember that I prefer my coffee black.",
|
||||
"expect": [BIO], "forbid": [LIB, HOUSE]},
|
||||
{"id": "bio_list", "group": "biographer", "query": "What do you know about me?",
|
||||
"expect": [BIO], "forbid": [LIB, HOUSE]},
|
||||
{"id": "bio_forget", "group": "biographer", "query": "Forget my old address.",
|
||||
"expect": [BIO], "forbid": [LIB, HOUSE]},
|
||||
|
||||
# --- research and current information -> librarian ------------------------
|
||||
{"id": "lib_weather", "group": "librarian", "query": "What's the weather in Rotterdam tomorrow?",
|
||||
"expect": [LIB], "forbid": [HOUSE]},
|
||||
{"id": "lib_news", "group": "librarian", "query": "What's in the news today?",
|
||||
"expect": [LIB], "forbid": [HOUSE, BIO]},
|
||||
{"id": "lib_url", "group": "librarian", "query": "Read https://example.com/article and summarise it.",
|
||||
"expect": [LIB], "forbid": [HOUSE, BIO]},
|
||||
{"id": "lib_research", "group": "librarian", "query": "Research how tidal power stations work.",
|
||||
"expect": [LIB], "forbid": [HOUSE, BIO]},
|
||||
{"id": "lib_wiki_create", "group": "librarian", "query": "Create a wiki page about our network topology.",
|
||||
"expect": [LIB], "forbid": [HOUSE, BIO]},
|
||||
|
||||
# --- home automation -> housekeeper --------------------------------------
|
||||
{"id": "house_lights_on", "group": "housekeeper", "query": "Turn on the kitchen lights.",
|
||||
"expect": [HOUSE], "forbid": [LIB, BIO, CORE]},
|
||||
{"id": "house_lights_off", "group": "housekeeper", "query": "Switch off all the lights downstairs.",
|
||||
"expect": [HOUSE], "forbid": [LIB, BIO, CORE]},
|
||||
{"id": "house_thermostat", "group": "housekeeper", "query": "Set the thermostat to 20 degrees.",
|
||||
"expect": [HOUSE], "forbid": [LIB, BIO]},
|
||||
{"id": "house_blinds", "group": "housekeeper", "query": "Close the blinds in the living room.",
|
||||
"expect": [HOUSE], "forbid": [LIB, BIO, CORE]},
|
||||
|
||||
# --- conversational -> nothing at all -------------------------------------
|
||||
# The expensive failure mode: a greeting that triggers a web search.
|
||||
{"id": "chat_greeting", "group": "conversational", "query": "Hello!",
|
||||
"expect": [], "forbid": ALL},
|
||||
{"id": "chat_thanks", "group": "conversational", "query": "Thanks, that's helpful.",
|
||||
"expect": [], "forbid": ALL},
|
||||
{"id": "chat_joke", "group": "conversational", "query": "Tell me a joke.",
|
||||
"expect": [], "forbid": ALL},
|
||||
{"id": "chat_howareyou", "group": "conversational", "query": "How are you doing today?",
|
||||
"expect": [], "forbid": ALL},
|
||||
{"id": "chat_prior_turn", "group": "conversational", "query": "What did I just say?",
|
||||
"expect": [], "forbid": ALL},
|
||||
|
||||
# --- genuinely multi-capability -------------------------------------------
|
||||
{"id": "multi_weather_home", "group": "multi",
|
||||
"query": "What's the weather here, and remember that I like it warm?",
|
||||
"expect": [LIB, BIO], "forbid": []},
|
||||
{"id": "multi_recall_search", "group": "multi",
|
||||
"query": "Look up the best route from my home address to Utrecht.",
|
||||
"expect": [BIO, LIB], "forbid": []},
|
||||
{"id": "multi_math_memory", "group": "multi",
|
||||
"query": "Remember that my budget is 500 euro, then work out 12% of it.",
|
||||
"expect": [BIO, CORE], "forbid": [LIB, HOUSE]},
|
||||
|
||||
# --- adversarial: vocabulary that used to select agents by substring ------
|
||||
# "temperature" is a housekeeper domain, but this is a unit conversion.
|
||||
{"id": "adv_temperature", "group": "adversarial", "query": "Convert 98.6 Fahrenheit to Celsius.",
|
||||
"expect": [CORE], "forbid": [HOUSE, LIB, BIO]},
|
||||
# "description" contains "script"; "discover" contains "cover".
|
||||
{"id": "adv_description", "group": "adversarial",
|
||||
"query": "Give me a short description of what 17 times 23 comes to.",
|
||||
"expect": [CORE], "forbid": [HOUSE, LIB]},
|
||||
# "acknowledge" contains "knowledge" and "know".
|
||||
{"id": "adv_acknowledge", "group": "adversarial",
|
||||
"query": "Just acknowledge this and add 5 and 6 for me.",
|
||||
"expect": [CORE], "forbid": [LIB, BIO]},
|
||||
# "my" appears inside "economy".
|
||||
{"id": "adv_economy", "group": "adversarial",
|
||||
"query": "How many zeros are in one trillion?",
|
||||
"expect": [CORE], "forbid": [BIO, HOUSE]},
|
||||
# "fan" inside "fantastic"; also a climate word without a home-control intent.
|
||||
{"id": "adv_fantastic", "group": "adversarial",
|
||||
"query": "That's fantastic. What is 8 squared?",
|
||||
"expect": [CORE], "forbid": [HOUSE, LIB]},
|
||||
# "home" without any actuation intent.
|
||||
{"id": "adv_home_word", "group": "adversarial", "query": "What time do I usually get home?",
|
||||
"expect": [BIO], "forbid": [HOUSE]},
|
||||
# "search" as ordinary English, not a web-search request.
|
||||
{"id": "adv_search_word", "group": "adversarial",
|
||||
"query": "No need to search anything, just tell me what 9 times 9 is.",
|
||||
"expect": [CORE], "forbid": [LIB]},
|
||||
# "create"/"write" are librarian domains but this is conversational.
|
||||
{"id": "adv_write_word", "group": "adversarial", "query": "Can you write that more simply?",
|
||||
"expect": [], "forbid": [LIB, HOUSE]},
|
||||
|
||||
# --- mutating intents: routing only, nothing is ever executed -------------
|
||||
{"id": "mutate_wiki_update", "group": "mutating", "query": "Update the dossier page with today's findings.",
|
||||
"expect": [LIB], "forbid": [HOUSE, CORE]},
|
||||
{"id": "mutate_scene", "group": "mutating", "query": "Run the movie night scene.",
|
||||
"expect": [HOUSE], "forbid": [LIB, BIO, CORE]},
|
||||
]
|
||||
|
||||
|
||||
GROUPS = sorted({f["group"] for f in FIXTURES})
|
||||
|
||||
assert len({f["id"] for f in FIXTURES}) == len(FIXTURES), "duplicate fixture id"
|
||||
Reference in New Issue
Block a user