test(bench): add labelled routing fixtures and router benchmark

Measures Steward routing against model and thinking settings by talking
to Ollama directly. No server, no agents, nothing executed — the
mutating fixtures only ever produce a routing decision — so the run is
cheap, repeatable and isolates routing from everything downstream. The
request body mirrors StewardAgent._call_ollama, so the `unset` cell is
exactly what production sends today.

Three thinking settings rather than two. `unset` is production, and it
is not neutral: gemma4 reasons by default and returns no `thinking`
field, so those tokens are generated and discarded.

Scoring is asymmetric on purpose. Each fixture carries `forbid` as well
as `expect`, because over-routing is the predicted failure when thinking
is off and it is the expensive one — a spurious librarian is a real web
call on a query that asked for arithmetic.

The adversarial group is regression coverage for the extraction fix in
a905363: those queries invite the vocabulary that used to select agents
by substring, so they now assert that routing follows what the Steward
decided rather than the words it used while explaining.

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
2026-08-08 15:22:01 +02:00
co-authored by Claude
parent a90536314e
commit 738ff10b93
3 changed files with 380 additions and 0 deletions
+221
View File
@@ -0,0 +1,221 @@
"""
Benchmark Steward routing quality against model and thinking settings.
Talks to Ollama directly. No Tatlock server, no agents, no tools, nothing is
executed — the mutating fixtures ("turn on the lights", "update the wiki") only
ever produce a routing decision. That makes this cheap and repeatable, and it
isolates the question: does the Steward still pick the right capabilities when
the model reasons less?
The request body is byte-identical to StewardAgent._call_ollama, plus the
`think` flag under test, so a cell labelled `unset` is exactly what production
sends today.
Three thinking settings, because "on vs off" hides the interesting case:
unset what production sends now. gemma4 reasons by default, and the
response carries no `thinking` field, so those tokens are generated
and discarded.
true reasoning requested explicitly and returned in `thinking`.
false reasoning suppressed.
Scoring is deliberately asymmetric. A missing capability under-routes and the
Butler answers without a tool it needed; a spurious one over-routes, and that is
a real agent call — a stray librarian is a multi-second web search on a query
that asked for arithmetic. Over-routing is the predicted failure when thinking
is off, so `forbid` violations are reported separately rather than folded into
one accuracy number.
Usage:
.venv/bin/python scripts/benchmark_routing.py
.venv/bin/python scripts/benchmark_routing.py --models gemma4:e2b
.venv/bin/python scripts/benchmark_routing.py --think false --repeats 3
"""
from __future__ import annotations
import argparse
import json
import statistics
import sys
import time
from datetime import UTC, datetime
from pathlib import Path
from typing import Any
import httpx
PROJECT_ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(PROJECT_ROOT))
from scripts.fixtures.routing_fixtures import FIXTURES # noqa: E402
from src.agents.steward.agent import build_steward_prompt # noqa: E402
from src.agents.steward.service import _DELEGATE_LINE_RE, _extract_capabilities # noqa: E402
from src.core.startup import register_household_members # noqa: E402
OLLAMA_URL = "http://localhost:11434"
DEFAULT_MODELS = ["gemma4:e2b", "gemma4:e4b"]
DEFAULT_THINK = ["unset", "true", "false"]
RESULTS_DIR = PROJECT_ROOT / "logs"
def build_body(model: str, prompt: str, think: str) -> dict[str, Any]:
"""Mirror StewardAgent._call_ollama exactly, then add the flag under test."""
body: dict[str, Any] = {
"model": model,
"prompt": prompt,
"stream": False,
"options": {
"temperature": 0.3, # Lower = more consistent
"top_p": 0.9,
},
}
if think != "unset":
body["think"] = think == "true"
return body
def call(client: httpx.Client, body: dict[str, Any]) -> dict[str, Any] | None:
try:
response = client.post(f"{OLLAMA_URL}/api/generate", json=body)
response.raise_for_status()
return response.json()
except Exception as exc: # noqa: BLE001 - a failed cell must not abort the run
print(f" ! {exc}", file=sys.stderr)
return None
def score(fixture: dict, found: list[str]) -> dict[str, Any]:
expected = set(fixture["expect"])
forbidden = set(fixture["forbid"])
got = set(found)
missing = sorted(expected - got)
spurious = sorted(got & forbidden)
return {
"found": found,
"missing": missing,
"spurious": spurious,
# Exact only when everything expected arrived and nothing forbidden did.
"exact": not missing and not spurious,
"under_routed": bool(missing),
"over_routed": bool(spurious),
}
def run_cell(client: httpx.Client, model: str, think: str, repeats: int) -> list[dict[str, Any]]:
rows: list[dict[str, Any]] = []
for fixture in FIXTURES:
prompt = build_steward_prompt(fixture["query"], [])
body = build_body(model, prompt, think)
for rep in range(repeats):
started = time.perf_counter()
data = call(client, body)
elapsed_ms = (time.perf_counter() - started) * 1000
if data is None:
rows.append({
"id": fixture["id"], "group": fixture["group"], "rep": rep,
"error": True, "exact": False, "under_routed": False, "over_routed": False,
})
continue
text = data.get("response", "") or ""
found = _extract_capabilities(text)
rows.append({
"id": fixture["id"],
"group": fixture["group"],
"rep": rep,
"error": False,
"latency_ms": round(elapsed_ms, 1),
"eval_tokens": data.get("eval_count"),
"prompt_tokens": data.get("prompt_eval_count"),
# Did the model obey the documented output shape at all?
"has_delegate_line": bool(_DELEGATE_LINE_RE.search(text)),
# Whether reasoning came back, as opposed to being generated and dropped.
"thinking_returned": bool(data.get("thinking")),
"response_chars": len(text),
**score(fixture, found),
})
return rows
def summarise(rows: list[dict[str, Any]]) -> dict[str, Any]:
ok = [r for r in rows if not r["error"]]
if not ok:
return {"n": 0, "errors": len(rows)}
latencies = [r["latency_ms"] for r in ok]
tokens = [r["eval_tokens"] for r in ok if r["eval_tokens"] is not None]
return {
"n": len(ok),
"errors": len(rows) - len(ok),
"exact_pct": round(100 * sum(r["exact"] for r in ok) / len(ok), 1),
"under_routed_pct": round(100 * sum(r["under_routed"] for r in ok) / len(ok), 1),
"over_routed_pct": round(100 * sum(r["over_routed"] for r in ok) / len(ok), 1),
"format_ok_pct": round(100 * sum(r["has_delegate_line"] for r in ok) / len(ok), 1),
"thinking_returned_pct": round(100 * sum(r["thinking_returned"] for r in ok) / len(ok), 1),
"latency_ms_median": round(statistics.median(latencies), 1),
"latency_ms_mean": round(statistics.fmean(latencies), 1),
"eval_tokens_median": round(statistics.median(tokens), 1) if tokens else None,
"eval_tokens_total": sum(tokens) if tokens else None,
}
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--models", default=",".join(DEFAULT_MODELS))
parser.add_argument("--think", default=",".join(DEFAULT_THINK),
help="comma-separated subset of unset,true,false")
parser.add_argument("--repeats", type=int, default=1)
parser.add_argument("--timeout", type=float, default=180.0)
args = parser.parse_args()
models = [m.strip() for m in args.models.split(",") if m.strip()]
think_modes = [t.strip() for t in args.think.split(",") if t.strip()]
# build_steward_prompt reads the registry, and the registry is populated at
# application startup. Without this the prompt lists no capabilities and every
# cell scores zero for reasons that have nothing to do with the model.
register_household_members()
print(f"{len(FIXTURES)} fixtures x {len(models)} models x {len(think_modes)} think "
f"x {args.repeats} repeats = {len(FIXTURES) * len(models) * len(think_modes) * args.repeats} calls\n")
cells: dict[str, Any] = {}
with httpx.Client(timeout=args.timeout) as client:
for model in models:
# Absorb the cold load (~36s) outside the measurements.
print(f"warming {model} ...", flush=True)
call(client, build_body(model, "hi", "false"))
for think in think_modes:
key = f"{model}|think={think}"
print(f" {key} ...", end=" ", flush=True)
started = time.perf_counter()
rows = run_cell(client, model, think, args.repeats)
summary = summarise(rows)
cells[key] = {"summary": summary, "rows": rows}
print(f"exact={summary.get('exact_pct')}% "
f"over={summary.get('over_routed_pct')}% "
f"median={summary.get('latency_ms_median')}ms "
f"({time.perf_counter() - started:.0f}s)")
RESULTS_DIR.mkdir(parents=True, exist_ok=True)
stamp = datetime.now(UTC).strftime("%Y%m%dT%H%M%SZ")
out = RESULTS_DIR / f"routing-bench-{stamp}.json"
out.write_text(json.dumps({
"generated_at": datetime.now(UTC).isoformat(),
"fixtures": len(FIXTURES),
"repeats": args.repeats,
"cells": cells,
}, indent=2))
print(f"\n{'cell':28} {'exact':>7} {'under':>7} {'over':>7} {'fmt':>6} {'tok':>7} {'ms':>8}")
print("-" * 76)
for key, cell in cells.items():
s = cell["summary"]
print(f"{key:28} {s.get('exact_pct'):>6}% {s.get('under_routed_pct'):>6}% "
f"{s.get('over_routed_pct'):>6}% {s.get('format_ok_pct'):>5}% "
f"{str(s.get('eval_tokens_median')):>7} {s.get('latency_ms_median'):>8}")
print(f"\nwritten to {out}")
return 0
if __name__ == "__main__":
sys.exit(main())
View File
+159
View File
@@ -0,0 +1,159 @@
"""
Labelled queries for the Steward routing benchmark.
Each fixture carries both `expect` and `forbid`:
expect capabilities that must appear. Missing one is under-routing — the
Butler answers without a tool it needed.
forbid capabilities that must not appear. Over-routing is not cosmetic: a
spurious librarian is a real multi-second web call, and a spurious
housekeeper can actuate hardware.
`forbid` matters more than `expect` here, because over-recommendation is the
predicted failure when model thinking is disabled and the Steward has less room
to discriminate.
The `adversarial` group deserves explanation. Until 2026-08-08 the extractor
substring-matched capability *domains* across the Steward's whole response, so
ordinary English in its REASON line selected agents: "description" contains the
housekeeper domain "script", "acknowledge" contains "knowledge" and "know",
"economy" contains the biographer domain "my". Those queries invite exactly that
vocabulary. They now serve as an end-to-end regression: routing must depend on
what the Steward *decided*, not on the words it happened to use while explaining.
Expectations follow the routing rules stated in the Steward prompt itself
(src/agents/steward/agent.py), not on what a capability could plausibly cover.
"""
CORE = "tatlock_core"
LIB = "librarian"
BIO = "biographer"
HOUSE = "housekeeper"
ALL = [CORE, LIB, BIO, HOUSE]
def _others(*keep: str) -> list[str]:
return [c for c in ALL if c not in keep]
FIXTURES: list[dict] = [
# --- arithmetic and computation -> tatlock_core --------------------------
{"id": "math_add", "group": "math", "query": "What is 61 plus 12?",
"expect": [CORE], "forbid": _others(CORE)},
{"id": "math_percent", "group": "math", "query": "What is 15% of 240?",
"expect": [CORE], "forbid": _others(CORE)},
{"id": "math_compound", "group": "math", "query": "If I save 200 a month for 3 years, how much is that?",
"expect": [CORE], "forbid": _others(CORE)},
{"id": "math_sqrt", "group": "math", "query": "What is the square root of 1764?",
"expect": [CORE], "forbid": _others(CORE)},
# --- date and time -> tatlock_core ---------------------------------------
{"id": "time_now", "group": "datetime", "query": "What time is it?",
"expect": [CORE], "forbid": _others(CORE)},
{"id": "time_date", "group": "datetime", "query": "What is today's date?",
"expect": [CORE], "forbid": _others(CORE)},
{"id": "time_delta", "group": "datetime", "query": "How many days until Christmas?",
"expect": [CORE], "forbid": _others(CORE)},
# --- personal memory -> biographer ---------------------------------------
{"id": "bio_location", "group": "biographer", "query": "Where do I live?",
"expect": [BIO], "forbid": [LIB, HOUSE]},
{"id": "bio_name", "group": "biographer", "query": "What's my name?",
"expect": [BIO], "forbid": [LIB, HOUSE]},
{"id": "bio_car", "group": "biographer", "query": "What car do I drive?",
"expect": [BIO], "forbid": [LIB, HOUSE]},
{"id": "bio_store", "group": "biographer", "query": "Remember that I prefer my coffee black.",
"expect": [BIO], "forbid": [LIB, HOUSE]},
{"id": "bio_list", "group": "biographer", "query": "What do you know about me?",
"expect": [BIO], "forbid": [LIB, HOUSE]},
{"id": "bio_forget", "group": "biographer", "query": "Forget my old address.",
"expect": [BIO], "forbid": [LIB, HOUSE]},
# --- research and current information -> librarian ------------------------
{"id": "lib_weather", "group": "librarian", "query": "What's the weather in Rotterdam tomorrow?",
"expect": [LIB], "forbid": [HOUSE]},
{"id": "lib_news", "group": "librarian", "query": "What's in the news today?",
"expect": [LIB], "forbid": [HOUSE, BIO]},
{"id": "lib_url", "group": "librarian", "query": "Read https://example.com/article and summarise it.",
"expect": [LIB], "forbid": [HOUSE, BIO]},
{"id": "lib_research", "group": "librarian", "query": "Research how tidal power stations work.",
"expect": [LIB], "forbid": [HOUSE, BIO]},
{"id": "lib_wiki_create", "group": "librarian", "query": "Create a wiki page about our network topology.",
"expect": [LIB], "forbid": [HOUSE, BIO]},
# --- home automation -> housekeeper --------------------------------------
{"id": "house_lights_on", "group": "housekeeper", "query": "Turn on the kitchen lights.",
"expect": [HOUSE], "forbid": [LIB, BIO, CORE]},
{"id": "house_lights_off", "group": "housekeeper", "query": "Switch off all the lights downstairs.",
"expect": [HOUSE], "forbid": [LIB, BIO, CORE]},
{"id": "house_thermostat", "group": "housekeeper", "query": "Set the thermostat to 20 degrees.",
"expect": [HOUSE], "forbid": [LIB, BIO]},
{"id": "house_blinds", "group": "housekeeper", "query": "Close the blinds in the living room.",
"expect": [HOUSE], "forbid": [LIB, BIO, CORE]},
# --- conversational -> nothing at all -------------------------------------
# The expensive failure mode: a greeting that triggers a web search.
{"id": "chat_greeting", "group": "conversational", "query": "Hello!",
"expect": [], "forbid": ALL},
{"id": "chat_thanks", "group": "conversational", "query": "Thanks, that's helpful.",
"expect": [], "forbid": ALL},
{"id": "chat_joke", "group": "conversational", "query": "Tell me a joke.",
"expect": [], "forbid": ALL},
{"id": "chat_howareyou", "group": "conversational", "query": "How are you doing today?",
"expect": [], "forbid": ALL},
{"id": "chat_prior_turn", "group": "conversational", "query": "What did I just say?",
"expect": [], "forbid": ALL},
# --- genuinely multi-capability -------------------------------------------
{"id": "multi_weather_home", "group": "multi",
"query": "What's the weather here, and remember that I like it warm?",
"expect": [LIB, BIO], "forbid": []},
{"id": "multi_recall_search", "group": "multi",
"query": "Look up the best route from my home address to Utrecht.",
"expect": [BIO, LIB], "forbid": []},
{"id": "multi_math_memory", "group": "multi",
"query": "Remember that my budget is 500 euro, then work out 12% of it.",
"expect": [BIO, CORE], "forbid": [LIB, HOUSE]},
# --- adversarial: vocabulary that used to select agents by substring ------
# "temperature" is a housekeeper domain, but this is a unit conversion.
{"id": "adv_temperature", "group": "adversarial", "query": "Convert 98.6 Fahrenheit to Celsius.",
"expect": [CORE], "forbid": [HOUSE, LIB, BIO]},
# "description" contains "script"; "discover" contains "cover".
{"id": "adv_description", "group": "adversarial",
"query": "Give me a short description of what 17 times 23 comes to.",
"expect": [CORE], "forbid": [HOUSE, LIB]},
# "acknowledge" contains "knowledge" and "know".
{"id": "adv_acknowledge", "group": "adversarial",
"query": "Just acknowledge this and add 5 and 6 for me.",
"expect": [CORE], "forbid": [LIB, BIO]},
# "my" appears inside "economy".
{"id": "adv_economy", "group": "adversarial",
"query": "How many zeros are in one trillion?",
"expect": [CORE], "forbid": [BIO, HOUSE]},
# "fan" inside "fantastic"; also a climate word without a home-control intent.
{"id": "adv_fantastic", "group": "adversarial",
"query": "That's fantastic. What is 8 squared?",
"expect": [CORE], "forbid": [HOUSE, LIB]},
# "home" without any actuation intent.
{"id": "adv_home_word", "group": "adversarial", "query": "What time do I usually get home?",
"expect": [BIO], "forbid": [HOUSE]},
# "search" as ordinary English, not a web-search request.
{"id": "adv_search_word", "group": "adversarial",
"query": "No need to search anything, just tell me what 9 times 9 is.",
"expect": [CORE], "forbid": [LIB]},
# "create"/"write" are librarian domains but this is conversational.
{"id": "adv_write_word", "group": "adversarial", "query": "Can you write that more simply?",
"expect": [], "forbid": [LIB, HOUSE]},
# --- mutating intents: routing only, nothing is ever executed -------------
{"id": "mutate_wiki_update", "group": "mutating", "query": "Update the dossier page with today's findings.",
"expect": [LIB], "forbid": [HOUSE, CORE]},
{"id": "mutate_scene", "group": "mutating", "query": "Run the movie night scene.",
"expect": [HOUSE], "forbid": [LIB, BIO, CORE]},
]
GROUPS = sorted({f["group"] for f in FIXTURES})
assert len({f["id"] for f in FIXTURES}) == len(FIXTURES), "duplicate fixture id"