diff --git a/scripts/benchmark_routing.py b/scripts/benchmark_routing.py new file mode 100644 index 0000000..8b5876a --- /dev/null +++ b/scripts/benchmark_routing.py @@ -0,0 +1,221 @@ +""" +Benchmark Steward routing quality against model and thinking settings. + +Talks to Ollama directly. No Tatlock server, no agents, no tools, nothing is +executed — the mutating fixtures ("turn on the lights", "update the wiki") only +ever produce a routing decision. That makes this cheap and repeatable, and it +isolates the question: does the Steward still pick the right capabilities when +the model reasons less? + +The request body is byte-identical to StewardAgent._call_ollama, plus the +`think` flag under test, so a cell labelled `unset` is exactly what production +sends today. + +Three thinking settings, because "on vs off" hides the interesting case: + + unset what production sends now. gemma4 reasons by default, and the + response carries no `thinking` field, so those tokens are generated + and discarded. + true reasoning requested explicitly and returned in `thinking`. + false reasoning suppressed. + +Scoring is deliberately asymmetric. A missing capability under-routes and the +Butler answers without a tool it needed; a spurious one over-routes, and that is +a real agent call — a stray librarian is a multi-second web search on a query +that asked for arithmetic. Over-routing is the predicted failure when thinking +is off, so `forbid` violations are reported separately rather than folded into +one accuracy number. + +Usage: + .venv/bin/python scripts/benchmark_routing.py + .venv/bin/python scripts/benchmark_routing.py --models gemma4:e2b + .venv/bin/python scripts/benchmark_routing.py --think false --repeats 3 +""" +from __future__ import annotations + +import argparse +import json +import statistics +import sys +import time +from datetime import UTC, datetime +from pathlib import Path +from typing import Any + +import httpx + +PROJECT_ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(PROJECT_ROOT)) + +from scripts.fixtures.routing_fixtures import FIXTURES # noqa: E402 +from src.agents.steward.agent import build_steward_prompt # noqa: E402 +from src.agents.steward.service import _DELEGATE_LINE_RE, _extract_capabilities # noqa: E402 +from src.core.startup import register_household_members # noqa: E402 + +OLLAMA_URL = "http://localhost:11434" +DEFAULT_MODELS = ["gemma4:e2b", "gemma4:e4b"] +DEFAULT_THINK = ["unset", "true", "false"] +RESULTS_DIR = PROJECT_ROOT / "logs" + + +def build_body(model: str, prompt: str, think: str) -> dict[str, Any]: + """Mirror StewardAgent._call_ollama exactly, then add the flag under test.""" + body: dict[str, Any] = { + "model": model, + "prompt": prompt, + "stream": False, + "options": { + "temperature": 0.3, # Lower = more consistent + "top_p": 0.9, + }, + } + if think != "unset": + body["think"] = think == "true" + return body + + +def call(client: httpx.Client, body: dict[str, Any]) -> dict[str, Any] | None: + try: + response = client.post(f"{OLLAMA_URL}/api/generate", json=body) + response.raise_for_status() + return response.json() + except Exception as exc: # noqa: BLE001 - a failed cell must not abort the run + print(f" ! {exc}", file=sys.stderr) + return None + + +def score(fixture: dict, found: list[str]) -> dict[str, Any]: + expected = set(fixture["expect"]) + forbidden = set(fixture["forbid"]) + got = set(found) + missing = sorted(expected - got) + spurious = sorted(got & forbidden) + return { + "found": found, + "missing": missing, + "spurious": spurious, + # Exact only when everything expected arrived and nothing forbidden did. + "exact": not missing and not spurious, + "under_routed": bool(missing), + "over_routed": bool(spurious), + } + + +def run_cell(client: httpx.Client, model: str, think: str, repeats: int) -> list[dict[str, Any]]: + rows: list[dict[str, Any]] = [] + for fixture in FIXTURES: + prompt = build_steward_prompt(fixture["query"], []) + body = build_body(model, prompt, think) + for rep in range(repeats): + started = time.perf_counter() + data = call(client, body) + elapsed_ms = (time.perf_counter() - started) * 1000 + if data is None: + rows.append({ + "id": fixture["id"], "group": fixture["group"], "rep": rep, + "error": True, "exact": False, "under_routed": False, "over_routed": False, + }) + continue + + text = data.get("response", "") or "" + found = _extract_capabilities(text) + rows.append({ + "id": fixture["id"], + "group": fixture["group"], + "rep": rep, + "error": False, + "latency_ms": round(elapsed_ms, 1), + "eval_tokens": data.get("eval_count"), + "prompt_tokens": data.get("prompt_eval_count"), + # Did the model obey the documented output shape at all? + "has_delegate_line": bool(_DELEGATE_LINE_RE.search(text)), + # Whether reasoning came back, as opposed to being generated and dropped. + "thinking_returned": bool(data.get("thinking")), + "response_chars": len(text), + **score(fixture, found), + }) + return rows + + +def summarise(rows: list[dict[str, Any]]) -> dict[str, Any]: + ok = [r for r in rows if not r["error"]] + if not ok: + return {"n": 0, "errors": len(rows)} + latencies = [r["latency_ms"] for r in ok] + tokens = [r["eval_tokens"] for r in ok if r["eval_tokens"] is not None] + return { + "n": len(ok), + "errors": len(rows) - len(ok), + "exact_pct": round(100 * sum(r["exact"] for r in ok) / len(ok), 1), + "under_routed_pct": round(100 * sum(r["under_routed"] for r in ok) / len(ok), 1), + "over_routed_pct": round(100 * sum(r["over_routed"] for r in ok) / len(ok), 1), + "format_ok_pct": round(100 * sum(r["has_delegate_line"] for r in ok) / len(ok), 1), + "thinking_returned_pct": round(100 * sum(r["thinking_returned"] for r in ok) / len(ok), 1), + "latency_ms_median": round(statistics.median(latencies), 1), + "latency_ms_mean": round(statistics.fmean(latencies), 1), + "eval_tokens_median": round(statistics.median(tokens), 1) if tokens else None, + "eval_tokens_total": sum(tokens) if tokens else None, + } + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--models", default=",".join(DEFAULT_MODELS)) + parser.add_argument("--think", default=",".join(DEFAULT_THINK), + help="comma-separated subset of unset,true,false") + parser.add_argument("--repeats", type=int, default=1) + parser.add_argument("--timeout", type=float, default=180.0) + args = parser.parse_args() + + models = [m.strip() for m in args.models.split(",") if m.strip()] + think_modes = [t.strip() for t in args.think.split(",") if t.strip()] + + # build_steward_prompt reads the registry, and the registry is populated at + # application startup. Without this the prompt lists no capabilities and every + # cell scores zero for reasons that have nothing to do with the model. + register_household_members() + + print(f"{len(FIXTURES)} fixtures x {len(models)} models x {len(think_modes)} think " + f"x {args.repeats} repeats = {len(FIXTURES) * len(models) * len(think_modes) * args.repeats} calls\n") + + cells: dict[str, Any] = {} + with httpx.Client(timeout=args.timeout) as client: + for model in models: + # Absorb the cold load (~36s) outside the measurements. + print(f"warming {model} ...", flush=True) + call(client, build_body(model, "hi", "false")) + for think in think_modes: + key = f"{model}|think={think}" + print(f" {key} ...", end=" ", flush=True) + started = time.perf_counter() + rows = run_cell(client, model, think, args.repeats) + summary = summarise(rows) + cells[key] = {"summary": summary, "rows": rows} + print(f"exact={summary.get('exact_pct')}% " + f"over={summary.get('over_routed_pct')}% " + f"median={summary.get('latency_ms_median')}ms " + f"({time.perf_counter() - started:.0f}s)") + + RESULTS_DIR.mkdir(parents=True, exist_ok=True) + stamp = datetime.now(UTC).strftime("%Y%m%dT%H%M%SZ") + out = RESULTS_DIR / f"routing-bench-{stamp}.json" + out.write_text(json.dumps({ + "generated_at": datetime.now(UTC).isoformat(), + "fixtures": len(FIXTURES), + "repeats": args.repeats, + "cells": cells, + }, indent=2)) + + print(f"\n{'cell':28} {'exact':>7} {'under':>7} {'over':>7} {'fmt':>6} {'tok':>7} {'ms':>8}") + print("-" * 76) + for key, cell in cells.items(): + s = cell["summary"] + print(f"{key:28} {s.get('exact_pct'):>6}% {s.get('under_routed_pct'):>6}% " + f"{s.get('over_routed_pct'):>6}% {s.get('format_ok_pct'):>5}% " + f"{str(s.get('eval_tokens_median')):>7} {s.get('latency_ms_median'):>8}") + print(f"\nwritten to {out}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/fixtures/__init__.py b/scripts/fixtures/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/scripts/fixtures/routing_fixtures.py b/scripts/fixtures/routing_fixtures.py new file mode 100644 index 0000000..e92ff64 --- /dev/null +++ b/scripts/fixtures/routing_fixtures.py @@ -0,0 +1,159 @@ +""" +Labelled queries for the Steward routing benchmark. + +Each fixture carries both `expect` and `forbid`: + + expect capabilities that must appear. Missing one is under-routing — the + Butler answers without a tool it needed. + forbid capabilities that must not appear. Over-routing is not cosmetic: a + spurious librarian is a real multi-second web call, and a spurious + housekeeper can actuate hardware. + +`forbid` matters more than `expect` here, because over-recommendation is the +predicted failure when model thinking is disabled and the Steward has less room +to discriminate. + +The `adversarial` group deserves explanation. Until 2026-08-08 the extractor +substring-matched capability *domains* across the Steward's whole response, so +ordinary English in its REASON line selected agents: "description" contains the +housekeeper domain "script", "acknowledge" contains "knowledge" and "know", +"economy" contains the biographer domain "my". Those queries invite exactly that +vocabulary. They now serve as an end-to-end regression: routing must depend on +what the Steward *decided*, not on the words it happened to use while explaining. + +Expectations follow the routing rules stated in the Steward prompt itself +(src/agents/steward/agent.py), not on what a capability could plausibly cover. +""" + +CORE = "tatlock_core" +LIB = "librarian" +BIO = "biographer" +HOUSE = "housekeeper" +ALL = [CORE, LIB, BIO, HOUSE] + + +def _others(*keep: str) -> list[str]: + return [c for c in ALL if c not in keep] + + +FIXTURES: list[dict] = [ + # --- arithmetic and computation -> tatlock_core -------------------------- + {"id": "math_add", "group": "math", "query": "What is 61 plus 12?", + "expect": [CORE], "forbid": _others(CORE)}, + {"id": "math_percent", "group": "math", "query": "What is 15% of 240?", + "expect": [CORE], "forbid": _others(CORE)}, + {"id": "math_compound", "group": "math", "query": "If I save 200 a month for 3 years, how much is that?", + "expect": [CORE], "forbid": _others(CORE)}, + {"id": "math_sqrt", "group": "math", "query": "What is the square root of 1764?", + "expect": [CORE], "forbid": _others(CORE)}, + + # --- date and time -> tatlock_core --------------------------------------- + {"id": "time_now", "group": "datetime", "query": "What time is it?", + "expect": [CORE], "forbid": _others(CORE)}, + {"id": "time_date", "group": "datetime", "query": "What is today's date?", + "expect": [CORE], "forbid": _others(CORE)}, + {"id": "time_delta", "group": "datetime", "query": "How many days until Christmas?", + "expect": [CORE], "forbid": _others(CORE)}, + + # --- personal memory -> biographer --------------------------------------- + {"id": "bio_location", "group": "biographer", "query": "Where do I live?", + "expect": [BIO], "forbid": [LIB, HOUSE]}, + {"id": "bio_name", "group": "biographer", "query": "What's my name?", + "expect": [BIO], "forbid": [LIB, HOUSE]}, + {"id": "bio_car", "group": "biographer", "query": "What car do I drive?", + "expect": [BIO], "forbid": [LIB, HOUSE]}, + {"id": "bio_store", "group": "biographer", "query": "Remember that I prefer my coffee black.", + "expect": [BIO], "forbid": [LIB, HOUSE]}, + {"id": "bio_list", "group": "biographer", "query": "What do you know about me?", + "expect": [BIO], "forbid": [LIB, HOUSE]}, + {"id": "bio_forget", "group": "biographer", "query": "Forget my old address.", + "expect": [BIO], "forbid": [LIB, HOUSE]}, + + # --- research and current information -> librarian ------------------------ + {"id": "lib_weather", "group": "librarian", "query": "What's the weather in Rotterdam tomorrow?", + "expect": [LIB], "forbid": [HOUSE]}, + {"id": "lib_news", "group": "librarian", "query": "What's in the news today?", + "expect": [LIB], "forbid": [HOUSE, BIO]}, + {"id": "lib_url", "group": "librarian", "query": "Read https://example.com/article and summarise it.", + "expect": [LIB], "forbid": [HOUSE, BIO]}, + {"id": "lib_research", "group": "librarian", "query": "Research how tidal power stations work.", + "expect": [LIB], "forbid": [HOUSE, BIO]}, + {"id": "lib_wiki_create", "group": "librarian", "query": "Create a wiki page about our network topology.", + "expect": [LIB], "forbid": [HOUSE, BIO]}, + + # --- home automation -> housekeeper -------------------------------------- + {"id": "house_lights_on", "group": "housekeeper", "query": "Turn on the kitchen lights.", + "expect": [HOUSE], "forbid": [LIB, BIO, CORE]}, + {"id": "house_lights_off", "group": "housekeeper", "query": "Switch off all the lights downstairs.", + "expect": [HOUSE], "forbid": [LIB, BIO, CORE]}, + {"id": "house_thermostat", "group": "housekeeper", "query": "Set the thermostat to 20 degrees.", + "expect": [HOUSE], "forbid": [LIB, BIO]}, + {"id": "house_blinds", "group": "housekeeper", "query": "Close the blinds in the living room.", + "expect": [HOUSE], "forbid": [LIB, BIO, CORE]}, + + # --- conversational -> nothing at all ------------------------------------- + # The expensive failure mode: a greeting that triggers a web search. + {"id": "chat_greeting", "group": "conversational", "query": "Hello!", + "expect": [], "forbid": ALL}, + {"id": "chat_thanks", "group": "conversational", "query": "Thanks, that's helpful.", + "expect": [], "forbid": ALL}, + {"id": "chat_joke", "group": "conversational", "query": "Tell me a joke.", + "expect": [], "forbid": ALL}, + {"id": "chat_howareyou", "group": "conversational", "query": "How are you doing today?", + "expect": [], "forbid": ALL}, + {"id": "chat_prior_turn", "group": "conversational", "query": "What did I just say?", + "expect": [], "forbid": ALL}, + + # --- genuinely multi-capability ------------------------------------------- + {"id": "multi_weather_home", "group": "multi", + "query": "What's the weather here, and remember that I like it warm?", + "expect": [LIB, BIO], "forbid": []}, + {"id": "multi_recall_search", "group": "multi", + "query": "Look up the best route from my home address to Utrecht.", + "expect": [BIO, LIB], "forbid": []}, + {"id": "multi_math_memory", "group": "multi", + "query": "Remember that my budget is 500 euro, then work out 12% of it.", + "expect": [BIO, CORE], "forbid": [LIB, HOUSE]}, + + # --- adversarial: vocabulary that used to select agents by substring ------ + # "temperature" is a housekeeper domain, but this is a unit conversion. + {"id": "adv_temperature", "group": "adversarial", "query": "Convert 98.6 Fahrenheit to Celsius.", + "expect": [CORE], "forbid": [HOUSE, LIB, BIO]}, + # "description" contains "script"; "discover" contains "cover". + {"id": "adv_description", "group": "adversarial", + "query": "Give me a short description of what 17 times 23 comes to.", + "expect": [CORE], "forbid": [HOUSE, LIB]}, + # "acknowledge" contains "knowledge" and "know". + {"id": "adv_acknowledge", "group": "adversarial", + "query": "Just acknowledge this and add 5 and 6 for me.", + "expect": [CORE], "forbid": [LIB, BIO]}, + # "my" appears inside "economy". + {"id": "adv_economy", "group": "adversarial", + "query": "How many zeros are in one trillion?", + "expect": [CORE], "forbid": [BIO, HOUSE]}, + # "fan" inside "fantastic"; also a climate word without a home-control intent. + {"id": "adv_fantastic", "group": "adversarial", + "query": "That's fantastic. What is 8 squared?", + "expect": [CORE], "forbid": [HOUSE, LIB]}, + # "home" without any actuation intent. + {"id": "adv_home_word", "group": "adversarial", "query": "What time do I usually get home?", + "expect": [BIO], "forbid": [HOUSE]}, + # "search" as ordinary English, not a web-search request. + {"id": "adv_search_word", "group": "adversarial", + "query": "No need to search anything, just tell me what 9 times 9 is.", + "expect": [CORE], "forbid": [LIB]}, + # "create"/"write" are librarian domains but this is conversational. + {"id": "adv_write_word", "group": "adversarial", "query": "Can you write that more simply?", + "expect": [], "forbid": [LIB, HOUSE]}, + + # --- mutating intents: routing only, nothing is ever executed ------------- + {"id": "mutate_wiki_update", "group": "mutating", "query": "Update the dossier page with today's findings.", + "expect": [LIB], "forbid": [HOUSE, CORE]}, + {"id": "mutate_scene", "group": "mutating", "query": "Run the movie night scene.", + "expect": [HOUSE], "forbid": [LIB, BIO, CORE]}, +] + + +GROUPS = sorted({f["group"] for f in FIXTURES}) + +assert len({f["id"] for f in FIXTURES}) == len(FIXTURES), "duplicate fixture id"