diff --git a/.env.example b/.env.example index 552727a..00b7fd6 100644 --- a/.env.example +++ b/.env.example @@ -8,11 +8,16 @@ API_HOST=0.0.0.0 API_PORT=8000 API_PREFIX=/v1 -# Ollama Configuration (local - primary backend) +# Local backend (primary). The var name is legacy: any OpenAI-compatible +# server works — Ollama, a bare llama-server, or the boilerroom wrapper +# (http://127.0.0.1:8602 on tower-of-joy). The flavor is probed at +# startup; through the wrapper, BACKEND_SLOT_PINNING=true sends named +# sessions with the T-4 eviction ranking instead of raw slot pins. OLLAMA_HOST=http://localhost:11434 OLLAMA_DEFAULT_MODEL=gemma4:e2b OLLAMA_TIMEOUT=120 STEWARD_TIMEOUT=60 +BACKEND_SLOT_PINNING=false # Anthropic Configuration (Claude - cloud fallback) # Set ANTHROPIC_API_KEY to keep the Claude fallback available: it is used diff --git a/.pql/changelog/ticket_history/2026-09.sql b/.pql/changelog/ticket_history/2026-09.sql new file mode 100644 index 0000000..e95f541 --- /dev/null +++ b/.pql/changelog/ticket_history/2026-09.sql @@ -0,0 +1,2 @@ +INSERT INTO ticket_history (ticket_record_id, field, old_value, new_value, changed_by, changed_at, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06G99D07EM5DZWTS0WD9V8PSKM', 'status', 'backlog', 'in_progress', NULL, '2026-09-12 13:50:48', '2026-09-12 13:50:48.289', '2026-09-12 13:50:48.289', NULL, '2e88d155325b3e704787f48b2aa97591', 2) ON CONFLICT(hash) DO NOTHING; +INSERT INTO ticket_history (ticket_record_id, field, old_value, new_value, changed_by, changed_at, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06G99D07EM5DZWTS0WD9V8PSKM', 'status', 'in_progress', 'done', NULL, '2026-09-12 14:13:29', '2026-09-12 14:13:29.027', '2026-09-12 14:13:29.027', NULL, '5b9dd9a62002db0fb4dc58f7f9ffcbb6', 2) ON CONFLICT(hash) DO NOTHING; diff --git a/.pql/changelog/tickets/2026-09.sql b/.pql/changelog/tickets/2026-09.sql index ddf9287..e4cd4a1 100644 --- a/.pql/changelog/tickets/2026-09.sql +++ b/.pql/changelog/tickets/2026-09.sql @@ -1,3 +1,5 @@ INSERT INTO tickets (record_id, type, parent_record_id, title, description, status, priority, assigned_to, team, decision_ref, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06G94KC9XM3PW91C6B7VJA6VRM', 'story', NULL, 'Stream the synthesis phase — TTFT equals wall time today', 'Nothing streams until steward-orchestrate-synthesize completes, so desklock''s time-to-first-word is the full pipeline (TTFT mean 14.9s, max 26.2s — measured 2026-09-11, workspace note cladmin/agent/notes/2026-09-11-serving-baseline.md). Stream the synthesis phase out as it generates: a first-class goal of the serving rework independent of KV reuse (workspace D-60). The backend now streams /v1 SSE natively (llama-gen).', 'backlog', 'high', NULL, NULL, NULL, '2026-09-11 21:09:10.253', '2026-09-11 21:09:10.253', NULL, 'f7d31ca4fe747835a32744c310f51674', 2) ON CONFLICT(record_id) DO UPDATE SET type=excluded.type, parent_record_id=excluded.parent_record_id, title=excluded.title, description=excluded.description, status=excluded.status, priority=excluded.priority, assigned_to=excluded.assigned_to, team=excluded.team, decision_ref=excluded.decision_ref, updated_at=excluded.updated_at, deleted_at=excluded.deleted_at, hash=excluded.hash, canonical_version=excluded.canonical_version WHERE excluded.updated_at >= tickets.updated_at; INSERT INTO tickets (record_id, type, parent_record_id, title, description, status, priority, assigned_to, team, decision_ref, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06G99D07EM5DZWTS0WD9V8PSKM', 'story', NULL, 'Adopt boilerroom named sessions with the usage ranking', 'Replace raw id_slot pinning (BACKEND_SLOT_PINNING) with the wrapper''s session extension fields, now that tatlock routes through boilerroom:8602 (since 2026-09-12). Ranking by likelihood of use, decided 2026-09-12 — lower eviction_order parks sooner: tatlock pipeline phases highest (eviction_order 40; e.g. sessions tatlock-steward/tatlock-orchestrate/tatlock-synthesize), librarian-routed calls as session librarian (30), webber will sit at 20 (workspace T-137), Open WebUI stays session-less by design (idle slots only, can never evict — and it is being deprecated for an own-webui build). Read the x-boilerroom-balancing and compaction_due signals; the compaction endpoint expects the client transcript (tatlock''s Redis/Qdrant context is the natural producer — boilerroom D-5 names tatlock the first producer).', 'backlog', 'high', NULL, NULL, NULL, '2026-09-12 08:20:21.493', '2026-09-12 08:20:21.493', NULL, '3fb8c350876fc056648a19e0f2b7718c', 2) ON CONFLICT(record_id) DO UPDATE SET type=excluded.type, parent_record_id=excluded.parent_record_id, title=excluded.title, description=excluded.description, status=excluded.status, priority=excluded.priority, assigned_to=excluded.assigned_to, team=excluded.team, decision_ref=excluded.decision_ref, updated_at=excluded.updated_at, deleted_at=excluded.deleted_at, hash=excluded.hash, canonical_version=excluded.canonical_version WHERE excluded.updated_at >= tickets.updated_at; INSERT INTO tickets (record_id, type, parent_record_id, title, description, status, priority, assigned_to, team, decision_ref, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06G99D07Z3FZY8EBT9PJ1RCGHW', 'bug', NULL, 'The container logs nowhere: stdout empty, /app/logs mount untouched', 'docker logs tatlock is 0 lines and the mounted ~/docker-data/tatlock/logs has been empty since 2025-12 — the running app writes logs to neither, which made the 2026-09-12 wrapper-cutover verification blind (flavor detection had to be proven behaviorally). Dev mode logs to build/logs/server.log via make run; production should log to stdout for docker logs at minimum.', 'backlog', 'medium', NULL, NULL, NULL, '2026-09-12 08:20:21.624', '2026-09-12 08:20:21.624', NULL, '3fd294efc9afa857aef15810d9003298', 2) ON CONFLICT(record_id) DO UPDATE SET type=excluded.type, parent_record_id=excluded.parent_record_id, title=excluded.title, description=excluded.description, status=excluded.status, priority=excluded.priority, assigned_to=excluded.assigned_to, team=excluded.team, decision_ref=excluded.decision_ref, updated_at=excluded.updated_at, deleted_at=excluded.deleted_at, hash=excluded.hash, canonical_version=excluded.canonical_version WHERE excluded.updated_at >= tickets.updated_at; +INSERT INTO tickets (record_id, type, parent_record_id, title, description, status, priority, assigned_to, team, decision_ref, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06G99D07EM5DZWTS0WD9V8PSKM', 'story', NULL, 'Adopt boilerroom named sessions with the usage ranking', 'Replace raw id_slot pinning (BACKEND_SLOT_PINNING) with the wrapper''s session extension fields, now that tatlock routes through boilerroom:8602 (since 2026-09-12). Ranking by likelihood of use, decided 2026-09-12 — lower eviction_order parks sooner: tatlock pipeline phases highest (eviction_order 40; e.g. sessions tatlock-steward/tatlock-orchestrate/tatlock-synthesize), librarian-routed calls as session librarian (30), webber will sit at 20 (workspace T-137), Open WebUI stays session-less by design (idle slots only, can never evict — and it is being deprecated for an own-webui build). Read the x-boilerroom-balancing and compaction_due signals; the compaction endpoint expects the client transcript (tatlock''s Redis/Qdrant context is the natural producer — boilerroom D-5 names tatlock the first producer).', 'in_progress', 'high', NULL, NULL, NULL, '2026-09-12 08:20:21.493', '2026-09-12 13:50:48.289', NULL, '62abeb4da2f1f7aedb8d92884e7e148c', 2) ON CONFLICT(record_id) DO UPDATE SET type=excluded.type, parent_record_id=excluded.parent_record_id, title=excluded.title, description=excluded.description, status=excluded.status, priority=excluded.priority, assigned_to=excluded.assigned_to, team=excluded.team, decision_ref=excluded.decision_ref, updated_at=excluded.updated_at, deleted_at=excluded.deleted_at, hash=excluded.hash, canonical_version=excluded.canonical_version WHERE excluded.updated_at >= tickets.updated_at; +INSERT INTO tickets (record_id, type, parent_record_id, title, description, status, priority, assigned_to, team, decision_ref, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06G99D07EM5DZWTS0WD9V8PSKM', 'story', NULL, 'Adopt boilerroom named sessions with the usage ranking', 'Replace raw id_slot pinning (BACKEND_SLOT_PINNING) with the wrapper''s session extension fields, now that tatlock routes through boilerroom:8602 (since 2026-09-12). Ranking by likelihood of use, decided 2026-09-12 — lower eviction_order parks sooner: tatlock pipeline phases highest (eviction_order 40; e.g. sessions tatlock-steward/tatlock-orchestrate/tatlock-synthesize), librarian-routed calls as session librarian (30), webber will sit at 20 (workspace T-137), Open WebUI stays session-less by design (idle slots only, can never evict — and it is being deprecated for an own-webui build). Read the x-boilerroom-balancing and compaction_due signals; the compaction endpoint expects the client transcript (tatlock''s Redis/Qdrant context is the natural producer — boilerroom D-5 names tatlock the first producer).', 'done', 'high', NULL, NULL, NULL, '2026-09-12 08:20:21.493', '2026-09-12 14:13:29.027', NULL, '89ce1be323ebb26f00c63569f197cab1', 2) ON CONFLICT(record_id) DO UPDATE SET type=excluded.type, parent_record_id=excluded.parent_record_id, title=excluded.title, description=excluded.description, status=excluded.status, priority=excluded.priority, assigned_to=excluded.assigned_to, team=excluded.team, decision_ref=excluded.decision_ref, updated_at=excluded.updated_at, deleted_at=excluded.deleted_at, hash=excluded.hash, canonical_version=excluded.canonical_version WHERE excluded.updated_at >= tickets.updated_at; diff --git a/CHANGELOG.md b/CHANGELOG.md index 0623d76..4cbd3dc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,24 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Added + +- Named sessions through the boilerroom wrapper (T-4): the backend + flavor probe now recognizes the wrapper by its `/health` service + name, and each pipeline phase sends `session`/`eviction_order` + extension fields instead of raw `id_slot` — `tatlock-steward`, + `tatlock-orchestrate`, `tatlock-synthesize` at rank 40 and the + librarian at 30, per the 2026-09-12 usage ranking (lower parks + sooner; webber will sit at 20, Open WebUI stays session-less). + Against a bare llama-server the old slot pins still apply, and + Ollama gets neither, so the backend remains swappable by env alone. + The wrapper's `x-boilerroom-balancing` and compaction-due signals + are read on every response (an httpx hook on the provider, body + extras on the steward's raw call) and logged — acting on + `compaction_due` is a future ticket. `get_model_info()` reports the + probed `local_flavor`. Wire-pinned by a new boilerroom wrapper + contract class in `make test-contracts`. + ## [2.6.0] - 2026-09-11 ### Added diff --git a/src/agents/librarian/agent.py b/src/agents/librarian/agent.py index 65f7a9a..88386b8 100644 --- a/src/agents/librarian/agent.py +++ b/src/agents/librarian/agent.py @@ -272,11 +272,14 @@ async def run_librarian( ) try: + from src.anthropic.model_selector import with_phase_binding + # One shared library-desk connection for all tool calls in this run async with library_client_session(): result = await agent.run( prompt, message_history=message_history, + model_settings=with_phase_binding(None, "librarian"), ) logger.info( diff --git a/src/agents/steward/agent.py b/src/agents/steward/agent.py index 31e965c..1f98f1e 100644 --- a/src/agents/steward/agent.py +++ b/src/agents/steward/agent.py @@ -162,15 +162,17 @@ class StewardAgent: OpenAI-compatible server (Ollama, llama-server) can sit behind OLLAMA_HOST without this method knowing which. """ + from src.anthropic.model_selector import phase_extra_body + payload: dict = { "model": self.ollama_model, "messages": [{"role": "user", "content": prompt}], "temperature": 0.3, # Lower = more consistent "top_p": 0.9, } - if config.BACKEND_SLOT_PINNING: - # The steward owns engine slot 0 (see BACKEND_SLOT_PINNING) - payload["id_slot"] = 0 + # Session through the wrapper, id_slot 0 against a bare + # llama-server, nothing elsewhere (repo T-4). + payload.update(phase_extra_body("steward")) async with httpx.AsyncClient(timeout=self.timeout) as client: response = await client.post( @@ -180,6 +182,17 @@ class StewardAgent: response.raise_for_status() result = response.json() + + # The wrapper reports on session-carrying responses (its + # D-4/D-5); this raw call bypasses the provider hook, so + # the body extras are read here. Acting on compaction_due + # is a later ticket — the signal must not pass silently. + balancing = result.get("balancing") + if balancing: + logger.info("backend_balancing", session="tatlock-steward", moves=balancing) + if result.get("compaction_due"): + logger.warning("backend_compaction_due", session="tatlock-steward") + return result["choices"][0]["message"]["content"].strip() async def analyze(self, query: str, conversation_history: list[dict] | None = None) -> str: diff --git a/src/agents/tatlock.py b/src/agents/tatlock.py index 4b2ce08..2e4c605 100644 --- a/src/agents/tatlock.py +++ b/src/agents/tatlock.py @@ -536,13 +536,13 @@ class TatlockAgent(AgentInterface): # Run with scoped tools and tracker # Force tool_choice to make LLM actually call tools - from src.anthropic.model_selector import get_tool_choice_settings, with_slot_pinning + from src.anthropic.model_selector import get_tool_choice_settings, with_phase_binding result = await scoped_agent.run( enriched_message, message_history=pydantic_history if pydantic_history else None, deps=tool_tracker, - model_settings=with_slot_pinning(get_tool_choice_settings(), slot=1), + model_settings=with_phase_binding(get_tool_choice_settings(), "orchestrate"), ) logger.info( @@ -725,13 +725,13 @@ class TatlockAgent(AgentInterface): pydantic_history.append(ModelResponse(parts=[TextPart(content=content)])) # Run with scoped tools and tracker - from src.anthropic.model_selector import get_tool_choice_settings, with_slot_pinning + from src.anthropic.model_selector import get_tool_choice_settings, with_phase_binding result = await scoped_agent.run( enriched_message, message_history=pydantic_history if pydantic_history else None, deps=tool_tracker, - model_settings=with_slot_pinning(get_tool_choice_settings(), slot=1), + model_settings=with_phase_binding(get_tool_choice_settings(), "orchestrate"), ) # Extract tool calls and results from the agent's messages @@ -888,12 +888,12 @@ class TatlockAgent(AgentInterface): pydantic_history.append(ModelResponse(parts=[TextPart(content=content)])) # Run synthesis - from src.anthropic.model_selector import with_slot_pinning + from src.anthropic.model_selector import with_phase_binding result = await synthesis_agent.run( synthesis_prompt, message_history=pydantic_history if pydantic_history else None, - model_settings=with_slot_pinning(None, slot=2), + model_settings=with_phase_binding(None, "synthesize"), ) logger.info( diff --git a/src/anthropic/model_selector.py b/src/anthropic/model_selector.py index a9d005f..567d471 100644 --- a/src/anthropic/model_selector.py +++ b/src/anthropic/model_selector.py @@ -28,9 +28,11 @@ logger = get_logger(__name__) # Cached health check results (set once at startup) _claude_available: bool | None = None _ollama_available: bool | None = None -# Which server answers behind OLLAMA_HOST: "llama-server" or "ollama". -# They disagree on tool_choice semantics (see get_tool_choice_settings), -# so the health check probes /props — served by llama-server only — and +# Which server answers behind OLLAMA_HOST: "boilerroom", "llama-server" +# or "ollama". They disagree on tool_choice semantics and on which +# extension fields do anything (see get_tool_choice_settings and +# phase_extra_body), so the health check probes for the wrapper's name +# on /health, then for /props — served by llama-server only — and # caches the answer here. _local_flavor: str | None = None @@ -58,12 +60,26 @@ async def check_ollama_health() -> bool: response.raise_for_status() names = [m.get("id", "") for m in response.json().get("data", [])] - # /props is llama-server's own surface; Ollama 404s it. + # The boilerroom wrapper names itself on /health; a bare + # llama-server answers /health without a service field and + # serves /props (its own surface); Ollama answers neither. + is_wrapper = False try: - props = await client.get(f"{host}/props") - _local_flavor = "llama-server" if props.status_code == 200 else "ollama" - except httpx.HTTPError: - _local_flavor = "ollama" + health = await client.get(f"{host}/health") + is_wrapper = ( + health.status_code == 200 and health.json().get("service") == "boilerroom" + ) + except (httpx.HTTPError, ValueError): + is_wrapper = False + + if is_wrapper: + _local_flavor = "boilerroom" + else: + try: + props = await client.get(f"{host}/props") + _local_flavor = "llama-server" if props.status_code == 200 else "ollama" + except httpx.HTTPError: + _local_flavor = "ollama" if model in names or f"{model}:latest" in names: _ollama_available = True @@ -271,36 +287,75 @@ def get_tool_choice_settings() -> ModelSettings: observed at cutover). Its --jinja template renders tool definitions the way gemma4 was trained, and the model calls tools reliably unforced (probed 3/3). + + boilerroom: same as llama-server — the wrapper forwards to it, and + its /props passthrough is what lets this detection keep working. """ from pydantic_ai.settings import ModelSettings if resolve_backend() == "claude": # PydanticAI's Anthropic model handles tool_choice internally return ModelSettings() - if _local_flavor == "llama-server": + if _local_flavor in ("llama-server", "boilerroom"): return ModelSettings() # Ollama needs explicit tool_choice via extra_body return ModelSettings(extra_body={"tool_choice": "required"}) -def with_slot_pinning(settings: ModelSettings | None, slot: int) -> ModelSettings | None: - """ - Merge llama-server slot pinning into model settings when enabled. +# Phase bindings for the local backend (this repo's T-4, decided +# 2026-09-12): through the boilerroom wrapper each phase is a named +# session with an eviction rank — lower parks sooner, so the most-used +# context keeps its KV warm. The pipeline phases outrank the librarian, +# which outranks webber (20, filed in its own repo); Open WebUI stays +# session-less and can never evict anyone (wrapper D-4). +# phase -> (direct llama-server slot, wrapper session name, eviction rank) +_PHASE_BINDINGS: dict[str, tuple[int | None, str, int]] = { + "steward": (0, "tatlock-steward", 40), + "orchestrate": (1, "tatlock-orchestrate", 40), + "synthesize": (2, "tatlock-synthesize", 40), + "librarian": (None, "librarian", 30), +} - Each pipeline phase owns one engine slot (steward=0, orchestrator=1, - synthesizer=2), so the phase's stable prompt prefix stays in that - slot's KV cache and a turn re-prefills only its new tokens. Off by - default (BACKEND_SLOT_PINNING); a no-op on the Claude backend, and - Ollama ignores the field, so enabling it is safe on any backend. + +def phase_extra_body(phase: str) -> dict[str, str | int]: + """ + The extension fields a pipeline phase sends to the local backend. + + Through the boilerroom wrapper: a named session with its eviction + rank — the wrapper owns the name-to-slot map and reports parks and + restores on the response. Against a bare llama-server: the raw + id_slot pin this replaces (the librarian floats there, as it always + did). Ollama and Claude get nothing, and an unprobed flavor sends + nothing rather than guessing. Gated by BACKEND_SLOT_PINNING like + the pinning it grew out of. """ if not config.BACKEND_SLOT_PINNING or resolve_backend() == "claude": + return {} + slot, session, rank = _PHASE_BINDINGS[phase] + if _local_flavor == "boilerroom": + return {"session": session, "eviction_order": rank} + if _local_flavor == "llama-server" and slot is not None: + return {"id_slot": slot} + return {} + + +def with_phase_binding(settings: ModelSettings | None, phase: str) -> ModelSettings | None: + """ + Merge a phase's local-backend binding into model settings. + + Session fields through the wrapper, id_slot against a bare + llama-server, settings untouched everywhere else — one call site + stays correct on any backend (see phase_extra_body). + """ + extra = phase_extra_body(phase) + if not extra: return settings from pydantic_ai.settings import ModelSettings merged = dict(settings or {}) extra_body = dict(merged.get("extra_body") or {}) - extra_body["id_slot"] = slot + extra_body.update(extra) merged["extra_body"] = extra_body return ModelSettings(**merged) @@ -337,5 +392,6 @@ def get_model_info() -> dict: "claude_configured": bool(config.ANTHROPIC_API_KEY), "ollama_available": is_ollama_available(), "ollama_model": config.OLLAMA_DEFAULT_MODEL, + "local_flavor": _local_flavor, "prefer_cloud": config.PREFER_CLOUD_BACKEND, } diff --git a/src/ollama/provider.py b/src/ollama/provider.py index 95c0e1d..93b4298 100644 --- a/src/ollama/provider.py +++ b/src/ollama/provider.py @@ -8,6 +8,7 @@ This provider sanitizes messages to use empty strings instead of null. from typing import Any +import httpx from openai import AsyncOpenAI from pydantic_ai.providers.ollama import OllamaProvider @@ -17,6 +18,27 @@ from src.core.logging_config import get_logger logger = get_logger(__name__) +async def _log_wrapper_signals(response: httpx.Response) -> None: + """ + Surface the boilerroom wrapper's per-response signals (repo T-4). + + balancing names the sessions parked or restored to serve this + request (wrapper D-4 — never silent); compaction-due says a session + crossed its authored context budget (wrapper D-5 — a signal, not an + interrupt). Reading them is this ticket; producing the compacted + transcript is a later one. The headers are absent on any other + backend, so the hook costs nothing there. + """ + balancing = response.headers.get("x-boilerroom-balancing") + if balancing and balancing != "[]": + logger.info("backend_balancing", moves=balancing) + if response.headers.get("x-boilerroom-compaction-due") == "true": + logger.warning( + "backend_compaction_due", + hint="GET /sessions on the wrapper names the session over budget", + ) + + class TatlockOllamaProvider(OllamaProvider): """ Custom OllamaProvider with message sanitization for Tatlock agents. @@ -56,6 +78,15 @@ class _SanitizedAsyncOpenAI(AsyncOpenAI): # configured Ollama timeout instead of the SDK default (~600s), # so one stuck request cannot eat the whole delegation budget. kwargs.setdefault("timeout", float(config.OLLAMA_TIMEOUT)) + # The response hook reads the wrapper's balancing/compaction + # headers on every call this client makes, streams included. + kwargs.setdefault( + "http_client", + httpx.AsyncClient( + event_hooks={"response": [_log_wrapper_signals]}, + timeout=float(config.OLLAMA_TIMEOUT), + ), + ) super().__init__(api_key="ollama", **kwargs) @property diff --git a/tests/anthropic/test_model_selector.py b/tests/anthropic/test_model_selector.py index ebcc60f..8aec232 100644 --- a/tests/anthropic/test_model_selector.py +++ b/tests/anthropic/test_model_selector.py @@ -67,7 +67,10 @@ class TestGetModel: class TestToolChoiceSettings: - def test_ollama_forces_tool_choice(self, local_first): + def test_ollama_forces_tool_choice(self, local_first, monkeypatch): + # Pin the flavor: the session-scoped app init probes the real + # dev backend, and this test's answer must not depend on it. + monkeypatch.setattr(model_selector, "_local_flavor", "ollama") settings = model_selector.get_tool_choice_settings() assert settings.get("extra_body") == {"tool_choice": "required"} @@ -93,33 +96,89 @@ class TestGetModelInfo: assert info["model"] == config.ANTHROPIC_MODEL -class TestWithSlotPinning: +class TestPhaseBinding: def test_disabled_returns_settings_unchanged(self, local_first, monkeypatch): monkeypatch.setattr(config, "BACKEND_SLOT_PINNING", False) base = model_selector.get_tool_choice_settings() - assert model_selector.with_slot_pinning(base, slot=1) is base - assert model_selector.with_slot_pinning(None, slot=2) is None + assert model_selector.with_phase_binding(base, "orchestrate") is base + assert model_selector.with_phase_binding(None, "synthesize") is None - def test_enabled_merges_slot_with_existing_extra_body(self, local_first, monkeypatch): + def test_wrapper_flavor_sends_session_and_rank(self, local_first, monkeypatch): monkeypatch.setattr(config, "BACKEND_SLOT_PINNING", True) - settings = model_selector.with_slot_pinning( - model_selector.get_tool_choice_settings(), slot=1 + monkeypatch.setattr(model_selector, "_local_flavor", "boilerroom") + settings = model_selector.with_phase_binding( + model_selector.get_tool_choice_settings(), "orchestrate" ) extra_body = settings["extra_body"] - assert extra_body["id_slot"] == 1 - # tool_choice from the base settings survives the merge - assert extra_body["tool_choice"] == "required" + assert extra_body["session"] == "tatlock-orchestrate" + assert extra_body["eviction_order"] == 40 + assert "id_slot" not in extra_body - def test_enabled_pins_bare_settings(self, local_first, monkeypatch): + def test_wrapper_ranking_matches_the_decision(self, local_first, monkeypatch): + # Decided 2026-09-12: pipeline over librarian over webber (20, + # filed in its own repo); lower eviction_order parks sooner. monkeypatch.setattr(config, "BACKEND_SLOT_PINNING", True) - settings = model_selector.with_slot_pinning(None, slot=2) + monkeypatch.setattr(model_selector, "_local_flavor", "boilerroom") + ranks = { + phase: model_selector.phase_extra_body(phase)["eviction_order"] + for phase in ("steward", "orchestrate", "synthesize", "librarian") + } + assert ranks["steward"] == ranks["orchestrate"] == ranks["synthesize"] == 40 + assert ranks["librarian"] == 30 + assert ranks["librarian"] < ranks["steward"] + assert ranks["librarian"] > 20 # webber's rank stays below + + def test_wrapper_session_names_are_stable(self, local_first, monkeypatch): + # The wrapper's map keys on these; renaming one orphans its slot. + monkeypatch.setattr(config, "BACKEND_SLOT_PINNING", True) + monkeypatch.setattr(model_selector, "_local_flavor", "boilerroom") + names = { + phase: model_selector.phase_extra_body(phase)["session"] + for phase in ("steward", "orchestrate", "synthesize", "librarian") + } + assert names == { + "steward": "tatlock-steward", + "orchestrate": "tatlock-orchestrate", + "synthesize": "tatlock-synthesize", + "librarian": "librarian", + } + + def test_direct_llama_server_keeps_id_slot(self, local_first, monkeypatch): + monkeypatch.setattr(config, "BACKEND_SLOT_PINNING", True) + monkeypatch.setattr(model_selector, "_local_flavor", "llama-server") + settings = model_selector.with_phase_binding(None, "synthesize") assert settings["extra_body"] == {"id_slot": 2} - def test_claude_backend_never_pinned(self, local_first, monkeypatch): + def test_librarian_floats_on_direct_llama_server(self, local_first, monkeypatch): + monkeypatch.setattr(config, "BACKEND_SLOT_PINNING", True) + monkeypatch.setattr(model_selector, "_local_flavor", "llama-server") + assert model_selector.phase_extra_body("librarian") == {} + + def test_ollama_flavor_gets_no_fields(self, local_first, monkeypatch): + monkeypatch.setattr(config, "BACKEND_SLOT_PINNING", True) + monkeypatch.setattr(model_selector, "_local_flavor", "ollama") + assert model_selector.phase_extra_body("steward") == {} + + def test_unprobed_flavor_sends_nothing_rather_than_guessing(self, local_first, monkeypatch): + monkeypatch.setattr(config, "BACKEND_SLOT_PINNING", True) + monkeypatch.setattr(model_selector, "_local_flavor", None) + assert model_selector.phase_extra_body("orchestrate") == {} + + def test_merge_preserves_existing_extra_body(self, local_first, monkeypatch): + from pydantic_ai.settings import ModelSettings + + monkeypatch.setattr(config, "BACKEND_SLOT_PINNING", True) + monkeypatch.setattr(model_selector, "_local_flavor", "boilerroom") + base = ModelSettings(extra_body={"marker": 1}) + merged = model_selector.with_phase_binding(base, "steward") + assert merged["extra_body"]["marker"] == 1 + assert merged["extra_body"]["session"] == "tatlock-steward" + + def test_claude_backend_never_bound(self, local_first, monkeypatch): monkeypatch.setattr(config, "BACKEND_SLOT_PINNING", True) monkeypatch.setattr(config, "PREFER_CLOUD_BACKEND", True) base = model_selector.get_tool_choice_settings() - assert model_selector.with_slot_pinning(base, slot=1) is base + assert model_selector.with_phase_binding(base, "orchestrate") is base class TestLocalFlavorToolChoice: @@ -128,6 +187,13 @@ class TestLocalFlavorToolChoice: settings = model_selector.get_tool_choice_settings() assert not settings.get("extra_body") + def test_boilerroom_flavor_sends_no_tool_choice(self, local_first, monkeypatch): + # The wrapper forwards to llama-server, which enforces + # tool_choice — the nudge through it is the tool-loop incident. + monkeypatch.setattr(model_selector, "_local_flavor", "boilerroom") + settings = model_selector.get_tool_choice_settings() + assert not settings.get("extra_body") + def test_ollama_flavor_keeps_advisory_required(self, local_first, monkeypatch): monkeypatch.setattr(model_selector, "_local_flavor", "ollama") settings = model_selector.get_tool_choice_settings() diff --git a/tests/contracts/test_service_contracts.py b/tests/contracts/test_service_contracts.py index 0b6c72a..695d57c 100644 --- a/tests/contracts/test_service_contracts.py +++ b/tests/contracts/test_service_contracts.py @@ -254,3 +254,62 @@ class TestRedisContract: await client.delete("contract-test-key") finally: await client.aclose() + + +@pytest.mark.contract +class TestBoilerroomWrapperContract: + """Boundary: the boilerroom wrapper surface session adoption rides on + (repo T-4). Skips when OLLAMA_HOST is not the wrapper — a direct + engine or Ollama is a valid deployment, not a violation. + """ + + async def _skip_unless_wrapper(self) -> None: + response = await _get_or_skip(f"{OLLAMA}/health", "local-backend") + try: + service = response.json().get("service") + except ValueError: + service = None + if service != "boilerroom": + pytest.skip("OLLAMA_HOST is not the boilerroom wrapper") + + async def test_wrapper_names_itself_on_health(self): + # Mirrors check_ollama_health()'s wrapper probe: the flavor + # detection keys on this exact field. + await self._skip_unless_wrapper() + + async def test_props_passes_through_for_flavor_detection(self): + # tatlock treats a non-200 /props as Ollama and applies the + # tool_choice nudge — enforced by llama-server into a tool + # loop. The wrapper passing /props through is what keeps that + # detection safe (boilerroom T-10). + await self._skip_unless_wrapper() + response = await _get_or_skip(f"{OLLAMA}/props", "local-backend") + assert response.status_code == 200 + + async def test_session_fields_answer_with_signals(self): + # Mirrors phase_extra_body() through the wrapper: a session- + # carrying request must come back with the balancing header and + # the compaction signal in the body (wrapper D-3/D-4/D-5). + await self._skip_unless_wrapper() + response = await _post_or_skip( + f"{OLLAMA}/v1/chat/completions", + "local-backend", + { + "model": config.OLLAMA_DEFAULT_MODEL, + # No max_tokens: gemma4 spends its first tokens thinking, + # and a tight cap returns empty content (not a wrapper + # fault — the sibling passthrough test runs uncapped too). + "messages": [{"role": "user", "content": "Reply with the single word: pong"}], + "session": "contract-probe", + "eviction_order": 1, + }, + timeout=config.OLLAMA_TIMEOUT, + ) + assert response.status_code == 200 + assert "x-boilerroom-balancing" in response.headers + assert response.headers.get("x-boilerroom-compaction-due") in ("true", "false") + body = response.json() + assert isinstance(body.get("compaction_due"), bool) + assert isinstance(body.get("balancing"), list) + # The extension fields must not leak into the engine's answer + assert body["choices"][0]["message"]["content"].strip()