Files
tatlock/tests/contracts/test_service_contracts.py
jpmschweitzerandClaude Fable 5 5b141669ce feat(backend): T-4 — named sessions through the boilerroom wrapper
The flavor probe gains a third answer: the wrapper names itself on
/health, so behind OLLAMA_HOST tatlock now distinguishes boilerroom,
a bare llama-server, and Ollama. Through the wrapper each pipeline
phase is a named session with the decided eviction ranking —
tatlock-steward/-orchestrate/-synthesize at 40, librarian at 30,
lower parks sooner (webber will sit at 20; Open WebUI stays
session-less and can never evict anyone). Against a bare llama-server
the raw id_slot pins survive unchanged, Ollama gets neither, and an
unprobed flavor sends nothing rather than guessing — the backend
stays swappable by env alone. tool_choice through the wrapper follows
the llama-server rule, since that is who answers.

The wrapper's balancing and compaction-due signals are read
everywhere: an httpx response hook on the provider covers every
PydanticAI call, streams included, and the steward's raw call reads
the body extras. Acting on compaction_due is a future ticket — the
signal just must not pass silently.

Verified end to end against the live wrapper: the dev server probed
flavor=boilerroom, a full pipeline turn answered in 5.7 s, and
GET /sessions showed all three phase sessions resident at rank 40
with engine-reported occupancies. The demonstration also filled the
production slot map — session-less prod delegations would have 503d
— cleared by a wrapper restart and filed as boilerroom T-11 (sessions
need an exit). A latent test flaw surfaced too: the ollama
tool_choice test relied on the dev backend probing as ollama; it now
pins the flavor it claims to test.

27 selector tests (9 new), 677 total green; three mutations shown to
fail their tests (librarian rank, the wrapper branch, the no-nudge
set); the new wrapper contract class runs 10/10 against the live
boundary.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-09-12 16:13:29 +02:00

316 lines
12 KiB
Python

"""
Wire-level contract tests for external service boundaries.
Each test sends the raw request the application code sends (no client
wrappers, no mocks) and asserts on the response shape, so boundary
breakage is caught directly instead of surfacing as agent misbehavior.
Semantics:
- Service unreachable -> skip (an outage is not a contract violation)
- Service reachable but wrong response shape -> fail
Run with: make test-contracts
"""
import json
import httpx
import pytest
from src.core.config import config
OLLAMA = str(config.OLLAMA_HOST).rstrip("/")
EMBEDDINGS = str(config.EMBEDDING_HOST or config.OLLAMA_HOST).rstrip("/")
QDRANT = f"http://{config.QDRANT_HOST}:{config.QDRANT_PORT}"
SEARXNG = str(config.SEARXNG_HOST).rstrip("/")
CALCULATOR_TOOL = {
"type": "function",
"function": {
"name": "calculator",
"description": "Evaluate a math expression",
"parameters": {
"type": "object",
"properties": {"expression": {"type": "string"}},
"required": ["expression"],
},
},
}
async def _get_or_skip(url: str, service: str, timeout: float = 5.0) -> httpx.Response:
"""GET a URL, skipping the test if the service is unreachable."""
try:
async with httpx.AsyncClient(timeout=timeout) as client:
return await client.get(url)
except httpx.TransportError as e:
pytest.skip(f"{service} unreachable at {url}: {e}")
async def _post_or_skip(
url: str, service: str, payload: dict, timeout: float, headers: dict | None = None
) -> httpx.Response:
"""POST a payload, skipping the test if the service is unreachable."""
try:
async with httpx.AsyncClient(timeout=timeout) as client:
return await client.post(url, json=payload, headers=headers)
except httpx.TransportError as e:
pytest.skip(f"{service} unreachable at {url}: {e}")
@pytest.mark.contract
class TestLocalBackendContract:
"""Boundary: the local OpenAI-compatible LLM backend.
Ollama today, llama-server after the serving cutover — every request
here is pure /v1, which is the whole point: the same contract must hold
whichever serves it, so the backend is swappable by env alone.
"""
async def test_models_lists_configured_model(self):
# Mirrors check_ollama_health()
response = await _get_or_skip(f"{OLLAMA}/v1/models", "local-backend")
assert response.status_code == 200
names = [m.get("id", "") for m in response.json()["data"]]
model = config.OLLAMA_DEFAULT_MODEL
assert (
model in names or f"{model}:latest" in names
), f"{model} not served; available: {names}"
async def test_chat_completion_returns_text(self):
# Mirrors StewardAgent._call_ollama()
response = await _post_or_skip(
f"{OLLAMA}/v1/chat/completions",
"local-backend",
{
"model": config.OLLAMA_DEFAULT_MODEL,
"messages": [{"role": "user", "content": "Reply with the single word: pong"}],
"temperature": 0.3,
"top_p": 0.9,
},
timeout=config.OLLAMA_TIMEOUT,
)
assert response.status_code == 200
assert response.json()["choices"][0]["message"]["content"].strip()
async def test_embeddings_shape(self):
# Mirrors OllamaEmbeddingClient.embed(): OpenAI shape, configured dim
response = await _post_or_skip(
f"{EMBEDDINGS}/v1/embeddings",
"embeddings-backend",
{
"model": config.OLLAMA_EMBEDDING_MODEL,
"input": "contract probe",
},
timeout=30.0,
)
assert response.status_code == 200
rows = response.json()["data"]
assert rows, "no embedding rows returned"
embedding = rows[0]["embedding"]
assert len(embedding) == config.QDRANT_EMBEDDING_DIM, (
f"dimension {len(embedding)} != configured {config.QDRANT_EMBEDDING_DIM} — "
"stored Qdrant vectors would be incompatible"
)
async def test_openai_compat_tool_calling(self):
# Mirrors the orchestration-phase request on llama-server: tools
# attached, NO tool_choice. The backend must call the tool unforced
# — "required" is deliberately absent because llama-server enforces
# it on every request in a run, which turns the tool loop
# unbreakable (observed at cutover: ~80 s turns).
response = await _post_or_skip(
f"{OLLAMA}/v1/chat/completions",
"local-backend",
{
"model": config.OLLAMA_DEFAULT_MODEL,
"messages": [{"role": "user", "content": "What is 6 * 7? Use the calculator."}],
"tools": [CALCULATOR_TOOL],
"stream": False,
},
timeout=config.OLLAMA_TIMEOUT,
)
assert response.status_code == 200
message = response.json()["choices"][0]["message"]
tool_calls = message.get("tool_calls")
assert tool_calls, f"model answered in text instead of calling the tool: {message}"
assert tool_calls[0]["function"]["name"] == "calculator"
arguments = json.loads(tool_calls[0]["function"]["arguments"])
assert "expression" in arguments
@pytest.mark.contract
class TestAnthropicContract:
"""Boundary: Anthropic Messages API (the Claude fallback backend)."""
HEADERS_KEY = "anthropic-version"
def _headers(self) -> dict:
if not config.ANTHROPIC_API_KEY:
pytest.skip("ANTHROPIC_API_KEY not configured")
return {
"x-api-key": config.ANTHROPIC_API_KEY,
"anthropic-version": "2023-06-01",
}
@staticmethod
def _skip_if_revoked(response: httpx.Response) -> None:
# The configured key is deliberately revoked (workspace D-11: the
# Claude migration is abandoned). A rejected credential means the
# fallback is disabled, not that the boundary broke.
if response.status_code == 401:
pytest.skip("Anthropic key rejected — fallback disabled per workspace D-11")
async def test_minimal_message_accepted(self):
# Mirrors check_claude_health(): tiny request, no sampling params
response = await _post_or_skip(
"https://api.anthropic.com/v1/messages",
"anthropic",
{
"model": config.ANTHROPIC_MODEL,
"max_tokens": 1,
"messages": [{"role": "user", "content": "hi"}],
},
timeout=30.0,
headers=self._headers(),
)
self._skip_if_revoked(response)
assert response.status_code == 200, response.text
async def test_temperature_rejected(self):
# Pins the Claude Sonnet 5+ contract that broke the Steward:
# sampling parameters are rejected with a 400 (and not billed).
response = await _post_or_skip(
"https://api.anthropic.com/v1/messages",
"anthropic",
{
"model": config.ANTHROPIC_MODEL,
"max_tokens": 1,
"messages": [{"role": "user", "content": "hi"}],
"temperature": 0.3,
},
timeout=30.0,
headers=self._headers(),
)
self._skip_if_revoked(response)
assert response.status_code == 400
assert "temperature" in response.text
@pytest.mark.contract
class TestQdrantContract:
"""Boundary: Qdrant REST API (Biographer's vector memory)."""
async def test_collections_endpoint(self):
response = await _get_or_skip(f"{QDRANT}/collections", "qdrant")
assert response.status_code == 200
assert "collections" in response.json()["result"]
@pytest.mark.contract
class TestSearxngContract:
"""Boundary: SearXNG JSON search API (web search tool)."""
async def test_json_search(self):
response = await _get_or_skip(
f"{SEARXNG}/search?q=test&format=json", "searxng", timeout=config.SEARXNG_TIMEOUT
)
assert response.status_code == 200
assert "results" in response.json()
@pytest.mark.contract
class TestLibraryDeskContract:
"""Boundary: library-desk research API (the Librarian's backend)."""
async def test_health(self):
host = getattr(config, "LIBRARY_DESK_HOST", None)
if not host:
pytest.skip("LIBRARY_DESK_HOST not configured")
response = await _get_or_skip(f"{str(host).rstrip('/')}/health", "library-desk")
assert response.status_code == 200
@pytest.mark.contract
class TestRedisContract:
"""Boundary: Redis on the configured memory DB."""
async def test_roundtrip(self):
import redis.asyncio as redis
client = redis.Redis(
host=config.REDIS_HOST,
port=config.REDIS_PORT,
db=config.REDIS_MEMORY_DB,
socket_connect_timeout=3,
)
try:
await client.ping()
except Exception as e:
pytest.skip(f"redis unreachable: {e}")
try:
await client.set("contract-test-key", "ok", ex=30)
assert await client.get("contract-test-key") == b"ok"
await client.delete("contract-test-key")
finally:
await client.aclose()
@pytest.mark.contract
class TestBoilerroomWrapperContract:
"""Boundary: the boilerroom wrapper surface session adoption rides on
(repo T-4). Skips when OLLAMA_HOST is not the wrapper — a direct
engine or Ollama is a valid deployment, not a violation.
"""
async def _skip_unless_wrapper(self) -> None:
response = await _get_or_skip(f"{OLLAMA}/health", "local-backend")
try:
service = response.json().get("service")
except ValueError:
service = None
if service != "boilerroom":
pytest.skip("OLLAMA_HOST is not the boilerroom wrapper")
async def test_wrapper_names_itself_on_health(self):
# Mirrors check_ollama_health()'s wrapper probe: the flavor
# detection keys on this exact field.
await self._skip_unless_wrapper()
async def test_props_passes_through_for_flavor_detection(self):
# tatlock treats a non-200 /props as Ollama and applies the
# tool_choice nudge — enforced by llama-server into a tool
# loop. The wrapper passing /props through is what keeps that
# detection safe (boilerroom T-10).
await self._skip_unless_wrapper()
response = await _get_or_skip(f"{OLLAMA}/props", "local-backend")
assert response.status_code == 200
async def test_session_fields_answer_with_signals(self):
# Mirrors phase_extra_body() through the wrapper: a session-
# carrying request must come back with the balancing header and
# the compaction signal in the body (wrapper D-3/D-4/D-5).
await self._skip_unless_wrapper()
response = await _post_or_skip(
f"{OLLAMA}/v1/chat/completions",
"local-backend",
{
"model": config.OLLAMA_DEFAULT_MODEL,
# No max_tokens: gemma4 spends its first tokens thinking,
# and a tight cap returns empty content (not a wrapper
# fault — the sibling passthrough test runs uncapped too).
"messages": [{"role": "user", "content": "Reply with the single word: pong"}],
"session": "contract-probe",
"eviction_order": 1,
},
timeout=config.OLLAMA_TIMEOUT,
)
assert response.status_code == 200
assert "x-boilerroom-balancing" in response.headers
assert response.headers.get("x-boilerroom-compaction-due") in ("true", "false")
body = response.json()
assert isinstance(body.get("compaction_due"), bool)
assert isinstance(body.get("balancing"), list)
# The extension fields must not leak into the engine's answer
assert body["choices"][0]["message"]["content"].strip()