The flavor probe gains a third answer: the wrapper names itself on /health, so behind OLLAMA_HOST tatlock now distinguishes boilerroom, a bare llama-server, and Ollama. Through the wrapper each pipeline phase is a named session with the decided eviction ranking — tatlock-steward/-orchestrate/-synthesize at 40, librarian at 30, lower parks sooner (webber will sit at 20; Open WebUI stays session-less and can never evict anyone). Against a bare llama-server the raw id_slot pins survive unchanged, Ollama gets neither, and an unprobed flavor sends nothing rather than guessing — the backend stays swappable by env alone. tool_choice through the wrapper follows the llama-server rule, since that is who answers. The wrapper's balancing and compaction-due signals are read everywhere: an httpx response hook on the provider covers every PydanticAI call, streams included, and the steward's raw call reads the body extras. Acting on compaction_due is a future ticket — the signal just must not pass silently. Verified end to end against the live wrapper: the dev server probed flavor=boilerroom, a full pipeline turn answered in 5.7 s, and GET /sessions showed all three phase sessions resident at rank 40 with engine-reported occupancies. The demonstration also filled the production slot map — session-less prod delegations would have 503d — cleared by a wrapper restart and filed as boilerroom T-11 (sessions need an exit). A latent test flaw surfaced too: the ollama tool_choice test relied on the dev backend probing as ollama; it now pins the flavor it claims to test. 27 selector tests (9 new), 677 total green; three mutations shown to fail their tests (librarian rank, the wrapper branch, the no-nudge set); the new wrapper contract class runs 10/10 against the live boundary. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
316 lines
12 KiB
Python
316 lines
12 KiB
Python
"""
|
|
Wire-level contract tests for external service boundaries.
|
|
|
|
Each test sends the raw request the application code sends (no client
|
|
wrappers, no mocks) and asserts on the response shape, so boundary
|
|
breakage is caught directly instead of surfacing as agent misbehavior.
|
|
|
|
Semantics:
|
|
- Service unreachable -> skip (an outage is not a contract violation)
|
|
- Service reachable but wrong response shape -> fail
|
|
|
|
Run with: make test-contracts
|
|
"""
|
|
|
|
import json
|
|
|
|
import httpx
|
|
import pytest
|
|
|
|
from src.core.config import config
|
|
|
|
OLLAMA = str(config.OLLAMA_HOST).rstrip("/")
|
|
EMBEDDINGS = str(config.EMBEDDING_HOST or config.OLLAMA_HOST).rstrip("/")
|
|
QDRANT = f"http://{config.QDRANT_HOST}:{config.QDRANT_PORT}"
|
|
SEARXNG = str(config.SEARXNG_HOST).rstrip("/")
|
|
|
|
CALCULATOR_TOOL = {
|
|
"type": "function",
|
|
"function": {
|
|
"name": "calculator",
|
|
"description": "Evaluate a math expression",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {"expression": {"type": "string"}},
|
|
"required": ["expression"],
|
|
},
|
|
},
|
|
}
|
|
|
|
|
|
async def _get_or_skip(url: str, service: str, timeout: float = 5.0) -> httpx.Response:
|
|
"""GET a URL, skipping the test if the service is unreachable."""
|
|
try:
|
|
async with httpx.AsyncClient(timeout=timeout) as client:
|
|
return await client.get(url)
|
|
except httpx.TransportError as e:
|
|
pytest.skip(f"{service} unreachable at {url}: {e}")
|
|
|
|
|
|
async def _post_or_skip(
|
|
url: str, service: str, payload: dict, timeout: float, headers: dict | None = None
|
|
) -> httpx.Response:
|
|
"""POST a payload, skipping the test if the service is unreachable."""
|
|
try:
|
|
async with httpx.AsyncClient(timeout=timeout) as client:
|
|
return await client.post(url, json=payload, headers=headers)
|
|
except httpx.TransportError as e:
|
|
pytest.skip(f"{service} unreachable at {url}: {e}")
|
|
|
|
|
|
@pytest.mark.contract
|
|
class TestLocalBackendContract:
|
|
"""Boundary: the local OpenAI-compatible LLM backend.
|
|
|
|
Ollama today, llama-server after the serving cutover — every request
|
|
here is pure /v1, which is the whole point: the same contract must hold
|
|
whichever serves it, so the backend is swappable by env alone.
|
|
"""
|
|
|
|
async def test_models_lists_configured_model(self):
|
|
# Mirrors check_ollama_health()
|
|
response = await _get_or_skip(f"{OLLAMA}/v1/models", "local-backend")
|
|
assert response.status_code == 200
|
|
names = [m.get("id", "") for m in response.json()["data"]]
|
|
model = config.OLLAMA_DEFAULT_MODEL
|
|
assert (
|
|
model in names or f"{model}:latest" in names
|
|
), f"{model} not served; available: {names}"
|
|
|
|
async def test_chat_completion_returns_text(self):
|
|
# Mirrors StewardAgent._call_ollama()
|
|
response = await _post_or_skip(
|
|
f"{OLLAMA}/v1/chat/completions",
|
|
"local-backend",
|
|
{
|
|
"model": config.OLLAMA_DEFAULT_MODEL,
|
|
"messages": [{"role": "user", "content": "Reply with the single word: pong"}],
|
|
"temperature": 0.3,
|
|
"top_p": 0.9,
|
|
},
|
|
timeout=config.OLLAMA_TIMEOUT,
|
|
)
|
|
assert response.status_code == 200
|
|
assert response.json()["choices"][0]["message"]["content"].strip()
|
|
|
|
async def test_embeddings_shape(self):
|
|
# Mirrors OllamaEmbeddingClient.embed(): OpenAI shape, configured dim
|
|
response = await _post_or_skip(
|
|
f"{EMBEDDINGS}/v1/embeddings",
|
|
"embeddings-backend",
|
|
{
|
|
"model": config.OLLAMA_EMBEDDING_MODEL,
|
|
"input": "contract probe",
|
|
},
|
|
timeout=30.0,
|
|
)
|
|
assert response.status_code == 200
|
|
rows = response.json()["data"]
|
|
assert rows, "no embedding rows returned"
|
|
embedding = rows[0]["embedding"]
|
|
assert len(embedding) == config.QDRANT_EMBEDDING_DIM, (
|
|
f"dimension {len(embedding)} != configured {config.QDRANT_EMBEDDING_DIM} — "
|
|
"stored Qdrant vectors would be incompatible"
|
|
)
|
|
|
|
async def test_openai_compat_tool_calling(self):
|
|
# Mirrors the orchestration-phase request on llama-server: tools
|
|
# attached, NO tool_choice. The backend must call the tool unforced
|
|
# — "required" is deliberately absent because llama-server enforces
|
|
# it on every request in a run, which turns the tool loop
|
|
# unbreakable (observed at cutover: ~80 s turns).
|
|
response = await _post_or_skip(
|
|
f"{OLLAMA}/v1/chat/completions",
|
|
"local-backend",
|
|
{
|
|
"model": config.OLLAMA_DEFAULT_MODEL,
|
|
"messages": [{"role": "user", "content": "What is 6 * 7? Use the calculator."}],
|
|
"tools": [CALCULATOR_TOOL],
|
|
"stream": False,
|
|
},
|
|
timeout=config.OLLAMA_TIMEOUT,
|
|
)
|
|
assert response.status_code == 200
|
|
message = response.json()["choices"][0]["message"]
|
|
tool_calls = message.get("tool_calls")
|
|
assert tool_calls, f"model answered in text instead of calling the tool: {message}"
|
|
assert tool_calls[0]["function"]["name"] == "calculator"
|
|
arguments = json.loads(tool_calls[0]["function"]["arguments"])
|
|
assert "expression" in arguments
|
|
|
|
|
|
@pytest.mark.contract
|
|
class TestAnthropicContract:
|
|
"""Boundary: Anthropic Messages API (the Claude fallback backend)."""
|
|
|
|
HEADERS_KEY = "anthropic-version"
|
|
|
|
def _headers(self) -> dict:
|
|
if not config.ANTHROPIC_API_KEY:
|
|
pytest.skip("ANTHROPIC_API_KEY not configured")
|
|
return {
|
|
"x-api-key": config.ANTHROPIC_API_KEY,
|
|
"anthropic-version": "2023-06-01",
|
|
}
|
|
|
|
@staticmethod
|
|
def _skip_if_revoked(response: httpx.Response) -> None:
|
|
# The configured key is deliberately revoked (workspace D-11: the
|
|
# Claude migration is abandoned). A rejected credential means the
|
|
# fallback is disabled, not that the boundary broke.
|
|
if response.status_code == 401:
|
|
pytest.skip("Anthropic key rejected — fallback disabled per workspace D-11")
|
|
|
|
async def test_minimal_message_accepted(self):
|
|
# Mirrors check_claude_health(): tiny request, no sampling params
|
|
response = await _post_or_skip(
|
|
"https://api.anthropic.com/v1/messages",
|
|
"anthropic",
|
|
{
|
|
"model": config.ANTHROPIC_MODEL,
|
|
"max_tokens": 1,
|
|
"messages": [{"role": "user", "content": "hi"}],
|
|
},
|
|
timeout=30.0,
|
|
headers=self._headers(),
|
|
)
|
|
self._skip_if_revoked(response)
|
|
assert response.status_code == 200, response.text
|
|
|
|
async def test_temperature_rejected(self):
|
|
# Pins the Claude Sonnet 5+ contract that broke the Steward:
|
|
# sampling parameters are rejected with a 400 (and not billed).
|
|
response = await _post_or_skip(
|
|
"https://api.anthropic.com/v1/messages",
|
|
"anthropic",
|
|
{
|
|
"model": config.ANTHROPIC_MODEL,
|
|
"max_tokens": 1,
|
|
"messages": [{"role": "user", "content": "hi"}],
|
|
"temperature": 0.3,
|
|
},
|
|
timeout=30.0,
|
|
headers=self._headers(),
|
|
)
|
|
self._skip_if_revoked(response)
|
|
assert response.status_code == 400
|
|
assert "temperature" in response.text
|
|
|
|
|
|
@pytest.mark.contract
|
|
class TestQdrantContract:
|
|
"""Boundary: Qdrant REST API (Biographer's vector memory)."""
|
|
|
|
async def test_collections_endpoint(self):
|
|
response = await _get_or_skip(f"{QDRANT}/collections", "qdrant")
|
|
assert response.status_code == 200
|
|
assert "collections" in response.json()["result"]
|
|
|
|
|
|
@pytest.mark.contract
|
|
class TestSearxngContract:
|
|
"""Boundary: SearXNG JSON search API (web search tool)."""
|
|
|
|
async def test_json_search(self):
|
|
response = await _get_or_skip(
|
|
f"{SEARXNG}/search?q=test&format=json", "searxng", timeout=config.SEARXNG_TIMEOUT
|
|
)
|
|
assert response.status_code == 200
|
|
assert "results" in response.json()
|
|
|
|
|
|
@pytest.mark.contract
|
|
class TestLibraryDeskContract:
|
|
"""Boundary: library-desk research API (the Librarian's backend)."""
|
|
|
|
async def test_health(self):
|
|
host = getattr(config, "LIBRARY_DESK_HOST", None)
|
|
if not host:
|
|
pytest.skip("LIBRARY_DESK_HOST not configured")
|
|
response = await _get_or_skip(f"{str(host).rstrip('/')}/health", "library-desk")
|
|
assert response.status_code == 200
|
|
|
|
|
|
@pytest.mark.contract
|
|
class TestRedisContract:
|
|
"""Boundary: Redis on the configured memory DB."""
|
|
|
|
async def test_roundtrip(self):
|
|
import redis.asyncio as redis
|
|
|
|
client = redis.Redis(
|
|
host=config.REDIS_HOST,
|
|
port=config.REDIS_PORT,
|
|
db=config.REDIS_MEMORY_DB,
|
|
socket_connect_timeout=3,
|
|
)
|
|
try:
|
|
await client.ping()
|
|
except Exception as e:
|
|
pytest.skip(f"redis unreachable: {e}")
|
|
try:
|
|
await client.set("contract-test-key", "ok", ex=30)
|
|
assert await client.get("contract-test-key") == b"ok"
|
|
await client.delete("contract-test-key")
|
|
finally:
|
|
await client.aclose()
|
|
|
|
|
|
@pytest.mark.contract
|
|
class TestBoilerroomWrapperContract:
|
|
"""Boundary: the boilerroom wrapper surface session adoption rides on
|
|
(repo T-4). Skips when OLLAMA_HOST is not the wrapper — a direct
|
|
engine or Ollama is a valid deployment, not a violation.
|
|
"""
|
|
|
|
async def _skip_unless_wrapper(self) -> None:
|
|
response = await _get_or_skip(f"{OLLAMA}/health", "local-backend")
|
|
try:
|
|
service = response.json().get("service")
|
|
except ValueError:
|
|
service = None
|
|
if service != "boilerroom":
|
|
pytest.skip("OLLAMA_HOST is not the boilerroom wrapper")
|
|
|
|
async def test_wrapper_names_itself_on_health(self):
|
|
# Mirrors check_ollama_health()'s wrapper probe: the flavor
|
|
# detection keys on this exact field.
|
|
await self._skip_unless_wrapper()
|
|
|
|
async def test_props_passes_through_for_flavor_detection(self):
|
|
# tatlock treats a non-200 /props as Ollama and applies the
|
|
# tool_choice nudge — enforced by llama-server into a tool
|
|
# loop. The wrapper passing /props through is what keeps that
|
|
# detection safe (boilerroom T-10).
|
|
await self._skip_unless_wrapper()
|
|
response = await _get_or_skip(f"{OLLAMA}/props", "local-backend")
|
|
assert response.status_code == 200
|
|
|
|
async def test_session_fields_answer_with_signals(self):
|
|
# Mirrors phase_extra_body() through the wrapper: a session-
|
|
# carrying request must come back with the balancing header and
|
|
# the compaction signal in the body (wrapper D-3/D-4/D-5).
|
|
await self._skip_unless_wrapper()
|
|
response = await _post_or_skip(
|
|
f"{OLLAMA}/v1/chat/completions",
|
|
"local-backend",
|
|
{
|
|
"model": config.OLLAMA_DEFAULT_MODEL,
|
|
# No max_tokens: gemma4 spends its first tokens thinking,
|
|
# and a tight cap returns empty content (not a wrapper
|
|
# fault — the sibling passthrough test runs uncapped too).
|
|
"messages": [{"role": "user", "content": "Reply with the single word: pong"}],
|
|
"session": "contract-probe",
|
|
"eviction_order": 1,
|
|
},
|
|
timeout=config.OLLAMA_TIMEOUT,
|
|
)
|
|
assert response.status_code == 200
|
|
assert "x-boilerroom-balancing" in response.headers
|
|
assert response.headers.get("x-boilerroom-compaction-due") in ("true", "false")
|
|
body = response.json()
|
|
assert isinstance(body.get("compaction_due"), bool)
|
|
assert isinstance(body.get("balancing"), list)
|
|
# The extension fields must not leak into the engine's answer
|
|
assert body["choices"][0]["message"]["content"].strip()
|