""" Wire-level contract tests for external service boundaries. Each test sends the raw request the application code sends (no client wrappers, no mocks) and asserts on the response shape, so boundary breakage is caught directly instead of surfacing as agent misbehavior. Semantics: - Service unreachable -> skip (an outage is not a contract violation) - Service reachable but wrong response shape -> fail Run with: make test-contracts """ import json import httpx import pytest from src.core.config import config OLLAMA = str(config.OLLAMA_HOST).rstrip("/") EMBEDDINGS = str(config.EMBEDDING_HOST or config.OLLAMA_HOST).rstrip("/") QDRANT = f"http://{config.QDRANT_HOST}:{config.QDRANT_PORT}" SEARXNG = str(config.SEARXNG_HOST).rstrip("/") CALCULATOR_TOOL = { "type": "function", "function": { "name": "calculator", "description": "Evaluate a math expression", "parameters": { "type": "object", "properties": {"expression": {"type": "string"}}, "required": ["expression"], }, }, } async def _get_or_skip(url: str, service: str, timeout: float = 5.0) -> httpx.Response: """GET a URL, skipping the test if the service is unreachable.""" try: async with httpx.AsyncClient(timeout=timeout) as client: return await client.get(url) except httpx.TransportError as e: pytest.skip(f"{service} unreachable at {url}: {e}") async def _post_or_skip( url: str, service: str, payload: dict, timeout: float, headers: dict | None = None ) -> httpx.Response: """POST a payload, skipping the test if the service is unreachable.""" try: async with httpx.AsyncClient(timeout=timeout) as client: return await client.post(url, json=payload, headers=headers) except httpx.TransportError as e: pytest.skip(f"{service} unreachable at {url}: {e}") @pytest.mark.contract class TestLocalBackendContract: """Boundary: the local OpenAI-compatible LLM backend. Ollama today, llama-server after the serving cutover — every request here is pure /v1, which is the whole point: the same contract must hold whichever serves it, so the backend is swappable by env alone. """ async def test_models_lists_configured_model(self): # Mirrors check_ollama_health() response = await _get_or_skip(f"{OLLAMA}/v1/models", "local-backend") assert response.status_code == 200 names = [m.get("id", "") for m in response.json()["data"]] model = config.OLLAMA_DEFAULT_MODEL assert ( model in names or f"{model}:latest" in names ), f"{model} not served; available: {names}" async def test_chat_completion_returns_text(self): # Mirrors StewardAgent._call_ollama() response = await _post_or_skip( f"{OLLAMA}/v1/chat/completions", "local-backend", { "model": config.OLLAMA_DEFAULT_MODEL, "messages": [{"role": "user", "content": "Reply with the single word: pong"}], "temperature": 0.3, "top_p": 0.9, }, timeout=config.OLLAMA_TIMEOUT, ) assert response.status_code == 200 assert response.json()["choices"][0]["message"]["content"].strip() async def test_embeddings_shape(self): # Mirrors OllamaEmbeddingClient.embed(): OpenAI shape, configured dim response = await _post_or_skip( f"{EMBEDDINGS}/v1/embeddings", "embeddings-backend", { "model": config.OLLAMA_EMBEDDING_MODEL, "input": "contract probe", }, timeout=30.0, ) assert response.status_code == 200 rows = response.json()["data"] assert rows, "no embedding rows returned" embedding = rows[0]["embedding"] assert len(embedding) == config.QDRANT_EMBEDDING_DIM, ( f"dimension {len(embedding)} != configured {config.QDRANT_EMBEDDING_DIM} — " "stored Qdrant vectors would be incompatible" ) async def test_openai_compat_tool_calling(self): # Mirrors the orchestration-phase request on llama-server: tools # attached, NO tool_choice. The backend must call the tool unforced # — "required" is deliberately absent because llama-server enforces # it on every request in a run, which turns the tool loop # unbreakable (observed at cutover: ~80 s turns). response = await _post_or_skip( f"{OLLAMA}/v1/chat/completions", "local-backend", { "model": config.OLLAMA_DEFAULT_MODEL, "messages": [{"role": "user", "content": "What is 6 * 7? Use the calculator."}], "tools": [CALCULATOR_TOOL], "stream": False, }, timeout=config.OLLAMA_TIMEOUT, ) assert response.status_code == 200 message = response.json()["choices"][0]["message"] tool_calls = message.get("tool_calls") assert tool_calls, f"model answered in text instead of calling the tool: {message}" assert tool_calls[0]["function"]["name"] == "calculator" arguments = json.loads(tool_calls[0]["function"]["arguments"]) assert "expression" in arguments @pytest.mark.contract class TestAnthropicContract: """Boundary: Anthropic Messages API (the Claude fallback backend).""" HEADERS_KEY = "anthropic-version" def _headers(self) -> dict: if not config.ANTHROPIC_API_KEY: pytest.skip("ANTHROPIC_API_KEY not configured") return { "x-api-key": config.ANTHROPIC_API_KEY, "anthropic-version": "2023-06-01", } @staticmethod def _skip_if_revoked(response: httpx.Response) -> None: # The configured key is deliberately revoked (workspace D-11: the # Claude migration is abandoned). A rejected credential means the # fallback is disabled, not that the boundary broke. if response.status_code == 401: pytest.skip("Anthropic key rejected — fallback disabled per workspace D-11") async def test_minimal_message_accepted(self): # Mirrors check_claude_health(): tiny request, no sampling params response = await _post_or_skip( "https://api.anthropic.com/v1/messages", "anthropic", { "model": config.ANTHROPIC_MODEL, "max_tokens": 1, "messages": [{"role": "user", "content": "hi"}], }, timeout=30.0, headers=self._headers(), ) self._skip_if_revoked(response) assert response.status_code == 200, response.text async def test_temperature_rejected(self): # Pins the Claude Sonnet 5+ contract that broke the Steward: # sampling parameters are rejected with a 400 (and not billed). response = await _post_or_skip( "https://api.anthropic.com/v1/messages", "anthropic", { "model": config.ANTHROPIC_MODEL, "max_tokens": 1, "messages": [{"role": "user", "content": "hi"}], "temperature": 0.3, }, timeout=30.0, headers=self._headers(), ) self._skip_if_revoked(response) assert response.status_code == 400 assert "temperature" in response.text @pytest.mark.contract class TestQdrantContract: """Boundary: Qdrant REST API (Biographer's vector memory).""" async def test_collections_endpoint(self): response = await _get_or_skip(f"{QDRANT}/collections", "qdrant") assert response.status_code == 200 assert "collections" in response.json()["result"] @pytest.mark.contract class TestSearxngContract: """Boundary: SearXNG JSON search API (web search tool).""" async def test_json_search(self): response = await _get_or_skip( f"{SEARXNG}/search?q=test&format=json", "searxng", timeout=config.SEARXNG_TIMEOUT ) assert response.status_code == 200 assert "results" in response.json() @pytest.mark.contract class TestLibraryDeskContract: """Boundary: library-desk research API (the Librarian's backend).""" async def test_health(self): host = getattr(config, "LIBRARY_DESK_HOST", None) if not host: pytest.skip("LIBRARY_DESK_HOST not configured") response = await _get_or_skip(f"{str(host).rstrip('/')}/health", "library-desk") assert response.status_code == 200 @pytest.mark.contract class TestRedisContract: """Boundary: Redis on the configured memory DB.""" async def test_roundtrip(self): import redis.asyncio as redis client = redis.Redis( host=config.REDIS_HOST, port=config.REDIS_PORT, db=config.REDIS_MEMORY_DB, socket_connect_timeout=3, ) try: await client.ping() except Exception as e: pytest.skip(f"redis unreachable: {e}") try: await client.set("contract-test-key", "ok", ex=30) assert await client.get("contract-test-key") == b"ok" await client.delete("contract-test-key") finally: await client.aclose() @pytest.mark.contract class TestBoilerroomWrapperContract: """Boundary: the boilerroom wrapper surface session adoption rides on (repo T-4). Skips when OLLAMA_HOST is not the wrapper — a direct engine or Ollama is a valid deployment, not a violation. """ async def _skip_unless_wrapper(self) -> None: response = await _get_or_skip(f"{OLLAMA}/health", "local-backend") try: service = response.json().get("service") except ValueError: service = None if service != "boilerroom": pytest.skip("OLLAMA_HOST is not the boilerroom wrapper") async def test_wrapper_names_itself_on_health(self): # Mirrors check_ollama_health()'s wrapper probe: the flavor # detection keys on this exact field. await self._skip_unless_wrapper() async def test_props_passes_through_for_flavor_detection(self): # tatlock treats a non-200 /props as Ollama and applies the # tool_choice nudge — enforced by llama-server into a tool # loop. The wrapper passing /props through is what keeps that # detection safe (boilerroom T-10). await self._skip_unless_wrapper() response = await _get_or_skip(f"{OLLAMA}/props", "local-backend") assert response.status_code == 200 async def test_session_fields_answer_with_signals(self): # Mirrors phase_extra_body() through the wrapper: a session- # carrying request must come back with the balancing header and # the compaction signal in the body (wrapper D-3/D-4/D-5). await self._skip_unless_wrapper() response = await _post_or_skip( f"{OLLAMA}/v1/chat/completions", "local-backend", { "model": config.OLLAMA_DEFAULT_MODEL, # No max_tokens: gemma4 spends its first tokens thinking, # and a tight cap returns empty content (not a wrapper # fault — the sibling passthrough test runs uncapped too). "messages": [{"role": "user", "content": "Reply with the single word: pong"}], "session": "contract-probe", "eviction_order": 1, }, timeout=config.OLLAMA_TIMEOUT, ) assert response.status_code == 200 assert "x-boilerroom-balancing" in response.headers assert response.headers.get("x-boilerroom-compaction-due") in ("true", "false") body = response.json() assert isinstance(body.get("compaction_due"), bool) assert isinstance(body.get("balancing"), list) # The extension fields must not leak into the engine's answer assert body["choices"][0]["message"]["content"].strip()