The provider assigned self._openai_client, an attribute nobody reads — OllamaProvider.client serves self._client — so the null-content sanitizer, the wrapper signal reading and the OLLAMA_TIMEOUT cap were all silently dead in production (sessions unaffected; they ride ModelSettings). Found via webber's identical bug on 2026-09-13. Three fixes travel together, because wiring alone would have traded a silent bypass for a loud crash: the client goes through the official openai_client constructor parameter; the parent-chat lookup survives openai's property/cached_property descriptor change (this venv's 2.11.0 is the latter — the old hardcoded .fget would have raised AttributeError on the first wired completion; ported from webber's 2026-08-11 fix); and the balancing/compaction signals are read from the response body's model_extra inside create() — the httpx event-hook approach demonstrably never fires under the openai SDK. The tests now assert through provider.client, the property pydantic_ai actually reads. The old ones asserted on the dead attribute and stayed green for the entire bypass — a check whose subject is not the real thing cannot fail for the real reason (workspace D-24's shape). Three mutations shown to fail their tests: the dead-attribute wiring, the .fget-only lookup, and removing the signal read from the request path. 683 tests green; a live dev turn through the wired client answers normally. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
212 lines
7.8 KiB
Python
212 lines
7.8 KiB
Python
"""
|
|
PydanticAI provider for Ollama with message sanitization.
|
|
|
|
Ollama's OpenAI-compatible API rejects messages with `content: null`,
|
|
which PydanticAI sends for assistant messages that only contain tool calls.
|
|
This provider sanitizes messages to use empty strings instead of null.
|
|
|
|
History worth keeping (repo T-6): until 2026-09-13 the sanitized client
|
|
was never in the request path — the provider assigned
|
|
`self._openai_client`, an attribute nobody reads, while
|
|
`OllamaProvider.client` serves `self._client`. The sanitizer, the
|
|
wrapper-signal reading and the OLLAMA_TIMEOUT cap were all silently
|
|
dead (sessions were unaffected — they ride ModelSettings). webber hit
|
|
the identical bug the same day; both now wire the client through the
|
|
constructor's official `openai_client` parameter, and the signals are
|
|
read from the response body — an httpx event-hook variant demonstrably
|
|
never fires under the openai SDK.
|
|
"""
|
|
|
|
from typing import Any
|
|
|
|
from openai import AsyncOpenAI
|
|
from pydantic_ai.providers.ollama import OllamaProvider
|
|
|
|
from src.core.config import config
|
|
from src.core.logging_config import get_logger
|
|
|
|
logger = get_logger(__name__)
|
|
|
|
|
|
def _log_wrapper_signals(response: Any) -> None:
|
|
"""
|
|
Surface the boilerroom wrapper's per-response signals (repo T-4).
|
|
|
|
balancing names the sessions parked or restored to serve this
|
|
request (wrapper D-4 — never silent); compaction-due says a session
|
|
crossed its authored context budget (wrapper D-5 — a signal, not an
|
|
interrupt). Read from the parsed body's extra fields: the wrapper
|
|
injects them into non-streamed JSON answers and openai's pydantic
|
|
models retain unknown fields in `model_extra`. Absent on any other
|
|
backend, so this costs nothing there. Streams carry the signals in
|
|
headers only and go unlogged here; the steward's raw call reads its
|
|
own body.
|
|
"""
|
|
extra = getattr(response, "model_extra", None) or {}
|
|
balancing = extra.get("balancing")
|
|
if balancing:
|
|
logger.info("backend_balancing", moves=balancing)
|
|
if extra.get("compaction_due"):
|
|
logger.warning(
|
|
"backend_compaction_due",
|
|
hint="GET /sessions on the wrapper names the session over budget",
|
|
)
|
|
|
|
|
|
class TatlockOllamaProvider(OllamaProvider):
|
|
"""
|
|
Custom OllamaProvider with message sanitization for Tatlock agents.
|
|
|
|
Fixes the 'invalid message content type: <nil>' error that occurs
|
|
when assistant messages have `content: null` with tool calls.
|
|
"""
|
|
|
|
def __init__(self, base_url: str | None = None):
|
|
"""
|
|
Initialize provider with Ollama base URL.
|
|
|
|
Args:
|
|
base_url: Ollama API URL (defaults to config.OLLAMA_HOST/v1)
|
|
"""
|
|
if base_url is None:
|
|
clean_host = str(config.OLLAMA_HOST).rstrip("/")
|
|
base_url = f"{clean_host}/v1"
|
|
|
|
# The sanitized client goes through the official constructor
|
|
# parameter: the provider's `.client` property serves `_client`,
|
|
# and poking a lookalike attribute after super().__init__ had
|
|
# built its own client is how the sanitizer sat bypassed in
|
|
# production (T-6).
|
|
super().__init__(openai_client=_SanitizedAsyncOpenAI(base_url=base_url))
|
|
|
|
logger.debug(
|
|
"tatlock_ollama_provider_created",
|
|
base_url=base_url,
|
|
timeout=config.OLLAMA_TIMEOUT,
|
|
)
|
|
|
|
|
|
class _SanitizedAsyncOpenAI(AsyncOpenAI):
|
|
"""AsyncOpenAI client that sanitizes messages before sending."""
|
|
|
|
def __init__(self, **kwargs: Any):
|
|
# Ollama doesn't need an API key. Cap each LLM call at the
|
|
# configured Ollama timeout instead of the SDK default (~600s),
|
|
# so one stuck request cannot eat the whole delegation budget.
|
|
kwargs.setdefault("timeout", float(config.OLLAMA_TIMEOUT))
|
|
super().__init__(api_key="ollama", **kwargs)
|
|
|
|
@property
|
|
def chat(self) -> "_SanitizedChat": # type: ignore[override]
|
|
"""Return sanitized chat interface.
|
|
|
|
Deliberately incompatible with AsyncOpenAI.chat, which is a Chat
|
|
resource. Replacing it is the entire mechanism of this class;
|
|
typing it as the parent's Chat would describe an object this
|
|
class does not return. The suppression is on this member alone.
|
|
"""
|
|
return _SanitizedChat(self)
|
|
|
|
|
|
def _parent_chat(client: AsyncOpenAI) -> Any:
|
|
"""Get AsyncOpenAI's own `chat`, bypassing the subclass override.
|
|
|
|
openai has shipped `chat` as both a property (getter `.fget`) and a
|
|
functools.cached_property (getter `.func`) — this venv's 2.11.0 is
|
|
the latter, so the old hardcoded `.fget` access raised
|
|
AttributeError the moment anything touched `.chat`. It went
|
|
unnoticed here for the same reason the whole class did: the client
|
|
was never wired (T-6). Reading whichever getter the descriptor
|
|
exposes keeps this working across both shapes, and raises something
|
|
legible if openai adopts a third. Ported from webber's 2026-08-11
|
|
fix.
|
|
"""
|
|
descriptor = AsyncOpenAI.__dict__["chat"]
|
|
getter = getattr(descriptor, "func", None) or getattr(descriptor, "fget", None)
|
|
if getter is None: # pragma: no cover - defensive
|
|
raise TypeError(
|
|
f"AsyncOpenAI.chat is a {type(descriptor).__name__} with neither "
|
|
"'func' nor 'fget'; the sanitising wrapper needs updating"
|
|
)
|
|
return getter(client)
|
|
|
|
|
|
class _SanitizedChat:
|
|
"""Chat interface wrapper with sanitized completions."""
|
|
|
|
def __init__(self, client: _SanitizedAsyncOpenAI):
|
|
self._client = client
|
|
self._original_chat = _parent_chat(client)
|
|
|
|
@property
|
|
def completions(self) -> "_SanitizedCompletions":
|
|
"""Return sanitized completions interface."""
|
|
return _SanitizedCompletions(self._original_chat.completions)
|
|
|
|
|
|
class _SanitizedCompletions:
|
|
"""Completions wrapper that sanitizes messages before API calls."""
|
|
|
|
def __init__(self, original_completions: Any):
|
|
self._original = original_completions
|
|
|
|
async def create(self, **kwargs: Any) -> Any:
|
|
"""
|
|
Create chat completion with sanitized messages.
|
|
|
|
Converts `content: null` to `content: ""` in assistant messages
|
|
to prevent Ollama's 'invalid message content type: <nil>' error.
|
|
"""
|
|
if "messages" in kwargs:
|
|
kwargs["messages"] = _sanitize_messages(kwargs["messages"])
|
|
|
|
response = await self._original.create(**kwargs)
|
|
_log_wrapper_signals(response)
|
|
return response
|
|
|
|
|
|
def _sanitize_messages(messages: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
"""
|
|
Sanitize messages to fix null content issues.
|
|
|
|
When an assistant message has tool_calls but no text content,
|
|
PydanticAI sets content to None. Ollama rejects this.
|
|
We convert None to empty string.
|
|
|
|
Args:
|
|
messages: List of chat messages
|
|
|
|
Returns:
|
|
Sanitized messages with null content replaced by empty strings
|
|
"""
|
|
sanitized = []
|
|
for msg in messages:
|
|
msg_copy = dict(msg)
|
|
|
|
# Fix null content in ANY message: Ollama rejects content: null with
|
|
# "invalid message content type: <nil>". The tool-call-only assistant
|
|
# case is the common one, but gemma thinking-only turns produce
|
|
# assistant messages with null content and NO tool_calls, which
|
|
# previously slipped through and 400'd the whole agent run.
|
|
if "content" in msg_copy and msg_copy.get("content") is None:
|
|
msg_copy["content"] = ""
|
|
logger.debug(
|
|
"sanitized_null_content",
|
|
role=msg_copy.get("role"),
|
|
tool_call_count=len(msg_copy.get("tool_calls") or []),
|
|
)
|
|
|
|
sanitized.append(msg_copy)
|
|
|
|
return sanitized
|
|
|
|
|
|
def get_ollama_provider() -> TatlockOllamaProvider:
|
|
"""
|
|
Get a configured Ollama provider for PydanticAI agents.
|
|
|
|
Returns:
|
|
TatlockOllamaProvider configured with sanitization
|
|
"""
|
|
return TatlockOllamaProvider()
|