release v2.6.0
The cutover's first real lesson ships: tool_choice "required" was a workaround for a backend that ignored it, and became an unbreakable tool loop on a backend that obeys it every request (~80 s arithmetic turns, observed). The health check now learns which server answers behind OLLAMA_HOST (/props is llama-server's own surface) and only Ollama gets the advisory nudge. Probed unforced on llama-server: 3/3 tool calls via the --jinja template. Also carries BACKEND_SLOT_PINNING (default off) for the P3 pilot. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -120,3 +120,20 @@ class TestWithSlotPinning:
|
||||
monkeypatch.setattr(config, "PREFER_CLOUD_BACKEND", True)
|
||||
base = model_selector.get_tool_choice_settings()
|
||||
assert model_selector.with_slot_pinning(base, slot=1) is base
|
||||
|
||||
|
||||
class TestLocalFlavorToolChoice:
|
||||
def test_llama_server_flavor_sends_no_tool_choice(self, local_first, monkeypatch):
|
||||
monkeypatch.setattr(model_selector, "_local_flavor", "llama-server")
|
||||
settings = model_selector.get_tool_choice_settings()
|
||||
assert not settings.get("extra_body")
|
||||
|
||||
def test_ollama_flavor_keeps_advisory_required(self, local_first, monkeypatch):
|
||||
monkeypatch.setattr(model_selector, "_local_flavor", "ollama")
|
||||
settings = model_selector.get_tool_choice_settings()
|
||||
assert settings["extra_body"] == {"tool_choice": "required"}
|
||||
|
||||
def test_unknown_flavor_defaults_to_ollama_semantics(self, local_first, monkeypatch):
|
||||
monkeypatch.setattr(model_selector, "_local_flavor", None)
|
||||
settings = model_selector.get_tool_choice_settings()
|
||||
assert settings["extra_body"] == {"tool_choice": "required"}
|
||||
|
||||
@@ -114,16 +114,18 @@ class TestLocalBackendContract:
|
||||
)
|
||||
|
||||
async def test_openai_compat_tool_calling(self):
|
||||
# Mirrors the request PydanticAI's OpenAIChatModel sends for the
|
||||
# orchestration phase, including the extra_body tool_choice.
|
||||
# Mirrors the orchestration-phase request on llama-server: tools
|
||||
# attached, NO tool_choice. The backend must call the tool unforced
|
||||
# — "required" is deliberately absent because llama-server enforces
|
||||
# it on every request in a run, which turns the tool loop
|
||||
# unbreakable (observed at cutover: ~80 s turns).
|
||||
response = await _post_or_skip(
|
||||
f"{OLLAMA}/v1/chat/completions",
|
||||
"ollama",
|
||||
"local-backend",
|
||||
{
|
||||
"model": config.OLLAMA_DEFAULT_MODEL,
|
||||
"messages": [{"role": "user", "content": "What is 6 * 7? Use the calculator."}],
|
||||
"tools": [CALCULATOR_TOOL],
|
||||
"tool_choice": "required",
|
||||
"stream": False,
|
||||
},
|
||||
timeout=config.OLLAMA_TIMEOUT,
|
||||
|
||||
Reference in New Issue
Block a user