fix(llm): alias tool names that collide with gpt-oss built-ins (#5878)

gpt-oss (harmony) ships BUILT-IN tools named python/browser, invoked
with the raw body as the argument (to=python + bare source), while
custom functions use to=functions.NAME + JSON. Exposing our own tools
under those names makes the model answer with the built-in convention:
it emits raw code, the server parses it as JSON, and the request dies
with 'error parsing tool call: raw=import sys, ...'. In streaming mode
Ollama does not report it at all — it truncates the stream, so the turn
arrives as an empty response and the agent loop reads it as a model
stall. bash collides the same way in practice.

Measured on gpt-oss:20b via Ollama /v1, fixed agentic prompt, 12 runs
per arm: python+bash as-is 2/12, python renamed 10/12, both renamed
12/12. 74 HTTP 500s were logged server-side during investigation with
zero surfaced to the client.

Rename the colliding tools on the outbound payload and map the names
back on responses. Transport-only and gated on gpt-oss: every other
model's schemas pass through untouched (asserted in tests).

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
Co-authored-by: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com>
This commit is contained in:
talentlesshack
2026-08-12 02:01:52 +01:00
committed by GitHub
co-authored by Claude Fable 5 Alexandre Teixeira
parent d87a913729
commit adfe3ab379
2 changed files with 137 additions and 4 deletions
+59 -4
View File
@@ -1,6 +1,7 @@
# src/llm_core.py
import httpx
import asyncio
import copy
import time
import json
import logging
@@ -644,7 +645,7 @@ def _build_ollama_payload(
if options:
payload["options"] = options
if tools:
payload["tools"] = tools
payload["tools"] = _alias_harmony_tools(tools, model)
return payload
@@ -1055,6 +1056,57 @@ def _model_disallows_reasoning_effort_with_chat_tools(model: str) -> bool:
return bool(re.match(r"^(?:openai/)?gpt-5(?:[.\-]\d+)?(?:[-_:].*)?$", m))
# gpt-oss (harmony) ships BUILT-IN tools named `python` and `browser`, invoked
# with the raw body as the argument (`to=python` + bare source), while custom
# functions use `to=functions.NAME` + JSON. A tool we expose under a built-in's
# name therefore gets called with the built-in convention: the model emits raw
# code, the server tries to parse it as JSON, and the whole request dies
# ("error parsing tool call: raw='import sys, ...'"). In streaming mode Ollama
# does not even report it — it truncates the stream, so the turn looks like an
# empty response. `bash` collides the same way in practice.
#
# Measured on gpt-oss:20b via Ollama /v1 with a fixed agentic prompt:
# tools named python+bash ............ 2/6 succeeded (4 parse failures)
# python renamed ..................... 5/6
# python and bash renamed ............ 6/6
#
# So rename the colliding tools on the way out and map the names back on the
# way in. Confined to the transport layer: callers keep using the real names.
_HARMONY_TOOL_ALIASES = {
"python": "run_python_code",
"bash": "run_shell_command",
"browser": "web_browser_tool",
}
_HARMONY_TOOL_ALIASES_REVERSE = {v: k for k, v in _HARMONY_TOOL_ALIASES.items()}
def _is_harmony_model(model: str) -> bool:
"""True for gpt-oss / harmony-format models, which have built-in tool names."""
return "gpt-oss" in (model or "").lower()
def _alias_harmony_tools(tools: Optional[List[Dict]], model: str) -> Optional[List[Dict]]:
"""Rename tools that collide with harmony built-ins. Returns a copy."""
if not tools or not _is_harmony_model(model):
return tools
out = []
for t in tools:
fn = t.get("function") or {}
alias = _HARMONY_TOOL_ALIASES.get(fn.get("name"))
if alias:
t = copy.deepcopy(t)
t["function"]["name"] = alias
out.append(t)
return out
def _unalias_harmony_tool_name(name: str, model: str) -> str:
"""Map an aliased tool name in a model response back to the real name."""
if not _is_harmony_model(model):
return name
return _HARMONY_TOOL_ALIASES_REVERSE.get(name, name)
def _scrub_openai_chat_tool_reasoning(payload: Dict, target_url: str, model: str) -> None:
if not payload.get("tools"):
return
@@ -2236,7 +2288,7 @@ async def _stream_llm_inner(url: str, model: str, messages: List[Dict], temperat
tok_key = "max_completion_tokens" if _uses_max_completion_tokens(model) else "max_tokens"
payload[tok_key] = max_tokens
if tools:
payload["tools"] = tools
payload["tools"] = _alias_harmony_tools(tools, model)
elif tool_choice_none:
payload["tool_choice"] = "none"
# Mistral thinking-capable models — send reasoning_effort so Mistral
@@ -2370,7 +2422,7 @@ async def _stream_llm_inner(url: str, model: str, messages: List[Dict], temperat
if fn.get("name"):
_ollama_tool_calls.append({
"id": tc.get("id") or f"call_{len(_ollama_tool_calls)}",
"name": fn.get("name") or "",
"name": _unalias_harmony_tool_name(fn.get("name") or "", model),
"arguments": json.dumps(fn.get("arguments") or {}),
})
if j.get("done"):
@@ -2750,7 +2802,10 @@ async def _stream_llm_inner(url: str, model: str, messages: List[Dict], temperat
if tc.get("extra_content"):
_tc_acc[idx]["extra_content"] = tc["extra_content"]
if func.get("name"):
_tc_acc[idx]["name"] = func["name"]
# Map harmony aliases back to real
# tool names before anything
# downstream sees them.
_tc_acc[idx]["name"] = _unalias_harmony_tool_name(func["name"], model)
if "arguments" in func:
# Guard against a null arguments delta: `func` can be
# {"arguments": None} (JSON null), and a raw `+= None`