Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9980e4764c | ||
|
|
4907798e74 |
@@ -7,6 +7,22 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
|
|
||||||
## [Unreleased]
|
## [Unreleased]
|
||||||
|
|
||||||
|
## [1.8.4] - 2025-12-16
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Remove `<think>` wrappers from think messages** - Messages in `reasoning_content` should be plain text
|
||||||
|
- Removed `<think>` wrappers from delegation.py household think messages
|
||||||
|
- Removed `<think>` wrappers from orchestration.py status messages
|
||||||
|
- Think messages now appear cleanly in Open WebUI's reasoning block
|
||||||
|
|
||||||
|
## [1.8.3] - 2025-12-16
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Open WebUI streaming rendering** - Use `reasoning_content` field for thinking (DeepSeek R1 format) instead of `<think>` tags in `content`
|
||||||
|
- Open WebUI now renders thinking as proper collapsible blocks instead of broken HTML
|
||||||
|
|
||||||
## [1.8.2] - 2025-12-16
|
## [1.8.2] - 2025-12-16
|
||||||
|
|
||||||
### Fixed
|
### Fixed
|
||||||
|
|||||||
+1
-1
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|||||||
|
|
||||||
[project]
|
[project]
|
||||||
name = "tatlock"
|
name = "tatlock"
|
||||||
version = "1.8.2"
|
version = "1.8.4"
|
||||||
description = "OpenAI-compatible API with Ollama backend"
|
description = "OpenAI-compatible API with Ollama backend"
|
||||||
requires-python = ">=3.12"
|
requires-python = ">=3.12"
|
||||||
dependencies = []
|
dependencies = []
|
||||||
|
|||||||
+23
-22
@@ -40,45 +40,46 @@ class ActionType(Enum):
|
|||||||
# =============================================================================
|
# =============================================================================
|
||||||
|
|
||||||
HOUSEHOLD_THINK_MESSAGES: dict[str, dict[ActionType, dict[str, str]]] = {
|
HOUSEHOLD_THINK_MESSAGES: dict[str, dict[ActionType, dict[str, str]]] = {
|
||||||
|
# Note: No <think> wrappers needed - these go to reasoning_content field
|
||||||
"librarian": {
|
"librarian": {
|
||||||
ActionType.RETRIEVE: {
|
ActionType.RETRIEVE: {
|
||||||
"start": "<think>Allow me to consult the archives, sir.</think>",
|
"start": "Allow me to consult the archives, sir.",
|
||||||
"success": "<think>The Librarian has compiled the relevant findings.</think>",
|
"success": "The Librarian has compiled the relevant findings.",
|
||||||
"error": "<think>I'm afraid the archives proved difficult to access.</think>",
|
"error": "I'm afraid the archives proved difficult to access.",
|
||||||
},
|
},
|
||||||
ActionType.RESEARCH: {
|
ActionType.RESEARCH: {
|
||||||
"start": "<think>I've dispatched the Librarian to conduct some fresh research.</think>",
|
"start": "I've dispatched the Librarian to conduct some fresh research.",
|
||||||
"success": "<think>The Librarian has returned with findings, sir.</think>",
|
"success": "The Librarian has returned with findings, sir.",
|
||||||
"error": "<think>The research proved inconclusive, I'm afraid.</think>",
|
"error": "The research proved inconclusive, I'm afraid.",
|
||||||
},
|
},
|
||||||
ActionType.CREATE: {
|
ActionType.CREATE: {
|
||||||
"start": "<think>I'm having the Librarian prepare a new entry.</think>",
|
"start": "I'm having the Librarian prepare a new entry.",
|
||||||
"success": "<think>The new material has been properly catalogued, sir.</think>",
|
"success": "The new material has been properly catalogued, sir.",
|
||||||
"error": "<think>I'm afraid there was difficulty filing the entry.</think>",
|
"error": "I'm afraid there was difficulty filing the entry.",
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
"biographer": {
|
"biographer": {
|
||||||
ActionType.RETRIEVE: {
|
ActionType.RETRIEVE: {
|
||||||
"start": "<think>Let me consult the household records.</think>",
|
"start": "Let me consult the household records.",
|
||||||
"success": "<think>The Biographer has located the relevant information, sir.</think>",
|
"success": "The Biographer has located the relevant information, sir.",
|
||||||
"error": "<think>I'm unable to locate those particular records.</think>",
|
"error": "I'm unable to locate those particular records.",
|
||||||
},
|
},
|
||||||
ActionType.RECORD: {
|
ActionType.RECORD: {
|
||||||
"start": "<think>I've asked the Biographer to take note of this, sir.</think>",
|
"start": "I've asked the Biographer to take note of this, sir.",
|
||||||
"success": "<think>The household records have been updated accordingly.</think>",
|
"success": "The household records have been updated accordingly.",
|
||||||
"error": "<think>I'm afraid there was difficulty recording the entry.</think>",
|
"error": "I'm afraid there was difficulty recording the entry.",
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
"housekeeper": {
|
"housekeeper": {
|
||||||
ActionType.RETRIEVE: {
|
ActionType.RETRIEVE: {
|
||||||
"start": "<think>Allow me to inquire with the household staff.</think>",
|
"start": "Allow me to inquire with the household staff.",
|
||||||
"success": "<think>The staff reports the current status, sir.</think>",
|
"success": "The staff reports the current status, sir.",
|
||||||
"error": "<think>The household staff is momentarily unavailable, I'm afraid.</think>",
|
"error": "The household staff is momentarily unavailable, I'm afraid.",
|
||||||
},
|
},
|
||||||
ActionType.CONTROL: {
|
ActionType.CONTROL: {
|
||||||
"start": "<think>I'm instructing the household staff now, sir.</think>",
|
"start": "I'm instructing the household staff now, sir.",
|
||||||
"success": "<think>The household has been configured as requested.</think>",
|
"success": "The household has been configured as requested.",
|
||||||
"error": "<think>I'm afraid the staff reports an issue with that request.</think>",
|
"error": "I'm afraid the staff reports an issue with that request.",
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
@@ -139,7 +140,7 @@ def get_think_message(expert: str, task: str, phase: str) -> str:
|
|||||||
action_type = _detect_action_type(expert, task)
|
action_type = _detect_action_type(expert, task)
|
||||||
expert_messages = HOUSEHOLD_THINK_MESSAGES.get(expert, {})
|
expert_messages = HOUSEHOLD_THINK_MESSAGES.get(expert, {})
|
||||||
action_messages = expert_messages.get(action_type, expert_messages.get(ActionType.RETRIEVE, {}))
|
action_messages = expert_messages.get(action_type, expert_messages.get(ActionType.RETRIEVE, {}))
|
||||||
return action_messages.get(phase, f"<think>Consulting {expert}...</think>")
|
return action_messages.get(phase, f"Consulting {expert}...")
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
|
|||||||
+13
-13
@@ -176,19 +176,19 @@ async def orchestrate_with_think_updates(
|
|||||||
if delegation_task.expert_name == "librarian":
|
if delegation_task.expert_name == "librarian":
|
||||||
expert_display_name = "The Librarian"
|
expert_display_name = "The Librarian"
|
||||||
|
|
||||||
yield f"<think>🤝 Consulting {expert_display_name}...</think>\n"
|
yield f"🤝 Consulting {expert_display_name}...\n"
|
||||||
|
|
||||||
# Execute delegation (uses run() internally)
|
# Execute delegation (uses run() internally)
|
||||||
result = await execute_delegation(delegation_task)
|
result = await execute_delegation(delegation_task)
|
||||||
|
|
||||||
if result.success:
|
if result.success:
|
||||||
yield f"<think>✅ {expert_display_name} completed research.</think>\n"
|
yield f"✅ {expert_display_name} completed research.\n"
|
||||||
|
|
||||||
# Yield the expert's findings
|
# Yield the expert's findings
|
||||||
if result.output:
|
if result.output:
|
||||||
yield f"\n{result.output}"
|
yield f"\n{result.output}"
|
||||||
else:
|
else:
|
||||||
yield f"<think>⚠️ {expert_display_name} encountered an issue: {result.error}</think>\n"
|
yield f"⚠️ {expert_display_name} encountered an issue: {result.error}\n"
|
||||||
|
|
||||||
logger.info(
|
logger.info(
|
||||||
"orchestration_complete",
|
"orchestration_complete",
|
||||||
@@ -449,12 +449,12 @@ async def orchestrate_multi_expert(
|
|||||||
return
|
return
|
||||||
|
|
||||||
# Stream: Starting multi-expert coordination
|
# Stream: Starting multi-expert coordination
|
||||||
yield f"<think>🎯 Starting multi-expert coordination ({len(tasks)} tasks, {mode.value})...</think>\n"
|
yield f"🎯 Starting multi-expert coordination ({len(tasks)} tasks, {mode.value})...\n"
|
||||||
|
|
||||||
if mode == ExecutionMode.PARALLEL:
|
if mode == ExecutionMode.PARALLEL:
|
||||||
# Parallel execution - emit one update then run all at once
|
# Parallel execution - emit one update then run all at once
|
||||||
expert_names = ", ".join(_get_display_name(t.expert_name) for t in tasks)
|
expert_names = ", ".join(_get_display_name(t.expert_name) for t in tasks)
|
||||||
yield f"<think>🔄 Consulting in parallel: {expert_names}...</think>\n"
|
yield f"🔄 Consulting in parallel: {expert_names}...\n"
|
||||||
|
|
||||||
result = await execute_parallel(tasks)
|
result = await execute_parallel(tasks)
|
||||||
|
|
||||||
@@ -462,9 +462,9 @@ async def orchestrate_multi_expert(
|
|||||||
for expert_name, expert_result in result.results.items():
|
for expert_name, expert_result in result.results.items():
|
||||||
display_name = _get_display_name(expert_name)
|
display_name = _get_display_name(expert_name)
|
||||||
if expert_result.success:
|
if expert_result.success:
|
||||||
yield f"<think>✅ {display_name} completed.</think>\n"
|
yield f"✅ {display_name} completed.\n"
|
||||||
else:
|
else:
|
||||||
yield f"<think>⚠️ {display_name} failed: {expert_result.error}</think>\n"
|
yield f"⚠️ {display_name} failed: {expert_result.error}\n"
|
||||||
|
|
||||||
else:
|
else:
|
||||||
# Sequential execution - emit updates for each task
|
# Sequential execution - emit updates for each task
|
||||||
@@ -472,27 +472,27 @@ async def orchestrate_multi_expert(
|
|||||||
|
|
||||||
for task in tasks:
|
for task in tasks:
|
||||||
display_name = _get_display_name(task.expert_name)
|
display_name = _get_display_name(task.expert_name)
|
||||||
yield f"<think>🤝 Consulting {display_name}...</think>\n"
|
yield f"🤝 Consulting {display_name}...\n"
|
||||||
|
|
||||||
task_result = await execute_delegation(task)
|
task_result = await execute_delegation(task)
|
||||||
result.add_result(task_result)
|
result.add_result(task_result)
|
||||||
|
|
||||||
if task_result.success:
|
if task_result.success:
|
||||||
yield f"<think>✅ {display_name} completed.</think>\n"
|
yield f"✅ {display_name} completed.\n"
|
||||||
else:
|
else:
|
||||||
yield f"<think>⚠️ {display_name} failed: {task_result.error}</think>\n"
|
yield f"⚠️ {display_name} failed: {task_result.error}\n"
|
||||||
if stop_on_failure:
|
if stop_on_failure:
|
||||||
yield "<think>🛑 Stopping due to failure.</think>\n"
|
yield "🛑 Stopping due to failure.\n"
|
||||||
break
|
break
|
||||||
|
|
||||||
result.aggregate_outputs()
|
result.aggregate_outputs()
|
||||||
|
|
||||||
# Stream: Summary
|
# Stream: Summary
|
||||||
if result.all_succeeded:
|
if result.all_succeeded:
|
||||||
yield "<think>🎉 All experts completed successfully.</think>\n"
|
yield "🎉 All experts completed successfully.\n"
|
||||||
else:
|
else:
|
||||||
failed_names = ", ".join(_get_display_name(e) for e in result.failed_experts)
|
failed_names = ", ".join(_get_display_name(e) for e in result.failed_experts)
|
||||||
yield f"<think>⚠️ Some experts failed: {failed_names}</think>\n"
|
yield f"⚠️ Some experts failed: {failed_names}\n"
|
||||||
|
|
||||||
# Yield combined output
|
# Yield combined output
|
||||||
if result.combined_output:
|
if result.combined_output:
|
||||||
|
|||||||
@@ -55,6 +55,7 @@ class ChatCompletionChunkDelta(CustomBaseModel):
|
|||||||
"""Delta in streaming chunk."""
|
"""Delta in streaming chunk."""
|
||||||
role: str | None = None
|
role: str | None = None
|
||||||
content: str | None = None
|
content: str | None = None
|
||||||
|
reasoning_content: str | None = None # For thinking/reasoning (DeepSeek R1 format)
|
||||||
|
|
||||||
|
|
||||||
class ChatCompletionChunkChoice(CustomBaseModel):
|
class ChatCompletionChunkChoice(CustomBaseModel):
|
||||||
|
|||||||
+6
-35
@@ -172,24 +172,9 @@ async def create_chat_completion_stream(
|
|||||||
|
|
||||||
async for event in stream_generator:
|
async for event in stream_generator:
|
||||||
if event.event == StreamEventType.REASONING_SUMMARY_DELTA:
|
if event.event == StreamEventType.REASONING_SUMMARY_DELTA:
|
||||||
# Start <think> block if needed
|
# Stream reasoning via reasoning_content field (DeepSeek R1 format)
|
||||||
if not in_reasoning:
|
# Open WebUI renders this as collapsible thinking block
|
||||||
yield ChatCompletionChunk(
|
in_reasoning = True
|
||||||
id=completion_id,
|
|
||||||
object=constants.CHAT_COMPLETION_CHUNK_OBJECT,
|
|
||||||
created=created_at,
|
|
||||||
model=request.model,
|
|
||||||
choices=[
|
|
||||||
ChatCompletionChunkChoice(
|
|
||||||
index=0,
|
|
||||||
delta=ChatCompletionChunkDelta(content="<think>\n"),
|
|
||||||
finish_reason=None,
|
|
||||||
)
|
|
||||||
],
|
|
||||||
)
|
|
||||||
in_reasoning = True
|
|
||||||
|
|
||||||
# Stream reasoning delta
|
|
||||||
yield ChatCompletionChunk(
|
yield ChatCompletionChunk(
|
||||||
id=completion_id,
|
id=completion_id,
|
||||||
object=constants.CHAT_COMPLETION_CHUNK_OBJECT,
|
object=constants.CHAT_COMPLETION_CHUNK_OBJECT,
|
||||||
@@ -198,29 +183,15 @@ async def create_chat_completion_stream(
|
|||||||
choices=[
|
choices=[
|
||||||
ChatCompletionChunkChoice(
|
ChatCompletionChunkChoice(
|
||||||
index=0,
|
index=0,
|
||||||
delta=ChatCompletionChunkDelta(content=event.delta),
|
delta=ChatCompletionChunkDelta(reasoning_content=event.delta),
|
||||||
finish_reason=None,
|
finish_reason=None,
|
||||||
)
|
)
|
||||||
],
|
],
|
||||||
)
|
)
|
||||||
|
|
||||||
elif event.event == StreamEventType.REASONING_SUMMARY_DONE:
|
elif event.event == StreamEventType.REASONING_SUMMARY_DONE:
|
||||||
# Close <think> block
|
# Signal end of reasoning block (no content needed)
|
||||||
if in_reasoning:
|
in_reasoning = False
|
||||||
yield ChatCompletionChunk(
|
|
||||||
id=completion_id,
|
|
||||||
object=constants.CHAT_COMPLETION_CHUNK_OBJECT,
|
|
||||||
created=created_at,
|
|
||||||
model=request.model,
|
|
||||||
choices=[
|
|
||||||
ChatCompletionChunkChoice(
|
|
||||||
index=0,
|
|
||||||
delta=ChatCompletionChunkDelta(content="</think>\n\n"),
|
|
||||||
finish_reason=None,
|
|
||||||
)
|
|
||||||
],
|
|
||||||
)
|
|
||||||
in_reasoning = False
|
|
||||||
|
|
||||||
elif event.event == StreamEventType.OUTPUT_TEXT_DELTA:
|
elif event.event == StreamEventType.OUTPUT_TEXT_DELTA:
|
||||||
# Stream message content
|
# Stream message content
|
||||||
|
|||||||
@@ -248,13 +248,16 @@ class TestHouseholdThinkMessages:
|
|||||||
assert "success" in messages, f"{expert}/{action_type} missing 'success'"
|
assert "success" in messages, f"{expert}/{action_type} missing 'success'"
|
||||||
assert "error" in messages, f"{expert}/{action_type} missing 'error'"
|
assert "error" in messages, f"{expert}/{action_type} missing 'error'"
|
||||||
|
|
||||||
def test_messages_are_think_tags(self):
|
def test_messages_are_plain_text(self):
|
||||||
"""Test messages are wrapped in <think> tags."""
|
"""Test messages are plain text (no <think> wrappers - those go to reasoning_content)."""
|
||||||
for expert, action_types in HOUSEHOLD_THINK_MESSAGES.items():
|
for expert, action_types in HOUSEHOLD_THINK_MESSAGES.items():
|
||||||
for action_type, messages in action_types.items():
|
for action_type, messages in action_types.items():
|
||||||
for phase, msg in messages.items():
|
for phase, msg in messages.items():
|
||||||
assert msg.startswith("<think>"), f"{expert}/{action_type}/{phase}"
|
# Messages should NOT have <think> wrappers - they go to reasoning_content field
|
||||||
assert msg.endswith("</think>"), f"{expert}/{action_type}/{phase}"
|
assert "<think>" not in msg, f"{expert}/{action_type}/{phase} should not have <think> wrapper"
|
||||||
|
assert "</think>" not in msg, f"{expert}/{action_type}/{phase} should not have </think> wrapper"
|
||||||
|
# Messages should be non-empty strings
|
||||||
|
assert isinstance(msg, str) and len(msg) > 0, f"{expert}/{action_type}/{phase}"
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.unit
|
@pytest.mark.unit
|
||||||
@@ -310,31 +313,32 @@ class TestGetThinkMessage:
|
|||||||
def test_librarian_retrieve_start(self):
|
def test_librarian_retrieve_start(self):
|
||||||
"""Test getting librarian retrieve start message."""
|
"""Test getting librarian retrieve start message."""
|
||||||
msg = get_think_message("librarian", "search for Docker", "start")
|
msg = get_think_message("librarian", "search for Docker", "start")
|
||||||
assert "<think>" in msg
|
# No <think> wrappers - messages go to reasoning_content field
|
||||||
assert "</think>" in msg
|
assert "<think>" not in msg
|
||||||
|
assert "archives" in msg.lower() or "consult" in msg.lower()
|
||||||
|
|
||||||
def test_librarian_create_success(self):
|
def test_librarian_create_success(self):
|
||||||
"""Test getting librarian create success message."""
|
"""Test getting librarian create success message."""
|
||||||
msg = get_think_message("librarian", "create a wiki page", "success")
|
msg = get_think_message("librarian", "create a wiki page", "success")
|
||||||
assert "<think>" in msg
|
assert "<think>" not in msg
|
||||||
assert "catalogued" in msg.lower()
|
assert "catalogued" in msg.lower()
|
||||||
|
|
||||||
def test_biographer_record_start(self):
|
def test_biographer_record_start(self):
|
||||||
"""Test getting biographer record start message."""
|
"""Test getting biographer record start message."""
|
||||||
msg = get_think_message("biographer", "remember my preference", "start")
|
msg = get_think_message("biographer", "remember my preference", "start")
|
||||||
assert "<think>" in msg
|
assert "<think>" not in msg
|
||||||
assert "note" in msg.lower() or "biographer" in msg.lower()
|
assert "note" in msg.lower() or "biographer" in msg.lower()
|
||||||
|
|
||||||
def test_housekeeper_control_success(self):
|
def test_housekeeper_control_success(self):
|
||||||
"""Test getting housekeeper control success message."""
|
"""Test getting housekeeper control success message."""
|
||||||
msg = get_think_message("housekeeper", "turn on the lights", "success")
|
msg = get_think_message("housekeeper", "turn on the lights", "success")
|
||||||
assert "<think>" in msg
|
assert "<think>" not in msg
|
||||||
assert "configured" in msg.lower()
|
assert "configured" in msg.lower()
|
||||||
|
|
||||||
def test_unknown_expert_fallback(self):
|
def test_unknown_expert_fallback(self):
|
||||||
"""Test unknown expert gets fallback message."""
|
"""Test unknown expert gets fallback message."""
|
||||||
msg = get_think_message("unknown_expert", "some task", "start")
|
msg = get_think_message("unknown_expert", "some task", "start")
|
||||||
assert "<think>" in msg
|
assert "<think>" not in msg
|
||||||
assert "unknown_expert" in msg.lower()
|
assert "unknown_expert" in msg.lower()
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -208,8 +208,8 @@ class TestOrchestrateWithThinkUpdates:
|
|||||||
):
|
):
|
||||||
updates.append(update)
|
updates.append(update)
|
||||||
|
|
||||||
# First update should be think tag about consulting
|
# First update should be about consulting (no <think> wrappers anymore)
|
||||||
assert any("<think>" in u and "Consulting" in u for u in updates)
|
assert any("Consulting" in u for u in updates)
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_orchestrate_emits_think_after_delegation(self):
|
async def test_orchestrate_emits_think_after_delegation(self):
|
||||||
@@ -233,8 +233,8 @@ class TestOrchestrateWithThinkUpdates:
|
|||||||
):
|
):
|
||||||
updates.append(update)
|
updates.append(update)
|
||||||
|
|
||||||
# Should have think tag about completion
|
# Should have message about completion (no <think> wrappers anymore)
|
||||||
assert any("<think>" in u and "completed" in u for u in updates)
|
assert any("completed" in u for u in updates)
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_orchestrate_yields_expert_output(self):
|
async def test_orchestrate_yields_expert_output(self):
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ Tests for chat completions streaming wrapper.
|
|||||||
Tests that the wrapper correctly:
|
Tests that the wrapper correctly:
|
||||||
- Wraps Responses API
|
- Wraps Responses API
|
||||||
- Enables reasoning automatically
|
- Enables reasoning automatically
|
||||||
- Converts reasoning to <think> tags
|
- Streams reasoning via reasoning_content field (DeepSeek R1 format)
|
||||||
- Streams both reasoning and content
|
- Streams both reasoning and content
|
||||||
"""
|
"""
|
||||||
import json
|
import json
|
||||||
@@ -17,7 +17,7 @@ from src.chat import constants
|
|||||||
@pytest.mark.unit
|
@pytest.mark.unit
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_streaming_wrapper_enables_reasoning(async_client: AsyncClient):
|
async def test_streaming_wrapper_enables_reasoning(async_client: AsyncClient):
|
||||||
"""Test that streaming wrapper automatically enables reasoning."""
|
"""Test that streaming wrapper automatically enables reasoning via reasoning_content."""
|
||||||
request_data = {
|
request_data = {
|
||||||
"model": "lorem-tester",
|
"model": "lorem-tester",
|
||||||
"messages": [
|
"messages": [
|
||||||
@@ -27,7 +27,7 @@ async def test_streaming_wrapper_enables_reasoning(async_client: AsyncClient):
|
|||||||
}
|
}
|
||||||
|
|
||||||
chunks_received = []
|
chunks_received = []
|
||||||
think_tags_found = False
|
reasoning_content_found = False
|
||||||
|
|
||||||
async with async_client.stream(
|
async with async_client.stream(
|
||||||
"POST",
|
"POST",
|
||||||
@@ -51,12 +51,12 @@ async def test_streaming_wrapper_enables_reasoning(async_client: AsyncClient):
|
|||||||
chunk = json.loads(data_str)
|
chunk = json.loads(data_str)
|
||||||
chunks_received.append(chunk)
|
chunks_received.append(chunk)
|
||||||
|
|
||||||
# Check for <think> tags in delta content
|
# Check for reasoning_content in delta (DeepSeek R1 format)
|
||||||
if "choices" in chunk and len(chunk["choices"]) > 0:
|
if "choices" in chunk and len(chunk["choices"]) > 0:
|
||||||
delta = chunk["choices"][0].get("delta", {})
|
delta = chunk["choices"][0].get("delta", {})
|
||||||
content = delta.get("content")
|
reasoning = delta.get("reasoning_content")
|
||||||
if content and ("<think>" in content or "</think>" in content):
|
if reasoning:
|
||||||
think_tags_found = True
|
reasoning_content_found = True
|
||||||
|
|
||||||
except json.JSONDecodeError:
|
except json.JSONDecodeError:
|
||||||
pass
|
pass
|
||||||
@@ -64,14 +64,14 @@ async def test_streaming_wrapper_enables_reasoning(async_client: AsyncClient):
|
|||||||
# Should have received chunks
|
# Should have received chunks
|
||||||
assert len(chunks_received) > 0
|
assert len(chunks_received) > 0
|
||||||
|
|
||||||
# Should have found <think> tags (reasoning enabled automatically)
|
# Should have found reasoning_content (reasoning enabled automatically)
|
||||||
assert think_tags_found, "Expected <think> tags in streaming output"
|
assert reasoning_content_found, "Expected reasoning_content in streaming output"
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.unit
|
@pytest.mark.unit
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_streaming_wrapper_reasoning_before_content(async_client: AsyncClient):
|
async def test_streaming_wrapper_reasoning_before_content(async_client: AsyncClient):
|
||||||
"""Test that reasoning (<think> tags) comes before actual content."""
|
"""Test that reasoning_content comes before regular content."""
|
||||||
request_data = {
|
request_data = {
|
||||||
"model": "lorem-tester",
|
"model": "lorem-tester",
|
||||||
"messages": [
|
"messages": [
|
||||||
@@ -80,10 +80,7 @@ async def test_streaming_wrapper_reasoning_before_content(async_client: AsyncCli
|
|||||||
"stream": True
|
"stream": True
|
||||||
}
|
}
|
||||||
|
|
||||||
all_content = []
|
chunk_types = [] # Track order: 'reasoning' or 'content'
|
||||||
found_think_opening = False
|
|
||||||
found_think_closing = False
|
|
||||||
found_content_after_think = False
|
|
||||||
|
|
||||||
async with async_client.stream(
|
async with async_client.stream(
|
||||||
"POST",
|
"POST",
|
||||||
@@ -106,28 +103,22 @@ async def test_streaming_wrapper_reasoning_before_content(async_client: AsyncCli
|
|||||||
chunk = json.loads(data_str)
|
chunk = json.loads(data_str)
|
||||||
if "choices" in chunk and len(chunk["choices"]) > 0:
|
if "choices" in chunk and len(chunk["choices"]) > 0:
|
||||||
delta = chunk["choices"][0].get("delta", {})
|
delta = chunk["choices"][0].get("delta", {})
|
||||||
content = delta.get("content", "")
|
reasoning = delta.get("reasoning_content")
|
||||||
if content:
|
content = delta.get("content")
|
||||||
all_content.append(content)
|
|
||||||
|
|
||||||
if "<think>" in content:
|
if reasoning:
|
||||||
found_think_opening = True
|
chunk_types.append("reasoning")
|
||||||
if "</think>" in content:
|
if content:
|
||||||
found_think_closing = True
|
chunk_types.append("content")
|
||||||
# Content after closing think tag
|
|
||||||
if found_think_closing and content.strip() and "<think>" not in content and "</think>" not in content:
|
|
||||||
found_content_after_think = True
|
|
||||||
|
|
||||||
except json.JSONDecodeError:
|
except json.JSONDecodeError:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
# Verify ordering
|
# Verify reasoning comes before content
|
||||||
full_text = "".join(all_content)
|
if "reasoning" in chunk_types and "content" in chunk_types:
|
||||||
if found_think_opening and found_think_closing:
|
first_reasoning = chunk_types.index("reasoning")
|
||||||
# Reasoning should come before main content
|
first_content = chunk_types.index("content")
|
||||||
think_start = full_text.index("<think>")
|
assert first_reasoning < first_content, "reasoning_content should come before content"
|
||||||
think_end = full_text.index("</think>")
|
|
||||||
assert think_start < think_end, "Opening <think> should come before closing </think>"
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.unit
|
@pytest.mark.unit
|
||||||
|
|||||||
Reference in New Issue
Block a user