style: apply ruff's automatic fixes and formatter

Mechanical only, and separated from the judgment calls that follow so the
reviewable changes are not buried in a 98-file whitespace diff.

227 automatic fixes: 60 blank lines carrying whitespace, 60 unsorted import
blocks, 34 Optional[X] to X | None, 28 unused imports, 16 deprecated typing
imports, 12 datetime.timezone.utc to datetime.UTC, and assorted smaller
modernisations. Then `ruff format` over src and tests: 98 files reformatted,
35 already conforming.

No file among the unused-import findings defines __all__ or is an __init__.py,
so nothing here removes a re-export.

`make test`: 658 passed, unchanged from HEAD.

Two things observed while verifying, neither addressed here:

`pytest tests/` cannot collect — tests/e2e/test_orchestration_e2e.py uses an
`e2e` marker that is not registered, and the config is strict about markers.
This fails identically at HEAD, so it predates this change; `make test` passes
because it ignores tests/e2e, tests/integration and tests/contracts.

test_tatlock_tool_call_logging_calculator is flaky. It failed once in a full run
with these changes and passed on the next, passes in isolation with them, and
fails in isolation at HEAD. It is order- or timing-dependent, not a regression
from this commit — established by running the full suite both ways rather than
by reasoning about which change could have caused it.

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
2026-08-11 17:25:18 +02:00
co-authored by Claude
parent 57fa6c13fc
commit 78066fab1b
103 changed files with 1601 additions and 1749 deletions
+19 -9
View File
@@ -3,12 +3,14 @@ Integration tests for Steward + Tatlock streaming.
Tests the complete streaming flow with Steward preprocessing.
"""
import pytest
from unittest.mock import AsyncMock, MagicMock, patch
from src.responses.schemas import ResponseRequest
from src.responses.streaming import StreamingCoordinator, StreamEventType
from unittest.mock import patch
import pytest
from src.core.startup import initialize_application
from src.responses.schemas import ResponseRequest
from src.responses.streaming import StreamEventType, StreamingCoordinator
@pytest.fixture(scope="module", autouse=True)
@@ -32,7 +34,9 @@ class TestStewardStreaming:
# Mock the Steward analysis
with patch("src.core.preprocessing.analyze_request") as mock_steward:
# Mock the streaming method (async generator)
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools_stream") as mock_tatlock_stream:
with patch(
"src.agents.tatlock.TatlockAgent.run_with_scoped_tools_stream"
) as mock_tatlock_stream:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
# Mock Steward recommendation
@@ -89,7 +93,9 @@ class TestStewardStreaming:
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools_stream") as mock_tatlock_stream:
with patch(
"src.agents.tatlock.TatlockAgent.run_with_scoped_tools_stream"
) as mock_tatlock_stream:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
@@ -99,7 +105,7 @@ class TestStewardStreaming:
conversation_context=ConversationContext(
has_previous_context=True,
relevant_turns=[0],
context_summary="Previous calculation in turn 0"
context_summary="Previous calculation in turn 0",
),
)
@@ -134,7 +140,9 @@ class TestStewardStreaming:
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools_stream") as mock_tatlock_stream:
with patch(
"src.agents.tatlock.TatlockAgent.run_with_scoped_tools_stream"
) as mock_tatlock_stream:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
@@ -173,7 +181,9 @@ class TestStewardStreaming:
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools_stream") as mock_tatlock_stream:
with patch(
"src.agents.tatlock.TatlockAgent.run_with_scoped_tools_stream"
) as mock_tatlock_stream:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
@@ -7,12 +7,14 @@ Tests the complete Phase 2 request pipeline:
3. Tatlock runs with scoped tools
4. Response includes both Steward reasoning and Tatlock output
"""
import pytest
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
from src.core.startup import initialize_application
from src.responses.schemas import ResponseRequest
from src.responses.service import create_response_with_steward
from src.core.startup import initialize_application
@pytest.fixture(scope="module", autouse=True)
@@ -56,8 +58,9 @@ class TestStewardTatlockIntegration:
assert mock_steward.called
# Note: preprocess_request injects temporal context
steward_call_arg = mock_steward.call_args[0][0]
assert steward_call_arg.startswith("What's 2 + 2?"), \
f"Expected request to start with original message, got: {steward_call_arg}"
assert steward_call_arg.startswith(
"What's 2 + 2?"
), f"Expected request to start with original message, got: {steward_call_arg}"
# Verify Tatlock was called with scoped tools
assert mock_tatlock.called
@@ -100,7 +103,7 @@ class TestStewardTatlockIntegration:
conversation_context=ConversationContext(
has_previous_context=True,
relevant_turns=[0],
context_summary="Previous calculation in turn 0"
context_summary="Previous calculation in turn 0",
),
)
@@ -161,7 +164,10 @@ class TestStewardTatlockIntegration:
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
with patch("src.core.tool_tracking.ToolCallTracker.finalize") as mock_finalize:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
from src.agents.steward.schemas import (
ConversationContext,
StewardRecommendation,
)
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
@@ -198,7 +204,9 @@ class TestStewardTatlockIntegration:
missing_capabilities="Image generation capability would be needed",
)
mock_tatlock.return_value = "I'm afraid I don't have image generation capabilities, sir."
mock_tatlock.return_value = (
"I'm afraid I don't have image generation capabilities, sir."
)
response = await create_response_with_steward(request)
@@ -220,7 +228,10 @@ class TestStewardTatlockIntegration:
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
with patch("src.responses.service.ToolCallTracker") as mock_tracker_class:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
from src.agents.steward.schemas import (
ConversationContext,
StewardRecommendation,
)
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
+28 -25
View File
@@ -5,10 +5,12 @@ These tests verify the complete streaming flow from API endpoint through
StreamingCoordinator to TatlockAgent, ensuring no text duplication and
proper delta calculation.
"""
import json
import pytest
from httpx import AsyncClient
from fastapi.testclient import TestClient
from httpx import AsyncClient
@pytest.mark.integration
@@ -24,7 +26,7 @@ async def test_tatlock_streaming_no_duplication(async_client: AsyncClient):
request_data = {
"model": "Tatlock",
"input": [{"role": "user", "content": "Say hello"}],
"stream": True
"stream": True,
}
collected_deltas = []
@@ -74,12 +76,12 @@ async def test_tatlock_streaming_no_duplication(async_client: AsyncClient):
if len(words) > 0:
# Check for consecutive duplicate words (sign of duplication bug)
consecutive_dupes = sum(
1 for i in range(len(words) - 1)
if words[i] == words[i + 1] and len(words[i]) > 3
1 for i in range(len(words) - 1) if words[i] == words[i + 1] and len(words[i]) > 3
)
# Allow a few duplicates (natural language), but not excessive
assert consecutive_dupes < len(words) * 0.1, \
f"Too many consecutive duplicate words: {consecutive_dupes}/{len(words)}"
assert (
consecutive_dupes < len(words) * 0.1
), f"Too many consecutive duplicate words: {consecutive_dupes}/{len(words)}"
@pytest.mark.integration
@@ -94,7 +96,7 @@ async def test_tatlock_chat_streaming_no_duplication(async_client: AsyncClient):
request_data = {
"model": "Tatlock",
"messages": [{"role": "user", "content": "Hello"}],
"stream": True
"stream": True,
}
collected_content = []
@@ -138,11 +140,11 @@ async def test_tatlock_chat_streaming_no_duplication(async_client: AsyncClient):
words = full_response.lower().split()
if len(words) > 0:
consecutive_dupes = sum(
1 for i in range(len(words) - 1)
if words[i] == words[i + 1] and len(words[i]) > 3
1 for i in range(len(words) - 1) if words[i] == words[i + 1] and len(words[i]) > 3
)
assert consecutive_dupes < len(words) * 0.1, \
f"Too many consecutive duplicate words in chat response: {consecutive_dupes}/{len(words)}"
assert (
consecutive_dupes < len(words) * 0.1
), f"Too many consecutive duplicate words in chat response: {consecutive_dupes}/{len(words)}"
@pytest.mark.integration
@@ -153,7 +155,7 @@ def test_tatlock_non_streaming_responses_api(client: TestClient):
request_data = {
"model": "Tatlock",
"input": [{"role": "user", "content": "Say hello"}],
"stream": False
"stream": False,
}
response = client.post("/v1/responses", json=request_data, timeout=30.0)
@@ -183,7 +185,7 @@ def test_tatlock_non_streaming_chat_api(client: TestClient):
request_data = {
"model": "Tatlock",
"messages": [{"role": "user", "content": "Hello"}],
"stream": False
"stream": False,
}
response = client.post("/v1/chat/completions", json=request_data, timeout=30.0)
@@ -216,7 +218,7 @@ async def test_tatlock_streaming_delta_accumulation(async_client: AsyncClient):
request_data = {
"model": "Tatlock",
"input": [{"role": "user", "content": "Count to three"}],
"stream": True
"stream": True,
}
collected_deltas = []
@@ -248,8 +250,9 @@ async def test_tatlock_streaming_delta_accumulation(async_client: AsyncClient):
# Verify each delta is new content
current_full = "".join(collected_deltas)
assert current_full.startswith(previous_full_text), \
"Deltas should accumulate progressively"
assert current_full.startswith(
previous_full_text
), "Deltas should accumulate progressively"
previous_full_text = current_full
except json.JSONDecodeError:
@@ -273,7 +276,7 @@ async def test_tatlock_with_reasoning(async_client: AsyncClient):
"model": "Tatlock",
"input": [{"role": "user", "content": "Hello"}],
"reasoning": {"effort": "medium", "summary": "auto"},
"stream": True
"stream": True,
}
has_reasoning = False
@@ -328,7 +331,7 @@ async def test_tatlock_markdown_formatting_preserved(async_client: AsyncClient):
request_data = {
"model": "Tatlock",
"input": [{"role": "user", "content": "Can you give me an HTML5 boilerplate template?"}],
"stream": True
"stream": True,
}
collected_deltas = []
@@ -365,15 +368,15 @@ async def test_tatlock_markdown_formatting_preserved(async_client: AsyncClient):
full_response = "".join(collected_deltas)
# Always print the response for debugging
print("\n" + "="*80)
print("\n" + "=" * 80)
print("FULL RESPONSE (repr):")
print("="*80)
print("=" * 80)
print(repr(full_response))
print("\n" + "="*80)
print("\n" + "=" * 80)
print("FULL RESPONSE (formatted):")
print("="*80)
print("=" * 80)
print(full_response)
print("="*80 + "\n")
print("=" * 80 + "\n")
# Verify we got a response (xfail if LLM didn't produce output)
if len(full_response) < 100:
@@ -384,7 +387,7 @@ async def test_tatlock_markdown_formatting_preserved(async_client: AsyncClient):
pytest.xfail("No markdown code blocks in response (LLM response varied)")
# Verify newlines are preserved (not all collapsed to spaces)
newline_count = full_response.count('\n')
newline_count = full_response.count("\n")
if newline_count < 5:
pytest.xfail(f"Only {newline_count} newlines, formatting may have been lost")
@@ -412,7 +415,7 @@ def test_tatlock_markdown_non_streaming(client: TestClient):
request_data = {
"model": "Tatlock",
"input": [{"role": "user", "content": "Give me a simple Python hello world code"}],
"stream": False
"stream": False,
}
response = client.post("/v1/responses", json=request_data, timeout=30.0)