style: apply ruff's automatic fixes and formatter

Mechanical only, and separated from the judgment calls that follow so the
reviewable changes are not buried in a 98-file whitespace diff.

227 automatic fixes: 60 blank lines carrying whitespace, 60 unsorted import
blocks, 34 Optional[X] to X | None, 28 unused imports, 16 deprecated typing
imports, 12 datetime.timezone.utc to datetime.UTC, and assorted smaller
modernisations. Then `ruff format` over src and tests: 98 files reformatted,
35 already conforming.

No file among the unused-import findings defines __all__ or is an __init__.py,
so nothing here removes a re-export.

`make test`: 658 passed, unchanged from HEAD.

Two things observed while verifying, neither addressed here:

`pytest tests/` cannot collect — tests/e2e/test_orchestration_e2e.py uses an
`e2e` marker that is not registered, and the config is strict about markers.
This fails identically at HEAD, so it predates this change; `make test` passes
because it ignores tests/e2e, tests/integration and tests/contracts.

test_tatlock_tool_call_logging_calculator is flaky. It failed once in a full run
with these changes and passed on the next, passes in isolation with them, and
fails in isolation at HEAD. It is order- or timing-dependent, not a regression
from this commit — established by running the full suite both ways rather than
by reasoning about which change could have caused it.

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
2026-08-11 17:25:18 +02:00
co-authored by Claude
parent 57fa6c13fc
commit 78066fab1b
103 changed files with 1601 additions and 1749 deletions
+66 -86
View File
@@ -7,10 +7,12 @@ These tests hit the actual running server and test the full stack:
- Tool execution
- Response formatting
"""
from collections.abc import AsyncGenerator
import httpx
import pytest
import pytest_asyncio
import httpx
from typing import AsyncGenerator
# Test server base URL (assumes server is running on localhost:8777 via ./wakeup.sh)
BASE_URL = "http://localhost:8777"
@@ -34,10 +36,8 @@ class TestChatCompletionsE2E:
"/v1/chat/completions",
json={
"model": "Tatlock",
"messages": [
{"role": "user", "content": "What is 144 divided by 12?"}
],
}
"messages": [{"role": "user", "content": "What is 144 divided by 12?"}],
},
)
assert response.status_code == 200
@@ -68,7 +68,9 @@ class TestChatCompletionsE2E:
assert "usage" in data
assert data["usage"]["total_tokens"] > 0
print(f"✓ Calculator test passed. Found '12' in response. Tool indicator: {has_calculator_indicator}")
print(
f"✓ Calculator test passed. Found '12' in response. Tool indicator: {has_calculator_indicator}"
)
@pytest.mark.asyncio
async def test_web_search(self, client: httpx.AsyncClient):
@@ -80,7 +82,7 @@ class TestChatCompletionsE2E:
"messages": [
{"role": "user", "content": "Search for the current population of Tokyo"}
],
}
},
)
assert response.status_code == 200
@@ -114,10 +116,8 @@ class TestChatCompletionsE2E:
"/v1/chat/completions",
json={
"model": "Tatlock",
"messages": [
{"role": "user", "content": "What is 15 times 4?"}
],
}
"messages": [{"role": "user", "content": "What is 15 times 4?"}],
},
)
assert response1.status_code == 200
@@ -135,9 +135,9 @@ class TestChatCompletionsE2E:
"messages": [
{"role": "user", "content": "What is 15 times 4?"},
{"role": "assistant", "content": message1},
{"role": "user", "content": "Now add 20 to that result."}
{"role": "user", "content": "Now add 20 to that result."},
],
}
},
)
assert response2.status_code == 200
@@ -152,7 +152,9 @@ class TestChatCompletionsE2E:
has_calculation = "60" in message2 and "20" in message2
assert has_answer or has_calculation, f"Expected '80' or calculation in: {message2}"
print(f"✓ Multi-turn test passed. Answer found: {has_answer}, Calculation shown: {has_calculation}")
print(
f"✓ Multi-turn test passed. Answer found: {has_answer}, Calculation shown: {has_calculation}"
)
@pytest.mark.asyncio
async def test_calculation_and_search(self, client: httpx.AsyncClient):
@@ -161,13 +163,8 @@ class TestChatCompletionsE2E:
"/v1/chat/completions",
json={
"model": "Tatlock",
"messages": [
{
"role": "user",
"content": "Calculate the square root of 256"
}
],
}
"messages": [{"role": "user", "content": "Calculate the square root of 256"}],
},
)
assert response.status_code == 200
@@ -194,10 +191,8 @@ class TestChatCompletionsE2E:
"/v1/chat/completions",
json={
"model": "Tatlock",
"messages": [
{"role": "user", "content": "Hello, how are you?"}
],
}
"messages": [{"role": "user", "content": "Hello, how are you?"}],
},
)
assert response.status_code == 200
@@ -223,10 +218,8 @@ class TestChatCompletionsE2E:
"/v1/chat/completions",
json={
"model": "Tatlock",
"messages": [
{"role": "user", "content": "What is today's date?"}
],
}
"messages": [{"role": "user", "content": "What is today's date?"}],
},
)
assert response.status_code == 200
@@ -242,11 +235,16 @@ class TestChatCompletionsE2E:
# Should contain some date/time information (flexible - varies in format)
import re
has_date = (
re.search(r'\d{4}', message) or # Year
re.search(r'\d{1,2}', message) or # Day/month number
re.search(r'(January|February|March|April|May|June|July|August|September|October|November|December)', message, re.IGNORECASE) or
"today" in message.lower()
re.search(r"\d{4}", message) # Year
or re.search(r"\d{1,2}", message) # Day/month number
or re.search(
r"(January|February|March|April|May|June|July|August|September|October|November|December)",
message,
re.IGNORECASE,
)
or "today" in message.lower()
)
assert has_date, f"Expected date/time information in: {message}"
@@ -263,11 +261,9 @@ class TestResponsesAPIE2E:
"/v1/responses",
json={
"model": "Tatlock",
"input": [
{"role": "user", "content": "Calculate 25 times 16"}
],
"reasoning": {"effort": "medium", "summary": "auto"}
}
"input": [{"role": "user", "content": "Calculate 25 times 16"}],
"reasoning": {"effort": "medium", "summary": "auto"},
},
)
assert response.status_code == 200
@@ -300,7 +296,7 @@ class TestResponsesAPIE2E:
assert "usage" in data
assert data["usage"]["total_tokens"] > 0
print(f"✓ Responses API test passed. Found '400' with Steward reasoning.")
print("✓ Responses API test passed. Found '400' with Steward reasoning.")
@pytest.mark.asyncio
async def test_response_multi_turn(self, client: httpx.AsyncClient):
@@ -312,10 +308,10 @@ class TestResponsesAPIE2E:
"input": [
{"role": "user", "content": "What is 7 times 8?"},
{"role": "assistant", "content": "Certainly, sir. 7 times 8 equals 56."},
{"role": "user", "content": "Double that number."}
{"role": "user", "content": "Double that number."},
],
"reasoning": {"effort": "medium", "summary": "auto"}
}
"reasoning": {"effort": "medium", "summary": "auto"},
},
)
assert response.status_code == 200
@@ -346,11 +342,9 @@ class TestStreamingE2E:
"/v1/chat/completions",
json={
"model": "Tatlock",
"messages": [
{"role": "user", "content": "What is 9 times 7?"}
],
"stream": True
}
"messages": [{"role": "user", "content": "What is 9 times 7?"}],
"stream": True,
},
) as response:
assert response.status_code == 200
@@ -362,6 +356,7 @@ class TestStreamingE2E:
break
import json
chunk = json.loads(data_str)
chunks.append(chunk)
@@ -373,8 +368,7 @@ class TestStreamingE2E:
# Should have received Steward's reasoning (in <think> tags)
full_content = "".join(
chunk["choices"][0]["delta"].get("content", "") or ""
for chunk in chunks
chunk["choices"][0]["delta"].get("content", "") or "" for chunk in chunks
)
assert "<think>" in full_content
assert "</think>" in full_content
@@ -393,10 +387,8 @@ class TestErrorHandling:
"/v1/chat/completions",
json={
"model": "nonexistent-model",
"messages": [
{"role": "user", "content": "Hello"}
],
}
"messages": [{"role": "user", "content": "Hello"}],
},
)
assert response.status_code == 404
@@ -411,7 +403,7 @@ class TestErrorHandling:
json={
"model": "Tatlock",
# Missing "messages" field
}
},
)
assert response.status_code == 422
@@ -425,11 +417,9 @@ class TestErrorHandling:
"/v1/chat/completions",
json={
"model": "Tatlock",
"messages": [
{"role": "user", "content": "Hello"}
],
"temperature": 5.0 # Max is 2.0
}
"messages": [{"role": "user", "content": "Hello"}],
"temperature": 5.0, # Max is 2.0
},
)
assert response.status_code == 422
@@ -447,11 +437,9 @@ class TestChatResponsesWrapper:
"/v1/responses",
json={
"model": "Tatlock",
"input": [
{"role": "user", "content": "Calculate 13 times 9"}
],
"reasoning": {"effort": "medium", "summary": "auto"}
}
"input": [{"role": "user", "content": "Calculate 13 times 9"}],
"reasoning": {"effort": "medium", "summary": "auto"},
},
)
assert response.status_code == 200
@@ -500,10 +488,8 @@ class TestChatResponsesWrapper:
"/v1/chat/completions",
json={
"model": "Tatlock",
"messages": [
{"role": "user", "content": "What is 5 plus 3?"}
],
}
"messages": [{"role": "user", "content": "What is 5 plus 3?"}],
},
)
assert response.status_code == 200
@@ -539,11 +525,9 @@ class TestChatResponsesWrapper:
"/v1/chat/completions",
json={
"model": "Tatlock",
"messages": [
{"role": "user", "content": "Count to 3"}
],
"stream": True
}
"messages": [{"role": "user", "content": "Count to 3"}],
"stream": True,
},
) as response:
assert response.status_code == 200
@@ -555,6 +539,7 @@ class TestChatResponsesWrapper:
break
import json
chunk = json.loads(data_str)
chunks.append(chunk)
@@ -574,10 +559,7 @@ class TestChatResponsesWrapper:
assert chunks[0]["choices"][0]["delta"]["role"] == "assistant"
# Should have content chunks
has_content = any(
"content" in chunk["choices"][0]["delta"]
for chunk in chunks
)
has_content = any("content" in chunk["choices"][0]["delta"] for chunk in chunks)
assert has_content
print(f"✓ Streaming format matches OpenAI spec ({len(chunks)} chunks)")
@@ -593,11 +575,9 @@ class TestStewardIntegration:
"/v1/responses",
json={
"model": "Tatlock",
"input": [
{"role": "user", "content": "Calculate 123 times 456"}
],
"reasoning": {"effort": "medium", "summary": "auto"}
}
"input": [{"role": "user", "content": "Calculate 123 times 456"}],
"reasoning": {"effort": "medium", "summary": "auto"},
},
)
assert response.status_code == 200
@@ -620,10 +600,10 @@ class TestStewardIntegration:
"input": [
{"role": "user", "content": "My favorite number is 42"},
{"role": "assistant", "content": "Noted, sir. 42 is an excellent choice."},
{"role": "user", "content": "What was that number again?"}
{"role": "user", "content": "What was that number again?"},
],
"reasoning": {"effort": "medium", "summary": "auto"}
}
"reasoning": {"effort": "medium", "summary": "auto"},
},
)
assert response.status_code == 200