Files
tatlock/tests/agents/test_tatlock_agent.py
jpmschweitzerandClaude 78066fab1b style: apply ruff's automatic fixes and formatter
Mechanical only, and separated from the judgment calls that follow so the
reviewable changes are not buried in a 98-file whitespace diff.

227 automatic fixes: 60 blank lines carrying whitespace, 60 unsorted import
blocks, 34 Optional[X] to X | None, 28 unused imports, 16 deprecated typing
imports, 12 datetime.timezone.utc to datetime.UTC, and assorted smaller
modernisations. Then `ruff format` over src and tests: 98 files reformatted,
35 already conforming.

No file among the unused-import findings defines __all__ or is an __init__.py,
so nothing here removes a re-export.

`make test`: 658 passed, unchanged from HEAD.

Two things observed while verifying, neither addressed here:

`pytest tests/` cannot collect — tests/e2e/test_orchestration_e2e.py uses an
`e2e` marker that is not registered, and the config is strict about markers.
This fails identically at HEAD, so it predates this change; `make test` passes
because it ignores tests/e2e, tests/integration and tests/contracts.

test_tatlock_tool_call_logging_calculator is flaky. It failed once in a full run
with these changes and passed on the next, passes in isolation with them, and
fails in isolation at HEAD. It is order- or timing-dependent, not a regression
from this commit — established by running the full suite both ways rather than
by reasoning about which change could have caused it.

Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-11 17:25:18 +02:00

397 lines
14 KiB
Python

"""
Tests for Tatlock agent conversation history and tool call logging.
These tests verify:
1. Conversation history is properly passed to PydanticAI (Tatlock remembers context)
2. Tool calls are logged to reasoning output (users see what tools are doing)
"""
import pytest
from httpx import AsyncClient
@pytest.mark.integration
@pytest.mark.asyncio
async def test_tatlock_conversation_history_memory(async_client: AsyncClient):
"""
Test that Tatlock remembers previous turns of the conversation.
This verifies the fix where Tatlock was only using the last user message
instead of the full conversation history.
Note: This test may fail due to LLM non-determinism.
"""
# First turn: User introduces themselves
request_data_1 = {
"model": "Tatlock",
"messages": [
{"role": "user", "content": "My name is Alice and I love Python programming."}
],
"stream": False,
}
response_1 = await async_client.post("/v1/chat/completions", json=request_data_1, timeout=120.0)
assert response_1.status_code == 200
data_1 = response_1.json()
first_response = data_1["choices"][0]["message"]["content"]
# Second turn: Ask about previous information
# Tatlock should remember the user's name and interest
request_data_2 = {
"model": "Tatlock",
"messages": [
{"role": "user", "content": "My name is Alice and I love Python programming."},
{"role": "assistant", "content": first_response},
{
"role": "user",
"content": "What did I say my name was? And what programming language did I mention?",
},
],
"stream": False,
}
response_2 = await async_client.post("/v1/chat/completions", json=request_data_2, timeout=120.0)
assert response_2.status_code == 200
data_2 = response_2.json()
second_response = data_2["choices"][0]["message"]["content"].lower()
# Verify Tatlock remembers the name and programming language
has_alice = "alice" in second_response
has_python = "python" in second_response
if not has_alice or not has_python:
pytest.xfail(
f"LLM did not remember context (non-deterministic): alice={has_alice}, python={has_python}, response: {second_response[:200]}"
)
@pytest.mark.integration
@pytest.mark.asyncio
async def test_tatlock_multi_turn_context(async_client: AsyncClient):
"""
Test that Tatlock maintains context over multiple turns.
Verifies conversation history is properly accumulated.
Note: This test may fail due to LLM non-determinism.
"""
# Build a multi-turn conversation
conversation = []
# Turn 1: Set up a topic
conversation.append({"role": "user", "content": "Let's talk about the number 42."})
request_1 = {"model": "Tatlock", "messages": conversation.copy(), "stream": False}
response_1 = await async_client.post("/v1/chat/completions", json=request_1, timeout=120.0)
assert response_1.status_code == 200
data_1 = response_1.json()
conversation.append(
{"role": "assistant", "content": data_1["choices"][0]["message"]["content"]}
)
# Turn 2: Reference "it" (should refer to 42)
conversation.append({"role": "user", "content": "What number did I just mention?"})
request_2 = {"model": "Tatlock", "messages": conversation.copy(), "stream": False}
response_2 = await async_client.post("/v1/chat/completions", json=request_2, timeout=120.0)
assert response_2.status_code == 200
data_2 = response_2.json()
final_response = data_2["choices"][0]["message"]["content"]
# Should reference 42 (check both as digit and word)
has_42 = (
"42" in final_response
or "forty-two" in final_response.lower()
or "forty two" in final_response.lower()
)
if not has_42:
pytest.xfail(
f"LLM did not mention 42 in response (non-deterministic): {final_response[:200]}"
)
@pytest.mark.integration
@pytest.mark.asyncio
async def test_tatlock_tool_call_logging_search(async_client: AsyncClient):
"""
Test that web search tool calls are logged to reasoning output.
This verifies that when Tatlock uses the search tool, the query
is visible in the chat response (in <think> tags).
"""
request_data = {
"model": "Tatlock",
"messages": [
{
"role": "user",
"content": "Search for current information about Python 3.13 release date",
}
],
"stream": False,
}
response = await async_client.post("/v1/chat/completions", json=request_data, timeout=120.0)
assert response.status_code == 200
data = response.json()
full_response = data["choices"][0]["message"]["content"]
# Tool calls should appear in <think> tags
assert "<think>" in full_response, "Should have reasoning/tool output in <think> tags"
# Should contain search indicator emoji (if search was used)
# OR the LLM might answer without searching if it has the info
# So we just verify the mechanism works by checking for think tags
print(f"\nFull response with tool logging:\n{full_response}")
# If search was used, should show the 🔍 emoji
if "🔍" in full_response:
assert (
"search" in full_response.lower() or "python" in full_response.lower()
), "Search query should be visible in the response"
@pytest.mark.integration
@pytest.mark.asyncio
async def test_tatlock_tool_call_logging_calculator(async_client: AsyncClient):
"""
Test that calculator requests are handled correctly.
Verifies that mathematical calculations produce correct results.
Note: Tool call logging visibility depends on execution path
(streaming vs run, scoped tools vs delegation).
"""
request_data = {
"model": "Tatlock",
"messages": [{"role": "user", "content": "What is the square root of 144 plus 25?"}],
"stream": False,
}
response = await async_client.post("/v1/chat/completions", json=request_data, timeout=120.0)
assert response.status_code == 200
data = response.json()
full_response = data["choices"][0]["message"]["content"]
# Should have reasoning in <think> tags (from Steward analysis)
assert (
"<think>" in full_response
), f"Should have reasoning output in <think> tags. Got: {full_response}"
# Should reference the calculation in some form
has_calculation_reference = (
"144" in full_response
or "sqrt" in full_response.lower()
or "square root" in full_response.lower()
)
assert has_calculation_reference, f"Should reference the calculation. Got: {full_response}"
# Should have the correct answer (37)
assert "37" in full_response, f"Should contain the answer 37. Got: {full_response}"
# Tool emoji is optional - depends on whether tool was used directly
# or computation was delegated to capability
if "🧮" in full_response:
print("\nCalculator tool was used directly")
else:
print("\nCalculation handled via tatlock_core capability")
print(f"\nCalculator response: {full_response}")
@pytest.mark.integration
@pytest.mark.asyncio
async def test_tatlock_tool_call_logging_datetime(async_client: AsyncClient):
"""
Test that date/time tool calls are logged to reasoning output.
"""
request_data = {
"model": "Tatlock",
"messages": [{"role": "user", "content": "What was the date exactly 2 weeks ago?"}],
"stream": False,
}
response = await async_client.post("/v1/chat/completions", json=request_data, timeout=120.0)
assert response.status_code == 200
data = response.json()
full_response = data["choices"][0]["message"]["content"]
# Should have reasoning in <think> tags
assert "<think>" in full_response, "Should have reasoning output in <think> tags"
# Check if date/time tool was used (LLM might calculate it itself sometimes)
used_date_tool = "🕐" in full_response
# Should mention the calculation or the timeframe
assert (
"2 weeks ago" in full_response.lower() or "weeks" in full_response.lower()
), f"Should reference the requested timeframe. Got: {full_response}"
# Should provide a specific date (either YYYY-MM-DD format or natural language like "November 23")
import re
has_iso_date = bool(re.search(r"\d{4}-\d{2}-\d{2}", full_response))
has_month_mention = any(
month in full_response.lower()
for month in [
"january",
"february",
"march",
"april",
"may",
"june",
"july",
"august",
"september",
"october",
"november",
"december",
]
)
has_date_number = bool(re.search(r"\b\d{1,2}(st|nd|rd|th)?\b", full_response.lower()))
assert (
has_iso_date or has_month_mention or has_date_number
), f"Should contain a specific date. Got: {full_response}"
print(f"\nDate/time response (tool used: {used_date_tool}): {full_response}")
@pytest.mark.integration
@pytest.mark.asyncio
async def test_tatlock_no_tool_calls_no_logging(async_client: AsyncClient):
"""
Test that when no tools are used, no tool logging appears.
Verifies the tool logging only appears when tools are actually called.
"""
request_data = {
"model": "Tatlock",
"messages": [{"role": "user", "content": "Just say hello to me."}],
"stream": False,
}
response = await async_client.post("/v1/chat/completions", json=request_data, timeout=120.0)
assert response.status_code == 200
data = response.json()
full_response = data["choices"][0]["message"]["content"]
# Should have basic reasoning in <think> tags
assert "<think>" in full_response, "Should have reasoning output in <think> tags"
# Should NOT have tool emojis (for a simple greeting)
has_tool_emoji = any(emoji in full_response for emoji in ["🔍", "🧮", "🕐"])
print(f"\nResponse without tools: {full_response}")
print(f"Has tool emojis: {has_tool_emoji}")
# Just verify we got a greeting response
assert len(full_response) > 0, "Should have a response"
@pytest.mark.integration
@pytest.mark.asyncio
async def test_tatlock_conversation_history_with_tools(async_client: AsyncClient):
"""
Test that conversation history works correctly when tools are used.
Combines both features: history + tool logging.
Note: This test may fail due to LLM non-determinism.
"""
conversation = []
# Turn 1: Do a calculation
conversation.append({"role": "user", "content": "Calculate 15 times 7 for me."})
request_1 = {"model": "Tatlock", "messages": conversation.copy(), "stream": False}
response_1 = await async_client.post("/v1/chat/completions", json=request_1, timeout=120.0)
assert response_1.status_code == 200
data_1 = response_1.json()
first_response = data_1["choices"][0]["message"]["content"]
# Should contain the answer (105) - allow for number formatting
has_105 = "105" in first_response.replace(",", "")
if not has_105:
pytest.xfail(f"LLM did not calculate 15*7=105 (non-deterministic): {first_response[:200]}")
conversation.append({"role": "assistant", "content": first_response})
# Turn 2: Ask about previous calculation
conversation.append({"role": "user", "content": "What calculation did I just ask you to do?"})
request_2 = {"model": "Tatlock", "messages": conversation.copy(), "stream": False}
response_2 = await async_client.post("/v1/chat/completions", json=request_2, timeout=120.0)
assert response_2.status_code == 200
data_2 = response_2.json()
second_response = data_2["choices"][0]["message"]["content"].lower()
# Should remember the calculation (either as digits or words)
has_calculation = (
("15" in second_response and "7" in second_response) # As digits
or ("fifteen" in second_response and "seven" in second_response) # As words
or "105" in second_response # As answer
or "multipl" in second_response # Mentions multiplication
)
if not has_calculation:
pytest.xfail(
f"LLM did not remember calculation (non-deterministic): {second_response[:200]}"
)
@pytest.mark.integration
@pytest.mark.asyncio
async def test_tatlock_ollama_fallback(async_client: AsyncClient):
"""
Test that Tatlock falls back to Ollama when Claude is unavailable.
Patches _claude_available to False to force the Ollama path,
then verifies the system still produces a valid response.
"""
import src.anthropic.model_selector as model_selector
# Save original value
original = model_selector._claude_available
try:
# Force Ollama fallback
model_selector._claude_available = False
# Verify we're actually using Ollama
info = model_selector.get_model_info()
assert info["backend"] == "ollama", f"Expected ollama backend, got {info['backend']}"
request_data = {
"model": "Tatlock",
"messages": [{"role": "user", "content": "Say hello to me."}],
"stream": False,
}
# 300s: this test forbids the Claude rescue, and the full local
# Steward -> orchestrate -> synthesize flow on gemma4 exceeds 120s
response = await async_client.post("/v1/chat/completions", json=request_data, timeout=300.0)
assert response.status_code == 200
data = response.json()
# Verify response structure is valid
assert "choices" in data
assert len(data["choices"]) == 1
full_response = data["choices"][0]["message"]["content"]
assert len(full_response) > 0, "Ollama should produce a non-empty response"
print(f"\nOllama fallback response: {full_response[:200]}")
finally:
# Restore original value
model_selector._claude_available = original