Files
tatlock/tests/agents/test_lorem_tester.py
T
jpmschweitzerandClaude 78066fab1b style: apply ruff's automatic fixes and formatter
Mechanical only, and separated from the judgment calls that follow so the
reviewable changes are not buried in a 98-file whitespace diff.

227 automatic fixes: 60 blank lines carrying whitespace, 60 unsorted import
blocks, 34 Optional[X] to X | None, 28 unused imports, 16 deprecated typing
imports, 12 datetime.timezone.utc to datetime.UTC, and assorted smaller
modernisations. Then `ruff format` over src and tests: 98 files reformatted,
35 already conforming.

No file among the unused-import findings defines __all__ or is an __init__.py,
so nothing here removes a re-export.

`make test`: 658 passed, unchanged from HEAD.

Two things observed while verifying, neither addressed here:

`pytest tests/` cannot collect — tests/e2e/test_orchestration_e2e.py uses an
`e2e` marker that is not registered, and the config is strict about markers.
This fails identically at HEAD, so it predates this change; `make test` passes
because it ignores tests/e2e, tests/integration and tests/contracts.

test_tatlock_tool_call_logging_calculator is flaky. It failed once in a full run
with these changes and passed on the next, passes in isolation with them, and
fails in isolation at HEAD. It is order- or timing-dependent, not a regression
from this commit — established by running the full suite both ways rather than
by reasoning about which change could have caused it.

Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-11 17:25:18 +02:00

197 lines
5.5 KiB
Python

"""
Tests for Lorem Tester agent.
"""
import pytest
from src.agents.lorem_tester import LoremTesterAgent
from src.core.exceptions import (
APIError,
ContextLengthError,
RateLimitError,
)
@pytest.mark.unit
@pytest.mark.asyncio
async def test_lorem_tester_basic_response():
"""Test basic lorem tester response without reasoning or tools."""
agent = LoremTesterAgent()
messages = [{"role": "user", "content": "Hello"}]
items = []
async for item in agent.generate_response(messages):
items.append(item)
# Should have at least one message item
assert len(items) >= 1
# Last item should be message
last_item = items[-1]
assert last_item.type == "message"
assert last_item.data["role"] == "assistant"
assert last_item.data["content"][0]["type"] == "output_text"
assert len(last_item.data["content"][0]["text"]) > 0
@pytest.mark.unit
@pytest.mark.asyncio
async def test_lorem_tester_with_reasoning():
"""Test lorem tester with reasoning enabled."""
agent = LoremTesterAgent()
messages = [{"role": "user", "content": "Explain something"}]
reasoning = {"summary": "auto", "effort": "medium"}
items = []
async for item in agent.generate_response(messages, reasoning=reasoning):
items.append(item)
# Should have reasoning item and message item
assert len(items) >= 2
# First item should be reasoning
reasoning_item = items[0]
assert reasoning_item.type == "reasoning"
assert "summary" in reasoning_item.data
assert isinstance(reasoning_item.data["summary"], list)
assert len(reasoning_item.data["summary"]) > 0
# Last item should be message
message_item = items[-1]
assert message_item.type == "message"
@pytest.mark.unit
@pytest.mark.asyncio
async def test_lorem_tester_reasoning_effort_levels():
"""Test different reasoning effort levels."""
agent = LoremTesterAgent()
messages = [{"role": "user", "content": "Test"}]
# Test different effort levels
efforts = ["minimal", "low", "medium", "high", "xhigh"]
for effort in efforts:
reasoning = {"summary": "auto", "effort": effort}
items = []
async for item in agent.generate_response(messages, reasoning=reasoning):
if item.type == "reasoning":
items.append(item)
# Should have reasoning item
assert len(items) >= 1
reasoning_item = items[0]
assert reasoning_item.type == "reasoning"
@pytest.mark.unit
@pytest.mark.asyncio
async def test_lorem_tester_with_tools():
"""Test lorem tester with tools (may or may not call them)."""
agent = LoremTesterAgent()
messages = [{"role": "user", "content": "Use a tool"}]
tools = [
{"name": "search_knowledge", "description": "Search knowledge base"},
{"name": "calculate", "description": "Do math"},
]
items = []
async for item in agent.generate_response(messages, tools=tools):
items.append(item)
# Should have at least message item
# May have function_call items (randomized)
assert len(items) >= 1
# Check item types
for item in items:
assert item.type in ["reasoning", "function_call", "message"]
@pytest.mark.unit
@pytest.mark.asyncio
async def test_lorem_tester_capabilities():
"""Test lorem tester capabilities."""
agent = LoremTesterAgent()
assert await agent.supports_tools() is True
assert await agent.supports_reasoning() is True
capabilities = await agent.get_capabilities()
assert capabilities["streaming"] is True
assert capabilities["reasoning"] is True
assert capabilities["tools"] is True
@pytest.mark.unit
@pytest.mark.asyncio
async def test_lorem_tester_rate_limit_trigger():
"""Test rate limit error trigger."""
agent = LoremTesterAgent()
messages = [{"role": "user", "content": "trigger_rate_limit"}]
with pytest.raises(RateLimitError) as exc_info:
async for item in agent.generate_response(messages):
pass
assert "rate limit" in str(exc_info.value).lower()
@pytest.mark.unit
@pytest.mark.asyncio
async def test_lorem_tester_context_overflow_trigger():
"""Test context length error trigger."""
agent = LoremTesterAgent()
messages = [{"role": "user", "content": "trigger_context_overflow"}]
with pytest.raises(ContextLengthError) as exc_info:
async for item in agent.generate_response(messages):
pass
assert "context" in str(exc_info.value).lower()
@pytest.mark.unit
@pytest.mark.asyncio
async def test_lorem_tester_invalid_tool_trigger():
"""Test invalid tool error trigger."""
agent = LoremTesterAgent()
messages = [{"role": "user", "content": "trigger_invalid_tool"}]
with pytest.raises(APIError) as exc_info:
async for item in agent.generate_response(messages):
pass
assert "tool" in str(exc_info.value).lower()
@pytest.mark.unit
@pytest.mark.asyncio
async def test_lorem_tester_temperature_variation():
"""Test temperature affects response variety."""
agent = LoremTesterAgent()
messages = [{"role": "user", "content": "Test"}]
# Low temperature
items_low = []
async for item in agent.generate_response(messages, temperature=0.1):
if item.type == "message":
items_low.append(item)
# High temperature
items_high = []
async for item in agent.generate_response(messages, temperature=1.5):
if item.type == "message":
items_high.append(item)
# Both should have responses
assert len(items_low) >= 1
assert len(items_high) >= 1