Mechanical only, and separated from the judgment calls that follow so the reviewable changes are not buried in a 98-file whitespace diff. 227 automatic fixes: 60 blank lines carrying whitespace, 60 unsorted import blocks, 34 Optional[X] to X | None, 28 unused imports, 16 deprecated typing imports, 12 datetime.timezone.utc to datetime.UTC, and assorted smaller modernisations. Then `ruff format` over src and tests: 98 files reformatted, 35 already conforming. No file among the unused-import findings defines __all__ or is an __init__.py, so nothing here removes a re-export. `make test`: 658 passed, unchanged from HEAD. Two things observed while verifying, neither addressed here: `pytest tests/` cannot collect — tests/e2e/test_orchestration_e2e.py uses an `e2e` marker that is not registered, and the config is strict about markers. This fails identically at HEAD, so it predates this change; `make test` passes because it ignores tests/e2e, tests/integration and tests/contracts. test_tatlock_tool_call_logging_calculator is flaky. It failed once in a full run with these changes and passed on the next, passes in isolation with them, and fails in isolation at HEAD. It is order- or timing-dependent, not a regression from this commit — established by running the full suite both ways rather than by reasoning about which change could have caused it. Co-Authored-By: Claude <noreply@anthropic.com>
166 lines
6.2 KiB
Python
166 lines
6.2 KiB
Python
"""
|
|
Tests for Steward schemas.
|
|
|
|
Tests the structured output models for conversation context and recommendations.
|
|
"""
|
|
|
|
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
|
|
|
|
|
|
class TestConversationContext:
|
|
"""Test ConversationContext model."""
|
|
|
|
def test_context_creation_with_defaults(self):
|
|
"""Test creating context with default values."""
|
|
context = ConversationContext(has_previous_context=False)
|
|
|
|
assert context.has_previous_context is False
|
|
assert context.relevant_turns == []
|
|
assert context.context_summary == ""
|
|
|
|
def test_context_creation_with_values(self):
|
|
"""Test creating context with explicit values."""
|
|
context = ConversationContext(
|
|
has_previous_context=True,
|
|
relevant_turns=[0, 2, 4],
|
|
context_summary="User discussed weather in turns 0 and 2",
|
|
)
|
|
|
|
assert context.has_previous_context is True
|
|
assert context.relevant_turns == [0, 2, 4]
|
|
assert "weather" in context.context_summary
|
|
|
|
|
|
class TestStewardRecommendation:
|
|
"""Test StewardRecommendation model."""
|
|
|
|
def test_recommendation_simple(self):
|
|
"""Test simple recommendation with no capabilities needed."""
|
|
rec = StewardRecommendation(
|
|
recommended_capabilities=[],
|
|
reasoning="Simple greeting requires no tools",
|
|
estimated_complexity="simple",
|
|
conversation_context=ConversationContext(has_previous_context=False),
|
|
)
|
|
|
|
assert rec.recommended_capabilities == []
|
|
assert rec.estimated_complexity == "simple"
|
|
assert rec.missing_capabilities is None
|
|
|
|
def test_recommendation_with_capabilities(self):
|
|
"""Test recommendation with specific capabilities."""
|
|
rec = StewardRecommendation(
|
|
recommended_capabilities=["tatlock_core"],
|
|
reasoning="Mathematical calculation requires calculator",
|
|
estimated_complexity="simple",
|
|
conversation_context=ConversationContext(has_previous_context=False),
|
|
)
|
|
|
|
assert "tatlock_core" in rec.recommended_capabilities
|
|
assert rec.estimated_complexity == "simple"
|
|
|
|
def test_recommendation_with_missing_capabilities(self):
|
|
"""Test recommendation noting missing capabilities."""
|
|
rec = StewardRecommendation(
|
|
recommended_capabilities=[],
|
|
reasoning="Image generation is not available",
|
|
estimated_complexity="simple",
|
|
conversation_context=ConversationContext(has_previous_context=False),
|
|
missing_capabilities="Image generation capability would be needed",
|
|
)
|
|
|
|
assert rec.missing_capabilities is not None
|
|
assert "Image generation" in rec.missing_capabilities
|
|
|
|
def test_recommendation_complexity_levels(self):
|
|
"""Test all complexity levels."""
|
|
for complexity in ["simple", "moderate", "complex"]:
|
|
rec = StewardRecommendation(
|
|
recommended_capabilities=[],
|
|
reasoning=f"Testing {complexity} complexity",
|
|
estimated_complexity=complexity,
|
|
conversation_context=ConversationContext(has_previous_context=False),
|
|
)
|
|
assert rec.estimated_complexity == complexity
|
|
|
|
def test_recommendation_with_context(self):
|
|
"""Test recommendation with conversation context."""
|
|
context = ConversationContext(
|
|
has_previous_context=True,
|
|
relevant_turns=[1, 3],
|
|
context_summary="User asked about calculation in turn 1, now wants explanation",
|
|
)
|
|
|
|
rec = StewardRecommendation(
|
|
recommended_capabilities=["tatlock_core"],
|
|
reasoning="User wants explanation of previous calculation",
|
|
estimated_complexity="moderate",
|
|
conversation_context=context,
|
|
)
|
|
|
|
assert rec.conversation_context.has_previous_context is True
|
|
assert len(rec.conversation_context.relevant_turns) == 2
|
|
|
|
def test_format_for_butler_simple(self):
|
|
"""Test formatting recommendation for Butler - simple case."""
|
|
rec = StewardRecommendation(
|
|
recommended_capabilities=["tatlock_core"],
|
|
reasoning="Math calculation needed",
|
|
estimated_complexity="simple",
|
|
conversation_context=ConversationContext(has_previous_context=False),
|
|
)
|
|
|
|
formatted = rec.format_for_butler()
|
|
|
|
assert "📋 Steward's Analysis" in formatted
|
|
assert "SIMPLE" in formatted
|
|
assert "tatlock_core" in formatted
|
|
|
|
def test_format_for_butler_with_context(self):
|
|
"""Test formatting with conversation context."""
|
|
context = ConversationContext(
|
|
has_previous_context=True,
|
|
relevant_turns=[0],
|
|
context_summary="Previous calculation mentioned",
|
|
)
|
|
|
|
rec = StewardRecommendation(
|
|
recommended_capabilities=["tatlock_core"],
|
|
reasoning="Follow-up calculation",
|
|
estimated_complexity="moderate",
|
|
conversation_context=context,
|
|
)
|
|
|
|
formatted = rec.format_for_butler()
|
|
|
|
assert "Context:" in formatted
|
|
assert "Previous calculation" in formatted
|
|
|
|
def test_format_for_butler_with_missing_capabilities(self):
|
|
"""Test formatting with missing capabilities warning."""
|
|
rec = StewardRecommendation(
|
|
recommended_capabilities=[],
|
|
reasoning="No suitable tools available",
|
|
estimated_complexity="simple",
|
|
conversation_context=ConversationContext(has_previous_context=False),
|
|
missing_capabilities="Image generation would be needed",
|
|
)
|
|
|
|
formatted = rec.format_for_butler()
|
|
|
|
assert "⚠️ Missing:" in formatted
|
|
assert "Image generation" in formatted
|
|
|
|
def test_format_for_butler_no_capabilities(self):
|
|
"""Test formatting when no tools needed (conversational)."""
|
|
rec = StewardRecommendation(
|
|
recommended_capabilities=[],
|
|
reasoning="Simple greeting",
|
|
estimated_complexity="simple",
|
|
conversation_context=ConversationContext(has_previous_context=False),
|
|
)
|
|
|
|
formatted = rec.format_for_butler()
|
|
|
|
assert "None (conversational response)" in formatted
|