feat: implement Phase 2 two-tier architecture with Steward

Add comprehensive two-tier architecture where Steward analyzes requests
and Tatlock executes with scoped tools. Includes full infrastructure for
request preprocessing, tool tracking, benchmarking, and streaming.

**Added:**
- Steward agent for request analysis and capability recommendation
- Household Registry for centralized capability management
- Request preprocessing pipeline (Steward → Tatlock flow)
- Tool usage tracking and benchmarking system
- Streaming transparency (Steward reasoning visible in streams)
- Structured logging with operation timing
- Redis benchmark storage with 30-day expiry
- Benchmark analysis CLI tools

**Infrastructure:**
- src/agents/steward/ - Steward agent implementation
- src/agents/tatlock_core/ - Tatlock capability domain
- src/core/preprocessing.py - Request preprocessing pipeline
- src/core/tool_tracking.py - Tool call tracking
- src/core/benchmarks.py - Benchmark recording system
- src/core/household_registry.py - Capability registry
- src/core/startup.py - Application startup coordination
- src/core/logging_config.py - Structured logging setup

**Integration:**
- Responses API uses Steward for Tatlock requests
- Chat Completions wraps Responses API for OpenAI compatibility
- Streaming coordinator supports Steward + Tatlock flow
- Tool scoping per request based on Steward recommendations

**Testing:**
- Integration tests for Steward-Tatlock flow
- Benchmark and registry unit tests
- Steward streaming tests

See PHASE2_PLAN.md and PHASE2_COMPLETE.md for detailed documentation.

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude Sonnet 4.5 <noreply@anthropic.com>
This commit is contained in:
2025-12-07 15:39:20 +01:00
co-authored by Claude Sonnet 4.5
parent 2577730546
commit 6eed5f4d13
34 changed files with 6362 additions and 133 deletions
+1
View File
@@ -0,0 +1 @@
"""Tests for the Steward agent."""
@@ -0,0 +1,166 @@
"""
Tests for Steward schemas.
Tests the structured output models for conversation context and recommendations.
"""
import pytest
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
class TestConversationContext:
"""Test ConversationContext model."""
def test_context_creation_with_defaults(self):
"""Test creating context with default values."""
context = ConversationContext(has_previous_context=False)
assert context.has_previous_context is False
assert context.relevant_turns == []
assert context.context_summary == ""
def test_context_creation_with_values(self):
"""Test creating context with explicit values."""
context = ConversationContext(
has_previous_context=True,
relevant_turns=[0, 2, 4],
context_summary="User discussed weather in turns 0 and 2"
)
assert context.has_previous_context is True
assert context.relevant_turns == [0, 2, 4]
assert "weather" in context.context_summary
class TestStewardRecommendation:
"""Test StewardRecommendation model."""
def test_recommendation_simple(self):
"""Test simple recommendation with no capabilities needed."""
rec = StewardRecommendation(
recommended_capabilities=[],
reasoning="Simple greeting requires no tools",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
assert rec.recommended_capabilities == []
assert rec.estimated_complexity == "simple"
assert rec.missing_capabilities is None
def test_recommendation_with_capabilities(self):
"""Test recommendation with specific capabilities."""
rec = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Mathematical calculation requires calculator",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
assert "tatlock_core" in rec.recommended_capabilities
assert rec.estimated_complexity == "simple"
def test_recommendation_with_missing_capabilities(self):
"""Test recommendation noting missing capabilities."""
rec = StewardRecommendation(
recommended_capabilities=[],
reasoning="Image generation is not available",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
missing_capabilities="Image generation capability would be needed",
)
assert rec.missing_capabilities is not None
assert "Image generation" in rec.missing_capabilities
def test_recommendation_complexity_levels(self):
"""Test all complexity levels."""
for complexity in ["simple", "moderate", "complex"]:
rec = StewardRecommendation(
recommended_capabilities=[],
reasoning=f"Testing {complexity} complexity",
estimated_complexity=complexity,
conversation_context=ConversationContext(has_previous_context=False),
)
assert rec.estimated_complexity == complexity
def test_recommendation_with_context(self):
"""Test recommendation with conversation context."""
context = ConversationContext(
has_previous_context=True,
relevant_turns=[1, 3],
context_summary="User asked about calculation in turn 1, now wants explanation"
)
rec = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="User wants explanation of previous calculation",
estimated_complexity="moderate",
conversation_context=context,
)
assert rec.conversation_context.has_previous_context is True
assert len(rec.conversation_context.relevant_turns) == 2
def test_format_for_butler_simple(self):
"""Test formatting recommendation for Butler - simple case."""
rec = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Math calculation needed",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
formatted = rec.format_for_butler()
assert "📋 Steward's Analysis" in formatted
assert "SIMPLE" in formatted
assert "tatlock_core" in formatted
def test_format_for_butler_with_context(self):
"""Test formatting with conversation context."""
context = ConversationContext(
has_previous_context=True,
relevant_turns=[0],
context_summary="Previous calculation mentioned"
)
rec = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Follow-up calculation",
estimated_complexity="moderate",
conversation_context=context,
)
formatted = rec.format_for_butler()
assert "Context:" in formatted
assert "Previous calculation" in formatted
def test_format_for_butler_with_missing_capabilities(self):
"""Test formatting with missing capabilities warning."""
rec = StewardRecommendation(
recommended_capabilities=[],
reasoning="No suitable tools available",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
missing_capabilities="Image generation would be needed",
)
formatted = rec.format_for_butler()
assert "⚠️ Missing:" in formatted
assert "Image generation" in formatted
def test_format_for_butler_no_capabilities(self):
"""Test formatting when no tools needed (conversational)."""
rec = StewardRecommendation(
recommended_capabilities=[],
reasoning="Simple greeting",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
formatted = rec.format_for_butler()
assert "None (conversational response)" in formatted
@@ -0,0 +1,201 @@
"""
Tests for Steward service layer.
Tests request analysis, logging, and benchmarking integration.
"""
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
from src.agents.steward.service import analyze_request, format_steward_note
from src.core.startup import initialize_application
@pytest.fixture(scope="module", autouse=True)
def setup_household_registry():
"""Initialize household registry before running tests."""
initialize_application()
class TestAnalyzeRequest:
"""Test the analyze_request service function."""
@pytest.mark.asyncio
async def test_analyze_simple_greeting(self):
"""Test analyzing a simple greeting."""
# Mock the Steward agent's analyze method (plain text approach)
mock_agent = MagicMock()
mock_agent.analyze = AsyncMock(return_value="Simple greeting requires no tools. This is a simple request.")
with patch("src.agents.steward.service.get_steward_agent", return_value=mock_agent):
with patch("src.agents.steward.service.get_benchmark_store") as mock_store:
mock_store.return_value.record = AsyncMock()
result = await analyze_request(
"Hello!",
conversation_history=[],
)
assert result.recommended_capabilities == []
assert result.estimated_complexity == "simple"
assert mock_agent.analyze.called
@pytest.mark.asyncio
async def test_analyze_math_request(self):
"""Test analyzing a mathematical request."""
mock_agent = MagicMock()
mock_agent.analyze = AsyncMock(
return_value="Mathematical calculation requires tatlock_core for solving this simple problem."
)
with patch("src.agents.steward.service.get_steward_agent", return_value=mock_agent):
with patch("src.agents.steward.service.get_benchmark_store") as mock_store:
mock_store.return_value.record = AsyncMock()
result = await analyze_request(
"What's sqrt(144)?",
conversation_history=[],
)
assert "tatlock_core" in result.recommended_capabilities
assert result.estimated_complexity == "simple"
@pytest.mark.asyncio
async def test_analyze_with_conversation_history(self):
"""Test analyzing with previous conversation context."""
mock_agent = MagicMock()
mock_agent.analyze = AsyncMock(
return_value="Follow-up to previous calculation in turn 0. Requires tatlock_core. Complexity: moderate."
)
conversation_history = [
{"role": "user", "content": "What's 2 + 2?"},
{"role": "assistant", "content": "4"},
]
with patch("src.agents.steward.service.get_steward_agent", return_value=mock_agent):
with patch("src.agents.steward.service.get_benchmark_store") as mock_store:
mock_store.return_value.record = AsyncMock()
result = await analyze_request(
"And what's that times 5?",
conversation_history=conversation_history,
)
assert result.conversation_context.has_previous_context is True
assert 0 in result.conversation_context.relevant_turns
# Verify conversation history was passed
call_kwargs = mock_agent.analyze.call_args.kwargs
assert "conversation_history" in call_kwargs
assert len(call_kwargs["conversation_history"]) == 2
@pytest.mark.asyncio
async def test_analyze_with_missing_capabilities(self):
"""Test analyzing request that needs unavailable capabilities."""
mock_agent = MagicMock()
mock_agent.analyze = AsyncMock(
return_value="Image generation not available. Would be needed for this request. Complexity: simple."
)
with patch("src.agents.steward.service.get_steward_agent", return_value=mock_agent):
with patch("src.agents.steward.service.get_benchmark_store") as mock_store:
mock_store.return_value.record = AsyncMock()
result = await analyze_request(
"Generate an image of a sunset",
conversation_history=[],
)
assert result.missing_capabilities is not None
assert "not available" in result.missing_capabilities
@pytest.mark.asyncio
async def test_analyze_with_conversation_id(self):
"""Test that analysis includes conversation ID in context."""
mock_agent = MagicMock()
mock_agent.analyze = AsyncMock(
return_value="This simple request requires tatlock_core to solve."
)
with patch("src.agents.steward.service.get_steward_agent", return_value=mock_agent):
with patch("src.agents.steward.service.get_benchmark_store") as mock_store:
mock_store.return_value.record = AsyncMock()
result = await analyze_request(
"Test request",
conversation_history=[],
conversation_id="test_conv_123",
)
# Verify analysis completed successfully
assert result.recommended_capabilities == ["tatlock_core"]
assert result.estimated_complexity == "simple"
@pytest.mark.asyncio
async def test_analyze_handles_errors(self):
"""Test error handling in analyze_request."""
mock_agent = MagicMock()
mock_agent.analyze = AsyncMock(side_effect=Exception("Test error"))
with patch("src.agents.steward.service.get_steward_agent", return_value=mock_agent):
with pytest.raises(Exception, match="Test error"):
await analyze_request("Test", conversation_history=[])
class TestFormatStewardNote:
"""Test the format_steward_note function."""
@pytest.mark.asyncio
async def test_format_simple_note(self):
"""Test formatting a simple recommendation."""
rec = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Math needed",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
note = await format_steward_note(rec)
assert "📋 Steward's Analysis" in note
assert "SIMPLE" in note
assert "tatlock_core" in note
@pytest.mark.asyncio
async def test_format_note_with_context(self):
"""Test formatting note with conversation context."""
context = ConversationContext(
has_previous_context=True,
relevant_turns=[0, 1],
context_summary="Previous discussion about calculations"
)
rec = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Follow-up calculation",
estimated_complexity="moderate",
conversation_context=context,
)
note = await format_steward_note(rec)
assert "Context:" in note
assert "Previous discussion" in note
@pytest.mark.asyncio
async def test_format_note_with_missing_capabilities(self):
"""Test formatting note with missing capabilities warning."""
rec = StewardRecommendation(
recommended_capabilities=[],
reasoning="Not available",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
missing_capabilities="Advanced research tools needed",
)
note = await format_steward_note(rec)
assert "⚠️ Missing:" in note
assert "Advanced research" in note