feat: implement Phase 2 two-tier architecture with Steward

Add comprehensive two-tier architecture where Steward analyzes requests
and Tatlock executes with scoped tools. Includes full infrastructure for
request preprocessing, tool tracking, benchmarking, and streaming.

**Added:**
- Steward agent for request analysis and capability recommendation
- Household Registry for centralized capability management
- Request preprocessing pipeline (Steward → Tatlock flow)
- Tool usage tracking and benchmarking system
- Streaming transparency (Steward reasoning visible in streams)
- Structured logging with operation timing
- Redis benchmark storage with 30-day expiry
- Benchmark analysis CLI tools

**Infrastructure:**
- src/agents/steward/ - Steward agent implementation
- src/agents/tatlock_core/ - Tatlock capability domain
- src/core/preprocessing.py - Request preprocessing pipeline
- src/core/tool_tracking.py - Tool call tracking
- src/core/benchmarks.py - Benchmark recording system
- src/core/household_registry.py - Capability registry
- src/core/startup.py - Application startup coordination
- src/core/logging_config.py - Structured logging setup

**Integration:**
- Responses API uses Steward for Tatlock requests
- Chat Completions wraps Responses API for OpenAI compatibility
- Streaming coordinator supports Steward + Tatlock flow
- Tool scoping per request based on Steward recommendations

**Testing:**
- Integration tests for Steward-Tatlock flow
- Benchmark and registry unit tests
- Steward streaming tests

See PHASE2_PLAN.md and PHASE2_COMPLETE.md for detailed documentation.

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude Sonnet 4.5 <noreply@anthropic.com>
This commit is contained in:
2025-12-07 15:39:20 +01:00
co-authored by Claude Sonnet 4.5
parent 2577730546
commit 6eed5f4d13
34 changed files with 6362 additions and 133 deletions
+190
View File
@@ -0,0 +1,190 @@
"""
Integration tests for Steward + Tatlock streaming.
Tests the complete streaming flow with Steward preprocessing.
"""
import pytest
from unittest.mock import AsyncMock, MagicMock, patch
from src.responses.schemas import ResponseRequest
from src.responses.streaming import StreamingCoordinator, StreamEventType
from src.core.startup import initialize_application
@pytest.fixture(scope="module", autouse=True)
def setup_household_registry():
"""Initialize household registry before running tests."""
initialize_application()
class TestStewardStreaming:
"""Test Steward + Tatlock streaming integration."""
@pytest.mark.asyncio
async def test_stream_with_steward_basic(self):
"""Test basic streaming with Steward preprocessing."""
request = ResponseRequest(
model="tatlock",
input=[{"role": "user", "content": "What's 2 + 2?"}],
stream=True,
)
# Mock the Steward analysis
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
# Mock Steward recommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Math calculation requires tatlock_core",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
# Mock Tatlock response
mock_tatlock.return_value = "Certainly, sir. 2 + 2 equals 4."
# Execute streaming
coordinator = StreamingCoordinator()
events = []
async for event in coordinator.stream_response_with_steward(request):
events.append(event)
# Verify event sequence
event_types = [e.event for e in events]
# Should have reasoning summary deltas
assert StreamEventType.REASONING_SUMMARY_DELTA in event_types
assert StreamEventType.REASONING_SUMMARY_DONE in event_types
# Should have output text deltas
assert StreamEventType.OUTPUT_TEXT_DELTA in event_types
assert StreamEventType.OUTPUT_TEXT_DONE in event_types
# Should end with response.done
assert events[-1].event == StreamEventType.RESPONSE_DONE
# Verify Steward and Tatlock were called
assert mock_steward.called
assert mock_tatlock.called
@pytest.mark.asyncio
async def test_stream_with_conversation_history(self):
"""Test streaming with conversation history."""
request = ResponseRequest(
model="tatlock",
input=[
{"role": "user", "content": "What's 5 times 3?"},
{"role": "assistant", "content": "That equals 15, sir."},
{"role": "user", "content": "And divided by 3?"},
],
stream=True,
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Follow-up calculation based on previous result of 15",
estimated_complexity="simple",
conversation_context=ConversationContext(
has_previous_context=True,
relevant_turns=[0],
context_summary="Previous calculation in turn 0"
),
)
mock_tatlock.return_value = "15 divided by 3 equals 5, sir."
coordinator = StreamingCoordinator()
events = []
async for event in coordinator.stream_response_with_steward(request):
events.append(event)
# Verify conversation history was passed to Steward
call_kwargs = mock_steward.call_args[1]
assert "conversation_history" in call_kwargs
assert len(call_kwargs["conversation_history"]) == 2 # First Q&A pair
# Verify final response includes both reasoning and message
final_event = events[-1]
assert final_event.event == StreamEventType.RESPONSE_DONE
assert len(final_event.response.output) == 2 # Reasoning + Message
@pytest.mark.asyncio
async def test_stream_reasoning_contains_steward_analysis(self):
"""Test that reasoning summary contains Steward's analysis."""
request = ResponseRequest(
model="tatlock",
input=[{"role": "user", "content": "Test request"}],
stream=True,
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="This is a test analysis with specific markers",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
mock_tatlock.return_value = "Test response"
coordinator = StreamingCoordinator()
reasoning_deltas = []
async for event in coordinator.stream_response_with_steward(request):
if event.event == StreamEventType.REASONING_SUMMARY_DELTA:
reasoning_deltas.append(event.delta)
# Combine all reasoning deltas
full_reasoning = "".join(reasoning_deltas)
# Should contain Steward's analysis
assert "test analysis" in full_reasoning.lower()
assert len(reasoning_deltas) > 0, "Should have streamed reasoning deltas"
@pytest.mark.asyncio
async def test_stream_with_missing_capabilities(self):
"""Test streaming when Steward detects missing capabilities."""
request = ResponseRequest(
model="tatlock",
input=[{"role": "user", "content": "Generate an image of a sunset"}],
stream=True,
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=[],
reasoning="Image generation not available in current toolset",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
missing_capabilities="Image generation capability would be needed",
)
mock_tatlock.return_value = "I'm afraid I don't have image generation capabilities, sir."
coordinator = StreamingCoordinator()
events = []
async for event in coordinator.stream_response_with_steward(request):
events.append(event)
# Should complete successfully even with missing capabilities
assert events[-1].event == StreamEventType.RESPONSE_DONE
# Verify empty scoped tools were passed
tatlock_kwargs = mock_tatlock.call_args[1]
assert "scoped_tools" in tatlock_kwargs
assert tatlock_kwargs["scoped_tools"] == []
@@ -0,0 +1,249 @@
"""
Integration tests for Steward → Tatlock flow.
Tests the complete Phase 2 request pipeline:
1. Steward analyzes request and recommends capabilities
2. Tool tracker monitors tool usage
3. Tatlock runs with scoped tools
4. Response includes both Steward reasoning and Tatlock output
"""
import pytest
from unittest.mock import AsyncMock, MagicMock, patch
from src.responses.schemas import ResponseRequest
from src.responses.service import create_response_with_steward
from src.core.startup import initialize_application
@pytest.fixture(scope="module", autouse=True)
def setup_household_registry():
"""Initialize household registry before running tests."""
initialize_application()
class TestStewardTatlockIntegration:
"""Test full Steward → Tatlock integration flow."""
@pytest.mark.asyncio
async def test_simple_math_request(self):
"""Test math request flows through Steward → Tatlock correctly."""
# Create a simple math request
request = ResponseRequest(
model="tatlock",
input=[{"role": "user", "content": "What's 2 + 2?"}],
)
# Mock the Steward analysis
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
# Mock Steward recommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Math calculation requires tatlock_core",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
# Mock Tatlock response
mock_tatlock.return_value = "Certainly, sir. 2 + 2 equals 4."
# Execute the flow
response = await create_response_with_steward(request)
# Verify Steward was called
assert mock_steward.called
assert mock_steward.call_args[0][0] == "What's 2 + 2?"
# Verify Tatlock was called with scoped tools
assert mock_tatlock.called
# Verify response structure
assert response.status == "completed"
assert len(response.output) == 2 # Reasoning + Message
# Check Steward reasoning output
reasoning_item = response.output[0]
assert reasoning_item.type == "reasoning"
assert "Math calculation" in reasoning_item.summary[1]
# Check Tatlock message output
message_item = response.output[1]
assert message_item.type == "message"
assert message_item.role == "assistant"
assert "4" in message_item.content[0].text
@pytest.mark.asyncio
async def test_request_with_conversation_history(self):
"""Test that conversation history flows through to Steward."""
request = ResponseRequest(
model="tatlock",
input=[
{"role": "user", "content": "What's 5 times 3?"},
{"role": "assistant", "content": "That equals 15, sir."},
{"role": "user", "content": "And divided by 3?"},
],
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Follow-up calculation",
estimated_complexity="simple",
conversation_context=ConversationContext(
has_previous_context=True,
relevant_turns=[0],
context_summary="Previous calculation in turn 0"
),
)
mock_tatlock.return_value = "15 divided by 3 equals 5, sir."
response = await create_response_with_steward(request)
# Verify Steward received conversation history
call_kwargs = mock_steward.call_args[1]
assert "conversation_history" in call_kwargs
assert len(call_kwargs["conversation_history"]) == 2 # First Q&A pair
# Verify Tatlock received history
tatlock_kwargs = mock_tatlock.call_args[1]
assert "message_history" in tatlock_kwargs
# Verify response completed
assert response.status == "completed"
@pytest.mark.asyncio
async def test_no_capabilities_needed(self):
"""Test simple conversational request that needs no tools."""
request = ResponseRequest(
model="tatlock",
input=[{"role": "user", "content": "Hello!"}],
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=[], # No tools needed
reasoning="Simple greeting, no tools required",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
mock_tatlock.return_value = "Good day, sir. How may I assist you?"
response = await create_response_with_steward(request)
# Verify empty scoped tools were passed
tatlock_kwargs = mock_tatlock.call_args[1]
assert "scoped_tools" in tatlock_kwargs
assert tatlock_kwargs["scoped_tools"] == [] # No tools
assert response.status == "completed"
@pytest.mark.asyncio
async def test_tool_tracker_integration(self):
"""Test that tool tracker is passed to Tatlock and finalized."""
request = ResponseRequest(
model="tatlock",
input=[{"role": "user", "content": "Calculate sqrt(16)"}],
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
with patch("src.core.tool_tracking.ToolCallTracker.finalize") as mock_finalize:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Calculator needed",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
mock_tatlock.return_value = "The square root of 16 is 4, sir."
response = await create_response_with_steward(request)
# Verify tool tracker was finalized
assert mock_finalize.called
assert response.status == "completed"
@pytest.mark.asyncio
async def test_missing_capabilities_warning(self):
"""Test that missing capabilities are included in Steward's reasoning."""
request = ResponseRequest(
model="tatlock",
input=[{"role": "user", "content": "Generate an image of a sunset"}],
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=[],
reasoning="Image generation not available",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
missing_capabilities="Image generation capability would be needed",
)
mock_tatlock.return_value = "I'm afraid I don't have image generation capabilities, sir."
response = await create_response_with_steward(request)
# Verify Steward's reasoning mentions missing capabilities
reasoning_item = response.output[0]
assert "not available" in reasoning_item.summary[1].lower()
assert response.status == "completed"
@pytest.mark.asyncio
async def test_conversation_id_propagation(self):
"""Test that conversation ID flows through entire pipeline."""
request = ResponseRequest(
model="tatlock",
input=[{"role": "user", "content": "Test request"}],
metadata={"conversation_id": "test_conv_123"},
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
with patch("src.responses.service.ToolCallTracker") as mock_tracker_class:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Test",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
mock_tatlock.return_value = "Test response"
mock_tracker = MagicMock()
mock_tracker.get_summary = MagicMock(return_value={})
mock_tracker.finalize = AsyncMock()
mock_tracker_class.return_value = mock_tracker
response = await create_response_with_steward(request)
# Verify conversation ID was passed to Steward
steward_kwargs = mock_steward.call_args[1]
assert steward_kwargs.get("conversation_id") == "test_conv_123"
# Verify conversation ID was passed to tracker
assert mock_tracker_class.called
tracker_call_args = mock_tracker_class.call_args
if tracker_call_args and len(tracker_call_args) > 1:
tracker_init_kwargs = tracker_call_args[1]
assert tracker_init_kwargs.get("conversation_id") == "test_conv_123"
assert response.status == "completed"