feat: implement Phase 2 two-tier architecture with Steward
Add comprehensive two-tier architecture where Steward analyzes requests and Tatlock executes with scoped tools. Includes full infrastructure for request preprocessing, tool tracking, benchmarking, and streaming. **Added:** - Steward agent for request analysis and capability recommendation - Household Registry for centralized capability management - Request preprocessing pipeline (Steward → Tatlock flow) - Tool usage tracking and benchmarking system - Streaming transparency (Steward reasoning visible in streams) - Structured logging with operation timing - Redis benchmark storage with 30-day expiry - Benchmark analysis CLI tools **Infrastructure:** - src/agents/steward/ - Steward agent implementation - src/agents/tatlock_core/ - Tatlock capability domain - src/core/preprocessing.py - Request preprocessing pipeline - src/core/tool_tracking.py - Tool call tracking - src/core/benchmarks.py - Benchmark recording system - src/core/household_registry.py - Capability registry - src/core/startup.py - Application startup coordination - src/core/logging_config.py - Structured logging setup **Integration:** - Responses API uses Steward for Tatlock requests - Chat Completions wraps Responses API for OpenAI compatibility - Streaming coordinator supports Steward + Tatlock flow - Tool scoping per request based on Steward recommendations **Testing:** - Integration tests for Steward-Tatlock flow - Benchmark and registry unit tests - Steward streaming tests See PHASE2_PLAN.md and PHASE2_COMPLETE.md for detailed documentation. 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude Sonnet 4.5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,190 @@
|
||||
"""
|
||||
Integration tests for Steward + Tatlock streaming.
|
||||
|
||||
Tests the complete streaming flow with Steward preprocessing.
|
||||
"""
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
from src.responses.schemas import ResponseRequest
|
||||
from src.responses.streaming import StreamingCoordinator, StreamEventType
|
||||
from src.core.startup import initialize_application
|
||||
|
||||
|
||||
@pytest.fixture(scope="module", autouse=True)
|
||||
def setup_household_registry():
|
||||
"""Initialize household registry before running tests."""
|
||||
initialize_application()
|
||||
|
||||
|
||||
class TestStewardStreaming:
|
||||
"""Test Steward + Tatlock streaming integration."""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_stream_with_steward_basic(self):
|
||||
"""Test basic streaming with Steward preprocessing."""
|
||||
request = ResponseRequest(
|
||||
model="tatlock",
|
||||
input=[{"role": "user", "content": "What's 2 + 2?"}],
|
||||
stream=True,
|
||||
)
|
||||
|
||||
# Mock the Steward analysis
|
||||
with patch("src.core.preprocessing.analyze_request") as mock_steward:
|
||||
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
|
||||
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
|
||||
|
||||
# Mock Steward recommendation
|
||||
mock_steward.return_value = StewardRecommendation(
|
||||
recommended_capabilities=["tatlock_core"],
|
||||
reasoning="Math calculation requires tatlock_core",
|
||||
estimated_complexity="simple",
|
||||
conversation_context=ConversationContext(has_previous_context=False),
|
||||
)
|
||||
|
||||
# Mock Tatlock response
|
||||
mock_tatlock.return_value = "Certainly, sir. 2 + 2 equals 4."
|
||||
|
||||
# Execute streaming
|
||||
coordinator = StreamingCoordinator()
|
||||
events = []
|
||||
|
||||
async for event in coordinator.stream_response_with_steward(request):
|
||||
events.append(event)
|
||||
|
||||
# Verify event sequence
|
||||
event_types = [e.event for e in events]
|
||||
|
||||
# Should have reasoning summary deltas
|
||||
assert StreamEventType.REASONING_SUMMARY_DELTA in event_types
|
||||
assert StreamEventType.REASONING_SUMMARY_DONE in event_types
|
||||
|
||||
# Should have output text deltas
|
||||
assert StreamEventType.OUTPUT_TEXT_DELTA in event_types
|
||||
assert StreamEventType.OUTPUT_TEXT_DONE in event_types
|
||||
|
||||
# Should end with response.done
|
||||
assert events[-1].event == StreamEventType.RESPONSE_DONE
|
||||
|
||||
# Verify Steward and Tatlock were called
|
||||
assert mock_steward.called
|
||||
assert mock_tatlock.called
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_stream_with_conversation_history(self):
|
||||
"""Test streaming with conversation history."""
|
||||
request = ResponseRequest(
|
||||
model="tatlock",
|
||||
input=[
|
||||
{"role": "user", "content": "What's 5 times 3?"},
|
||||
{"role": "assistant", "content": "That equals 15, sir."},
|
||||
{"role": "user", "content": "And divided by 3?"},
|
||||
],
|
||||
stream=True,
|
||||
)
|
||||
|
||||
with patch("src.core.preprocessing.analyze_request") as mock_steward:
|
||||
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
|
||||
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
|
||||
|
||||
mock_steward.return_value = StewardRecommendation(
|
||||
recommended_capabilities=["tatlock_core"],
|
||||
reasoning="Follow-up calculation based on previous result of 15",
|
||||
estimated_complexity="simple",
|
||||
conversation_context=ConversationContext(
|
||||
has_previous_context=True,
|
||||
relevant_turns=[0],
|
||||
context_summary="Previous calculation in turn 0"
|
||||
),
|
||||
)
|
||||
|
||||
mock_tatlock.return_value = "15 divided by 3 equals 5, sir."
|
||||
|
||||
coordinator = StreamingCoordinator()
|
||||
events = []
|
||||
|
||||
async for event in coordinator.stream_response_with_steward(request):
|
||||
events.append(event)
|
||||
|
||||
# Verify conversation history was passed to Steward
|
||||
call_kwargs = mock_steward.call_args[1]
|
||||
assert "conversation_history" in call_kwargs
|
||||
assert len(call_kwargs["conversation_history"]) == 2 # First Q&A pair
|
||||
|
||||
# Verify final response includes both reasoning and message
|
||||
final_event = events[-1]
|
||||
assert final_event.event == StreamEventType.RESPONSE_DONE
|
||||
assert len(final_event.response.output) == 2 # Reasoning + Message
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_stream_reasoning_contains_steward_analysis(self):
|
||||
"""Test that reasoning summary contains Steward's analysis."""
|
||||
request = ResponseRequest(
|
||||
model="tatlock",
|
||||
input=[{"role": "user", "content": "Test request"}],
|
||||
stream=True,
|
||||
)
|
||||
|
||||
with patch("src.core.preprocessing.analyze_request") as mock_steward:
|
||||
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
|
||||
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
|
||||
|
||||
mock_steward.return_value = StewardRecommendation(
|
||||
recommended_capabilities=["tatlock_core"],
|
||||
reasoning="This is a test analysis with specific markers",
|
||||
estimated_complexity="simple",
|
||||
conversation_context=ConversationContext(has_previous_context=False),
|
||||
)
|
||||
|
||||
mock_tatlock.return_value = "Test response"
|
||||
|
||||
coordinator = StreamingCoordinator()
|
||||
reasoning_deltas = []
|
||||
|
||||
async for event in coordinator.stream_response_with_steward(request):
|
||||
if event.event == StreamEventType.REASONING_SUMMARY_DELTA:
|
||||
reasoning_deltas.append(event.delta)
|
||||
|
||||
# Combine all reasoning deltas
|
||||
full_reasoning = "".join(reasoning_deltas)
|
||||
|
||||
# Should contain Steward's analysis
|
||||
assert "test analysis" in full_reasoning.lower()
|
||||
assert len(reasoning_deltas) > 0, "Should have streamed reasoning deltas"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_stream_with_missing_capabilities(self):
|
||||
"""Test streaming when Steward detects missing capabilities."""
|
||||
request = ResponseRequest(
|
||||
model="tatlock",
|
||||
input=[{"role": "user", "content": "Generate an image of a sunset"}],
|
||||
stream=True,
|
||||
)
|
||||
|
||||
with patch("src.core.preprocessing.analyze_request") as mock_steward:
|
||||
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
|
||||
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
|
||||
|
||||
mock_steward.return_value = StewardRecommendation(
|
||||
recommended_capabilities=[],
|
||||
reasoning="Image generation not available in current toolset",
|
||||
estimated_complexity="simple",
|
||||
conversation_context=ConversationContext(has_previous_context=False),
|
||||
missing_capabilities="Image generation capability would be needed",
|
||||
)
|
||||
|
||||
mock_tatlock.return_value = "I'm afraid I don't have image generation capabilities, sir."
|
||||
|
||||
coordinator = StreamingCoordinator()
|
||||
events = []
|
||||
|
||||
async for event in coordinator.stream_response_with_steward(request):
|
||||
events.append(event)
|
||||
|
||||
# Should complete successfully even with missing capabilities
|
||||
assert events[-1].event == StreamEventType.RESPONSE_DONE
|
||||
|
||||
# Verify empty scoped tools were passed
|
||||
tatlock_kwargs = mock_tatlock.call_args[1]
|
||||
assert "scoped_tools" in tatlock_kwargs
|
||||
assert tatlock_kwargs["scoped_tools"] == []
|
||||
@@ -0,0 +1,249 @@
|
||||
"""
|
||||
Integration tests for Steward → Tatlock flow.
|
||||
|
||||
Tests the complete Phase 2 request pipeline:
|
||||
1. Steward analyzes request and recommends capabilities
|
||||
2. Tool tracker monitors tool usage
|
||||
3. Tatlock runs with scoped tools
|
||||
4. Response includes both Steward reasoning and Tatlock output
|
||||
"""
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
from src.responses.schemas import ResponseRequest
|
||||
from src.responses.service import create_response_with_steward
|
||||
from src.core.startup import initialize_application
|
||||
|
||||
|
||||
@pytest.fixture(scope="module", autouse=True)
|
||||
def setup_household_registry():
|
||||
"""Initialize household registry before running tests."""
|
||||
initialize_application()
|
||||
|
||||
|
||||
class TestStewardTatlockIntegration:
|
||||
"""Test full Steward → Tatlock integration flow."""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_simple_math_request(self):
|
||||
"""Test math request flows through Steward → Tatlock correctly."""
|
||||
# Create a simple math request
|
||||
request = ResponseRequest(
|
||||
model="tatlock",
|
||||
input=[{"role": "user", "content": "What's 2 + 2?"}],
|
||||
)
|
||||
|
||||
# Mock the Steward analysis
|
||||
with patch("src.core.preprocessing.analyze_request") as mock_steward:
|
||||
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
|
||||
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
|
||||
|
||||
# Mock Steward recommendation
|
||||
mock_steward.return_value = StewardRecommendation(
|
||||
recommended_capabilities=["tatlock_core"],
|
||||
reasoning="Math calculation requires tatlock_core",
|
||||
estimated_complexity="simple",
|
||||
conversation_context=ConversationContext(has_previous_context=False),
|
||||
)
|
||||
|
||||
# Mock Tatlock response
|
||||
mock_tatlock.return_value = "Certainly, sir. 2 + 2 equals 4."
|
||||
|
||||
# Execute the flow
|
||||
response = await create_response_with_steward(request)
|
||||
|
||||
# Verify Steward was called
|
||||
assert mock_steward.called
|
||||
assert mock_steward.call_args[0][0] == "What's 2 + 2?"
|
||||
|
||||
# Verify Tatlock was called with scoped tools
|
||||
assert mock_tatlock.called
|
||||
|
||||
# Verify response structure
|
||||
assert response.status == "completed"
|
||||
assert len(response.output) == 2 # Reasoning + Message
|
||||
|
||||
# Check Steward reasoning output
|
||||
reasoning_item = response.output[0]
|
||||
assert reasoning_item.type == "reasoning"
|
||||
assert "Math calculation" in reasoning_item.summary[1]
|
||||
|
||||
# Check Tatlock message output
|
||||
message_item = response.output[1]
|
||||
assert message_item.type == "message"
|
||||
assert message_item.role == "assistant"
|
||||
assert "4" in message_item.content[0].text
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_request_with_conversation_history(self):
|
||||
"""Test that conversation history flows through to Steward."""
|
||||
request = ResponseRequest(
|
||||
model="tatlock",
|
||||
input=[
|
||||
{"role": "user", "content": "What's 5 times 3?"},
|
||||
{"role": "assistant", "content": "That equals 15, sir."},
|
||||
{"role": "user", "content": "And divided by 3?"},
|
||||
],
|
||||
)
|
||||
|
||||
with patch("src.core.preprocessing.analyze_request") as mock_steward:
|
||||
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
|
||||
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
|
||||
|
||||
mock_steward.return_value = StewardRecommendation(
|
||||
recommended_capabilities=["tatlock_core"],
|
||||
reasoning="Follow-up calculation",
|
||||
estimated_complexity="simple",
|
||||
conversation_context=ConversationContext(
|
||||
has_previous_context=True,
|
||||
relevant_turns=[0],
|
||||
context_summary="Previous calculation in turn 0"
|
||||
),
|
||||
)
|
||||
|
||||
mock_tatlock.return_value = "15 divided by 3 equals 5, sir."
|
||||
|
||||
response = await create_response_with_steward(request)
|
||||
|
||||
# Verify Steward received conversation history
|
||||
call_kwargs = mock_steward.call_args[1]
|
||||
assert "conversation_history" in call_kwargs
|
||||
assert len(call_kwargs["conversation_history"]) == 2 # First Q&A pair
|
||||
|
||||
# Verify Tatlock received history
|
||||
tatlock_kwargs = mock_tatlock.call_args[1]
|
||||
assert "message_history" in tatlock_kwargs
|
||||
|
||||
# Verify response completed
|
||||
assert response.status == "completed"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_no_capabilities_needed(self):
|
||||
"""Test simple conversational request that needs no tools."""
|
||||
request = ResponseRequest(
|
||||
model="tatlock",
|
||||
input=[{"role": "user", "content": "Hello!"}],
|
||||
)
|
||||
|
||||
with patch("src.core.preprocessing.analyze_request") as mock_steward:
|
||||
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
|
||||
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
|
||||
|
||||
mock_steward.return_value = StewardRecommendation(
|
||||
recommended_capabilities=[], # No tools needed
|
||||
reasoning="Simple greeting, no tools required",
|
||||
estimated_complexity="simple",
|
||||
conversation_context=ConversationContext(has_previous_context=False),
|
||||
)
|
||||
|
||||
mock_tatlock.return_value = "Good day, sir. How may I assist you?"
|
||||
|
||||
response = await create_response_with_steward(request)
|
||||
|
||||
# Verify empty scoped tools were passed
|
||||
tatlock_kwargs = mock_tatlock.call_args[1]
|
||||
assert "scoped_tools" in tatlock_kwargs
|
||||
assert tatlock_kwargs["scoped_tools"] == [] # No tools
|
||||
|
||||
assert response.status == "completed"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_tool_tracker_integration(self):
|
||||
"""Test that tool tracker is passed to Tatlock and finalized."""
|
||||
request = ResponseRequest(
|
||||
model="tatlock",
|
||||
input=[{"role": "user", "content": "Calculate sqrt(16)"}],
|
||||
)
|
||||
|
||||
with patch("src.core.preprocessing.analyze_request") as mock_steward:
|
||||
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
|
||||
with patch("src.core.tool_tracking.ToolCallTracker.finalize") as mock_finalize:
|
||||
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
|
||||
|
||||
mock_steward.return_value = StewardRecommendation(
|
||||
recommended_capabilities=["tatlock_core"],
|
||||
reasoning="Calculator needed",
|
||||
estimated_complexity="simple",
|
||||
conversation_context=ConversationContext(has_previous_context=False),
|
||||
)
|
||||
|
||||
mock_tatlock.return_value = "The square root of 16 is 4, sir."
|
||||
|
||||
response = await create_response_with_steward(request)
|
||||
|
||||
# Verify tool tracker was finalized
|
||||
assert mock_finalize.called
|
||||
assert response.status == "completed"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_missing_capabilities_warning(self):
|
||||
"""Test that missing capabilities are included in Steward's reasoning."""
|
||||
request = ResponseRequest(
|
||||
model="tatlock",
|
||||
input=[{"role": "user", "content": "Generate an image of a sunset"}],
|
||||
)
|
||||
|
||||
with patch("src.core.preprocessing.analyze_request") as mock_steward:
|
||||
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
|
||||
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
|
||||
|
||||
mock_steward.return_value = StewardRecommendation(
|
||||
recommended_capabilities=[],
|
||||
reasoning="Image generation not available",
|
||||
estimated_complexity="simple",
|
||||
conversation_context=ConversationContext(has_previous_context=False),
|
||||
missing_capabilities="Image generation capability would be needed",
|
||||
)
|
||||
|
||||
mock_tatlock.return_value = "I'm afraid I don't have image generation capabilities, sir."
|
||||
|
||||
response = await create_response_with_steward(request)
|
||||
|
||||
# Verify Steward's reasoning mentions missing capabilities
|
||||
reasoning_item = response.output[0]
|
||||
assert "not available" in reasoning_item.summary[1].lower()
|
||||
|
||||
assert response.status == "completed"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_conversation_id_propagation(self):
|
||||
"""Test that conversation ID flows through entire pipeline."""
|
||||
request = ResponseRequest(
|
||||
model="tatlock",
|
||||
input=[{"role": "user", "content": "Test request"}],
|
||||
metadata={"conversation_id": "test_conv_123"},
|
||||
)
|
||||
|
||||
with patch("src.core.preprocessing.analyze_request") as mock_steward:
|
||||
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
|
||||
with patch("src.responses.service.ToolCallTracker") as mock_tracker_class:
|
||||
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
|
||||
|
||||
mock_steward.return_value = StewardRecommendation(
|
||||
recommended_capabilities=["tatlock_core"],
|
||||
reasoning="Test",
|
||||
estimated_complexity="simple",
|
||||
conversation_context=ConversationContext(has_previous_context=False),
|
||||
)
|
||||
|
||||
mock_tatlock.return_value = "Test response"
|
||||
|
||||
mock_tracker = MagicMock()
|
||||
mock_tracker.get_summary = MagicMock(return_value={})
|
||||
mock_tracker.finalize = AsyncMock()
|
||||
mock_tracker_class.return_value = mock_tracker
|
||||
|
||||
response = await create_response_with_steward(request)
|
||||
|
||||
# Verify conversation ID was passed to Steward
|
||||
steward_kwargs = mock_steward.call_args[1]
|
||||
assert steward_kwargs.get("conversation_id") == "test_conv_123"
|
||||
|
||||
# Verify conversation ID was passed to tracker
|
||||
assert mock_tracker_class.called
|
||||
tracker_call_args = mock_tracker_class.call_args
|
||||
if tracker_call_args and len(tracker_call_args) > 1:
|
||||
tracker_init_kwargs = tracker_call_args[1]
|
||||
assert tracker_init_kwargs.get("conversation_id") == "test_conv_123"
|
||||
|
||||
assert response.status == "completed"
|
||||
Reference in New Issue
Block a user