Files
tatlock/tests/integration/test_steward_tatlock_integration.py
T
jpmschweitzerandClaude Opus 4.5 64cad4500a
Build and Push / build (release) Successful in 52s
feat: environment-aware config, direct delegation, E2E test suite (v1.4.0)
### Added
- Environment-aware configuration:
  - Auto-selected logging (DEBUG for dev, WARNING for prod)
  - Auto-selected default user (llm_tester for dev isolation)
  - User context logging at request entry
- Direct delegation bypass:
  - Pure memory/librarian requests skip Tatlock LLM
  - Reduces latency for memory-only requests
- Text-based delegation fallback:
  - Parse [DELEGATE:agent] patterns from LLM output
  - Sequential and parallel execution support
- Comprehensive E2E test suite:
  - 22 orchestration tests with QdrantVerifier
  - assert_llm_behavior() for flexible pattern matching
  - Tests for memory, delegation, isolation, scenarios

### Fixed
- Unit test mocks for streaming (async generator)
- Temporal context handling in tests
- LLM non-determinism with pytest.xfail()
- Streaming test timeouts increased

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-14 21:19:47 +01:00

253 lines
11 KiB
Python

"""
Integration tests for Steward → Tatlock flow.
Tests the complete Phase 2 request pipeline:
1. Steward analyzes request and recommends capabilities
2. Tool tracker monitors tool usage
3. Tatlock runs with scoped tools
4. Response includes both Steward reasoning and Tatlock output
"""
import pytest
from unittest.mock import AsyncMock, MagicMock, patch
from src.responses.schemas import ResponseRequest
from src.responses.service import create_response_with_steward
from src.core.startup import initialize_application
@pytest.fixture(scope="module", autouse=True)
def setup_household_registry():
"""Initialize household registry before running tests."""
initialize_application()
class TestStewardTatlockIntegration:
"""Test full Steward → Tatlock integration flow."""
@pytest.mark.asyncio
async def test_simple_math_request(self):
"""Test math request flows through Steward → Tatlock correctly."""
# Create a simple math request
request = ResponseRequest(
model="tatlock",
input=[{"role": "user", "content": "What's 2 + 2?"}],
)
# Mock the Steward analysis
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
# Mock Steward recommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Math calculation requires tatlock_core",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
# Mock Tatlock response
mock_tatlock.return_value = "Certainly, sir. 2 + 2 equals 4."
# Execute the flow
response = await create_response_with_steward(request)
# Verify Steward was called
assert mock_steward.called
# Note: preprocess_request injects temporal context
steward_call_arg = mock_steward.call_args[0][0]
assert steward_call_arg.startswith("What's 2 + 2?"), \
f"Expected request to start with original message, got: {steward_call_arg}"
# Verify Tatlock was called with scoped tools
assert mock_tatlock.called
# Verify response structure
assert response.status == "completed"
assert len(response.output) == 2 # Reasoning + Message
# Check Steward reasoning output
reasoning_item = response.output[0]
assert reasoning_item.type == "reasoning"
assert "Math calculation" in reasoning_item.summary[1]
# Check Tatlock message output
message_item = response.output[1]
assert message_item.type == "message"
assert message_item.role == "assistant"
assert "4" in message_item.content[0].text
@pytest.mark.asyncio
async def test_request_with_conversation_history(self):
"""Test that conversation history flows through to Steward."""
request = ResponseRequest(
model="tatlock",
input=[
{"role": "user", "content": "What's 5 times 3?"},
{"role": "assistant", "content": "That equals 15, sir."},
{"role": "user", "content": "And divided by 3?"},
],
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Follow-up calculation",
estimated_complexity="simple",
conversation_context=ConversationContext(
has_previous_context=True,
relevant_turns=[0],
context_summary="Previous calculation in turn 0"
),
)
mock_tatlock.return_value = "15 divided by 3 equals 5, sir."
response = await create_response_with_steward(request)
# Verify Steward received conversation history
call_kwargs = mock_steward.call_args[1]
assert "conversation_history" in call_kwargs
assert len(call_kwargs["conversation_history"]) == 2 # First Q&A pair
# Verify Tatlock received history
tatlock_kwargs = mock_tatlock.call_args[1]
assert "message_history" in tatlock_kwargs
# Verify response completed
assert response.status == "completed"
@pytest.mark.asyncio
async def test_no_capabilities_needed(self):
"""Test simple conversational request that needs no tools."""
request = ResponseRequest(
model="tatlock",
input=[{"role": "user", "content": "Hello!"}],
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=[], # No tools needed
reasoning="Simple greeting, no tools required",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
mock_tatlock.return_value = "Good day, sir. How may I assist you?"
response = await create_response_with_steward(request)
# Verify empty scoped tools were passed
tatlock_kwargs = mock_tatlock.call_args[1]
assert "scoped_tools" in tatlock_kwargs
assert tatlock_kwargs["scoped_tools"] == [] # No tools
assert response.status == "completed"
@pytest.mark.asyncio
async def test_tool_tracker_integration(self):
"""Test that tool tracker is passed to Tatlock and finalized."""
request = ResponseRequest(
model="tatlock",
input=[{"role": "user", "content": "Calculate sqrt(16)"}],
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
with patch("src.core.tool_tracking.ToolCallTracker.finalize") as mock_finalize:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Calculator needed",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
mock_tatlock.return_value = "The square root of 16 is 4, sir."
response = await create_response_with_steward(request)
# Verify tool tracker was finalized
assert mock_finalize.called
assert response.status == "completed"
@pytest.mark.asyncio
async def test_missing_capabilities_warning(self):
"""Test that missing capabilities are included in Steward's reasoning."""
request = ResponseRequest(
model="tatlock",
input=[{"role": "user", "content": "Generate an image of a sunset"}],
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=[],
reasoning="Image generation not available",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
missing_capabilities="Image generation capability would be needed",
)
mock_tatlock.return_value = "I'm afraid I don't have image generation capabilities, sir."
response = await create_response_with_steward(request)
# Verify Steward's reasoning mentions missing capabilities
reasoning_item = response.output[0]
assert "not available" in reasoning_item.summary[1].lower()
assert response.status == "completed"
@pytest.mark.asyncio
async def test_conversation_id_propagation(self):
"""Test that conversation ID flows through entire pipeline."""
request = ResponseRequest(
model="tatlock",
input=[{"role": "user", "content": "Test request"}],
metadata={"conversation_id": "test_conv_123"},
)
with patch("src.core.preprocessing.analyze_request") as mock_steward:
with patch("src.agents.tatlock.TatlockAgent.run_with_scoped_tools") as mock_tatlock:
with patch("src.responses.service.ToolCallTracker") as mock_tracker_class:
from src.agents.steward.schemas import ConversationContext, StewardRecommendation
mock_steward.return_value = StewardRecommendation(
recommended_capabilities=["tatlock_core"],
reasoning="Test",
estimated_complexity="simple",
conversation_context=ConversationContext(has_previous_context=False),
)
mock_tatlock.return_value = "Test response"
mock_tracker = MagicMock()
mock_tracker.get_summary = MagicMock(return_value={})
mock_tracker.finalize = AsyncMock()
mock_tracker_class.return_value = mock_tracker
response = await create_response_with_steward(request)
# Verify conversation ID was passed to Steward
steward_kwargs = mock_steward.call_args[1]
assert steward_kwargs.get("conversation_id") == "test_conv_123"
# Verify conversation ID was passed to tracker
assert mock_tracker_class.called
tracker_call_args = mock_tracker_class.call_args
if tracker_call_args and len(tracker_call_args) > 1:
tracker_init_kwargs = tracker_call_args[1]
assert tracker_init_kwargs.get("conversation_id") == "test_conv_123"
assert response.status == "completed"