feat: integrate tracing throughout request pipeline

Instrument the full request flow with trace spans for debugging:

- Wrap expert delegations (librarian/biographer/housekeeper) in spans
- Add orchestrate and synthesize spans to TatlockAgent
- Trace Steward analysis in preprocessing
- Start/end traces in response service with context management
- Simplify router by moving context handling to service layer
- Include tracing router in debug mode
- Remove benchmark recording from tool_tracking and steward service

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
This commit is contained in:
2025-12-22 10:26:37 +01:00
co-authored by Claude Opus 4.5
parent 2a9449bc81
commit 87f2926db2
8 changed files with 502 additions and 369 deletions
+56 -2
View File
@@ -20,6 +20,11 @@ from src.agents.tatlock_core.tools import (
)
from src.core.config import config
from src.core.logging_config import get_logger
from src.core.tracing import (
start_span, end_span, get_current_span,
add_tool_spans_from_messages,
SpanType, SpanStatus,
)
logger = get_logger(__name__)
@@ -433,7 +438,7 @@ class TatlockAgent(AgentInterface):
steward_note: Note from Steward (prepended to request, invisible to user)
scoped_tools: List of tool definitions from household registry
message_history: Conversation history in PydanticAI format
tool_tracker: Optional tool call tracker for benchmarking
tool_tracker: Optional tool call tracker for analysis
Returns:
str: Tatlock's response text
@@ -627,7 +632,7 @@ class TatlockAgent(AgentInterface):
steward_note: Note from Steward (invisible to user)
scoped_tools: List of tool definitions from household registry
message_history: Conversation history
tool_tracker: Optional tool call tracker for benchmarking
tool_tracker: Optional tool call tracker for analysis
Returns:
dict with:
@@ -655,6 +660,16 @@ class TatlockAgent(AgentInterface):
history_length=len(message_history),
)
# Start tracing span for orchestration phase
orchestrate_span = start_span(
"tatlock_orchestrate",
SpanType.TATLOCK,
metadata={
"scoped_tool_count": len(scoped_tools),
"tool_names": [getattr(t, '__name__', str(t)) for t in scoped_tools[:5]],
},
)
# Create a fresh agent instance with scoped tools only
clean_host = self.ollama_host.rstrip('/')
base_url = f"{clean_host}/v1"
@@ -731,6 +746,23 @@ class TatlockAgent(AgentInterface):
tool_output_count=len(tool_outputs),
)
# Add tool-level spans from result messages
if orchestrate_span:
add_tool_spans_from_messages(result.new_messages(), orchestrate_span)
# End orchestration span with results
end_span(
orchestrate_span,
metadata_update={
"tools_called": tools_called,
"expert_count": len(expert_results),
"tool_output_count": len(tool_outputs),
},
details_update={
"steward_note_preview": steward_note[:500] if steward_note else None,
},
)
return {
"tools_called": tools_called,
"expert_results": expert_results,
@@ -769,6 +801,16 @@ class TatlockAgent(AgentInterface):
tool_count=len(orchestration_results.get("tool_outputs", {})),
)
# Start tracing span for synthesis phase
synthesize_span = start_span(
"tatlock_synthesize",
SpanType.TATLOCK,
metadata={
"expert_count": len(orchestration_results.get("expert_results", {})),
"tool_output_count": len(orchestration_results.get("tool_outputs", {})),
},
)
# Build synthesis prompt with all available information
synthesis_parts = []
synthesis_parts.append(f"The user asked: {user_message}")
@@ -841,6 +883,18 @@ class TatlockAgent(AgentInterface):
response_preview=result.output[:100],
)
# End synthesis span with result
end_span(
synthesize_span,
metadata_update={
"response_length": len(result.output),
},
details_update={
"synthesis_prompt": synthesis_prompt[:1000],
"response_preview": result.output[:500],
},
)
return result.output
async def get_capabilities(self) -> dict: