feat(responses): pass conversation context and stream thinks in real time

- delegate_to_* now receives a trimmed conversation history (last ~6
  turns, 500 chars/turn) as context on both live direct-delegation
  paths (streaming and steward non-streaming), via new
  build_delegation_context helper
- _stream_direct_delegation restructured as an async generator: the
  butler 'start' think message streams BEFORE the expert runs and the
  success/error message right after it finishes, instead of all
  messages arriving after the research completed

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
2026-07-14 10:30:34 +02:00
co-authored by Claude Fable 5
parent 0708c759fc
commit 7ce1c1a314
6 changed files with 355 additions and 41 deletions
+52 -35
View File
@@ -189,20 +189,18 @@ class StreamingCoordinator:
tatlock = TatlockAgent()
if delegation_only:
# Direct delegation path with streaming think slugs
orchestration_results = await self._stream_direct_delegation(
# Direct delegation path - think slugs stream in real time,
# BEFORE and after each expert runs (not after the fact)
orchestration_results: dict = {}
async for event in self._stream_direct_delegation(
user_message=user_message,
recommendation=enriched.recommendation,
tracker=tracker,
conversation_id=conversation_id,
)
# Stream think slugs that were collected during delegation
# Each think message is complete, so we signal done after each
for think_msg in orchestration_results.get("think_messages", []):
yield ReasoningSummaryDelta(delta=think_msg)
yield ReasoningSummaryDone()
await asyncio.sleep(0.05)
conversation_history=conversation_history,
results=orchestration_results,
):
yield event
else:
# Phase 1: Orchestrate tool calls
@@ -272,24 +270,33 @@ class StreamingCoordinator:
recommendation: "StewardRecommendation", # type: ignore
tracker: "ToolCallTracker", # type: ignore
conversation_id: str,
) -> dict:
conversation_history: list | None = None,
results: dict | None = None,
) -> AsyncGenerator[StreamEvent, None]:
"""
Execute direct delegation with streaming think messages.
Execute direct delegation, streaming think messages in real time.
Collects think messages as delegations execute for streaming to client.
An async generator: the "start" think message for each expert is
yielded BEFORE its research runs (so the user sees 'Allow me to
consult the archives, sir.' while waiting), and the success/error
message right after it finishes.
Args:
user_message: User's request
recommendation: Steward's recommendation
tracker: Tool call tracker
conversation_id: Conversation ID
conversation_history: Prior turns, trimmed into expert context
results: Mutable dict populated with orchestration results
(expert_results, tools_called, think_messages, ...)
Returns:
dict: Orchestration results with think_messages list
Yields:
StreamEvent: Reasoning summary events as delegation progresses
"""
import time as time_module
from src.agents.delegation import (
build_delegation_context,
delegate_to_biographer,
delegate_to_housekeeper,
delegate_to_librarian,
@@ -299,21 +306,30 @@ class StreamingCoordinator:
expert_results = {}
tools_called = []
think_messages = []
context = build_delegation_context(conversation_history)
for agent in recommendation.recommended_capabilities:
# Emit start think message
# Emit start think message BEFORE the expert runs
start_msg = get_think_message(agent, user_message, "start")
think_messages.append(start_msg + "\n")
yield ReasoningSummaryDelta(delta=start_msg + "\n")
yield ReasoningSummaryDone()
start_time = time_module.time()
try:
# Execute delegation
if agent == "librarian":
result = await delegate_to_librarian(task=user_message)
result = await delegate_to_librarian(
task=user_message, context=context
)
elif agent == "biographer":
result = await delegate_to_biographer(task=user_message)
result = await delegate_to_biographer(
task=user_message, context=context
)
elif agent == "housekeeper":
result = await delegate_to_housekeeper(task=user_message)
result = await delegate_to_housekeeper(
task=user_message, context=context
)
else:
result = None
@@ -324,18 +340,15 @@ class StreamingCoordinator:
expert_results[agent] = result.output
tools_called.append(f"delegate_to_{agent}")
# Emit success think message
success_msg = get_think_message(agent, user_message, "success")
think_messages.append(success_msg + "\n")
phase_msg = get_think_message(agent, user_message, "success")
else:
# Failed delegations carry a curated user-safe sentence
# in output; exception detail is already in the logs.
error_think = get_think_message(agent, user_message, "error")
phase_msg = get_think_message(agent, user_message, "error")
if result and result.output:
expert_results[agent] = result.output
else:
expert_results[agent] = error_think
# Emit error think message
think_messages.append(error_think + "\n")
expert_results[agent] = phase_msg
except Exception as e:
logger.error(
@@ -345,17 +358,21 @@ class StreamingCoordinator:
conversation_id=conversation_id,
exc_info=True,
)
error_think = get_think_message(agent, user_message, "error")
expert_results[agent] = error_think
think_messages.append(error_think + "\n")
phase_msg = get_think_message(agent, user_message, "error")
expert_results[agent] = phase_msg
return {
"tools_called": tools_called,
"expert_results": expert_results,
"tool_outputs": {},
"raw_output": "",
"think_messages": think_messages,
}
think_messages.append(phase_msg + "\n")
yield ReasoningSummaryDelta(delta=phase_msg + "\n")
yield ReasoningSummaryDone()
if results is not None:
results.update({
"tools_called": tools_called,
"expert_results": expert_results,
"tool_outputs": {},
"raw_output": "",
"think_messages": think_messages,
})
async def stream_response(
self,