feat(responses): pass conversation context and stream thinks in real time

- delegate_to_* now receives a trimmed conversation history (last ~6
  turns, 500 chars/turn) as context on both live direct-delegation
  paths (streaming and steward non-streaming), via new
  build_delegation_context helper
- _stream_direct_delegation restructured as an async generator: the
  butler 'start' think message streams BEFORE the expert runs and the
  success/error message right after it finishes, instead of all
  messages arriving after the research completed

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
2026-07-14 10:30:34 +02:00
co-authored by Claude Fable 5
parent 0708c759fc
commit 7ce1c1a314
6 changed files with 355 additions and 41 deletions
+44
View File
@@ -128,6 +128,50 @@ def _detect_action_type(expert: str, task: str) -> ActionType:
return ActionType.RETRIEVE
def build_delegation_context(
conversation_history: list[dict] | None,
max_turns: int = 6,
max_chars_per_turn: int = 500,
) -> str:
"""
Format the most recent conversation turns as delegation context.
Experts accept a context string but the live paths never passed the
in-scope conversation history; this trims it to the last few turns
so follow-up questions ("and what about X?") keep their referent.
Args:
conversation_history: Prior messages as {"role", "content"} dicts
max_turns: How many trailing turns to include
max_chars_per_turn: Truncation limit per turn
Returns:
str: Newline-joined "role: content" lines ("" when no history)
"""
if not conversation_history:
return ""
lines = []
for msg in conversation_history[-max_turns:]:
if not isinstance(msg, dict):
continue
role = msg.get("role", "user")
content = msg.get("content", "")
if isinstance(content, list):
# Tolerate structured content parts
content = " ".join(
part.get("text", "") if isinstance(part, dict) else str(part)
for part in content
)
content = str(content).strip()
if content:
lines.append(f"{role}: {content[:max_chars_per_turn]}")
if not lines:
return ""
return "Recent conversation:\n" + "\n".join(lines)
def get_think_message(expert: str, task: str, phase: str) -> str:
"""
Get the appropriate think message for an expert delegation.