""" agent_loop.py Streaming agent loop for odysseus-ui. Wraps stream_llm() with multi-round tool execution. The LLM decides when to use tools by writing fenced code blocks. """ import ast import asyncio import collections import contextlib import csv import difflib import html import json import os import re import shlex import shutil import time import logging import hashlib from src.web_recovery import WebRecoveryBudget from itertools import count from datetime import date, datetime, timedelta from dataclasses import replace from pathlib import Path from typing import Any, AsyncGenerator, Dict, Iterable, List, Mapping, Optional, Sequence, Set from urllib.parse import parse_qs, parse_qsl, quote, unquote, urlparse from src.llm_core import ( dedupe_model_candidates, stream_llm, stream_llm_with_fallback, _strip_visible_chat_template_artifacts, _is_ollama_native_url, _normalize_http_status, _normalize_usage_counts, ) from src.model_context import estimate_tokens, is_local_endpoint from src.model_profiles import ( ODYSSEUS_COMPACT_TOOL_SCHEMA_PROFILE, is_odysseus_merged_tools_model, tool_schema_profile, ) from src.agent_evidence import ( EvidenceLedger, command_has_mutation_effect, command_is_test, command_is_validation, requirements_from_runtime_context, ) from src.context_compactor import ( apply_compaction_state, apply_compaction_state_for_session, maybe_compact, ) from src.settings import get_setting from src.prompt_security import untrusted_context_message from src.tool_security import ( blocked_tools_for_owner, email_tool_policy_names, plan_mode_disabled_tools, ) from src.tool_policy import GUIDE_ONLY_DIRECTIVE, WEB_TOOL_NAMES, ToolPolicy, known_tool_names from src.client_tool_contract import TUI_CLIENT_TOOL_NAMES from src.tool_capabilities import ( ResultIntegrity, ToolRunSecurityContext, blocked_tool_result, capabilities_for_action, capabilities_for_tool, messages_contain_external_untrusted_context, tool_result_is_successful, tool_result_should_arm_gate, ) from src.tool_approvals import ( ExactToolApproval, document_content_digest, tool_approval_store, ) from src.tool_types import ToolBlock from src.turn_contract import selected_tools_for_request, with_turn_contract from src.agent_runtime.journal import propose_action, execute_action from src.agent_runtime.completion import with_completion_gate from src.tool_utils import _truncate, get_mcp_manager from src.agent_tools import ( parse_tool_blocks, strip_tool_blocks, execute_tool_block, format_tool_result, set_active_document, set_active_model, function_call_to_tool_block, FUNCTION_TOOL_SCHEMAS, TOOL_TAGS, MAX_AGENT_ROUNDS, ) def _local_media_discovery_call_allowed(tool_name: str, command: str) -> bool: """Allow harmless workspace discovery before media evidence is acquired. The local-media evidence gate must prevent answering from a filename and must block content-reading or mutating side channels. It should not turn a benign directory listing into a failed recovery path: models commonly inspect the workspace first and select ``inspect_media`` on the next turn. Keep shell support deliberately narrow and side-effect free. """ name = str(tool_name or "").strip().lower() if name in {"ls", "glob", "get_workspace"}: return True if name != "bash": return False text = str(command or "").strip() # ``#!bg`` is a parser marker emitted in some fenced shell blocks. text = re.sub(r"^#!\s*bg\s*\n?", "", text, count=1).strip() if not text or "\n" in text: return False if re.search(r"[;&|<>`$()]", text): return False return bool(re.fullmatch(r"(?:ls|stat|file)(?:\s+-[A-Za-z0-9./_-]+)*\s+[^\s]+", text)) def _resolved_tool_call_id( native_call: Optional[Mapping[str, Any]], *, session_id: str, round_num: int, tool_index: int, tool_name: str, ) -> str: """Return one stable SSE correlation ID for every executed tool call. Native model calls already carry an ID and must retain it. Harness-generated follow-through calls (artifact verification, recovery, and deterministic routing) do not, but downstream trace consumers still need matching ``tool_start`` and ``tool_output`` identities. """ native_id = str((native_call or {}).get("id") or "").strip() if native_id: return native_id seed = f"{session_id}\0{round_num}\0{tool_index}\0{tool_name}" digest = hashlib.sha256(seed.encode("utf-8")).hexdigest()[:24] return f"odysseus-auto-{digest}" logger = logging.getLogger(__name__) _MODEL_TOOL_SURFACES = {"none", "compact", "full"} _ROUTE_THINKING_MODES = {"auto", "on", "off"} _NO_THINKING_COMPACT_DOMAINS = { "email", "notes_calendar_tasks", "memory", "contacts", "documents", } def _normalize_model_tool_surface(value: Any) -> str: value = str(value or "").strip().lower() return value if value in _MODEL_TOOL_SURFACES else "" def _route_thinking_policy() -> str: mode = os.getenv("ODYSSEUS_QWEN_ROUTE_THINKING", "auto").strip().lower() return mode if mode in _ROUTE_THINKING_MODES else "auto" def _thinking_mode_for_route( *, model: str, tool_surface: str, domains: Set[str], direct: bool = False, ) -> Optional[str]: """Select Qwen thinking mode for the current agent route. ``auto`` keeps thinking available for broad/search/coding routes but turns it off for compact personal-tool surfaces where we want direct tool calls and concise final answers. Teacher/data-generation runs can set ``ODYSSEUS_QWEN_ROUTE_THINKING=on``; production can force ``off``. """ model_name = str(model or "").lower() qwen35_family = bool(re.search(r"(?:qwen3\.5|qwen35)", model_name)) if not (_is_qwen38_tool_router(model) or qwen35_family): return None # The pre-Heretic control is served by vLLM without a verified reasoning # parser. If thinking is enabled, its private analysis is returned as # ordinary content and the WebUI buffers a long pre-answer transcript. if is_odysseus_merged_tools_model(model_name): return "off" policy = _route_thinking_policy() if policy in {"on", "off"}: return policy if tool_surface == "compact" and (set(domains or set()) & _NO_THINKING_COMPACT_DOMAINS): return "off" if direct and "qwen35-email" in model_name: return "off" return None def _qwen_tool_router_output_budget(requested: int | None) -> int: """Keep an explicit agent budget; default only when none was requested.""" try: value = int(requested or 0) except (TypeError, ValueError): value = 0 return value if value > 0 else 1024 def _allow_visual_tool_evidence_for_model(model: str) -> bool: """Keep pixels for multimodal Odysseus routers; legacy routers stay text-only.""" return is_odysseus_merged_tools_model(model) or not _is_qwen38_tool_router(model) def _malformed_native_tool_recovery_instruction(names: Set[str]) -> str: """Return targeted, schema-level recovery for dropped native calls.""" if "write_file" in set(names or ()): return ( "Your previous write_file call was incomplete or malformed. Call " "write_file once with both path and content. Keep the file within " "the output budget by using loops, reusable functions, CSS, or data " "arrays instead of repeating generated markup. Do not restate the plan." ) return "" def _looks_like_explicit_web_search_request( text: str, *, local_media_turn: bool = False, ) -> bool: """Recognize explicit public-web intent without hijacking local media work.""" if local_media_turn: return False value = str(text or "") return bool( re.search( r"\b(?:latest|current|today|online|internet|web|search|look\s+up)\b" r"|\bfind\b.{0,80}\b(?:official\s+)?(?:website|site|page|url|link)\b", value, re.IGNORECASE, ) and not re.search( r"\b(?:email|mail|inbox|calendar|meeting|task|note|memory|saved\s+research|" r"skills?|procedures?|documents?|docs?|past\s+chat|prior\s+chat|" r"previous\s+conversation|research|deep\s+dive|investigate)\b", value, re.IGNORECASE, ) ) def _repeated_artifact_mutation_can_finish( names: Sequence[str], *, html_verified: bool, ) -> bool: """Stop after a verified artifact is regenerated byte-for-byte.""" normalized = {str(name or "").strip().lower() for name in names} return bool( html_verified and normalized and normalized <= {"write_file", "edit_file", "apply_patch"} ) def _malformed_write_needs_body_handoff( names: Set[str], missing_artifacts: Sequence[str], *, attempts: int, ) -> bool: """Use raw-body recovery once instead of repeating truncated tool JSON.""" missing = [str(path or "").strip() for path in missing_artifacts] return bool( attempts == 0 and "write_file" in set(names or ()) and len(missing) == 1 and missing[0] and not _binary_artifact_path(missing[0]) ) def _post_finish_inspection_should_converge( *, finish_nudge_sent: bool, correction_seen: bool, force_answer: bool, verification_only: bool, current_inspection: bool, can_complete: bool, ) -> bool: """Bound repeated inspection after a completed artifact's finish nudge.""" return bool( finish_nudge_sent and not correction_seen and not force_answer and verification_only and current_inspection and can_complete ) def _parse_model_tool_modes(raw: Any) -> Dict[str, str]: if not raw: return {} try: data = json.loads(raw) if isinstance(raw, str) else raw except Exception: return {} if not isinstance(data, dict): return {} modes: Dict[str, str] = {} for key, value in data.items(): model_id = str(key or "").strip() mode = _normalize_model_tool_surface(value) if model_id and mode: modes[model_id] = mode return modes def _model_id_tokens(value: Any) -> List[str]: leaf = os.path.basename(str(value or "").strip().rstrip("/")).lower() return [part for part in re.split(r"[^a-z0-9]+", leaf) if part] def _model_tool_mode_for_model(modes: Dict[str, str], model: str) -> str: """Resolve a per-model tool mode across exact ids and runtime aliases.""" model = str(model or "").strip() if not model or not modes: return "" exact = modes.get(model) if exact: return exact lowered = model.lower() for key, mode in modes.items(): if str(key or "").strip().lower() == lowered: return mode requested_tokens = _model_id_tokens(model) if not requested_tokens: return "" matches: List[str] = [] for key, mode in modes.items(): configured_tokens = _model_id_tokens(key) if not configured_tokens: continue if configured_tokens == requested_tokens: matches.append(mode) elif ( len(requested_tokens) >= 2 and len(configured_tokens) > len(requested_tokens) and configured_tokens[: len(requested_tokens)] == requested_tokens ): matches.append(mode) return matches[0] if len(matches) == 1 else "" def _apply_tool_surface_to_schemas( schemas: List[Dict[str, Any]], surface: str, ) -> List[Dict[str, Any]]: surface = _normalize_model_tool_surface(surface) if surface == "none": return [] if surface == "compact": return [_compact_openai_tool_schema(schema) for schema in (schemas or [])] return list(schemas or []) def _contract_allows_early_completion(contract) -> bool: # A shortcut cannot prove it completed every action, including multiple # actions within one family. Let the normal loop handle contract work. if contract is None: return True active = getattr(contract, "active_capabilities", None) if active is not None: return not active and not contract.required return not (contract.capabilities or contract.required or contract.offered) def _contract_prompt_domains(contract) -> Set[str]: """Adapt the resolved capabilities to legacy prompt-domain vocabulary.""" aliases = { "notes": "notes_calendar_tasks", "calendar": "notes_calendar_tasks", "tasks": "notes_calendar_tasks", "search_browser": "web", "shell_files": "files", "cookbook_admin": "cookbook", } return {aliases.get(family, family) for family in contract.capabilities} def _contract_allows_single_action_terminal(contract) -> bool: return contract is None or len(contract.capabilities) <= 1 def _request_has_compound_actions(text: str) -> bool: """Return whether a turn explicitly requests multiple semantic operations.""" value = str(text or "") if ( len(re.findall(r"\bhttps?://[^\s<>\"']+", value, re.IGNORECASE)) >= 2 and re.search( r"\b(?:compare|contrast|synthesi[sz]e|cite|citing|evidence)\b", value, re.IGNORECASE, ) ): return True groups = ( r"\b(?:create|add|make|write|draft|schedule|book|set\s+up)\b", r"\b(?:list|search|find|locate|look\s+up)\b", r"\b(?:read|open|inspect|view|download)\b", r"\b(?:edit|update|change|replace|rewrite|append)\b", r"\b(?:suggest|recommend|propose)\b", r"\b(?:pause|disable|suspend)\b", r"\b(?:resume|re-enable|bring\s+(?:it|them)\s+back)\b", r"\b(?:delete|remove|cancel|get\s+rid\s+of)\b", r"\b(?:verify|confirm|check)\b", ) return sum(bool(re.search(pattern, value, re.IGNORECASE)) for pattern in groups) >= 2 def _request_forbids_execution_retry(text: str) -> bool: """Return whether the user explicitly bounded command execution to one try.""" value = str(text or "") return bool( re.search( r"\b(?:do\s+not|don['’]?t|dont|never)\s+" r"(?:retry|re-?run|run\s+(?:it|that|the\s+command)\s+again)\b", value, re.IGNORECASE, ) or re.search( r"\b(?:run|execute|try)\b[^.!?\n]{0,120}\b(?:once|one\s+time)\b", value, re.IGNORECASE, ) ) def _contract_mutation_signature(block, contract): """Deduplicate an exact successful mutation for the rest of this turn. A model may continue after a successful write in order to verify or summarize it. That continuation must never execute the same state-changing call again, regardless of whether the turn contract names one capability or several. """ from src.tool_capabilities import ToolEffect, capabilities_for_action effects = capabilities_for_action(block.tool_type, block.content).effects if not effects & {ToolEffect.WRITE_PRIVATE, ToolEffect.WRITE_WORKSPACE, ToolEffect.EXTERNAL_SIDE_EFFECT, ToolEffect.ADMIN_CHANGE, ToolEffect.DESTRUCTIVE}: return None content = block.content or "" try: content = json.dumps(json.loads(content), sort_keys=True, separators=(",", ":")) except (TypeError, ValueError): pass return block.tool_type, content def _has_accepted_contract_tool_call(contract, tool_blocks) -> bool: """Accepted calls own their arguments; intent recovery only fills a gap.""" return contract is not None and any( contract.permits(block.tool_type) for block in (tool_blocks or ()) ) def _required_safe_read_operation(contract): """Consume the optional operation without expanding permissions or scope.""" operation = getattr(contract, "required_operation", None) if operation is None: operation = getattr(contract, "required_read_operation", None) active = getattr(contract, "active_capabilities", None) operation_scope = active if active else getattr(contract, "capabilities", ()) if operation is None or len(operation_scope) > 1: return None def field(name, default=None): return operation.get(name, default) if isinstance(operation, Mapping) else getattr(operation, name, default) name, args, limit = field("tool_name", field("tool")), field("args"), field("max_items") # Email account metadata is safe; mailbox contents and mutations stay out. # Deliberately exclude web/search and shell/files. supported = { "manage_notes", "manage_calendar", "manage_tasks", "manage_documents", "manage_memory", "manage_skills", "list_models", "list_cookbook_servers", "list_cached_models", "list_served_models", "list_serve_presets", "list_downloads", "list_email_accounts", "mcp__email__list_email_accounts", } if name not in supported or not isinstance(args, Mapping) or not contract.permits(name): return None if limit is not None and (type(limit) is not int or limit < 0): return None try: content = json.dumps(dict(args), sort_keys=True, ensure_ascii=False, allow_nan=False) except (TypeError, ValueError): return None from src.tool_capabilities import ToolEffect capability = capabilities_for_action(name, content) if not capability.known or capability.effects != frozenset({ToolEffect.READ_PRIVATE}): return None return ToolBlock(name, content), limit def _required_read_native_id(block, native_calls): """Keep the native ID only when the model supplied the immutable operation.""" expected = json.loads(block.content) def canonical_name(name): return "list_email_accounts" if name == "mcp__email__list_email_accounts" else name for call in native_calls or (): function = call.get("function") or call if canonical_name(function.get("name")) != canonical_name(block.tool_type): continue args = function.get("arguments") try: args = json.loads(args) if isinstance(args, str) else args except (TypeError, ValueError): continue if args == expected: return call.get("id") return None def _required_read_summary(block, result, max_items=None): raw = next((result.get(key) for key in ("output", "response", "results", "content") if result.get(key)), "") if not isinstance(raw, str): raw = json.dumps(raw, ensure_ascii=False, default=str) raw = _strip_think_blocks(strip_tool_blocks(raw)).removeprefix("AI: ").strip() if max_items == 0: return "Read completed; no items displayed." args = json.loads(block.content) action = str(args.get("action") or "").lower() summary = "" bounded_helpers = { "manage_notes": _note_list_summary_from_tool_output, "manage_calendar": _calendar_list_summary_from_tool_output, "manage_documents": _document_list_summary_from_tool_output, "manage_skills": _skills_list_summary_from_tool_output, } if max_items is not None and action in {"list", "list_events", "index", "search", "find", "lis"}: helper = bounded_helpers.get(block.tool_type) if helper: summary = helper(raw, max_items=max_items) if not summary: summary = _ody_qwen_terminal_tool_summary({ "tool": block.tool_type, "command": block.content, "output": raw, }) or raw if max_items is not None: # Existing renderers embed overflow items in expandable HTML comments. # A contract cap bounds the actual answer payload, including overflow. summary = summary.split("\n[...and {len(hidden)} more notes](#notes-more-{hidden_id})") return "\n".join(lines) def _note_title_id_pairs_from_tool_output(raw: str) -> list[tuple[str, str]]: if not isinstance(raw, str) or not raw.strip(): return [] pairs: list[tuple[str, str]] = [] seen: set[tuple[str, str]] = set() def add_pair(title: Any, note_id: Any) -> None: clean_title = re.sub(r"\s+", " ", str(title or "")).strip() clean_id = str(note_id or "").strip() if len(clean_title) < 2 or not clean_id: return key = (clean_title, clean_id) if key not in seen: pairs.append(key) seen.add(key) for match in re.finditer(r"\[([^\]]+)\]\(#note-([^)]+)\)", raw): add_pair(match.group(1), match.group(2)) for line in raw.splitlines(): match = re.match(r"^\s*-\s+\[([^\]]+)\]\s+\*\*(.*?)\*\*", line) if match: add_pair(match.group(2), match.group(1)) return pairs def _linkify_note_titles_from_tool_events(answer: str, tool_events: list[dict[str, Any]]) -> str: """Add #note links to synthesized note answers using real note tool output.""" text = str(answer or "") if not text.strip() or not tool_events: return text title_to_id: dict[str, str] = {} for event in tool_events or []: if _resolved_tool_event_name(event) != "manage_notes": continue if not tool_result_is_successful(event): continue if event.get("note_id") and event.get("note_title"): title_to_id.setdefault( str(event.get("note_title") or "").strip(), str(event.get("note_id") or "").strip(), ) for title, note_id in _note_title_id_pairs_from_tool_output(event.get("output") or ""): title_to_id.setdefault(title, note_id) title_to_id = {title: note_id for title, note_id in title_to_id.items() if title and note_id} if not title_to_id: return text titles = sorted(title_to_id, key=len, reverse=True) linked_lines: list[str] = [] for line in text.splitlines(): if "#note-" in line: linked_lines.append(line) continue updated = line for title in titles: if title not in updated: continue note_id = title_to_id[title] label = title.replace("\\", "\\\\").replace("[", "\\[").replace("]", "\\]") link = f"[{label}](#note-{note_id})" bold_pattern = re.compile(rf"\*\*{re.escape(title)}\*\*") if bold_pattern.search(updated): updated = bold_pattern.sub(f"**{link}**", updated, count=1) continue updated = updated.replace(title, link, 1) linked_lines.append(updated) return "\n".join(linked_lines) def _notes_expected_actions(user_text: str) -> set[str]: value = str(user_text or "").strip().lower() if not value: return set() if re.search(r"\b(?:delete|remove|clear)\b", value): return {"delete", "remove"} if re.search(r"\b(?:check\s+off|mark\s+(?:done|complete)|toggle|uncheck)\b", value): return {"toggle_item", "update"} if re.search(r"\b(?:update|change|edit|rename|tag|retag|pin|unpin|color|colour)\b", value): return {"update", "edit"} if re.search(r"\b(?:add|create|make|write\s+down|jot|save|remind)\b", value): return {"add", "create", "save", "remind"} if re.search(r"\b(?:show|list|search|find|open|view|read|what|which)\b", value): return {"list", "search", "find", "view", "lis"} return set() def _split_note_items(value: str) -> list[dict[str, Any]]: parts = [ re.sub(r"\s+", " ", part).strip(" .") for part in re.split(r"\s*,\s*|\s+\band\b\s+", str(value or "")) ] return [{"text": part, "done": False} for part in parts if part] def _clean_notes_search_query(value: str) -> str: query = re.sub(r"\s+", " ", str(value or "")).strip(" .\"'") query = re.sub(r"^(?:the|my|a|an)\s+", "", query, flags=re.IGNORECASE) query = re.sub(r"\s+(?:note|notes|checklist|list|reminder)\s*$", "", query, flags=re.IGNORECASE) query = re.sub(r"\s+", " ", query).strip(" .\"'") return query def _notes_general_definition_answer(text: str) -> Optional[str]: """Answer note-like word questions that are not saved-note requests.""" value = re.sub(r"\s+", " ", str(text or "")).strip() lower = value.lower() if not value: return None if re.search(r"\b(?:my|saved|open|show|list|search|find|create|add|delete|archive|pin|tag)\s+(?:notes?|checklists?)\b", lower): return None if not re.search(r"\b(?:what(?:'s| is)?|define|explain|meaning|mean|difference|synonym|sentence)\b", lower): return None if re.search(r"\bmusical\s+note\b|\bnote\s+in\s+music\b|\bmusic\s+theory\b", lower): return "A musical note is a written or sounded pitch with a duration." if re.search(r"\bpinned\b|\bpinning\b", lower): return "Pinned usually means an item is kept fixed, visible, or prioritized in place." if re.search(r"\barchiv(?:e|ed|ing)\b", lower): return "Archive means store something for later reference instead of keeping it active." if re.search(r"\bchecklist\b", lower) and not re.search( r"\b(?:left|remaining|complete|completed|done|unfinished|pending)\b", lower, ): return "A checklist is a list where items can be marked complete." if re.search(r"\btag\b|\btagged\b", lower): return "A tag is a label used to categorize or find an item." if re.search(r"\bcolor coding\b|\bcolour coding\b", lower): return "Color coding means using colors to classify or distinguish information." if re.search(r"\b(?:word\s+)?note\b", lower): if re.search(r"\bsentence\b", lower): return "Please note that the meeting starts at noon." if re.search(r"\bsynonym\b", lower): return "A useful synonym for note is memo, comment, or remark depending on context." return "A note can mean a short written record, a comment, or a musical pitch depending on context." return None def _is_personal_tool_definition_turn(text: str) -> bool: """Recognize definitions that mention app nouns without requesting app data.""" q = re.sub(r"\s+", " ", str(text or "").lower()).strip() return bool( re.match( r"^(?:what(?:'s| is)|define|explain)\s+(?:(?:a|an|the)\s+)?" r"(?:calendar|event|meeting|appointment|schedule|note|task|memory|skill)\b", q, ) or re.match( r"^what\s+does\s+(?:(?:computer|human|working|long[- ]term)\s+)?" r"(?:memory|calendar|event|schedule|note|task|skill)\s+mean\b", q, ) ) def _parse_simple_notes_tool_request(text: str) -> Optional[tuple[str, str]]: """Deterministic fallback for obvious notes commands when a model stalls.""" value = str(text or "").strip() lower = value.lower() if not value: return None if ( _parse_explicit_open_panel_request(value) and not re.search( r"\b(?:create|add|make|save|write|edit|update|change|delete|remove|archive|pin|tag)\b", lower, ) ): return None if _notes_general_definition_answer(value): return None explicit_note_create = bool( re.search(r"\b(?:create|add|make|save|write\s+down|jot)\b.{0,80}\bnotes?\b", lower) or re.search(r"\bnotes?\b.{0,80}\b(?:create|add|make|save|write\s+down|jot)\b", lower) ) if re.search(r"\b(?:email|mail|inbox)\b", lower): return None label_match = re.search( r"\b(?:tagged|under)\s+#?([a-zA-Z0-9_-]{2,40})\b" r"|\b(?:tag|label(?:ed)?)\s+(?:it\s+)?(?:as\s+)?#?([a-zA-Z0-9_-]{2,40})\b", value, re.IGNORECASE, ) if label_match: label = next((g for g in label_match.groups() if g), "").lower() else: label = "" checklist_match = re.search( r"\b(?:make|create|add)\s+(?:a\s+)?checklist\s+(?:called|titled|named)\s+(.+?)\s+with\s+(.+?)\s*$", value, re.IGNORECASE, ) if checklist_match: title = re.sub(r"\s+", " ", checklist_match.group(1)).strip(" .\"'") items = _split_note_items(checklist_match.group(2)) if title and items: return "manage_notes", json.dumps({ "action": "add", "title": title, "note_type": "checklist", "checklist_items": items, }) note_named_match = re.search( r"\b(?:create|add|make|save)\s+(?:a\s+|the\s+)?(?:short\s+)?note\s+" r"(?:called|titled|named)\s+(.+?)" r"(?:\s+(?:with|saying|that says|summari[sz]ing|about)\s+(.+?))?\s*$", value, re.IGNORECASE, ) if note_named_match: title = re.sub(r"\s+", " ", note_named_match.group(1)).strip(" .\"'") body = re.sub(r"\s+", " ", note_named_match.group(2) or title).strip(" .\"'") if title: args = {"action": "add", "title": title, "content": body or title} if label: args["label"] = label return "manage_notes", json.dumps(args) remaining_match = re.search( r"\b(?:what(?:'s| is)?|show|tell\s+me)\b.*?\b(?:left|remaining)\b.*?\b(?:on|in)\s+(?:the\s+)?(.+?)\s+checklist\b", value, re.IGNORECASE, ) if remaining_match: query = _clean_notes_search_query(remaining_match.group(1)) if query: return "manage_notes", json.dumps({"action": "search", "query": query}) note_saying_match = re.search( r"\b(?:create|add|make|save)\s+(?:a\s+)?note\s+(?:saying|that says|with)\s+(.+?)\s*$", value, re.IGNORECASE, ) if note_saying_match: body = re.sub( r"\s+(?:and\s+)?(?:tag|label)\s+(?:it\s+)?(?:as\s+)?#?[a-zA-Z0-9_-]{2,40}\s*$", "", note_saying_match.group(1), flags=re.IGNORECASE, ) title = re.sub(r"\s+", " ", body).strip(" .\"'") if title: args: dict[str, Any] = {"action": "add", "title": title, "content": title} if label: args["label"] = label return "manage_notes", json.dumps(args) if re.search(r"\b(?:show|list|see|what(?:'s| is)?)\b", lower) and re.search(r"\b(?:notes?|checklists?|reminders?)\b", lower): args = {"action": "list"} if label: args["label"] = label if re.search(r"\bpinned\b", lower): args["pinned"] = True if re.search(r"\breminders?\b", lower): args["reminders"] = True return "manage_notes", json.dumps(args) search_match = re.search( r"\b(?:search|find|open|view|read)\b(?:\s+(?:my\s+)?notes?)?(?:\s+(?:for|about))?\s+(.+?)\s*$", value, re.IGNORECASE, ) if search_match and re.search(r"\b(?:notes?|note|checklist|reminder)\b", lower): query = re.sub(r"\bnotes?\b", "", search_match.group(1), flags=re.IGNORECASE) query = _clean_notes_search_query(query) if query: args = {"action": "search", "query": query} if label: args["label"] = label return "manage_notes", json.dumps(args) delete_match = re.search( r"\b(?:delete|remove|clear)\s+(?:the\s+)?(.+?)\s*$", value, re.IGNORECASE, ) if delete_match and re.search(r"\b(?:notes?|note|checklist|list|reminder)\b", lower): title = re.sub(r"\b(?:note|checklist|list|reminder)\b", "", delete_match.group(1), flags=re.IGNORECASE) title = re.sub(r"\s+", " ", title).strip(" .\"'") if title: return "manage_notes", json.dumps({"action": "delete", "title": title}) if re.search(r"\b(?:calendar|events?|meeting|appointment)\b", lower) and not explicit_note_create: return None return None def _notes_body_requested(text: str) -> bool: value = str(text or "") return bool( re.search(r"\b(?:read|open|view)\b", value, re.IGNORECASE) or re.search(r"\b(?:what(?:'s| is)?|show|tell\s+me)\b.*?\b(?:left|remaining)\b", value, re.IGNORECASE) ) def _notes_request_requires_fresh_tool( user_text: str, intent_domains: Set[str], relevant_tools: Any, ) -> bool: if "notes_calendar_tasks" not in set(intent_domains or set()): return False try: if "manage_notes" not in set(relevant_tools or set()): return False except TypeError: return False value = str(user_text or "").strip().lower() if not value: return False if not ( re.search(r"\b(?:notes?|todos?|to-dos?|checklists?|reminders?)\b", value) or re.search(r"\b(?:packing|shopping|grocery)\s+list\b", value) or _looks_like_implicit_notes_turn(value) ): return False explicit_note_create = bool( re.search(r"\b(?:create|add|make|save|write\s+down|jot)\b.{0,80}\bnotes?\b", value) or re.search(r"\bnotes?\b.{0,80}\b(?:create|add|make|save|write\s+down|jot)\b", value) ) if re.search(r"\b(?:email|mail|inbox)\b", value): return False if re.search(r"\b(?:calendar|events?|meeting|appointment)\b", value) and not explicit_note_create: return False return bool(_notes_expected_actions(value)) def _has_successful_notes_action_evidence( tool_events: list[dict[str, Any]], expected_actions: set[str], ) -> bool: expected = {str(a or "").strip().lower() for a in expected_actions if a} if not expected: expected = {"list", "search", "find", "view", "add", "create", "update", "edit", "delete", "remove", "toggle_item"} aliases = { "create": "add", "new": "add", "save": "add", "remind": "add", "remove": "delete", } expected = {aliases.get(action, action) for action in expected} for event in tool_events or []: if not isinstance(event, dict): continue if _resolved_tool_event_name(event) != "manage_notes": continue if not tool_result_is_successful(event): continue command = str(event.get("command") or "").strip() action = "" try: parsed = json.loads(command or "{}") if isinstance(parsed, dict): action = str(parsed.get("action") or "").strip().lower() except Exception: action = command.splitlines()[0].strip().lower() if command else "" action = aliases.get(action, action) if action in expected: return True return False def _memory_list_summary_from_tool_output(raw: str, max_items: int = 20) -> str: """Keep broad memory listings reviewable without dumping the whole store.""" if not isinstance(raw, str) or not raw.strip(): return "" # The memory tool may already return the compact form. Treat it as a # complete answer so the agent does not spend a second round asking the # model to summarize an answer that is already summarized. compact_match = re.fullmatch( r"Memory:\s+\d+\s+saved\s+entries?(?:\s+\([^\n]+\))?\.?", raw.strip(), re.IGNORECASE, ) if compact_match: return raw.strip() if re.search(r"\bno memories found\b", raw, re.IGNORECASE): return "No saved memories found." count_match = re.search(r"Found\s+(\d+)\s+memory entries", raw, re.IGNORECASE) compact_count_match = re.search(r"Memory:\s+(\d+)\s+saved\s+entries?", raw, re.IGNORECASE) if not count_match: if not compact_count_match: return "" total = int((count_match or compact_count_match).group(1)) categories: collections.Counter[str] = collections.Counter() items: list[str] = [] all_items: list[str] = [] for line in raw.splitlines(): match = re.match(r"^\s*-\s+\[([^\]]+)\]", line) if match: categories[match.group(1).strip().lower()] += 1 item_match = re.match( r"^\s*-\s+\[([^\]]+)\]\s+`([^`]+)`\s+[—-]\s+(.+?)\s*$", line, ) if item_match: category = item_match.group(1).strip() memory_id = item_match.group(2).strip() text = re.sub(r"\s+", " ", item_match.group(3)).strip() row = f"- [{category} {memory_id}](#memory-{quote(memory_id, safe='')}) — {text}" all_items.append(row) if len(items) < max_items: items.append(row) compact_header_match = re.search( r"^(Memory:\s+\d+\s+saved\s+entr(?:y|ies)(?:\s+\([^\n]+\))?\.?)", raw.strip(), re.IGNORECASE, ) if compact_header_match: header = compact_header_match.group(1).strip() else: category_text = ", ".join( f"{name} {count}" for name, count in sorted(categories.items()) ) suffix = f" ({category_text})" if category_text else "" header = f"Memory: {total} saved entr{'y' if total == 1 else 'ies'}{suffix}." if not items: return header remaining = total - len(items) if remaining > 0: # The Memory panel owns the complete browser. Embedding every omitted # memory in an invisible chat payload turned a simple list into a huge # terminal SSE event and copied private text into chat history. items.append( f"...and {remaining} more saved memories. Open Memory to browse all." ) return "\n".join([header, *items]) def _document_list_summary_from_tool_output(raw: str, max_items: int = 8) -> str: """Format manage_documents list output for chat without an LLM pass.""" if not isinstance(raw, str) or not raw.strip(): return "" text = raw.strip() if text.startswith("AI: "): text = text[4:].strip() if re.search(r"\b(no documents|0 documents|found 0)\b", text, re.IGNORECASE): return "No documents found." lines = [line.strip() for line in text.splitlines() if line.strip()] if not lines: return "" # manage_documents already returns click-ready markdown rows. Keep its # compact shape, but cap very large libraries for chat. heading = lines[0] rows = [line for line in lines[1:] if line.startswith(("-", "*"))] if rows: continuation = next( ( row for row in rows if re.match(r"^[-*]\s+\.\.\.and\s+\d+\s+more\b", row, re.IGNORECASE) ), "", ) real_rows = [ row for row in rows if not re.match(r"^[-*]\s+\.\.\.and\s+\d+\s+more\b", row, re.IGNORECASE) ] clipped = real_rows[:max_items] if continuation: clipped.append(continuation) elif len(real_rows) > len(clipped): clipped.append(f"- ...and {len(real_rows) - len(clipped)} more") return "\n".join([heading, *clipped]) return "\n".join(lines[: max_items + 1]) def _document_read_summary_from_tool_output(raw: str) -> str: """Return document read output as the answer body.""" if not isinstance(raw, str) or not raw.strip(): return "" text = raw[4:].strip() if raw.startswith("AI: ") else raw.strip() return text def _document_detail_requested(text: str) -> bool: """Whether a document locator must be followed by a read/open call.""" t = (text or "").lower() if not re.search(r"\b(doc|docs|document|documents|library|file|files)\b", t): return False return bool( re.search( r"\b(read|open|view|show|display|summari[sz]e|quote|contents?|body|text|inside|passphrase|phrase|detail|details)\b", t, ) ) def _single_document_id_from_tool_output(raw: str) -> str: """Extract the sole document id from a manage_documents list/search result.""" if not isinstance(raw, str) or not raw.strip(): return "" ids = { match.group(1).strip() for match in re.finditer(r"#document-([A-Za-z0-9][A-Za-z0-9_.:-]*)", raw) } return next(iter(ids)) if len(ids) == 1 else "" def _session_list_summary_from_tool_output(raw: str, max_items: int = 12) -> str: """Keep a broad session listing readable and terminal for small routers.""" if not isinstance(raw, str) or not raw.strip(): return "" text = raw.strip() if text.startswith("AI: "): text = text[4:].strip() lines = [line.strip() for line in text.splitlines() if line.strip()] if not lines: return "" if re.search(r"\b(no chats|no sessions|0 sessions)\b", text, re.IGNORECASE): return lines[0] rows = [line for line in lines[1:] if line.startswith("-")] if not rows: return "\n".join(lines[: max_items + 1]) formatted_rows: list[str] = [] for row in rows: link_match = re.search(r"(\[(?:\\.|[^\]])+\]\(#session-[^)]+\))", row) if link_match: meta_match = re.search(r"\(([^()]*(?:last active|msgs|model|id:)[^()]*)\)", row) meta = meta_match.group(1) if meta_match else "" active = re.search(r"last active [^)]+", meta) suffix = f" ({active.group(0)})" if active else "" formatted_rows.append(f"- {link_match.group(1)}{suffix}") else: formatted_rows.append(row[:180].rstrip() + ("..." if len(row) > 180 else "")) shown = formatted_rows[:max_items] hidden = formatted_rows[max_items:] if hidden: shown.append(f"- ...and more sessions ({len(hidden)} hidden)") return "\n".join([lines[0], *shown]) def _registry_list_summary_from_tool_output(raw: str, max_items: int = 12) -> str: """Bound simple list/read registry output without another model round.""" if not isinstance(raw, str) or not raw.strip(): return "" text = raw[4:].strip() if raw.startswith("AI: ") else raw.strip() lines = [line.strip() for line in text.splitlines() if line.strip()] # A registry can return one enormous JSON/markdown line, so a line-count # limit alone is not a size bound. Preserve useful leading fields while # keeping the terminal SSE event comfortably below a normal model chunk. clipped = [ line if len(line) <= 320 else line[:317].rstrip() + "..." for line in lines[: max_items + 1] ] if len(lines) > len(clipped): clipped.append("- ...and more") summary = "\n".join(clipped) return summary if len(summary) <= 3200 else summary[:3197].rstrip() + "..." def _research_list_summary_from_tool_output(raw: str, max_items: int = 6) -> str: """Keep saved research listings concise while preserving report anchors.""" if not isinstance(raw, str) or not raw.strip(): return "" text = raw[4:].strip() if raw.startswith("AI: ") else raw.strip() lines = [line.strip() for line in text.splitlines() if line.strip()] if not lines: return "" if re.search(r"\b(no research|0 research|0 items)\b", text, re.IGNORECASE): return lines[0] rows: list[str] = [] for line in lines[1:]: match = re.match(r"^-\s+\[(.*?)\]\(#research-([^)]+)\)(.*)$", line) if not match: continue title = re.sub(r"\s+", " ", match.group(1)).strip() if len(title) > 110: title = title[:107].rstrip() + "..." suffix = re.sub(r"\s+", " ", match.group(3) or "").strip() rows.append(f"- [{title}](#research-{match.group(2)}) {suffix}".rstrip()) if len(rows) >= max_items: break if not rows: return "\n".join(lines[: max_items + 1]) total_match = re.search(r"\((\d+)\s+items?\)", lines[0], re.IGNORECASE) total = int(total_match.group(1)) if total_match else len(rows) if total > len(rows): rows.append(f"- ...and {total - len(rows)} more research reports") return "\n".join([lines[0], *rows]) def _skills_list_summary_from_tool_output(raw: str, max_items: int = 8) -> str: """Keep the skill index visible without dumping the full registry.""" if not isinstance(raw, str) or not raw.strip(): return "" text = raw[4:].strip() if raw.startswith("AI: ") else raw.strip() lines = [line.strip() for line in text.splitlines() if line.strip()] if not lines: return "" section = "" rows: list[tuple[str, str]] = [] totals = {"Published": 0, "Drafts": 0} for line in lines: section_match = re.match(r"^##\s+(Published|Drafts)\b", line, re.IGNORECASE) if section_match: section = section_match.group(1).title() continue if not line.startswith("-"): continue label = section or "Skills" if label in totals: totals[label] += 1 match = re.match(r"^-\s+\*\*(.*?)\*\*(?:\s+\((.*?)\)|\s+\[(draft)\])?(?::\s*(.*))?$", line) if match: name = re.sub(r"\s+", " ", match.group(1)).strip() meta = re.sub(r"\s+", " ", (match.group(2) or match.group(3) or label).strip()) rows.append((label, f"- [{name}](#skill-{quote(name, safe='')}) ({meta})")) else: rows.append((label, line[:96].rstrip() + ("..." if len(line) > 96 else ""))) if not rows: clipped = lines[:max_items] if len(lines) > len(clipped): clipped.append("- ...and more skills") return "Available skills:\n" + "\n".join(clipped) shown = rows[:max_items] total = len(rows) heading_bits = [] if totals["Published"]: heading_bits.append(f"{totals['Published']} published") if totals["Drafts"]: heading_bits.append(f"{totals['Drafts']} drafts") heading = "Available skills" if heading_bits: heading += f" ({', '.join(heading_bits)})" out = [heading + ":"] current = "" for label, row in shown: if label != current: out.append(f"## {label}") current = label out.append(row) if total > len(shown): # Keep the terminal event genuinely compact; the Skills panel remains # the complete registry browser. out.append( f"...and {total - len(shown)} more skills. Open Skills to browse all." ) return "\n".join(out) def _calendar_detail_requested(text: str) -> bool: """Whether a calendar listing answer should preserve event details.""" t = (text or "").lower() if not re.search(r"\b(calendar|event|events|schedule|appointment|appointments)\b", t): return False return bool( re.search( r"\b(description|descriptions|detail|details|note|notes|passphrase|phrase|where|location|agenda|about)\b", t, ) ) def _calendar_list_summary_from_tool_output( raw: str, max_items: int = 20, include_details: bool = False, user_text: str = "", ) -> str: """Format manage_calendar list_events output for chat without an LLM pass.""" if not isinstance(raw, str) or not raw.strip(): return "" text = raw[4:].strip() if raw.startswith("AI: ") else raw.strip() if re.search(r"\bno events between\b", text, re.IGNORECASE): query = str(user_text or "").lower() if re.search(r"\btoday(?:'?s)?\b", query): return "You have no events today." if re.search(r"\btomorrow(?:'?s)?\b", query): return "You have no events tomorrow." return text.splitlines()[0] def format_when(value: str) -> str: raw_when = re.sub(r"\s+", " ", value or "").strip() all_day_match = re.match(r"^(\d{4}-\d{2}-\d{2})\s*\(all day\)$", raw_when, re.IGNORECASE) if all_day_match: try: parsed = datetime.fromisoformat(all_day_match.group(1)) return f"{parsed.strftime('%b')} {parsed.day} · All day" except ValueError: return raw_when parts = re.split(r"\s*->\s*", raw_when, maxsplit=1) if len(parts) != 2: return raw_when try: start = datetime.fromisoformat(parts[0].replace("Z", "+00:00")) end = datetime.fromisoformat(parts[1].replace("Z", "+00:00")) if start.tzinfo is not None: from src.user_time import user_timezone start = start.astimezone(user_timezone()) end = end.astimezone(user_timezone()) except (TypeError, ValueError): return raw_when def time_label(dt: datetime) -> str: return dt.strftime("%-I:%M %p") start_date = f"{start.strftime('%b')} {start.day}" if start.date() == end.date(): return f"{start_date}, {time_label(start)}–{time_label(end)}" end_date = f"{end.strftime('%b')} {end.day}" return f"{start_date}, {time_label(start)}–{end_date}, {time_label(end)}" items: list[str] = [] current_item_idx = -1 for line in text.splitlines(): m = re.match(r"^\s*-\s+(.+?):\s+\[(.*?)\]\(#event-([^)]+)\)(.*)$", line) if not m: if include_details and current_item_idx >= 0: detail = re.sub(r"\s+", " ", line).strip() if detail and not detail.startswith("-"): items[current_item_idx] = f"{items[current_item_idx]} — {detail}" continue when = re.sub(r"\s+", " ", m.group(1)).strip() title = re.sub(r"\s+", " ", m.group(2)).strip() event_id = m.group(3).strip() suffix = re.sub(r"\s+", " ", m.group(4) or "").strip() label = f"[{title}](#event-{event_id}) — {format_when(when)}" if suffix: label += f" {suffix}" items.append(label) current_item_idx = len(items) - 1 if not items: return "" total_match = re.search(r"Found\s+(\d+)\s+event", text, re.IGNORECASE) total = int(total_match.group(1)) if total_match else len(items) lines = [f"I found {total} calendar event{'s' if total != 1 else ''} in that range:"] shown = items[:max_items] hidden = items[max_items:] lines.extend(f"- {item}" for item in shown) if hidden: hidden_text = "\n".join(f"- {item}" for item in hidden) hidden_id = hashlib.sha1(hidden_text.encode("utf-8")).hexdigest()[:12] lines.append( f"\n" f"[...and {len(hidden)} more events](#events-more-{hidden_id})" ) elif total > len(items): lines.append(f"...and {total - len(items)} more events") return "\n".join(lines) _ORDINAL_WEEKDAY_CODES = { "monday": "MO", "tuesday": "TU", "wednesday": "WE", "thursday": "TH", "friday": "FR", "saturday": "SA", "sunday": "SU", } _ORDINAL_RRULE_PREFIXES = { "first": "1", "1st": "1", "second": "2", "2nd": "2", "third": "3", "3rd": "3", "fourth": "4", "4th": "4", "fifth": "5", "5th": "5", "last": "-1", "final": "-1", } def _ordinal_weekday_monthly_rrule_from_text(text: str) -> Optional[str]: """Return an RRULE for "2nd Thursday of the month" style requests.""" q = re.sub(r"\s+", " ", str(text or "").lower()).strip() if not q or "month" not in q: return None byday: list[str] = [] for name, code in _ORDINAL_WEEKDAY_CODES.items(): for ordinal, prefix in _ORDINAL_RRULE_PREFIXES.items(): if re.search(rf"\b{ordinal}\s+{name}\b(?:\s+of\s+(?:the\s+)?month)?", q): token = f"{prefix}{code}" if token not in byday: byday.append(token) if byday: return f"FREQ=MONTHLY;BYDAY={','.join(byday)}" return None def _ambiguous_ordinal_weekday_of_week(text: str) -> Optional[str]: """Detect contradictory "first and last Monday of the week" requests.""" q = re.sub(r"\s+", " ", str(text or "").lower()).strip() if not q or "month" in q: return None if not re.search(r"\bweek\b", q): return None if not re.search(r"\bfirst\b", q) or not re.search(r"\blast\b", q): return None for name in _ORDINAL_WEEKDAY_CODES: if re.search(rf"\b{name}\b", q): return name return None def _normalize_calendar_ordinal_weekday_rrule( args: dict[str, Any], last_user: str, ) -> tuple[dict[str, Any], bool]: if not isinstance(args, dict): return args, False action = str(args.get("action") or "").strip().lower() action = { "create": "create_event", "update": "update_event", }.get(action, action) if action not in {"create_event", "update_event"}: return args, False rrule = _ordinal_weekday_monthly_rrule_from_text(last_user) if not rrule: return args, False normalized = dict(args) normalized["action"] = action normalized["rrule"] = rrule return normalized, normalized != args def _calendar_ordinal_week_ask_user_block(last_user: str) -> Optional[ToolBlock]: weekday = _ambiguous_ordinal_weekday_of_week(last_user) if not weekday: return None cap = weekday.capitalize() payload = { "question": ( f"A week only has one {cap}. Did you mean the first and last " f"{cap} of each month?" ), "options": [ {"label": "Each month", "description": f"Create a monthly event on the first and last {cap}."}, {"label": "Every week", "description": f"Create a weekly event every {cap}."}, {"label": "Exact rule", "description": "I'll type the recurrence I want."}, ], } return ToolBlock("ask_user", json.dumps(payload, ensure_ascii=False)) def _normalize_calendar_list_range_args( args: dict[str, Any], *, today: Any = None, user_text: str = "", ) -> tuple[dict[str, Any], bool]: """Convert obvious relative calendar list ranges to concrete ISO dates.""" if not isinstance(args, dict): return args, False action = str(args.get("action") or "").strip().lower() if action not in {"list", "list_events", "lis_events"}: return args, False from datetime import date, datetime, timedelta if today is None: try: from src.user_time import now_user_local today_date = now_user_local().date() except Exception: today_date = date.today() elif isinstance(today, datetime): today_date = today.date() elif isinstance(today, date): today_date = today else: today_date = datetime.strptime(str(today)[:10], "%Y-%m-%d").date() def _week_bounds(offset_weeks: int = 0) -> tuple[str, str]: monday = today_date - timedelta(days=today_date.weekday()) + timedelta(days=7 * offset_weeks) return monday.isoformat(), (monday + timedelta(days=7)).isoformat() def _day_bounds(offset_days: int = 0) -> tuple[str, str]: start = today_date + timedelta(days=offset_days) return start.isoformat(), (start + timedelta(days=1)).isoformat() relative_start = str( args.get("start") or args.get("start_date") or args.get("from") or "" ).strip().lower() start: str | None = None end: str | None = None if relative_start in {"next week", "the next week"}: start, end = _week_bounds(1) elif relative_start in {"this week", "current week"}: start, end = _week_bounds(0) elif relative_start == "today": start, end = _day_bounds(0) elif relative_start == "tomorrow": start, end = _day_bounds(1) elif relative_start in {"next 7 days", "the next 7 days", "coming week"}: start = today_date.isoformat() end = (today_date + timedelta(days=7)).isoformat() if not start or not end: broad_calendar_read = bool(re.search( r"\bwhat(?:['’]?s|\s+is)\s+on\s+(?:my|our|the)\s+calendar\b|" r"\b(?:list|show|check)\s+(?:me\s+)?(?:my|our|the)?\s*" r"(?:calendar|calendar\s+events|schedule)\b", str(user_text or ""), re.IGNORECASE, )) if not broad_calendar_read: return args, False prompt_bounds = _calendar_bounds_for_prompt(user_text, today=today_date) if not prompt_bounds: return args, False # A broad listing has no user-authored title filter. Discard model # guesses such as a fabricated schedule string or narrow clock range. return { "action": "list_events", "start": prompt_bounds[0], "end": prompt_bounds[1], }, True normalized = dict(args) normalized["action"] = "list_events" normalized["start"] = start normalized["end"] = end for alias in ("start_date", "end_date", "from", "to"): normalized.pop(alias, None) return normalized, normalized != args def _calendar_bounds_for_prompt(text: str, *, today: Any = None) -> Optional[tuple[str, str]]: from datetime import date, datetime, timedelta if today is None: try: from src.user_time import now_user_local today_date = now_user_local().date() except Exception: today_date = date.today() elif isinstance(today, datetime): today_date = today.date() elif isinstance(today, date): today_date = today else: today_date = datetime.strptime(str(today)[:10], "%Y-%m-%d").date() q = re.sub(r"\s+", " ", str(text or "").lower()).strip() q = re.sub(r"\btodays\b", "today's", q) q = re.sub(r"\btomorrows\b", "tomorrow's", q) if not q: return None if re.search(r"\btoday\b", q) and re.search(r"\btomorrow\b", q): return today_date.isoformat(), (today_date + timedelta(days=2)).isoformat() if re.search(r"\btoday\b|\btonight\b", q): return today_date.isoformat(), (today_date + timedelta(days=1)).isoformat() if re.search(r"\btomorrow\b", q): day = today_date + timedelta(days=1) return day.isoformat(), (day + timedelta(days=1)).isoformat() if re.search(r"\b(?:latest|upcoming|coming up|next events?|next appointments?)\b", q): return today_date.isoformat(), (today_date + timedelta(days=14)).isoformat() month_names = { "january": 1, "february": 2, "march": 3, "april": 4, "may": 5, "june": 6, "july": 7, "august": 8, "september": 9, "october": 10, "november": 11, "december": 12, } for name, month in month_names.items(): if re.search(rf"\b{name}\b", q): year_match = re.search(r"\b(20\d{2})\b", q) year = int(year_match.group(1)) if year_match else today_date.year start = date(year, month, 1) end = date(year + (1 if month == 12 else 0), 1 if month == 12 else month + 1, 1) return start.isoformat(), end.isoformat() if re.search(r"\b(?:recurring|repeat(?:ing)?|trash|travel)\b", q): return today_date.isoformat(), (today_date + timedelta(days=365)).isoformat() return today_date.isoformat(), (today_date + timedelta(days=30)).isoformat() def _parse_simple_calendar_tool_request( text: str, messages: Optional[List[Dict]] = None, history_session: Any = None, ) -> Optional[tuple[str, str]]: """Deterministic fallback for obvious calendar lookup/update prompts.""" value = str(text or "").strip() q = value.lower() # Chat input commonly omits apostrophes. Normalize only these intent # words so "whats my calendar" and "whats todays calendar" retain the # same semantics as their punctuated forms. q = re.sub(r"\bwhats\b", "what's", q) q = re.sub(r"\btodays\b", "today's", q) if not q: return None # Definitions are no-tool questions, not requests to inspect the user's # calendar. Without this boundary, "What is a calendar?" causes a lookup. if _is_personal_tool_definition_turn(q): return None calendar_mutation_requested = bool(re.search( r"\b(?:add|create|schedule|book|move|reschedule|rename|update|change|edit|delete|remove|cancel)\b", q, )) refs = _recent_odysseus_anchor_refs(messages or [], history_session) contextual_event_lookup = bool( refs.get("event_uid") and re.search(r"\b(?:show|list|check|what(?:'s| is| are)?|when|find|see)\b", q) and re.search(r"\b(?:it|this|that|entry|item|prep|block)\b", q) ) if not calendar_mutation_requested and ( ( re.search( r"\b(?:show|list|check|what(?:'s| is| are)?|when|find|see)\b" r"|\b(?:do\s+i\s+have|are\s+there)\b", q, ) and re.search( r"\b(?:calendar|events?|meetings?|appointments?|schedule|recurring|trash|travel)\b", q, ) ) or contextual_event_lookup ): bounds = _calendar_bounds_for_prompt(value) if not bounds: return None args: dict[str, Any] = {"action": "list_events", "start": bounds[0], "end": bounds[1]} if contextual_event_lookup and refs.get("event_title"): args["query"] = refs["event_title"] elif re.search(r"\btrash\b", q): args["query"] = "trash" elif re.search(r"\btravel\b", q): args["query"] = "travel" return "manage_calendar", json.dumps(args, ensure_ascii=False) tag_match = re.search( r"\b(?:change|update|set|retag)\b\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b", value, re.IGNORECASE, ) if tag_match and re.search(r"\b(?:calendar|event|trip|meeting|appointment)\b", q): title = re.sub(r"\s+", " ", tag_match.group(1)).strip(" .") if title: return "manage_calendar", json.dumps({ "action": "update_event", "summary": title, "tag": tag_match.group(2).lower(), }, ensure_ascii=False) return None def _parse_ambiguous_calendar_date_ask_user(text: str) -> Optional[tuple[str, str]]: value = str(text or "").strip() q = value.lower() if not q or not re.search(r"\b(?:event|calendar|reservation|dinner|lunch|meeting|appointment)\b", q): return None if not re.search(r"\b(?:add|create|schedule|book|event)\b", q): return None if not re.search(r"\bnext\s+month\b", q): return None # An ordinal weekday is a complete, deterministic date specification once # the request supplies "next month" (for example, "the last Wednesday of # next month"). Do not preempt a capable model with an unnecessary # ask_user turn merely because the user did not spell out a calendar day. if re.search( r"\b(?:first|second|third|fourth|last)\s+" r"(?:monday|tuesday|wednesday|thursday|friday|saturday|sunday)\b" r"(?:\s+of\s+(?:the\s+)?next\s+month)?", q, ): return None if re.search(r"\b(?:20\d{2}-\d{2}-\d{2}|\b\d{1,2}/\d{1,2}\b|jan(?:uary)?|feb(?:ruary)?|mar(?:ch)?|apr(?:il)?|may|jun(?:e)?|jul(?:y)?|aug(?:ust)?|sep(?:tember)?|oct(?:ober)?|nov(?:ember)?|dec(?:ember)?)\s+\d{1,2}\b", q): return None if not re.search(r"\b\d{1,2}(?::\d{2})?\s*(?:am|pm)?\b", q): return None try: from src.user_time import now_user_local today = now_user_local().date() except Exception: from datetime import date today = date.today() month = today.month + 1 year = today.year if month == 13: month = 1 year += 1 month_name = [ "", "January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December", ][month] place_match = re.search(r"\b(?:at|in)\s+(.+?)(?:\s+\d{1,2}(?::\d{2})?\s*(?:am|pm)?|\s+reservation|\s+remind|$)", value, re.IGNORECASE) place = place_match.group(1).strip(" .") if place_match else "the event" question = f"What day in {month_name} {year} is {place}?" return "ask_user", json.dumps({ "question": question, "options": [ {"label": "Exact date", "description": f"Type the date, e.g. {month_name} 12"}, {"label": "Cancel", "description": "Don't create the event yet"}, ], }, ensure_ascii=False) def _normalize_calendar_create_relative_args( args: dict[str, Any], last_user: str, ) -> tuple[dict[str, Any], bool]: """Clamp obvious relative create-event dates to the user's current date. Small local tool-router adapters can emit stale absolute dates learned from training examples. If the user said "tomorrow", the harness has enough trusted clock context to correct the date while preserving the chosen time. """ if not isinstance(args, dict): return args, False action = str(args.get("action") or "").strip().lower() action = { "create": "create_event", "update": "update_event", "delete": "delete_event", }.get(action, action) if action not in {"create_event", "update_event"}: return args, False raw_start = args.get("dtstart") or args.get("start") or args.get("start_time") if not raw_start: return args, False from datetime import date, datetime, timedelta user_text = last_user or "" user_mentions_timezone = bool(re.search( r"\b(?:utc|gmt|jst|pst|pdt|est|edt|cst|cdt|mst|mdt|" r"[a-z]+/[a-z_]+|timezone|time\s*zone)\b", user_text, re.IGNORECASE, )) def _strip_iso_timezone(value: Any) -> tuple[Any, bool]: text = str(value or "").strip() if not text: return value, False stripped = re.sub(r"(?:[Zz]|[+\-]\d{2}:?\d{2})$", "", text).strip() return stripped, stripped != text mentions_tomorrow = bool( re.search(r"\b(?:tomorrow|tmrw|tmr)\b", user_text, re.IGNORECASE) ) weekday_match = re.search( r"\b(?:(?:this|next)\s+)?(monday|tuesday|wednesday|thursday|friday|saturday|sunday)\b", user_text, re.IGNORECASE, ) if weekday_match and re.search(r"\b(?:every|each|weekly|recurr(?:ing|ence)?)\b", user_text, re.IGNORECASE): weekday_match = None if not user_mentions_timezone and not mentions_tomorrow and not weekday_match: # Tool schemas require local wall-time ISO for user-entered calendar # times. Small routers sometimes append "Z" anyway, which shifts an # "8am" request to another local hour in the browser. Strip accidental # timezone suffixes unless the user explicitly asked for a timezone. normalized = dict(args) changed = False stripped_start, stripped_changed = _strip_iso_timezone(raw_start) if stripped_changed: normalized["dtstart"] = stripped_start changed = True for alias in ("start", "start_time"): if alias in normalized: normalized.pop(alias, None) changed = True raw_end = args.get("dtend") or args.get("end") or args.get("end_time") stripped_end, end_changed = _strip_iso_timezone(raw_end) if end_changed: normalized["dtend"] = stripped_end changed = True for alias in ("end", "end_time"): if alias in normalized: normalized.pop(alias, None) changed = True if "timezone" in normalized: normalized.pop("timezone", None) changed = True normalized["action"] = action return normalized, changed if not mentions_tomorrow and not weekday_match: return args, False if re.search(r"\b20\d{2}-\d{1,2}-\d{1,2}\b", user_text): return args, False try: from src.user_time import now_user_local today = now_user_local().date() except Exception: today = date.today() if mentions_tomorrow: expected_date = today + timedelta(days=1) else: weekday = { "monday": 0, "tuesday": 1, "wednesday": 2, "thursday": 3, "friday": 4, "saturday": 5, "sunday": 6, }[weekday_match.group(1).lower()] days = (weekday - today.weekday()) % 7 expected_date = today + timedelta(days=days or 7) def _parse_iso(value: Any) -> datetime | None: text = str(value or "").strip() if not text: return None if text.endswith("Z"): text = text[:-1] + "+00:00" try: return datetime.fromisoformat(text) except ValueError: return None start_dt = _parse_iso(raw_start) if start_dt is None: return args, False normalized = dict(args) delta = expected_date - start_dt.date() normalized_start = start_dt + delta if not user_mentions_timezone: normalized_start = normalized_start.replace(tzinfo=None) normalized.pop("timezone", None) normalized["action"] = action normalized["dtstart"] = normalized_start.isoformat(timespec="seconds") for alias in ("start", "start_time"): normalized.pop(alias, None) raw_end = args.get("dtend") or args.get("end") or args.get("end_time") end_dt = _parse_iso(raw_end) if end_dt is not None: normalized_end = end_dt + delta if not user_mentions_timezone: normalized_end = normalized_end.replace(tzinfo=None) normalized["dtend"] = normalized_end.isoformat(timespec="seconds") for alias in ("end", "end_time"): normalized.pop(alias, None) for optional_key in ("location", "description", "uid"): if str(normalized.get(optional_key) or "").strip().lower() in {"none", "null", "n/a"}: normalized.pop(optional_key, None) return normalized, normalized != args def _recover_manage_email_tool_block( block: ToolBlock, *, active_document: Any = None, last_user: str = "", ) -> ToolBlock: """Map stale compact-router manage_email aliases onto real tools.""" if block.tool_type in {"mark_email_state", "mcp__email__mark_email_state"}: raw = block.content or "" try: args = json.loads(raw or "{}") except (TypeError, ValueError, json.JSONDecodeError): args = {} if not isinstance(args, dict): args = {} action = str(args.get("action") or "").strip().lower() if action not in {"mark_read", "mark_unread"}: action = "mark_unread" if re.search(r"\bunread\b", last_user or "", re.IGNORECASE) else "mark_read" normalized = { "action": action, "uid": args.get("uid") or args.get("message_uid") or args.get("id"), "folder": args.get("folder") or "INBOX", } if args.get("account"): normalized["account"] = args.get("account") return ToolBlock("mcp__email__manage_email_state", json.dumps(normalized)) if block.tool_type != "manage_email": return block raw = block.content or "" try: args = json.loads(raw or "{}") except (TypeError, ValueError, json.JSONDecodeError): args = {} if not isinstance(args, dict): args = {} action = str(args.get("action") or "").strip().lower() if action in {"list", "list_email", "list_emails", "latest", "latest_email"}: unread = args.get("unread_only", False) if isinstance(unread, str): unread = unread.strip().lower() in {"1", "true", "yes"} max_results = args.get("max_results", 1) with contextlib.suppress(Exception): max_results = int(max_results) return ToolBlock("mcp__email__list_emails", json.dumps({ "folder": str(args.get("folder") or "INBOX"), "max_results": max_results or 1, "unread_only": bool(unread), })) if action in {"reply", "reply_to_email", "draft_reply"} and _is_email_document_obj(active_document): reply_text = str(args.get("body") or args.get("content") or args.get("message") or "").strip() if not reply_text: reply_text = _extract_followup_content_update(last_user) if reply_text: return ToolBlock("update_document", json.dumps({ "content": _build_active_email_draft_reply_content( getattr(active_document, "current_content", "") or "", reply_text, ) })) return block def _collapse_repeated_email_singletons( tool_blocks: list[ToolBlock], ) -> list[ToolBlock]: """Collapse repeated one-message email mutations into one bulk_email call.""" if len(tool_blocks) < 2: return tool_blocks action_by_tool = { "archive_email": "archive", "mcp__email__archive_email": "archive", "delete_email": "delete", "mcp__email__delete_email": "delete", "mark_email_read": "mark_read", "mcp__email__mark_email_read": "mark_read", } if any(block.tool_type not in action_by_tool for block in tool_blocks): return tool_blocks parsed: list[dict[str, Any]] = [] for block in tool_blocks: try: args = json.loads(block.content or "{}") except (TypeError, ValueError, json.JSONDecodeError): return tool_blocks if not isinstance(args, dict) or not args.get("uid"): return tool_blocks parsed.append(args) actions = {action_by_tool[block.tool_type] for block in tool_blocks} if len(actions) != 1: return tool_blocks action = next(iter(actions)) if action == "mark_read": read_values = {bool(args.get("read", True)) for args in parsed} if len(read_values) != 1: return tool_blocks action = "mark_read" if next(iter(read_values)) else "mark_unread" folders = {str(args.get("folder") or "INBOX") for args in parsed} accounts = {str(args.get("account") or "") for args in parsed} if len(folders) != 1 or len(accounts) != 1: return tool_blocks bulk_args: dict[str, Any] = { "action": action, "uids": [str(args["uid"]) for args in parsed], "folder": next(iter(folders)), } account = next(iter(accounts)) if account: bulk_args["account"] = account if action == "delete" and any(bool(args.get("permanent", False)) for args in parsed): bulk_args["permanent"] = True return [ToolBlock("mcp__email__bulk_email", json.dumps(bulk_args))] def _email_list_summary_from_tool_output( raw: str, max_items: int = 10, *, attachments_only: bool = False, unread_requested: bool = False, ) -> str: """Format list_emails output for chat without an LLM pass.""" if not isinstance(raw, str) or not raw.strip(): return "" account_errors = bool(re.search(r"\[EMAIL ACCOUNT ERRORS:", raw, re.IGNORECASE)) if account_errors and not re.search(r"^\s*\d+\.\s+\*\*", raw, re.MULTILINE): return ( "I couldn't check the inbox because one or more email accounts are " "currently unavailable. No reliable empty-inbox result was returned." ) if (not account_errors and re.search(r"\b(no emails?|found 0 email|0 email)\b", raw, re.IGNORECASE)): return "No emails found." parsed: list[dict[str, str]] = [] current: dict[str, str] | None = None for line in raw.splitlines(): m = re.match(r"^\s*\d+\.\s+\*\*(.*?)\*\*\s*$", line) if m: if current: parsed.append(current) current = {"subject": re.sub(r"\s+", " ", m.group(1)).strip()} continue if current is None: continue fm = re.match(r"^\s*From:\s*(.+?)\s*$", line) if fm: current["from"] = re.sub(r"\s+", " ", fm.group(1)).strip() continue dm = re.match(r"^\s*Date:\s*(.+?)\s*$", line) if dm: current["date"] = re.sub(r"\s+", " ", dm.group(1)).strip() continue um = re.match(r"^\s*UID:\s*(.+?)\s*$", line) if um: current["uid"] = re.sub(r"\s+", " ", um.group(1)).strip() continue am = re.match(r"^\s*Account:\s*(.+?)\s*$", line) if am: current["account"] = re.sub(r"\s+", " ", am.group(1)).strip() continue atm = re.match(r"^\s*Attachments?:\s*(.+?)\s*$", line, re.IGNORECASE) if atm: current["attachments"] = re.sub(r"\s+", " ", atm.group(1)).strip() continue sm = re.match(r"^\s*Summary:\s*(.+?)\s*$", line) if sm: current["summary"] = re.sub(r"\s+", " ", sm.group(1)).strip() continue if current: parsed.append(current) if attachments_only: parsed = [item for item in parsed if item.get("attachments")] if not parsed: if attachments_only: return "No emails with attachments found." return "" total_match = re.search(r"Found\s+(\d+)\s+email", raw, re.IGNORECASE) raw_total = int(total_match.group(1)) if total_match else len(parsed) total = len(parsed) if attachments_only else raw_total account_context = bool(re.search(r"\[EMAIL ACCOUNT CONTEXT:", raw)) if unread_requested and account_context and not attachments_only: grouped: dict[str, list[dict[str, str]]] = {} for item in parsed: account = item.get("account") or "Mailbox" grouped.setdefault(account, []).append(item) lines = [f"You have {total} unread email{'s' if total != 1 else ''} across {len(grouped)} account{'s' if len(grouped) != 1 else ''}:"] display_limit = max_items if total > 20 else max(max_items, total) shown = 0 for account, account_items in grouped.items(): if shown >= display_limit: break lines.append("") lines.append(f"**{account} — {len(account_items)} unread**") for item in account_items: if shown >= display_limit: break lines.append(f"- {_format_email_summary_item(item, include_account=False)}") shown += 1 if total > shown: lines.append(f"- ...and {total - shown} more") return "\n".join(lines) if attachments_only: items = [_format_email_attachment_summary_item(item) for item in parsed[:max_items]] heading = ( "Latest email with attachments:" if total == 1 else f"Latest emails with attachments ({total}):" ) else: items = [_format_email_summary_item(item) for item in parsed[:max_items]] heading = "Here is your latest email:" if total == 1 else f"Here are your emails ({total}):" lines = [heading] lines.extend(f"{idx}. {item}" for idx, item in enumerate(items, start=1)) if total > len(items): lines.append(f"- ...and {total - len(items)} more") return "\n".join(lines) def _single_email_uid_from_tool_output(raw: str) -> str: """Return the only UID in a one-result email list/search output.""" text = str(raw or "") if not re.search(r"\bFound\s+1\s+email", text, re.IGNORECASE): return "" matches = re.findall(r"^\s*UID:\s*(.+?)\s*$", text, re.MULTILINE) return matches[0].strip() if len(matches) == 1 else "" _INVISIBLE_RESPONSE_CHARS = "\u2063\u200b\u200c\u200d\ufeff" def _visible_response_text(text: str) -> str: """Return model-visible prose, ignoring invisible provider separators.""" value = _strip_think_blocks(strip_tool_blocks(str(text or ""))) # Some local Qwen chat templates suppress the opening token while # still emitting its closing token. Everything before that orphan closer # is internal analysis; only the text after it belongs in chat. if "" in value.lower(): value = re.split(r"", value, flags=re.IGNORECASE)[-1] for char in _INVISIBLE_RESPONSE_CHARS: value = value.replace(char, "") value = _strip_incomplete_tool_markup_tail(value) return value.strip() def _format_email_summary_item(item: dict[str, str], *, include_account: bool = True) -> str: subject = item.get("subject") or "(no subject)" uid = str(item.get("uid") or "").strip() if uid: label = str(subject).replace("\\", "\\\\").replace("[", "\\[").replace("]", "\\]") subject = f"[{label}](#email-{uid})" parts = [subject] if item.get("from"): parts.append(f"from {item['from']}") if item.get("date"): parts.append(item["date"]) if uid: parts.append(f"UID {uid}") text = " — ".join(parts) if include_account and item.get("account"): text += f"\n Account: {item['account']}" if item.get("attachments"): text += f"\n Attachments: {item['attachments']}" return text def _email_subject_uid_pairs_from_tool_output(raw: str) -> list[tuple[str, str]]: """Extract subject/UID pairs from email list/search/read tool output.""" if not isinstance(raw, str) or not raw.strip(): return [] pairs: list[tuple[str, str]] = [] current_subject = "" current_uid = "" def flush_current() -> None: nonlocal current_subject, current_uid subject = re.sub(r"\s+", " ", current_subject or "").strip() uid = re.sub(r"\s+", " ", current_uid or "").strip() if subject and uid: pairs.append((subject, uid)) current_subject = "" current_uid = "" for line in raw.splitlines(): list_match = re.match(r"^\s*\d+\.\s+\*\*(.*?)\*\*\s*$", line) if list_match: flush_current() current_subject = list_match.group(1).strip() continue subject_match = re.match(r"^\s*\*\*Subject:\*\*\s*(.*?)\s*$", line) if subject_match: flush_current() current_subject = subject_match.group(1).strip() continue uid_match = re.match(r"^\s*(?:\*\*)?UID(?:\*\*)?:\s*(.+?)\s*$", line) if uid_match: current_uid = uid_match.group(1).strip() continue flush_current() return pairs def _linkify_email_titles_from_tool_events(answer: str, tool_events: list[dict[str, Any]]) -> str: """Add #email links to synthesized answers using the latest email tool data.""" text = str(answer or "") if not text.strip() or not tool_events: return text subject_to_uid: dict[str, str] = {} for event in tool_events or []: if _resolved_tool_event_name(event) not in { "list_emails", "mcp__email__list_emails", "search_emails", "mcp__email__search_emails", "read_email", "mcp__email__read_email", }: continue if not tool_result_is_successful(event): continue for subject, uid in _email_subject_uid_pairs_from_tool_output(event.get("output") or ""): if len(subject.strip()) < 3: continue subject_to_uid.setdefault(subject, uid) if not subject_to_uid: return text subjects = sorted(subject_to_uid, key=len, reverse=True) linked_lines: list[str] = [] for line in text.splitlines(): if "#email-" in line: linked_lines.append(line) continue updated = line for subject in subjects: if subject not in updated: continue uid = subject_to_uid[subject] label = subject.replace("\\", "\\\\").replace("[", "\\[").replace("]", "\\]") link = f"[{label}](#email-{uid})" bold_pattern = re.compile(rf"\*\*{re.escape(subject)}\*\*") if bold_pattern.search(updated): updated = bold_pattern.sub(f"**{link}**", updated, count=1) break updated = updated.replace(subject, link, 1) break linked_lines.append(updated) return "\n".join(linked_lines) def _calendar_title_uid_pairs_from_tool_event(event: dict[str, Any]) -> list[tuple[str, str]]: pairs: list[tuple[str, str]] = [] seen: set[tuple[str, str]] = set() def add_pair(title: Any, uid: Any) -> None: clean_title = re.sub(r"\s+", " ", str(title or "")).strip() clean_uid = str(uid or "").strip() if len(clean_title) < 3 or not clean_uid: return key = (clean_title, clean_uid) if key not in seen: pairs.append(key) seen.add(key) for row in event.get("events") or []: if not isinstance(row, dict): continue add_pair(row.get("summary") or row.get("title"), row.get("uid") or row.get("id")) raw = str(event.get("output") or "") for match in re.finditer(r"\[([^\]]+)\]\(#event-([^)]+)\)", raw): add_pair(match.group(1), match.group(2)) return pairs def _single_calendar_uid_from_tool_event(event: dict[str, Any]) -> str: pairs = _calendar_title_uid_pairs_from_tool_event(event) unique_uids = [] for _title, uid in pairs: if uid and uid not in unique_uids: unique_uids.append(uid) return unique_uids[0] if len(unique_uids) == 1 else "" def _linkify_calendar_titles_from_tool_events(answer: str, tool_events: list[dict[str, Any]]) -> str: """Add #event links to synthesized calendar answers using real tool results.""" text = str(answer or "") if not text.strip() or not tool_events: return text title_to_uid: dict[str, str] = {} for event in tool_events or []: if _resolved_tool_event_name(event) != "manage_calendar": continue if not tool_result_is_successful(event): continue for title, uid in _calendar_title_uid_pairs_from_tool_event(event): title_to_uid.setdefault(title, uid) if not title_to_uid: return text titles = sorted(title_to_uid, key=len, reverse=True) linked_lines: list[str] = [] for line in text.splitlines(): if "#event-" in line: linked_lines.append(line) continue updated = line for title in titles: if title not in updated: continue uid = title_to_uid[title] label = title.replace("\\", "\\\\").replace("[", "\\[").replace("]", "\\]") link = f"[{label}](#event-{uid})" bold_pattern = re.compile(rf"\*\*{re.escape(title)}\*\*") if bold_pattern.search(updated): updated = bold_pattern.sub(f"**{link}**", updated, count=1) break updated = updated.replace(title, link, 1) break linked_lines.append(updated) return "\n".join(linked_lines) def _has_successful_calendar_list_evidence(tool_events: list[dict[str, Any]]) -> bool: """True after manage_calendar has successfully listed events for this turn.""" for event in tool_events or []: if not isinstance(event, dict): continue if _resolved_tool_event_name(event) != "manage_calendar": continue if not tool_result_is_successful(event): continue command = str(event.get("command") or "").strip() output = str(event.get("output") or "").strip() action = "" try: parsed = json.loads(command) if isinstance(parsed, dict): action = str(parsed.get("action") or "").strip().lower() except Exception: action = command.splitlines()[0].strip().lower() if command else "" if action in {"list", "list_events"}: return True if output.startswith("Found ") and "event" in output.lower(): return True return False def _has_successful_calendar_tool_evidence(tool_events: list[dict[str, Any]]) -> bool: """True after any successful manage_calendar call in this turn.""" for event in tool_events or []: if not isinstance(event, dict): continue if _resolved_tool_event_name(event) != "manage_calendar": continue if tool_result_is_successful(event): return True return False def _friendly_email_date(value: str) -> str: text = str(value or "").strip() if not text: return "" try: parsed = datetime.fromisoformat(text.replace("Z", "+00:00")) return parsed.strftime("%b %-d, %-I:%M %p") except Exception: try: parsed = datetime.fromisoformat(text[:19]) return parsed.strftime("%b %-d, %-I:%M %p") except Exception: return text def _email_sender_name(value: str) -> str: text = re.sub(r"\s+", " ", str(value or "")).strip() if not text: return "" text = re.sub(r"\s*\([^)]*@[^)]*\)\s*$", "", text).strip() text = re.sub(r"\s*<[^>]*>\s*$", "", text).strip() return text or str(value or "").strip() def _email_account_label(value: str) -> str: text = re.sub(r"\s+", " ", str(value or "")).strip() if not text: return "" return re.sub(r"\s*<[^>]+>\s*$", "", text).strip() or text def _format_email_attachment_summary_item(item: dict[str, str]) -> str: subject = item.get("subject") or "(no subject)" uid = str(item.get("uid") or "").strip() if uid: label = str(subject).replace("\\", "\\\\").replace("[", "\\[").replace("]", "\\]") subject = f"[{label}](#email-{uid})" meta: list[str] = [] sender = _email_sender_name(item.get("from") or "") if sender: meta.append(sender) friendly_date = _friendly_email_date(item.get("date") or "") if friendly_date: meta.append(friendly_date) account = _email_account_label(item.get("account") or "") if account: meta.append(account) files = [ part.strip() for part in str(item.get("attachments") or "").split(",") if part.strip() ] file_text = ", ".join(f"`{name}`" for name in files) if files else "`attachment`" suffix = f" — {' — '.join(meta)}" if meta else "" return f"{subject}{suffix}\n Files: {file_text}" def _email_attachment_list_requested(user_text: str) -> bool: text = str(user_text or "") return bool(re.search(r"\battachments?\b|\battached\b|\bpdfs?\b|\bfiles?\b", text, re.IGNORECASE)) def _email_read_summary_from_tool_output(raw: str) -> str: """Format read_email output for chat without requiring a second LLM round.""" if not isinstance(raw, str) or not raw.strip(): return "" subject = from_ = date = uid = "" body_lines: list[str] = [] in_body = False for line in raw.splitlines(): if line.strip() == "---": in_body = True continue if in_body: body_lines.append(line) continue m = re.match(r"^\*\*Subject:\*\*\s*(.*)$", line) if m: subject = re.sub(r"\s+", " ", m.group(1)).strip() continue m = re.match(r"^\*\*From:\*\*\s*(.*)$", line) if m: from_ = re.sub(r"\s+", " ", m.group(1)).strip() continue m = re.match(r"^\*\*Date:\*\*\s*(.*)$", line) if m: date = re.sub(r"\s+", " ", m.group(1)).strip() continue m = re.match(r"^\*\*UID:\*\*\s*(.*)$", line) if m: uid = re.sub(r"\s+", " ", m.group(1)).strip() continue if not any((subject, from_, date, uid, body_lines)): return "" lines = [f"Email: {subject or '(no subject)'}"] meta = [] if from_: meta.append(f"From: {from_}") if date: meta.append(f"Date: {date}") if uid: meta.append(f"UID: {uid}") lines.extend(meta) body = "\n".join(body_lines).strip() if body: # read_email returns a metadata block followed by the original RFC-ish # message headers. The chat answer should show the message content, not # duplicate From/To/Subject/Message-ID boilerplate. cleaned_lines = [] skipping_headers = True for body_line in body.splitlines(): stripped = body_line.strip() if skipping_headers and ( not stripped or re.match( r"^(?:From|To|Cc|Bcc|Subject|Message-ID|In-Reply-To|References|Date):\s*", stripped, re.IGNORECASE, ) ): continue skipping_headers = False cleaned_lines.append(body_line) body = "\n".join(cleaned_lines).strip() if body: if len(body) > 1200: body = body[:1200].rstrip() + "\n..." lines.append("") lines.append(body) return "\n".join(lines) def _email_attachment_summary_from_tool_output(raw: str) -> str: """Format download_attachment output for chat without a second LLM round.""" if not isinstance(raw, str) or not raw.strip(): return "" if raw.strip().lower().startswith("error:"): return raw.strip() filename = path = size = "" content_lines: list[str] = [] in_content = False for line in raw.splitlines(): if in_content: content_lines.append(line) continue m = re.match(r"^Attachment downloaded to:\s*`?(.+?)`?\s*$", line) if m: path = m.group(1).strip() continue m = re.match(r"^Filename:\s*(.+?)\s*$", line) if m: filename = m.group(1).strip() continue m = re.match(r"^Size:\s*(.+?)\s*$", line) if m: size = m.group(1).strip() continue if line.strip() == "Content:": in_content = True continue lines = [] if filename: lines.append(f"Attachment: {filename}") if size: lines.append(f"Size: {size}") content = "\n".join(content_lines).strip() if content: if len(content) > 1600: content = content[:1600].rstrip() + "\n..." if lines: lines.append("") lines.append(content) elif path: lines.append(f"Downloaded to: {path}") return "\n".join(lines).strip() def _email_read_summaries_from_tool_events(tool_events: list[dict[str, Any]]) -> list[str]: summaries: list[str] = [] for event in tool_events or []: if _resolved_tool_event_name(event) not in {"read_email", "mcp__email__read_email"}: continue if not tool_result_is_successful(event): continue summary = _email_read_summary_from_tool_output(event.get("output") or "") if summary: summaries.append(summary) return summaries def _email_read_evidence_from_tool_output(raw: str, *, max_body_chars: int = 6000) -> str: """Return bounded, plain-text evidence for a final email lookup synthesis.""" if not isinstance(raw, str) or not raw.strip(): return "" text = raw.strip() # Older cached messages can contain a non-multipart HTML body. Keep the # factual text but never feed style tags and Outlook markup into another # model round. text = re.sub(r"", "\n", text, flags=re.IGNORECASE) text = re.sub(r"", "\n", text, flags=re.IGNORECASE) text = re.sub(r"<[^>]+>", "", text) text = html.unescape(text) text = re.sub(r"[ \t]+\n", "\n", text) text = re.sub(r"\n{3,}", "\n\n", text).strip() if len(text) > max_body_chars: text = text[:max_body_chars].rstrip() + "\n[...email truncated]" return text def _email_lookup_request_from_messages(messages: list[dict], last_user: str) -> str: """Recover the substantive request behind terse follow-ups such as 'and?'.""" terse = re.compile( r"^\s*(?:and|so|well|still|then|okay|ok|did you find it(?: yet)?|what did you find)\s*[?.!]*\s*$", re.IGNORECASE, ) current = str(last_user or "").strip() if current and not terse.match(current): return current for message in reversed(messages or []): if not isinstance(message, dict) or message.get("role") != "user": continue content = message.get("content") if not isinstance(content, str): continue candidate = content.strip() if candidate and not terse.match(candidate): return candidate return current def _email_fact_lookup_requested(user_text: str) -> bool: """Distinguish extracting a fact from mail from displaying the email itself.""" text = str(user_text or "").strip() if not text: return False if re.search( r"\b(?:open|show|display|read)\b.{0,24}\b(?:email|message|thread|it|them)\b", text, re.IGNORECASE, ): return False return bool(re.search( r"\b(?:find|locate|which|where|what|when|who|how much|address|amount|date|deadline|" r"reservation|invoice|receipt|property|contract|attachment|said|say|mention|contained?)\b", text, re.IGNORECASE, )) _EMAIL_TERMINAL_ACTION_TOOLS = { "draft_email", "mcp__email__draft_email", "draft_email_reply", "mcp__email__draft_email_reply", "ai_draft_email_reply", "mcp__email__ai_draft_email_reply", "send_email", "mcp__email__send_email", "reply_to_email", "mcp__email__reply_to_email", } def _email_lookup_needs_post_synthesis( user_text: str, tool_events: list[dict[str, Any]], ) -> bool: """Avoid a redundant lookup synthesis after a completed email action.""" if not _email_fact_lookup_requested(user_text): return False return not any( _resolved_tool_event_name(event) in _EMAIL_TERMINAL_ACTION_TOOLS and tool_result_is_successful(event) for event in (tool_events or []) ) def _email_attachment_summaries_from_tool_events(tool_events: list[dict[str, Any]]) -> list[str]: summaries: list[str] = [] for event in tool_events or []: if _resolved_tool_event_name(event) not in {"download_attachment", "mcp__email__download_attachment"}: continue if not tool_result_is_successful(event): continue summary = _email_attachment_summary_from_tool_output(event.get("output") or "") if summary: summaries.append(summary) return summaries def _email_compact_summary_from_read_summaries(summaries: list[str], user_text: str = "") -> str: items: list[dict[str, str]] = [] for summary in summaries: lines = summary.splitlines() subject = from_ = date = uid = "" body_start = 0 for idx, line in enumerate(lines): if line.startswith("Email: "): subject = line.removeprefix("Email: ").strip() elif line.startswith("From: "): from_ = line.removeprefix("From: ").strip() elif line.startswith("Date: "): date = line.removeprefix("Date: ").strip() elif line.startswith("UID: "): uid = line.removeprefix("UID: ").strip() elif not line.strip(): body_start = idx + 1 break body = "\n".join(lines[body_start:]).strip() if body_start else "" body = re.sub(r"\s+", " ", body).strip() body = re.sub(r"(?i)\bplease capture the action, deadline, and owner if present\..*?$", "", body).strip() body = re.sub(r"(?i)\breference item \d+ in the follow-up notes\.", "", body).strip() body = re.sub(r"\s+", " ", body).strip() if len(body) > 180: body = body[:180].rsplit(" ", 1)[0].rstrip() + "..." items.append({ "subject": subject or "(no subject)", "from": from_, "date": date, "uid": uid, "body": body, }) if not items: return "" noun = "emails" if len(items) != 1 else "email" scope = "latest " if re.search(r"\blast\s+week\b", user_text or "", re.IGNORECASE): scope = "last week's " elif re.search(r"\blast\s+month\b", user_text or "", re.IGNORECASE): scope = "last month's " elif re.search(r"\blast\s+year\b", user_text or "", re.IGNORECASE): scope = "last year's " lines = [f"Summary of your {scope}{noun}:"] for item in items: subject = item["subject"] uid = item.get("uid", "").strip() title = f"[{subject}](#email-{uid})" if uid else subject meta = [] if item.get("from"): meta.append(f"from {item['from']}") if item.get("date"): meta.append(item["date"]) prefix = " -- ".join(meta) body = item.get("body") or "No body text was returned." if prefix: lines.append(f"- {title} -- {prefix}: {body}") else: lines.append(f"- {title}: {body}") return "\n".join(lines) def _email_summary_requested(text: str) -> bool: return bool(re.search(r"\b(?:summari[sz]e|summary|tldr|recap|brief|rundown)\b", str(text or ""), re.IGNORECASE)) def _email_count_requested(text: str) -> bool: return bool(re.search(r"\b(?:how\s+many|count|number\s+of|total)\b.{0,60}\b(?:emails?|messages?|mail)\b|\b(?:emails?|messages?|mail)\b.{0,60}\b(?:how\s+many|count|number\s+of|total)\b", str(text or ""), re.IGNORECASE)) def _email_direct_listing_requested(text: str) -> bool: q = str(text or "").strip().lower() if not q: return False if _email_summary_requested(q) or _email_count_requested(q): return False if re.search(r"\b(?:urgent|important|priority|spam|junk|phishing|unsubscribe|attachment\s+content|what\s+does|what\s+did|say|said|says)\b", q): return False return bool( re.search(r"\b(?:show|list|display|view)\b.{0,50}\b(?:my\s+)?(?:inbox|emails?|mail|messages)\b", q) or re.search(r"\b(?:what(?:'s|\s+is|\s+are)?|check)\b.{0,30}\b(?:my\s+)?(?:inbox|emails?|mail|messages)\b", q) or re.search(r"\b(?:latest|newest|recent|last\s+\d+)\s+(?:emails?|messages|mail)\b", q) or re.search(r"\b(?:emails?|messages|mail)\s+(?:from\s+)?(?:today|yesterday|last\s+week|last\s+month|last\s+year)\b", q) ) def _email_urgent_summary_from_read_summaries(summaries: list[str]) -> str: items: list[dict[str, str | int]] = [] for summary in summaries: lines = summary.splitlines() subject = from_ = date = uid = "" body_start = 0 for idx, line in enumerate(lines): if line.startswith("Email: "): subject = line.removeprefix("Email: ").strip() elif line.startswith("From: "): from_ = line.removeprefix("From: ").strip() elif line.startswith("Date: "): date = line.removeprefix("Date: ").strip() elif line.startswith("UID: "): uid = line.removeprefix("UID: ").strip() elif not line.strip(): body_start = idx + 1 break body = "\n".join(lines[body_start:]).strip() if body_start else "" haystack = f"{subject}\n{body}".lower() score = 0 reasons: list[str] = [] if re.search(r"\bdeadline\b|\btomorrow\b|\bby\s+\d{1,2}:?\d{0,2}\b", haystack): score += 40 reasons.append("has a deadline") if re.search(r"\baction needed\b|\bplease review\b|\bsend\b|\bconfirm\b", haystack): score += 30 reasons.append("asks for action") if re.search(r"\bbefore sending\b|\bsanity-check\b|\bwider team\b", haystack): score += 25 reasons.append("blocks an outbound send") if re.search(r"\bchanged\b|\blatest version\b|\bnumbers\b", haystack): score += 15 reasons.append("may affect dependent work") if not reasons: reasons.append("needs follow-up") items.append({ "subject": subject or "(no subject)", "from": from_, "date": date, "uid": uid, "score": score, "reason": "; ".join(dict.fromkeys(reasons)), }) if not items: return "" items.sort(key=lambda item: int(item.get("score") or 0), reverse=True) lines = ["Most urgent emails I found:"] for idx, item in enumerate(items, start=1): subject = str(item.get("subject") or "(no subject)") uid = str(item.get("uid") or "").strip() title = f"[{subject}](#email-{uid})" if uid else subject meta = [] if item.get("from"): meta.append(f"from {item['from']}") if item.get("date"): meta.append(str(item["date"])) if item.get("reason"): meta.append(str(item["reason"])) lines.append(f"{idx}. {title} — " + " — ".join(meta)) return "\n".join(lines) def _email_accounts_summary_from_tool_output(raw: str, max_items: int = 8) -> str: """Format list_email_accounts output without a second model round.""" if not isinstance(raw, str) or not raw.strip(): return "" text = raw[4:].strip() if raw.startswith("AI: ") else raw.strip() rows: list[str] = [] current = "" for line in text.splitlines(): stripped = line.strip() if not stripped: continue m = re.match(r"^-\s+\*\*(.*?)\*\*(.*)$", stripped) if m: if current: rows.append(current) if len(rows) >= max_items: break current = re.sub(r"\s+", " ", (m.group(1) + m.group(2)).strip()) continue if current and stripped.lower().startswith("email:"): email = stripped.split(":", 1)[1].strip() if email and email not in current: current = f"{current} <{email}>" if current and len(rows) < max_items: rows.append(current) if not rows: return text.splitlines()[0] if text else "" total_match = re.search(r"Found\s+(\d+)\s+email account", text, re.IGNORECASE) total = int(total_match.group(1)) if total_match else len(rows) lines = [f"Email accounts ({total}):"] lines.extend(f"- {row}" for row in rows) if total > len(rows): lines.append(f"- ...and {total - len(rows)} more") return "\n".join(lines) def _web_fetch_summary_from_tool_output(raw: str) -> str: """Render a bounded answer from web_fetch output for simple URL fetches.""" if not isinstance(raw, str) or not raw.strip(): return "" lines = [line.rstrip() for line in raw.strip().splitlines()] title = "" source = "" body_lines: list[str] = [] for line in lines: stripped = line.strip() if not stripped: continue if not title and stripped.startswith("#"): title = stripped.lstrip("#").strip() continue if stripped.lower().startswith("source:"): source = stripped.split(":", 1)[1].strip() continue body_lines.append(stripped) body = re.sub(r"\s+", " ", " ".join(body_lines)).strip() if len(body) > 600: body = body[:600].rstrip() + "..." if title and source: return f"{title}\nSource: {source}" + (f"\n\n{body}" if body else "") if title: return title + (f"\n\n{body}" if body else "") return body[:700] if body else "" def _load_mcp_disabled_map() -> Dict[str, set]: """Load per-server disabled tool sets from the database.""" from core.database import McpServer, SessionLocal disabled_map: Dict[str, set] = {} db = SessionLocal() try: for srv in db.query(McpServer).all(): if srv.disabled_tools: try: names = json.loads(srv.disabled_tools) if names: disabled_map[srv.id] = set(names) except (json.JSONDecodeError, TypeError): pass finally: db.close() return disabled_map # System prompt that tells the LLM about available tools. # Always injected — the LLM decides whether to use them. _AGENT_PREAMBLE = """\ You are an AI assistant with tool access. You can run shell commands, execute Python, search the web, \ read/write files, create and edit documents, generate images, manage memories, and more. \ To use a tool, write a fenced code block with the tool name as the language tag. \ The block executes automatically and you see the output.""" _AGENT_RULES = """\ ## Rules - Only use tools when needed. Don't search for things you already know. - For web lookup/search/latest/current requests, use `web_search` or `web_fetch`. Do NOT use `bash`, `python`, `curl`, `requests`, or scraping code for web lookup unless web tools are disabled or already failed. - If `web_search` is listed in this prompt, web search is available. Do NOT tell the user search/web tools are unavailable. - These exact tags execute automatically. For showing code examples, use ```shell, ```sh, ```py, etc. instead. - Multiple tool blocks per response OK. 60s timeout per tool, 10K char output limit. - Code/content >15 lines → ```create_document (NOT in chat). Short snippets OK in chat. - Long-form or structured writing is a document by default when the user asks to write/create/make/generate it and the answer would be more than a short paragraph. Use create_document instead of dumping the full content in chat. - Editing an existing document: ALWAYS use ```edit_document with FIND/REPLACE blocks. Do NOT rewrite the whole document with ```update_document unless genuinely changing more than half of it. - BIAS TOWARD ACTION on edit requests. If the user says "edit out X", "remove the Y paragraph", "change Z" — JUST DO IT with your best interpretation. Don't ask for clarification on minor ambiguity. The user can undo or re-prompt if wrong. - AFTER A TOOL SUCCEEDS, do not second-guess. The success message ("Document edited: v2, 1 edit") means it worked. Reply in ONE short sentence confirming what was done. No re-checking, no replaying the diff in your head, no validation theater. - AFTER A TOOL FAILS (timeout, error, "Unknown action", "not found"), DO NOT GO SILENT. The user expects a follow-up: either retry with a fix (e.g. correct args, longer-running form, run `tail -f /tmp/foo.log` to see progress, split into smaller steps), OR explicitly tell them "this didn't work, want me to try X instead?". A failed tool is not a stopping condition — only a successful one is. - YOU DECLARE WHEN THE JOB IS DONE — not a timer. Keep taking concrete steps while the task still needs them; you have plenty of rounds, so don't rush to quit just because you've made a few calls. There are exactly three ways to end a turn: (1) DONE — before you declare it, sanity-check that every concrete thing the user asked for actually exists or succeeded (file written, edit applied, command exited clean); then stop calling tools and write the final answer (that IS your "done" signal); (2) BLOCKED — you genuinely can't proceed (a capability is missing, permission denied, or data you can't obtain), so say plainly what's blocking you, in a sentence or two, and stop; (3) keep going with the single most useful next step. The only wrong moves are trailing off mid-task without one of these, and repeating a call you already ran. - Calendar: call `manage_calendar` with `action=list_calendars` FIRST before create/update/delete operations. If a create/update request is missing a required date, time, or target event, use `ask_user` once with a short question; do not guess a reservation/event date, and do not write a long ambiguity analysis. For open-ended dates, include an option like "Exact date" and ask the user to type it. - BULK email actions ("delete all those", "mark all as read", "archive these", "delete all spam", "mark these 19 read") → use the `bulk_email` tool ONCE with either the exact `uids` list from the latest `list_emails` result or `all_unread: true`. NEVER just say you deleted/archived/marked messages unless a delete/archive/mark/bulk email tool call succeeded. NEVER loop mark_email_read / archive_email / delete_email one message at a time — that floods the context and can blow the token budget. One bulk_email call handles the whole set. - Suspected spam workflow: first list/search/scan and explain suspicious candidates with UID, sender, subject, and reason. Before deleting, moving to Junk, unsubscribing, or blocking a sender, ask for confirmation with `ask_user` unless the user explicitly commanded the exact action. After approval, use `bulk_email` with action="junk" for messages and `block_sender` for sender rules. Do not block senders silently. - Email UIDs are the values after `UID:` in tool output, not list row numbers. For example, row `1.` with `UID: 90186` must use `"90186"`, never `"1"`. - "Last/latest/newest email" means call `list_emails` with `max_results: 1`, `unread_only: false`, and the right `account`, then read the UID returned by that tool if full content is needed. NEVER use a table row number like "#18" as an email UID. - Plain "list/show/check my inbox/emails" means latest inbox mail, including read messages. Do not set `unread_only: true` unless the user explicitly asks for unread/needs attention. - If the user asks for multiple specific emails and you call `read_email` more than once, your final answer MUST include every successfully read email, clearly separated and linked by UID. Do not answer with only the last email you read. - Multiple email accounts: if tool output says "Other accounts" or the user asks "my Gmail?", "other inbox?", "work mail?", "custom domain mail?", or names any mailbox/account, DO NOT answer from memory. Call `list_email_accounts` if needed, then call `list_emails`/`read_email`/`bulk_email` with the exact `account` value for that mailbox. Account names are user-defined labels; if the user typo-matches a known account, use the closest listed account instead of claiming it does not exist. NEVER use `app_api` or `/api/email/accounts` to discover email accounts; that route is owner-filtered in tool context and can falsely return empty. - User identity facts/preferences ("my name is ", "I live in ", "I prefer concise replies", "call me ") → use `manage_memory` with action=add. NEVER use `manage_contact` for facts about the user unless the user explicitly says to create/update a contact and provides contact details such as an email or phone. - "Create/add/write a note" / "notes" / "todos" / "remind me to X at