"""Opt-in v3 tool loop. No intent routing or argument substitutions.""" import asyncio import copy import calendar as month_calendar import base64 import importlib.util import io import json import logging import os import re import sys import time from contextlib import asynccontextmanager from dataclasses import dataclass, replace from datetime import datetime, timezone from functools import lru_cache from pathlib import Path import httpx import jsonschema from src.context_compactor import prune_multimodal_images, trim_for_context from src.agent_evidence import command_has_mutation_effect, workspace_artifact_is_usable from src.tool_capabilities import ToolEffect, ToolRunSecurityContext, capabilities_for_action from src.tool_execution import execute_tool_block from src.tool_schemas import ( function_call_to_tool_block, normalize_native_function_args, normalized_native_function_argument_error, ) from src.tool_types import ToolBlock from src.tool_parsing import parse_tool_blocks, strip_tool_blocks from src.turn_contract import ( FAMILY_TOOLS, broad_web_briefing_request, required_read_operation_for_request, targets_bound_editor_request, inline_text_transformation, ) from src.prompt_security import untrusted_context_message from src.model_profiles import ( is_odysseus_merged_tools_model, uses_odysseus_progressive_thinking, ) ENDPOINT_ID = 'cleanv3' MODE = 'clean_compact_v3_preview' # Native unattended workspaces routinely require several inspections followed # by several artifact writes. The interactive preview keeps its six-call # limit below; this larger budget applies only after server-side validation of # a confined native workspace. Duplicate-call suppression still bounds loops. NATIVE_TOOL_CALL_LIMIT = 32 NATIVE_ROUND_LIMIT = 64 # Interactive turns still have duplicate-call and round guards, but legitimate # multi-step work should not be cut off after only a handful of executions. # Keep browser workflows proportionally larger because navigation, inspection, # and interaction are separate observable actions. INTERACTIVE_TOOL_CALL_LIMIT = 18 INTERACTIVE_BROWSER_TOOL_CALL_LIMIT = 30 INTERACTIVE_ROUND_LIMIT = 8 NATIVE_ARTIFACT_RESEARCH_LIMIT = 12 SAME_TARGET_WRITE_LIMIT = 3 ARTIFACT_RESEARCH_TOOLS = frozenset({ 'web_search', 'web_fetch', 'private_browser', 'pdf_extract', 'youtube_tool', 'inspect_media', 'extract_text', 'transcribe_media', }) DETAILED_VIDEO_REQUEST = re.compile( r"\b(?:how\s+many|count|sequence|in\s+order|chronological|" r"timestamps?|what\s+time|at\s+what\s+time|when\s+.*(?:end|happen)|" r"first\s+.*(?:save|attempt|event)|score(?:board)?s?)\b|" r"(?:多少|几次|何时|什么时候|时间|顺序)", re.IGNORECASE, ) READ_TOOLS = frozenset({ 'manage_notes', 'manage_calendar', 'manage_memory', 'manage_skills', 'manage_tasks', 'manage_documents', 'manage_research', 'manage_contact', 'list_sessions', 'search_chats', 'list_email_accounts', 'list_emails', 'search_emails', 'read_email', 'download_attachment', 'scan_spam', 'scan_email_unsubscribes', 'manage_email_state', 'web_search', 'web_fetch', 'youtube_tool', 'search_hf_models', 'pdf_extract', 'private_browser', 'list_cookbook_servers', 'list_models', 'list_served_models', 'list_cached_models', 'list_serve_presets', 'list_downloads', 'tail_serve_output', 'extract_text', 'manage_endpoints', 'manage_mcp', 'manage_tokens', 'manage_webhooks', 'manage_settings', 'app_api', }) SAFE_WRITE_TOOLS = frozenset({ 'manage_notes', 'manage_calendar', 'manage_memory', 'manage_skills', 'manage_tasks', 'create_document', 'manage_documents', 'edit_document', 'update_document', 'suggest_document', 'draft_email', 'draft_email_reply', 'edit_image', }) EXPLICIT_EXECUTE_TOOLS = frozenset({'bash', 'python'}) SAFE_UI_TOOLS = frozenset({'ui_control'}) BROKERED_JOB_TOOLS = frozenset({'trigger_research'}) CONTRACT_REQUIRED_TOOLS = frozenset({ 'send_email', 'reply_to_email', 'create_session', 'send_to_session', 'manage_session', 'chat_with_model', 'pipeline', 'serve_preset', 'stop_served_model', 'download_model', 'ask_teacher', }) from src.turn_contract import CONTRACT_CORE_TOOLS PREVIEW_TOOLS = ( READ_TOOLS | SAFE_WRITE_TOOLS | EXPLICIT_EXECUTE_TOOLS | SAFE_UI_TOOLS | BROKERED_JOB_TOOLS | CONTRACT_REQUIRED_TOOLS | CONTRACT_CORE_TOOLS ) # Keep a small recovery-capable surface on every interactive compact agent # turn. Routing still adds domain tools, while policy and action guards remain # authoritative for execution. Use the same definition as contract resolution. INTERACTIVE_CORE_TOOLS = CONTRACT_CORE_TOOLS # The interactive compact-v5 surface above stays unchanged. These tools are # added only for a server-validated ``odysseus-native`` request with an active, # confined workspace. This lets the model-specific clean runtime serve native # media/artifact tasks without granting the WebUI arbitrary filesystem access. NATIVE_WORKSPACE_READ_TOOLS = frozenset({ 'inspect_media', 'extract_text', 'transcribe_media', 'read_file', 'ls', 'get_workspace', 'pdf_extract', 'glob', 'grep', }) NATIVE_WORKSPACE_WRITE_TOOLS = frozenset({'write_file', 'edit_file'}) NATIVE_WORKSPACE_EXECUTE_TOOLS = frozenset({'python'}) NATIVE_WORKSPACE_TOOLS = ( NATIVE_WORKSPACE_READ_TOOLS | NATIVE_WORKSPACE_WRITE_TOOLS | NATIVE_WORKSPACE_EXECUTE_TOOLS ) ALLOWED_EFFECTS = frozenset({ ToolEffect.READ_PUBLIC, ToolEffect.READ_PRIVATE, ToolEffect.READ_WORKSPACE, ToolEffect.BROKERED_NETWORK_READ, ToolEffect.WRITE_PRIVATE, }) SAFE_ACTIONS = { 'manage_notes': frozenset({'list', 'search', 'find', 'view', 'add', 'update', 'delete', 'toggle_item'}), 'manage_calendar': frozenset({'list_calendars', 'list_events', 'create_event', 'update_event', 'delete_event'}), 'manage_memory': frozenset({'list', 'search', 'add', 'edit', 'delete'}), 'manage_skills': frozenset({'list', 'index', 'view', 'view_ref', 'search', 'add', 'edit', 'patch', 'delete'}), 'manage_tasks': frozenset({'list', 'create', 'edit', 'delete', 'pause', 'resume'}), 'manage_documents': frozenset({'list', 'read', 'view', 'open', 'get', 'delete'}), 'manage_research': frozenset({'list', 'read', 'open', 'view', 'get'}), 'manage_contact': frozenset({'list', 'search', 'find'}), 'private_browser': frozenset({ 'open', 'read', 'snapshot', 'find', 'evaluate', 'click', 'fill', 'press', 'scroll', 'wait', 'screenshot', 'close', 'batch', }), # These UI effects are reversible. A model switch is additionally bound # below to explicit user wording; keep toggle mutation, mode changes, and # email-draft actions outside this subset. 'ui_control': frozenset({ 'open_panel', 'set_theme', 'create_theme', 'get_theme', 'get_toggles', 'switch_model', }), 'manage_endpoints': frozenset({'list'}), 'manage_mcp': frozenset({'list', 'list_tools'}), 'manage_tokens': frozenset({'list'}), 'manage_webhooks': frozenset({'list'}), 'manage_settings': frozenset({'list', 'get', 'list_tools'}), 'manage_email_state': frozenset({'list_blocked'}), 'manage_session': frozenset({ 'rename', 'archive', 'unarchive', 'delete', 'important', 'unimportant', 'truncate', 'fork', }), } def search_tool_choice_request(request): """Enforce a search via one offered tool, not named-tool argument decoding. The served model emits missing query fields under named search choice. Required choice over the same single schema preserves the policy intent. Other tools and auto/none requests retain their existing dispatch. """ choice = request.get('tool_choice') if not isinstance(choice, dict) or choice.get('type') != 'function': return request name = (choice.get('function') or {}).get('name') if name != 'web_search': return request selected = [s for s in request.get('tools', []) if s.get('function', {}).get('name') == name] if len(selected) != 1: return request return {**request, 'tools': selected, 'tool_choice': 'required'} def provider_compatible_tool_choice_request(request, model): """Keep tools but avoid forced choice unsupported by thinking providers.""" model_name = canonical(str(model or '')).casefold() if model_name.startswith('deepseek') and 'tool_choice' in request: compatible = dict(request) choice = compatible.get('tool_choice') selected_name = ( (choice.get('function') or {}).get('name') if isinstance(choice, dict) else None ) if selected_name: selected = [ schema for schema in compatible.get('tools') or [] if (schema.get('function') or {}).get('name') == selected_name ] if selected: compatible['tools'] = selected compatible.pop('tool_choice', None) return compatible if ( request.get('tool_choice') == 'required' and len(request.get('tools') or []) == 1 and ('qwen' in model_name or model_name.startswith('odysseus-')) ): # Raw-policy capture accepts a named tool constraint (and records that # turn as excluded from policy loss), but deliberately rejects the # distribution-wide ``required`` mode. Search recovery narrows the # schema to one tool before reaching this boundary, so preserving that # exact name has the same runtime intent without a transport failure. compatible = dict(request) name = request['tools'][0]['function']['name'] compatible['tool_choice'] = { 'type': 'function', 'function': {'name': name}, } return compatible return request def bounded_search_observation(output, budget=8000): from src.search_passages import bounded_search_observation as compact return compact(output, budget) def preview_tool_result_text(result, tool, args): """Preserve failure evidence before applying the observation budget.""" output = result.get('output') or result.get('error') or result if result.get('error') or result.get('exit_code') not in (None, 0): # A nonempty stdout is not proof of success. This text is also the # model's saved tool message; SSE-only status cannot inform follow-ups. # Put the status first so a long output cannot truncate it away. output = {'exit_code': result.get('exit_code', 1), 'error': result.get('error'), **result} elif (canonical(tool) == 'manage_skills' and args.get('action') in {'list', 'index'} and not result.get('error') and not result.get('output') and isinstance(result.get('results'), str)): output = result['results'] elif ( canonical(tool) == 'manage_memory' and str(args.get('action') or '').replace('-', '_').casefold() in {'list', 'index'} and not result.get('error') and isinstance(result.get('results'), str) ): # Keep the row-oriented payload parseable. JSON-encoding hundreds of # entries before the observation cap can cut inside a quoted string, # leaving neither the model nor canonical renderer usable evidence. output = result['results'] output = output if isinstance(output, str) else json.dumps(output, ensure_ascii=False) if canonical(tool) == 'web_search' and not result.get('error') and result.get('exit_code') in (None, 0): output = bounded_search_observation(output) if len(output) > 8000: output = output[:8000] + '\n[Tool result truncated at 8000 characters.]' return output def canonical(name): return name.removeprefix('mcp__email__') def semantic_repeat_scope(name, args): """Identify narrow repeated actions whose changing text hides one intent.""" tool = canonical(str(name or '')) if not isinstance(args, dict): return None if tool == 'write_file': raw_path = str(args.get('path') or '').strip() if raw_path: return ('write_target', os.path.normpath(raw_path)) if tool == 'inspect_media': raw_path = str(args.get('path') or '').strip() suffix = os.path.splitext(raw_path.casefold())[1] if raw_path and suffix in {'.jpg', '.jpeg', '.png', '.webp', '.gif', '.bmp'}: query = re.sub(r'\s+', ' ', str(args.get('query') or '')).strip().casefold() try: max_dimension = int(args.get('max_dimension') or 0) except (TypeError, ValueError): max_dimension = 0 return ( 'still_image_inspection', os.path.normpath(raw_path), query, max_dimension, ) if tool == 'bash': command = str(args.get('command') or '') lowered = command.casefold() image_glob = re.search(r'\*\.(?:jpe?g|png|webp|gif|bmp)', lowered) filename_probe = ( 'find ' in lowered and 'grep ' in lowered and ('echo "$1"' in lowered or "echo '$1'" in lowered) ) if image_glob and filename_probe: return ('media_filename_inference', 'bash') return None def shell_native_tool_misuse(output, offered_schemas): """Return an offered native tool incorrectly invoked as a shell binary.""" text = str(output or '') offered = { canonical(schema.get('function', {}).get('name', '')) for schema in (offered_schemas or []) } for match in re.finditer( r'(?:^|\n)(?:bash: line \d+: )?([A-Za-z_][\w.-]*): command not found\b', text, ): name = canonical(match.group(1)) if name in offered: return name return '' def shell_native_tool_command_misuse(command, offered_schemas): """Return an offered native tool treated as a package or Python module.""" text = str(command or '') offered = { canonical(schema.get('function', {}).get('name', '')) for schema in (offered_schemas or []) } candidates = set() for match in re.finditer( r'\b(?:python\d*(?:\.\d+)?\s+-m\s+)?pip\d*(?:\.\d+)?\s+install\b([^;&|\n]*)', text, re.I, ): for token in re.findall(r'(? 1 and str(command[0]).lower() in {'open', 'read'}: child['url'] = command[1] else: continue child_changed, next_url = private_browser_state_transition(child, next_url) changed = changed or child_changed return changed, next_url return action in {'click', 'fill', 'press', 'scroll', 'wait', 'evaluate', 'close'}, current_url def private_browser_success_repeat_limit(args): """Permit a few fresh DOM observations while keeping retries bounded.""" if not isinstance(args, dict): return 1 action = str(args.get('action') or '').strip().lower() if action == 'snapshot': return 3 if action == 'batch': commands = args.get('commands') or args.get('steps') or () actions = { str(command.get('action') or command.get('command') or '').strip().lower() if isinstance(command, dict) else str(command[0]).strip().lower() for command in commands if isinstance(command, dict) or (isinstance(command, (list, tuple)) and command) } if actions and actions <= {'snapshot'}: return 3 return 1 def email_account_backend_unavailable(result): """Treat a merged all-account transport outage as failure, not zero rows.""" if not isinstance(result, dict): return False text = "\n".join(str(result.get(key) or "") for key in ("output", "stdout", "error")) return bool( re.search(r"\[EMAIL ACCOUNT ERRORS:", text, re.IGNORECASE) and not re.search(r"^\s*\d+\.\s+\*\*", text, re.MULTILINE) ) def requested_item_limit(user_text, *, default, maximum=50): """Resolve an explicit user-facing result cap for canonical renderers.""" text = str(user_text or '') number_words = { 'one': 1, 'two': 2, 'three': 3, 'four': 4, 'five': 5, 'six': 6, 'seven': 7, 'eight': 8, 'nine': 9, 'ten': 10, } count = r'(\d+|one|two|three|four|five|six|seven|eight|nine|ten)' match = re.search( r'\b(?:at\s+most|up\s+to|no\s+more\s+than|' r'cap(?:\s+(?:it|them|the\s+(?:answer|list)))?\s+at|' r'max(?:imum)?(?:\s+of)?|(?:i\s+)?only(?:\s+(?:need|want|show))?' r'(?:\s+(?:the\s+)?first)?|need\s+only|just(?:\s+(?:the\s+)?first)?|' r'limit(?:ed)?\s+to|trim(?:\s+it|\s+them|\s+the\s+list)?\s+to|return|show|list)\s+' + count + r'\b', text, re.IGNORECASE, ) if not match: match = re.search( r'\b' + count + r'\s+(?:short\s+)?(?:titles?|items?|results?|entries?|names?|things?)?' r'(?:\s*(?:and|\+|with)\s+(?:their\s+)?(?:status(?:es)?|states?|times?))?\s*' r'(?:at\s+most|max(?:imum)?|only|tops?)\b' r'(?:\s*,\s*(?:no\s+edits?|read[- ]only))?', text, re.IGNORECASE, ) if not match: match = re.search( r'\b(?:titles?|items?|results?|entries?|names?|ones?|things?)\s*[,;:-]?\s*' + count + r'\s*(?:at\s+most|max(?:imum)?|only|tops?)\b', text, re.IGNORECASE, ) if not match: match = re.search( r'\b' + count + r'\s+short\s+(?:ones?|items?|entries?|bits?)\b', text, re.IGNORECASE, ) if not match: match = re.search( r'\b' + count + r'\s+is\s+(?:fine|enough|plenty)\b', text, re.IGNORECASE, ) if not match: match = re.search( r'\b(?:first|same)\s+' + count + r'(?:\s+(?:short\s+)?(?:titles?|items?|results?|entries?|names?|ones?|bits?))?\b', text, re.IGNORECASE, ) if not match: match = re.search( r'\b(?:like\s+)?' + count + r'\s+(?:titles?|items?|results?|entries?|names?|ones?|bits?|things?)' r'(?:\s*(?:and|\+)\s+(?:their\s+)?(?:status(?:es)?|states?))?' r'(?:\s*(?:\+|and)\s+(?:whether|if)\b[^.!?]*)?[.!?]*\s*$', text, re.IGNORECASE, ) if not match and re.search(r'\b(?:(?:only|just)\s+a|first|top)\s+(?:few|couple)\b', text, re.I): return min(maximum, 3) if not match and re.search(r'\b(?:just\s+)?(?:list|show)(?:\s+me)?\s+a\s+few\b', text, re.I): return min(maximum, 3) if not match: match = re.search( r'\b(?:keep\s+it\s+to|(?:maybe\s+)?(?:first|top)|(?:the\s+)?next|same)\s+' + count + r'\b', text, re.I, ) if not match: match = re.search(r'[,;]\s*' + count + r'[.!?]*\s*$', text, re.I) if not match: return default token = match.group(1).casefold() value = int(token) if token.isdigit() else number_words[token] return max(0, min(maximum, value)) def contract_item_limit(turn_contract, default): operation = getattr(turn_contract, 'required_read_operation', None) value = getattr(operation, 'max_items', None) if operation is not None else None return value if isinstance(value, int) and value >= 0 else default def _bounded_structured_list(summary, *, user_text): """Keep capped list history identical to the rows visible to the user.""" if requested_item_limit(user_text, default=None) is None: return summary return str(summary or '').split('