feat(runtime): record execution evidence and gate completion

This commit is contained in:
Alexandre Teixeira
2026-09-26 13:48:12 +01:00
parent ea847e6c0b
commit eaa5668baa
16 changed files with 1105 additions and 154 deletions
+59 -89
View File
@@ -727,49 +727,13 @@ async def _route_tool_via_bridge(tool: str, content: str, session_id: Optional[s
# "tool_path_extra_roots" setting (list of path strings).
# ---------------------------------------------------------------------------
_SENSITIVE_BASENAMES: set[str] = {
".ssh", ".gnupg", ".gitconfig",
".bashrc", ".bash_profile", ".bash_logout",
".zshrc", ".zprofile", ".zshenv",
".profile", ".tcshrc", ".cshrc",
".env", ".netrc",
}
_SENSITIVE_FILE_PATTERNS: tuple[str, ...] = (
"authorized_keys", "id_rsa", "id_ed25519", "id_ecdsa",
"known_hosts", "auth.json", "app.db", "settings.json",
# Compatibility exports: the same deny predicate protects tool access and
# artifact observations, so hashing cannot become a sensitive-file side channel.
from src.agent_runtime.path_policy import (
_SENSITIVE_BASENAMES, _SENSITIVE_FILE_PATTERNS,
_SENSITIVE_BASENAMES_CF, _SENSITIVE_FILE_PATTERNS_CF, _is_sensitive_path,
)
# Case-folded views used for matching. On a case-insensitive filesystem
# (Windows, default macOS) ".SSH/AUTHORIZED_KEYS" and ".env" resolve to the
# same protected files as their lowercase forms, so the deny-list has to fold
# case before comparing — the sibling resolver already normcases paths for the
# same reason. casefold (not os.path.normcase) because normcase is a no-op on
# POSIX, which is exactly where the macOS read-exfil path lives.
_SENSITIVE_BASENAMES_CF: frozenset[str] = frozenset(b.casefold() for b in _SENSITIVE_BASENAMES)
_SENSITIVE_FILE_PATTERNS_CF: frozenset[str] = frozenset(p.casefold() for p in _SENSITIVE_FILE_PATTERNS)
def _is_sensitive_path(resolved: str) -> bool:
"""Return True if *resolved* falls under a sensitive directory or
matches a sensitive filename — regardless of what root it sits under.
Matching is case-insensitive: on Windows / default macOS a case-variant
name (``.SSH``, ``AUTHORIZED_KEYS``, ``Id_Rsa``) points at the same file as
the lowercase form, so a case-sensitive check would let it slip past the
deny-list in every file tool that relies on it.
"""
parts = [p.casefold() for p in resolved.split(os.sep)]
filename = parts[-1] if parts else ""
# Check if any path component is a sensitive directory.
for part in parts:
if part in _SENSITIVE_BASENAMES_CF:
return True
# Check filename against known sensitive files.
return filename in _SENSITIVE_FILE_PATTERNS_CF
def _tool_path_roots() -> list[str]:
"""Return the list of directory roots that read_file / write_file
@@ -1290,6 +1254,10 @@ async def _document_tool_dispatch(
# Dispatcher
# ---------------------------------------------------------------------------
from src.agent_runtime.journal import dispatched, mark_authorized, mark_dispatch, record_action
@record_action
async def execute_tool_block(
block: Any,
session_id: Optional[str] = None,
@@ -1615,14 +1583,15 @@ async def _execute_tool_block_impl(
if rejected is not None:
return rejected
mark_authorized()
if bridge_owns_tool:
try:
return await execution_bridge.route_tool(
return await dispatched(execution_bridge.route_tool(
tool,
content,
session_id,
client_runtime_context,
)
))
except asyncio.CancelledError:
raise
except Exception as exc:
@@ -1643,7 +1612,7 @@ async def _execute_tool_block_impl(
)
if tool in _ROUTED_BRIDGE_TOOLS and _client_bridge(client_runtime_context) is not None:
return await _route_tool_via_bridge(tool, content, session_id, client_runtime_context)
return await dispatched(_route_tool_via_bridge(tool, content, session_id, client_runtime_context))
# Background execution: a `bash` block whose first line is the `#!bg`
# marker runs DETACHED — returns a job id immediately so the chat stream
@@ -1653,6 +1622,7 @@ async def _execute_tool_block_impl(
_is_bg, _bg_cmd = _split_bg_marker(content)
if _is_bg and _bg_cmd:
from src import bg_jobs
mark_dispatch()
rec = bg_jobs.launch(_bg_cmd, session_id=session_id, cwd=agent_cwd())
short = _bg_cmd.strip().split(chr(10))[0][:80]
desc = f"bash (background): {short}"
@@ -1678,37 +1648,37 @@ async def _execute_tool_block_impl(
if tool in _MCP_TOOL_MAP:
first_line = content.split(chr(10))[0][:80]
desc = f"{tool}: {first_line}"
result = await _call_mcp_tool(tool, content, progress_cb=progress_cb)
result = await dispatched(_call_mcp_tool(tool, content, progress_cb=progress_cb))
elif tool in ("grep", "glob", "ls", "get_workspace", "host_shell"):
# Code-navigation tools — no MCP server; run the direct implementation.
first_line = content.split(chr(10))[0][:80]
desc = f"{tool}: {first_line}"
result = await _direct_fallback(
result = await dispatched(_direct_fallback(
tool,
content,
progress_cb=progress_cb,
owner=owner,
client_runtime_context=client_runtime_context,
) \
)) \
or {"error": f"{tool}: execution failed", "exit_code": 1}
elif tool == "apply_patch" and _tui_host_bridge_patch_url(client_runtime_context):
first_line = content.split(chr(10))[0][:80]
desc = f"{tool}: {first_line}" if first_line else tool
result = await _apply_patch_via_tui_host_bridge(content, client_runtime_context)
result = await dispatched(_apply_patch_via_tui_host_bridge(content, client_runtime_context))
elif tool in ("apply_patch", "todowrite"):
first_line = content.split(chr(10))[0][:80]
desc = f"{tool}: {first_line}" if first_line else tool
result = await _direct_fallback(tool, content, session_id=session_id, owner=owner) \
result = await dispatched(_direct_fallback(tool, content, session_id=session_id, owner=owner)) \
or {"error": f"{tool}: execution failed", "exit_code": 1}
elif tool == "manage_bg_jobs":
# Inspect/kill detached `bash` jobs; needs session_id to scope to chat.
desc = f"manage_bg_jobs: {content.split(chr(10))[0][:80]}"
result = await _direct_fallback(tool, content, session_id=session_id, owner=owner) \
result = await dispatched(_direct_fallback(tool, content, session_id=session_id, owner=owner)) \
or {"error": "manage_bg_jobs: execution failed", "exit_code": 1}
elif tool in ("create_document", "update_document", "edit_document",
"suggest_document", "manage_documents"):
desc = f"{tool}: {content.split(chr(10))[0][:80]}"
result = await _document_tool_dispatch(
result = await dispatched(_document_tool_dispatch(
tool,
content,
session_id,
@@ -1716,14 +1686,14 @@ async def _execute_tool_block_impl(
document_id=approved_document_id or active_document_id,
document_version=approved_document_version,
document_digest=approved_document_digest,
) \
)) \
or {"error": f"{tool}: execution failed", "exit_code": 1}
if tool in ("edit_document", "suggest_document") and "title" in (result or {}):
desc = f"{tool}: {result.get('title', '')}"
elif tool == "search_chats":
query = content.split("\n")[0].strip()
desc = f"search_chats: {query[:80]}"
result = await do_search_chats(query, owner=owner)
result = await dispatched(do_search_chats(query, owner=owner))
elif tool in ("chat_with_model", "ask_teacher", "list_models"):
# Migrated to the agent_tools registry (#3629): dispatched through
# TOOL_HANDLERS with the owner/session ctx these tools need, instead
@@ -1731,7 +1701,7 @@ async def _execute_tool_block_impl(
# src/agent_tools/model_interaction_tools.py.
first_line = content.split(chr(10))[0].strip()[:60]
desc = f"{tool}: {first_line}" if first_line else tool
result = await _document_tool_dispatch(tool, content, session_id, owner) \
result = await dispatched(_document_tool_dispatch(tool, content, session_id, owner)) \
or {"error": f"{tool}: execution failed", "exit_code": 1}
elif tool in ("create_session", "list_sessions", "send_to_session", "manage_session"):
# Migrated to the agent_tools registry (#3629): dispatched through
@@ -1739,101 +1709,101 @@ async def _execute_tool_block_impl(
# live in src/agent_tools/session_tools.py.
first_line = content.split(chr(10))[0].strip()[:60]
desc = f"{tool}: {first_line}" if first_line else tool
result = await _document_tool_dispatch(tool, content, session_id, owner) \
result = await dispatched(_document_tool_dispatch(tool, content, session_id, owner)) \
or {"error": f"{tool}: execution failed", "exit_code": 1}
elif tool in ("pipeline", "manage_memory", "ui_control"):
from src.ai_interaction import dispatch_ai_tool
desc, result = await dispatch_ai_tool(tool, content, session_id, owner=owner)
desc, result = await dispatched(dispatch_ai_tool(tool, content, session_id, owner=owner))
elif tool == "manage_tasks":
desc = "manage_tasks"
result = await do_manage_tasks(content, owner=owner)
result = await dispatched(do_manage_tasks(content, owner=owner))
elif tool == "manage_skills":
desc = "manage_skills"
result = await do_manage_skills(content, owner=owner)
result = await dispatched(do_manage_skills(content, owner=owner))
elif tool == "api_call":
first_line = content.split("\n")[0].strip()[:60]
desc = f"api_call: {first_line}"
result = await do_api_call(content)
result = await dispatched(do_api_call(content))
elif tool in ("manage_endpoints", "manage_mcp", "manage_webhooks", "manage_tokens", "manage_settings"):
# Registry-dispatched (agent_tools.admin_tools); owner threaded for ownership/admin checks.
desc = tool
result = await _direct_fallback(tool, content, owner=owner) \
result = await dispatched(_direct_fallback(tool, content, owner=owner)) \
or {"error": f"{tool}: execution failed", "exit_code": 1}
elif tool == "manage_notes":
desc = "manage_notes"
result = await do_manage_notes(content, owner=owner)
result = await dispatched(do_manage_notes(content, owner=owner))
elif tool == "manage_calendar":
desc = "manage_calendar"
result = await do_manage_calendar(content, owner=owner)
result = await dispatched(do_manage_calendar(content, owner=owner))
elif tool == "download_model":
desc = "download_model"
result = await do_download_model(content, owner=owner)
result = await dispatched(do_download_model(content, owner=owner))
elif tool == "serve_model":
desc = "serve_model"
result = await do_serve_model(content, owner=owner)
result = await dispatched(do_serve_model(content, owner=owner))
elif tool == "list_served_models":
desc = "list_served_models"
result = await do_list_served_models(content, owner=owner)
result = await dispatched(do_list_served_models(content, owner=owner))
elif tool == "stop_served_model":
desc = "stop_served_model"
result = await do_stop_served_model(content, owner=owner)
result = await dispatched(do_stop_served_model(content, owner=owner))
elif tool == "tail_serve_output":
desc = "tail_serve_output"
result = await do_tail_serve_output(content, owner=owner)
result = await dispatched(do_tail_serve_output(content, owner=owner))
elif tool == "list_downloads":
desc = "list_downloads"
result = await do_list_downloads(content, owner=owner)
result = await dispatched(do_list_downloads(content, owner=owner))
elif tool == "cancel_download":
desc = "cancel_download"
result = await do_cancel_download(content, owner=owner)
result = await dispatched(do_cancel_download(content, owner=owner))
elif tool == "search_hf_models":
desc = "search_hf_models"
result = await do_search_hf_models(content, owner=owner)
result = await dispatched(do_search_hf_models(content, owner=owner))
elif tool == "list_cached_models":
desc = "list_cached_models"
result = await do_list_cached_models(content, owner=owner)
result = await dispatched(do_list_cached_models(content, owner=owner))
elif tool == "app_api":
desc = "app_api"
result = await do_app_api(content, owner=owner)
result = await dispatched(do_app_api(content, owner=owner))
elif tool == "list_serve_presets":
desc = "list_serve_presets"
result = await do_list_serve_presets(content, owner=owner)
result = await dispatched(do_list_serve_presets(content, owner=owner))
elif tool == "serve_preset":
desc = "serve_preset"
result = await do_serve_preset(content, owner=owner)
result = await dispatched(do_serve_preset(content, owner=owner))
elif tool == "adopt_served_model":
desc = "adopt_served_model"
result = await do_adopt_served_model(content, owner=owner)
result = await dispatched(do_adopt_served_model(content, owner=owner))
elif tool == "list_cookbook_servers":
desc = "list_cookbook_servers"
result = await do_list_cookbook_servers(content, owner=owner)
result = await dispatched(do_list_cookbook_servers(content, owner=owner))
elif tool == "edit_image":
desc = "edit_image"
result = await do_edit_image(content, owner=owner)
result = await dispatched(do_edit_image(content, owner=owner))
elif tool == "edit_file":
result = await _direct_fallback(tool, content) or {"error": "edit failed", "exit_code": 1}
result = await dispatched(_direct_fallback(tool, content)) or {"error": "edit failed", "exit_code": 1}
desc = result.get("output") or result.get("error") or "edit_file"
elif tool == "trigger_research":
desc = "trigger_research"
result = await do_trigger_research(content, owner=owner, chat_session_id=session_id)
result = await dispatched(do_trigger_research(content, owner=owner, chat_session_id=session_id))
elif tool == "manage_research":
desc = "manage_research"
result = await do_manage_research(content, owner=owner)
result = await dispatched(do_manage_research(content, owner=owner))
elif tool == "resolve_contact":
desc = "resolve_contact"
result = await do_resolve_contact(content, owner=owner)
result = await dispatched(do_resolve_contact(content, owner=owner))
elif tool == "manage_contact":
desc = "manage_contact"
result = await do_manage_contact(content, owner=owner)
result = await dispatched(do_manage_contact(content, owner=owner))
elif tool == "vault_search":
desc = "vault_search"
result = await do_vault_search(content, owner=owner)
result = await dispatched(do_vault_search(content, owner=owner))
elif tool == "vault_get":
desc = "vault_get"
result = await do_vault_get(content, owner=owner)
result = await dispatched(do_vault_get(content, owner=owner))
elif tool == "vault_unlock":
desc = "vault_unlock"
result = await do_vault_unlock(content, owner=owner)
result = await dispatched(do_vault_unlock(content, owner=owner))
elif tool in BUILTIN_EMAIL_TOOLS:
# Bare email tool name from fenced-block models (e.g. Ollama) — route to MCP email server.
# Non-admin owners never reach here: BUILTIN_EMAIL_TOOLS ⊆ NON_ADMIN_BLOCKED_TOOLS,
@@ -1879,7 +1849,7 @@ async def _execute_tool_block_impl(
if session_id:
args = dict(args)
args[_EMAIL_MCP_SESSION_ARG] = session_id
result = await mcp.call_tool(qualified, args)
result = await dispatched(mcp.call_tool(qualified, args))
else:
result = {"error": "MCP manager not available", "exit_code": 1}
elif tool.startswith("mcp__"):
@@ -1898,7 +1868,7 @@ async def _execute_tool_block_impl(
if session_id:
args = dict(args)
args[_EMAIL_MCP_SESSION_ARG] = session_id
result = _normalize_mcp_text_error(await mcp.call_tool(tool, args))
result = _normalize_mcp_text_error(await dispatched(mcp.call_tool(tool, args)))
else:
desc = f"mcp: {tool}"
result = {"error": "MCP manager not available", "exit_code": 1}
@@ -1907,14 +1877,14 @@ async def _execute_tool_block_impl(
elif tool in dynamic_handlers:
first_line = content.split(chr(10))[0][:80]
desc = f"registry: {tool} {first_line}".strip()
res = await _direct_fallback(
res = await dispatched(_direct_fallback(
tool,
content,
progress_cb=progress_cb,
session_id=session_id,
owner=owner,
client_runtime_context=client_runtime_context,
)
))
if isinstance(res, tuple):
desc, result = res