Merge commit 'refs/phase3/pre-ajax/publication-tip' into integration/pre-ajax-release

# Conflicts:
#	routes/chat_routes.py
#	routes/session_routes.py
#	src/agent_loop.py
#	src/agent_tools/filesystem_tools.py
#	src/teacher_escalation.py
#	src/tool_capabilities.py
#	src/tool_execution.py
#	tests/test_mcp_add_server_args_validation.py
#	tests/test_token_cache_atomic_swap.py
This commit is contained in:
Alexandre Teixeira
2026-10-05 15:59:59 +01:00
1395 changed files with 360455 additions and 105938 deletions
+443 -237
View File
@@ -6,9 +6,11 @@ import difflib
import secrets
import shutil
import time
import tempfile
from typing import Optional, Dict, Any, Tuple, List
from src.constants import MAX_READ_CHARS, MAX_DIFF_LINES, MAX_OUTPUT_CHARS
from src.path_confinement import is_inside
_CODENAV_SKIP_DIRS = frozenset({
".git", ".hg", ".svn", "node_modules", "venv", ".venv", "__pycache__",
@@ -19,6 +21,50 @@ _CODENAV_MAX_HITS = 200
_CODENAV_MAX_LINE = 400
_GREP_TIMEOUT_SECONDS = 20
_GREP_STDERR_PREFIX = 20_000
_STRUCTURED_DOCUMENT_SUFFIXES = frozenset({
".doc", ".docx", ".epub", ".pdf", ".pptx", ".xls", ".xlsx",
})
_BINARY_ARTIFACT_SUFFIXES = _STRUCTURED_DOCUMENT_SUFFIXES | frozenset({
".bmp", ".gif", ".ico", ".jpeg", ".jpg", ".mp3", ".mp4", ".ogg",
".png", ".wav", ".webm", ".webp", ".zip",
})
def _visible_bound_resource(path):
from src.agent_runtime.resource_binding import active_resource_operation
bound = active_resource_operation()
if bound is None:
return True
try:
bound.resolve_path(path)
return True
except (ValueError, OSError, RuntimeError):
return False
# Models frequently put source artifacts in a Markdown code fence even when a
# tool schema asks for the raw file body. Persisting that fence makes HTML,
# CSS, JavaScript, and source files invalid. Restrict normalization to
# code-like targets so a user can still write a literal fence to Markdown.
_FENCED_SOURCE_SUFFIXES = frozenset({
".css", ".csv", ".html", ".htm", ".js", ".json", ".jsx", ".mjs",
".py", ".sh", ".sql", ".svg", ".ts", ".tsx", ".xml", ".yaml", ".yml",
})
def _unwrap_fenced_source_body(body: str, path: str) -> str:
"""Remove an accidental outer Markdown fence from a source artifact.
An opening fence is enough to normalize: generation can end during a tool
call while its argument remains otherwise usable, and retaining the fence
corrupts the artifact. This only applies to source-like file extensions.
"""
if os.path.splitext(path)[1].casefold() not in _FENCED_SOURCE_SUFFIXES:
return body
match = re.match(r"^(\s*)```[^\r\n]*\r?\n", body)
if not match:
return body
unwrapped = body[match.end():]
return re.sub(r"\r?\n```\s*$", "", unwrapped)
def _glob_to_regex(pat: str) -> "re.Pattern":
@@ -46,6 +92,22 @@ def _glob_to_regex(pat: str) -> "re.Pattern":
return re.compile("".join(out))
def _validate_grep_descriptor(descriptor):
"""Revalidate an inert parent snapshot without inherited ContextVars."""
path, identity, ancestors = descriptor
if os.path.islink(path) or os.path.realpath(path) != path:
raise ValueError("grep: resource path changed")
for parent, observed in ancestors:
info = os.stat(parent, follow_symlinks=False)
if not os.path.isdir(parent) or (info.st_dev, info.st_ino) != tuple(observed):
raise ValueError("grep: resource ancestor changed")
info = os.stat(path, follow_symlinks=False)
import stat
if (not stat.S_ISREG(info.st_mode) or info.st_nlink > 1
or (info.st_dev, info.st_ino) != tuple(identity)):
raise ValueError("grep: resource identity changed")
def _python_grep_worker(payload: dict, output_queue) -> None:
"""Spawn-safe fallback grep worker used when ripgrep is unavailable.
@@ -66,85 +128,35 @@ def _python_grep_worker(payload: dict, output_queue) -> None:
output_queue.put(("error", f"grep: bad pattern: {exc}"))
return
requested_root = payload["root"]
skip_dirs = set(payload["skip_dirs"])
sensitive = {name.casefold() for name in payload["sensitive_names"]}
max_hits = payload["max_hits"]
hits = 0
def within(path: str, root: str) -> bool:
try:
return os.path.commonpath(
[os.path.normcase(path), os.path.normcase(root)]
) == os.path.normcase(root)
except ValueError:
return False
def safe_file(path: str, target: str) -> Optional[str]:
if os.path.islink(path):
return None
canonical = os.path.realpath(path)
if not within(canonical, requested_root) or not within(canonical, target):
return None
parts = [part.casefold() for part in canonical.split(os.sep)]
if any(part in sensitive for part in parts):
return None
try:
if not os.path.isfile(canonical) or os.stat(canonical).st_nlink > 1:
return None
except OSError:
return None
return canonical
for target in payload["targets"]:
for descriptor in payload["files"]:
if hits >= max_hits:
break
if os.path.isfile(target):
file_iter = iter((target,))
else:
def walk_files():
for directory, dirnames, filenames in os.walk(
target, followlinks=False
):
dirnames[:] = [
name
for name in dirnames
if name not in skip_dirs
and name.casefold() not in sensitive
and not os.path.islink(os.path.join(directory, name))
]
for name in filenames:
yield os.path.join(directory, name)
file_iter = walk_files()
for candidate in file_iter:
path = safe_file(candidate, target)
if path is None:
continue
relative = os.path.relpath(path, requested_root).replace(os.sep, "/")
if glob_regex and not (
glob_regex.fullmatch(relative)
or glob_regex.fullmatch(os.path.basename(path))
):
continue
try:
with open(path, "r", encoding="utf-8", errors="strict") as handle:
for number, line in enumerate(handle, 1):
if regex.search(line):
output_queue.put((
"match",
path,
number,
line.rstrip()[:_CODENAV_MAX_LINE],
))
hits += 1
if hits >= max_hits:
break
except (UnicodeDecodeError, OSError):
continue
if hits >= max_hits:
break
path, identity, ancestors = descriptor
_validate_grep_descriptor(descriptor)
relative = os.path.relpath(path, payload["base"]).replace(os.sep, "/")
if glob_regex and not (
glob_regex.fullmatch(relative) or glob_regex.fullmatch(os.path.basename(path))
):
continue
try:
with open(path, "r", encoding="utf-8", errors="strict") as handle:
info = os.fstat(handle.fileno())
if (info.st_dev, info.st_ino) != tuple(identity) or info.st_nlink > 1:
raise ValueError("grep: resource identity changed before read")
for number, line in enumerate(handle, 1):
if regex.search(line):
output_queue.put(("match", path, number, line.rstrip()[:_CODENAV_MAX_LINE]))
hits += 1
if hits >= max_hits:
break
except UnicodeDecodeError:
continue
except OSError as error:
output_queue.put(("error", f"grep: {error}"))
return
output_queue.put(("done",))
except BaseException as exc:
try:
@@ -180,52 +192,75 @@ def _unified_diff(old: str, new: str, path: str) -> Optional[Dict[str, Any]]:
"file": os.path.basename(path) or (path or "file"),
}
def _edit_file_text(original: str, old: str, new: str, replace_all: bool) -> tuple[str | None, str]:
"""The exact text edit_file writes for ``original``, or None and why not.
Pure: the effect adapter derives the requested post-state from this same
function, so the postcondition is the producer's own transformation.
"""
count = original.count(old)
if count == 0:
return None, "not_found"
if count > 1 and not replace_all:
return None, f"not_unique:{count}"
return (original.replace(old, new) if replace_all else original.replace(old, new, 1)), "ok"
class EditFileTool:
async def execute(self, content: str, ctx: dict) -> dict:
from src.tool_execution import _resolve_tool_path, _resolve_search_root, _truncate
try:
args = json.loads(content) if content.strip().startswith("{") else {}
except (json.JSONDecodeError, TypeError):
args = {}
raw_path = (args.get("path") or "").strip()
old = args.get("old_string", "")
new = args.get("new_string", "")
replace_all = bool(args.get("replace_all", False))
return {"error": "edit_file: expected valid JSON arguments", "exit_code": 1}
if not isinstance(args, dict):
return {"error": "edit_file: expected a JSON object", "exit_code": 1}
raw_path_value = args.get("path")
raw_path = raw_path_value.strip() if isinstance(raw_path_value, str) else ""
old = args.get("old_string")
new = args.get("new_string")
replace_all = args.get("replace_all", False)
if not raw_path:
return {"error": "edit_file: path required", "exit_code": 1}
if not isinstance(old, str) or not old:
return {"error": "edit_file: old_string required (use write_file to create a file)", "exit_code": 1}
if not isinstance(new, str):
return {"error": "edit_file: new_string required", "exit_code": 1}
if not isinstance(replace_all, bool):
return {"error": "edit_file: replace_all must be a boolean", "exit_code": 1}
try:
path = _resolve_tool_path(raw_path)
except ValueError as e:
return {"error": f"edit_file: {e}", "exit_code": 1}
if old == "":
return {"error": "edit_file: old_string required (use write_file to create a file)", "exit_code": 1}
if old == new:
return {"error": "edit_file: old_string and new_string are identical", "exit_code": 1}
def _apply():
"""Helper function that performs the actual string replacement and file writing logic."""
with open(path, "r", encoding="utf-8") as f:
# Exact replacement must not normalize unrelated CRLF/CR newlines.
with open(path, "r", encoding="utf-8", newline="") as f:
original = f.read()
count = original.count(old)
if count == 0:
return original, None, "not_found"
if count > 1 and not replace_all:
return original, None, f"not_unique:{count}"
updated = original.replace(old, new) if replace_all else original.replace(old, new, 1)
with open(path, "w", encoding="utf-8") as f:
updated, status = _edit_file_text(original, old, new, replace_all)
if updated is None:
return original, None, status
attempted.append(True)
with open(path, "w", encoding="utf-8", newline="") as f:
f.write(updated)
return original, updated, "ok"
# In-place rewrite: a failure after truncation may leave partial bytes.
attempted = []
partial = lambda: {"mutation_attempted": True} if attempted else {}
try:
original, updated, status = await asyncio.to_thread(_apply)
except FileNotFoundError:
return {"error": f"edit_file: {path}: not found (use write_file to create it)", "exit_code": 1}
return {"error": f"edit_file: {path}: not found (use write_file to create it)", "exit_code": 1, **partial()}
except (IsADirectoryError, UnicodeDecodeError):
return {"error": f"edit_file: {path}: not an editable text file", "exit_code": 1}
return {"error": f"edit_file: {path}: not an editable text file", "exit_code": 1, **partial()}
except PermissionError:
return {"error": f"edit_file: {path}: permission denied", "exit_code": 1}
return {"error": f"edit_file: {path}: permission denied", "exit_code": 1, **partial()}
except OSError as e:
return {"error": f"edit_file: {path}: {e}", "exit_code": 1}
return {"error": f"edit_file: {path}: {e}", "exit_code": 1, **partial()}
if status == "not_found":
return {"error": f"edit_file: old_string not found in {path}. Read the file and match it exactly.", "exit_code": 1}
@@ -248,17 +283,36 @@ class ReadFileTool:
if _stripped.startswith("{"):
try:
_a = json.loads(_stripped)
raw_path = str(_a.get("path", "")).strip()
if not isinstance(_a, dict):
return {"error": "read_file: expected a JSON object", "exit_code": 1}
raw_path_value = _a.get("path")
raw_path = raw_path_value.strip() if isinstance(raw_path_value, str) else ""
offset = int(_a.get("offset") or 0)
limit = int(_a.get("limit") or 0)
except (json.JSONDecodeError, TypeError, ValueError):
pass
return {"error": "read_file: expected valid JSON arguments", "exit_code": 1}
if not raw_path:
return {"error": "read_file: path required", "exit_code": 1}
try:
path = _resolve_tool_path(raw_path)
except ValueError as e:
return {"error": f"read_file: {e}", "exit_code": 1}
try:
def _read():
if os.path.splitext(path)[1].lower() in _STRUCTURED_DOCUMENT_SUFFIXES:
from src.document_processor import extract_local_document
extracted = extract_local_document(
path,
display_name=os.path.basename(path),
analyze_embedded_images=False,
)
if offset > 0 or limit > 0:
lines = extracted.splitlines(keepends=True)
start = max(offset, 1) - 1
stop = start + limit if limit > 0 else None
return "".join(lines[start:stop])[:MAX_READ_CHARS]
return extracted[:MAX_READ_CHARS + 1]
if offset > 0 or limit > 0:
start = max(offset, 1)
out, n, budget = [], 0, MAX_READ_CHARS
@@ -331,44 +385,91 @@ class _EmptyBodyWouldTruncate(Exception):
self.path = path
self.existing_bytes = existing_bytes
def _parse_write_intent(content: str) -> tuple[str, str, bool, bool]:
"""Classify the original transport, before binding or source normalization.
The returned clear flag is derived here, never from a caller-supplied key.
"""
if not isinstance(content, str):
raise ValueError("write_file: expected string arguments")
if content.lstrip().startswith("{"):
try:
args = json.loads(content)
except (TypeError, ValueError) as error:
raise ValueError("write_file: expected valid JSON arguments") from error
if not isinstance(args, dict):
raise ValueError("write_file: expected a JSON object")
path, body = args.get("path"), args.get("content")
if not isinstance(body, str):
raise ValueError("write_file: content required and must be a string")
if not isinstance(path, str):
raise ValueError("write_file: path required and must be a string")
return path.strip(), body, True, not body.strip()
path, delimiter, body = content.partition("\n")
return path.strip(), body, bool(delimiter), False
class WriteFileTool:
async def execute(self, content: str, ctx: dict) -> dict:
from src.tool_execution import _resolve_tool_path, _resolve_search_root, _truncate
lines = content.split("\n", 1)
raw_path = lines[0].strip()
body = lines[1] if len(lines) > 1 else ""
# Only the fenced inline-JSON form can say "this file is meant to be empty":
# the text form's `path\n` and a body a parser dropped look identical here.
declared_clear = False
# Decode JSON-object args (the fenced inline-args shape
# ```write_file {"path": "...", "content": "..."}```), matching
# ReadFileTool above. Without this the whole JSON string becomes the
# path and the file is written under a garbage name. This is the live
# path: there is no filesystem MCP server, so write_file always runs
# here via _direct_fallback, not through _build_mcp_args.
_stripped = content.strip()
if _stripped.startswith("{"):
try:
_a = json.loads(_stripped)
if isinstance(_a, dict) and "path" in _a:
raw_path = str(_a.get("path", "")).strip()
_content = _a.get("content")
# A `content` key that is literally an empty (or whitespace-only)
# string is the caller declaring the file should be cleared. A
# missing key or a null is what a parser that lost the body leaves
# behind, so neither declares anything. The old
# `str(_a.get("content", ""))` also turned null into the 4 bytes
# "None", which could be neither refused nor honoured.
declared_clear = isinstance(_content, str) and not _content.strip()
body = "" if _content is None else (
_content if isinstance(_content, str) else str(_content)
)
except (json.JSONDecodeError, TypeError, ValueError):
pass
from src.tool_execution import _display_tool_path, _resolve_tool_path
from src.agent_runtime.resource_binding import active_resource_operation
bound = active_resource_operation()
original = bound.operation.input if bound is not None else content
try:
_, _, has_section, declared_clear = _parse_write_intent(original)
raw_path, body, _, _ = _parse_write_intent(content)
except ValueError as error:
return {"error": str(error), "exit_code": 1}
if not raw_path:
return {"error": "write_file: path required", "exit_code": 1}
try:
path = _resolve_tool_path(raw_path)
except ValueError as e:
return {"error": f"write_file: {e}", "exit_code": 1}
body = _unwrap_fenced_source_body(body, path)
# A frequent multimodal artifact failure is writing SVG markup to a
# path whose extension promises a raster image. The file exists, so
# ordinary artifact checks pass, but image judges cannot decode it.
# Reject the mismatch with an actionable native-tool recovery path:
# save the SVG with an .svg suffix, then use inspect_media to render
# it to the requested PNG/JPEG path.
image_suffixes = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp"}
body_probe = body.lstrip().casefold()
if os.path.splitext(path)[1].casefold() in image_suffixes and (
body_probe.startswith("<svg")
or (body_probe.startswith("<?xml") and "<svg" in body_probe[:2000])
):
return {
"error": (
f"write_file: {path} contains SVG markup but has a raster "
"image extension. Write the SVG to a .svg path first, "
"then call inspect_media with that SVG as path and this "
"path as output_path to render a real raster image."
),
"exit_code": 1,
"artifact_format_error": True,
}
# write_file is a UTF-8 text writer. Refuse to silently destroy an
# existing PDF, image, archive, or media artifact produced by a
# format-aware tool, especially after the agent has verified it.
suffix = os.path.splitext(path)[1].casefold()
if suffix in _BINARY_ARTIFACT_SUFFIXES:
target_existed = os.path.isfile(path)
return {
"error": (
f"write_file: refusing UTF-8 text for binary artifact path {path}. "
"Use Python or a format-specific creation tool, then inspect the result."
),
"exit_code": 1,
"binary_artifact_preserved": target_existed,
}
if not has_section:
return {"error": "write_file: content required; missing content section", "exit_code": 1}
if raw_path.endswith(("/", "\\")) or os.path.isdir(path):
return {"error": "write_file: target is a directory", "exit_code": 1}
# This writer truncates in place. Once that stage is reached, a failure
# may leave a partial file; report it so effect evidence stays honest.
attempted = []
try:
def _write():
old = ""
@@ -395,10 +496,12 @@ class WriteFileTool:
# Publish whitespace content atomically. Writing it
# after exclusive creation could overwrite bytes from
# a writer that filled the new placeholder meanwhile.
attempted.append(True)
_write_new_file_without_overwrite(path, body)
else:
# An exact empty body needs no staged data, so create
# the file exclusively and never write through it.
attempted.append(True)
with open(path, "x", encoding="utf-8"):
pass
except FileExistsError:
@@ -410,6 +513,7 @@ class WriteFileTool:
raise
return old, len(body)
attempted.append(True)
with open(path, "w", encoding="utf-8") as f:
f.write(body)
return old, len(body)
@@ -426,11 +530,19 @@ class WriteFileTool:
"exit_code": 1,
}
except PermissionError:
return {"error": f"write_file: {path}: permission denied", "exit_code": 1}
return {"error": f"write_file: {path}: permission denied", "exit_code": 1,
**({"mutation_attempted": True} if attempted else {})}
except OSError as e:
return {"error": f"write_file: {path}: {e}", "exit_code": 1}
diff = _unified_diff(old_content, body, path)
result = {"output": f"Wrote {size} bytes to {path}", "exit_code": 0}
return {"error": f"write_file: {path}: {e}", "exit_code": 1,
**({"mutation_attempted": True} if attempted else {})}
committed_body = old_content if size == 0 and not declared_clear and not body.strip() else body
diff = _unified_diff(old_content, committed_body, path)
result = {
"output": (f"Wrote {size} bytes to {_display_tool_path(path)}" if attempted else
f"No write performed for {_display_tool_path(path)} (implicit empty body)"),
"exit_code": 0,
**({"write_noop": True} if not attempted else {}),
}
if diff:
result["diff"] = diff
return result
@@ -485,16 +597,77 @@ class ApplyPatchTool:
new = _apply_patch_hunks(old, op["hunks"], op["path"])
prepared.append((kind, path, old, new))
staged: list[tuple[str, str]] = []
backups: list[tuple[str, str | None]] = []
try:
for kind, path, _old, new in prepared:
if kind == "delete":
continue
directory = os.path.dirname(path) or "."
os.makedirs(directory, exist_ok=True)
fd, temp_path = tempfile.mkstemp(
prefix=f".{os.path.basename(path)}.odysseus-",
dir=directory,
)
try:
with os.fdopen(fd, "w", encoding="utf-8", newline="") as handle:
handle.write(new)
handle.flush()
os.fsync(handle.fileno())
if os.path.exists(path):
shutil.copymode(path, temp_path)
except BaseException:
try:
os.unlink(temp_path)
except OSError:
pass
raise
staged.append((path, temp_path))
for _kind, path, _old, _new in prepared:
if os.path.exists(path):
directory = os.path.dirname(path) or "."
fd, backup_path = tempfile.mkstemp(
prefix=f".{os.path.basename(path)}.odysseus-backup-",
dir=directory,
)
os.close(fd)
os.unlink(backup_path)
os.replace(path, backup_path)
backups.append((path, backup_path))
else:
backups.append((path, None))
staged_by_path = dict(staged)
for kind, path, _old, _new in prepared:
if kind != "delete":
os.replace(staged_by_path[path], path)
staged.clear()
except BaseException:
for path, backup_path in reversed(backups):
try:
if os.path.exists(path):
os.unlink(path)
if backup_path and os.path.exists(backup_path):
os.replace(backup_path, path)
except OSError:
pass
raise
finally:
for _path, temp_path in staged:
try:
os.unlink(temp_path)
except OSError:
pass
for _path, backup_path in backups:
if backup_path:
try:
os.unlink(backup_path)
except OSError:
pass
diffs = []
for kind, path, old, new in prepared:
if kind == "delete":
os.remove(path)
else:
directory = os.path.dirname(path)
if directory:
os.makedirs(directory, exist_ok=True)
with open(path, "w", encoding="utf-8") as f:
f.write(new)
for _kind, path, old, new in prepared:
diff = _unified_diff(old, new, path)
if diff:
diffs.append(diff)
@@ -612,11 +785,7 @@ def _apply_patch_hunks(original: str, hunks: List[List[str]], label: str) -> str
class LsTool:
async def execute(self, content: str, ctx: dict) -> dict:
from src.tool_execution import (
_is_denied_tool_path,
_resolve_search_root,
_truncate,
)
from src.tool_execution import _display_tool_path, _is_denied_tool_path, _resolve_search_root, _truncate
raw_path = ""
_s = (content or "").strip()
if _s.startswith("{"):
@@ -640,7 +809,7 @@ class LsTool:
for entry in it:
if entry.name.startswith("."):
continue
if _is_denied_tool_path(os.path.realpath(entry.path)):
if _is_denied_tool_path(os.path.realpath(entry.path)) or not _visible_bound_resource(entry.path):
continue
try:
is_dir = entry.is_dir(follow_symlinks=False)
@@ -651,7 +820,7 @@ class LsTool:
except (PermissionError, OSError) as _e:
return None, f"ls: {_e}"
rows.sort(key=lambda r: (not r[0], r[1].lower()))
lines = [f"{root}:"]
lines = [f"{_display_tool_path(root)}:"]
for is_dir, name, size in rows[:_CODENAV_MAX_HITS]:
lines.append(f" {name}/" if is_dir else f" {name} ({size} B)")
if len(rows) > _CODENAV_MAX_HITS:
@@ -671,6 +840,8 @@ class GlobTool:
_SENSITIVE_BASENAMES,
_can_traverse_tool_path,
_is_denied_tool_path,
_display_tool_path,
_is_sensitive_path,
_resolve_tool_path,
_resolve_search_root,
_truncate,
@@ -708,18 +879,12 @@ class GlobTool:
# confinement that _resolve_search_root applies to the root.
# An escaping literal falls through to the walk, which only ever
# yields paths under base.
nbase = os.path.normcase(rbase)
try:
inside = cand == rbase or os.path.commonpath(
[os.path.normcase(cand), nbase]
) == nbase
except ValueError:
inside = False
inside = is_inside(rbase, cand)
# A literal that names a deny-listed sensitive file (.env,
# .ssh/id_rsa, …) falls through to the walk, which skips it —
# otherwise glob would surface secret paths that read_file /
# grep already refuse to touch.
if inside and os.path.exists(cand) and not _is_denied_tool_path(cand):
if inside and os.path.exists(cand) and not _is_denied_tool_path(cand) and _visible_bound_resource(cand):
return [cand], None
# Literal not at exact path — fall through to walk so
# e.g. "foo.py" still matches at any depth (like rglob).
@@ -748,7 +913,7 @@ class GlobTool:
if regex.fullmatch(rel) or regex.fullmatch(name):
# Skip deny-listed sensitive files (.env, id_rsa,
# known_hosts, …) the same way grep does.
if _is_denied_tool_path(os.path.realpath(full)):
if _is_denied_tool_path(os.path.realpath(full)) or not _visible_bound_resource(full):
continue
try:
mtime = os.stat(full).st_mtime
@@ -766,8 +931,8 @@ class GlobTool:
if err:
return {"error": err, "exit_code": 1}
if not paths:
return {"output": f"No files matching {pattern!r} under {root}", "exit_code": 0}
out = "\n".join(paths)
return {"output": f"No files matching {pattern!r} under {_display_tool_path(root)}", "exit_code": 0}
out = "\n".join(_display_tool_path(path) for path in paths)
if len(paths) >= _CODENAV_MAX_HITS:
out += f"\n... [capped at {_CODENAV_MAX_HITS} files]"
return {"output": _truncate(out), "exit_code": 0}
@@ -779,6 +944,8 @@ class GrepTool:
_SENSITIVE_FILE_PATTERNS,
_agent_readable_data_subdirs,
_is_denied_tool_path,
_display_tool_path,
_can_traverse_tool_path,
_is_sensitive_path,
_path_within,
_resolve_search_root,
@@ -818,72 +985,86 @@ class GrepTool:
rg = shutil.which("rg")
real_root = os.path.realpath(root)
data_dir = os.path.realpath(DATA_DIR)
spans_state = _path_within(data_dir, real_root)
def is_top_level_safe(path: str, *, partition_generated: bool) -> bool:
lexical = os.path.abspath(path)
if os.path.islink(lexical):
return False
canonical = os.path.realpath(lexical)
if not _path_within(canonical, real_root):
return False
if partition_generated and os.path.basename(lexical) in _CODENAV_SKIP_DIRS:
return False
if _is_sensitive_path(canonical) or _is_denied_tool_path(canonical):
return False
return True
def safe_targets() -> tuple[list[str], Optional[str]]:
candidates: list[tuple[str, bool]] = []
if not spans_state:
# Preserve direct-root compatibility: skip-directory policy
# prunes descendants, but an explicitly requested allowed
# root named node_modules remains searchable.
candidates.append((real_root, False))
else:
current = real_root
if current != data_dir:
for part in os.path.relpath(data_dir, current).split(os.sep):
try:
with os.scandir(current) as entries:
for entry in entries:
if entry.name != part:
# Reject a sibling link lexically before
# canonicalizing or treating it as a target.
if entry.is_symlink():
continue
candidates.append((entry.path, True))
except OSError as exc:
return [], f"grep: {exc}"
current = os.path.join(current, part)
for readable in _agent_readable_data_subdirs():
if (
_path_within(readable, data_dir)
and _path_within(readable, real_root)
and os.path.exists(readable)
):
candidates.append((readable, True))
targets: list[str] = []
seen: set[str] = set()
for candidate, partition_generated in candidates:
if not is_top_level_safe(
candidate, partition_generated=partition_generated
):
continue
canonical = os.path.realpath(candidate)
if canonical not in seen:
seen.add(canonical)
targets.append(canonical)
return targets, None
targets, target_error = safe_targets()
if target_error:
return None, target_error
base = real_root if os.path.isdir(real_root) else os.path.dirname(real_root)
deadline = time.monotonic() + _GREP_TIMEOUT_SECONDS
base = real_root if os.path.isdir(real_root) else os.path.dirname(real_root)
from src.agent_runtime.resource_binding import active_resource_operation
bound = active_resource_operation()
if bound is not None:
bound.validate()
files = []
def check_deadline():
if time.monotonic() >= deadline:
raise TimeoutError("grep: timed out")
def observe_file(path):
check_deadline()
if os.path.islink(path):
return
canonical = os.path.realpath(path)
if not _path_within(canonical, base) or _is_denied_tool_path(canonical):
return
if bound is not None:
try:
bound.resolve_path(canonical)
except ValueError:
# Intentionally denied publication control-plane paths
# are omitted before any producer is allowed to read.
return
info = os.stat(canonical, follow_symlinks=False)
if not os.path.isfile(canonical):
return
ancestors = []
parent = os.path.dirname(canonical)
while _path_within(parent, base):
observed = os.stat(parent, follow_symlinks=False)
ancestors.append((parent, (observed.st_dev, observed.st_ino)))
if parent == base:
break
parent = os.path.dirname(parent)
files.append((canonical, (info.st_dev, info.st_ino), tuple(ancestors)))
if len(files) > 100_000:
raise ValueError("grep: enumeration limit exceeded; scan incomplete")
try:
if os.path.islink(root):
raise ValueError("grep: symlink search root is not allowed")
if os.path.isfile(real_root):
observe_file(real_root)
elif os.path.isdir(real_root):
pending_directories = [real_root]
enumerated = 0
while pending_directories:
check_deadline()
directory = pending_directories.pop()
if not _can_traverse_tool_path(directory):
continue
with os.scandir(directory) as entries:
for entry in entries:
check_deadline()
enumerated += 1
if enumerated > 100_000:
raise ValueError("grep: enumeration limit exceeded; scan incomplete")
if entry.is_symlink():
continue
canonical = os.path.realpath(entry.path)
if not _path_within(canonical, base):
raise ValueError("grep: directory identity changed during enumeration")
if entry.is_dir(follow_symlinks=False):
if (entry.name not in _CODENAV_SKIP_DIRS
and _can_traverse_tool_path(canonical)
and (bound is None or _is_denied_tool_path(canonical)
or _visible_bound_resource(canonical))):
pending_directories.append(canonical)
else:
observe_file(entry.path)
else:
raise FileNotFoundError(f"grep: {root}: not found")
check_deadline()
except (OSError, ValueError) as error:
return None, str(error) if str(error).startswith("grep:") else f"grep: {error}"
descriptors = {record[0]: record for record in files}
targets = list(descriptors)
lines: list[str] = []
def parse_rg_result(raw: str) -> Optional[str]:
@@ -901,9 +1082,12 @@ class GrepTool:
return None
absolute = path if os.path.isabs(path) else os.path.join(base, path)
canonical = os.path.realpath(absolute)
if not _path_within(canonical, real_root) or _is_denied_tool_path(canonical):
return None
return f"{os.path.abspath(absolute)}:{number}:{text_value.rstrip()[:_CODENAV_MAX_LINE]}"
if canonical not in descriptors:
raise ValueError("grep: producer returned an undeclared resource")
_validate_grep_descriptor(descriptors[canonical])
if bound is not None:
bound.resolve_path(canonical)
return f"{_display_tool_path(canonical)}:{number}:{text_value.rstrip()[:_CODENAV_MAX_LINE]}"
def run_rg(cmd: list[str]) -> Optional[str]:
try:
@@ -1000,7 +1184,7 @@ class GrepTool:
if rg:
# Validate even when policy filtering leaves no search targets.
if not targets:
error = run_rg([rg, "--json", "--no-config", "--regexp", pattern])
error = run_rg([rg, "--json", "--no-config", "--regexp", pattern, "--", "-"])
return (None, error) if error else ([], None)
relative_targets = [os.path.relpath(target, base) for target in targets]
for offset in range(0, len(relative_targets), 128):
@@ -1021,7 +1205,15 @@ class GrepTool:
for skipped_dir in _CODENAV_SKIP_DIRS:
cmd += ["--glob", f"!**/{skipped_dir}/**"]
cmd += ["--regexp", pattern, "--", *relative_targets[offset:offset + 128]]
error = run_rg(cmd)
try:
for path in targets[offset:offset + 128]:
check_deadline()
_validate_grep_descriptor(descriptors[path])
if bound is not None:
bound.resolve_path(path)
error = run_rg(cmd)
except (OSError, ValueError) as exc:
return None, f"grep: {exc}"
if error:
return None, error
return lines, None
@@ -1030,8 +1222,9 @@ class GrepTool:
# multithreaded process and can deadlock. Spawn is platform-safe and
# PyInstaller-compatible via launcher's early freeze_support().
payload = {
"root": real_root,
"targets": targets,
"root": base,
"base": base,
"files": tuple(files),
"pattern": pattern,
"ignore_case": ignore_case,
"glob": glob_pat,
@@ -1090,9 +1283,17 @@ class GrepTool:
break
_, path, number, text_value = record
canonical = os.path.realpath(path)
if not _path_within(canonical, real_root) or _is_denied_tool_path(canonical):
continue
rendered = f"{path}:{number}:{text_value}"
if canonical not in descriptors:
error = "grep: fallback returned an undeclared resource"
break
try:
_validate_grep_descriptor(descriptors[canonical])
if bound is not None:
bound.resolve_path(canonical)
except (OSError, ValueError) as exc:
error = f"grep: {exc}"
break
rendered = f"{_display_tool_path(canonical)}:{number}:{text_value}"
if rendered not in lines:
lines.append(rendered)
finally:
@@ -1107,7 +1308,7 @@ class GrepTool:
output_queue.close()
if error:
return None, error
if worker.exitcode not in (0, None) and len(lines) < max_hits:
if (not completed or worker.exitcode not in (0, None)) and len(lines) < max_hits:
return None, f"grep: fallback worker exited {worker.exitcode}"
return lines, None
@@ -1115,8 +1316,13 @@ class GrepTool:
if err:
return {"error": err, "exit_code": 1}
if not lines:
return {"output": f"No matches for {pattern!r} under {root}", "exit_code": 0}
out = "\n".join(ln[:_CODENAV_MAX_LINE] for ln in lines)
return {"output": f"No matches for {pattern!r} under {_display_tool_path(root)}", "exit_code": 0}
physical_root = os.path.realpath(root)
display_root = _display_tool_path(physical_root)
out = "\n".join(
(display_root + ln[len(physical_root):] if ln.startswith(physical_root) else ln)[:_CODENAV_MAX_LINE]
for ln in lines
)
if len(lines) >= max_hits:
out += f"\n... [capped at {max_hits} matches]"
return {"output": _truncate(out), "exit_code": 0}
@@ -1129,7 +1335,7 @@ class GetWorkspaceTool:
ws = get_active_workspace()
if ws:
return {
"output": f"{ws}\n(File tools are confined to this folder; the shell starts "
"output": "/workspace\n(File tools are confined to this folder; the shell starts "
f"here but is not sandboxed and can reach outside it.)",
"exit_code": 0,
}