import asyncio import json import os import re import difflib import fnmatch import shutil import tempfile from typing import Optional, Dict, Any, Tuple, List from src.constants import MAX_READ_CHARS, MAX_DIFF_LINES, MAX_OUTPUT_CHARS from src.path_confinement import is_inside _CODENAV_SKIP_DIRS = frozenset({ ".git", ".hg", ".svn", "node_modules", "venv", ".venv", "__pycache__", ".mypy_cache", ".pytest_cache", ".ruff_cache", "dist", "build", ".next", ".cache", "site-packages", ".idea", ".tox", }) _CODENAV_MAX_HITS = 200 _CODENAV_MAX_LINE = 400 _STRUCTURED_DOCUMENT_SUFFIXES = frozenset({ ".doc", ".docx", ".epub", ".pdf", ".pptx", ".xls", ".xlsx", }) _BINARY_ARTIFACT_SUFFIXES = _STRUCTURED_DOCUMENT_SUFFIXES | frozenset({ ".bmp", ".gif", ".ico", ".jpeg", ".jpg", ".mp3", ".mp4", ".ogg", ".png", ".wav", ".webm", ".webp", ".zip", }) def _visible_bound_resource(path): from src.agent_runtime.resource_binding import active_resource_operation bound = active_resource_operation() if bound is None: return True try: bound.resolve_path(path) return True except (ValueError, OSError, RuntimeError): return False # Models frequently put source artifacts in a Markdown code fence even when a # tool schema asks for the raw file body. Persisting that fence makes HTML, # CSS, JavaScript, and source files invalid. Restrict normalization to # code-like targets so a user can still write a literal fence to Markdown. _FENCED_SOURCE_SUFFIXES = frozenset({ ".css", ".csv", ".html", ".htm", ".js", ".json", ".jsx", ".mjs", ".py", ".sh", ".sql", ".svg", ".ts", ".tsx", ".xml", ".yaml", ".yml", }) def _unwrap_fenced_source_body(body: str, path: str) -> str: """Remove an accidental outer Markdown fence from a source artifact. An opening fence is enough to normalize: generation can end during a tool call while its argument remains otherwise usable, and retaining the fence corrupts the artifact. This only applies to source-like file extensions. """ if os.path.splitext(path)[1].casefold() not in _FENCED_SOURCE_SUFFIXES: return body match = re.match(r"^(\s*)```[^\r\n]*\r?\n", body) if not match: return body unwrapped = body[match.end():] return re.sub(r"\r?\n```\s*$", "", unwrapped) def _glob_to_regex(pat: str) -> "re.Pattern": """Translate a forward-slash glob (**, *, ?) into a compiled regex. `**/` matches zero or more complete directories. `*` matches within a single path segment (does not cross /). """ i, n, out = 0, len(pat), [] while i < n: if pat[i : i + 3] == "**/": out.append("(?:[^/]+/)*") i += 3 elif pat[i : i + 2] == "**": out.append(".*") i += 2 elif pat[i] == "*": out.append("[^/]*") i += 1 elif pat[i] == "?": out.append("[^/]") i += 1 else: out.append(re.escape(pat[i])) i += 1 return re.compile("".join(out)) def _unified_diff(old: str, new: str, path: str) -> Optional[Dict[str, Any]]: if old == new: return None old_lines = old.splitlines() new_lines = new.splitlines() label = path or "file" diff_lines = list(difflib.unified_diff( old_lines, new_lines, fromfile=f"a/{label}", tofile=f"b/{label}", lineterm="", )) added = sum(1 for line in diff_lines if line.startswith("+") and not line.startswith("+++")) removed = sum(1 for line in diff_lines if line.startswith("-") and not line.startswith("---")) truncated = False if len(diff_lines) > MAX_DIFF_LINES: diff_lines = diff_lines[:MAX_DIFF_LINES] truncated = True text = "\n".join(diff_lines) if truncated: text += f"\n… diff truncated at {MAX_DIFF_LINES} lines" return { "text": text, "added": added, "removed": removed, "new_file": old == "", "file": os.path.basename(path) or (path or "file"), } class EditFileTool: async def execute(self, content: str, ctx: dict) -> dict: from src.tool_execution import _resolve_tool_path, _resolve_search_root, _truncate try: args = json.loads(content) if content.strip().startswith("{") else {} except (json.JSONDecodeError, TypeError): return {"error": "edit_file: expected valid JSON arguments", "exit_code": 1} if not isinstance(args, dict): return {"error": "edit_file: expected a JSON object", "exit_code": 1} raw_path_value = args.get("path") raw_path = raw_path_value.strip() if isinstance(raw_path_value, str) else "" old = args.get("old_string") new = args.get("new_string") replace_all = args.get("replace_all", False) if not raw_path: return {"error": "edit_file: path required", "exit_code": 1} if not isinstance(old, str) or not old: return {"error": "edit_file: old_string required (use write_file to create a file)", "exit_code": 1} if not isinstance(new, str): return {"error": "edit_file: new_string required", "exit_code": 1} if not isinstance(replace_all, bool): return {"error": "edit_file: replace_all must be a boolean", "exit_code": 1} try: path = _resolve_tool_path(raw_path) except ValueError as e: return {"error": f"edit_file: {e}", "exit_code": 1} if old == new: return {"error": "edit_file: old_string and new_string are identical", "exit_code": 1} def _apply(): """Helper function that performs the actual string replacement and file writing logic.""" # Exact replacement must not normalize unrelated CRLF/CR newlines. with open(path, "r", encoding="utf-8", newline="") as f: original = f.read() count = original.count(old) if count == 0: return original, None, "not_found" if count > 1 and not replace_all: return original, None, f"not_unique:{count}" updated = original.replace(old, new) if replace_all else original.replace(old, new, 1) with open(path, "w", encoding="utf-8", newline="") as f: f.write(updated) return original, updated, "ok" try: original, updated, status = await asyncio.to_thread(_apply) except FileNotFoundError: return {"error": f"edit_file: {path}: not found (use write_file to create it)", "exit_code": 1} except (IsADirectoryError, UnicodeDecodeError): return {"error": f"edit_file: {path}: not an editable text file", "exit_code": 1} except PermissionError: return {"error": f"edit_file: {path}: permission denied", "exit_code": 1} except OSError as e: return {"error": f"edit_file: {path}: {e}", "exit_code": 1} if status == "not_found": return {"error": f"edit_file: old_string not found in {path}. Read the file and match it exactly.", "exit_code": 1} if status.startswith("not_unique"): n = status.split(":", 1)[1] return {"error": f"edit_file: old_string is not unique in {path} ({n} matches). Add surrounding context or set replace_all=true.", "exit_code": 1} n = original.count(old) result = {"output": f"Edited {path} ({n} replacement{'s' if n != 1 else ''})", "exit_code": 0} diff = _unified_diff(original, updated, path) if diff: result["diff"] = diff return result class ReadFileTool: async def execute(self, content: str, ctx: dict) -> dict: from src.tool_execution import _resolve_tool_path, _resolve_search_root, _truncate raw_path, offset, limit = content.split("\n", 1)[0].strip(), 0, 0 _stripped = content.strip() if _stripped.startswith("{"): try: _a = json.loads(_stripped) if not isinstance(_a, dict): return {"error": "read_file: expected a JSON object", "exit_code": 1} raw_path_value = _a.get("path") raw_path = raw_path_value.strip() if isinstance(raw_path_value, str) else "" offset = int(_a.get("offset") or 0) limit = int(_a.get("limit") or 0) except (json.JSONDecodeError, TypeError, ValueError): return {"error": "read_file: expected valid JSON arguments", "exit_code": 1} if not raw_path: return {"error": "read_file: path required", "exit_code": 1} try: path = _resolve_tool_path(raw_path) except ValueError as e: return {"error": f"read_file: {e}", "exit_code": 1} try: def _read(): if os.path.splitext(path)[1].lower() in _STRUCTURED_DOCUMENT_SUFFIXES: from src.document_processor import extract_local_document extracted = extract_local_document( path, display_name=os.path.basename(path), analyze_embedded_images=False, ) if offset > 0 or limit > 0: lines = extracted.splitlines(keepends=True) start = max(offset, 1) - 1 stop = start + limit if limit > 0 else None return "".join(lines[start:stop])[:MAX_READ_CHARS] return extracted[:MAX_READ_CHARS + 1] if offset > 0 or limit > 0: start = max(offset, 1) out, n, budget = [], 0, MAX_READ_CHARS with open(path, "r", encoding="utf-8", errors="replace") as f: for i, line in enumerate(f, 1): if i < start: continue if limit > 0 and n >= limit: break out.append(line) n += 1 budget -= len(line) if budget <= 0: out.append(f"\n... [truncated at {MAX_READ_CHARS} chars]") break return "".join(out) with open(path, "r", encoding="utf-8", errors="replace") as f: return f.read(MAX_READ_CHARS + 1) data = await asyncio.to_thread(_read) except FileNotFoundError: return {"error": f"read_file: {path}: not found", "exit_code": 1} except PermissionError: return {"error": f"read_file: {path}: permission denied", "exit_code": 1} except IsADirectoryError: return {"error": f"read_file: {path}: is a directory (use ls)", "exit_code": 1} except OSError as e: return {"error": f"read_file: {path}: {e}", "exit_code": 1} if not (offset > 0 or limit > 0) and len(data) > MAX_READ_CHARS: data = data[:MAX_READ_CHARS] + f"\n... [truncated at {MAX_READ_CHARS} chars]" return {"output": data, "exit_code": 0} class WriteFileTool: async def execute(self, content: str, ctx: dict) -> dict: from src.tool_execution import _display_tool_path, _resolve_tool_path, _resolve_search_root, _truncate lines = content.split("\n", 1) raw_path = lines[0].strip() body = lines[1] if len(lines) > 1 else "" # Decode JSON-object args (the fenced inline-args shape # ```write_file {"path": "...", "content": "..."}```), matching # ReadFileTool above. Without this the whole JSON string becomes the # path and the file is written under a garbage name. This is the live # path: there is no filesystem MCP server, so write_file always runs # here via _direct_fallback, not through _build_mcp_args. _stripped = content.strip() if _stripped.startswith("{"): try: _a = json.loads(_stripped) if not isinstance(_a, dict): return {"error": "write_file: expected a JSON object", "exit_code": 1} raw_path_value = _a.get("path") body_value = _a.get("content") raw_path = raw_path_value.strip() if isinstance(raw_path_value, str) else "" if not isinstance(body_value, str): return {"error": "write_file: content required", "exit_code": 1} body = body_value except (json.JSONDecodeError, TypeError, ValueError): return {"error": "write_file: expected valid JSON arguments", "exit_code": 1} if not raw_path: return {"error": "write_file: path required", "exit_code": 1} try: path = _resolve_tool_path(raw_path) except ValueError as e: return {"error": f"write_file: {e}", "exit_code": 1} body = _unwrap_fenced_source_body(body, path) # A frequent multimodal artifact failure is writing SVG markup to a # path whose extension promises a raster image. The file exists, so # ordinary artifact checks pass, but image judges cannot decode it. # Reject the mismatch with an actionable native-tool recovery path: # save the SVG with an .svg suffix, then use inspect_media to render # it to the requested PNG/JPEG path. image_suffixes = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp"} body_probe = body.lstrip().casefold() if os.path.splitext(path)[1].casefold() in image_suffixes and ( body_probe.startswith(" dict: """Apply a small Codex-style patch using exact context matching. This is deliberately stricter than git-apply: if an update hunk's old text is not found exactly once, the whole patch is rejected before any file is changed. That keeps agent edits reviewable and avoids fuzzy corruption when the model patches stale context. """ from src.tool_execution import _resolve_tool_path patch_text = content or "" stripped = patch_text.strip() if stripped.startswith("{"): try: args = json.loads(stripped) if isinstance(args, dict): patch_text = str(args.get("patch_text") or args.get("patchText") or args.get("patch") or "") except (json.JSONDecodeError, TypeError): pass if not patch_text.strip(): return {"error": "apply_patch: patch_text required", "exit_code": 1} try: ops = _parse_agent_patch(patch_text) if not ops: return {"error": "apply_patch: no file operations found", "exit_code": 1} prepared = [] for op in ops: path = _resolve_tool_path(op["path"]) kind = op["kind"] if kind == "add": if os.path.exists(path): return {"error": f"apply_patch: {op['path']}: already exists", "exit_code": 1} old = "" new = op["content"] elif kind == "delete": if not os.path.isfile(path): return {"error": f"apply_patch: {op['path']}: not found", "exit_code": 1} with open(path, "r", encoding="utf-8") as f: old = f.read() new = "" else: if not os.path.isfile(path): return {"error": f"apply_patch: {op['path']}: not found", "exit_code": 1} with open(path, "r", encoding="utf-8") as f: old = f.read() new = _apply_patch_hunks(old, op["hunks"], op["path"]) prepared.append((kind, path, old, new)) staged: list[tuple[str, str]] = [] backups: list[tuple[str, str | None]] = [] try: for kind, path, _old, new in prepared: if kind == "delete": continue directory = os.path.dirname(path) or "." os.makedirs(directory, exist_ok=True) fd, temp_path = tempfile.mkstemp( prefix=f".{os.path.basename(path)}.odysseus-", dir=directory, ) try: with os.fdopen(fd, "w", encoding="utf-8", newline="") as handle: handle.write(new) handle.flush() os.fsync(handle.fileno()) if os.path.exists(path): shutil.copymode(path, temp_path) except BaseException: try: os.unlink(temp_path) except OSError: pass raise staged.append((path, temp_path)) for _kind, path, _old, _new in prepared: if os.path.exists(path): directory = os.path.dirname(path) or "." fd, backup_path = tempfile.mkstemp( prefix=f".{os.path.basename(path)}.odysseus-backup-", dir=directory, ) os.close(fd) os.unlink(backup_path) os.replace(path, backup_path) backups.append((path, backup_path)) else: backups.append((path, None)) staged_by_path = dict(staged) for kind, path, _old, _new in prepared: if kind != "delete": os.replace(staged_by_path[path], path) staged.clear() except BaseException: for path, backup_path in reversed(backups): try: if os.path.exists(path): os.unlink(path) if backup_path and os.path.exists(backup_path): os.replace(backup_path, path) except OSError: pass raise finally: for _path, temp_path in staged: try: os.unlink(temp_path) except OSError: pass for _path, backup_path in backups: if backup_path: try: os.unlink(backup_path) except OSError: pass diffs = [] for _kind, path, old, new in prepared: diff = _unified_diff(old, new, path) if diff: diffs.append(diff) except (ValueError, UnicodeDecodeError, PermissionError, OSError) as e: return {"error": f"apply_patch: {e}", "exit_code": 1} added = sum(int(d.get("added") or 0) for d in diffs) removed = sum(int(d.get("removed") or 0) for d in diffs) text_parts = [d.get("text", "") for d in diffs if d.get("text")] diff_text = "\n".join(text_parts) if len(diff_text.splitlines()) > MAX_DIFF_LINES: diff_text = "\n".join(diff_text.splitlines()[:MAX_DIFF_LINES]) + f"\n... diff truncated at {MAX_DIFF_LINES} lines" result = { "output": f"Applied patch ({len(prepared)} file{'s' if len(prepared) != 1 else ''}, +{added}/-{removed})", "exit_code": 0, } if diffs: result["diff"] = { "text": diff_text, "added": added, "removed": removed, "new_file": any(d.get("new_file") for d in diffs), "file": "patch", } return result def _parse_agent_patch(patch_text: str) -> List[Dict[str, Any]]: lines = patch_text.replace("\r\n", "\n").replace("\r", "\n").split("\n") while lines and not lines[0].strip(): lines.pop(0) while lines and not lines[-1].strip(): lines.pop() if not lines or lines[0].strip() != "*** Begin Patch": raise ValueError("patch must start with *** Begin Patch") if lines[-1].strip() != "*** End Patch": raise ValueError("patch must end with *** End Patch") ops: List[Dict[str, Any]] = [] i = 1 while i < len(lines) - 1: line = lines[i] if not line: i += 1 continue if line.startswith("*** Add File: "): path = line[len("*** Add File: "):].strip() body = [] i += 1 while i < len(lines) - 1 and not lines[i].startswith("*** "): if not lines[i].startswith("+"): raise ValueError(f"add file {path}: every content line must start with +") body.append(lines[i][1:]) i += 1 ops.append({"kind": "add", "path": path, "content": "\n".join(body) + ("\n" if body else "")}) continue if line.startswith("*** Delete File: "): path = line[len("*** Delete File: "):].strip() ops.append({"kind": "delete", "path": path}) i += 1 continue if line.startswith("*** Update File: "): path = line[len("*** Update File: "):].strip() hunks = [] current = [] i += 1 if i < len(lines) - 1 and lines[i].startswith("*** Move to: "): raise ValueError("move operations are not supported") while i < len(lines) - 1 and not lines[i].startswith("*** "): if lines[i].startswith("@@"): if current: hunks.append(current) current = [] elif lines[i].startswith((" ", "-", "+")): current.append(lines[i]) elif lines[i] == "": current.append(" ") else: raise ValueError(f"update file {path}: invalid patch line {lines[i]!r}") i += 1 if current: hunks.append(current) if not hunks: raise ValueError(f"update file {path}: no hunks") ops.append({"kind": "update", "path": path, "hunks": hunks}) continue raise ValueError(f"unexpected patch line: {line!r}") return ops def _apply_patch_hunks(original: str, hunks: List[List[str]], label: str) -> str: updated = original for idx, hunk in enumerate(hunks, 1): old_lines = [] new_lines = [] for line in hunk: prefix, body = line[:1], line[1:] if prefix in (" ", "-"): old_lines.append(body) if prefix in (" ", "+"): new_lines.append(body) old_text = "\n".join(old_lines) new_text = "\n".join(new_lines) if old_text and old_text in updated: occurrences = updated.count(old_text) if occurrences != 1: raise ValueError(f"{label}: hunk {idx} context matched {occurrences} times") updated = updated.replace(old_text, new_text, 1) elif old_text + "\n" in updated: occurrences = updated.count(old_text + "\n") if occurrences != 1: raise ValueError(f"{label}: hunk {idx} context matched {occurrences} times") updated = updated.replace(old_text + "\n", new_text + "\n", 1) else: raise ValueError(f"{label}: hunk {idx} context not found") return updated class LsTool: async def execute(self, content: str, ctx: dict) -> dict: from src.tool_execution import _display_tool_path, _resolve_search_root, _truncate raw_path = "" _s = (content or "").strip() if _s.startswith("{"): try: raw_path = str(json.loads(_s).get("path", "")).strip() except json.JSONDecodeError: raw_path = "" else: raw_path = _s.split("\n", 1)[0].strip() try: root = _resolve_search_root(raw_path) except ValueError as e: return {"error": f"ls: {e}", "exit_code": 1} def _ls(): if not os.path.isdir(root): return None, f"ls: {root}: not a directory" rows = [] try: with os.scandir(root) as it: for entry in it: if entry.name.startswith("."): continue if not _visible_bound_resource(entry.path): continue try: is_dir = entry.is_dir(follow_symlinks=False) size = entry.stat(follow_symlinks=False).st_size if not is_dir else 0 except OSError: continue rows.append((is_dir, entry.name, size)) except (PermissionError, OSError) as _e: return None, f"ls: {_e}" rows.sort(key=lambda r: (not r[0], r[1].lower())) lines = [f"{_display_tool_path(root)}:"] for is_dir, name, size in rows[:_CODENAV_MAX_HITS]: lines.append(f" {name}/" if is_dir else f" {name} ({size} B)") if len(rows) > _CODENAV_MAX_HITS: lines.append(f" ... [{len(rows) - _CODENAV_MAX_HITS} more]") if not rows: lines.append(" (empty)") return "\n".join(lines), None out, err = await asyncio.to_thread(_ls) if err: return {"error": err, "exit_code": 1} return {"output": _truncate(out), "exit_code": 0} class GlobTool: async def execute(self, content: str, ctx: dict) -> dict: from src.tool_execution import ( _SENSITIVE_BASENAMES, _display_tool_path, _is_sensitive_path, _resolve_tool_path, _resolve_search_root, _truncate, ) args = {} _s = (content or "").strip() if _s.startswith("{"): try: args = json.loads(_s) except json.JSONDecodeError: args = {} else: args = {"pattern": _s} pattern = str(args.get("pattern", "")).strip() if not pattern: return {"error": "glob: pattern is required", "exit_code": 1} try: root = _resolve_search_root(str(args.get("path", ""))) except ValueError as e: return {"error": f"glob: {e}", "exit_code": 1} def _glob(): base = os.path.abspath(root) if not os.path.isdir(base): return None, f"glob: {root}: not a directory" rbase = os.path.realpath(base) norm_pat = pattern.replace("\\", "/") # Fast path: literal pattern (no wildcards) → direct path lookup. if not any(c in norm_pat for c in "*?["): cand = os.path.realpath(os.path.join(base, norm_pat)) # Keep the literal lookup inside the search root. os.path.join # lets an absolute pattern (or one containing ../) escape `base`, # which would turn glob into an existence/path oracle for # arbitrary host files — bypassing the workspace/allowlist # confinement that _resolve_search_root applies to the root. # An escaping literal falls through to the walk, which only ever # yields paths under base. inside = is_inside(rbase, cand) # A literal that names a deny-listed sensitive file (.env, # .ssh/id_rsa, …) falls through to the walk, which skips it — # otherwise glob would surface secret paths that read_file / # grep already refuse to touch. if inside and os.path.exists(cand) and not _is_sensitive_path(cand) and _visible_bound_resource(cand): return [cand], None # Literal not at exact path — fall through to walk so # e.g. "foo.py" still matches at any depth (like rglob). # Compile glob to regex: * stays within one segment, **/ spans dirs. regex = _glob_to_regex(norm_pat) matched = [] cap = _CODENAV_MAX_HITS * 5 try: for dp, dns, fns in os.walk(base): # Prune skipped dirs before descending (unlike rglob which # descends first then filters — fatal on large node_modules). # Sensitive dirs (.ssh, .gnupg, …) are pruned too so glob # never enumerates the keys/tokens inside them. dns[:] = [ d for d in dns if d not in _CODENAV_SKIP_DIRS and d not in _SENSITIVE_BASENAMES ] for name in fns + dns: full = os.path.join(dp, name) rel = os.path.relpath(full, base).replace(os.sep, "/") if regex.fullmatch(rel) or regex.fullmatch(name): # Skip deny-listed sensitive files (.env, id_rsa, # known_hosts, …) the same way grep does. if _is_sensitive_path(os.path.realpath(full)) or not _visible_bound_resource(full): continue try: mtime = os.stat(full).st_mtime except OSError: mtime = 0 matched.append((mtime, full)) if len(matched) > cap: break except OSError as _e: return None, f"glob: {_e}" matched.sort(key=lambda t: t[0], reverse=True) return [pth for _, pth in matched[:_CODENAV_MAX_HITS]], None paths, err = await asyncio.to_thread(_glob) if err: return {"error": err, "exit_code": 1} if not paths: return {"output": f"No files matching {pattern!r} under {_display_tool_path(root)}", "exit_code": 0} out = "\n".join(_display_tool_path(path) for path in paths) if len(paths) >= _CODENAV_MAX_HITS: out += f"\n... [capped at {_CODENAV_MAX_HITS} files]" return {"output": _truncate(out), "exit_code": 0} class GrepTool: async def execute(self, content: str, ctx: dict) -> dict: from src.tool_execution import ( _SENSITIVE_FILE_PATTERNS, _display_tool_path, _is_sensitive_path, _resolve_tool_path, _resolve_search_root, _truncate, ) args: Dict[str, Any] = {} _s = (content or "").strip() if _s.startswith("{"): try: args = json.loads(_s) except json.JSONDecodeError: args = {} else: args = {"pattern": _s} pattern = str(args.get("pattern", "")).strip() if not pattern: return {"error": "grep: pattern is required", "exit_code": 1} ignore_case = bool(args.get("ignore_case")) glob_pat = str(args.get("glob", "") or "").strip() try: max_hits = int(args.get("max_results") or _CODENAV_MAX_HITS) except (TypeError, ValueError): max_hits = _CODENAV_MAX_HITS max_hits = max(1, min(max_hits, _CODENAV_MAX_HITS)) try: root = _resolve_search_root(str(args.get("path", ""))) except ValueError as e: return {"error": f"grep: {e}", "exit_code": 1} def _grep(): import re as _re import shutil from src.agent_runtime.resource_binding import active_resource_operation if not os.path.exists(root): return None, f"grep: search target not found: {_display_tool_path(root)}" # The pathname-only fast path scans before individual resources can # be checked. Bound searches must validate every file before read. rg = None if active_resource_operation() is not None else shutil.which("rg") if rg: cmd = [rg, "--line-number", "--with-filename", "--no-heading", "--color=never", "--max-count", str(max_hits)] if ignore_case: cmd.append("--ignore-case") if glob_pat: cmd += ["--glob", glob_pat] # --iglob (not --glob) so the exclusion is case-insensitive: # on a case-insensitive filesystem "ID_RSA"/"Known_Hosts" # resolve to the same secret as their lowercase forms, and the # Python fallback below already folds case via _is_sensitive_path. for _pat in _SENSITIVE_FILE_PATTERNS: cmd += ["--iglob", f"!*{_pat}*"] for _d in _CODENAV_SKIP_DIRS: cmd += ["--glob", f"!**/{_d}/**"] cmd += ["--regexp", pattern, root] try: import subprocess p = subprocess.run(cmd, capture_output=True, text=True, timeout=20) # ripgrep: 0 = matches, 1 = no matches, 2 = failed scan. # Do not present invalid patterns or IO failures as absence. if p.returncode not in (0, 1): detail = (p.stderr or '').strip()[:1200] return None, f"grep: search failed (exit {p.returncode}): {detail or 'no diagnostic available'}" lines = [ln for ln in (p.stdout or "").splitlines() if ln][:max_hits] return lines, None except subprocess.TimeoutExpired: return None, "grep: timed out" except Exception as _e: return None, f"grep: {_e}" try: rx = _re.compile(pattern, _re.IGNORECASE if ignore_case else 0) except _re.error as _e: return None, f"grep: bad pattern: {_e}" hits = [] scan_errors = [] if os.path.isfile(root): file_iter = [root] else: file_iter = [] for dp, dns, fns in os.walk(root, onerror=scan_errors.append): dns[:] = [d for d in dns if d not in _CODENAV_SKIP_DIRS] for fn in fns: if glob_pat and not fnmatch.fnmatch(fn, glob_pat): continue file_iter.append(os.path.join(dp, fn)) for fp in file_iter: if len(hits) >= max_hits: break try: resolved_file = _resolve_tool_path(fp) except ValueError: # Apply the same workspace/sensitive-path checks to each # discovered file, not just the initial search directory. continue try: with open(resolved_file, "r", encoding="utf-8", errors="strict") as f: for i, line in enumerate(f, 1): if rx.search(line): hits.append(f"{fp}:{i}:{line.rstrip()[:_CODENAV_MAX_LINE]}") if len(hits) >= max_hits: break except UnicodeDecodeError: continue except OSError as error: scan_errors.append(error) if scan_errors: return None, "grep: search incomplete; one or more files or directories could not be read" return hits, None lines, err = await asyncio.to_thread(_grep) if err: return {"error": err, "exit_code": 1} if not lines: return {"output": f"No matches for {pattern!r} under {_display_tool_path(root)}", "exit_code": 0} physical_root = os.path.realpath(root) display_root = _display_tool_path(physical_root) out = "\n".join( (display_root + ln[len(physical_root):] if ln.startswith(physical_root) else ln)[:_CODENAV_MAX_LINE] for ln in lines ) if len(lines) >= max_hits: out += f"\n... [capped at {max_hits} matches]" return {"output": _truncate(out), "exit_code": 0} class GetWorkspaceTool: """Report the active workspace folder (no args). File tools are confined to it; the shell starts there (cwd) but is NOT sandboxed.""" async def execute(self, content: str, ctx: dict) -> dict: from src.tool_execution import get_active_workspace ws = get_active_workspace() if ws: return { "output": "/workspace\n(File tools are confined to this folder; the shell starts " f"here but is not sandboxed and can reach outside it.)", "exit_code": 0, } return { "output": "No workspace is set. File tools use the default allowed roots; " "resolve paths from the user or use absolute paths.", "exit_code": 0, }