mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-09-10 10:12:20 +02:00
Merge commit from fork
* fix(security): keep agent file tools out of the app state directory
The agent's read tools (read_file, grep, glob, ls) resolved model-supplied
paths against a root list whose first entry was the whole data directory.
That directory holds the session store, the auth database, the app
encryption key and the settings file, so prompt-injected content could ask
for any of them. No approval prompt stood in the way: reads are classified
read_workspace and pass the untrusted-context gate untouched, which is
correct for reading a workspace and wrong for reading the app's own state.
The agent gets data/agent_workspace/ instead, and the subprocess cwd and
HOME move with it so bash and read_file agree on where scratch files live.
The deny itself is a property of the path, not of the root it arrived
through, because three routes reach the same bytes and closing only the
first leaves the other two working:
- the default root list
- a workspace bound at or above the data directory, which vet_workspace
accepted and chat_routes auto-binds from a path named in the message
- a tool_path_extra_roots setting covering the data directory
_resolve_search_root also returned the workspace root unchecked when the
path was empty, so a bare ls enumerated the directory whatever the deny
list said. It now resolves that case through the same guards.
A containment rule rather than a filename deny list, so state files added
later are covered without anyone remembering to list them, and so a user's
own settings.json or app.db inside a real workspace is not caught.
Four directories of user content stay readable, because the application
hands their paths to the model and tells it to open them: the chat upload
manifest, downloaded mail attachments, personal docs (which covers the
runbook) and personal uploads.
* fix: enforce state deny during recursive file search
* fix: bound protected filesystem searches
* fix(security): reject inode aliases and workspace redirects
* fix(security): harden partitioned agent searches
* fix(security): report fallback worker exits promptly
* fix(security): clean up search readers and retain relative data roots
---------
Co-authored-by: RaresKeY <158580472+RaresKeY@users.noreply.github.com>
This commit is contained in:
@@ -14,6 +14,13 @@ import threading
|
|||||||
import time
|
import time
|
||||||
import webbrowser
|
import webbrowser
|
||||||
|
|
||||||
|
# PyInstaller multiprocessing children re-enter this executable with a private
|
||||||
|
# bootstrap argument. Consume it before splash/UI or application imports so a
|
||||||
|
# spawn-based worker does not relaunch the full desktop application.
|
||||||
|
if __name__ == "__main__":
|
||||||
|
import multiprocessing
|
||||||
|
multiprocessing.freeze_support()
|
||||||
|
|
||||||
# Define a dummy NullWriter to suppress standard stream crashes (isatty etc.) in GUI mode
|
# Define a dummy NullWriter to suppress standard stream crashes (isatty etc.) in GUI mode
|
||||||
class NullWriter:
|
class NullWriter:
|
||||||
def write(self, text):
|
def write(self, text):
|
||||||
|
|||||||
@@ -16,7 +16,7 @@ sys.path.insert(0, BASE_DIR)
|
|||||||
from src.constants import (
|
from src.constants import (
|
||||||
DATA_DIR, AUTH_FILE, UPLOAD_DIR, PERSONAL_DIR, PERSONAL_UPLOADS_DIR,
|
DATA_DIR, AUTH_FILE, UPLOAD_DIR, PERSONAL_DIR, PERSONAL_UPLOADS_DIR,
|
||||||
TTS_CACHE_DIR, GENERATED_IMAGES_DIR, DEEP_RESEARCH_DIR, CHROMA_DIR,
|
TTS_CACHE_DIR, GENERATED_IMAGES_DIR, DEEP_RESEARCH_DIR, CHROMA_DIR,
|
||||||
RAG_DIR, MEMORY_VECTORS_DIR, PASSWORD_MIN_LENGTH,
|
RAG_DIR, MEMORY_VECTORS_DIR, AGENT_WORKSPACE_DIR, PASSWORD_MIN_LENGTH,
|
||||||
)
|
)
|
||||||
from core.auth import RESERVED_USERNAMES
|
from core.auth import RESERVED_USERNAMES
|
||||||
|
|
||||||
@@ -31,6 +31,7 @@ DIRS = [
|
|||||||
CHROMA_DIR,
|
CHROMA_DIR,
|
||||||
RAG_DIR,
|
RAG_DIR,
|
||||||
MEMORY_VECTORS_DIR,
|
MEMORY_VECTORS_DIR,
|
||||||
|
AGENT_WORKSPACE_DIR,
|
||||||
os.path.join(BASE_DIR, "logs"),
|
os.path.join(BASE_DIR, "logs"),
|
||||||
]
|
]
|
||||||
|
|
||||||
|
|||||||
@@ -3,8 +3,8 @@ import json
|
|||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
import difflib
|
import difflib
|
||||||
import fnmatch
|
|
||||||
import shutil
|
import shutil
|
||||||
|
import time
|
||||||
from typing import Optional, Dict, Any, Tuple, List
|
from typing import Optional, Dict, Any, Tuple, List
|
||||||
|
|
||||||
from src.constants import MAX_READ_CHARS, MAX_DIFF_LINES, MAX_OUTPUT_CHARS
|
from src.constants import MAX_READ_CHARS, MAX_DIFF_LINES, MAX_OUTPUT_CHARS
|
||||||
@@ -16,6 +16,8 @@ _CODENAV_SKIP_DIRS = frozenset({
|
|||||||
})
|
})
|
||||||
_CODENAV_MAX_HITS = 200
|
_CODENAV_MAX_HITS = 200
|
||||||
_CODENAV_MAX_LINE = 400
|
_CODENAV_MAX_LINE = 400
|
||||||
|
_GREP_TIMEOUT_SECONDS = 20
|
||||||
|
_GREP_STDERR_PREFIX = 20_000
|
||||||
|
|
||||||
|
|
||||||
def _glob_to_regex(pat: str) -> "re.Pattern":
|
def _glob_to_regex(pat: str) -> "re.Pattern":
|
||||||
@@ -42,6 +44,113 @@ def _glob_to_regex(pat: str) -> "re.Pattern":
|
|||||||
i += 1
|
i += 1
|
||||||
return re.compile("".join(out))
|
return re.compile("".join(out))
|
||||||
|
|
||||||
|
|
||||||
|
def _python_grep_worker(payload: dict, output_queue) -> None:
|
||||||
|
"""Spawn-safe fallback grep worker used when ripgrep is unavailable.
|
||||||
|
|
||||||
|
Keep this at module scope: a frozen Windows executable cannot safely be
|
||||||
|
relaunched as ``sys.executable -c ...``, while multiprocessing can invoke a
|
||||||
|
top-level target through its frozen-process bootstrap.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
flags = re.IGNORECASE if payload["ignore_case"] else 0
|
||||||
|
try:
|
||||||
|
regex = re.compile(payload["pattern"], flags)
|
||||||
|
glob_regex = (
|
||||||
|
_glob_to_regex(payload["glob"].replace("\\", "/"))
|
||||||
|
if payload["glob"]
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
except re.error as exc:
|
||||||
|
output_queue.put(("error", f"grep: bad pattern: {exc}"))
|
||||||
|
return
|
||||||
|
|
||||||
|
requested_root = payload["root"]
|
||||||
|
skip_dirs = set(payload["skip_dirs"])
|
||||||
|
sensitive = {name.casefold() for name in payload["sensitive_names"]}
|
||||||
|
max_hits = payload["max_hits"]
|
||||||
|
hits = 0
|
||||||
|
|
||||||
|
def within(path: str, root: str) -> bool:
|
||||||
|
try:
|
||||||
|
return os.path.commonpath(
|
||||||
|
[os.path.normcase(path), os.path.normcase(root)]
|
||||||
|
) == os.path.normcase(root)
|
||||||
|
except ValueError:
|
||||||
|
return False
|
||||||
|
|
||||||
|
def safe_file(path: str, target: str) -> Optional[str]:
|
||||||
|
if os.path.islink(path):
|
||||||
|
return None
|
||||||
|
canonical = os.path.realpath(path)
|
||||||
|
if not within(canonical, requested_root) or not within(canonical, target):
|
||||||
|
return None
|
||||||
|
parts = [part.casefold() for part in canonical.split(os.sep)]
|
||||||
|
if any(part in sensitive for part in parts):
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
if not os.path.isfile(canonical) or os.stat(canonical).st_nlink > 1:
|
||||||
|
return None
|
||||||
|
except OSError:
|
||||||
|
return None
|
||||||
|
return canonical
|
||||||
|
|
||||||
|
for target in payload["targets"]:
|
||||||
|
if hits >= max_hits:
|
||||||
|
break
|
||||||
|
if os.path.isfile(target):
|
||||||
|
file_iter = iter((target,))
|
||||||
|
else:
|
||||||
|
def walk_files():
|
||||||
|
for directory, dirnames, filenames in os.walk(
|
||||||
|
target, followlinks=False
|
||||||
|
):
|
||||||
|
dirnames[:] = [
|
||||||
|
name
|
||||||
|
for name in dirnames
|
||||||
|
if name not in skip_dirs
|
||||||
|
and name.casefold() not in sensitive
|
||||||
|
and not os.path.islink(os.path.join(directory, name))
|
||||||
|
]
|
||||||
|
for name in filenames:
|
||||||
|
yield os.path.join(directory, name)
|
||||||
|
|
||||||
|
file_iter = walk_files()
|
||||||
|
|
||||||
|
for candidate in file_iter:
|
||||||
|
path = safe_file(candidate, target)
|
||||||
|
if path is None:
|
||||||
|
continue
|
||||||
|
relative = os.path.relpath(path, requested_root).replace(os.sep, "/")
|
||||||
|
if glob_regex and not (
|
||||||
|
glob_regex.fullmatch(relative)
|
||||||
|
or glob_regex.fullmatch(os.path.basename(path))
|
||||||
|
):
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
with open(path, "r", encoding="utf-8", errors="strict") as handle:
|
||||||
|
for number, line in enumerate(handle, 1):
|
||||||
|
if regex.search(line):
|
||||||
|
output_queue.put((
|
||||||
|
"match",
|
||||||
|
path,
|
||||||
|
number,
|
||||||
|
line.rstrip()[:_CODENAV_MAX_LINE],
|
||||||
|
))
|
||||||
|
hits += 1
|
||||||
|
if hits >= max_hits:
|
||||||
|
break
|
||||||
|
except (UnicodeDecodeError, OSError):
|
||||||
|
continue
|
||||||
|
if hits >= max_hits:
|
||||||
|
break
|
||||||
|
output_queue.put(("done",))
|
||||||
|
except BaseException as exc:
|
||||||
|
try:
|
||||||
|
output_queue.put(("error", f"grep: fallback worker failed: {exc}"))
|
||||||
|
except BaseException:
|
||||||
|
pass
|
||||||
|
|
||||||
def _unified_diff(old: str, new: str, path: str) -> Optional[Dict[str, Any]]:
|
def _unified_diff(old: str, new: str, path: str) -> Optional[Dict[str, Any]]:
|
||||||
if old == new:
|
if old == new:
|
||||||
return None
|
return None
|
||||||
@@ -407,7 +516,11 @@ def _apply_patch_hunks(original: str, hunks: List[List[str]], label: str) -> str
|
|||||||
|
|
||||||
class LsTool:
|
class LsTool:
|
||||||
async def execute(self, content: str, ctx: dict) -> dict:
|
async def execute(self, content: str, ctx: dict) -> dict:
|
||||||
from src.tool_execution import _resolve_tool_path, _resolve_search_root, _truncate
|
from src.tool_execution import (
|
||||||
|
_is_denied_tool_path,
|
||||||
|
_resolve_search_root,
|
||||||
|
_truncate,
|
||||||
|
)
|
||||||
raw_path = ""
|
raw_path = ""
|
||||||
_s = (content or "").strip()
|
_s = (content or "").strip()
|
||||||
if _s.startswith("{"):
|
if _s.startswith("{"):
|
||||||
@@ -431,6 +544,8 @@ class LsTool:
|
|||||||
for entry in it:
|
for entry in it:
|
||||||
if entry.name.startswith("."):
|
if entry.name.startswith("."):
|
||||||
continue
|
continue
|
||||||
|
if _is_denied_tool_path(os.path.realpath(entry.path)):
|
||||||
|
continue
|
||||||
try:
|
try:
|
||||||
is_dir = entry.is_dir(follow_symlinks=False)
|
is_dir = entry.is_dir(follow_symlinks=False)
|
||||||
size = entry.stat(follow_symlinks=False).st_size if not is_dir else 0
|
size = entry.stat(follow_symlinks=False).st_size if not is_dir else 0
|
||||||
@@ -458,7 +573,8 @@ class GlobTool:
|
|||||||
async def execute(self, content: str, ctx: dict) -> dict:
|
async def execute(self, content: str, ctx: dict) -> dict:
|
||||||
from src.tool_execution import (
|
from src.tool_execution import (
|
||||||
_SENSITIVE_BASENAMES,
|
_SENSITIVE_BASENAMES,
|
||||||
_is_sensitive_path,
|
_can_traverse_tool_path,
|
||||||
|
_is_denied_tool_path,
|
||||||
_resolve_tool_path,
|
_resolve_tool_path,
|
||||||
_resolve_search_root,
|
_resolve_search_root,
|
||||||
_truncate,
|
_truncate,
|
||||||
@@ -507,7 +623,7 @@ class GlobTool:
|
|||||||
# .ssh/id_rsa, …) falls through to the walk, which skips it —
|
# .ssh/id_rsa, …) falls through to the walk, which skips it —
|
||||||
# otherwise glob would surface secret paths that read_file /
|
# otherwise glob would surface secret paths that read_file /
|
||||||
# grep already refuse to touch.
|
# grep already refuse to touch.
|
||||||
if inside and os.path.exists(cand) and not _is_sensitive_path(cand):
|
if inside and os.path.exists(cand) and not _is_denied_tool_path(cand):
|
||||||
return [cand], None
|
return [cand], None
|
||||||
# Literal not at exact path — fall through to walk so
|
# Literal not at exact path — fall through to walk so
|
||||||
# e.g. "foo.py" still matches at any depth (like rglob).
|
# e.g. "foo.py" still matches at any depth (like rglob).
|
||||||
@@ -517,13 +633,18 @@ class GlobTool:
|
|||||||
cap = _CODENAV_MAX_HITS * 5
|
cap = _CODENAV_MAX_HITS * 5
|
||||||
try:
|
try:
|
||||||
for dp, dns, fns in os.walk(base):
|
for dp, dns, fns in os.walk(base):
|
||||||
|
if not _can_traverse_tool_path(os.path.realpath(dp)):
|
||||||
|
dns[:] = []
|
||||||
|
continue
|
||||||
# Prune skipped dirs before descending (unlike rglob which
|
# Prune skipped dirs before descending (unlike rglob which
|
||||||
# descends first then filters — fatal on large node_modules).
|
# descends first then filters — fatal on large node_modules).
|
||||||
# Sensitive dirs (.ssh, .gnupg, …) are pruned too so glob
|
# Sensitive dirs (.ssh, .gnupg, …) are pruned too so glob
|
||||||
# never enumerates the keys/tokens inside them.
|
# never enumerates the keys/tokens inside them.
|
||||||
dns[:] = [
|
dns[:] = [
|
||||||
d for d in dns
|
d for d in dns
|
||||||
if d not in _CODENAV_SKIP_DIRS and d not in _SENSITIVE_BASENAMES
|
if d not in _CODENAV_SKIP_DIRS
|
||||||
|
and d not in _SENSITIVE_BASENAMES
|
||||||
|
and _can_traverse_tool_path(os.path.realpath(os.path.join(dp, d)))
|
||||||
]
|
]
|
||||||
for name in fns + dns:
|
for name in fns + dns:
|
||||||
full = os.path.join(dp, name)
|
full = os.path.join(dp, name)
|
||||||
@@ -531,7 +652,7 @@ class GlobTool:
|
|||||||
if regex.fullmatch(rel) or regex.fullmatch(name):
|
if regex.fullmatch(rel) or regex.fullmatch(name):
|
||||||
# Skip deny-listed sensitive files (.env, id_rsa,
|
# Skip deny-listed sensitive files (.env, id_rsa,
|
||||||
# known_hosts, …) the same way grep does.
|
# known_hosts, …) the same way grep does.
|
||||||
if _is_sensitive_path(os.path.realpath(full)):
|
if _is_denied_tool_path(os.path.realpath(full)):
|
||||||
continue
|
continue
|
||||||
try:
|
try:
|
||||||
mtime = os.stat(full).st_mtime
|
mtime = os.stat(full).st_mtime
|
||||||
@@ -558,9 +679,12 @@ class GlobTool:
|
|||||||
class GrepTool:
|
class GrepTool:
|
||||||
async def execute(self, content: str, ctx: dict) -> dict:
|
async def execute(self, content: str, ctx: dict) -> dict:
|
||||||
from src.tool_execution import (
|
from src.tool_execution import (
|
||||||
|
_SENSITIVE_BASENAMES,
|
||||||
_SENSITIVE_FILE_PATTERNS,
|
_SENSITIVE_FILE_PATTERNS,
|
||||||
|
_agent_readable_data_subdirs,
|
||||||
|
_is_denied_tool_path,
|
||||||
_is_sensitive_path,
|
_is_sensitive_path,
|
||||||
_resolve_tool_path,
|
_path_within,
|
||||||
_resolve_search_root,
|
_resolve_search_root,
|
||||||
_truncate,
|
_truncate,
|
||||||
)
|
)
|
||||||
@@ -589,64 +713,307 @@ class GrepTool:
|
|||||||
return {"error": f"grep: {e}", "exit_code": 1}
|
return {"error": f"grep: {e}", "exit_code": 1}
|
||||||
|
|
||||||
def _grep():
|
def _grep():
|
||||||
import re as _re
|
import multiprocessing
|
||||||
import shutil
|
import queue
|
||||||
|
import subprocess
|
||||||
|
import threading
|
||||||
|
|
||||||
|
from src.constants import DATA_DIR
|
||||||
|
|
||||||
rg = shutil.which("rg")
|
rg = shutil.which("rg")
|
||||||
|
real_root = os.path.realpath(root)
|
||||||
|
data_dir = os.path.realpath(DATA_DIR)
|
||||||
|
spans_state = _path_within(data_dir, real_root)
|
||||||
|
|
||||||
|
def is_top_level_safe(path: str, *, partition_generated: bool) -> bool:
|
||||||
|
lexical = os.path.abspath(path)
|
||||||
|
if os.path.islink(lexical):
|
||||||
|
return False
|
||||||
|
canonical = os.path.realpath(lexical)
|
||||||
|
if not _path_within(canonical, real_root):
|
||||||
|
return False
|
||||||
|
if partition_generated and os.path.basename(lexical) in _CODENAV_SKIP_DIRS:
|
||||||
|
return False
|
||||||
|
if _is_sensitive_path(canonical) or _is_denied_tool_path(canonical):
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
def safe_targets() -> tuple[list[str], Optional[str]]:
|
||||||
|
candidates: list[tuple[str, bool]] = []
|
||||||
|
if not spans_state:
|
||||||
|
# Preserve direct-root compatibility: skip-directory policy
|
||||||
|
# prunes descendants, but an explicitly requested allowed
|
||||||
|
# root named node_modules remains searchable.
|
||||||
|
candidates.append((real_root, False))
|
||||||
|
else:
|
||||||
|
current = real_root
|
||||||
|
if current != data_dir:
|
||||||
|
for part in os.path.relpath(data_dir, current).split(os.sep):
|
||||||
|
try:
|
||||||
|
with os.scandir(current) as entries:
|
||||||
|
for entry in entries:
|
||||||
|
if entry.name != part:
|
||||||
|
# Reject a sibling link lexically before
|
||||||
|
# canonicalizing or treating it as a target.
|
||||||
|
if entry.is_symlink():
|
||||||
|
continue
|
||||||
|
candidates.append((entry.path, True))
|
||||||
|
except OSError as exc:
|
||||||
|
return [], f"grep: {exc}"
|
||||||
|
current = os.path.join(current, part)
|
||||||
|
for readable in _agent_readable_data_subdirs():
|
||||||
|
if (
|
||||||
|
_path_within(readable, data_dir)
|
||||||
|
and _path_within(readable, real_root)
|
||||||
|
and os.path.exists(readable)
|
||||||
|
):
|
||||||
|
candidates.append((readable, True))
|
||||||
|
|
||||||
|
targets: list[str] = []
|
||||||
|
seen: set[str] = set()
|
||||||
|
for candidate, partition_generated in candidates:
|
||||||
|
if not is_top_level_safe(
|
||||||
|
candidate, partition_generated=partition_generated
|
||||||
|
):
|
||||||
|
continue
|
||||||
|
canonical = os.path.realpath(candidate)
|
||||||
|
if canonical not in seen:
|
||||||
|
seen.add(canonical)
|
||||||
|
targets.append(canonical)
|
||||||
|
return targets, None
|
||||||
|
|
||||||
|
targets, target_error = safe_targets()
|
||||||
|
if target_error:
|
||||||
|
return None, target_error
|
||||||
|
|
||||||
|
base = real_root if os.path.isdir(real_root) else os.path.dirname(real_root)
|
||||||
|
deadline = time.monotonic() + _GREP_TIMEOUT_SECONDS
|
||||||
|
lines: list[str] = []
|
||||||
|
|
||||||
|
def parse_rg_result(raw: str) -> Optional[str]:
|
||||||
|
try:
|
||||||
|
record = json.loads(raw)
|
||||||
|
except (TypeError, json.JSONDecodeError):
|
||||||
|
return None
|
||||||
|
if record.get("type") != "match":
|
||||||
|
return None
|
||||||
|
data = record.get("data") or {}
|
||||||
|
path = (data.get("path") or {}).get("text")
|
||||||
|
text_value = (data.get("lines") or {}).get("text")
|
||||||
|
number = data.get("line_number")
|
||||||
|
if not isinstance(path, str) or not isinstance(text_value, str):
|
||||||
|
return None
|
||||||
|
absolute = path if os.path.isabs(path) else os.path.join(base, path)
|
||||||
|
canonical = os.path.realpath(absolute)
|
||||||
|
if not _path_within(canonical, real_root) or _is_denied_tool_path(canonical):
|
||||||
|
return None
|
||||||
|
return f"{os.path.abspath(absolute)}:{number}:{text_value.rstrip()[:_CODENAV_MAX_LINE]}"
|
||||||
|
|
||||||
|
def run_rg(cmd: list[str]) -> Optional[str]:
|
||||||
|
try:
|
||||||
|
process = subprocess.Popen(
|
||||||
|
cmd,
|
||||||
|
cwd=base,
|
||||||
|
stdin=subprocess.DEVNULL,
|
||||||
|
stdout=subprocess.PIPE,
|
||||||
|
stderr=subprocess.PIPE,
|
||||||
|
text=True,
|
||||||
|
bufsize=1,
|
||||||
|
)
|
||||||
|
except Exception as exc:
|
||||||
|
return f"grep: {exc}"
|
||||||
|
output: queue.Queue[Optional[str]] = queue.Queue(maxsize=max_hits + 2)
|
||||||
|
stderr_prefix: list[str] = []
|
||||||
|
stderr_size = 0
|
||||||
|
stop_reader = threading.Event()
|
||||||
|
|
||||||
|
def enqueue_stdout(value: Optional[str]) -> bool:
|
||||||
|
# The consumer stops at the result cap or deadline. Never
|
||||||
|
# leave a producer blocked on its bounded queue afterward.
|
||||||
|
while not stop_reader.is_set():
|
||||||
|
try:
|
||||||
|
output.put(value, timeout=0.05)
|
||||||
|
return True
|
||||||
|
except queue.Full:
|
||||||
|
continue
|
||||||
|
return False
|
||||||
|
|
||||||
|
def read_stdout() -> None:
|
||||||
|
assert process.stdout is not None
|
||||||
|
try:
|
||||||
|
for line in process.stdout:
|
||||||
|
if not enqueue_stdout(line.rstrip("\n")):
|
||||||
|
break
|
||||||
|
finally:
|
||||||
|
enqueue_stdout(None)
|
||||||
|
|
||||||
|
def read_stderr() -> None:
|
||||||
|
nonlocal stderr_size
|
||||||
|
assert process.stderr is not None
|
||||||
|
while True:
|
||||||
|
chunk = process.stderr.read(4096)
|
||||||
|
if not chunk:
|
||||||
|
break
|
||||||
|
if stderr_size < _GREP_STDERR_PREFIX:
|
||||||
|
kept = chunk[:_GREP_STDERR_PREFIX - stderr_size]
|
||||||
|
stderr_prefix.append(kept)
|
||||||
|
stderr_size += len(kept)
|
||||||
|
|
||||||
|
stdout_thread = threading.Thread(target=read_stdout, daemon=True)
|
||||||
|
stderr_thread = threading.Thread(target=read_stderr, daemon=True)
|
||||||
|
stdout_thread.start()
|
||||||
|
stderr_thread.start()
|
||||||
|
timed_out = False
|
||||||
|
capped = False
|
||||||
|
try:
|
||||||
|
while len(lines) < max_hits:
|
||||||
|
remaining = deadline - time.monotonic()
|
||||||
|
if remaining <= 0:
|
||||||
|
timed_out = True
|
||||||
|
break
|
||||||
|
try:
|
||||||
|
raw = output.get(timeout=remaining)
|
||||||
|
except queue.Empty:
|
||||||
|
timed_out = True
|
||||||
|
break
|
||||||
|
if raw is None:
|
||||||
|
break
|
||||||
|
parsed = parse_rg_result(raw)
|
||||||
|
if parsed and parsed not in lines:
|
||||||
|
lines.append(parsed)
|
||||||
|
capped = len(lines) >= max_hits
|
||||||
|
finally:
|
||||||
|
stop_reader.set()
|
||||||
|
if (timed_out or capped) and process.poll() is None:
|
||||||
|
process.terminate()
|
||||||
|
try:
|
||||||
|
remaining = max(0.01, deadline - time.monotonic())
|
||||||
|
return_code = process.wait(timeout=min(1, remaining))
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
process.kill()
|
||||||
|
return_code = process.wait()
|
||||||
|
stdout_thread.join()
|
||||||
|
stderr_thread.join()
|
||||||
|
if timed_out:
|
||||||
|
return "grep: timed out"
|
||||||
|
if not capped and return_code not in (0, 1):
|
||||||
|
detail = "".join(stderr_prefix).strip()
|
||||||
|
return f"grep: {detail or f'process exited {return_code}'}"
|
||||||
|
return None
|
||||||
|
|
||||||
if rg:
|
if rg:
|
||||||
cmd = [rg, "--line-number", "--no-heading", "--color=never",
|
# Validate even when policy filtering leaves no search targets.
|
||||||
"--max-count", str(max_hits)]
|
if not targets:
|
||||||
|
error = run_rg([rg, "--json", "--no-config", "--regexp", pattern])
|
||||||
|
return (None, error) if error else ([], None)
|
||||||
|
relative_targets = [os.path.relpath(target, base) for target in targets]
|
||||||
|
for offset in range(0, len(relative_targets), 128):
|
||||||
|
if len(lines) >= max_hits:
|
||||||
|
break
|
||||||
|
cmd = [
|
||||||
|
rg, "--json", "--no-config", "--no-follow",
|
||||||
|
"--max-count", str(max_hits - len(lines)),
|
||||||
|
"--max-columns", str(_CODENAV_MAX_LINE),
|
||||||
|
"--max-columns-preview",
|
||||||
|
]
|
||||||
if ignore_case:
|
if ignore_case:
|
||||||
cmd.append("--ignore-case")
|
cmd.append("--ignore-case")
|
||||||
if glob_pat:
|
if glob_pat:
|
||||||
cmd += ["--glob", glob_pat]
|
cmd += ["--glob", glob_pat]
|
||||||
# --iglob (not --glob) so the exclusion is case-insensitive:
|
for sensitive_pattern in _SENSITIVE_FILE_PATTERNS:
|
||||||
# on a case-insensitive filesystem "ID_RSA"/"Known_Hosts"
|
cmd += ["--iglob", f"!{sensitive_pattern}"]
|
||||||
# resolve to the same secret as their lowercase forms, and the
|
for skipped_dir in _CODENAV_SKIP_DIRS:
|
||||||
# Python fallback below already folds case via _is_sensitive_path.
|
cmd += ["--glob", f"!**/{skipped_dir}/**"]
|
||||||
for _pat in _SENSITIVE_FILE_PATTERNS:
|
cmd += ["--regexp", pattern, "--", *relative_targets[offset:offset + 128]]
|
||||||
cmd += ["--iglob", f"!*{_pat}*"]
|
error = run_rg(cmd)
|
||||||
for _d in _CODENAV_SKIP_DIRS:
|
if error:
|
||||||
cmd += ["--glob", f"!**/{_d}/**"]
|
return None, error
|
||||||
cmd += ["--regexp", pattern, root]
|
|
||||||
try:
|
|
||||||
import subprocess
|
|
||||||
p = subprocess.run(cmd, capture_output=True, text=True, timeout=20)
|
|
||||||
lines = [ln for ln in (p.stdout or "").splitlines() if ln][:max_hits]
|
|
||||||
return lines, None
|
return lines, None
|
||||||
except subprocess.TimeoutExpired:
|
|
||||||
return None, "grep: timed out"
|
# This runs inside asyncio.to_thread(), so forking would clone a
|
||||||
except Exception as _e:
|
# multithreaded process and can deadlock. Spawn is platform-safe and
|
||||||
return None, f"grep: {_e}"
|
# PyInstaller-compatible via launcher's early freeze_support().
|
||||||
|
payload = {
|
||||||
|
"root": real_root,
|
||||||
|
"targets": targets,
|
||||||
|
"pattern": pattern,
|
||||||
|
"ignore_case": ignore_case,
|
||||||
|
"glob": glob_pat,
|
||||||
|
"max_hits": max_hits,
|
||||||
|
"skip_dirs": tuple(_CODENAV_SKIP_DIRS),
|
||||||
|
"sensitive_names": tuple(
|
||||||
|
set(_SENSITIVE_BASENAMES) | set(_SENSITIVE_FILE_PATTERNS)
|
||||||
|
),
|
||||||
|
}
|
||||||
try:
|
try:
|
||||||
rx = _re.compile(pattern, _re.IGNORECASE if ignore_case else 0)
|
context = multiprocessing.get_context("spawn")
|
||||||
except _re.error as _e:
|
output_queue = context.Queue(maxsize=max_hits + 2)
|
||||||
return None, f"grep: bad pattern: {_e}"
|
worker = context.Process(
|
||||||
hits = []
|
target=_python_grep_worker, args=(payload, output_queue)
|
||||||
if os.path.isfile(root):
|
)
|
||||||
file_iter = [root]
|
worker.start()
|
||||||
else:
|
except Exception as exc:
|
||||||
file_iter = []
|
|
||||||
for dp, dns, fns in os.walk(root):
|
|
||||||
dns[:] = [d for d in dns if d not in _CODENAV_SKIP_DIRS]
|
|
||||||
for fn in fns:
|
|
||||||
if glob_pat and not fnmatch.fnmatch(fn, glob_pat):
|
|
||||||
continue
|
|
||||||
file_iter.append(os.path.join(dp, fn))
|
|
||||||
for fp in file_iter:
|
|
||||||
if len(hits) >= max_hits:
|
|
||||||
break
|
|
||||||
if _is_sensitive_path(os.path.realpath(fp)):
|
|
||||||
continue
|
|
||||||
try:
|
try:
|
||||||
with open(fp, "r", encoding="utf-8", errors="strict") as f:
|
output_queue.close()
|
||||||
for i, line in enumerate(f, 1):
|
except (NameError, OSError, ValueError):
|
||||||
if rx.search(line):
|
pass
|
||||||
hits.append(f"{fp}:{i}:{line.rstrip()[:_CODENAV_MAX_LINE]}")
|
return None, f"grep: could not start fallback worker: {exc}"
|
||||||
if len(hits) >= max_hits:
|
error = None
|
||||||
|
completed = False
|
||||||
|
try:
|
||||||
|
while len(lines) < max_hits:
|
||||||
|
remaining = deadline - time.monotonic()
|
||||||
|
if remaining <= 0:
|
||||||
|
error = "grep: timed out"
|
||||||
break
|
break
|
||||||
except (UnicodeDecodeError, OSError):
|
try:
|
||||||
|
# Keep queue waits short enough to observe a spawn
|
||||||
|
# worker that dies during bootstrap/import before it
|
||||||
|
# can enqueue either an error or the done sentinel.
|
||||||
|
record = output_queue.get(timeout=min(0.05, remaining))
|
||||||
|
except queue.Empty:
|
||||||
|
if worker.is_alive():
|
||||||
continue
|
continue
|
||||||
return hits, None
|
worker.join(timeout=0)
|
||||||
|
try:
|
||||||
|
# A multiprocessing queue's feeder can make the
|
||||||
|
# final record visible at process-exit time. Give
|
||||||
|
# that record precedence over the exit status.
|
||||||
|
remaining = deadline - time.monotonic()
|
||||||
|
record = output_queue.get(
|
||||||
|
timeout=min(0.05, max(0, remaining))
|
||||||
|
)
|
||||||
|
except queue.Empty:
|
||||||
|
error = f"grep: fallback worker exited {worker.exitcode}"
|
||||||
|
break
|
||||||
|
if record[0] == "done":
|
||||||
|
completed = True
|
||||||
|
break
|
||||||
|
if record[0] == "error":
|
||||||
|
error = record[1]
|
||||||
|
break
|
||||||
|
_, path, number, text_value = record
|
||||||
|
canonical = os.path.realpath(path)
|
||||||
|
if not _path_within(canonical, real_root) or _is_denied_tool_path(canonical):
|
||||||
|
continue
|
||||||
|
rendered = f"{path}:{number}:{text_value}"
|
||||||
|
if rendered not in lines:
|
||||||
|
lines.append(rendered)
|
||||||
|
finally:
|
||||||
|
if completed:
|
||||||
|
worker.join(timeout=min(1, max(0.01, deadline - time.monotonic())))
|
||||||
|
if worker.is_alive():
|
||||||
|
worker.terminate()
|
||||||
|
worker.join(timeout=1)
|
||||||
|
if worker.is_alive():
|
||||||
|
worker.kill()
|
||||||
|
worker.join()
|
||||||
|
output_queue.close()
|
||||||
|
if error:
|
||||||
|
return None, error
|
||||||
|
if worker.exitcode not in (0, None) and len(lines) < max_hits:
|
||||||
|
return None, f"grep: fallback worker exited {worker.exitcode}"
|
||||||
|
return lines, None
|
||||||
|
|
||||||
lines, err = await asyncio.to_thread(_grep)
|
lines, err = await asyncio.to_thread(_grep)
|
||||||
if err:
|
if err:
|
||||||
|
|||||||
+30
-1
@@ -2,10 +2,11 @@
|
|||||||
"""Initialize all application components and dependencies."""
|
"""Initialize all application components and dependencies."""
|
||||||
import os
|
import os
|
||||||
import logging
|
import logging
|
||||||
|
import stat
|
||||||
from typing import Dict, Any
|
from typing import Dict, Any
|
||||||
|
|
||||||
from src.constants import (
|
from src.constants import (
|
||||||
DATA_DIR, PERSONAL_DIR, RUNBOOK_DIR, UPLOAD_DIR,
|
DATA_DIR, PERSONAL_DIR, RUNBOOK_DIR, UPLOAD_DIR, AGENT_WORKSPACE_DIR,
|
||||||
SESSIONS_FILE, DEFAULT_HOST, OPENAI_API_KEY
|
SESSIONS_FILE, DEFAULT_HOST, OPENAI_API_KEY
|
||||||
)
|
)
|
||||||
from src.memory import MemoryManager
|
from src.memory import MemoryManager
|
||||||
@@ -31,6 +32,34 @@ def create_directories():
|
|||||||
for directory in (DATA_DIR, PERSONAL_DIR, RUNBOOK_DIR, UPLOAD_DIR):
|
for directory in (DATA_DIR, PERSONAL_DIR, RUNBOOK_DIR, UPLOAD_DIR):
|
||||||
os.makedirs(directory, exist_ok=True)
|
os.makedirs(directory, exist_ok=True)
|
||||||
|
|
||||||
|
# The model-controlled workspace must be a real child of DATA_DIR. Never
|
||||||
|
# follow a pre-existing symlink here: it would silently move the default
|
||||||
|
# native-file root outside the application volume before any resolver runs.
|
||||||
|
data_root = os.path.realpath(os.path.abspath(os.path.expanduser(DATA_DIR)))
|
||||||
|
workspace = os.path.abspath(os.path.expanduser(AGENT_WORKSPACE_DIR))
|
||||||
|
expected_workspace = os.path.join(data_root, "agent_workspace")
|
||||||
|
# Validate the real parent so a supported DATA_DIR bind/symlink works, but
|
||||||
|
# require the fixed internal carve-out name and reject a link at the model-
|
||||||
|
# controlled workspace entry itself.
|
||||||
|
if (
|
||||||
|
os.path.basename(workspace) != "agent_workspace"
|
||||||
|
or os.path.realpath(os.path.dirname(workspace)) != data_root
|
||||||
|
):
|
||||||
|
raise RuntimeError("agent workspace must be the canonical child of DATA_DIR")
|
||||||
|
if os.path.lexists(workspace):
|
||||||
|
mode = os.lstat(workspace).st_mode
|
||||||
|
if stat.S_ISLNK(mode) or not stat.S_ISDIR(mode):
|
||||||
|
raise RuntimeError("agent workspace must be a real directory")
|
||||||
|
else:
|
||||||
|
os.mkdir(workspace, 0o700)
|
||||||
|
resolved_workspace = os.path.realpath(workspace)
|
||||||
|
if resolved_workspace != expected_workspace:
|
||||||
|
raise RuntimeError("agent workspace must be the canonical child of DATA_DIR")
|
||||||
|
try:
|
||||||
|
os.chmod(workspace, 0o700)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
|
||||||
def initialize_managers(base_dir: str, rag_manager=None) -> Dict[str, Any]:
|
def initialize_managers(base_dir: str, rag_manager=None) -> Dict[str, Any]:
|
||||||
"""
|
"""
|
||||||
Initialize all manager and handler instances.
|
Initialize all manager and handler instances.
|
||||||
|
|||||||
@@ -54,6 +54,11 @@ GALLERY_DIR = os.path.join(DATA_DIR, "gallery")
|
|||||||
GALLERY_UPLOADS_DIR = os.path.join(DATA_DIR, "gallery_uploads")
|
GALLERY_UPLOADS_DIR = os.path.join(DATA_DIR, "gallery_uploads")
|
||||||
MEMORY_VECTORS_DIR = os.path.join(DATA_DIR, "memory_vectors")
|
MEMORY_VECTORS_DIR = os.path.join(DATA_DIR, "memory_vectors")
|
||||||
|
|
||||||
|
# The only part of DATA_DIR the agent's file tools and subprocesses may touch.
|
||||||
|
# Everything else under DATA_DIR is application state (session store, auth
|
||||||
|
# database, encryption key, settings), and the agent has no business reading it.
|
||||||
|
AGENT_WORKSPACE_DIR = os.path.join(DATA_DIR, "agent_workspace")
|
||||||
|
|
||||||
# Paths with an intentional dedicated env override, defaulting under DATA_DIR.
|
# Paths with an intentional dedicated env override, defaulting under DATA_DIR.
|
||||||
MAIL_ATTACHMENTS_DIR = os.getenv("ODYSSEUS_MAIL_ATTACHMENTS_DIR", os.path.join(DATA_DIR, "mail-attachments"))
|
MAIL_ATTACHMENTS_DIR = os.getenv("ODYSSEUS_MAIL_ATTACHMENTS_DIR", os.path.join(DATA_DIR, "mail-attachments"))
|
||||||
# `or` (not os.getenv's default arg) so a PRESENT-but-EMPTY value falls back to
|
# `or` (not os.getenv's default arg) so a PRESENT-but-EMPTY value falls back to
|
||||||
|
|||||||
+235
-18
@@ -15,6 +15,7 @@ import logging
|
|||||||
import os
|
import os
|
||||||
import pathlib
|
import pathlib
|
||||||
import re
|
import re
|
||||||
|
import stat
|
||||||
import sys
|
import sys
|
||||||
import time
|
import time
|
||||||
from typing import Any, Awaitable, Callable, Dict, Optional, Tuple
|
from typing import Any, Awaitable, Callable, Dict, Optional, Tuple
|
||||||
@@ -30,7 +31,12 @@ from src.tool_security import (
|
|||||||
from src.tool_capabilities import ToolRunSecurityContext, blocked_tool_result
|
from src.tool_capabilities import ToolRunSecurityContext, blocked_tool_result
|
||||||
from src.tool_approvals import ExactToolApproval
|
from src.tool_approvals import ExactToolApproval
|
||||||
from src.tool_policy import ToolPolicy
|
from src.tool_policy import ToolPolicy
|
||||||
from src.constants import MAX_OUTPUT_CHARS, MAX_READ_CHARS, MAX_DIFF_LINES, DATA_DIR
|
from src.constants import (
|
||||||
|
MAX_OUTPUT_CHARS,
|
||||||
|
MAX_READ_CHARS,
|
||||||
|
MAX_DIFF_LINES,
|
||||||
|
AGENT_WORKSPACE_DIR,
|
||||||
|
)
|
||||||
from src.tool_utils import _truncate, get_mcp_manager
|
from src.tool_utils import _truncate, get_mcp_manager
|
||||||
|
|
||||||
|
|
||||||
@@ -46,11 +52,11 @@ _MISSING_TOOL_SECURITY_CONTEXT = _MissingToolSecurityContext()
|
|||||||
NO_TOOL_SECURITY_CONTEXT = _NoToolSecurityContext()
|
NO_TOOL_SECURITY_CONTEXT = _NoToolSecurityContext()
|
||||||
|
|
||||||
# Persistent working directory for agent subprocesses.
|
# Persistent working directory for agent subprocesses.
|
||||||
# Resolves to <repo_root>/data, which is the bind-mounted volume in Docker
|
# Resolves to <repo_root>/data/agent_workspace, inside the bind-mounted volume
|
||||||
# (/app/data) and the local data directory for manual installs.
|
# in Docker (/app/data), so files survive a rebuild as before. The subdirectory
|
||||||
# Using this as cwd and HOME prevents the agent from silently creating files
|
# rather than data/ itself keeps agent scratch files and dotfiles out of the
|
||||||
# in ephemeral container layers that are lost on the next rebuild.
|
# directory holding the session store and the auth database.
|
||||||
_AGENT_WORKDIR = DATA_DIR
|
_AGENT_WORKDIR = AGENT_WORKSPACE_DIR
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
@@ -66,10 +72,15 @@ _AGENT_WORKDIR = DATA_DIR
|
|||||||
# 1. Sensitive-subpath deny list — checked FIRST. Blocks .ssh,
|
# 1. Sensitive-subpath deny list — checked FIRST. Blocks .ssh,
|
||||||
# .gnupg, shell rc files, token/env files even if the root above
|
# .gnupg, shell rc files, token/env files even if the root above
|
||||||
# them is on the allowlist.
|
# them is on the allowlist.
|
||||||
# 2. Allowlist — only the directories the agent legitimately needs
|
# 2. Application-state deny (_is_app_state_path) - DATA_DIR holds the
|
||||||
# (project data/, system tmp). $HOME is NOT on the default list.
|
# session store, auth database, app key and settings, so only
|
||||||
# 3. Opt-in extra roots — admin can add broader roots via the
|
# _agent_readable_data_subdirs() is readable inside it.
|
||||||
# "tool_path_extra_roots" setting (list of path strings).
|
# 3. Allowlist - only the directories the agent legitimately needs
|
||||||
|
# (its data/ workspace, user content, system tmp). $HOME is NOT on
|
||||||
|
# the default list.
|
||||||
|
# 4. Opt-in extra roots - admin can add broader roots via the
|
||||||
|
# "tool_path_extra_roots" setting. These cannot re-open DATA_DIR;
|
||||||
|
# rule 2 is independent of which root a path arrived through.
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
_SENSITIVE_BASENAMES: set[str] = {
|
_SENSITIVE_BASENAMES: set[str] = {
|
||||||
@@ -116,6 +127,184 @@ def _is_sensitive_path(resolved: str) -> bool:
|
|||||||
return filename in _SENSITIVE_FILE_PATTERNS_CF
|
return filename in _SENSITIVE_FILE_PATTERNS_CF
|
||||||
|
|
||||||
|
|
||||||
|
def _path_within(resolved: str, root: str) -> bool:
|
||||||
|
"""True when *resolved* is *root* itself or sits underneath it.
|
||||||
|
|
||||||
|
Use the platform's path-case rules. This helper participates in allow
|
||||||
|
decisions, so unconditional case-folding would let a distinct ``/DATA``
|
||||||
|
tree masquerade as a descendant of ``/data`` on case-sensitive systems.
|
||||||
|
"""
|
||||||
|
resolved, root = os.path.normcase(resolved), os.path.normcase(root)
|
||||||
|
if resolved == root:
|
||||||
|
return True
|
||||||
|
try:
|
||||||
|
if os.path.commonpath([resolved, root]) == root:
|
||||||
|
return True
|
||||||
|
except ValueError:
|
||||||
|
return False
|
||||||
|
# normcase is intentionally conservative about assumptions (notably on
|
||||||
|
# POSIX), so consult the filesystem when paths exist. This recognizes a
|
||||||
|
# case alias on a case-insensitive volume without treating distinct
|
||||||
|
# case-sensitive paths as the same allow root.
|
||||||
|
if os.path.exists(root):
|
||||||
|
candidate = resolved
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
if os.path.exists(candidate) and os.path.samefile(candidate, root):
|
||||||
|
return True
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
parent = os.path.dirname(candidate)
|
||||||
|
if parent == candidate:
|
||||||
|
break
|
||||||
|
candidate = parent
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def _path_within_conservative(resolved: str, root: str) -> bool:
|
||||||
|
"""Containment for deny decisions, folding case to fail closed."""
|
||||||
|
resolved, root = resolved.casefold(), root.casefold()
|
||||||
|
if resolved == root:
|
||||||
|
return True
|
||||||
|
try:
|
||||||
|
return os.path.commonpath([resolved, root]) == root
|
||||||
|
except ValueError:
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def _agent_readable_data_subdirs() -> tuple[str, ...]:
|
||||||
|
"""The only parts of DATA_DIR the agent's file tools may reach.
|
||||||
|
|
||||||
|
The agent's own scratch folder, plus the directories of user content whose
|
||||||
|
paths the application itself gives to the model, which it would then be
|
||||||
|
unable to open. These normally live under DATA_DIR; the documented mail
|
||||||
|
attachment override may instead name a disjoint external directory:
|
||||||
|
|
||||||
|
UPLOAD_DIR the chat upload manifest renders "path=<p>" and
|
||||||
|
says to read it with read_file (agent_loop.py)
|
||||||
|
MAIL_ATTACHMENTS_DIR download_attachment returns the path and its own
|
||||||
|
description tells the model to read it
|
||||||
|
PERSONAL_DIR GET /api/personal returns a path per file and is
|
||||||
|
reachable through the app_api tool; RUNBOOK_DIR
|
||||||
|
nests under it
|
||||||
|
PERSONAL_UPLOADS_DIR indexed as a personal-docs directory, which
|
||||||
|
manage_rag lists as an absolute path
|
||||||
|
|
||||||
|
Order matters: the first entry is roots[0], which _resolve_search_root uses
|
||||||
|
when grep/glob/ls are called with no path.
|
||||||
|
"""
|
||||||
|
from src.constants import (
|
||||||
|
DATA_DIR,
|
||||||
|
MAIL_ATTACHMENTS_DIR,
|
||||||
|
PERSONAL_DIR,
|
||||||
|
PERSONAL_UPLOADS_DIR,
|
||||||
|
UPLOAD_DIR,
|
||||||
|
)
|
||||||
|
configured = (
|
||||||
|
(AGENT_WORKSPACE_DIR, "agent_workspace", False),
|
||||||
|
(UPLOAD_DIR, "uploads", False),
|
||||||
|
# This has a documented environment override and may legitimately
|
||||||
|
# live outside DATA_DIR, but it must never equal/contain DATA_DIR.
|
||||||
|
(MAIL_ATTACHMENTS_DIR, "mail-attachments", True),
|
||||||
|
(PERSONAL_DIR, "personal_docs", False),
|
||||||
|
(PERSONAL_UPLOADS_DIR, "personal_uploads", False),
|
||||||
|
)
|
||||||
|
configured_data_dir = os.path.abspath(os.path.expanduser(str(DATA_DIR)))
|
||||||
|
data_dir = os.path.realpath(configured_data_dir)
|
||||||
|
safe: list[str] = []
|
||||||
|
for raw, internal_name, external_ok in configured:
|
||||||
|
value = str(raw or "").strip()
|
||||||
|
# These paths are security-policy roots, not ordinary allowlist
|
||||||
|
# entries. Internal roles may inherit a relative DATA_DIR, but must
|
||||||
|
# still resolve to their exact canonical child below. External mail
|
||||||
|
# overrides require an absolute, disjoint directory.
|
||||||
|
if not value:
|
||||||
|
continue
|
||||||
|
expanded = os.path.abspath(os.path.expanduser(value))
|
||||||
|
# A policy root must not acquire an exemption by redirecting its final
|
||||||
|
# path component to protected state or to an unrelated external tree.
|
||||||
|
if os.path.islink(expanded):
|
||||||
|
continue
|
||||||
|
resolved = os.path.realpath(expanded)
|
||||||
|
if os.path.exists(resolved) and not os.path.isdir(resolved):
|
||||||
|
continue
|
||||||
|
expected_internal = os.path.join(data_dir, internal_name)
|
||||||
|
expected_configured = os.path.join(configured_data_dir, internal_name)
|
||||||
|
inside_data = (
|
||||||
|
os.path.normcase(expanded)
|
||||||
|
in {
|
||||||
|
os.path.normcase(expected_configured),
|
||||||
|
os.path.normcase(expected_internal),
|
||||||
|
}
|
||||||
|
and resolved == expected_internal
|
||||||
|
)
|
||||||
|
external_safe = (
|
||||||
|
external_ok
|
||||||
|
and os.path.isabs(os.path.expanduser(value))
|
||||||
|
and resolved != data_dir
|
||||||
|
and os.path.dirname(resolved) != resolved
|
||||||
|
and not _path_within(data_dir, resolved)
|
||||||
|
and not _path_within(resolved, data_dir)
|
||||||
|
)
|
||||||
|
if not (inside_data or external_safe) or _is_sensitive_path(resolved):
|
||||||
|
continue
|
||||||
|
safe.append(resolved)
|
||||||
|
return tuple(safe)
|
||||||
|
|
||||||
|
|
||||||
|
def _is_app_state_path(resolved: str) -> bool:
|
||||||
|
"""True for anything under DATA_DIR that is not agent-readable.
|
||||||
|
|
||||||
|
DATA_DIR holds the session store, the auth database, the app encryption key
|
||||||
|
and the settings file. A model-supplied path must not reach those through
|
||||||
|
any root, so this is checked in both resolvers rather than expressed as an
|
||||||
|
absence from the allowlist: a workspace bound at or above the data
|
||||||
|
directory, or an opt-in tool_path_extra_roots entry covering it, would
|
||||||
|
otherwise put them back in reach.
|
||||||
|
|
||||||
|
A containment rule rather than a filename deny list, so state files added
|
||||||
|
later are covered without anyone remembering to list them, and so a user's
|
||||||
|
own settings.json or app.db inside a real workspace is not caught.
|
||||||
|
"""
|
||||||
|
from src.constants import DATA_DIR
|
||||||
|
if not _path_within_conservative(resolved, os.path.realpath(DATA_DIR)):
|
||||||
|
return False
|
||||||
|
return not any(
|
||||||
|
_path_within(resolved, d)
|
||||||
|
for d in _agent_readable_data_subdirs()
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _is_hardlinked_regular_file(resolved: str) -> bool:
|
||||||
|
"""Reject inode aliases that can smuggle DATA_DIR state into an allow root."""
|
||||||
|
try:
|
||||||
|
target = os.stat(resolved, follow_symlinks=False)
|
||||||
|
except OSError:
|
||||||
|
return False
|
||||||
|
return stat.S_ISREG(target.st_mode) and getattr(target, "st_nlink", 1) > 1
|
||||||
|
|
||||||
|
|
||||||
|
def _is_denied_tool_path(resolved: str) -> bool:
|
||||||
|
"""Apply every path deny to a canonical traversal result."""
|
||||||
|
return (
|
||||||
|
_is_sensitive_path(resolved)
|
||||||
|
or _is_app_state_path(resolved)
|
||||||
|
or _is_hardlinked_regular_file(resolved)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _can_traverse_tool_path(resolved: str) -> bool:
|
||||||
|
"""Allow walking a denied state parent only to reach safe carve-outs."""
|
||||||
|
if _is_sensitive_path(resolved):
|
||||||
|
return False
|
||||||
|
if not _is_app_state_path(resolved):
|
||||||
|
return True
|
||||||
|
return any(
|
||||||
|
_path_within(readable, resolved)
|
||||||
|
for readable in _agent_readable_data_subdirs()
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def _tool_path_roots() -> list[str]:
|
def _tool_path_roots() -> list[str]:
|
||||||
"""Return the list of directory roots that read_file / write_file
|
"""Return the list of directory roots that read_file / write_file
|
||||||
may touch. Default: project data/ + system temp dirs. Extra roots
|
may touch. Default: project data/ + system temp dirs. Extra roots
|
||||||
@@ -123,9 +312,9 @@ def _tool_path_roots() -> list[str]:
|
|||||||
"""
|
"""
|
||||||
roots: list[str] = []
|
roots: list[str] = []
|
||||||
|
|
||||||
# Project data directory — the agent's primary workspace.
|
# The agent's workspace plus the user-content directories inside data/.
|
||||||
from src.constants import DATA_DIR
|
# The rest of DATA_DIR is denied by _is_app_state_path.
|
||||||
roots.append(DATA_DIR)
|
roots.extend(_agent_readable_data_subdirs())
|
||||||
|
|
||||||
# /tmp (and its macOS realpath /private/tmp).
|
# /tmp (and its macOS realpath /private/tmp).
|
||||||
roots.append("/tmp")
|
roots.append("/tmp")
|
||||||
@@ -193,6 +382,12 @@ def _resolve_tool_path(raw_path: str) -> str:
|
|||||||
f"path '{raw_path}' is inside a sensitive directory "
|
f"path '{raw_path}' is inside a sensitive directory "
|
||||||
f"(e.g. .ssh, .gnupg) or matches a sensitive filename"
|
f"(e.g. .ssh, .gnupg) or matches a sensitive filename"
|
||||||
)
|
)
|
||||||
|
if _is_app_state_path(resolved):
|
||||||
|
raise ValueError(
|
||||||
|
f"path '{raw_path}' is inside the application state directory"
|
||||||
|
)
|
||||||
|
if _is_hardlinked_regular_file(resolved):
|
||||||
|
raise ValueError(f"path '{raw_path}' is a hard-linked file")
|
||||||
|
|
||||||
for root in _tool_path_roots():
|
for root in _tool_path_roots():
|
||||||
if resolved == root:
|
if resolved == root:
|
||||||
@@ -228,6 +423,12 @@ def _resolve_tool_path_in_workspace(workspace: str, raw_path: str) -> str:
|
|||||||
f"path '{raw_path}' is inside a sensitive directory "
|
f"path '{raw_path}' is inside a sensitive directory "
|
||||||
f"(e.g. .ssh, .gnupg) or matches a sensitive filename"
|
f"(e.g. .ssh, .gnupg) or matches a sensitive filename"
|
||||||
)
|
)
|
||||||
|
if _is_app_state_path(resolved):
|
||||||
|
raise ValueError(
|
||||||
|
f"path '{raw_path}' is inside the application state directory"
|
||||||
|
)
|
||||||
|
if _is_hardlinked_regular_file(resolved):
|
||||||
|
raise ValueError(f"path '{raw_path}' is a hard-linked file")
|
||||||
if resolved != base:
|
if resolved != base:
|
||||||
# normcase so containment holds on case-insensitive filesystems
|
# normcase so containment holds on case-insensitive filesystems
|
||||||
# (Windows, default macOS): it lowercases on Windows and is a no-op on
|
# (Windows, default macOS): it lowercases on Windows and is a no-op on
|
||||||
@@ -277,6 +478,10 @@ def vet_workspace(raw: str) -> Optional[str]:
|
|||||||
resolved = os.path.realpath(os.path.expanduser(raw))
|
resolved = os.path.realpath(os.path.expanduser(raw))
|
||||||
if not os.path.isdir(resolved) or _is_sensitive_path(resolved):
|
if not os.path.isdir(resolved) or _is_sensitive_path(resolved):
|
||||||
return None
|
return None
|
||||||
|
# Refuse the bind rather than binding a workspace where every subsequent
|
||||||
|
# tool call would fail on the same deny list.
|
||||||
|
if _is_app_state_path(resolved):
|
||||||
|
return None
|
||||||
# Reject filesystem roots: binding / (or a Windows drive/UNC root) as the
|
# Reject filesystem roots: binding / (or a Windows drive/UNC root) as the
|
||||||
# workspace would make every absolute path "inside" it, collapsing the
|
# workspace would make every absolute path "inside" it, collapsing the
|
||||||
# confinement into host-wide file access. A root is its own dirname, which
|
# confinement into host-wide file access. A root is its own dirname, which
|
||||||
@@ -289,7 +494,13 @@ def vet_workspace(raw: str) -> Optional[str]:
|
|||||||
def agent_cwd() -> str:
|
def agent_cwd() -> str:
|
||||||
"""Working directory for agent subprocesses (bash/python/background jobs):
|
"""Working directory for agent subprocesses (bash/python/background jobs):
|
||||||
the active workspace when set, else the persistent data dir."""
|
the active workspace when set, else the persistent data dir."""
|
||||||
return get_active_workspace() or _AGENT_WORKDIR
|
workspace = get_active_workspace()
|
||||||
|
if workspace:
|
||||||
|
return workspace
|
||||||
|
resolved = os.path.realpath(_AGENT_WORKDIR)
|
||||||
|
if resolved not in _agent_readable_data_subdirs():
|
||||||
|
raise RuntimeError("agent workspace is not a safe real directory")
|
||||||
|
return resolved
|
||||||
|
|
||||||
|
|
||||||
def get_mcp_manager():
|
def get_mcp_manager():
|
||||||
@@ -304,16 +515,22 @@ def _resolve_search_root(raw_path: str) -> str:
|
|||||||
|
|
||||||
With a workspace active, the workspace folder is the root and a supplied
|
With a workspace active, the workspace folder is the root and a supplied
|
||||||
path is confined inside it. Otherwise an empty path defaults to the agent's
|
path is confined inside it. Otherwise an empty path defaults to the agent's
|
||||||
primary root (project data dir) and a supplied path is confined by the
|
primary root (its workspace under the project data dir) and a supplied path
|
||||||
global allowlist + sensitive-file policy.
|
is confined by the global allowlist + sensitive-file policy.
|
||||||
"""
|
"""
|
||||||
raw = (raw_path or "").strip()
|
raw = (raw_path or "").strip()
|
||||||
ws = get_active_workspace()
|
ws = get_active_workspace()
|
||||||
if ws:
|
if ws:
|
||||||
return os.path.realpath(ws) if not raw else _resolve_tool_path_in_workspace(ws, raw)
|
# Resolve the empty case as the workspace path rather than returning
|
||||||
|
# it directly: returned unchecked it skipped both deny lists, so a
|
||||||
|
# bare ls listed whatever the workspace was bound to.
|
||||||
|
return _resolve_tool_path_in_workspace(ws, raw or ws)
|
||||||
if not raw:
|
if not raw:
|
||||||
roots = _tool_path_roots()
|
roots = _tool_path_roots()
|
||||||
return roots[0] if roots else os.path.realpath(".")
|
default_root = os.path.realpath(AGENT_WORKSPACE_DIR)
|
||||||
|
if default_root in roots and not _is_denied_tool_path(default_root):
|
||||||
|
return default_root
|
||||||
|
raise ValueError("default agent workspace is not a safe readable data subdirectory")
|
||||||
return _resolve_tool_path(raw)
|
return _resolve_tool_path(raw)
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -91,6 +91,17 @@ def test_grep_python_fallback_when_no_rg(repo, monkeypatch):
|
|||||||
assert ".git/config" not in r["output"]
|
assert ".git/config" not in r["output"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_grep_python_fallback_uses_relative_glob_paths(repo, monkeypatch):
|
||||||
|
monkeypatch.setattr(shutil, "which", lambda name: None)
|
||||||
|
r = _run(
|
||||||
|
"grep",
|
||||||
|
f'{{"pattern": "needle|python", "glob": "**/*.py", "path": "{repo}"}}',
|
||||||
|
)
|
||||||
|
assert r["exit_code"] == 0
|
||||||
|
assert "a.py" in r["output"]
|
||||||
|
assert "sub/deep/c.py" in r["output"]
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.skipif(shutil.which("rg") is None, reason="targets the ripgrep fast-path")
|
@pytest.mark.skipif(shutil.which("rg") is None, reason="targets the ripgrep fast-path")
|
||||||
def test_grep_skips_case_variant_sensitive_files_rg(repo):
|
def test_grep_skips_case_variant_sensitive_files_rg(repo):
|
||||||
"""The rg fast-path must exclude deny-listed key files case-insensitively.
|
"""The rg fast-path must exclude deny-listed key files case-insensitively.
|
||||||
|
|||||||
@@ -1,12 +1,22 @@
|
|||||||
# tests/test_launcher.py
|
# tests/test_launcher.py
|
||||||
import sys
|
import sys
|
||||||
import os
|
import os
|
||||||
|
from pathlib import Path
|
||||||
from unittest import mock
|
from unittest import mock
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from launcher import NullWriter, create_tray_image, on_open_browser, on_exit, open_browser
|
from launcher import NullWriter, create_tray_image, on_open_browser, on_exit, open_browser
|
||||||
|
|
||||||
|
|
||||||
|
def test_frozen_multiprocessing_bootstrap_precedes_gui_and_app_imports():
|
||||||
|
source = Path("launcher.py").read_text(encoding="utf-8")
|
||||||
|
|
||||||
|
freeze = source.index("multiprocessing.freeze_support()")
|
||||||
|
splash = source.index("if getattr(sys, 'frozen', False):")
|
||||||
|
app_import = source.index("from app import app")
|
||||||
|
assert freeze < splash < app_import
|
||||||
|
|
||||||
|
|
||||||
def test_null_writer():
|
def test_null_writer():
|
||||||
writer = NullWriter()
|
writer = NullWriter()
|
||||||
# writing and flushing should not raise any exceptions
|
# writing and flushing should not raise any exceptions
|
||||||
|
|||||||
@@ -161,12 +161,14 @@ def test_blocks_netrc():
|
|||||||
_resolve_tool_path("~/.netrc")
|
_resolve_tool_path("~/.netrc")
|
||||||
|
|
||||||
|
|
||||||
def test_allows_project_data(tmp_path):
|
def test_allows_agent_workspace(tmp_path):
|
||||||
"""Paths under project data/ must resolve cleanly."""
|
"""Paths under the agent's workspace in project data/ must resolve
|
||||||
|
cleanly. The rest of data/ is application state and is rejected;
|
||||||
|
tests/test_agent_state_dir_confinement.py covers that side."""
|
||||||
from src.tool_execution import _resolve_tool_path
|
from src.tool_execution import _resolve_tool_path
|
||||||
from src.constants import DATA_DIR
|
from src.constants import AGENT_WORKSPACE_DIR
|
||||||
target = os.path.join(DATA_DIR, "test-confinement-ok.txt")
|
target = os.path.join(AGENT_WORKSPACE_DIR, "test-confinement-ok.txt")
|
||||||
os.makedirs(DATA_DIR, exist_ok=True)
|
os.makedirs(AGENT_WORKSPACE_DIR, exist_ok=True)
|
||||||
with open(target, "w") as f:
|
with open(target, "w") as f:
|
||||||
f.write("ok")
|
f.write("ok")
|
||||||
try:
|
try:
|
||||||
|
|||||||
Reference in New Issue
Block a user