Merge commit from fork

* fix(security): keep agent file tools out of the app state directory

The agent's read tools (read_file, grep, glob, ls) resolved model-supplied
paths against a root list whose first entry was the whole data directory.
That directory holds the session store, the auth database, the app
encryption key and the settings file, so prompt-injected content could ask
for any of them. No approval prompt stood in the way: reads are classified
read_workspace and pass the untrusted-context gate untouched, which is
correct for reading a workspace and wrong for reading the app's own state.

The agent gets data/agent_workspace/ instead, and the subprocess cwd and
HOME move with it so bash and read_file agree on where scratch files live.

The deny itself is a property of the path, not of the root it arrived
through, because three routes reach the same bytes and closing only the
first leaves the other two working:

  - the default root list
  - a workspace bound at or above the data directory, which vet_workspace
    accepted and chat_routes auto-binds from a path named in the message
  - a tool_path_extra_roots setting covering the data directory

_resolve_search_root also returned the workspace root unchecked when the
path was empty, so a bare ls enumerated the directory whatever the deny
list said. It now resolves that case through the same guards.

A containment rule rather than a filename deny list, so state files added
later are covered without anyone remembering to list them, and so a user's
own settings.json or app.db inside a real workspace is not caught.

Four directories of user content stay readable, because the application
hands their paths to the model and tells it to open them: the chat upload
manifest, downloaded mail attachments, personal docs (which covers the
runbook) and personal uploads.

* fix: enforce state deny during recursive file search

* fix: bound protected filesystem searches

* fix(security): reject inode aliases and workspace redirects

* fix(security): harden partitioned agent searches

* fix(security): report fallback worker exits promptly

* fix(security): clean up search readers and retain relative data roots

---------

Co-authored-by: RaresKeY <158580472+RaresKeY@users.noreply.github.com>
This commit is contained in:
nopoz
2026-09-05 19:21:12 +02:00
committed by GitHub
co-authored by RaresKeY
parent f88e2d1f7f
commit 934d23c0be
10 changed files with 1780 additions and 87 deletions
+7
View File
@@ -14,6 +14,13 @@ import threading
import time import time
import webbrowser import webbrowser
# PyInstaller multiprocessing children re-enter this executable with a private
# bootstrap argument. Consume it before splash/UI or application imports so a
# spawn-based worker does not relaunch the full desktop application.
if __name__ == "__main__":
import multiprocessing
multiprocessing.freeze_support()
# Define a dummy NullWriter to suppress standard stream crashes (isatty etc.) in GUI mode # Define a dummy NullWriter to suppress standard stream crashes (isatty etc.) in GUI mode
class NullWriter: class NullWriter:
def write(self, text): def write(self, text):
+2 -1
View File
@@ -16,7 +16,7 @@ sys.path.insert(0, BASE_DIR)
from src.constants import ( from src.constants import (
DATA_DIR, AUTH_FILE, UPLOAD_DIR, PERSONAL_DIR, PERSONAL_UPLOADS_DIR, DATA_DIR, AUTH_FILE, UPLOAD_DIR, PERSONAL_DIR, PERSONAL_UPLOADS_DIR,
TTS_CACHE_DIR, GENERATED_IMAGES_DIR, DEEP_RESEARCH_DIR, CHROMA_DIR, TTS_CACHE_DIR, GENERATED_IMAGES_DIR, DEEP_RESEARCH_DIR, CHROMA_DIR,
RAG_DIR, MEMORY_VECTORS_DIR, PASSWORD_MIN_LENGTH, RAG_DIR, MEMORY_VECTORS_DIR, AGENT_WORKSPACE_DIR, PASSWORD_MIN_LENGTH,
) )
from core.auth import RESERVED_USERNAMES from core.auth import RESERVED_USERNAMES
@@ -31,6 +31,7 @@ DIRS = [
CHROMA_DIR, CHROMA_DIR,
RAG_DIR, RAG_DIR,
MEMORY_VECTORS_DIR, MEMORY_VECTORS_DIR,
AGENT_WORKSPACE_DIR,
os.path.join(BASE_DIR, "logs"), os.path.join(BASE_DIR, "logs"),
] ]
+428 -61
View File
@@ -3,8 +3,8 @@ import json
import os import os
import re import re
import difflib import difflib
import fnmatch
import shutil import shutil
import time
from typing import Optional, Dict, Any, Tuple, List from typing import Optional, Dict, Any, Tuple, List
from src.constants import MAX_READ_CHARS, MAX_DIFF_LINES, MAX_OUTPUT_CHARS from src.constants import MAX_READ_CHARS, MAX_DIFF_LINES, MAX_OUTPUT_CHARS
@@ -16,6 +16,8 @@ _CODENAV_SKIP_DIRS = frozenset({
}) })
_CODENAV_MAX_HITS = 200 _CODENAV_MAX_HITS = 200
_CODENAV_MAX_LINE = 400 _CODENAV_MAX_LINE = 400
_GREP_TIMEOUT_SECONDS = 20
_GREP_STDERR_PREFIX = 20_000
def _glob_to_regex(pat: str) -> "re.Pattern": def _glob_to_regex(pat: str) -> "re.Pattern":
@@ -42,6 +44,113 @@ def _glob_to_regex(pat: str) -> "re.Pattern":
i += 1 i += 1
return re.compile("".join(out)) return re.compile("".join(out))
def _python_grep_worker(payload: dict, output_queue) -> None:
"""Spawn-safe fallback grep worker used when ripgrep is unavailable.
Keep this at module scope: a frozen Windows executable cannot safely be
relaunched as ``sys.executable -c ...``, while multiprocessing can invoke a
top-level target through its frozen-process bootstrap.
"""
try:
flags = re.IGNORECASE if payload["ignore_case"] else 0
try:
regex = re.compile(payload["pattern"], flags)
glob_regex = (
_glob_to_regex(payload["glob"].replace("\\", "/"))
if payload["glob"]
else None
)
except re.error as exc:
output_queue.put(("error", f"grep: bad pattern: {exc}"))
return
requested_root = payload["root"]
skip_dirs = set(payload["skip_dirs"])
sensitive = {name.casefold() for name in payload["sensitive_names"]}
max_hits = payload["max_hits"]
hits = 0
def within(path: str, root: str) -> bool:
try:
return os.path.commonpath(
[os.path.normcase(path), os.path.normcase(root)]
) == os.path.normcase(root)
except ValueError:
return False
def safe_file(path: str, target: str) -> Optional[str]:
if os.path.islink(path):
return None
canonical = os.path.realpath(path)
if not within(canonical, requested_root) or not within(canonical, target):
return None
parts = [part.casefold() for part in canonical.split(os.sep)]
if any(part in sensitive for part in parts):
return None
try:
if not os.path.isfile(canonical) or os.stat(canonical).st_nlink > 1:
return None
except OSError:
return None
return canonical
for target in payload["targets"]:
if hits >= max_hits:
break
if os.path.isfile(target):
file_iter = iter((target,))
else:
def walk_files():
for directory, dirnames, filenames in os.walk(
target, followlinks=False
):
dirnames[:] = [
name
for name in dirnames
if name not in skip_dirs
and name.casefold() not in sensitive
and not os.path.islink(os.path.join(directory, name))
]
for name in filenames:
yield os.path.join(directory, name)
file_iter = walk_files()
for candidate in file_iter:
path = safe_file(candidate, target)
if path is None:
continue
relative = os.path.relpath(path, requested_root).replace(os.sep, "/")
if glob_regex and not (
glob_regex.fullmatch(relative)
or glob_regex.fullmatch(os.path.basename(path))
):
continue
try:
with open(path, "r", encoding="utf-8", errors="strict") as handle:
for number, line in enumerate(handle, 1):
if regex.search(line):
output_queue.put((
"match",
path,
number,
line.rstrip()[:_CODENAV_MAX_LINE],
))
hits += 1
if hits >= max_hits:
break
except (UnicodeDecodeError, OSError):
continue
if hits >= max_hits:
break
output_queue.put(("done",))
except BaseException as exc:
try:
output_queue.put(("error", f"grep: fallback worker failed: {exc}"))
except BaseException:
pass
def _unified_diff(old: str, new: str, path: str) -> Optional[Dict[str, Any]]: def _unified_diff(old: str, new: str, path: str) -> Optional[Dict[str, Any]]:
if old == new: if old == new:
return None return None
@@ -407,7 +516,11 @@ def _apply_patch_hunks(original: str, hunks: List[List[str]], label: str) -> str
class LsTool: class LsTool:
async def execute(self, content: str, ctx: dict) -> dict: async def execute(self, content: str, ctx: dict) -> dict:
from src.tool_execution import _resolve_tool_path, _resolve_search_root, _truncate from src.tool_execution import (
_is_denied_tool_path,
_resolve_search_root,
_truncate,
)
raw_path = "" raw_path = ""
_s = (content or "").strip() _s = (content or "").strip()
if _s.startswith("{"): if _s.startswith("{"):
@@ -431,6 +544,8 @@ class LsTool:
for entry in it: for entry in it:
if entry.name.startswith("."): if entry.name.startswith("."):
continue continue
if _is_denied_tool_path(os.path.realpath(entry.path)):
continue
try: try:
is_dir = entry.is_dir(follow_symlinks=False) is_dir = entry.is_dir(follow_symlinks=False)
size = entry.stat(follow_symlinks=False).st_size if not is_dir else 0 size = entry.stat(follow_symlinks=False).st_size if not is_dir else 0
@@ -458,7 +573,8 @@ class GlobTool:
async def execute(self, content: str, ctx: dict) -> dict: async def execute(self, content: str, ctx: dict) -> dict:
from src.tool_execution import ( from src.tool_execution import (
_SENSITIVE_BASENAMES, _SENSITIVE_BASENAMES,
_is_sensitive_path, _can_traverse_tool_path,
_is_denied_tool_path,
_resolve_tool_path, _resolve_tool_path,
_resolve_search_root, _resolve_search_root,
_truncate, _truncate,
@@ -507,7 +623,7 @@ class GlobTool:
# .ssh/id_rsa, …) falls through to the walk, which skips it — # .ssh/id_rsa, …) falls through to the walk, which skips it —
# otherwise glob would surface secret paths that read_file / # otherwise glob would surface secret paths that read_file /
# grep already refuse to touch. # grep already refuse to touch.
if inside and os.path.exists(cand) and not _is_sensitive_path(cand): if inside and os.path.exists(cand) and not _is_denied_tool_path(cand):
return [cand], None return [cand], None
# Literal not at exact path — fall through to walk so # Literal not at exact path — fall through to walk so
# e.g. "foo.py" still matches at any depth (like rglob). # e.g. "foo.py" still matches at any depth (like rglob).
@@ -517,13 +633,18 @@ class GlobTool:
cap = _CODENAV_MAX_HITS * 5 cap = _CODENAV_MAX_HITS * 5
try: try:
for dp, dns, fns in os.walk(base): for dp, dns, fns in os.walk(base):
if not _can_traverse_tool_path(os.path.realpath(dp)):
dns[:] = []
continue
# Prune skipped dirs before descending (unlike rglob which # Prune skipped dirs before descending (unlike rglob which
# descends first then filters — fatal on large node_modules). # descends first then filters — fatal on large node_modules).
# Sensitive dirs (.ssh, .gnupg, …) are pruned too so glob # Sensitive dirs (.ssh, .gnupg, …) are pruned too so glob
# never enumerates the keys/tokens inside them. # never enumerates the keys/tokens inside them.
dns[:] = [ dns[:] = [
d for d in dns d for d in dns
if d not in _CODENAV_SKIP_DIRS and d not in _SENSITIVE_BASENAMES if d not in _CODENAV_SKIP_DIRS
and d not in _SENSITIVE_BASENAMES
and _can_traverse_tool_path(os.path.realpath(os.path.join(dp, d)))
] ]
for name in fns + dns: for name in fns + dns:
full = os.path.join(dp, name) full = os.path.join(dp, name)
@@ -531,7 +652,7 @@ class GlobTool:
if regex.fullmatch(rel) or regex.fullmatch(name): if regex.fullmatch(rel) or regex.fullmatch(name):
# Skip deny-listed sensitive files (.env, id_rsa, # Skip deny-listed sensitive files (.env, id_rsa,
# known_hosts, …) the same way grep does. # known_hosts, …) the same way grep does.
if _is_sensitive_path(os.path.realpath(full)): if _is_denied_tool_path(os.path.realpath(full)):
continue continue
try: try:
mtime = os.stat(full).st_mtime mtime = os.stat(full).st_mtime
@@ -558,9 +679,12 @@ class GlobTool:
class GrepTool: class GrepTool:
async def execute(self, content: str, ctx: dict) -> dict: async def execute(self, content: str, ctx: dict) -> dict:
from src.tool_execution import ( from src.tool_execution import (
_SENSITIVE_BASENAMES,
_SENSITIVE_FILE_PATTERNS, _SENSITIVE_FILE_PATTERNS,
_agent_readable_data_subdirs,
_is_denied_tool_path,
_is_sensitive_path, _is_sensitive_path,
_resolve_tool_path, _path_within,
_resolve_search_root, _resolve_search_root,
_truncate, _truncate,
) )
@@ -589,64 +713,307 @@ class GrepTool:
return {"error": f"grep: {e}", "exit_code": 1} return {"error": f"grep: {e}", "exit_code": 1}
def _grep(): def _grep():
import re as _re import multiprocessing
import shutil import queue
import subprocess
import threading
from src.constants import DATA_DIR
rg = shutil.which("rg") rg = shutil.which("rg")
if rg: real_root = os.path.realpath(root)
cmd = [rg, "--line-number", "--no-heading", "--color=never", data_dir = os.path.realpath(DATA_DIR)
"--max-count", str(max_hits)] spans_state = _path_within(data_dir, real_root)
if ignore_case:
cmd.append("--ignore-case") def is_top_level_safe(path: str, *, partition_generated: bool) -> bool:
if glob_pat: lexical = os.path.abspath(path)
cmd += ["--glob", glob_pat] if os.path.islink(lexical):
# --iglob (not --glob) so the exclusion is case-insensitive: return False
# on a case-insensitive filesystem "ID_RSA"/"Known_Hosts" canonical = os.path.realpath(lexical)
# resolve to the same secret as their lowercase forms, and the if not _path_within(canonical, real_root):
# Python fallback below already folds case via _is_sensitive_path. return False
for _pat in _SENSITIVE_FILE_PATTERNS: if partition_generated and os.path.basename(lexical) in _CODENAV_SKIP_DIRS:
cmd += ["--iglob", f"!*{_pat}*"] return False
for _d in _CODENAV_SKIP_DIRS: if _is_sensitive_path(canonical) or _is_denied_tool_path(canonical):
cmd += ["--glob", f"!**/{_d}/**"] return False
cmd += ["--regexp", pattern, root] return True
def safe_targets() -> tuple[list[str], Optional[str]]:
candidates: list[tuple[str, bool]] = []
if not spans_state:
# Preserve direct-root compatibility: skip-directory policy
# prunes descendants, but an explicitly requested allowed
# root named node_modules remains searchable.
candidates.append((real_root, False))
else:
current = real_root
if current != data_dir:
for part in os.path.relpath(data_dir, current).split(os.sep):
try:
with os.scandir(current) as entries:
for entry in entries:
if entry.name != part:
# Reject a sibling link lexically before
# canonicalizing or treating it as a target.
if entry.is_symlink():
continue
candidates.append((entry.path, True))
except OSError as exc:
return [], f"grep: {exc}"
current = os.path.join(current, part)
for readable in _agent_readable_data_subdirs():
if (
_path_within(readable, data_dir)
and _path_within(readable, real_root)
and os.path.exists(readable)
):
candidates.append((readable, True))
targets: list[str] = []
seen: set[str] = set()
for candidate, partition_generated in candidates:
if not is_top_level_safe(
candidate, partition_generated=partition_generated
):
continue
canonical = os.path.realpath(candidate)
if canonical not in seen:
seen.add(canonical)
targets.append(canonical)
return targets, None
targets, target_error = safe_targets()
if target_error:
return None, target_error
base = real_root if os.path.isdir(real_root) else os.path.dirname(real_root)
deadline = time.monotonic() + _GREP_TIMEOUT_SECONDS
lines: list[str] = []
def parse_rg_result(raw: str) -> Optional[str]:
try: try:
import subprocess record = json.loads(raw)
p = subprocess.run(cmd, capture_output=True, text=True, timeout=20) except (TypeError, json.JSONDecodeError):
lines = [ln for ln in (p.stdout or "").splitlines() if ln][:max_hits] return None
return lines, None if record.get("type") != "match":
except subprocess.TimeoutExpired: return None
return None, "grep: timed out" data = record.get("data") or {}
except Exception as _e: path = (data.get("path") or {}).get("text")
return None, f"grep: {_e}" text_value = (data.get("lines") or {}).get("text")
try: number = data.get("line_number")
rx = _re.compile(pattern, _re.IGNORECASE if ignore_case else 0) if not isinstance(path, str) or not isinstance(text_value, str):
except _re.error as _e: return None
return None, f"grep: bad pattern: {_e}" absolute = path if os.path.isabs(path) else os.path.join(base, path)
hits = [] canonical = os.path.realpath(absolute)
if os.path.isfile(root): if not _path_within(canonical, real_root) or _is_denied_tool_path(canonical):
file_iter = [root] return None
else: return f"{os.path.abspath(absolute)}:{number}:{text_value.rstrip()[:_CODENAV_MAX_LINE]}"
file_iter = []
for dp, dns, fns in os.walk(root): def run_rg(cmd: list[str]) -> Optional[str]:
dns[:] = [d for d in dns if d not in _CODENAV_SKIP_DIRS] try:
for fn in fns: process = subprocess.Popen(
if glob_pat and not fnmatch.fnmatch(fn, glob_pat): cmd,
cwd=base,
stdin=subprocess.DEVNULL,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
bufsize=1,
)
except Exception as exc:
return f"grep: {exc}"
output: queue.Queue[Optional[str]] = queue.Queue(maxsize=max_hits + 2)
stderr_prefix: list[str] = []
stderr_size = 0
stop_reader = threading.Event()
def enqueue_stdout(value: Optional[str]) -> bool:
# The consumer stops at the result cap or deadline. Never
# leave a producer blocked on its bounded queue afterward.
while not stop_reader.is_set():
try:
output.put(value, timeout=0.05)
return True
except queue.Full:
continue continue
file_iter.append(os.path.join(dp, fn)) return False
for fp in file_iter:
if len(hits) >= max_hits: def read_stdout() -> None:
break assert process.stdout is not None
if _is_sensitive_path(os.path.realpath(fp)): try:
continue for line in process.stdout:
if not enqueue_stdout(line.rstrip("\n")):
break
finally:
enqueue_stdout(None)
def read_stderr() -> None:
nonlocal stderr_size
assert process.stderr is not None
while True:
chunk = process.stderr.read(4096)
if not chunk:
break
if stderr_size < _GREP_STDERR_PREFIX:
kept = chunk[:_GREP_STDERR_PREFIX - stderr_size]
stderr_prefix.append(kept)
stderr_size += len(kept)
stdout_thread = threading.Thread(target=read_stdout, daemon=True)
stderr_thread = threading.Thread(target=read_stderr, daemon=True)
stdout_thread.start()
stderr_thread.start()
timed_out = False
capped = False
try: try:
with open(fp, "r", encoding="utf-8", errors="strict") as f: while len(lines) < max_hits:
for i, line in enumerate(f, 1): remaining = deadline - time.monotonic()
if rx.search(line): if remaining <= 0:
hits.append(f"{fp}:{i}:{line.rstrip()[:_CODENAV_MAX_LINE]}") timed_out = True
if len(hits) >= max_hits: break
break try:
except (UnicodeDecodeError, OSError): raw = output.get(timeout=remaining)
continue except queue.Empty:
return hits, None timed_out = True
break
if raw is None:
break
parsed = parse_rg_result(raw)
if parsed and parsed not in lines:
lines.append(parsed)
capped = len(lines) >= max_hits
finally:
stop_reader.set()
if (timed_out or capped) and process.poll() is None:
process.terminate()
try:
remaining = max(0.01, deadline - time.monotonic())
return_code = process.wait(timeout=min(1, remaining))
except subprocess.TimeoutExpired:
process.kill()
return_code = process.wait()
stdout_thread.join()
stderr_thread.join()
if timed_out:
return "grep: timed out"
if not capped and return_code not in (0, 1):
detail = "".join(stderr_prefix).strip()
return f"grep: {detail or f'process exited {return_code}'}"
return None
if rg:
# Validate even when policy filtering leaves no search targets.
if not targets:
error = run_rg([rg, "--json", "--no-config", "--regexp", pattern])
return (None, error) if error else ([], None)
relative_targets = [os.path.relpath(target, base) for target in targets]
for offset in range(0, len(relative_targets), 128):
if len(lines) >= max_hits:
break
cmd = [
rg, "--json", "--no-config", "--no-follow",
"--max-count", str(max_hits - len(lines)),
"--max-columns", str(_CODENAV_MAX_LINE),
"--max-columns-preview",
]
if ignore_case:
cmd.append("--ignore-case")
if glob_pat:
cmd += ["--glob", glob_pat]
for sensitive_pattern in _SENSITIVE_FILE_PATTERNS:
cmd += ["--iglob", f"!{sensitive_pattern}"]
for skipped_dir in _CODENAV_SKIP_DIRS:
cmd += ["--glob", f"!**/{skipped_dir}/**"]
cmd += ["--regexp", pattern, "--", *relative_targets[offset:offset + 128]]
error = run_rg(cmd)
if error:
return None, error
return lines, None
# This runs inside asyncio.to_thread(), so forking would clone a
# multithreaded process and can deadlock. Spawn is platform-safe and
# PyInstaller-compatible via launcher's early freeze_support().
payload = {
"root": real_root,
"targets": targets,
"pattern": pattern,
"ignore_case": ignore_case,
"glob": glob_pat,
"max_hits": max_hits,
"skip_dirs": tuple(_CODENAV_SKIP_DIRS),
"sensitive_names": tuple(
set(_SENSITIVE_BASENAMES) | set(_SENSITIVE_FILE_PATTERNS)
),
}
try:
context = multiprocessing.get_context("spawn")
output_queue = context.Queue(maxsize=max_hits + 2)
worker = context.Process(
target=_python_grep_worker, args=(payload, output_queue)
)
worker.start()
except Exception as exc:
try:
output_queue.close()
except (NameError, OSError, ValueError):
pass
return None, f"grep: could not start fallback worker: {exc}"
error = None
completed = False
try:
while len(lines) < max_hits:
remaining = deadline - time.monotonic()
if remaining <= 0:
error = "grep: timed out"
break
try:
# Keep queue waits short enough to observe a spawn
# worker that dies during bootstrap/import before it
# can enqueue either an error or the done sentinel.
record = output_queue.get(timeout=min(0.05, remaining))
except queue.Empty:
if worker.is_alive():
continue
worker.join(timeout=0)
try:
# A multiprocessing queue's feeder can make the
# final record visible at process-exit time. Give
# that record precedence over the exit status.
remaining = deadline - time.monotonic()
record = output_queue.get(
timeout=min(0.05, max(0, remaining))
)
except queue.Empty:
error = f"grep: fallback worker exited {worker.exitcode}"
break
if record[0] == "done":
completed = True
break
if record[0] == "error":
error = record[1]
break
_, path, number, text_value = record
canonical = os.path.realpath(path)
if not _path_within(canonical, real_root) or _is_denied_tool_path(canonical):
continue
rendered = f"{path}:{number}:{text_value}"
if rendered not in lines:
lines.append(rendered)
finally:
if completed:
worker.join(timeout=min(1, max(0.01, deadline - time.monotonic())))
if worker.is_alive():
worker.terminate()
worker.join(timeout=1)
if worker.is_alive():
worker.kill()
worker.join()
output_queue.close()
if error:
return None, error
if worker.exitcode not in (0, None) and len(lines) < max_hits:
return None, f"grep: fallback worker exited {worker.exitcode}"
return lines, None
lines, err = await asyncio.to_thread(_grep) lines, err = await asyncio.to_thread(_grep)
if err: if err:
+30 -1
View File
@@ -2,10 +2,11 @@
"""Initialize all application components and dependencies.""" """Initialize all application components and dependencies."""
import os import os
import logging import logging
import stat
from typing import Dict, Any from typing import Dict, Any
from src.constants import ( from src.constants import (
DATA_DIR, PERSONAL_DIR, RUNBOOK_DIR, UPLOAD_DIR, DATA_DIR, PERSONAL_DIR, RUNBOOK_DIR, UPLOAD_DIR, AGENT_WORKSPACE_DIR,
SESSIONS_FILE, DEFAULT_HOST, OPENAI_API_KEY SESSIONS_FILE, DEFAULT_HOST, OPENAI_API_KEY
) )
from src.memory import MemoryManager from src.memory import MemoryManager
@@ -31,6 +32,34 @@ def create_directories():
for directory in (DATA_DIR, PERSONAL_DIR, RUNBOOK_DIR, UPLOAD_DIR): for directory in (DATA_DIR, PERSONAL_DIR, RUNBOOK_DIR, UPLOAD_DIR):
os.makedirs(directory, exist_ok=True) os.makedirs(directory, exist_ok=True)
# The model-controlled workspace must be a real child of DATA_DIR. Never
# follow a pre-existing symlink here: it would silently move the default
# native-file root outside the application volume before any resolver runs.
data_root = os.path.realpath(os.path.abspath(os.path.expanduser(DATA_DIR)))
workspace = os.path.abspath(os.path.expanduser(AGENT_WORKSPACE_DIR))
expected_workspace = os.path.join(data_root, "agent_workspace")
# Validate the real parent so a supported DATA_DIR bind/symlink works, but
# require the fixed internal carve-out name and reject a link at the model-
# controlled workspace entry itself.
if (
os.path.basename(workspace) != "agent_workspace"
or os.path.realpath(os.path.dirname(workspace)) != data_root
):
raise RuntimeError("agent workspace must be the canonical child of DATA_DIR")
if os.path.lexists(workspace):
mode = os.lstat(workspace).st_mode
if stat.S_ISLNK(mode) or not stat.S_ISDIR(mode):
raise RuntimeError("agent workspace must be a real directory")
else:
os.mkdir(workspace, 0o700)
resolved_workspace = os.path.realpath(workspace)
if resolved_workspace != expected_workspace:
raise RuntimeError("agent workspace must be the canonical child of DATA_DIR")
try:
os.chmod(workspace, 0o700)
except OSError:
pass
def initialize_managers(base_dir: str, rag_manager=None) -> Dict[str, Any]: def initialize_managers(base_dir: str, rag_manager=None) -> Dict[str, Any]:
""" """
Initialize all manager and handler instances. Initialize all manager and handler instances.
+5
View File
@@ -54,6 +54,11 @@ GALLERY_DIR = os.path.join(DATA_DIR, "gallery")
GALLERY_UPLOADS_DIR = os.path.join(DATA_DIR, "gallery_uploads") GALLERY_UPLOADS_DIR = os.path.join(DATA_DIR, "gallery_uploads")
MEMORY_VECTORS_DIR = os.path.join(DATA_DIR, "memory_vectors") MEMORY_VECTORS_DIR = os.path.join(DATA_DIR, "memory_vectors")
# The only part of DATA_DIR the agent's file tools and subprocesses may touch.
# Everything else under DATA_DIR is application state (session store, auth
# database, encryption key, settings), and the agent has no business reading it.
AGENT_WORKSPACE_DIR = os.path.join(DATA_DIR, "agent_workspace")
# Paths with an intentional dedicated env override, defaulting under DATA_DIR. # Paths with an intentional dedicated env override, defaulting under DATA_DIR.
MAIL_ATTACHMENTS_DIR = os.getenv("ODYSSEUS_MAIL_ATTACHMENTS_DIR", os.path.join(DATA_DIR, "mail-attachments")) MAIL_ATTACHMENTS_DIR = os.getenv("ODYSSEUS_MAIL_ATTACHMENTS_DIR", os.path.join(DATA_DIR, "mail-attachments"))
# `or` (not os.getenv's default arg) so a PRESENT-but-EMPTY value falls back to # `or` (not os.getenv's default arg) so a PRESENT-but-EMPTY value falls back to
+235 -18
View File
@@ -15,6 +15,7 @@ import logging
import os import os
import pathlib import pathlib
import re import re
import stat
import sys import sys
import time import time
from typing import Any, Awaitable, Callable, Dict, Optional, Tuple from typing import Any, Awaitable, Callable, Dict, Optional, Tuple
@@ -30,7 +31,12 @@ from src.tool_security import (
from src.tool_capabilities import ToolRunSecurityContext, blocked_tool_result from src.tool_capabilities import ToolRunSecurityContext, blocked_tool_result
from src.tool_approvals import ExactToolApproval from src.tool_approvals import ExactToolApproval
from src.tool_policy import ToolPolicy from src.tool_policy import ToolPolicy
from src.constants import MAX_OUTPUT_CHARS, MAX_READ_CHARS, MAX_DIFF_LINES, DATA_DIR from src.constants import (
MAX_OUTPUT_CHARS,
MAX_READ_CHARS,
MAX_DIFF_LINES,
AGENT_WORKSPACE_DIR,
)
from src.tool_utils import _truncate, get_mcp_manager from src.tool_utils import _truncate, get_mcp_manager
@@ -46,11 +52,11 @@ _MISSING_TOOL_SECURITY_CONTEXT = _MissingToolSecurityContext()
NO_TOOL_SECURITY_CONTEXT = _NoToolSecurityContext() NO_TOOL_SECURITY_CONTEXT = _NoToolSecurityContext()
# Persistent working directory for agent subprocesses. # Persistent working directory for agent subprocesses.
# Resolves to <repo_root>/data, which is the bind-mounted volume in Docker # Resolves to <repo_root>/data/agent_workspace, inside the bind-mounted volume
# (/app/data) and the local data directory for manual installs. # in Docker (/app/data), so files survive a rebuild as before. The subdirectory
# Using this as cwd and HOME prevents the agent from silently creating files # rather than data/ itself keeps agent scratch files and dotfiles out of the
# in ephemeral container layers that are lost on the next rebuild. # directory holding the session store and the auth database.
_AGENT_WORKDIR = DATA_DIR _AGENT_WORKDIR = AGENT_WORKSPACE_DIR
@@ -66,10 +72,15 @@ _AGENT_WORKDIR = DATA_DIR
# 1. Sensitive-subpath deny list — checked FIRST. Blocks .ssh, # 1. Sensitive-subpath deny list — checked FIRST. Blocks .ssh,
# .gnupg, shell rc files, token/env files even if the root above # .gnupg, shell rc files, token/env files even if the root above
# them is on the allowlist. # them is on the allowlist.
# 2. Allowlist — only the directories the agent legitimately needs # 2. Application-state deny (_is_app_state_path) - DATA_DIR holds the
# (project data/, system tmp). $HOME is NOT on the default list. # session store, auth database, app key and settings, so only
# 3. Opt-in extra roots — admin can add broader roots via the # _agent_readable_data_subdirs() is readable inside it.
# "tool_path_extra_roots" setting (list of path strings). # 3. Allowlist - only the directories the agent legitimately needs
# (its data/ workspace, user content, system tmp). $HOME is NOT on
# the default list.
# 4. Opt-in extra roots - admin can add broader roots via the
# "tool_path_extra_roots" setting. These cannot re-open DATA_DIR;
# rule 2 is independent of which root a path arrived through.
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
_SENSITIVE_BASENAMES: set[str] = { _SENSITIVE_BASENAMES: set[str] = {
@@ -116,6 +127,184 @@ def _is_sensitive_path(resolved: str) -> bool:
return filename in _SENSITIVE_FILE_PATTERNS_CF return filename in _SENSITIVE_FILE_PATTERNS_CF
def _path_within(resolved: str, root: str) -> bool:
"""True when *resolved* is *root* itself or sits underneath it.
Use the platform's path-case rules. This helper participates in allow
decisions, so unconditional case-folding would let a distinct ``/DATA``
tree masquerade as a descendant of ``/data`` on case-sensitive systems.
"""
resolved, root = os.path.normcase(resolved), os.path.normcase(root)
if resolved == root:
return True
try:
if os.path.commonpath([resolved, root]) == root:
return True
except ValueError:
return False
# normcase is intentionally conservative about assumptions (notably on
# POSIX), so consult the filesystem when paths exist. This recognizes a
# case alias on a case-insensitive volume without treating distinct
# case-sensitive paths as the same allow root.
if os.path.exists(root):
candidate = resolved
while True:
try:
if os.path.exists(candidate) and os.path.samefile(candidate, root):
return True
except OSError:
pass
parent = os.path.dirname(candidate)
if parent == candidate:
break
candidate = parent
return False
def _path_within_conservative(resolved: str, root: str) -> bool:
"""Containment for deny decisions, folding case to fail closed."""
resolved, root = resolved.casefold(), root.casefold()
if resolved == root:
return True
try:
return os.path.commonpath([resolved, root]) == root
except ValueError:
return False
def _agent_readable_data_subdirs() -> tuple[str, ...]:
"""The only parts of DATA_DIR the agent's file tools may reach.
The agent's own scratch folder, plus the directories of user content whose
paths the application itself gives to the model, which it would then be
unable to open. These normally live under DATA_DIR; the documented mail
attachment override may instead name a disjoint external directory:
UPLOAD_DIR the chat upload manifest renders "path=<p>" and
says to read it with read_file (agent_loop.py)
MAIL_ATTACHMENTS_DIR download_attachment returns the path and its own
description tells the model to read it
PERSONAL_DIR GET /api/personal returns a path per file and is
reachable through the app_api tool; RUNBOOK_DIR
nests under it
PERSONAL_UPLOADS_DIR indexed as a personal-docs directory, which
manage_rag lists as an absolute path
Order matters: the first entry is roots[0], which _resolve_search_root uses
when grep/glob/ls are called with no path.
"""
from src.constants import (
DATA_DIR,
MAIL_ATTACHMENTS_DIR,
PERSONAL_DIR,
PERSONAL_UPLOADS_DIR,
UPLOAD_DIR,
)
configured = (
(AGENT_WORKSPACE_DIR, "agent_workspace", False),
(UPLOAD_DIR, "uploads", False),
# This has a documented environment override and may legitimately
# live outside DATA_DIR, but it must never equal/contain DATA_DIR.
(MAIL_ATTACHMENTS_DIR, "mail-attachments", True),
(PERSONAL_DIR, "personal_docs", False),
(PERSONAL_UPLOADS_DIR, "personal_uploads", False),
)
configured_data_dir = os.path.abspath(os.path.expanduser(str(DATA_DIR)))
data_dir = os.path.realpath(configured_data_dir)
safe: list[str] = []
for raw, internal_name, external_ok in configured:
value = str(raw or "").strip()
# These paths are security-policy roots, not ordinary allowlist
# entries. Internal roles may inherit a relative DATA_DIR, but must
# still resolve to their exact canonical child below. External mail
# overrides require an absolute, disjoint directory.
if not value:
continue
expanded = os.path.abspath(os.path.expanduser(value))
# A policy root must not acquire an exemption by redirecting its final
# path component to protected state or to an unrelated external tree.
if os.path.islink(expanded):
continue
resolved = os.path.realpath(expanded)
if os.path.exists(resolved) and not os.path.isdir(resolved):
continue
expected_internal = os.path.join(data_dir, internal_name)
expected_configured = os.path.join(configured_data_dir, internal_name)
inside_data = (
os.path.normcase(expanded)
in {
os.path.normcase(expected_configured),
os.path.normcase(expected_internal),
}
and resolved == expected_internal
)
external_safe = (
external_ok
and os.path.isabs(os.path.expanduser(value))
and resolved != data_dir
and os.path.dirname(resolved) != resolved
and not _path_within(data_dir, resolved)
and not _path_within(resolved, data_dir)
)
if not (inside_data or external_safe) or _is_sensitive_path(resolved):
continue
safe.append(resolved)
return tuple(safe)
def _is_app_state_path(resolved: str) -> bool:
"""True for anything under DATA_DIR that is not agent-readable.
DATA_DIR holds the session store, the auth database, the app encryption key
and the settings file. A model-supplied path must not reach those through
any root, so this is checked in both resolvers rather than expressed as an
absence from the allowlist: a workspace bound at or above the data
directory, or an opt-in tool_path_extra_roots entry covering it, would
otherwise put them back in reach.
A containment rule rather than a filename deny list, so state files added
later are covered without anyone remembering to list them, and so a user's
own settings.json or app.db inside a real workspace is not caught.
"""
from src.constants import DATA_DIR
if not _path_within_conservative(resolved, os.path.realpath(DATA_DIR)):
return False
return not any(
_path_within(resolved, d)
for d in _agent_readable_data_subdirs()
)
def _is_hardlinked_regular_file(resolved: str) -> bool:
"""Reject inode aliases that can smuggle DATA_DIR state into an allow root."""
try:
target = os.stat(resolved, follow_symlinks=False)
except OSError:
return False
return stat.S_ISREG(target.st_mode) and getattr(target, "st_nlink", 1) > 1
def _is_denied_tool_path(resolved: str) -> bool:
"""Apply every path deny to a canonical traversal result."""
return (
_is_sensitive_path(resolved)
or _is_app_state_path(resolved)
or _is_hardlinked_regular_file(resolved)
)
def _can_traverse_tool_path(resolved: str) -> bool:
"""Allow walking a denied state parent only to reach safe carve-outs."""
if _is_sensitive_path(resolved):
return False
if not _is_app_state_path(resolved):
return True
return any(
_path_within(readable, resolved)
for readable in _agent_readable_data_subdirs()
)
def _tool_path_roots() -> list[str]: def _tool_path_roots() -> list[str]:
"""Return the list of directory roots that read_file / write_file """Return the list of directory roots that read_file / write_file
may touch. Default: project data/ + system temp dirs. Extra roots may touch. Default: project data/ + system temp dirs. Extra roots
@@ -123,9 +312,9 @@ def _tool_path_roots() -> list[str]:
""" """
roots: list[str] = [] roots: list[str] = []
# Project data directory — the agent's primary workspace. # The agent's workspace plus the user-content directories inside data/.
from src.constants import DATA_DIR # The rest of DATA_DIR is denied by _is_app_state_path.
roots.append(DATA_DIR) roots.extend(_agent_readable_data_subdirs())
# /tmp (and its macOS realpath /private/tmp). # /tmp (and its macOS realpath /private/tmp).
roots.append("/tmp") roots.append("/tmp")
@@ -193,6 +382,12 @@ def _resolve_tool_path(raw_path: str) -> str:
f"path '{raw_path}' is inside a sensitive directory " f"path '{raw_path}' is inside a sensitive directory "
f"(e.g. .ssh, .gnupg) or matches a sensitive filename" f"(e.g. .ssh, .gnupg) or matches a sensitive filename"
) )
if _is_app_state_path(resolved):
raise ValueError(
f"path '{raw_path}' is inside the application state directory"
)
if _is_hardlinked_regular_file(resolved):
raise ValueError(f"path '{raw_path}' is a hard-linked file")
for root in _tool_path_roots(): for root in _tool_path_roots():
if resolved == root: if resolved == root:
@@ -228,6 +423,12 @@ def _resolve_tool_path_in_workspace(workspace: str, raw_path: str) -> str:
f"path '{raw_path}' is inside a sensitive directory " f"path '{raw_path}' is inside a sensitive directory "
f"(e.g. .ssh, .gnupg) or matches a sensitive filename" f"(e.g. .ssh, .gnupg) or matches a sensitive filename"
) )
if _is_app_state_path(resolved):
raise ValueError(
f"path '{raw_path}' is inside the application state directory"
)
if _is_hardlinked_regular_file(resolved):
raise ValueError(f"path '{raw_path}' is a hard-linked file")
if resolved != base: if resolved != base:
# normcase so containment holds on case-insensitive filesystems # normcase so containment holds on case-insensitive filesystems
# (Windows, default macOS): it lowercases on Windows and is a no-op on # (Windows, default macOS): it lowercases on Windows and is a no-op on
@@ -277,6 +478,10 @@ def vet_workspace(raw: str) -> Optional[str]:
resolved = os.path.realpath(os.path.expanduser(raw)) resolved = os.path.realpath(os.path.expanduser(raw))
if not os.path.isdir(resolved) or _is_sensitive_path(resolved): if not os.path.isdir(resolved) or _is_sensitive_path(resolved):
return None return None
# Refuse the bind rather than binding a workspace where every subsequent
# tool call would fail on the same deny list.
if _is_app_state_path(resolved):
return None
# Reject filesystem roots: binding / (or a Windows drive/UNC root) as the # Reject filesystem roots: binding / (or a Windows drive/UNC root) as the
# workspace would make every absolute path "inside" it, collapsing the # workspace would make every absolute path "inside" it, collapsing the
# confinement into host-wide file access. A root is its own dirname, which # confinement into host-wide file access. A root is its own dirname, which
@@ -289,7 +494,13 @@ def vet_workspace(raw: str) -> Optional[str]:
def agent_cwd() -> str: def agent_cwd() -> str:
"""Working directory for agent subprocesses (bash/python/background jobs): """Working directory for agent subprocesses (bash/python/background jobs):
the active workspace when set, else the persistent data dir.""" the active workspace when set, else the persistent data dir."""
return get_active_workspace() or _AGENT_WORKDIR workspace = get_active_workspace()
if workspace:
return workspace
resolved = os.path.realpath(_AGENT_WORKDIR)
if resolved not in _agent_readable_data_subdirs():
raise RuntimeError("agent workspace is not a safe real directory")
return resolved
def get_mcp_manager(): def get_mcp_manager():
@@ -304,16 +515,22 @@ def _resolve_search_root(raw_path: str) -> str:
With a workspace active, the workspace folder is the root and a supplied With a workspace active, the workspace folder is the root and a supplied
path is confined inside it. Otherwise an empty path defaults to the agent's path is confined inside it. Otherwise an empty path defaults to the agent's
primary root (project data dir) and a supplied path is confined by the primary root (its workspace under the project data dir) and a supplied path
global allowlist + sensitive-file policy. is confined by the global allowlist + sensitive-file policy.
""" """
raw = (raw_path or "").strip() raw = (raw_path or "").strip()
ws = get_active_workspace() ws = get_active_workspace()
if ws: if ws:
return os.path.realpath(ws) if not raw else _resolve_tool_path_in_workspace(ws, raw) # Resolve the empty case as the workspace path rather than returning
# it directly: returned unchecked it skipped both deny lists, so a
# bare ls listed whatever the workspace was bound to.
return _resolve_tool_path_in_workspace(ws, raw or ws)
if not raw: if not raw:
roots = _tool_path_roots() roots = _tool_path_roots()
return roots[0] if roots else os.path.realpath(".") default_root = os.path.realpath(AGENT_WORKSPACE_DIR)
if default_root in roots and not _is_denied_tool_path(default_root):
return default_root
raise ValueError("default agent workspace is not a safe readable data subdirectory")
return _resolve_tool_path(raw) return _resolve_tool_path(raw)
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
File diff suppressed because it is too large Load Diff
+11
View File
@@ -91,6 +91,17 @@ def test_grep_python_fallback_when_no_rg(repo, monkeypatch):
assert ".git/config" not in r["output"] assert ".git/config" not in r["output"]
def test_grep_python_fallback_uses_relative_glob_paths(repo, monkeypatch):
monkeypatch.setattr(shutil, "which", lambda name: None)
r = _run(
"grep",
f'{{"pattern": "needle|python", "glob": "**/*.py", "path": "{repo}"}}',
)
assert r["exit_code"] == 0
assert "a.py" in r["output"]
assert "sub/deep/c.py" in r["output"]
@pytest.mark.skipif(shutil.which("rg") is None, reason="targets the ripgrep fast-path") @pytest.mark.skipif(shutil.which("rg") is None, reason="targets the ripgrep fast-path")
def test_grep_skips_case_variant_sensitive_files_rg(repo): def test_grep_skips_case_variant_sensitive_files_rg(repo):
"""The rg fast-path must exclude deny-listed key files case-insensitively. """The rg fast-path must exclude deny-listed key files case-insensitively.
+10
View File
@@ -1,12 +1,22 @@
# tests/test_launcher.py # tests/test_launcher.py
import sys import sys
import os import os
from pathlib import Path
from unittest import mock from unittest import mock
import pytest import pytest
from launcher import NullWriter, create_tray_image, on_open_browser, on_exit, open_browser from launcher import NullWriter, create_tray_image, on_open_browser, on_exit, open_browser
def test_frozen_multiprocessing_bootstrap_precedes_gui_and_app_imports():
source = Path("launcher.py").read_text(encoding="utf-8")
freeze = source.index("multiprocessing.freeze_support()")
splash = source.index("if getattr(sys, 'frozen', False):")
app_import = source.index("from app import app")
assert freeze < splash < app_import
def test_null_writer(): def test_null_writer():
writer = NullWriter() writer = NullWriter()
# writing and flushing should not raise any exceptions # writing and flushing should not raise any exceptions
+7 -5
View File
@@ -161,12 +161,14 @@ def test_blocks_netrc():
_resolve_tool_path("~/.netrc") _resolve_tool_path("~/.netrc")
def test_allows_project_data(tmp_path): def test_allows_agent_workspace(tmp_path):
"""Paths under project data/ must resolve cleanly.""" """Paths under the agent's workspace in project data/ must resolve
cleanly. The rest of data/ is application state and is rejected;
tests/test_agent_state_dir_confinement.py covers that side."""
from src.tool_execution import _resolve_tool_path from src.tool_execution import _resolve_tool_path
from src.constants import DATA_DIR from src.constants import AGENT_WORKSPACE_DIR
target = os.path.join(DATA_DIR, "test-confinement-ok.txt") target = os.path.join(AGENT_WORKSPACE_DIR, "test-confinement-ok.txt")
os.makedirs(DATA_DIR, exist_ok=True) os.makedirs(AGENT_WORKSPACE_DIR, exist_ok=True)
with open(target, "w") as f: with open(target, "w") as f:
f.write("ok") f.write("ok")
try: try: