mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-08 07:52:20 +02:00
Squash Odysseus development history
This commit is contained in:
@@ -5,6 +5,7 @@ import re
|
||||
import difflib
|
||||
import fnmatch
|
||||
import shutil
|
||||
import tempfile
|
||||
from typing import Optional, Dict, Any, Tuple, List
|
||||
|
||||
from src.constants import MAX_READ_CHARS, MAX_DIFF_LINES, MAX_OUTPUT_CHARS
|
||||
@@ -16,6 +17,9 @@ _CODENAV_SKIP_DIRS = frozenset({
|
||||
})
|
||||
_CODENAV_MAX_HITS = 200
|
||||
_CODENAV_MAX_LINE = 400
|
||||
_STRUCTURED_DOCUMENT_SUFFIXES = frozenset({
|
||||
".doc", ".docx", ".epub", ".pdf", ".pptx", ".xls", ".xlsx",
|
||||
})
|
||||
|
||||
|
||||
def _glob_to_regex(pat: str) -> "re.Pattern":
|
||||
@@ -76,25 +80,33 @@ class EditFileTool:
|
||||
try:
|
||||
args = json.loads(content) if content.strip().startswith("{") else {}
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
args = {}
|
||||
raw_path = (args.get("path") or "").strip()
|
||||
old = args.get("old_string", "")
|
||||
new = args.get("new_string", "")
|
||||
replace_all = bool(args.get("replace_all", False))
|
||||
return {"error": "edit_file: expected valid JSON arguments", "exit_code": 1}
|
||||
if not isinstance(args, dict):
|
||||
return {"error": "edit_file: expected a JSON object", "exit_code": 1}
|
||||
raw_path_value = args.get("path")
|
||||
raw_path = raw_path_value.strip() if isinstance(raw_path_value, str) else ""
|
||||
old = args.get("old_string")
|
||||
new = args.get("new_string")
|
||||
replace_all = args.get("replace_all", False)
|
||||
if not raw_path:
|
||||
return {"error": "edit_file: path required", "exit_code": 1}
|
||||
if not isinstance(old, str) or not old:
|
||||
return {"error": "edit_file: old_string required (use write_file to create a file)", "exit_code": 1}
|
||||
if not isinstance(new, str):
|
||||
return {"error": "edit_file: new_string required", "exit_code": 1}
|
||||
if not isinstance(replace_all, bool):
|
||||
return {"error": "edit_file: replace_all must be a boolean", "exit_code": 1}
|
||||
try:
|
||||
path = _resolve_tool_path(raw_path)
|
||||
except ValueError as e:
|
||||
return {"error": f"edit_file: {e}", "exit_code": 1}
|
||||
if old == "":
|
||||
return {"error": "edit_file: old_string required (use write_file to create a file)", "exit_code": 1}
|
||||
if old == new:
|
||||
return {"error": "edit_file: old_string and new_string are identical", "exit_code": 1}
|
||||
|
||||
def _apply():
|
||||
"""Helper function that performs the actual string replacement and file writing logic."""
|
||||
with open(path, "r", encoding="utf-8") as f:
|
||||
# Exact replacement must not normalize unrelated CRLF/CR newlines.
|
||||
with open(path, "r", encoding="utf-8", newline="") as f:
|
||||
original = f.read()
|
||||
count = original.count(old)
|
||||
if count == 0:
|
||||
@@ -102,7 +114,7 @@ class EditFileTool:
|
||||
if count > 1 and not replace_all:
|
||||
return original, None, f"not_unique:{count}"
|
||||
updated = original.replace(old, new) if replace_all else original.replace(old, new, 1)
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
with open(path, "w", encoding="utf-8", newline="") as f:
|
||||
f.write(updated)
|
||||
return original, updated, "ok"
|
||||
|
||||
@@ -138,17 +150,36 @@ class ReadFileTool:
|
||||
if _stripped.startswith("{"):
|
||||
try:
|
||||
_a = json.loads(_stripped)
|
||||
raw_path = str(_a.get("path", "")).strip()
|
||||
if not isinstance(_a, dict):
|
||||
return {"error": "read_file: expected a JSON object", "exit_code": 1}
|
||||
raw_path_value = _a.get("path")
|
||||
raw_path = raw_path_value.strip() if isinstance(raw_path_value, str) else ""
|
||||
offset = int(_a.get("offset") or 0)
|
||||
limit = int(_a.get("limit") or 0)
|
||||
except (json.JSONDecodeError, TypeError, ValueError):
|
||||
pass
|
||||
return {"error": "read_file: expected valid JSON arguments", "exit_code": 1}
|
||||
if not raw_path:
|
||||
return {"error": "read_file: path required", "exit_code": 1}
|
||||
try:
|
||||
path = _resolve_tool_path(raw_path)
|
||||
except ValueError as e:
|
||||
return {"error": f"read_file: {e}", "exit_code": 1}
|
||||
try:
|
||||
def _read():
|
||||
if os.path.splitext(path)[1].lower() in _STRUCTURED_DOCUMENT_SUFFIXES:
|
||||
from src.document_processor import extract_local_document
|
||||
|
||||
extracted = extract_local_document(
|
||||
path,
|
||||
display_name=os.path.basename(path),
|
||||
analyze_embedded_images=False,
|
||||
)
|
||||
if offset > 0 or limit > 0:
|
||||
lines = extracted.splitlines(keepends=True)
|
||||
start = max(offset, 1) - 1
|
||||
stop = start + limit if limit > 0 else None
|
||||
return "".join(lines[start:stop])[:MAX_READ_CHARS]
|
||||
return extracted[:MAX_READ_CHARS + 1]
|
||||
if offset > 0 or limit > 0:
|
||||
start = max(offset, 1)
|
||||
out, n, budget = [], 0, MAX_READ_CHARS
|
||||
@@ -196,15 +227,54 @@ class WriteFileTool:
|
||||
if _stripped.startswith("{"):
|
||||
try:
|
||||
_a = json.loads(_stripped)
|
||||
if isinstance(_a, dict) and "path" in _a:
|
||||
raw_path = str(_a.get("path", "")).strip()
|
||||
body = str(_a.get("content", ""))
|
||||
if not isinstance(_a, dict):
|
||||
return {"error": "write_file: expected a JSON object", "exit_code": 1}
|
||||
raw_path_value = _a.get("path")
|
||||
body_value = _a.get("content")
|
||||
raw_path = raw_path_value.strip() if isinstance(raw_path_value, str) else ""
|
||||
if not isinstance(body_value, str):
|
||||
return {"error": "write_file: content required", "exit_code": 1}
|
||||
body = body_value
|
||||
except (json.JSONDecodeError, TypeError, ValueError):
|
||||
pass
|
||||
return {"error": "write_file: expected valid JSON arguments", "exit_code": 1}
|
||||
if not raw_path:
|
||||
return {"error": "write_file: path required", "exit_code": 1}
|
||||
try:
|
||||
path = _resolve_tool_path(raw_path)
|
||||
except ValueError as e:
|
||||
return {"error": f"write_file: {e}", "exit_code": 1}
|
||||
# A frequent multimodal artifact failure is writing SVG markup to a
|
||||
# path whose extension promises a raster image. The file exists, so
|
||||
# ordinary artifact checks pass, but image judges cannot decode it.
|
||||
# Reject the mismatch with an actionable native-tool recovery path:
|
||||
# save the SVG with an .svg suffix, then use inspect_media to render
|
||||
# it to the requested PNG/JPEG path.
|
||||
image_suffixes = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp"}
|
||||
body_probe = body.lstrip().casefold()
|
||||
if os.path.splitext(path)[1].casefold() in image_suffixes and (
|
||||
body_probe.startswith("<svg")
|
||||
or (body_probe.startswith("<?xml") and "<svg" in body_probe[:2000])
|
||||
):
|
||||
return {
|
||||
"error": (
|
||||
f"write_file: {path} contains SVG markup but has a raster "
|
||||
"image extension. Write the SVG to a .svg path first, "
|
||||
"then call inspect_media with that SVG as path and this "
|
||||
"path as output_path to render a real raster image."
|
||||
),
|
||||
"exit_code": 1,
|
||||
"artifact_format_error": True,
|
||||
}
|
||||
if not body:
|
||||
return {
|
||||
"error": (
|
||||
f"write_file: {path}: content required; refusing to create an "
|
||||
"empty file. Call write_file again with the exact filename and "
|
||||
"non-empty content. If you need a directory, create it from "
|
||||
"bash/python instead."
|
||||
),
|
||||
"exit_code": 1,
|
||||
}
|
||||
try:
|
||||
def _write():
|
||||
old = ""
|
||||
@@ -280,16 +350,77 @@ class ApplyPatchTool:
|
||||
new = _apply_patch_hunks(old, op["hunks"], op["path"])
|
||||
prepared.append((kind, path, old, new))
|
||||
|
||||
staged: list[tuple[str, str]] = []
|
||||
backups: list[tuple[str, str | None]] = []
|
||||
try:
|
||||
for kind, path, _old, new in prepared:
|
||||
if kind == "delete":
|
||||
continue
|
||||
directory = os.path.dirname(path) or "."
|
||||
os.makedirs(directory, exist_ok=True)
|
||||
fd, temp_path = tempfile.mkstemp(
|
||||
prefix=f".{os.path.basename(path)}.odysseus-",
|
||||
dir=directory,
|
||||
)
|
||||
try:
|
||||
with os.fdopen(fd, "w", encoding="utf-8", newline="") as handle:
|
||||
handle.write(new)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
if os.path.exists(path):
|
||||
shutil.copymode(path, temp_path)
|
||||
except BaseException:
|
||||
try:
|
||||
os.unlink(temp_path)
|
||||
except OSError:
|
||||
pass
|
||||
raise
|
||||
staged.append((path, temp_path))
|
||||
|
||||
for _kind, path, _old, _new in prepared:
|
||||
if os.path.exists(path):
|
||||
directory = os.path.dirname(path) or "."
|
||||
fd, backup_path = tempfile.mkstemp(
|
||||
prefix=f".{os.path.basename(path)}.odysseus-backup-",
|
||||
dir=directory,
|
||||
)
|
||||
os.close(fd)
|
||||
os.unlink(backup_path)
|
||||
os.replace(path, backup_path)
|
||||
backups.append((path, backup_path))
|
||||
else:
|
||||
backups.append((path, None))
|
||||
|
||||
staged_by_path = dict(staged)
|
||||
for kind, path, _old, _new in prepared:
|
||||
if kind != "delete":
|
||||
os.replace(staged_by_path[path], path)
|
||||
staged.clear()
|
||||
except BaseException:
|
||||
for path, backup_path in reversed(backups):
|
||||
try:
|
||||
if os.path.exists(path):
|
||||
os.unlink(path)
|
||||
if backup_path and os.path.exists(backup_path):
|
||||
os.replace(backup_path, path)
|
||||
except OSError:
|
||||
pass
|
||||
raise
|
||||
finally:
|
||||
for _path, temp_path in staged:
|
||||
try:
|
||||
os.unlink(temp_path)
|
||||
except OSError:
|
||||
pass
|
||||
for _path, backup_path in backups:
|
||||
if backup_path:
|
||||
try:
|
||||
os.unlink(backup_path)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
diffs = []
|
||||
for kind, path, old, new in prepared:
|
||||
if kind == "delete":
|
||||
os.remove(path)
|
||||
else:
|
||||
directory = os.path.dirname(path)
|
||||
if directory:
|
||||
os.makedirs(directory, exist_ok=True)
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
f.write(new)
|
||||
for _kind, path, old, new in prepared:
|
||||
diff = _unified_diff(old, new, path)
|
||||
if diff:
|
||||
diffs.append(diff)
|
||||
@@ -407,7 +538,7 @@ def _apply_patch_hunks(original: str, hunks: List[List[str]], label: str) -> str
|
||||
|
||||
class LsTool:
|
||||
async def execute(self, content: str, ctx: dict) -> dict:
|
||||
from src.tool_execution import _resolve_tool_path, _resolve_search_root, _truncate
|
||||
from src.tool_execution import _display_tool_path, _resolve_search_root, _truncate
|
||||
raw_path = ""
|
||||
_s = (content or "").strip()
|
||||
if _s.startswith("{"):
|
||||
@@ -440,7 +571,7 @@ class LsTool:
|
||||
except (PermissionError, OSError) as _e:
|
||||
return None, f"ls: {_e}"
|
||||
rows.sort(key=lambda r: (not r[0], r[1].lower()))
|
||||
lines = [f"{root}:"]
|
||||
lines = [f"{_display_tool_path(root)}:"]
|
||||
for is_dir, name, size in rows[:_CODENAV_MAX_HITS]:
|
||||
lines.append(f" {name}/" if is_dir else f" {name} ({size} B)")
|
||||
if len(rows) > _CODENAV_MAX_HITS:
|
||||
@@ -458,6 +589,7 @@ class GlobTool:
|
||||
async def execute(self, content: str, ctx: dict) -> dict:
|
||||
from src.tool_execution import (
|
||||
_SENSITIVE_BASENAMES,
|
||||
_display_tool_path,
|
||||
_is_sensitive_path,
|
||||
_resolve_tool_path,
|
||||
_resolve_search_root,
|
||||
@@ -549,8 +681,8 @@ class GlobTool:
|
||||
if err:
|
||||
return {"error": err, "exit_code": 1}
|
||||
if not paths:
|
||||
return {"output": f"No files matching {pattern!r} under {root}", "exit_code": 0}
|
||||
out = "\n".join(paths)
|
||||
return {"output": f"No files matching {pattern!r} under {_display_tool_path(root)}", "exit_code": 0}
|
||||
out = "\n".join(_display_tool_path(path) for path in paths)
|
||||
if len(paths) >= _CODENAV_MAX_HITS:
|
||||
out += f"\n... [capped at {_CODENAV_MAX_HITS} files]"
|
||||
return {"output": _truncate(out), "exit_code": 0}
|
||||
@@ -559,6 +691,7 @@ class GrepTool:
|
||||
async def execute(self, content: str, ctx: dict) -> dict:
|
||||
from src.tool_execution import (
|
||||
_SENSITIVE_FILE_PATTERNS,
|
||||
_display_tool_path,
|
||||
_is_sensitive_path,
|
||||
_resolve_tool_path,
|
||||
_resolve_search_root,
|
||||
@@ -591,9 +724,11 @@ class GrepTool:
|
||||
def _grep():
|
||||
import re as _re
|
||||
import shutil
|
||||
if not os.path.exists(root):
|
||||
return None, f"grep: search target not found: {_display_tool_path(root)}"
|
||||
rg = shutil.which("rg")
|
||||
if rg:
|
||||
cmd = [rg, "--line-number", "--no-heading", "--color=never",
|
||||
cmd = [rg, "--line-number", "--with-filename", "--no-heading", "--color=never",
|
||||
"--max-count", str(max_hits)]
|
||||
if ignore_case:
|
||||
cmd.append("--ignore-case")
|
||||
@@ -611,6 +746,11 @@ class GrepTool:
|
||||
try:
|
||||
import subprocess
|
||||
p = subprocess.run(cmd, capture_output=True, text=True, timeout=20)
|
||||
# ripgrep: 0 = matches, 1 = no matches, 2 = failed scan.
|
||||
# Do not present invalid patterns or IO failures as absence.
|
||||
if p.returncode not in (0, 1):
|
||||
detail = (p.stderr or '').strip()[:1200]
|
||||
return None, f"grep: search failed (exit {p.returncode}): {detail or 'no diagnostic available'}"
|
||||
lines = [ln for ln in (p.stdout or "").splitlines() if ln][:max_hits]
|
||||
return lines, None
|
||||
except subprocess.TimeoutExpired:
|
||||
@@ -622,11 +762,12 @@ class GrepTool:
|
||||
except _re.error as _e:
|
||||
return None, f"grep: bad pattern: {_e}"
|
||||
hits = []
|
||||
scan_errors = []
|
||||
if os.path.isfile(root):
|
||||
file_iter = [root]
|
||||
else:
|
||||
file_iter = []
|
||||
for dp, dns, fns in os.walk(root):
|
||||
for dp, dns, fns in os.walk(root, onerror=scan_errors.append):
|
||||
dns[:] = [d for d in dns if d not in _CODENAV_SKIP_DIRS]
|
||||
for fn in fns:
|
||||
if glob_pat and not fnmatch.fnmatch(fn, glob_pat):
|
||||
@@ -635,25 +776,38 @@ class GrepTool:
|
||||
for fp in file_iter:
|
||||
if len(hits) >= max_hits:
|
||||
break
|
||||
if _is_sensitive_path(os.path.realpath(fp)):
|
||||
try:
|
||||
resolved_file = _resolve_tool_path(fp)
|
||||
except ValueError:
|
||||
# Apply the same workspace/sensitive-path checks to each
|
||||
# discovered file, not just the initial search directory.
|
||||
continue
|
||||
try:
|
||||
with open(fp, "r", encoding="utf-8", errors="strict") as f:
|
||||
with open(resolved_file, "r", encoding="utf-8", errors="strict") as f:
|
||||
for i, line in enumerate(f, 1):
|
||||
if rx.search(line):
|
||||
hits.append(f"{fp}:{i}:{line.rstrip()[:_CODENAV_MAX_LINE]}")
|
||||
if len(hits) >= max_hits:
|
||||
break
|
||||
except (UnicodeDecodeError, OSError):
|
||||
except UnicodeDecodeError:
|
||||
continue
|
||||
except OSError as error:
|
||||
scan_errors.append(error)
|
||||
if scan_errors:
|
||||
return None, "grep: search incomplete; one or more files or directories could not be read"
|
||||
return hits, None
|
||||
|
||||
lines, err = await asyncio.to_thread(_grep)
|
||||
if err:
|
||||
return {"error": err, "exit_code": 1}
|
||||
if not lines:
|
||||
return {"output": f"No matches for {pattern!r} under {root}", "exit_code": 0}
|
||||
out = "\n".join(ln[:_CODENAV_MAX_LINE] for ln in lines)
|
||||
return {"output": f"No matches for {pattern!r} under {_display_tool_path(root)}", "exit_code": 0}
|
||||
physical_root = os.path.realpath(root)
|
||||
display_root = _display_tool_path(physical_root)
|
||||
out = "\n".join(
|
||||
(display_root + ln[len(physical_root):] if ln.startswith(physical_root) else ln)[:_CODENAV_MAX_LINE]
|
||||
for ln in lines
|
||||
)
|
||||
if len(lines) >= max_hits:
|
||||
out += f"\n... [capped at {max_hits} matches]"
|
||||
return {"output": _truncate(out), "exit_code": 0}
|
||||
@@ -666,7 +820,7 @@ class GetWorkspaceTool:
|
||||
ws = get_active_workspace()
|
||||
if ws:
|
||||
return {
|
||||
"output": f"{ws}\n(File tools are confined to this folder; the shell starts "
|
||||
"output": "/workspace\n(File tools are confined to this folder; the shell starts "
|
||||
f"here but is not sandboxed and can reach outside it.)",
|
||||
"exit_code": 0,
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user