Squash Odysseus development history

This commit is contained in:
pewdiepie-archdaemon
2026-09-11 06:04:19 +00:00
parent c9dd68d890
commit 84aa9a91de
871 changed files with 265870 additions and 27854 deletions
+190 -36
View File
@@ -5,6 +5,7 @@ import re
import difflib
import fnmatch
import shutil
import tempfile
from typing import Optional, Dict, Any, Tuple, List
from src.constants import MAX_READ_CHARS, MAX_DIFF_LINES, MAX_OUTPUT_CHARS
@@ -16,6 +17,9 @@ _CODENAV_SKIP_DIRS = frozenset({
})
_CODENAV_MAX_HITS = 200
_CODENAV_MAX_LINE = 400
_STRUCTURED_DOCUMENT_SUFFIXES = frozenset({
".doc", ".docx", ".epub", ".pdf", ".pptx", ".xls", ".xlsx",
})
def _glob_to_regex(pat: str) -> "re.Pattern":
@@ -76,25 +80,33 @@ class EditFileTool:
try:
args = json.loads(content) if content.strip().startswith("{") else {}
except (json.JSONDecodeError, TypeError):
args = {}
raw_path = (args.get("path") or "").strip()
old = args.get("old_string", "")
new = args.get("new_string", "")
replace_all = bool(args.get("replace_all", False))
return {"error": "edit_file: expected valid JSON arguments", "exit_code": 1}
if not isinstance(args, dict):
return {"error": "edit_file: expected a JSON object", "exit_code": 1}
raw_path_value = args.get("path")
raw_path = raw_path_value.strip() if isinstance(raw_path_value, str) else ""
old = args.get("old_string")
new = args.get("new_string")
replace_all = args.get("replace_all", False)
if not raw_path:
return {"error": "edit_file: path required", "exit_code": 1}
if not isinstance(old, str) or not old:
return {"error": "edit_file: old_string required (use write_file to create a file)", "exit_code": 1}
if not isinstance(new, str):
return {"error": "edit_file: new_string required", "exit_code": 1}
if not isinstance(replace_all, bool):
return {"error": "edit_file: replace_all must be a boolean", "exit_code": 1}
try:
path = _resolve_tool_path(raw_path)
except ValueError as e:
return {"error": f"edit_file: {e}", "exit_code": 1}
if old == "":
return {"error": "edit_file: old_string required (use write_file to create a file)", "exit_code": 1}
if old == new:
return {"error": "edit_file: old_string and new_string are identical", "exit_code": 1}
def _apply():
"""Helper function that performs the actual string replacement and file writing logic."""
with open(path, "r", encoding="utf-8") as f:
# Exact replacement must not normalize unrelated CRLF/CR newlines.
with open(path, "r", encoding="utf-8", newline="") as f:
original = f.read()
count = original.count(old)
if count == 0:
@@ -102,7 +114,7 @@ class EditFileTool:
if count > 1 and not replace_all:
return original, None, f"not_unique:{count}"
updated = original.replace(old, new) if replace_all else original.replace(old, new, 1)
with open(path, "w", encoding="utf-8") as f:
with open(path, "w", encoding="utf-8", newline="") as f:
f.write(updated)
return original, updated, "ok"
@@ -138,17 +150,36 @@ class ReadFileTool:
if _stripped.startswith("{"):
try:
_a = json.loads(_stripped)
raw_path = str(_a.get("path", "")).strip()
if not isinstance(_a, dict):
return {"error": "read_file: expected a JSON object", "exit_code": 1}
raw_path_value = _a.get("path")
raw_path = raw_path_value.strip() if isinstance(raw_path_value, str) else ""
offset = int(_a.get("offset") or 0)
limit = int(_a.get("limit") or 0)
except (json.JSONDecodeError, TypeError, ValueError):
pass
return {"error": "read_file: expected valid JSON arguments", "exit_code": 1}
if not raw_path:
return {"error": "read_file: path required", "exit_code": 1}
try:
path = _resolve_tool_path(raw_path)
except ValueError as e:
return {"error": f"read_file: {e}", "exit_code": 1}
try:
def _read():
if os.path.splitext(path)[1].lower() in _STRUCTURED_DOCUMENT_SUFFIXES:
from src.document_processor import extract_local_document
extracted = extract_local_document(
path,
display_name=os.path.basename(path),
analyze_embedded_images=False,
)
if offset > 0 or limit > 0:
lines = extracted.splitlines(keepends=True)
start = max(offset, 1) - 1
stop = start + limit if limit > 0 else None
return "".join(lines[start:stop])[:MAX_READ_CHARS]
return extracted[:MAX_READ_CHARS + 1]
if offset > 0 or limit > 0:
start = max(offset, 1)
out, n, budget = [], 0, MAX_READ_CHARS
@@ -196,15 +227,54 @@ class WriteFileTool:
if _stripped.startswith("{"):
try:
_a = json.loads(_stripped)
if isinstance(_a, dict) and "path" in _a:
raw_path = str(_a.get("path", "")).strip()
body = str(_a.get("content", ""))
if not isinstance(_a, dict):
return {"error": "write_file: expected a JSON object", "exit_code": 1}
raw_path_value = _a.get("path")
body_value = _a.get("content")
raw_path = raw_path_value.strip() if isinstance(raw_path_value, str) else ""
if not isinstance(body_value, str):
return {"error": "write_file: content required", "exit_code": 1}
body = body_value
except (json.JSONDecodeError, TypeError, ValueError):
pass
return {"error": "write_file: expected valid JSON arguments", "exit_code": 1}
if not raw_path:
return {"error": "write_file: path required", "exit_code": 1}
try:
path = _resolve_tool_path(raw_path)
except ValueError as e:
return {"error": f"write_file: {e}", "exit_code": 1}
# A frequent multimodal artifact failure is writing SVG markup to a
# path whose extension promises a raster image. The file exists, so
# ordinary artifact checks pass, but image judges cannot decode it.
# Reject the mismatch with an actionable native-tool recovery path:
# save the SVG with an .svg suffix, then use inspect_media to render
# it to the requested PNG/JPEG path.
image_suffixes = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp"}
body_probe = body.lstrip().casefold()
if os.path.splitext(path)[1].casefold() in image_suffixes and (
body_probe.startswith("<svg")
or (body_probe.startswith("<?xml") and "<svg" in body_probe[:2000])
):
return {
"error": (
f"write_file: {path} contains SVG markup but has a raster "
"image extension. Write the SVG to a .svg path first, "
"then call inspect_media with that SVG as path and this "
"path as output_path to render a real raster image."
),
"exit_code": 1,
"artifact_format_error": True,
}
if not body:
return {
"error": (
f"write_file: {path}: content required; refusing to create an "
"empty file. Call write_file again with the exact filename and "
"non-empty content. If you need a directory, create it from "
"bash/python instead."
),
"exit_code": 1,
}
try:
def _write():
old = ""
@@ -280,16 +350,77 @@ class ApplyPatchTool:
new = _apply_patch_hunks(old, op["hunks"], op["path"])
prepared.append((kind, path, old, new))
staged: list[tuple[str, str]] = []
backups: list[tuple[str, str | None]] = []
try:
for kind, path, _old, new in prepared:
if kind == "delete":
continue
directory = os.path.dirname(path) or "."
os.makedirs(directory, exist_ok=True)
fd, temp_path = tempfile.mkstemp(
prefix=f".{os.path.basename(path)}.odysseus-",
dir=directory,
)
try:
with os.fdopen(fd, "w", encoding="utf-8", newline="") as handle:
handle.write(new)
handle.flush()
os.fsync(handle.fileno())
if os.path.exists(path):
shutil.copymode(path, temp_path)
except BaseException:
try:
os.unlink(temp_path)
except OSError:
pass
raise
staged.append((path, temp_path))
for _kind, path, _old, _new in prepared:
if os.path.exists(path):
directory = os.path.dirname(path) or "."
fd, backup_path = tempfile.mkstemp(
prefix=f".{os.path.basename(path)}.odysseus-backup-",
dir=directory,
)
os.close(fd)
os.unlink(backup_path)
os.replace(path, backup_path)
backups.append((path, backup_path))
else:
backups.append((path, None))
staged_by_path = dict(staged)
for kind, path, _old, _new in prepared:
if kind != "delete":
os.replace(staged_by_path[path], path)
staged.clear()
except BaseException:
for path, backup_path in reversed(backups):
try:
if os.path.exists(path):
os.unlink(path)
if backup_path and os.path.exists(backup_path):
os.replace(backup_path, path)
except OSError:
pass
raise
finally:
for _path, temp_path in staged:
try:
os.unlink(temp_path)
except OSError:
pass
for _path, backup_path in backups:
if backup_path:
try:
os.unlink(backup_path)
except OSError:
pass
diffs = []
for kind, path, old, new in prepared:
if kind == "delete":
os.remove(path)
else:
directory = os.path.dirname(path)
if directory:
os.makedirs(directory, exist_ok=True)
with open(path, "w", encoding="utf-8") as f:
f.write(new)
for _kind, path, old, new in prepared:
diff = _unified_diff(old, new, path)
if diff:
diffs.append(diff)
@@ -407,7 +538,7 @@ def _apply_patch_hunks(original: str, hunks: List[List[str]], label: str) -> str
class LsTool:
async def execute(self, content: str, ctx: dict) -> dict:
from src.tool_execution import _resolve_tool_path, _resolve_search_root, _truncate
from src.tool_execution import _display_tool_path, _resolve_search_root, _truncate
raw_path = ""
_s = (content or "").strip()
if _s.startswith("{"):
@@ -440,7 +571,7 @@ class LsTool:
except (PermissionError, OSError) as _e:
return None, f"ls: {_e}"
rows.sort(key=lambda r: (not r[0], r[1].lower()))
lines = [f"{root}:"]
lines = [f"{_display_tool_path(root)}:"]
for is_dir, name, size in rows[:_CODENAV_MAX_HITS]:
lines.append(f" {name}/" if is_dir else f" {name} ({size} B)")
if len(rows) > _CODENAV_MAX_HITS:
@@ -458,6 +589,7 @@ class GlobTool:
async def execute(self, content: str, ctx: dict) -> dict:
from src.tool_execution import (
_SENSITIVE_BASENAMES,
_display_tool_path,
_is_sensitive_path,
_resolve_tool_path,
_resolve_search_root,
@@ -549,8 +681,8 @@ class GlobTool:
if err:
return {"error": err, "exit_code": 1}
if not paths:
return {"output": f"No files matching {pattern!r} under {root}", "exit_code": 0}
out = "\n".join(paths)
return {"output": f"No files matching {pattern!r} under {_display_tool_path(root)}", "exit_code": 0}
out = "\n".join(_display_tool_path(path) for path in paths)
if len(paths) >= _CODENAV_MAX_HITS:
out += f"\n... [capped at {_CODENAV_MAX_HITS} files]"
return {"output": _truncate(out), "exit_code": 0}
@@ -559,6 +691,7 @@ class GrepTool:
async def execute(self, content: str, ctx: dict) -> dict:
from src.tool_execution import (
_SENSITIVE_FILE_PATTERNS,
_display_tool_path,
_is_sensitive_path,
_resolve_tool_path,
_resolve_search_root,
@@ -591,9 +724,11 @@ class GrepTool:
def _grep():
import re as _re
import shutil
if not os.path.exists(root):
return None, f"grep: search target not found: {_display_tool_path(root)}"
rg = shutil.which("rg")
if rg:
cmd = [rg, "--line-number", "--no-heading", "--color=never",
cmd = [rg, "--line-number", "--with-filename", "--no-heading", "--color=never",
"--max-count", str(max_hits)]
if ignore_case:
cmd.append("--ignore-case")
@@ -611,6 +746,11 @@ class GrepTool:
try:
import subprocess
p = subprocess.run(cmd, capture_output=True, text=True, timeout=20)
# ripgrep: 0 = matches, 1 = no matches, 2 = failed scan.
# Do not present invalid patterns or IO failures as absence.
if p.returncode not in (0, 1):
detail = (p.stderr or '').strip()[:1200]
return None, f"grep: search failed (exit {p.returncode}): {detail or 'no diagnostic available'}"
lines = [ln for ln in (p.stdout or "").splitlines() if ln][:max_hits]
return lines, None
except subprocess.TimeoutExpired:
@@ -622,11 +762,12 @@ class GrepTool:
except _re.error as _e:
return None, f"grep: bad pattern: {_e}"
hits = []
scan_errors = []
if os.path.isfile(root):
file_iter = [root]
else:
file_iter = []
for dp, dns, fns in os.walk(root):
for dp, dns, fns in os.walk(root, onerror=scan_errors.append):
dns[:] = [d for d in dns if d not in _CODENAV_SKIP_DIRS]
for fn in fns:
if glob_pat and not fnmatch.fnmatch(fn, glob_pat):
@@ -635,25 +776,38 @@ class GrepTool:
for fp in file_iter:
if len(hits) >= max_hits:
break
if _is_sensitive_path(os.path.realpath(fp)):
try:
resolved_file = _resolve_tool_path(fp)
except ValueError:
# Apply the same workspace/sensitive-path checks to each
# discovered file, not just the initial search directory.
continue
try:
with open(fp, "r", encoding="utf-8", errors="strict") as f:
with open(resolved_file, "r", encoding="utf-8", errors="strict") as f:
for i, line in enumerate(f, 1):
if rx.search(line):
hits.append(f"{fp}:{i}:{line.rstrip()[:_CODENAV_MAX_LINE]}")
if len(hits) >= max_hits:
break
except (UnicodeDecodeError, OSError):
except UnicodeDecodeError:
continue
except OSError as error:
scan_errors.append(error)
if scan_errors:
return None, "grep: search incomplete; one or more files or directories could not be read"
return hits, None
lines, err = await asyncio.to_thread(_grep)
if err:
return {"error": err, "exit_code": 1}
if not lines:
return {"output": f"No matches for {pattern!r} under {root}", "exit_code": 0}
out = "\n".join(ln[:_CODENAV_MAX_LINE] for ln in lines)
return {"output": f"No matches for {pattern!r} under {_display_tool_path(root)}", "exit_code": 0}
physical_root = os.path.realpath(root)
display_root = _display_tool_path(physical_root)
out = "\n".join(
(display_root + ln[len(physical_root):] if ln.startswith(physical_root) else ln)[:_CODENAV_MAX_LINE]
for ln in lines
)
if len(lines) >= max_hits:
out += f"\n... [capped at {max_hits} matches]"
return {"output": _truncate(out), "exit_code": 0}
@@ -666,7 +820,7 @@ class GetWorkspaceTool:
ws = get_active_workspace()
if ws:
return {
"output": f"{ws}\n(File tools are confined to this folder; the shell starts "
"output": "/workspace\n(File tools are confined to this folder; the shell starts "
f"here but is not sandboxed and can reach outside it.)",
"exit_code": 0,
}