mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 06:52:20 +02:00
Three of the six recorded failures were the same test bug: an unresolved /tmp path compared against a resolved /private/tmp one. macOS makes /tmp a symlink, so a fixture built with tempfile.mkdtemp(dir="/tmp") and a code path that resolves what it reports disagree about a file both found correctly. test_code_nav_tools builds its fixture unresolved and compares it against the reported path. One realpath fixes both of its failures. test_glob_confined_e2e is the same cause through a longer route: it mixed os.path.realpath(ws) with an unresolved secret directory, so relpath emitted "../../../../tmp/<absolute path>" and the assertion that the absolute path was absent from the output matched it as a substring. Resolving the secret directory puts both sides in one tree and the relative path stays short. macOS full suite goes from 6 failures to 3. The remaining three are an ffmpeg build without a WebP encoder, a socket test that needs a fast connection refusal, and the rich-text colour test that is still unexplained. The ledger is updated in the same change so it does not describe failures that no longer happen.
248 lines
8.9 KiB
Python
248 lines
8.9 KiB
Python
"""Tests for the code-navigation tools (grep, glob, ls) + read_file line range."""
|
|
import os
|
|
import shutil
|
|
import asyncio
|
|
import tempfile
|
|
import pytest
|
|
|
|
os.environ.setdefault("DATABASE_URL", "sqlite:////tmp/test_code_nav.db")
|
|
|
|
from src.tool_execution import _direct_fallback
|
|
|
|
|
|
def _run(tool, content):
|
|
return asyncio.run(_direct_fallback(tool, content))
|
|
|
|
|
|
@pytest.fixture
|
|
def repo():
|
|
# Built under /tmp, which is on the default tool-path allowlist.
|
|
# realpath because the code under test resolves the path it reports, and on
|
|
# macOS /tmp is a symlink to /private/tmp: comparing the unresolved path
|
|
# against the resolved one fails on a file both sides found correctly.
|
|
root = os.path.realpath(tempfile.mkdtemp(dir="/tmp", prefix="codenav_"))
|
|
try:
|
|
with open(os.path.join(root, "a.py"), "w") as f:
|
|
f.write("import os\n# needle here\nprint('x')\n")
|
|
os.mkdir(os.path.join(root, "sub"))
|
|
with open(os.path.join(root, "sub", "b.txt"), "w") as f:
|
|
f.write("nothing\nNEEDLE upper\n")
|
|
os.mkdir(os.path.join(root, "sub", "deep"))
|
|
with open(os.path.join(root, "sub", "deep", "c.py"), "w") as f:
|
|
f.write("# deep python\n")
|
|
os.mkdir(os.path.join(root, "node_modules"))
|
|
with open(os.path.join(root, "node_modules", "dep.py"), "w") as f:
|
|
f.write("needle in dep\n")
|
|
g = os.path.join(root, ".git")
|
|
os.mkdir(g)
|
|
with open(os.path.join(g, "config"), "w") as f:
|
|
f.write("needle in git\n")
|
|
yield root
|
|
finally:
|
|
shutil.rmtree(root, ignore_errors=True)
|
|
|
|
|
|
# ── grep ──────────────────────────────────────────────────────────────────
|
|
|
|
def test_grep_finds_match(repo):
|
|
r = _run("grep", f'{{"pattern": "needle", "path": "{repo}"}}')
|
|
assert r["exit_code"] == 0
|
|
assert "a.py:2:" in r["output"]
|
|
|
|
|
|
def test_grep_skips_junk_dirs(repo):
|
|
r = _run("grep", f'{{"pattern": "needle", "path": "{repo}"}}')
|
|
assert "node_modules" not in r["output"]
|
|
assert ".git/config" not in r["output"]
|
|
|
|
|
|
def test_grep_ignore_case(repo):
|
|
r = _run("grep", f'{{"pattern": "needle", "ignore_case": true, "path": "{repo}"}}')
|
|
assert "b.txt:2:" in r["output"]
|
|
|
|
|
|
def test_grep_glob_filter(repo):
|
|
r = _run("grep", f'{{"pattern": "needle", "ignore_case": true, "glob": "*.py", "path": "{repo}"}}')
|
|
assert "a.py" in r["output"]
|
|
assert "b.txt" not in r["output"]
|
|
|
|
|
|
def test_grep_no_match(repo):
|
|
r = _run("grep", f'{{"pattern": "zzzznotfound", "path": "{repo}"}}')
|
|
assert r["exit_code"] == 0
|
|
assert "No matches" in r["output"]
|
|
|
|
|
|
def test_grep_requires_pattern(repo):
|
|
r = _run("grep", "{}")
|
|
assert r["exit_code"] == 1
|
|
assert "pattern is required" in r["error"]
|
|
|
|
|
|
def test_grep_path_outside_roots_rejected(repo):
|
|
r = _run("grep", '{"pattern": "x", "path": "/etc"}')
|
|
assert r["exit_code"] == 1
|
|
assert "outside the allowed roots" in r["error"]
|
|
|
|
|
|
def test_grep_python_fallback_when_no_rg(repo, monkeypatch):
|
|
monkeypatch.setattr(shutil, "which", lambda name: None)
|
|
r = _run("grep", f'{{"pattern": "needle", "path": "{repo}"}}')
|
|
assert r["exit_code"] == 0
|
|
assert "a.py:2:" in r["output"]
|
|
assert "node_modules" not in r["output"]
|
|
assert ".git/config" not in r["output"]
|
|
|
|
|
|
@pytest.mark.skipif(shutil.which("rg") is None, reason="targets the ripgrep fast-path")
|
|
def test_grep_skips_case_variant_sensitive_files_rg(repo):
|
|
"""The rg fast-path must exclude deny-listed key files case-insensitively.
|
|
|
|
A file whose name is a case variant of a sensitive pattern (e.g. ID_RSA vs
|
|
id_rsa, Known_Hosts vs known_hosts) points at the same secret on a
|
|
case-insensitive filesystem, so grep must not return its contents. The
|
|
Python fallback already folds case via _is_sensitive_path; a plain --glob
|
|
exclusion is case-sensitive, so it would leak these — this pins the rg path.
|
|
"""
|
|
token = "GREPSECRET_TOKEN_ZZZ"
|
|
with open(os.path.join(repo, "notes.txt"), "w") as f:
|
|
f.write(f"see {token}\n")
|
|
with open(os.path.join(repo, "ID_RSA"), "w") as f:
|
|
f.write(f"PRIVATE {token}\n")
|
|
with open(os.path.join(repo, "Known_Hosts"), "w") as f:
|
|
f.write(f"host {token}\n")
|
|
r = _run("grep", f'{{"pattern": "{token}", "path": "{repo}"}}')
|
|
assert r["exit_code"] == 0
|
|
assert "notes.txt" in r["output"] # ordinary matches still returned
|
|
assert "ID_RSA" not in r["output"] # case-variant key excluded
|
|
assert "Known_Hosts" not in r["output"]
|
|
|
|
|
|
# ── glob ──────────────────────────────────────────────────────────────────
|
|
|
|
def test_glob_py(repo):
|
|
r = _run("glob", f'{{"pattern": "*.py", "path": "{repo}"}}')
|
|
assert r["exit_code"] == 0
|
|
assert "a.py" in r["output"]
|
|
|
|
|
|
def test_glob_recursive_skips_junk(repo):
|
|
r = _run("glob", f'{{"pattern": "**/*.py", "path": "{repo}"}}')
|
|
assert "a.py" in r["output"]
|
|
assert "node_modules" not in r["output"]
|
|
|
|
|
|
def test_glob_requires_pattern(repo):
|
|
r = _run("glob", "{}")
|
|
assert r["exit_code"] == 1
|
|
|
|
|
|
def test_glob_literal_in_subdir(repo):
|
|
"""Bare literal should match at any depth (like rglob), not only at root."""
|
|
r = _run("glob", f'{{"pattern": "b.txt", "path": "{repo}"}}')
|
|
assert r["exit_code"] == 0
|
|
assert "b.txt" in r["output"]
|
|
|
|
|
|
def test_glob_multi_segment_single_star(repo):
|
|
"""sub/*.txt matches sub/b.txt but NOT sub/deep/c.py (single * stays in one segment)."""
|
|
r = _run("glob", f'{{"pattern": "sub/*.txt", "path": "{repo}"}}')
|
|
assert r["exit_code"] == 0
|
|
assert "b.txt" in r["output"]
|
|
assert "c.py" not in r["output"]
|
|
|
|
|
|
def test_glob_star_does_not_cross_slash(repo):
|
|
"""src/*.py must NOT match src/a/b/x.py — * is single-segment only."""
|
|
r = _run("glob", f'{{"pattern": "sub/*.py", "path": "{repo}"}}')
|
|
assert r["exit_code"] == 0
|
|
# sub/ has no .py directly, only sub/deep/c.py — should NOT match
|
|
assert "No files matching" in r["output"]
|
|
|
|
|
|
def test_glob_double_star_matches_deep(repo):
|
|
"""**/*.py should match files at any depth."""
|
|
r = _run("glob", f'{{"pattern": "**/*.py", "path": "{repo}"}}')
|
|
assert r["exit_code"] == 0
|
|
assert "a.py" in r["output"]
|
|
assert "c.py" in r["output"]
|
|
|
|
|
|
# ── ls ────────────────────────────────────────────────────────────────────
|
|
|
|
def test_ls_lists_entries(repo):
|
|
r = _run("ls", f'{{"path": "{repo}"}}')
|
|
assert r["exit_code"] == 0
|
|
assert "a.py" in r["output"]
|
|
assert "sub/" in r["output"]
|
|
assert ".git" not in r["output"] # hidden skipped
|
|
|
|
|
|
def test_ls_path_outside_rejected(repo):
|
|
r = _run("ls", '{"path": "/etc"}')
|
|
assert r["exit_code"] == 1
|
|
assert "outside the allowed roots" in r["error"]
|
|
|
|
|
|
# ── read_file line range ───────────────────────────────────────────────────
|
|
|
|
def test_read_file_offset_limit(repo):
|
|
p = os.path.join(repo, "lines.txt")
|
|
with open(p, "w") as f:
|
|
f.write("\n".join(f"line{i}" for i in range(1, 11)) + "\n")
|
|
r = _run("read_file", f'{{"path": "{p}", "offset": 3, "limit": 2}}')
|
|
assert r["exit_code"] == 0
|
|
assert r["output"] == "line3\nline4\n"
|
|
|
|
|
|
def test_read_file_plain_path_backcompat(repo):
|
|
r = _run("read_file", os.path.join(repo, "a.py"))
|
|
assert r["exit_code"] == 0
|
|
assert "needle" in r["output"]
|
|
|
|
|
|
def test_read_file_extracts_structured_documents(repo, monkeypatch):
|
|
p = os.path.join(repo, "report.docx")
|
|
with open(p, "wb") as f:
|
|
f.write(b"PK fake structured document")
|
|
|
|
calls = []
|
|
|
|
def _fake_extract(path, **kwargs):
|
|
calls.append((path, kwargs))
|
|
return "Heading\nFirst fact\nSecond fact\n"
|
|
|
|
monkeypatch.setattr(
|
|
"src.document_processor.extract_local_document",
|
|
_fake_extract,
|
|
)
|
|
|
|
r = _run("read_file", f'{{"path": "{p}", "offset": 2, "limit": 2}}')
|
|
|
|
assert r == {"output": "First fact\nSecond fact\n", "exit_code": 0}
|
|
assert calls == [(p, {
|
|
"display_name": "report.docx",
|
|
"analyze_embedded_images": False,
|
|
})]
|
|
|
|
|
|
def test_read_file_extracts_legacy_word_documents(repo, monkeypatch):
|
|
p = os.path.join(repo, "report.doc")
|
|
with open(p, "wb") as f:
|
|
f.write(b"binary OLE document")
|
|
|
|
calls = []
|
|
|
|
def _fake_extract(path, **kwargs):
|
|
calls.append((path, kwargs))
|
|
return "Legacy Word content\nShipping price\n"
|
|
|
|
monkeypatch.setattr("src.document_processor.extract_local_document", _fake_extract)
|
|
r = _run("read_file", p)
|
|
|
|
assert r == {"output": "Legacy Word content\nShipping price\n", "exit_code": 0}
|
|
assert calls == [(p, {
|
|
"display_name": "report.doc",
|
|
"analyze_embedded_images": False,
|
|
})]
|