mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 06:52:20 +02:00
Completes the lane alteixeira20 asked for. Test-only: tests/ and test helpers,
no production runtime code, no benchmark fixtures or allowlists.
Supplied workspace context must not produce a clarification. Pins
_looks_like_unattended_clarification on four shapes that hand the decision
back ("could you please share", "shall I", "which approach do you prefer")
and three ordinary answers that must not trip it.
Repeated update_plan is not the turn's work. ask_user and update_plan are
permitted on nearly every turn, so if they counted as execution a model could
loop on them and look busy. Pins that _tool_rejection_reason does not
advertise either as an available tool, and that update_plan is permitted
without ever being in required.
Request-scoped tool authority. _request_scoped_allowed_tool_names must not
make an undeclared tool executable; the native-terminal widening is pinned
separately so it stays opt-in rather than drifting into the default.
Foreign-process safety. The Chrome sweep matches this runtime's own profile
prefix, so a fake procfs with our pid, a user's ordinary Chrome and another
worktree's agent browser must leave exactly two of the three alone.
14 passed, 2 xfailed. The xfails are the negative-wording cases from the first
commit that do not hold yet.
269 lines
10 KiB
Python
269 lines
10 KiB
Python
"""Deterministic regression coverage for runtime behaviours, test-only.
|
|
|
|
These drive ``stream_agent_loop`` with a fake model so the assertions are about
|
|
what the runtime offers and does, not about what a model happens to answer.
|
|
Nothing here imports benchmark fixtures or allowlists, and nothing here touches
|
|
production runtime code: the lane exists so the implementation side can move
|
|
without losing the behaviours underneath it.
|
|
|
|
The fake-model pattern is the one already used by tests/test_tool_policy.py:
|
|
patch ``stream_llm_with_fallback`` and inspect the ``tools`` kwarg the loop
|
|
hands it, which is the runtime's decision about what the turn may do.
|
|
"""
|
|
|
|
import asyncio
|
|
import json
|
|
|
|
import pytest
|
|
|
|
import src.agent_loop as al
|
|
import src.agent_tools.web_tools as al_web
|
|
|
|
|
|
def _collect(gen):
|
|
async def _run():
|
|
return [c async for c in gen]
|
|
|
|
return asyncio.run(_run())
|
|
|
|
|
|
def _delta_chunk(text):
|
|
payload = {"choices": [{"delta": {"content": text}}]}
|
|
return f"data: {json.dumps(payload)}\n\n"
|
|
|
|
|
|
def _schema_names(tools):
|
|
return {
|
|
tool.get("function", {}).get("name") or tool.get("name")
|
|
for tool in (tools or [])
|
|
}
|
|
|
|
|
|
def _patch_loop_basics(monkeypatch):
|
|
monkeypatch.setattr(al, "get_setting", lambda key, default=None: default, raising=False)
|
|
monkeypatch.setattr(al, "get_mcp_manager", lambda: None, raising=False)
|
|
monkeypatch.setattr(al, "estimate_tokens", lambda *a, **k: 10, raising=False)
|
|
|
|
|
|
def _run_turn(monkeypatch, messages, **kwargs):
|
|
"""Drive one agent turn and return the tool sets offered to the model."""
|
|
_patch_loop_basics(monkeypatch)
|
|
offered = []
|
|
|
|
async def _fake_stream(_candidates, _messages, **kw):
|
|
offered.append(kw.get("tools"))
|
|
yield _delta_chunk("ok")
|
|
yield "data: [DONE]\n\n"
|
|
|
|
monkeypatch.setattr(al, "stream_llm_with_fallback", _fake_stream, raising=False)
|
|
_collect(
|
|
al.stream_agent_loop(
|
|
"http://local.test/v1",
|
|
"moonshotai/kimi-k3",
|
|
messages,
|
|
max_rounds=kwargs.pop("max_rounds", 1),
|
|
relevant_tools=kwargs.pop("relevant_tools", {"ask_user", "update_plan"}),
|
|
owner=kwargs.pop("owner", "sft_alex_creator"),
|
|
**kwargs,
|
|
)
|
|
)
|
|
return offered
|
|
|
|
|
|
|
|
def _contract(offered=("ask_user", "update_plan", "manage_notes"),
|
|
required=("manage_notes",)):
|
|
"""A minimal valid TurnContract.
|
|
|
|
The dataclass validates required <= offered <= executable and that the
|
|
schema inventory matches offered exactly, so the schemas are built from
|
|
the same names rather than hand-written.
|
|
"""
|
|
from src.turn_contract import TurnContract
|
|
|
|
return TurnContract(
|
|
capabilities=frozenset({"notes"}),
|
|
required=frozenset(required),
|
|
offered=frozenset(offered),
|
|
executable=frozenset(offered),
|
|
unavailable=frozenset(),
|
|
schema_json=tuple(
|
|
json.dumps({"type": "function", "function": {"name": n, "parameters": {}}})
|
|
for n in offered
|
|
),
|
|
)
|
|
|
|
|
|
# ── negative capability wording ─────────────────────────────────────────────
|
|
# A turn that says not to search must not be handed the search tools. The
|
|
# failure this guards is a model that obeys the wording while the runtime
|
|
# contradicted it by offering the tool anyway.
|
|
#
|
|
# SCOPE, and it matters: these cover the inferred path, where the turn has no
|
|
# explicit web toggle and the runtime decides from intent. Measured on
|
|
# lab@c499c01b, detection there is partial: "don't search online" suppresses
|
|
# the intent and the web tools are withheld; "Do not search the web" and "No
|
|
# web search please" do not, and the tools are offered.
|
|
#
|
|
# When the user explicitly enables web for the turn, wording does not withhold
|
|
# anything: confirmed end to end against a local Qwen3.5-9B Q4_K_M, where all
|
|
# three phrasings were offered web_search, web_fetch and private_browser. That
|
|
# may well be correct, an explicit toggle beating an inferred negative, so it
|
|
# is recorded here rather than asserted either way.
|
|
#
|
|
# The two inferred-path cases that do not hold are xfail(strict=True): they
|
|
# document the target, run on every suite, and fail the moment the behaviour
|
|
# lands. Delete the marker then.
|
|
|
|
HELD = ["Answer from memory only, don't search online."]
|
|
|
|
NOT_HELD_YET = [
|
|
"Summarise what you already know. Do not search the web.",
|
|
"No web search please, just tell me what you know about Python decorators.",
|
|
]
|
|
|
|
|
|
@pytest.mark.parametrize("phrasing", HELD)
|
|
def test_negative_web_wording_withholds_the_web_tools(monkeypatch, phrasing):
|
|
offered = _run_turn(monkeypatch, [{"role": "user", "content": phrasing}])
|
|
|
|
names = _schema_names(offered[0])
|
|
assert "web_search" not in names, f"web_search offered despite: {phrasing!r}"
|
|
assert "web_fetch" not in names, f"web_fetch offered despite: {phrasing!r}"
|
|
|
|
|
|
@pytest.mark.xfail(
|
|
strict=True,
|
|
reason="negative web wording is only partially detected on lab@c499c01b; "
|
|
"these phrasings still get the web tools offered",
|
|
)
|
|
@pytest.mark.parametrize("phrasing", NOT_HELD_YET)
|
|
def test_negative_web_wording_withholds_the_web_tools_unhandled(monkeypatch, phrasing):
|
|
offered = _run_turn(monkeypatch, [{"role": "user", "content": phrasing}])
|
|
|
|
names = _schema_names(offered[0])
|
|
assert "web_search" not in names, f"web_search offered despite: {phrasing!r}"
|
|
assert "web_fetch" not in names, f"web_fetch offered despite: {phrasing!r}"
|
|
|
|
|
|
def test_plain_web_request_still_offers_search(monkeypatch):
|
|
"""The guard above must not become a blanket removal of the web tools."""
|
|
offered = _run_turn(
|
|
monkeypatch,
|
|
[{"role": "user", "content": "Search the web for the latest Python release."}],
|
|
)
|
|
|
|
assert "web_search" in _schema_names(offered[0])
|
|
|
|
|
|
# ── supplied workspace context must not produce a clarification ─────────────
|
|
# When the turn already carries what it needs, an answer that hands the next
|
|
# decision back to the user is a failed turn, not a polite one. The runtime
|
|
# detects that shape; these pin the detector so a reworded prompt cannot slip
|
|
# past it silently.
|
|
|
|
@pytest.mark.parametrize(
|
|
"answer",
|
|
[
|
|
"Could you please share the file you want me to edit?",
|
|
"Would you like me to go ahead and refactor it?",
|
|
"Shall I start with the parser?",
|
|
"Please let me know which approach you prefer.",
|
|
],
|
|
)
|
|
def test_handing_the_decision_back_is_recognised_as_clarification(answer):
|
|
assert al._looks_like_unattended_clarification(answer) is True
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"answer",
|
|
[
|
|
"I read config.py and the timeout is set to 30 seconds.",
|
|
"The parser fails on empty input because it indexes before checking length.",
|
|
"Done. The workspace now has three files.",
|
|
],
|
|
)
|
|
def test_ordinary_answers_are_not_clarifications(answer):
|
|
assert al._looks_like_unattended_clarification(answer) is False
|
|
|
|
|
|
# ── repeated update_plan is not the turn's actionable work ─────────────────
|
|
# update_plan and ask_user are permitted on almost every turn, so if they
|
|
# counted as execution a model could loop on them forever and look busy. The
|
|
# runtime must not advertise them as the tools that satisfy the request.
|
|
|
|
def test_plan_and_ask_are_not_advertised_as_the_turns_available_tools():
|
|
contract = _contract()
|
|
|
|
reason = al._tool_rejection_reason("web_search", set(), None, contract=contract)
|
|
|
|
assert "manage_notes" in reason
|
|
assert "update_plan" not in reason, "update_plan advertised as actionable work"
|
|
assert "ask_user" not in reason, "ask_user advertised as actionable work"
|
|
|
|
|
|
def test_update_plan_is_permitted_but_never_the_requirement():
|
|
contract = _contract()
|
|
|
|
assert contract.permits("update_plan") is True
|
|
assert "update_plan" not in contract.required
|
|
|
|
|
|
# ── request-scoped tool authority ──────────────────────────────────────────
|
|
# An external contract names what the request may do. A tool the caller never
|
|
# declared must not become executable just because the runtime knows it.
|
|
|
|
def test_request_scope_excludes_tools_the_caller_never_declared():
|
|
declared = [{"function": {"name": "write_file"}}]
|
|
offered = [{"function": {"name": "write_file"}}, {"function": {"name": "bash"}}]
|
|
|
|
allowed = al._request_scoped_allowed_tool_names(
|
|
declared, offered, native_terminal_runtime=False
|
|
)
|
|
|
|
assert allowed == {"write_file"}
|
|
assert "bash" not in allowed, "an undeclared tool became executable"
|
|
|
|
|
|
def test_native_terminal_runtime_adds_offered_tools_deliberately():
|
|
"""The widening exists, so pin it: it is opt-in, not the default."""
|
|
declared = [{"function": {"name": "write_file"}}]
|
|
offered = [{"function": {"name": "write_file"}}, {"function": {"name": "bash"}}]
|
|
|
|
allowed = al._request_scoped_allowed_tool_names(
|
|
declared, offered, native_terminal_runtime=True
|
|
)
|
|
|
|
assert allowed == {"write_file", "bash"}
|
|
|
|
|
|
# ── owned process cleanup, foreign-process safety ──────────────────────────
|
|
# The Chrome sweep matches on this runtime's own profile prefix. A browser
|
|
# belonging to the user, or to another worktree, must survive it.
|
|
|
|
def test_chrome_sweep_kills_only_this_runtimes_profile(monkeypatch, tmp_path):
|
|
from src.agent_tools.web_tools import PrivateBrowserTool
|
|
|
|
proc = tmp_path / "proc"
|
|
tmpdir = tmp_path / "runtime-tmp"
|
|
tmpdir.mkdir()
|
|
ours = str(tmpdir.resolve() / "agent-browser-chrome-")
|
|
|
|
def _pid(pid, cmdline):
|
|
entry = proc / pid
|
|
entry.mkdir(parents=True)
|
|
(entry / "cmdline").write_bytes(cmdline.replace(" ", "\0").encode())
|
|
|
|
_pid("101", f"chrome --user-data-dir={ours}session-a")
|
|
_pid("202", "chrome --user-data-dir=/Users/someone/Library/Chrome")
|
|
_pid("303", "chrome --user-data-dir=/tmp/other-worktree/agent-browser-chrome-x")
|
|
(proc / "self").mkdir()
|
|
|
|
monkeypatch.setattr(al_web, "_PROC_ROOT", proc)
|
|
killed = []
|
|
monkeypatch.setattr(al_web.os, "kill", lambda pid, sig: killed.append(pid))
|
|
|
|
PrivateBrowserTool._terminate_owned_chrome({"TMPDIR": str(tmpdir)})
|
|
|
|
assert killed == [101], f"swept a process that was not ours: {killed}"
|