From 5fd911488231eaedb9cb20b79b67624b23053985 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?L=C3=A9o?= Date: Wed, 30 Sep 2026 17:11:10 +0200 Subject: [PATCH 1/3] test(runtime): pin negative capability wording, test-only First slice of the runtime regression lane. Drives stream_agent_loop with a fake model and asserts on the tools the runtime offers, which is its decision about what the turn may do. No production runtime code is touched and no benchmark fixture or allowlist is imported. The fake-model pattern is the one tests/test_tool_policy.py already uses: patch stream_llm_with_fallback and inspect the tools kwarg. Measured on lab@c499c01b, negative web wording is only partially detected: "Answer from memory only, don't search online." web tools withheld "Summarise what you already know. Do not search the web." web tools OFFERED "No web search please, just tell me what you know..." web tools OFFERED The case that holds is a plain regression guard. The two that do not are xfail(strict=True): they run on every suite, document the target, and fail the moment the behaviour lands so the marker gets removed rather than lingering. A positive control keeps the guard from being satisfied by removing the web tools altogether. --- tests/test_runtime_behavior_regressions.py | 121 +++++++++++++++++++++ 1 file changed, 121 insertions(+) create mode 100644 tests/test_runtime_behavior_regressions.py diff --git a/tests/test_runtime_behavior_regressions.py b/tests/test_runtime_behavior_regressions.py new file mode 100644 index 000000000..389f8dfdf --- /dev/null +++ b/tests/test_runtime_behavior_regressions.py @@ -0,0 +1,121 @@ +"""Deterministic regression coverage for runtime behaviours, test-only. + +These drive ``stream_agent_loop`` with a fake model so the assertions are about +what the runtime offers and does, not about what a model happens to answer. +Nothing here imports benchmark fixtures or allowlists, and nothing here touches +production runtime code: the lane exists so the implementation side can move +without losing the behaviours underneath it. + +The fake-model pattern is the one already used by tests/test_tool_policy.py: +patch ``stream_llm_with_fallback`` and inspect the ``tools`` kwarg the loop +hands it, which is the runtime's decision about what the turn may do. +""" + +import asyncio +import json + +import pytest + +import src.agent_loop as al + + +def _collect(gen): + async def _run(): + return [c async for c in gen] + + return asyncio.run(_run()) + + +def _delta_chunk(text): + payload = {"choices": [{"delta": {"content": text}}]} + return f"data: {json.dumps(payload)}\n\n" + + +def _schema_names(tools): + return { + tool.get("function", {}).get("name") or tool.get("name") + for tool in (tools or []) + } + + +def _patch_loop_basics(monkeypatch): + monkeypatch.setattr(al, "get_setting", lambda key, default=None: default, raising=False) + monkeypatch.setattr(al, "get_mcp_manager", lambda: None, raising=False) + monkeypatch.setattr(al, "estimate_tokens", lambda *a, **k: 10, raising=False) + + +def _run_turn(monkeypatch, messages, **kwargs): + """Drive one agent turn and return the tool sets offered to the model.""" + _patch_loop_basics(monkeypatch) + offered = [] + + async def _fake_stream(_candidates, _messages, **kw): + offered.append(kw.get("tools")) + yield _delta_chunk("ok") + yield "data: [DONE]\n\n" + + monkeypatch.setattr(al, "stream_llm_with_fallback", _fake_stream, raising=False) + _collect( + al.stream_agent_loop( + "http://local.test/v1", + "moonshotai/kimi-k3", + messages, + max_rounds=kwargs.pop("max_rounds", 1), + relevant_tools=kwargs.pop("relevant_tools", {"ask_user", "update_plan"}), + owner=kwargs.pop("owner", "sft_alex_creator"), + **kwargs, + ) + ) + return offered + + +# ── negative capability wording ───────────────────────────────────────────── +# A turn that says not to search must not be handed the search tools. The +# failure this guards is a model that obeys the wording while the runtime +# contradicted it by offering the tool anyway. +# +# Measured on lab@c499c01b: the detection is partial. "don't search online" is +# caught; "Do not search the web" and "No web search please" are not, and the +# web tools are offered for both. The two that do not hold yet are marked +# xfail(strict=True), so they document the target, run on every suite, and fail +# loudly the moment the behaviour lands. Delete the marker then. + +HELD = ["Answer from memory only, don't search online."] + +NOT_HELD_YET = [ + "Summarise what you already know. Do not search the web.", + "No web search please, just tell me what you know about Python decorators.", +] + + +@pytest.mark.parametrize("phrasing", HELD) +def test_negative_web_wording_withholds_the_web_tools(monkeypatch, phrasing): + offered = _run_turn(monkeypatch, [{"role": "user", "content": phrasing}]) + + names = _schema_names(offered[0]) + assert "web_search" not in names, f"web_search offered despite: {phrasing!r}" + assert "web_fetch" not in names, f"web_fetch offered despite: {phrasing!r}" + + +@pytest.mark.xfail( + strict=True, + reason="negative web wording is only partially detected on lab@c499c01b; " + "these phrasings still get the web tools offered", +) +@pytest.mark.parametrize("phrasing", NOT_HELD_YET) +def test_negative_web_wording_withholds_the_web_tools_unhandled(monkeypatch, phrasing): + offered = _run_turn(monkeypatch, [{"role": "user", "content": phrasing}]) + + names = _schema_names(offered[0]) + assert "web_search" not in names, f"web_search offered despite: {phrasing!r}" + assert "web_fetch" not in names, f"web_fetch offered despite: {phrasing!r}" + + +def test_plain_web_request_still_offers_search(monkeypatch): + """The guard above must not become a blanket removal of the web tools.""" + offered = _run_turn( + monkeypatch, + [{"role": "user", "content": "Search the web for the latest Python release."}], + ) + + assert "web_search" in _schema_names(offered[0]) From d423632559ab7331ff8c7c237ae42797e58854e4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?L=C3=A9o?= Date: Wed, 30 Sep 2026 17:19:07 +0200 Subject: [PATCH 2/3] test(runtime): scope the negative-wording claim to the inferred path Verified end to end against a local Qwen3.5-9B Q4_K_M that when the user explicitly enables web for the turn, none of the three phrasings withholds web_search, web_fetch or private_browser, including the one these tests record as held. The held case holds on the inferred path only, where no toggle is set and the runtime decides from intent. That distinction was missing and the file read as a stronger claim than the measurement supports. An explicit toggle beating an inferred negative may be the intended semantics, so it is recorded rather than asserted. --- tests/test_runtime_behavior_regressions.py | 20 +++++++++++++++----- 1 file changed, 15 insertions(+), 5 deletions(-) diff --git a/tests/test_runtime_behavior_regressions.py b/tests/test_runtime_behavior_regressions.py index 389f8dfdf..6c991fdb7 100644 --- a/tests/test_runtime_behavior_regressions.py +++ b/tests/test_runtime_behavior_regressions.py @@ -74,11 +74,21 @@ def _run_turn(monkeypatch, messages, **kwargs): # failure this guards is a model that obeys the wording while the runtime # contradicted it by offering the tool anyway. # -# Measured on lab@c499c01b: the detection is partial. "don't search online" is -# caught; "Do not search the web" and "No web search please" are not, and the -# web tools are offered for both. The two that do not hold yet are marked -# xfail(strict=True), so they document the target, run on every suite, and fail -# loudly the moment the behaviour lands. Delete the marker then. +# SCOPE, and it matters: these cover the inferred path, where the turn has no +# explicit web toggle and the runtime decides from intent. Measured on +# lab@c499c01b, detection there is partial: "don't search online" suppresses +# the intent and the web tools are withheld; "Do not search the web" and "No +# web search please" do not, and the tools are offered. +# +# When the user explicitly enables web for the turn, wording does not withhold +# anything: confirmed end to end against a local Qwen3.5-9B Q4_K_M, where all +# three phrasings were offered web_search, web_fetch and private_browser. That +# may well be correct, an explicit toggle beating an inferred negative, so it +# is recorded here rather than asserted either way. +# +# The two inferred-path cases that do not hold are xfail(strict=True): they +# document the target, run on every suite, and fail the moment the behaviour +# lands. Delete the marker then. HELD = ["Answer from memory only, don't search online."] From bf5d8e400119661da06f2b825c596ab1d4106dc6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?L=C3=A9o?= Date: Wed, 30 Sep 2026 17:38:57 +0200 Subject: [PATCH 3/3] test(runtime): cover the remaining four requested behaviours Completes the lane alteixeira20 asked for. Test-only: tests/ and test helpers, no production runtime code, no benchmark fixtures or allowlists. Supplied workspace context must not produce a clarification. Pins _looks_like_unattended_clarification on four shapes that hand the decision back ("could you please share", "shall I", "which approach do you prefer") and three ordinary answers that must not trip it. Repeated update_plan is not the turn's work. ask_user and update_plan are permitted on nearly every turn, so if they counted as execution a model could loop on them and look busy. Pins that _tool_rejection_reason does not advertise either as an available tool, and that update_plan is permitted without ever being in required. Request-scoped tool authority. _request_scoped_allowed_tool_names must not make an undeclared tool executable; the native-terminal widening is pinned separately so it stays opt-in rather than drifting into the default. Foreign-process safety. The Chrome sweep matches this runtime's own profile prefix, so a fake procfs with our pid, a user's ordinary Chrome and another worktree's agent browser must leave exactly two of the three alone. 14 passed, 2 xfailed. The xfails are the negative-wording cases from the first commit that do not hold yet. --- tests/test_runtime_behavior_regressions.py | 137 +++++++++++++++++++++ 1 file changed, 137 insertions(+) diff --git a/tests/test_runtime_behavior_regressions.py b/tests/test_runtime_behavior_regressions.py index 6c991fdb7..84ca3ad15 100644 --- a/tests/test_runtime_behavior_regressions.py +++ b/tests/test_runtime_behavior_regressions.py @@ -17,6 +17,7 @@ import json import pytest import src.agent_loop as al +import src.agent_tools.web_tools as al_web def _collect(gen): @@ -69,6 +70,30 @@ def _run_turn(monkeypatch, messages, **kwargs): return offered + +def _contract(offered=("ask_user", "update_plan", "manage_notes"), + required=("manage_notes",)): + """A minimal valid TurnContract. + + The dataclass validates required <= offered <= executable and that the + schema inventory matches offered exactly, so the schemas are built from + the same names rather than hand-written. + """ + from src.turn_contract import TurnContract + + return TurnContract( + capabilities=frozenset({"notes"}), + required=frozenset(required), + offered=frozenset(offered), + executable=frozenset(offered), + unavailable=frozenset(), + schema_json=tuple( + json.dumps({"type": "function", "function": {"name": n, "parameters": {}}}) + for n in offered + ), + ) + + # ── negative capability wording ───────────────────────────────────────────── # A turn that says not to search must not be handed the search tools. The # failure this guards is a model that obeys the wording while the runtime @@ -129,3 +154,115 @@ def test_plain_web_request_still_offers_search(monkeypatch): ) assert "web_search" in _schema_names(offered[0]) + + +# ── supplied workspace context must not produce a clarification ───────────── +# When the turn already carries what it needs, an answer that hands the next +# decision back to the user is a failed turn, not a polite one. The runtime +# detects that shape; these pin the detector so a reworded prompt cannot slip +# past it silently. + +@pytest.mark.parametrize( + "answer", + [ + "Could you please share the file you want me to edit?", + "Would you like me to go ahead and refactor it?", + "Shall I start with the parser?", + "Please let me know which approach you prefer.", + ], +) +def test_handing_the_decision_back_is_recognised_as_clarification(answer): + assert al._looks_like_unattended_clarification(answer) is True + + +@pytest.mark.parametrize( + "answer", + [ + "I read config.py and the timeout is set to 30 seconds.", + "The parser fails on empty input because it indexes before checking length.", + "Done. The workspace now has three files.", + ], +) +def test_ordinary_answers_are_not_clarifications(answer): + assert al._looks_like_unattended_clarification(answer) is False + + +# ── repeated update_plan is not the turn's actionable work ───────────────── +# update_plan and ask_user are permitted on almost every turn, so if they +# counted as execution a model could loop on them forever and look busy. The +# runtime must not advertise them as the tools that satisfy the request. + +def test_plan_and_ask_are_not_advertised_as_the_turns_available_tools(): + contract = _contract() + + reason = al._tool_rejection_reason("web_search", set(), None, contract=contract) + + assert "manage_notes" in reason + assert "update_plan" not in reason, "update_plan advertised as actionable work" + assert "ask_user" not in reason, "ask_user advertised as actionable work" + + +def test_update_plan_is_permitted_but_never_the_requirement(): + contract = _contract() + + assert contract.permits("update_plan") is True + assert "update_plan" not in contract.required + + +# ── request-scoped tool authority ────────────────────────────────────────── +# An external contract names what the request may do. A tool the caller never +# declared must not become executable just because the runtime knows it. + +def test_request_scope_excludes_tools_the_caller_never_declared(): + declared = [{"function": {"name": "write_file"}}] + offered = [{"function": {"name": "write_file"}}, {"function": {"name": "bash"}}] + + allowed = al._request_scoped_allowed_tool_names( + declared, offered, native_terminal_runtime=False + ) + + assert allowed == {"write_file"} + assert "bash" not in allowed, "an undeclared tool became executable" + + +def test_native_terminal_runtime_adds_offered_tools_deliberately(): + """The widening exists, so pin it: it is opt-in, not the default.""" + declared = [{"function": {"name": "write_file"}}] + offered = [{"function": {"name": "write_file"}}, {"function": {"name": "bash"}}] + + allowed = al._request_scoped_allowed_tool_names( + declared, offered, native_terminal_runtime=True + ) + + assert allowed == {"write_file", "bash"} + + +# ── owned process cleanup, foreign-process safety ────────────────────────── +# The Chrome sweep matches on this runtime's own profile prefix. A browser +# belonging to the user, or to another worktree, must survive it. + +def test_chrome_sweep_kills_only_this_runtimes_profile(monkeypatch, tmp_path): + from src.agent_tools.web_tools import PrivateBrowserTool + + proc = tmp_path / "proc" + tmpdir = tmp_path / "runtime-tmp" + tmpdir.mkdir() + ours = str(tmpdir.resolve() / "agent-browser-chrome-") + + def _pid(pid, cmdline): + entry = proc / pid + entry.mkdir(parents=True) + (entry / "cmdline").write_bytes(cmdline.replace(" ", "\0").encode()) + + _pid("101", f"chrome --user-data-dir={ours}session-a") + _pid("202", "chrome --user-data-dir=/Users/someone/Library/Chrome") + _pid("303", "chrome --user-data-dir=/tmp/other-worktree/agent-browser-chrome-x") + (proc / "self").mkdir() + + monkeypatch.setattr(al_web, "_PROC_ROOT", proc) + killed = [] + monkeypatch.setattr(al_web.os, "kill", lambda pid, sig: killed.append(pid)) + + PrivateBrowserTool._terminate_owned_chrome({"TMPDIR": str(tmpdir)}) + + assert killed == [101], f"swept a process that was not ours: {killed}"