Harden maintainer-preview harness and review fixes

Unify conversational domain routing, preserve artifact completion evidence, replace provisional tool-round prose with terminal synthesis, and resolve verified maintainer review findings across search, frontend module identity, path policy, configuration, and built-in skill startup.
This commit is contained in:
pewdiepie-archdaemon
2026-09-21 06:54:03 +00:00
parent d7cad0621f
commit 297ad19248
54 changed files with 1223 additions and 110 deletions
+49
View File
@@ -15,6 +15,45 @@ def _events(chunks):
return parsed
def test_deepseek_flash_visual_continuation_flattens_only_tool_visual_history():
messages = [
{"role": "system", "content": "system contract"},
{"role": "user", "content": "original task"},
{
"role": "assistant",
"content": None,
"tool_calls": [{"id": "c1", "type": "function", "function": {
"name": "inspect_media", "arguments": "{}",
}}],
},
{"role": "tool", "tool_call_id": "c1", "content": "preview follows"},
{
"role": "user",
"metadata": {"source": "tool visual evidence", "trusted": False},
"content": [
{"type": "text", "text": "Visual evidence returned by tool execution."},
{"type": "image_url", "image_url": {"url": "data:image/jpeg;base64,AAA"}},
],
},
]
flattened = agent_loop._deepseek_flash_visual_continuation(messages, "original task")
assert [message["role"] for message in flattened] == ["system", "user"]
assert "original task" in flattened[-1]["content"][0]["text"]
assert flattened[-1]["content"][1]["image_url"]["url"].endswith("AAA")
assert agent_loop._deepseek_flash_visual_continuation(
messages[:-1], "original task"
) is None
def test_deepseek_flash_vision_compatibility_is_exact_model_only():
assert agent_loop._is_deepseek_flash_vision_model("deepseek-flash")
assert agent_loop._is_deepseek_flash_vision_model("provider/deepseek-flash")
assert not agent_loop._is_deepseek_flash_vision_model("deepseek-v4-pro")
assert not agent_loop._is_deepseek_flash_vision_model("deepseek-flash-preview")
def _patch_loop(monkeypatch, responses, captured_kwargs=None):
monkeypatch.setattr(agent_loop, "get_setting", lambda key, default=None: default)
monkeypatch.setattr(agent_loop, "get_mcp_manager", lambda: None)
@@ -796,3 +835,13 @@ def test_run_the_script_is_not_inferred_when_multiple_scripts_are_named():
)
assert agent_loop._requested_verification_command(request) == ""
def test_inspect_saved_file_requests_artifact_readback_verification():
request = (
"Create /workspace/output/results.csv, inspect the saved file, "
"then summarize completion."
)
assert agent_loop._requested_post_edit_verification(request)
assert agent_loop._requested_artifact_readback(request)
+4 -2
View File
@@ -27,8 +27,10 @@ def test_compact_footer_and_details_show_real_performance_counters():
assert "`${Number(tps).toFixed(2)} tok/s`" in RENDERER
assert "`${Number(ttft).toFixed(3)}s TTFT`" in RENDERER
assert "`${Number(injectedTokens).toLocaleString()} in`" in RENDERER
assert 'Input (all rounds)' in RENDERER
assert 'Injected (first request)' in RENDERER
assert '<span class="ctx-label">Input</span>' in RENDERER
assert '<span class="ctx-label">Injected</span>' in RENDERER
assert 'all rounds' not in RENDERER
assert 'first request' not in RENDERER
assert 'Tool schemas' in RENDERER
assert 'Agent rounds' in RENDERER
assert 'Tool calls' in RENDERER
+110 -7
View File
@@ -382,13 +382,36 @@ def test_native_execution_limits_allow_multi_artifact_work_without_unbounded_rou
def test_compact_preview_honors_configured_interactive_round_limit():
assert interactive_execution_limit(100) == 8
assert interactive_execution_limit(1000) == 8
# A configured budget is the user's explicit instruction, honored up to
# the same 200 ceiling the settings endpoint enforces.
assert interactive_execution_limit(100) == 100
assert interactive_execution_limit(1000) == 200
assert interactive_execution_limit(0) == 1
# No resolvable budget falls back to the bounded default.
assert interactive_execution_limit(None) == 8
assert interactive_execution_limit("invalid") == 8
def test_compact_preview_honors_configured_tool_call_budget():
from src.clean_agent_preview import (
INTERACTIVE_BROWSER_TOOL_CALL_LIMIT,
INTERACTIVE_TOOL_CALL_LIMIT,
UNLIMITED_TOOL_CALL_LIMIT,
interactive_tool_call_limit,
)
# 0 means unlimited, matching the main agent loop's max_tool_calls <= 0.
assert interactive_tool_call_limit(0) == UNLIMITED_TOOL_CALL_LIMIT
assert interactive_tool_call_limit(0, browser_offered=True) == UNLIMITED_TOOL_CALL_LIMIT
# An explicit finite budget is honored in both directions.
assert interactive_tool_call_limit(5) == 5
assert interactive_tool_call_limit(250) == 250
# A malformed value keeps the bounded default.
assert interactive_tool_call_limit("invalid") == INTERACTIVE_TOOL_CALL_LIMIT
assert interactive_tool_call_limit(None, browser_offered=True) == (
INTERACTIVE_BROWSER_TOOL_CALL_LIMIT
)
@pytest.mark.parametrize('text,expected', [
('helo', True), ('Hello!', True), ('thanks', True),
('hello, find the latest news', False), ('thanks, now open the source', False),
@@ -5256,7 +5279,9 @@ async def test_model_choice_budget_preserves_evidence_for_final_answer_without_e
import src.clean_agent_preview as module
# Exercise the boundary deterministically without coupling this recovery
# test to the larger production allowance for interactive turns.
monkeypatch.setattr(module, 'INTERACTIVE_TOOL_CALL_LIMIT', 6)
# The budget is the configured agent_max_tool_calls setting; the module
# constant is only the fallback when no budget is resolvable.
_tool_budget = 6
def call(i):
return {'index': i, 'id': f'call-{i}', 'function': {
'name': 'web_fetch', 'arguments': json.dumps({'url': f'https://example.org/{i}'})}}
@@ -5291,7 +5316,8 @@ async def test_model_choice_budget_preserves_evidence_for_final_answer_without_e
routing_experiment='recent_model_choice')
raw = [chunk async for chunk in stream_preview(
endpoint_url='http://test', model='test', messages=[{'role':'user','content':'Read these public pages.'}],
headers={}, turn_contract=contract, session_id='test', owner='test', disabled_tools=set(), tool_policy=ToolPolicy())]
headers={}, turn_contract=contract, session_id='test', owner='test', disabled_tools=set(),
tool_policy=ToolPolicy(), max_tool_calls=_tool_budget)]
assert len(executions) == 6
assert len(requests) == 3
assert 'tools' not in requests[-1]
@@ -5898,9 +5924,9 @@ async def test_context_overflow_retries_model_request_without_replaying_tool(mon
import src.clean_agent_preview as module
requests, executions = [], []
overflow_request = 2 if recovery_kind == 'none' else 3
_tool_budget = 0
if recovery_kind == 'budget':
monkeypatch.setattr(module, 'INTERACTIVE_TOOL_CALL_LIMIT', 1)
monkeypatch.setattr(module, 'INTERACTIVE_BROWSER_TOOL_CALL_LIMIT', 1)
_tool_budget = 1
async def handle(request):
payload = json.loads(request.content)
requests.append(payload)
@@ -5936,7 +5962,7 @@ async def test_context_overflow_retries_model_request_without_replaying_tool(mon
raw = [chunk async for chunk in stream_preview(
endpoint_url='http://test', model='test', headers={}, turn_contract=contract,
messages=[{'role': 'user', 'content': prompt}], session_id='test', owner='test',
disabled_tools=set(), tool_policy=ToolPolicy())]
disabled_tools=set(), tool_policy=ToolPolicy(), max_tool_calls=_tool_budget)]
assert any('The page was read.' in chunk for chunk in raw)
assert len(executions) == 1
assert len(requests) == overflow_request + 1
@@ -7075,3 +7101,80 @@ async def test_parallel_tool_results_precede_visual_evidence(monkeypatch):
]
assert "Final completion round" in messages[-1]["content"]
assert messages[-2]["content"][1]["type"] == "image_url"
@pytest.mark.asyncio
async def test_tool_round_leadin_is_replaced_by_terminal_synthesis(monkeypatch):
from dataclasses import replace
import src.clean_agent_preview as module
packets = iter([
{'choices': [{'delta': {
'content': "I'll browse IKEA and look at their chairs.",
'tool_calls': [{'index': 0, 'id': 'browse-1', 'function': {
'name': 'private_browser',
'arguments': json.dumps({
'action': 'open',
'url': 'https://www.ikea.com/us/en/cat/armchairs-16239/',
}),
}}],
}}]},
{'choices': [{'delta': {
'reasoning_content': 'I have enough. Now compose the final answer.',
'content': 'The most epic option is the DYVLINGE swivel chair.',
}}]},
])
class Response:
def __init__(self, payload): self.payload = payload
async def __aenter__(self): return self
async def __aexit__(self, *args): pass
def raise_for_status(self): pass
async def aiter_lines(self):
yield 'data: ' + json.dumps(self.payload)
yield 'data: [DONE]'
class Client:
def __init__(self, **kwargs): pass
async def __aenter__(self): return self
async def __aexit__(self, *args): pass
def stream(self, *args, **kwargs): return Response(next(packets))
async def execute(block, **kwargs):
return 'private_browser', {'output': 'IKEA chair evidence', 'exit_code': 0}
monkeypatch.setattr(module.httpx, 'AsyncClient', Client)
monkeypatch.setattr(module, 'execute_tool_block', execute)
schema = next(
item for item in FUNCTION_TOOL_SCHEMAS
if item['function']['name'] == 'private_browser'
)
contract = replace(
resolve_full_inventory_contract(schemas=[schema], policy=ToolPolicy()),
routing_experiment='recent_model_choice',
)
raw = [chunk async for chunk in stream_preview(
endpoint_url='http://test', model='deepseek-test', headers={},
turn_contract=contract,
messages=[{'role': 'user', 'content': 'Go to ikea.com and find the most epic chair.'}],
session_id='test', owner='test', disabled_tools=set(),
tool_policy=ToolPolicy(), max_rounds=3,
)]
events = [json.loads(chunk[6:]) for chunk in raw if '[DONE]' not in chunk]
leadin = next(event for event in events if event.get('delta', '').startswith("I'll browse"))
assert 'replacement_scope' not in leadin
synthesis = next(
event for event in events
if event.get('delta', '').startswith('The most epic option')
)
assert synthesis['render_owner'] == 'streamed'
assert synthesis['replacement_scope'] == 'turn'
metrics = next(event['data'] for event in events if event.get('type') == 'metrics')
assistant_rounds = [
item.get('content')
for item in metrics['clean_v3_turn']
if item.get('role') == 'assistant' and item.get('content')
]
assert assistant_rounds == [
"I'll browse IKEA and look at their chairs.",
'The most epic option is the DYVLINGE swivel chair.',
]
+51
View File
@@ -0,0 +1,51 @@
import sys
import types
import pytest
from src import embeddings
def _install_fastembed(monkeypatch):
calls = []
module = types.ModuleType("fastembed")
class TextEmbedding:
def __init__(self, **kwargs):
calls.append(kwargs)
module.TextEmbedding = TextEmbedding
monkeypatch.setitem(sys.modules, "fastembed", module)
return calls
def test_fastembed_threads_default_is_unchanged(monkeypatch, tmp_path):
calls = _install_fastembed(monkeypatch)
monkeypatch.delenv("FASTEMBED_THREADS", raising=False)
monkeypatch.setattr(embeddings, "FASTEMBED_CACHE_DIR", str(tmp_path))
embeddings.FastEmbedClient(model="test-model")
assert calls == [{"model_name": "test-model", "cache_dir": str(tmp_path)}]
def test_fastembed_threads_can_be_bounded(monkeypatch, tmp_path):
calls = _install_fastembed(monkeypatch)
monkeypatch.setenv("FASTEMBED_THREADS", "2")
monkeypatch.setattr(embeddings, "FASTEMBED_CACHE_DIR", str(tmp_path))
embeddings.FastEmbedClient(model="test-model")
assert calls == [
{"model_name": "test-model", "cache_dir": str(tmp_path), "threads": 2}
]
@pytest.mark.parametrize("value", ["0", "257", "not-a-number"])
def test_fastembed_threads_rejects_invalid_values(monkeypatch, tmp_path, value):
_install_fastembed(monkeypatch)
monkeypatch.setenv("FASTEMBED_THREADS", value)
monkeypatch.setattr(embeddings, "FASTEMBED_CACHE_DIR", str(tmp_path))
with pytest.raises(ValueError, match="FASTEMBED_THREADS"):
embeddings.FastEmbedClient(model="test-model")
+2
View File
@@ -95,6 +95,8 @@ class TestModelSupportsVision:
def test_falls_back_to_name_when_no_endpoint(self):
# No endpoint URL → pure name heuristic.
assert chat_helpers.model_supports_vision("llava-1.6", "") is True
assert chat_helpers.model_supports_vision("deepseek-flash", "") is True
assert chat_helpers.model_supports_vision("deepseek-v4-pro", "") is False
assert chat_helpers.model_supports_vision("mistral-7b", "") is False
def test_falls_back_to_name_when_endpoint_unknown(self, monkeypatch):
+108
View File
@@ -0,0 +1,108 @@
"""Regressions for the 2026-09-19 review fixes.
Each test fails on the pre-fix tree; see ~/odysseus-review-20260919.md.
"""
import json
import pytest
from services.search import core as search_core
from src.agent_loop import _read_file_block_path, _read_file_targets_artifact
from src.turn_contract import personal_store_families, requested_capabilities
@pytest.mark.parametrize("message,family", [
("is there anything new in my inbox?", "email"),
("check my email for news from Bob", "email"),
("show me my latest email updates", "email"),
("give me an update on my tasks", "tasks"),
("what's new on my calendar this week?", "calendar"),
("catch me up on my meetings", "calendar"),
("anything new in my documents?", "documents"),
("what should i know about my notes", "notes"),
])
def test_briefing_phrase_does_not_hijack_personal_store(message, family):
"""A broad-briefing phrase must not route the user's own store to the Web."""
capabilities = requested_capabilities(message)
assert family in capabilities
assert "search_browser" not in capabilities
@pytest.mark.parametrize("message", [
"what's new in AI this week",
"give me an update on the war in Ukraine",
"what are the latest events in Kyiv",
"news about the fed rate decision",
"what should i know about rust 2.0",
"catch me up on the latest AI news",
"latest info on the iphone 18",
])
def test_open_web_briefings_keep_their_web_route(message):
"""Open-web subjects that borrow a product noun keep search_browser."""
assert requested_capabilities(message) == frozenset({"search_browser"})
def test_personal_store_families_requires_possessive():
assert personal_store_families("my calendar") == frozenset({"calendar"})
assert personal_store_families("the latest events in Kyiv") == frozenset()
NGINX = {
"title": "NGINX Documentation",
"snippet": "Official configuration docs for NGINX",
"url": "https://nginx.org/en/docs/",
}
IKEA = {
"title": "BILLY Bookcase Assembly Manual",
"snippet": "IKEA BILLY instructions PDF",
"url": "https://ikea.com/manuals/billy",
}
FORD = {
"title": "Ford Ranger Owner Manual",
"snippet": "Operator manual for the Ford Ranger",
"url": "https://ford.com/manual",
}
@pytest.mark.parametrize("query,result", [
("how to configure nginx docs", NGINX),
("where can i find the ikea billy manual", IKEA),
("best guide for sourdough", {"title": "Sourdough Starter Guide",
"snippet": "A complete guide to sourdough",
"url": "https://ex.com/sourdough"}),
("nginx configuration docs", NGINX),
])
def test_document_lookups_accept_relevant_results(query, result):
"""Leading function words and task verbs are not the query entity."""
assert search_core._result_has_query_overlap(query, result) is True
@pytest.mark.parametrize("query", [
"ikea billy manual",
"nginx docs",
"sourdough guide",
])
def test_document_lookups_still_reject_homonyms(query):
"""A result naming none of the entity terms is still not evidence."""
assert search_core._result_has_query_overlap(query, FORD) is False
def test_read_file_block_path_accepts_json_and_bare_text():
assert _read_file_block_path(
json.dumps({"path": "/workspace/out/report.md"})
) == "/workspace/out/report.md"
assert _read_file_block_path(
"/workspace/out/report.md\ntrailing"
) == "/workspace/out/report.md"
assert _read_file_block_path("") == ""
def test_reading_the_input_does_not_verify_the_output():
"""The forced read-back must target the deliverable, not the source."""
target = "/workspace/out/report.md"
assert _read_file_targets_artifact(json.dumps({"path": target}), target)
assert _read_file_targets_artifact(json.dumps({"path": "./report.md"}), target)
assert not _read_file_targets_artifact(
json.dumps({"path": "/workspace/input/data.csv"}), target
)
assert not _read_file_targets_artifact(json.dumps({"path": target}), None)
@@ -752,3 +752,21 @@ def test_service_ddg_html_fallback_sends_safesearch(monkeypatch):
assert seen["params"]["kp"] == "-2"
assert seen["timeout"] <= 5
assert results[0]["url"].startswith("https://notduckduckgo.com/")
def test_relevance_filter_falls_back_when_heuristic_rejects_every_result():
rows = [{
"title": "Bicycle buying guide",
"snippet": "Road bikes and city bikes",
"url": "https://example.test/bicycle-guide",
}]
assert core._filter_low_relevance_results("bicycles", rows) == rows
def test_relevance_filter_does_not_bypass_explicit_site_scope():
rows = [{
"title": "Bicycle buying guide",
"snippet": "Road bikes and city bikes",
"url": "https://example.test/bicycle-guide",
}]
assert core._filter_low_relevance_results("site:official.test bicycles", rows) == []
+5
View File
@@ -338,3 +338,8 @@ async def test_write_file_dispatch_blocks_cron(monkeypatch):
)
assert "outside the allowed roots" in (result.get("error") or "")
assert result.get("exit_code") == 1
@pytest.mark.parametrize("filename", ["auth.json", "app.db", "settings.json"])
def test_application_secrets_are_sensitive_paths(filename):
from src.tool_execution import _is_sensitive_path
assert _is_sensitive_path(f"/tmp/odysseus-data/{filename}")
+40
View File
@@ -21,6 +21,46 @@ def test_disabled_ocr_is_not_reintroduced_by_explicit_request_or_warm_history():
assert 'extract_text' not in offered.executable
def test_web_subject_followup_does_not_offer_shell_or_private_stores():
from src.tool_routing_experiment import MODEL_CHOICE_MODE
from src.turn_contract import requested_capabilities
policy = ToolPolicy()
history = [
{'role': 'user', 'content': 'Is there Rocket League for Switch 2?'},
{'role': 'assistant', 'content': 'It uses the Switch listing.', 'metadata': {
'tool_events': [
{'tool': 'web_search', 'exit_code': 0, 'error': False},
{'tool': 'web_fetch', 'exit_code': 0, 'error': False},
],
}},
]
capabilities = requested_capabilities(
'I searched rocket and cannot find it', history,
)
inventory = resolve_full_inventory_contract(
schemas=FUNCTION_TOOL_SCHEMAS, policy=policy,
)
routed = resolve_turn_contract(
capabilities=capabilities, schemas=FUNCTION_TOOL_SCHEMAS, policy=policy,
)
result = select_experiment_inventory(
inventory, routed, history, MODEL_CHOICE_MODE,
user_text='I searched rocket and cannot find it',
)
assert capabilities == {'search_browser'}
assert result.offered
assert {schema.rsplit('__', 1)[-1] for schema in result.offered} <= {
'web_search', 'web_fetch', 'private_browser', 'youtube_tool',
'search_hf_models', 'pdf_extract',
}
assert not result.offered.intersection({
'bash', 'python', 'read_file', 'manage_notes', 'manage_documents',
'search_emails', 'search_chats',
})
def test_model_choice_rollout_is_account_model_and_header_scoped():
from src.tool_routing_experiment import MODEL_CHOICE_MODE, MODEL_CHOICE_MODEL
assert experiment_mode(None, 'pewds', MODEL_CHOICE_MODEL) == MODEL_CHOICE_MODE
+47 -1
View File
@@ -12,13 +12,59 @@ from src.tool_policy import ToolPolicy, WEB_ACCESS_TOOL_NAMES
from src.tool_schemas import FUNCTION_TOOL_SCHEMAS
from src.turn_contract import (
FAMILY_TOOLS, RequiredReadOperation, active_turn_contract, bind_turn_contract, canonical_tool,
requested_capabilities, required_read_operation_for_request,
immediately_established_family, requested_capabilities, required_read_operation_for_request,
requests_independent_web_source, requests_supporting_web_source, resolve_turn_contract,
preserve_bound_editor_selected_tools, selected_tools_for_request,
targets_bound_editor_request,
)
def _completed_tool_turn(user_text, *tools):
return [
{"role": "user", "content": user_text},
{"role": "assistant", "content": "Done", "metadata": {"tool_events": [
{"tool": tool, "exit_code": 0, "error": False} for tool in tools
]}},
]
@pytest.mark.parametrize(("prior", "followup"), [
("Is there Rocket League for Switch 2?", "I searched rocket and cannot find it"),
("Find the current price of the Framework laptop", "framework is not showing for me"),
("Look up the Kyoto railway museum opening hours", "I cannot find the kyoto museum result"),
("Check whether Aurora 7 is available on PlayStation", "where is aurora 7 listed"),
])
def test_subject_continuity_keeps_public_web_followups_out_of_shell(prior, followup):
history = _completed_tool_turn(prior, "web_search", "web_fetch")
assert immediately_established_family(followup, history) == "search_browser"
assert requested_capabilities(followup, history) == frozenset({"search_browser"})
def test_subject_continuity_uses_typed_private_domain_without_crossing_to_shell():
history = _completed_tool_turn("Find my Project Juniper note", "manage_notes")
assert requested_capabilities(
"juniper is not showing in the results", history,
) == frozenset({"notes"})
def test_explicit_domain_switch_overrides_subject_continuity():
history = _completed_tool_turn("Find the current Orion browser release", "web_search")
assert requested_capabilities(
"search my documents for Orion", history,
) == frozenset({"documents"})
def test_mixed_domain_turn_is_not_inherited_as_one_domain():
history = _completed_tool_turn(
"Find sources for Atlas and save them to my notes", "web_search", "manage_notes",
)
assert immediately_established_family("atlas is missing", history) is None
@pytest.mark.parametrize('prompt', [
"What's new in Sweden?",
"What's happening in Sweden?",
+21
View File
@@ -0,0 +1,21 @@
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
def test_yoyo_is_a_builtin_theme_with_its_saved_effect_defaults():
source = (ROOT / "static/js/theme.js").read_text(encoding="utf-8")
assert "yoyo: { bg:'#211f23', fg:'#dfdbd7'" in source
assert "yoyo: 'ascii-fireflies'" in source
assert "yoyo: '#b8e6c1'" in source
assert ".filter(([name]) => !THEMES[name])" in source
def test_agent_theme_inventory_includes_yoyo():
interaction = (ROOT / "src/ai_interaction.py").read_text(encoding="utf-8")
schemas = (ROOT / "src/tool_schemas.py").read_text(encoding="utf-8")
assert '"monolith", "yoyo"' in interaction
assert "blueprint, monolith, yoyo" in schemas