from types import SimpleNamespace
from dataclasses import replace
import json
import jsonschema
import pytest
import re
from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, execution_targets_required_artifact, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, bounded_web_evidence_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, web_fetch_observation_is_boilerplate, broad_current_web_request, record_tool_execution, align_structured_tool_history, provider_request_messages, provider_compatible_tool_choice_request, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, retrieved_source_urls, serialize_required_email_attachment_chain
from src.tool_capabilities import capabilities_for_tool
def test_native_python_write_command_counts_as_successful_write_effect():
assert execution_has_write_effect(
"python",
{"code": "open('/workspace/results/out.txt', 'w').write('ok')"},
capabilities_for_tool("python"),
native_workspace_enabled=True,
)
def test_read_only_code_does_not_count_as_successful_write_effect():
assert not execution_has_write_effect(
"bash",
{"command": "cat /workspace/input.txt"},
capabilities_for_tool("bash"),
native_workspace_enabled=True,
)
def test_code_mutation_is_not_promoted_outside_native_workspace():
assert not execution_has_write_effect(
"bash",
{"command": "printf ok > /workspace/results/out.txt"},
capabilities_for_tool("bash"),
native_workspace_enabled=False,
)
def test_create_draft_alias_resolves_only_when_reviewable_draft_is_offered():
offered = [{'function': {'name': 'mcp__email__draft_email'}}]
assert offered_tool_alias('mcp__email__create_draft', offered) == 'mcp__email__draft_email'
assert offered_tool_alias('mcp__email__create_draft', []) == 'mcp__email__create_draft'
def test_calendar_dependent_draft_requires_successful_calendar_evidence():
class Contract:
required_read_operation = type('Operation', (), {'tool': 'manage_calendar'})()
assert dependent_write_prerequisite_error(Contract(), 'mcp__email__draft_email', set())
assert dependent_write_prerequisite_error(
Contract(), 'mcp__email__draft_email', {'manage_calendar'}
) is None
class RequiredSetContract:
required_read_operation = None
required = {'manage_calendar', 'mcp__email__draft_email'}
assert dependent_write_prerequisite_error(
RequiredSetContract(), 'mcp__email__draft_email', set()
)
def test_required_email_attachment_batch_is_serialized_by_successful_stage():
required = {'search_emails', 'read_email', 'download_attachment', 'draft_email'}
calls = [
{'function': {'name': f'mcp__email__{name}', 'arguments': '{}'}}
for name in ('search_emails', 'read_email', 'download_attachment', 'draft_email')
]
assert serialize_required_email_attachment_chain(calls, required, [])[0]['function']['name'].endswith('search_emails')
done = [{'tool': 'mcp__email__search_emails', 'exit_code': 0, 'error': False}]
assert serialize_required_email_attachment_chain(calls, required, done)[0]['function']['name'].endswith('read_email')
def test_provider_request_messages_strips_internal_metadata_without_mutating_history():
history = [{
'role': 'user',
'content': [{'type': 'text', 'text': 'evidence'}],
'metadata': {'trusted': False, 'source': 'tool visual evidence'},
}]
assert provider_request_messages(history) == [{
'role': 'user',
'content': [{'type': 'text', 'text': 'evidence'}],
}]
assert history[0]['metadata']['trusted'] is False
def test_provider_wire_messages_drops_empty_assistant_placeholder():
from src.clean_agent_preview import provider_wire_messages
history = [
{'role': 'assistant', 'content': None},
{'role': 'user', 'content': 'Completion check: create the artifact.'},
]
assert provider_request_messages(history) == history
assert provider_wire_messages(history) == [history[1]]
def test_deepseek_flash_keeps_tools_but_drops_unsupported_forced_choice():
request = {
'model': 'deepseek-flash',
'tools': [{'type': 'function', 'function': {'name': 'inspect_media'}}],
'tool_choice': {'type': 'function', 'function': {'name': 'inspect_media'}},
}
compatible = provider_compatible_tool_choice_request(request, 'deepseek-flash')
assert compatible['tools'] == request['tools']
assert 'tool_choice' not in compatible
assert request['tool_choice']['function']['name'] == 'inspect_media'
assert provider_compatible_tool_choice_request(request, 'qwen3.5-9b') is request
def test_private_browser_observations_do_not_advance_page_revision():
assert private_browser_state_transition({'action': 'snapshot'}, 'https://example.org') == (
False, 'https://example.org')
assert private_browser_state_transition({'action': 'find', 'text': 'heading'}, None) == (
False, None)
assert private_browser_state_transition(
{'action': 'batch', 'commands': [['snapshot'], ['snapshot']]},
'https://example.org',
) == (False, 'https://example.org')
def test_private_browser_interactions_and_new_navigation_advance_page_revision():
assert private_browser_state_transition(
{'action': 'open', 'url': 'https://example.org'}, None,
) == (True, 'https://example.org')
assert private_browser_state_transition(
{'action': 'open', 'url': 'https://example.org'}, 'https://example.org',
) == (False, 'https://example.org')
assert private_browser_state_transition(
{'action': 'click', 'target': '@e2'}, 'https://example.org',
) == (True, 'https://example.org')
assert private_browser_state_transition(
{'action': 'click', 'target': '@e2'}, 'https://example.org',
{'exit_code': 1, 'error': None, 'output': 'Unknown ref: e2'},
) == (False, 'https://example.org')
def test_private_browser_snapshot_repeat_limit_is_bounded():
assert private_browser_success_repeat_limit({'action': 'snapshot'}) == 3
assert private_browser_success_repeat_limit(
{'action': 'batch', 'commands': [['snapshot'], ['snapshot']]},
) == 3
assert private_browser_success_repeat_limit({'action': 'open', 'url': 'https://example.org'}) == 1
def test_private_browser_covered_click_does_not_advance_dom_revision():
changed, current_url = private_browser_state_transition(
{'action': 'click', 'target': '@e100'},
'https://www.ikea.com/',
{
'exit_code': 1,
'output': (
"Element '@e100' is covered by at its click point, "
'so the input would land on that element instead.'
),
},
)
assert changed is False
assert current_url == 'https://www.ikea.com/'
def test_compact_notes_preserves_create_vs_edit_and_replacement_semantics():
schema = next(s for s in FUNCTION_TOOL_SCHEMAS if s['function']['name'] == 'manage_notes')
compact = compact_schemas([schema])[0]['function']
assert 'add creates a new note' in compact['description']
assert 'update with id' in compact['description']
assert 'replaces the whole checklist' in compact['parameters']['properties']['checklist_items']['description']
def test_compact_notes_exposes_optional_explicit_checked_state():
schema = next(s for s in FUNCTION_TOOL_SCHEMAS if s['function']['name'] == 'manage_notes')
for candidate in (schema, compact_schemas([schema])[0]):
parameters = candidate['function']['parameters']
assert parameters['properties']['done']['type'] == 'boolean'
assert 'omit to toggle' in parameters['properties']['done']['description']
assert 'done' not in parameters['required']
def test_compact_skills_distinguishes_field_edits_from_raw_text_patches():
schema = next(s for s in FUNCTION_TOOL_SCHEMAS if s['function']['name'] == 'manage_skills')
compact = compact_schemas([schema])[0]['function']
props = compact['parameters']['properties']
assert props['procedure']['type'] == 'array'
assert 'add/edit' in props['procedure']['description']
assert 'not a flag' in props['procedure']['description']
assert 'exactly once' in props['old_string']['description']
assert 'full SKILL.md' in props['old_string']['description']
def test_compact_skills_preserves_reference_read_contract():
schema = next(s for s in FUNCTION_TOOL_SCHEMAS if s['function']['name'] == 'manage_skills')
params = compact_schemas([schema])[0]['function']['parameters']
props = params['properties']
assert 'view = SKILL.md' in props['action']['description']
assert 'view_ref = supporting file' in props['action']['description']
assert 'not a file path' in props['name']['description']
assert 'view_ref only' in props['path']['description']
assert params['required'] == ['action'] # listing still needs no name/path
@pytest.mark.parametrize('name,key', [
('edit_document', 'edits'),
('suggest_document', 'suggestions'),
])
def test_preview_normalizes_json_encoded_document_arrays_before_schema_validation(name, key):
value = [{'find': 'old', 'replace': 'new'}]
if name == 'suggest_document':
value[0]['reason'] = 'clearer'
tool_type, normalized = normalize_preview_function_args(
name,
{key: json.dumps(value)},
user_text='Improve the active draft.',
)
assert tool_type == name
assert normalized[key] == value
schema = next(
item for item in compact_schemas(FUNCTION_TOOL_SCHEMAS)
if item['function']['name'] == name
)
jsonschema.validate(normalized, schema['function']['parameters'])
@pytest.mark.parametrize('message', [
'run that list again, three only, read-only',
'do that again pls, just three, not touching anything',
])
def test_explicit_operation_repeat_is_not_replaced_with_stale_collection(message):
history = [{
'role': 'assistant',
'tool_calls': [{
'id': 'call-1', 'type': 'function',
'function': {'name': 'manage_documents', 'arguments': '{"action":"list"}'},
}],
}, {
'role': 'tool', 'tool_call_id': 'call-1',
'content': '{"response":"Old rows","exit_code":0}',
}]
assert prior_collection_repeat_answer(message, history) == ''
@pytest.mark.parametrize(('name', 'field'), [
('manage_documents', 'limit'),
])
def test_model_choice_runtime_still_normalizes_transport_integer_strings(name, field):
tool_type, normalized = normalize_preview_call_args(
name, {'action': 'list', field: '3'},
user_text='list three', model_choice_experiment=True,
)
assert tool_type == name
assert normalized[field] == 3
def test_model_choice_drops_unsupported_task_list_limit_alias():
tool_type, normalized = normalize_preview_call_args(
'manage_tasks', {'action': 'list', 'max_results': '3'},
user_text='list three', model_choice_experiment=True,
)
assert tool_type == 'manage_tasks'
assert normalized == {'action': 'list'}
def test_contract_sealed_hwfit_get_is_allowed_but_generic_app_api_is_not():
args = {
'action': 'call',
'method': 'GET',
'path': '/api/hwfit/models?fit_only=true&limit=10&sort=fit',
}
allowed = evaluate_preview_call(
'app_api', args, 'find the best model to run on my hardware',
contract_required_tools={'app_api'},
turn_authorized_families={'cookbook_admin'},
)
assert allowed.allowed
denied = evaluate_preview_call(
'app_api', {'action': 'call', 'method': 'GET', 'path': '/api/cookbook/state'},
'show cookbook state', contract_required_tools={'app_api'},
turn_authorized_families={'cookbook_admin'},
)
assert not denied.allowed
def test_contract_sealed_gallery_read_and_explicit_image_edit_are_allowed():
gallery = evaluate_preview_call(
'app_api', {
'action': 'call', 'method': 'GET', 'path': '/api/gallery/library',
},
'list my gallery images through the internal app api',
contract_required_tools={'app_api'},
turn_authorized_families={'cookbook_admin'},
)
assert gallery.allowed
upscale = evaluate_preview_call(
'edit_image', {'image_id': 'owned-image', 'action': 'upscale', 'scale': 2},
'upscale that image 2x',
contract_required_tools={'edit_image'},
turn_authorized_families={'image_editing'},
model_choice_private_tools={'edit_image'},
)
assert upscale.allowed
def test_compact_suggestion_contract_forbids_noop_replacements():
schema = next(
item for item in compact_schemas(FUNCTION_TOOL_SCHEMAS)
if item['function']['name'] == 'suggest_document'
)['function']
assert 'never emit a no-op suggestion' in schema['description']
replace = schema['parameters']['properties']['suggestions']['items']['properties']['replace']
assert 'MUST be materially different from find' in replace['description']
def test_interactive_ocr_uses_upload_references_not_arbitrary_workspace_reads():
owned_ref = evaluate_preview_call('extract_text', {'path': 'odysseus://attachment/fixture-upload'}, 'OCR this image')
assert owned_ref.allowed
assert owned_ref.effects == ('read_private',)
assert not evaluate_preview_call('extract_text', {'path': '/etc/passwd'}, 'OCR this image').allowed
assert not evaluate_preview_call('extract_text', {'path': '/workspace/image.png'}, 'OCR this image').allowed
from src.tool_schemas import FUNCTION_TOOL_SCHEMAS, function_call_to_tool_block
from src.tool_policy import ToolPolicy
from src.turn_contract import resolve_full_inventory_contract
def test_native_execution_limits_allow_multi_artifact_work_without_unbounded_rounds():
assert native_execution_limits(64) == (64, 32)
assert native_execution_limits(1000) == (64, 32)
assert native_execution_limits("invalid") == (8, 32)
def test_compact_preview_honors_configured_interactive_round_limit():
assert interactive_execution_limit(100) == 8
assert interactive_execution_limit(1000) == 8
assert interactive_execution_limit(0) == 1
assert interactive_execution_limit(None) == 8
assert interactive_execution_limit("invalid") == 8
@pytest.mark.parametrize('text,expected', [
('helo', True), ('Hello!', True), ('thanks', True),
('hello, find the latest news', False), ('thanks, now open the source', False),
('search for the song Hello', False),
])
def test_social_turn_requires_the_entire_request(text, expected):
from src.clean_agent_preview import standalone_social_turn
assert standalone_social_turn(text) is expected
@pytest.mark.parametrize('prompt,expected', [
('Read /workspace/fixtures/paper.pdf and create /workspace/results.csv and /workspace/chart.png',
('/workspace/results.csv', '/workspace/chart.png')),
('Create /workspace/results.csv using /workspace/source.csv', ('/workspace/results.csv',)),
('Create /workspace/results.csv. Read /workspace/source.csv and /workspace/other.csv',
('/workspace/results.csv',)),
('Read /workspace/source.csv and /workspace/other.csv', ()),
('Create /workspace/results.v2.csv and /workspace/chart.v2.png',
('/workspace/results.v2.csv', '/workspace/chart.v2.png')),
])
def test_all_requested_outputs_survive_filename_periods(prompt, expected):
from src.clean_agent_preview import declared_workspace_artifacts
assert declared_workspace_artifacts(prompt) == expected
def test_notes_terminal_response_preserves_links_from_wrapped_executor_output():
from src.clean_agent_preview import notes_terminal_response
raw = json.dumps({
'results': '- [note-123] **Fixture note**',
'exit_code': 0,
})
assert notes_terminal_response(raw) == (
'Here are your notes (1):\nš [Fixture note](#note-note-123)'
)
@pytest.mark.parametrize('tool,field,anchor,expected', [
('manage_notes', 'id', '#note-note-123', 'note-123'),
('manage_calendar', 'uid', '#event-event-123', 'event-123'),
('manage_tasks', 'task_id', '#task-task-123', 'task-123'),
('manage_memory', 'memory_id', '#memory-memory-123', 'memory-123'),
('manage_documents', 'document_id', '#document-document-123', 'document-123'),
('read_email', 'uid', '#email-104', '104'),
])
def test_clickable_entity_anchor_is_normalized_back_to_its_server_id(
tool, field, anchor, expected,
):
_, args = normalize_preview_function_args(tool, {field: anchor})
assert args[field] == expected
def test_successful_document_suggestion_has_one_browser_owned_event():
suggestions = [{'find': 'wordy', 'replace': 'concise', 'reason': 'clarity'}]
assert document_suggestions_event({'doc_id': 'doc-1', 'suggestions': suggestions}) == {
'type': 'doc_suggestions', 'doc_id': 'doc-1', 'suggestions': suggestions,
}
assert document_suggestions_event(
{'doc_id': 'doc-1', 'suggestions': suggestions}, failed=True,
) is None
assert document_suggestions_event({'error': 'no match'}, failed=True) is None
def test_preserve_meaning_rejects_repeated_destructive_suggestion_replacement():
args = {'suggestions': [
{'find': 'First passage ' * 12, 'replace': 'This book is fiction.', 'reason': 'Concise.'},
{'find': 'Different second passage ' * 8, 'replace': 'This book is fiction.', 'reason': 'Concise.'},
]}
assert document_suggestion_quality_error(
'suggest_document', args, user_text='Rewrite this but preserve the meaning.'
)
assert document_suggestion_quality_error(
'suggest_document', args, user_text='Make suggestions.'
) is None
def test_runtime_required_artifacts_includes_runner_declared_directory():
assert runtime_required_artifacts(
'Create the requested output.',
{'completion_requirements': {'required_artifacts': ['/tmp_workspace/results/']}},
) == ('/tmp_workspace/results',)
def test_runtime_required_artifacts_does_not_promote_inputs_to_outputs():
assert runtime_required_artifacts(
'Read /workspace/input/data.json and write /workspace/results/report.json.',
{'completion_requirements': {'required_artifacts': ['/workspace/results/report.json']}},
) == ('/workspace/results/report.json',)
def test_prompt_input_paths_are_not_misclassified_as_required_artifacts():
assert runtime_required_artifacts(
'Use read_file to read /workspace/sample.txt. Read only.', {}
) == ()
assert runtime_required_artifacts(
'Use OCR to extract text from /workspace/receipt.png. Read only.', {}
) == ()
def test_prompt_output_path_is_tracked_without_its_source_path():
assert runtime_required_artifacts(
'Create an HTML report at /workspace/output.html from /workspace/input.csv.',
{},
) == ('/workspace/output.html',)
def test_preferences_are_inputs_not_required_outputs_for_a_saved_digest():
from src.clean_agent_preview import declared_workspace_artifacts
assert declared_workspace_artifacts(
'Local preferences live at /workspace/config/interests.json and '
'/workspace/config/categories.json; they contain input data. '
'Save everything to /workspace/results/paper_digest.md.'
) == ('/workspace/results/paper_digest.md',)
def test_compact_writer_advertises_parallel_independent_file_calls():
writer = next(
schema for schema in compact_schemas(FUNCTION_TOOL_SCHEMAS)
if schema['function']['name'] == 'write_file'
)
assert 'multiple write_file calls in the same response' in writer['function']['description']
def test_read_only_python_does_not_satisfy_required_artifact():
required = ('/workspace/output.html',)
assert not execution_targets_required_artifact(
'python', {'code': "Image.open('/workspace/input/reference.png')"}, required,
)
assert execution_targets_required_artifact(
'python', {'code': "open('/workspace/output.html', 'w').write(body)"}, required,
)
assert execution_targets_required_artifact(
'write_file', {'path': '/workspace/output.html', 'content': ''}, required,
)
assert not execution_targets_required_artifact(
'write_file', {'path': '/workspace/analyze.py', 'content': 'print(1)'}, required,
)
def test_read_only_gate_blocks_mutation_and_network_shell():
assert readonly_call('manage_notes', {'action': 'list'})
assert readonly_call('manage_notes', {'action': 'view', 'id': 'abc'})
assert not readonly_call('manage_notes', {'action': 'delete', 'id': 'abc'})
assert not readonly_call('bash', {'command': 'curl https://example.com'})
assert not readonly_call('mcp__email__send_email', {})
def test_preview_allows_safe_personal_writes_only():
assert preview_call_allowed('manage_notes', {'action': 'add', 'title': 'x'}, 'add a note')
assert preview_call_allowed('manage_tasks', {'action': 'create', 'task': 'x'}, 'add a task')
assert preview_call_allowed('manage_calendar', {'action': 'create_event', 'summary': 'x'}, 'add a calendar event')
assert preview_call_allowed('manage_memory', {'action': 'edit', 'id': 'x'}, 'edit my memory')
assert preview_call_allowed('create_document', {'title': 'x'}, 'make a document')
assert preview_call_allowed('manage_notes', {'action': 'delete', 'id': 'x'}, 'delete a note')
assert not preview_call_allowed('manage_tasks', {'action': 'run', 'id': 'x'}, 'run task')
assert not preview_call_allowed('send_email', {'to': 'x@example.com'}, 'send email')
assert not preview_call_allowed('bash', {'command': 'true'}, 'run shell')
def test_preview_allows_read_only_email_screening_tools():
assert preview_call_allowed(
'scan_spam', {'account': 'Primary Inbox', 'folder': 'INBOX'},
'anything junky in the first inbox?',
)
assert preview_call_allowed(
'scan_email_unsubscribes', {'account': 'Primary Inbox'},
'scan newsletters for unsubscribe links',
)
def test_explicit_primary_inbox_is_preserved_when_email_search_omits_account():
from src.clean_agent_preview import preserve_requested_email_account
assert preserve_requested_email_account(
'mcp__email__search_emails', {'query': 'vendor constraints'},
user_text='Search my primary inbox for the latest constraints',
) == {'query': 'vendor constraints', 'account': 'Primary Inbox'}
assert preserve_requested_email_account(
'mcp__email__search_emails', {'query': 'vendor constraints'},
user_text='Search all my mailboxes',
) == {'query': 'vendor constraints'}
@pytest.mark.parametrize("tool,args,prompt", [
('send_email', {'to': 'x@example.com', 'subject': 'Hi', 'body': 'Ready'}, 'send email to x@example.com'),
('reply_to_email', {'uid': '1001', 'body': 'Ready'}, 'reply now to UID 1001'),
('send_to_session', {'session_id': 'chat-1', 'message': 'Ready'}, 'send the chat this message'),
('chat_with_model', {'model': 'provider/model', 'message': 'Question'}, 'ask provider/model this question'),
('pipeline', {'steps': [{'model': 'provider/model', 'prompt': 'Question'}]}, 'run a model pipeline'),
])
def test_contract_required_external_operations_are_preview_safe(tool, args, prompt):
decision = evaluate_preview_call(
tool, args, prompt,
turn_authorized_families={'email', 'sessions'},
contract_required_tools={tool},
)
assert decision.allowed, decision.audit()
def test_external_operations_remain_blocked_without_exact_contract_requirement():
decision = evaluate_preview_call(
'send_email',
{'to': 'x@example.com', 'subject': 'Hi', 'body': 'Ready'},
'send email to x@example.com',
turn_authorized_families={'email'},
)
assert not decision.allowed
@pytest.mark.parametrize("tool,args", [
('manage_research', {'action': 'list'}),
('manage_research', {'action': 'read', 'id': 'report-1'}),
('list_sessions', {}),
('manage_contact', {'action': 'list'}),
('manage_contact', {'action': 'search', 'query': 'Alex'}),
])
def test_supplemental_private_inventory_reads_are_preview_safe(tool, args):
decision = evaluate_preview_call(tool, args, 'list my saved data')
assert decision.allowed and decision.reason == 'allowed'
@pytest.mark.parametrize("tool,args", [
('manage_research', {'action': 'delete', 'id': 'report-1'}),
('manage_contact', {'action': 'add', 'name': 'Alex', 'email': 'a@example.com'}),
])
def test_supplemental_private_inventory_mutations_remain_blocked(tool, args):
decision = evaluate_preview_call(tool, args, 'list my saved data')
assert not decision.allowed
@pytest.mark.parametrize('sentinel', ['', 'all', 'all sessions', 'all_sessions', 'no_filter', '*'])
def test_preview_unfiltered_session_sentinels_do_not_become_literal_title_filters(sentinel):
tool, args = normalize_preview_function_args('list_sessions', {'filter': sentinel})
assert tool == 'list_sessions'
assert args == {}
def test_preview_preserves_real_session_title_filter():
tool, args = normalize_preview_function_args('list_sessions', {'filter': 'audit'})
assert tool == 'list_sessions'
assert args == {'filter': 'audit'}
def test_plain_note_list_drops_model_invented_search_and_type_filters():
tool, args = normalize_preview_function_args(
'manage_notes',
{
'action': 'list', 'title': 'Top Three Notes', 'note_type': 'note',
'pinned': True,
},
user_text='list those again, top three only',
)
assert tool == 'manage_notes'
assert args == {'action': 'list'}
def test_grounded_note_list_filters_are_preserved():
tool, args = normalize_preview_function_args(
'manage_notes',
{'action': 'list', 'query': 'packing', 'pinned': True, 'archived': True},
user_text='list archived pinned notes matching packing',
)
assert tool == 'manage_notes'
assert args == {
'action': 'list', 'query': 'packing', 'pinned': True, 'archived': True,
}
def test_skill_walkthrough_normalizes_invented_reference_to_skill_body_view():
tool, args = normalize_preview_function_args(
'manage_skills',
{
'action': 'view_ref', 'name': 'action-evidence-synthesis',
'path': 'references/details.md',
},
user_text='what does the first one actually do? walk me thru it',
)
assert tool == 'manage_skills'
assert args == {'action': 'view', 'name': 'action-evidence-synthesis'}
@pytest.mark.parametrize('message', [
'Which one runs most often?',
'do those show next run time too or only status?',
'how often does that task execute?',
])
def test_task_questions_over_list_evidence_require_synthesis(message):
assert task_list_requires_synthesis(message)
def test_plain_task_inventory_keeps_canonical_renderer():
assert not task_list_requires_synthesis('list my first three tasks and statuses')
def test_empty_note_search_blocks_referential_view_of_older_list_item():
history = [
{'role': 'assistant', 'tool_calls': [{
'id': 'search-1', 'function': {
'name': 'manage_notes',
'arguments': json.dumps({'action': 'search', 'query': 'apartment lease'}),
},
}]},
{'role': 'tool', 'tool_call_id': 'search-1', 'content': json.dumps({
'response': 'No notes found.', 'exit_code': 0,
})},
]
assert note_search_result_empty(history[-1]['content'])
assert note_referent_error(
'manage_notes', {'action': 'view', 'id': 'older-note'},
user_text='open that one', history=history,
)
assert note_referent_error(
'manage_notes', {'action': 'view', 'id': 'explicit-note'},
user_text='open note explicit-note', history=history,
) is None
def test_server_sealed_read_forces_exact_first_tool_then_releases_choice():
from src.turn_contract import RequiredReadOperation, resolve_turn_contract
contract = resolve_turn_contract(
capabilities={'contacts'}, schemas=FUNCTION_TOOL_SCHEMAS,
policy=ToolPolicy(),
required_read_operation=RequiredReadOperation(
'manage_contact', {'action': 'list'}, max_items=3,
),
)
offered = compact_schemas(contract.schemas())
assert required_read_tool_choice(contract, offered) == {
'type': 'function', 'function': {'name': 'manage_contact'},
}
assert required_read_tool_choice(contract, offered, calls=1) is None
assert sealed_read_arguments(
contract, 'manage_contact', {'action': 'delete', 'id': 'wrong'}
) == {'action': 'list'}
assert sealed_read_arguments(
contract, 'manage_contact', {'action': 'delete'}, calls=1
) == {'action': 'delete'}
def test_server_sealed_document_list_enforces_contract_limit():
from src.turn_contract import RequiredReadOperation, resolve_turn_contract
contract = resolve_turn_contract(
capabilities={'documents'}, schemas=FUNCTION_TOOL_SCHEMAS,
policy=ToolPolicy(),
required_read_operation=RequiredReadOperation(
'manage_documents', {'action': 'list'}, max_items=3,
),
)
assert sealed_read_arguments(
contract, 'manage_documents', {'action': 'list'}
) == {'action': 'list', 'limit': 3}
def test_server_sealed_calendar_read_preserves_model_resolved_range():
from src.turn_contract import RequiredReadOperation, resolve_turn_contract
contract = resolve_turn_contract(
capabilities={'calendar'}, schemas=FUNCTION_TOOL_SCHEMAS,
policy=ToolPolicy(),
required_read_operation=RequiredReadOperation(
'manage_calendar', {'action': 'list_events'}, max_items=3,
),
)
assert sealed_read_arguments(
contract,
'manage_calendar',
{
'action': 'list_events',
'start': '2026-09-12T00:00:00',
'end': '2026-09-13T00:00:00',
'summary': 'unsafe mutation field',
},
) == {
'action': 'list_events',
'start': '2026-09-12T00:00:00',
'end': '2026-09-13T00:00:00',
}
def test_email_identifier_guard_rejects_placeholder_after_failed_listing():
history = [
{'role': 'assistant', 'tool_calls': [{
'id': 'call-list', 'function': {'name': 'list_emails', 'arguments': '{}'},
}]},
{'role': 'tool', 'tool_call_id': 'call-list', 'content': json.dumps({
'exit_code': 1, 'error': 'email backend unavailable',
})},
]
error = email_identifier_error(
'mcp__email__read_email', {'message_id': ''}, history=history,
)
assert error and 'placeholders are not executable' in error
def test_email_identifier_guard_accepts_successful_result_user_id_and_active_email():
history = [
{'role': 'assistant', 'tool_calls': [{
'id': 'call-list', 'function': {'name': 'list_emails', 'arguments': '{}'},
}]},
{'role': 'tool', 'tool_call_id': 'call-list', 'content': 'Subject: Hello\nUID: 8492'},
{'role': 'system', 'content': 'Active email context\nMessage UID: open-44'},
]
assert email_identifier_error('read_email', {'uid': '8492'}, history=history) is None
assert email_identifier_error('draft_email_reply', {'uid': 'open-44'}, history=history) is None
assert email_identifier_error(
'read_email', {'uid': 'user-77'}, user_text='read email UID user-77', history=history,
) is None
assert email_identifier_error('read_email', {'uid': 'invented'}, history=history)
def test_email_identifier_guard_decodes_mcp_stdout_before_extracting_uid():
history = [
{'role': 'assistant', 'tool_calls': [{
'id': 'call-search', 'function': {
'name': 'mcp__email__search_emails', 'arguments': '{}',
},
}]},
{'role': 'tool', 'tool_call_id': 'call-search', 'content': json.dumps({
'stdout': (
'Found 1 email(s):\nUID: 104\n'
'Account: Research Mail '
),
'stderr': '', 'exit_code': 0,
})},
]
assert email_identifier_error(
'mcp__email__draft_email_reply', {'uid': '104'}, history=history,
) is None
def test_email_identifier_guard_accepts_uid_nested_in_successful_stdout_result():
history = [
{'role': 'assistant', 'tool_calls': [{
'id': 'call-search',
'function': {'name': 'mcp__email__search_emails', 'arguments': '{}'},
}]},
{'role': 'tool', 'tool_call_id': 'call-search', 'content': json.dumps({
'stdout': 'Found 1 email\nUID: 104\nAccount: Research Mail',
'stderr': '',
'exit_code': 0,
})},
]
assert email_identifier_error(
'mcp__email__draft_email_reply', {'uid': '104'}, history=history,
) is None
def test_canonical_result_renderers_honor_explicit_user_count_limits():
assert requested_item_limit('return at most three titles', default=20) == 3
assert requested_item_limit('whats on my notes list? three at most, dont touch anything', default=20) == 3
assert requested_item_limit('show only 2', default=20) == 2
assert requested_item_limit('pull them up, just 3 short ones', default=20) == 3
assert requested_item_limit('list my notes please, 3 titles max', default=20) == 3
assert requested_item_limit('list those again, three only', default=20) == 3
assert requested_item_limit('maybe first three titles', default=20) == 3
assert requested_item_limit('keep it to three titles', default=20) == 3
assert requested_item_limit('top 3 titles', default=20) == 3
assert requested_item_limit('list those again, 3', default=20) == 3
assert requested_item_limit('three names and their statuses max', default=20) == 3
assert requested_item_limit('just titles, 3 max, read only pls', default=20) == 3
assert requested_item_limit(
'what tasks do i have scheduled? 3 is fine, need name + status', default=20,
) == 3
assert requested_item_limit(
'can you check my calendar and give me the next 3 events? titles and times only',
default=20,
) == 3
assert requested_item_limit('show me my saved memories, only a few', default=20) == 3
assert requested_item_limit("what's in my memory? just a few", default=20) == 3
assert requested_item_limit(
'Can you show my notes? I only need three titles. Read-only, keep it short.',
default=20,
) == 3
assert requested_item_limit('what notes have i got? 3 titles tops', default=20) == 3
assert requested_item_limit(
'list my memorries, three short ones, read-only please', default=20,
) == 3
assert requested_item_limit('list those again, three tops', default=20) == 3
assert requested_item_limit('same again but cap it at three, read only', default=20) == 3
assert requested_item_limit('what tasks are set up? no more than 3 names', default=20) == 3
assert requested_item_limit('i only want 3 names and statuses', default=20) == 3
assert requested_item_limit(
"only need the first three names + whether theyre running or paused", default=20,
) == 3
assert requested_item_limit('three short bits, dont change anything', default=20) == 3
assert requested_item_limit(
'what do you remember about me? show me like three things max', default=20,
) == 3
assert requested_item_limit(
'what scheduled tasks do i have? three names + status max', default=20,
) == 3
assert requested_item_limit(
'can u list my calendar? three titles and times max, no edits', default=20,
) == 3
notes = '- [a] **One**\n- [b] **Two**\n- [c] **Three**\n- [d] **Four**'
rendered_notes = notes_terminal_response(notes, user_text='List at most three titles')
assert 'Four' not in rendered_notes
assert '