from types import SimpleNamespace from dataclasses import replace import json import jsonschema import pytest import re from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, execution_targets_required_artifact, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, direct_shell_output_request, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, bounded_web_evidence_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, web_fetch_observation_is_boilerplate, broad_current_web_request, record_tool_execution, align_structured_tool_history, provider_request_messages, provider_compatible_tool_choice_request, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, retrieved_source_urls, serialize_required_email_attachment_chain from src.tool_capabilities import capabilities_for_tool def test_native_python_write_command_counts_as_successful_write_effect(): assert execution_has_write_effect( "python", {"code": "open('/workspace/results/out.txt', 'w').write('ok')"}, capabilities_for_tool("python"), native_workspace_enabled=True, ) def test_read_only_code_does_not_count_as_successful_write_effect(): assert not execution_has_write_effect( "bash", {"command": "cat /workspace/input.txt"}, capabilities_for_tool("bash"), native_workspace_enabled=True, ) def test_code_mutation_is_not_promoted_outside_native_workspace(): assert not execution_has_write_effect( "bash", {"command": "printf ok > /workspace/results/out.txt"}, capabilities_for_tool("bash"), native_workspace_enabled=False, ) def test_create_draft_alias_resolves_only_when_reviewable_draft_is_offered(): offered = [{'function': {'name': 'mcp__email__draft_email'}}] assert offered_tool_alias('mcp__email__create_draft', offered) == 'mcp__email__draft_email' assert offered_tool_alias('mcp__email__create_draft', []) == 'mcp__email__create_draft' def test_calendar_dependent_draft_requires_successful_calendar_evidence(): class Contract: required_read_operation = type('Operation', (), {'tool': 'manage_calendar'})() assert dependent_write_prerequisite_error(Contract(), 'mcp__email__draft_email', set()) assert dependent_write_prerequisite_error( Contract(), 'mcp__email__draft_email', {'manage_calendar'} ) is None class RequiredSetContract: required_read_operation = None required = {'manage_calendar', 'mcp__email__draft_email'} assert dependent_write_prerequisite_error( RequiredSetContract(), 'mcp__email__draft_email', set() ) def test_required_email_attachment_batch_is_serialized_by_successful_stage(): required = {'search_emails', 'read_email', 'download_attachment', 'draft_email'} calls = [ {'function': {'name': f'mcp__email__{name}', 'arguments': '{}'}} for name in ('search_emails', 'read_email', 'download_attachment', 'draft_email') ] assert serialize_required_email_attachment_chain(calls, required, [])[0]['function']['name'].endswith('search_emails') done = [{'tool': 'mcp__email__search_emails', 'exit_code': 0, 'error': False}] assert serialize_required_email_attachment_chain(calls, required, done)[0]['function']['name'].endswith('read_email') def test_provider_request_messages_strips_internal_metadata_without_mutating_history(): history = [{ 'role': 'user', 'content': [{'type': 'text', 'text': 'evidence'}], 'metadata': {'trusted': False, 'source': 'tool visual evidence'}, }] assert provider_request_messages(history) == [{ 'role': 'user', 'content': [{'type': 'text', 'text': 'evidence'}], }] assert history[0]['metadata']['trusted'] is False def test_provider_wire_messages_drops_empty_assistant_placeholder(): from src.clean_agent_preview import provider_wire_messages history = [ {'role': 'assistant', 'content': None}, {'role': 'user', 'content': 'Completion check: create the artifact.'}, ] assert provider_request_messages(history) == history assert provider_wire_messages(history) == [history[1]] def test_deepseek_flash_keeps_tools_but_drops_unsupported_forced_choice(): request = { 'model': 'deepseek-flash', 'tools': [ {'type': 'function', 'function': {'name': 'inspect_media'}}, {'type': 'function', 'function': {'name': 'python'}}, ], 'tool_choice': {'type': 'function', 'function': {'name': 'inspect_media'}}, } compatible = provider_compatible_tool_choice_request(request, 'deepseek-flash') assert [tool['function']['name'] for tool in compatible['tools']] == ['inspect_media'] assert 'tool_choice' not in compatible assert request['tool_choice']['function']['name'] == 'inspect_media' assert provider_compatible_tool_choice_request(request, 'qwen3.5-9b') is request def test_deepseek_v4_flash_drops_named_choice_for_thinking_mode(): request = { 'tools': [{'type': 'function', 'function': {'name': 'write_file'}}], 'tool_choice': {'type': 'function', 'function': {'name': 'write_file'}}, } compatible = provider_compatible_tool_choice_request(request, 'deepseek-v4-flash') assert [tool['function']['name'] for tool in compatible['tools']] == ['write_file'] assert 'tool_choice' not in compatible def test_qwen_capture_converts_single_required_tool_to_named_constraint(): request = { 'tools': [{'type': 'function', 'function': {'name': 'web_search'}}], 'tool_choice': 'required', } compatible = provider_compatible_tool_choice_request( request, 'odysseus-qwen3.5-tools-pre-heretic' ) assert compatible['tool_choice'] == { 'type': 'function', 'function': {'name': 'web_search'}, } def test_private_browser_observations_do_not_advance_page_revision(): assert private_browser_state_transition({'action': 'snapshot'}, 'https://example.org') == ( False, 'https://example.org') assert private_browser_state_transition({'action': 'find', 'text': 'heading'}, None) == ( False, None) assert private_browser_state_transition( {'action': 'batch', 'commands': [['snapshot'], ['snapshot']]}, 'https://example.org', ) == (False, 'https://example.org') def test_private_browser_interactions_and_new_navigation_advance_page_revision(): assert private_browser_state_transition( {'action': 'open', 'url': 'https://example.org'}, None, ) == (True, 'https://example.org') assert private_browser_state_transition( {'action': 'open', 'url': 'https://example.org'}, 'https://example.org', ) == (False, 'https://example.org') assert private_browser_state_transition( {'action': 'click', 'target': '@e2'}, 'https://example.org', ) == (True, 'https://example.org') assert private_browser_state_transition( {'action': 'click', 'target': '@e2'}, 'https://example.org', {'exit_code': 1, 'error': None, 'output': 'Unknown ref: e2'}, ) == (False, 'https://example.org') def test_private_browser_snapshot_repeat_limit_is_bounded(): assert private_browser_success_repeat_limit({'action': 'snapshot'}) == 3 assert private_browser_success_repeat_limit( {'action': 'batch', 'commands': [['snapshot'], ['snapshot']]}, ) == 3 assert private_browser_success_repeat_limit({'action': 'open', 'url': 'https://example.org'}) == 1 def test_private_browser_covered_click_does_not_advance_dom_revision(): changed, current_url = private_browser_state_transition( {'action': 'click', 'target': '@e100'}, 'https://www.ikea.com/', { 'exit_code': 1, 'output': ( "Element '@e100' is covered by at its click point, " 'so the input would land on that element instead.' ), }, ) assert changed is False assert current_url == 'https://www.ikea.com/' def test_compact_notes_preserves_create_vs_edit_and_replacement_semantics(): schema = next(s for s in FUNCTION_TOOL_SCHEMAS if s['function']['name'] == 'manage_notes') compact = compact_schemas([schema])[0]['function'] assert 'add creates a new note' in compact['description'] assert 'update with id' in compact['description'] assert 'replaces the whole checklist' in compact['parameters']['properties']['checklist_items']['description'] def test_compact_notes_exposes_optional_explicit_checked_state(): schema = next(s for s in FUNCTION_TOOL_SCHEMAS if s['function']['name'] == 'manage_notes') for candidate in (schema, compact_schemas([schema])[0]): parameters = candidate['function']['parameters'] assert parameters['properties']['done']['type'] == 'boolean' assert 'omit to toggle' in parameters['properties']['done']['description'] assert 'done' not in parameters['required'] def test_compact_skills_distinguishes_field_edits_from_raw_text_patches(): schema = next(s for s in FUNCTION_TOOL_SCHEMAS if s['function']['name'] == 'manage_skills') compact = compact_schemas([schema])[0]['function'] props = compact['parameters']['properties'] assert props['procedure']['type'] == 'array' assert 'add/edit' in props['procedure']['description'] assert 'not a flag' in props['procedure']['description'] assert 'exactly once' in props['old_string']['description'] assert 'full SKILL.md' in props['old_string']['description'] def test_compact_skills_preserves_reference_read_contract(): schema = next(s for s in FUNCTION_TOOL_SCHEMAS if s['function']['name'] == 'manage_skills') params = compact_schemas([schema])[0]['function']['parameters'] props = params['properties'] assert 'view = SKILL.md' in props['action']['description'] assert 'view_ref = supporting file' in props['action']['description'] assert 'not a file path' in props['name']['description'] assert 'view_ref only' in props['path']['description'] assert params['required'] == ['action'] # listing still needs no name/path @pytest.mark.parametrize('name,key', [ ('edit_document', 'edits'), ('suggest_document', 'suggestions'), ]) def test_preview_normalizes_json_encoded_document_arrays_before_schema_validation(name, key): value = [{'find': 'old', 'replace': 'new'}] if name == 'suggest_document': value[0]['reason'] = 'clearer' tool_type, normalized = normalize_preview_function_args( name, {key: json.dumps(value)}, user_text='Improve the active draft.', ) assert tool_type == name assert normalized[key] == value schema = next( item for item in compact_schemas(FUNCTION_TOOL_SCHEMAS) if item['function']['name'] == name ) jsonschema.validate(normalized, schema['function']['parameters']) @pytest.mark.parametrize('message', [ 'run that list again, three only, read-only', 'do that again pls, just three, not touching anything', ]) def test_explicit_operation_repeat_is_not_replaced_with_stale_collection(message): history = [{ 'role': 'assistant', 'tool_calls': [{ 'id': 'call-1', 'type': 'function', 'function': {'name': 'manage_documents', 'arguments': '{"action":"list"}'}, }], }, { 'role': 'tool', 'tool_call_id': 'call-1', 'content': '{"response":"Old rows","exit_code":0}', }] assert prior_collection_repeat_answer(message, history) == '' @pytest.mark.parametrize(('name', 'field'), [ ('manage_documents', 'limit'), ]) def test_model_choice_runtime_still_normalizes_transport_integer_strings(name, field): tool_type, normalized = normalize_preview_call_args( name, {'action': 'list', field: '3'}, user_text='list three', model_choice_experiment=True, ) assert tool_type == name assert normalized[field] == 3 def test_model_choice_drops_unsupported_task_list_limit_alias(): tool_type, normalized = normalize_preview_call_args( 'manage_tasks', {'action': 'list', 'max_results': '3'}, user_text='list three', model_choice_experiment=True, ) assert tool_type == 'manage_tasks' assert normalized == {'action': 'list'} def test_contract_sealed_hwfit_get_is_allowed_but_generic_app_api_is_not(): args = { 'action': 'call', 'method': 'GET', 'path': '/api/hwfit/models?fit_only=true&limit=10&sort=fit', } allowed = evaluate_preview_call( 'app_api', args, 'find the best model to run on my hardware', contract_required_tools={'app_api'}, turn_authorized_families={'cookbook_admin'}, ) assert allowed.allowed denied = evaluate_preview_call( 'app_api', {'action': 'call', 'method': 'GET', 'path': '/api/cookbook/state'}, 'show cookbook state', contract_required_tools={'app_api'}, turn_authorized_families={'cookbook_admin'}, ) assert not denied.allowed def test_contract_sealed_gallery_read_and_explicit_image_edit_are_allowed(): gallery = evaluate_preview_call( 'app_api', { 'action': 'call', 'method': 'GET', 'path': '/api/gallery/library', }, 'list my gallery images through the internal app api', contract_required_tools={'app_api'}, turn_authorized_families={'cookbook_admin'}, ) assert gallery.allowed upscale = evaluate_preview_call( 'edit_image', {'image_id': 'owned-image', 'action': 'upscale', 'scale': 2}, 'upscale that image 2x', contract_required_tools={'edit_image'}, turn_authorized_families={'image_editing'}, model_choice_private_tools={'edit_image'}, ) assert upscale.allowed def test_compact_suggestion_contract_forbids_noop_replacements(): schema = next( item for item in compact_schemas(FUNCTION_TOOL_SCHEMAS) if item['function']['name'] == 'suggest_document' )['function'] assert 'never emit a no-op suggestion' in schema['description'] replace = schema['parameters']['properties']['suggestions']['items']['properties']['replace'] assert 'MUST be materially different from find' in replace['description'] def test_interactive_ocr_uses_upload_references_not_arbitrary_workspace_reads(): owned_ref = evaluate_preview_call('extract_text', {'path': 'odysseus://attachment/fixture-upload'}, 'OCR this image') assert owned_ref.allowed assert owned_ref.effects == ('read_private',) assert not evaluate_preview_call('extract_text', {'path': '/etc/passwd'}, 'OCR this image').allowed assert not evaluate_preview_call('extract_text', {'path': '/workspace/image.png'}, 'OCR this image').allowed from src.tool_schemas import FUNCTION_TOOL_SCHEMAS, function_call_to_tool_block from src.tool_policy import ToolPolicy from src.turn_contract import resolve_full_inventory_contract def test_native_execution_limits_allow_multi_artifact_work_without_unbounded_rounds(): assert native_execution_limits(64) == (64, 32) assert native_execution_limits(1000) == (64, 32) assert native_execution_limits("invalid") == (8, 32) def test_compact_preview_honors_configured_interactive_round_limit(): assert interactive_execution_limit(100) == 8 assert interactive_execution_limit(1000) == 8 assert interactive_execution_limit(0) == 1 assert interactive_execution_limit(None) == 8 assert interactive_execution_limit("invalid") == 8 @pytest.mark.parametrize('text,expected', [ ('helo', True), ('Hello!', True), ('thanks', True), ('hello, find the latest news', False), ('thanks, now open the source', False), ('search for the song Hello', False), ]) def test_social_turn_requires_the_entire_request(text, expected): from src.clean_agent_preview import standalone_social_turn assert standalone_social_turn(text) is expected @pytest.mark.parametrize('prompt,expected', [ ('Read /workspace/fixtures/paper.pdf and create /workspace/results.csv and /workspace/chart.png', ('/workspace/results.csv', '/workspace/chart.png')), ('Create /workspace/results.csv using /workspace/source.csv', ('/workspace/results.csv',)), ('Create /workspace/results.csv. Read /workspace/source.csv and /workspace/other.csv', ('/workspace/results.csv',)), ('Read /workspace/source.csv and /workspace/other.csv', ()), ('Create /workspace/results.v2.csv and /workspace/chart.v2.png', ('/workspace/results.v2.csv', '/workspace/chart.v2.png')), ]) def test_all_requested_outputs_survive_filename_periods(prompt, expected): from src.clean_agent_preview import declared_workspace_artifacts assert declared_workspace_artifacts(prompt) == expected def test_notes_terminal_response_preserves_links_from_wrapped_executor_output(): from src.clean_agent_preview import notes_terminal_response raw = json.dumps({ 'results': '- [note-123] **Fixture note**', 'exit_code': 0, }) assert notes_terminal_response(raw) == ( 'Here are your notes (1):\nšŸ“ [Fixture note](#note-note-123)' ) @pytest.mark.parametrize('tool,field,anchor,expected', [ ('manage_notes', 'id', '#note-note-123', 'note-123'), ('manage_calendar', 'uid', '#event-event-123', 'event-123'), ('manage_tasks', 'task_id', '#task-task-123', 'task-123'), ('manage_memory', 'memory_id', '#memory-memory-123', 'memory-123'), ('manage_documents', 'document_id', '#document-document-123', 'document-123'), ('read_email', 'uid', '#email-104', '104'), ]) def test_clickable_entity_anchor_is_normalized_back_to_its_server_id( tool, field, anchor, expected, ): _, args = normalize_preview_function_args(tool, {field: anchor}) assert args[field] == expected def test_successful_document_suggestion_has_one_browser_owned_event(): suggestions = [{'find': 'wordy', 'replace': 'concise', 'reason': 'clarity'}] assert document_suggestions_event({'doc_id': 'doc-1', 'suggestions': suggestions}) == { 'type': 'doc_suggestions', 'doc_id': 'doc-1', 'suggestions': suggestions, } assert document_suggestions_event( {'doc_id': 'doc-1', 'suggestions': suggestions}, failed=True, ) is None assert document_suggestions_event({'error': 'no match'}, failed=True) is None def test_preserve_meaning_rejects_repeated_destructive_suggestion_replacement(): args = {'suggestions': [ {'find': 'First passage ' * 12, 'replace': 'This book is fiction.', 'reason': 'Concise.'}, {'find': 'Different second passage ' * 8, 'replace': 'This book is fiction.', 'reason': 'Concise.'}, ]} assert document_suggestion_quality_error( 'suggest_document', args, user_text='Rewrite this but preserve the meaning.' ) assert document_suggestion_quality_error( 'suggest_document', args, user_text='Make suggestions.' ) is None def test_runtime_required_artifacts_includes_runner_declared_directory(): assert runtime_required_artifacts( 'Create the requested output.', {'completion_requirements': {'required_artifacts': ['/tmp_workspace/results/']}}, ) == ('/tmp_workspace/results',) def test_runtime_required_artifacts_does_not_promote_inputs_to_outputs(): assert runtime_required_artifacts( 'Read /workspace/input/data.json and write /workspace/results/report.json.', {'completion_requirements': {'required_artifacts': ['/workspace/results/report.json']}}, ) == ('/workspace/results/report.json',) def test_prompt_input_paths_are_not_misclassified_as_required_artifacts(): assert runtime_required_artifacts( 'Use read_file to read /workspace/sample.txt. Read only.', {} ) == () assert runtime_required_artifacts( 'Use OCR to extract text from /workspace/receipt.png. Read only.', {} ) == () def test_prompt_output_path_is_tracked_without_its_source_path(): assert runtime_required_artifacts( 'Create an HTML report at /workspace/output.html from /workspace/input.csv.', {}, ) == ('/workspace/output.html',) def test_empty_runner_artifact_list_falls_back_to_prompt_output_path(): assert runtime_required_artifacts( 'Create an HTML report at /workspace/output.html from /workspace/input.csv.', {'completion_requirements': {'required_artifacts': []}}, ) == ('/workspace/output.html',) def test_preferences_are_inputs_not_required_outputs_for_a_saved_digest(): from src.clean_agent_preview import declared_workspace_artifacts assert declared_workspace_artifacts( 'Local preferences live at /workspace/config/interests.json and ' '/workspace/config/categories.json; they contain input data. ' 'Save everything to /workspace/results/paper_digest.md.' ) == ('/workspace/results/paper_digest.md',) def test_compact_writer_advertises_parallel_independent_file_calls(): writer = next( schema for schema in compact_schemas(FUNCTION_TOOL_SCHEMAS) if schema['function']['name'] == 'write_file' ) assert 'multiple write_file calls in the same response' in writer['function']['description'] def test_read_only_python_does_not_satisfy_required_artifact(): required = ('/workspace/output.html',) assert not execution_targets_required_artifact( 'python', {'code': "Image.open('/workspace/input/reference.png')"}, required, ) assert execution_targets_required_artifact( 'python', {'code': "open('/workspace/output.html', 'w').write(body)"}, required, ) assert execution_targets_required_artifact( 'write_file', {'path': '/workspace/output.html', 'content': ''}, required, ) assert not execution_targets_required_artifact( 'write_file', {'path': '/workspace/analyze.py', 'content': 'print(1)'}, required, ) def test_read_only_gate_blocks_mutation_and_network_shell(): assert readonly_call('manage_notes', {'action': 'list'}) assert readonly_call('manage_notes', {'action': 'view', 'id': 'abc'}) assert not readonly_call('manage_notes', {'action': 'delete', 'id': 'abc'}) assert not readonly_call('bash', {'command': 'curl https://example.com'}) assert not readonly_call('mcp__email__send_email', {}) def test_preview_allows_safe_personal_writes_only(): assert preview_call_allowed('manage_notes', {'action': 'add', 'title': 'x'}, 'add a note') assert preview_call_allowed('manage_tasks', {'action': 'create', 'task': 'x'}, 'add a task') assert preview_call_allowed('manage_calendar', {'action': 'create_event', 'summary': 'x'}, 'add a calendar event') assert preview_call_allowed('manage_memory', {'action': 'edit', 'id': 'x'}, 'edit my memory') assert preview_call_allowed('create_document', {'title': 'x'}, 'make a document') assert preview_call_allowed('manage_notes', {'action': 'delete', 'id': 'x'}, 'delete a note') assert not preview_call_allowed('manage_tasks', {'action': 'run', 'id': 'x'}, 'run task') assert not preview_call_allowed('send_email', {'to': 'x@example.com'}, 'send email') assert not preview_call_allowed('bash', {'command': 'true'}, 'run shell') def test_preview_allows_read_only_email_screening_tools(): assert preview_call_allowed( 'scan_spam', {'account': 'Primary Inbox', 'folder': 'INBOX'}, 'anything junky in the first inbox?', ) assert preview_call_allowed( 'scan_email_unsubscribes', {'account': 'Primary Inbox'}, 'scan newsletters for unsubscribe links', ) def test_explicit_primary_inbox_is_preserved_when_email_search_omits_account(): from src.clean_agent_preview import preserve_requested_email_account assert preserve_requested_email_account( 'mcp__email__search_emails', {'query': 'vendor constraints'}, user_text='Search my primary inbox for the latest constraints', ) == {'query': 'vendor constraints', 'account': 'Primary Inbox'} assert preserve_requested_email_account( 'mcp__email__search_emails', {'query': 'vendor constraints'}, user_text='Search all my mailboxes', ) == {'query': 'vendor constraints'} @pytest.mark.parametrize("tool,args,prompt", [ ('send_email', {'to': 'x@example.com', 'subject': 'Hi', 'body': 'Ready'}, 'send email to x@example.com'), ('reply_to_email', {'uid': '1001', 'body': 'Ready'}, 'reply now to UID 1001'), ('send_to_session', {'session_id': 'chat-1', 'message': 'Ready'}, 'send the chat this message'), ('chat_with_model', {'model': 'provider/model', 'message': 'Question'}, 'ask provider/model this question'), ('pipeline', {'steps': [{'model': 'provider/model', 'prompt': 'Question'}]}, 'run a model pipeline'), ]) def test_contract_required_external_operations_are_preview_safe(tool, args, prompt): decision = evaluate_preview_call( tool, args, prompt, turn_authorized_families={'email', 'sessions'}, contract_required_tools={tool}, ) assert decision.allowed, decision.audit() def test_external_operations_remain_blocked_without_exact_contract_requirement(): decision = evaluate_preview_call( 'send_email', {'to': 'x@example.com', 'subject': 'Hi', 'body': 'Ready'}, 'send email to x@example.com', turn_authorized_families={'email'}, ) assert not decision.allowed @pytest.mark.parametrize("tool,args", [ ('manage_research', {'action': 'list'}), ('manage_research', {'action': 'read', 'id': 'report-1'}), ('list_sessions', {}), ('manage_contact', {'action': 'list'}), ('manage_contact', {'action': 'search', 'query': 'Alex'}), ]) def test_supplemental_private_inventory_reads_are_preview_safe(tool, args): decision = evaluate_preview_call(tool, args, 'list my saved data') assert decision.allowed and decision.reason == 'allowed' @pytest.mark.parametrize("tool,args", [ ('manage_research', {'action': 'delete', 'id': 'report-1'}), ('manage_contact', {'action': 'add', 'name': 'Alex', 'email': 'a@example.com'}), ]) def test_supplemental_private_inventory_mutations_remain_blocked(tool, args): decision = evaluate_preview_call(tool, args, 'list my saved data') assert not decision.allowed @pytest.mark.parametrize('sentinel', ['', 'all', 'all sessions', 'all_sessions', 'no_filter', '*']) def test_preview_unfiltered_session_sentinels_do_not_become_literal_title_filters(sentinel): tool, args = normalize_preview_function_args('list_sessions', {'filter': sentinel}) assert tool == 'list_sessions' assert args == {} def test_preview_preserves_real_session_title_filter(): tool, args = normalize_preview_function_args('list_sessions', {'filter': 'audit'}) assert tool == 'list_sessions' assert args == {'filter': 'audit'} def test_plain_note_list_drops_model_invented_search_and_type_filters(): tool, args = normalize_preview_function_args( 'manage_notes', { 'action': 'list', 'title': 'Top Three Notes', 'note_type': 'note', 'pinned': True, }, user_text='list those again, top three only', ) assert tool == 'manage_notes' assert args == {'action': 'list'} def test_grounded_note_list_filters_are_preserved(): tool, args = normalize_preview_function_args( 'manage_notes', {'action': 'list', 'query': 'packing', 'pinned': True, 'archived': True}, user_text='list archived pinned notes matching packing', ) assert tool == 'manage_notes' assert args == { 'action': 'list', 'query': 'packing', 'pinned': True, 'archived': True, } def test_skill_walkthrough_normalizes_invented_reference_to_skill_body_view(): tool, args = normalize_preview_function_args( 'manage_skills', { 'action': 'view_ref', 'name': 'action-evidence-synthesis', 'path': 'references/details.md', }, user_text='what does the first one actually do? walk me thru it', ) assert tool == 'manage_skills' assert args == {'action': 'view', 'name': 'action-evidence-synthesis'} @pytest.mark.parametrize('message', [ 'Which one runs most often?', 'do those show next run time too or only status?', 'how often does that task execute?', ]) def test_task_questions_over_list_evidence_require_synthesis(message): assert task_list_requires_synthesis(message) def test_plain_task_inventory_keeps_canonical_renderer(): assert not task_list_requires_synthesis('list my first three tasks and statuses') def test_empty_note_search_blocks_referential_view_of_older_list_item(): history = [ {'role': 'assistant', 'tool_calls': [{ 'id': 'search-1', 'function': { 'name': 'manage_notes', 'arguments': json.dumps({'action': 'search', 'query': 'apartment lease'}), }, }]}, {'role': 'tool', 'tool_call_id': 'search-1', 'content': json.dumps({ 'response': 'No notes found.', 'exit_code': 0, })}, ] assert note_search_result_empty(history[-1]['content']) assert note_referent_error( 'manage_notes', {'action': 'view', 'id': 'older-note'}, user_text='open that one', history=history, ) assert note_referent_error( 'manage_notes', {'action': 'view', 'id': 'explicit-note'}, user_text='open note explicit-note', history=history, ) is None def test_server_sealed_read_forces_exact_first_tool_then_releases_choice(): from src.turn_contract import RequiredReadOperation, resolve_turn_contract contract = resolve_turn_contract( capabilities={'contacts'}, schemas=FUNCTION_TOOL_SCHEMAS, policy=ToolPolicy(), required_read_operation=RequiredReadOperation( 'manage_contact', {'action': 'list'}, max_items=3, ), ) offered = compact_schemas(contract.schemas()) assert required_read_tool_choice(contract, offered) == { 'type': 'function', 'function': {'name': 'manage_contact'}, } assert required_read_tool_choice(contract, offered, calls=1) is None assert sealed_read_arguments( contract, 'manage_contact', {'action': 'delete', 'id': 'wrong'} ) == {'action': 'list'} assert sealed_read_arguments( contract, 'manage_contact', {'action': 'delete'}, calls=1 ) == {'action': 'delete'} def test_server_sealed_document_list_enforces_contract_limit(): from src.turn_contract import RequiredReadOperation, resolve_turn_contract contract = resolve_turn_contract( capabilities={'documents'}, schemas=FUNCTION_TOOL_SCHEMAS, policy=ToolPolicy(), required_read_operation=RequiredReadOperation( 'manage_documents', {'action': 'list'}, max_items=3, ), ) assert sealed_read_arguments( contract, 'manage_documents', {'action': 'list'} ) == {'action': 'list', 'limit': 3} def test_server_sealed_calendar_read_preserves_model_resolved_range(): from src.turn_contract import RequiredReadOperation, resolve_turn_contract contract = resolve_turn_contract( capabilities={'calendar'}, schemas=FUNCTION_TOOL_SCHEMAS, policy=ToolPolicy(), required_read_operation=RequiredReadOperation( 'manage_calendar', {'action': 'list_events'}, max_items=3, ), ) assert sealed_read_arguments( contract, 'manage_calendar', { 'action': 'list_events', 'start': '2026-09-12T00:00:00', 'end': '2026-09-13T00:00:00', 'summary': 'unsafe mutation field', }, ) == { 'action': 'list_events', 'start': '2026-09-12T00:00:00', 'end': '2026-09-13T00:00:00', } def test_email_identifier_guard_rejects_placeholder_after_failed_listing(): history = [ {'role': 'assistant', 'tool_calls': [{ 'id': 'call-list', 'function': {'name': 'list_emails', 'arguments': '{}'}, }]}, {'role': 'tool', 'tool_call_id': 'call-list', 'content': json.dumps({ 'exit_code': 1, 'error': 'email backend unavailable', })}, ] error = email_identifier_error( 'mcp__email__read_email', {'message_id': ''}, history=history, ) assert error and 'placeholders are not executable' in error def test_email_identifier_guard_accepts_successful_result_user_id_and_active_email(): history = [ {'role': 'assistant', 'tool_calls': [{ 'id': 'call-list', 'function': {'name': 'list_emails', 'arguments': '{}'}, }]}, {'role': 'tool', 'tool_call_id': 'call-list', 'content': 'Subject: Hello\nUID: 8492'}, {'role': 'system', 'content': 'Active email context\nMessage UID: open-44'}, ] assert email_identifier_error('read_email', {'uid': '8492'}, history=history) is None assert email_identifier_error('draft_email_reply', {'uid': 'open-44'}, history=history) is None assert email_identifier_error( 'read_email', {'uid': 'user-77'}, user_text='read email UID user-77', history=history, ) is None assert email_identifier_error('read_email', {'uid': 'invented'}, history=history) def test_email_identifier_guard_decodes_mcp_stdout_before_extracting_uid(): history = [ {'role': 'assistant', 'tool_calls': [{ 'id': 'call-search', 'function': { 'name': 'mcp__email__search_emails', 'arguments': '{}', }, }]}, {'role': 'tool', 'tool_call_id': 'call-search', 'content': json.dumps({ 'stdout': ( 'Found 1 email(s):\nUID: 104\n' 'Account: Research Mail ' ), 'stderr': '', 'exit_code': 0, })}, ] assert email_identifier_error( 'mcp__email__draft_email_reply', {'uid': '104'}, history=history, ) is None def test_email_identifier_guard_accepts_uid_nested_in_successful_stdout_result(): history = [ {'role': 'assistant', 'tool_calls': [{ 'id': 'call-search', 'function': {'name': 'mcp__email__search_emails', 'arguments': '{}'}, }]}, {'role': 'tool', 'tool_call_id': 'call-search', 'content': json.dumps({ 'stdout': 'Found 1 email\nUID: 104\nAccount: Research Mail', 'stderr': '', 'exit_code': 0, })}, ] assert email_identifier_error( 'mcp__email__draft_email_reply', {'uid': '104'}, history=history, ) is None def test_canonical_result_renderers_honor_explicit_user_count_limits(): assert requested_item_limit('return at most three titles', default=20) == 3 assert requested_item_limit('whats on my notes list? three at most, dont touch anything', default=20) == 3 assert requested_item_limit('show only 2', default=20) == 2 assert requested_item_limit('pull them up, just 3 short ones', default=20) == 3 assert requested_item_limit('list my notes please, 3 titles max', default=20) == 3 assert requested_item_limit('list those again, three only', default=20) == 3 assert requested_item_limit('maybe first three titles', default=20) == 3 assert requested_item_limit('keep it to three titles', default=20) == 3 assert requested_item_limit('top 3 titles', default=20) == 3 assert requested_item_limit('list those again, 3', default=20) == 3 assert requested_item_limit('three names and their statuses max', default=20) == 3 assert requested_item_limit('just titles, 3 max, read only pls', default=20) == 3 assert requested_item_limit( 'what tasks do i have scheduled? 3 is fine, need name + status', default=20, ) == 3 assert requested_item_limit( 'can you check my calendar and give me the next 3 events? titles and times only', default=20, ) == 3 assert requested_item_limit('show me my saved memories, only a few', default=20) == 3 assert requested_item_limit("what's in my memory? just a few", default=20) == 3 assert requested_item_limit( 'Can you show my notes? I only need three titles. Read-only, keep it short.', default=20, ) == 3 assert requested_item_limit('what notes have i got? 3 titles tops', default=20) == 3 assert requested_item_limit( 'list my memorries, three short ones, read-only please', default=20, ) == 3 assert requested_item_limit('list those again, three tops', default=20) == 3 assert requested_item_limit('same again but cap it at three, read only', default=20) == 3 assert requested_item_limit('what tasks are set up? no more than 3 names', default=20) == 3 assert requested_item_limit('i only want 3 names and statuses', default=20) == 3 assert requested_item_limit( "only need the first three names + whether theyre running or paused", default=20, ) == 3 assert requested_item_limit('three short bits, dont change anything', default=20) == 3 assert requested_item_limit( 'what do you remember about me? show me like three things max', default=20, ) == 3 assert requested_item_limit( 'what scheduled tasks do i have? three names + status max', default=20, ) == 3 assert requested_item_limit( 'can u list my calendar? three titles and times max, no edits', default=20, ) == 3 notes = '- [a] **One**\n- [b] **Two**\n- [c] **Three**\n- [d] **Four**' rendered_notes = notes_terminal_response(notes, user_text='List at most three titles') assert 'Four' not in rendered_notes assert '