diff --git a/src/turn_contract.py b/src/turn_contract.py index f0ad72c76..67cf0905e 100644 --- a/src/turn_contract.py +++ b/src/turn_contract.py @@ -48,9 +48,9 @@ _FAMILY_WORDS = { "memory": r"\b(?:memory|memories|memores|remember|forget|past\s+chats?|previous\s+conversations?)\b", "documents": r"\b(?:documents?|documets?|docs?|editor)\b", "email": r"\b(?:emails?|inbox|mailbox|mail|spam)\b", - "search_browser": r"\b(?:search\s+(?:the\s+)?web|web|online|browse|browser|websites?|sites?|news|weather|youtube|hugging\s*face)\b|https?://|\b\w+\.(?:com|org|net|io)\b", + "search_browser": r"\b(?:search\s+(?:the\s+)?web|web|online|browse|browser|websites?|sites?|news|weather|youtube|arxiv|hugging\s*face)\b|https?://|\b\w+\.(?:com|org|net|io)\b", "shell_files": r"\b(?:files?|folders?|directory|shell|terminal|workspace|repo|repository|python|hostname|b?ssh|bash)\b", - "cookbook_admin": r"\b(?:cookbo{1,2}k|endpoints?|models?|servers?|settings|downloads?|integrations?)\b", + "cookbook_admin": r"\b(?:cookbo{1,2}k|endpoints?|models?|servers?|settings|integrations?)\b", "research": r"\bresearch\b", "contacts": r"\bcontacts?\b", "sessions": r"\b(?:sessions?|chats?|conversations?)\b", @@ -933,6 +933,7 @@ def selected_tools_for_request(message: str) -> frozenset[str] | None: return frozenset({"web_search"}) if ( re.match(r"^\s*" + _REQUEST_PREFIX + r"(?:open|navigate|browse|visit|go\s+to)\b", text, re.I) + and not re.search(r"(?:file://)?/(?:tmp_)?workspace/", text, re.I) and ( re.search(r"\bhttps?://[^\s<>\"']+", text, re.I) or re.search(r"\b(?:[a-z0-9-]+\.)+(?:com|org|net|io|ai|jp|co\.jp)\b", text, re.I) @@ -1115,20 +1116,22 @@ def selected_tools_for_request(message: str) -> frozenset[str] | None: re.I, ): return frozenset({"list_models"}) - if ( - re.search(r"\b(?:models?|qwen|llama|gemma|mistral|instruct)\b", text, re.I) + model_discovery_clauses = re.split(r"[\n.!?;]+", text) + if any( + re.search(r"\b(?:models?|qwen|llama|gemma|mistral|instruct)\b", clause, re.I) and re.search( r"\b(?:i(?:['’]?m|\s+am)\s+after|look(?:ing)?\s+for|find|search|show|" r"recommend|suggest|anything\s+in)\b", - text, + clause, re.I, ) and re.search( r"\b(?:hugging\s*face|huggingface|hf|small|local(?:ly)?|at\s+home|" r"\d+(?:\.\d+)?\s*[-–]\s*\d+(?:\.\d+)?\s*b|\d+(?:\.\d+)?b)\b", - text, + clause, re.I, ) + for clause in model_discovery_clauses ): # Model discovery belongs to the Hugging Face catalog. This covers # natural recommendation wording, not only the literal phrase @@ -1336,6 +1339,12 @@ def selected_tools_for_request(message: str) -> frozenset[str] | None: re.search(r"\b(?:grab|download)\b", text, re.I) and re.search(r"\b[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+\b", text) and re.search(r"\b(?:locally|local|download)\b", text, re.I) + and re.search( + r"\b(?:models?|qwen|llama|gemma|mistral|safetensors|gguf|" + r"hugging\s*face|huggingface|hf\s+hub|model\s+hub)\b", + text, + re.I, + ) ): return frozenset({"download_model"}) if ( @@ -1603,7 +1612,7 @@ def selected_tools_for_request(message: str) -> frozenset[str] | None: re.I, ) and not re.search(r"(?:\.pdf(?:[?#]|$)|/pdf/)", urls[0], re.I) - and not re.search(r"(?:^|\s)(?:file://)?/workspace/", text, re.I) + and not re.search(r"(?:file://)?/(?:tmp_)?workspace/", text, re.I) and not re.search( r"\b(?:browse|navigate|click|fill|submit|private[_ -]?browser|" r"save|write|create|export|render|generate|send|email)\b", @@ -3135,15 +3144,17 @@ def required_read_operation_for_request(message: str, history: Iterable = ()) -> and not re.search(r"\b(?:add|delete|remove|enable|disable|reconnect|change)\b", text, re.I) ): return RequiredReadOperation("manage_mcp", {"action": "list_tools"}, maximum) - if ( - re.search(r"\b(?:agent\s+)?tools?\b", text, re.I) + tool_inventory_clauses = re.split(r"[\n.!?;]+", text) + if any( + re.search(r"\b(?:agent\s+)?tools?\b", clause, re.I) and re.search( r"\b(?:disabled|enabled|available|unavailable|toggles?|" r"switched\s+(?:off|on)|turned\s+(?:off|on))\b", - text, + clause, re.I, ) - and re.search(r"\b(?:what|which|wich|show|list|check)\b", text, re.I) + and re.search(r"\b(?:what|which|wich|show|list|check)\b", clause, re.I) + for clause in tool_inventory_clauses ): return RequiredReadOperation("manage_settings", {"action": "list_tools"}, maximum) if re.match( @@ -3646,6 +3657,23 @@ def _clause_capabilities(text: str) -> set[str]: # only as the forbidden side effect (for example, "do not create a file"). if _PURE_ACTION_PROHIBITION.fullmatch(text): return set() + if re.fullmatch( + r"\s*(?:please\s+)?solve\s+(?:the|this)\s+task\s+efficiently\s+" + r"before\s+(?:the\s+)?timeout(?:\s*\([^)]*\))?\s*", + text, + re.I, + ): + # Execution boilerplate describes the current turn; it is not a + # request to operate on the user's background-task scheduler. + return set() + if delegated := re.match( + r"^\s*(?:your|the)\s+task\s+is\s+to\s+(?P[\s\S]+)$", + text, + re.I, + ): + # ``task`` labels the current instruction here; route the actual + # request body instead of granting background-scheduler authority. + return _clause_capabilities(delegated["request"]) if conditional := _CONDITIONAL_ACTION.fullmatch(text): # The premise supplies context; the post-condition clause owns the # requested side effect and therefore its product family. @@ -4837,7 +4865,26 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum ): # A filename is workspace data, even when its stem is a product name # such as notes.txt or calendar.json. - return frozenset({"shell_files"}) + families = {"shell_files"} + if concrete_urls: + families.add("search_browser") + elif ( + re.search(r"\barxiv\b", text, re.I) + and re.search( + r"\b(?:fetch|retrieve|get|download|search|find|read|inspect|prepare|digest|identify|recover)\b", + text, + re.I, + ) + ): + # Creating a local artifact does not replace the explicitly named + # external source needed to populate it. + families.add("search_browser") + elif ( + re.search(r"\bgithub\b", text, re.I) + and re.search(r"\b(?:repositor(?:y|ies)|repos?|contributors?|commits?|pushed_at)\b", text, re.I) + ): + families.add("search_browser") + return frozenset(families) if re.match( r"^\s*what(?:['’]?s|\s+is)\s+happening\s+(?:in|with|around)\b" r"[^?!.]{1,180}\b(?:lately|recently|right\s+now)\b", @@ -5311,6 +5358,25 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum clauses = re.split(r"[;\n]|[.!?]\s+|\b(?:and|then)\s+(?=" + _ACTION_REQUEST + r")", text, flags=re.I) families = set().union(*(_clause_capabilities(clause) for clause in clauses)) + if re.search(r"(?:file://)?/tmp_workspace(?:/|\b)", text, re.I): + families.add("shell_files") + if ( + re.search(r"\bgithub\b", text, re.I) + and re.search(r"\b(?:repositor(?:y|ies)|repos?|contributors?|commits?|pushed_at)\b", text, re.I) + ): + families.add("search_browser") + if ( + re.search(r"\barxiv\b", text, re.I) + and re.search( + r"\b(?:fetch|retrieve|get|download|search|find|read|inspect|prepare|digest|identify|recover)\b", + text, + re.I, + ) + ): + # arXiv is an external paper source. Long artifact requests often put + # the retrieval verb and ``arXiv`` in different list items, so routing + # each clause independently can otherwise leave only local file tools. + families.add("search_browser") if (recent == ("email",) and re.search(r"\b(?:from\s+them|latest\s+one|that\s+(?:message|email))\b", text, re.I)): families.add("email") diff --git a/tests/test_turn_contract.py b/tests/test_turn_contract.py index 4065c3135..cb1b46b09 100644 --- a/tests/test_turn_contract.py +++ b/tests/test_turn_contract.py @@ -523,6 +523,28 @@ def test_local_model_discovery_selects_hugging_face_search_and_persistence_switc ) == {"notes"} +def test_unrelated_model_terms_do_not_combine_into_hugging_face_discovery(): + prompt = ( + "You are in a restricted environment. Use the available tools. Prepare my daily " + "arXiv paper digest. Classify papers under Multimodal / Vision-Language Models. Based " + "on my research interests, highlight papers I might find interesting. If any " + "paper benchmarks against CapRL, extract the comparison results. Save the digest " + "to /tmp_workspace/results/digest.md.\n" + "| CapRL-3B | result |" + ) + assert selected_tools_for_request(prompt) != {"search_hf_models"} + assert "search_browser" in requested_capabilities(prompt) + + +def test_execution_boilerplate_does_not_route_to_background_tasks(): + prompt = ( + "Solve the task efficiently before the timeout (600s). Use the available tools. " + "Unpack /tmp_workspace/images.tar and classify the images into output folders." + ) + capabilities = requested_capabilities(prompt) + assert capabilities == {"shell_files"} + + def test_referential_web_source_relationship_keeps_web_tools_warm(): history = [{"role": "assistant", "content": "Official link.", "metadata": { "tool_events": [{"tool": "web_search", "exit_code": 0}], @@ -1967,6 +1989,24 @@ def test_explicit_repo_download_selects_tracked_cookbook_download(): assert requested_capabilities(message) == {"cookbook_admin"} +def test_arxiv_source_download_is_not_a_model_download(): + message = ( + "Download the source package from https://arxiv.org/abs/2501.07888, " + "extract every table, and save each one to /tmp_workspace/results/1.tex." + ) + assert selected_tools_for_request(message) != {"download_model"} + assert requested_capabilities(message) == {"search_browser", "shell_files"} + + +def test_single_web_page_with_workspace_outputs_is_not_sealed_to_fetch_only(): + message = ( + "Visit https://example.org/catalog and save every item under " + "/tmp_workspace/results/items/ plus a summary.jsonl file." + ) + assert selected_tools_for_request(message) is None + assert requested_capabilities(message) == {"search_browser", "shell_files"} + + def test_even_have_email_accounts_selects_account_inventory(): message = "do i even have any email accounts connected here?" assert selected_tools_for_request(message) == {"list_email_accounts"} diff --git a/tests/test_turn_contract_read_operations.py b/tests/test_turn_contract_read_operations.py index 997e7de45..6e5c6e24e 100644 --- a/tests/test_turn_contract_read_operations.py +++ b/tests/test_turn_contract_read_operations.py @@ -278,6 +278,15 @@ def test_quick_web_lookup_with_official_link_selects_search(): assert requested_capabilities(message) == {"search_browser"} +def test_available_tools_execution_boilerplate_is_not_a_tool_inventory_request(): + message = ( + "Solve the task efficiently before the timeout. Use the available tools and " + "as many iterative steps as needed. Fetch today's papers, classify every paper " + "into exactly one category, and report which paper is most relevant." + ) + assert required_read_operation_for_request(message) is None + + @pytest.mark.parametrize(("message", "expected"), [ ( "whats on my agenda today? just the titles, dont change anything",