"""Preserve legacy text-call/listing semantics and bound hostile scans.""" import itertools import random import re import subprocess import sys import textwrap import pytest from src.agent_loop import ( _calendar_listing_row, _captures_after_first_prefix, _contains_email_draft_headers, _looks_like_agent_reasoning_preamble, _looks_like_ody_qwen_leaked_tool_text, _email_account_label, _email_sender_name, _contextual_summary_fragment, _is_terse_link_request, _is_terse_email_lookup_followup, _looks_like_destructive_request, _looks_like_youtube_tool_turn, _mentions_pdf_url, _mentions_workspace_script, _numbered_row_parenthesized_ids, _pipeline_request_parts, _parse_explicit_open_panel_request, _private_browser_product_query, _read_only_shell_command, _remaining_checklist_name, _research_listing_row, _session_link_from_row, _session_find_query_capture, _session_list_summary_from_tool_output, _split_before_assistant_prompt, _split_note_items, _strip_trailing_done, _strip_horizontal_space_before_lf, _skill_listing_row, ) from src.clean_agent_preview import ( _page_listing_request, _prior_web_source_request, _terminal_source_link_clause, _web_source_rows, declared_workspace_artifacts, ) from src.text_scanning import ( contains_detailed_sequence_request, contains_search_engine_navigation, iter_angle_contents, iter_markdown_links, first_tag_content, replace_markdown_links_with_labels, ) from src.turn_contract import ( _WARM_RECALL, _WARM_RECALL_WITH_FOLLOWUP, _is_pure_action_prohibition, _mentions_under_budget, _mentions_workspace_artifact, _mentions_workspace_output, _strip_cancelled_request_lead, _strip_terminal_but, _terminal_clause_match, _warm_recall_parts, ) from src.tool_parsing import ( _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE, _QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE, _QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE, _iter_named_blocks, _iter_qwen_python_args, _strip_delimited, parse_tool_blocks, strip_tool_blocks, strip_angle_tags, iter_email_addresses, ) _SESSION_LINK = re.compile(r"(\[(?:\\.|[^\]])+\]\(#session-[^)]+\))") _GEMMA = re.compile( r"<\|?tool_call\|?>\s*call:([\w\d_-]+)\s*(\{[\s\S]*?\})\s*<\|?tool_call\|?>", re.I, ) _QWEN_FUNCTION = re.compile(r"\s*([\s\S]*?)\s*") _QWEN_PARAMETER = re.compile(r"\s*([\s\S]*?)\s*") _QWEN_ARGS = re.compile(r"([A-Za-z_]\w*)\s*=\s*(['\"].*?['\"]|[^,]+)") _WORKSPACE_SCRIPT = re.compile( r"/workspace/[^\s`\"']+\.(?:py|pyw|sh|bash|js|mjs|ts|rb|pl)\b", re.I ) _PDF_URL = re.compile(r"https?://\S+(?:\.pdf\b|/pdf/)", re.I) _WORKSPACE_ARTIFACT = re.compile( r"(?:file://)?/workspace/[^\s`\"']+\.(?:csv|html?|json|md|svg|txt)\b", re.I ) _WORKSPACE_OUTPUT = re.compile( r"(?:file://)?/workspace/(?!input/)[^\s`\"']+\." r"(?:csv|html?|json|md|svg|txt|avif|bmp|gif|jpe?g|png|webp|pdf|mp4|webm)\b", re.I, ) _SEARCH_ENGINE_URL = re.compile( r"https?://(?:[^/]+\.)?(?:google\.[^/]+|bing\.com|duckduckgo\.com)" r"/(?:search|sorry|html|lite|\?)", re.I, ) _LEAKED_TOOL_TEXT = re.compile( r"(<\s*/?\s*(?:function|parameter|tool_call)\b|(?:^|\n)\s*(?:function|parameter)\s*=" r"|\bmanage_(?:notes|calendar|memory|documents|contact)\s*\(|\"function\"\s*:\s*\"(?:manage_|mcp__)" r"|mcp__email__|(?:^|\n)\s*(?:web_search|web_fetch|private_browser)\s*:)", re.I, ) _VIDEO_DETAIL_BASE = re.compile( r"\b(?:how\s+many|count|break\s*points?|timestamps?|what\s+time|" r"when\s+.*(?:end|happen)|score(?:board)?s?)\b", re.I, ) _VIDEO_DETAIL_EXTENDED = re.compile( r"\b(?:how\s+many|count|sequence|in\s+order|chronological|timestamps?|" r"what\s+time|at\s+what\s+time|when\s+.*(?:end|happen)|" r"first\s+.*(?:save|attempt|event)|score(?:board)?s?)\b|" r"(?:多少|几次|何时|什么时候|时间|顺序)", re.I, ) def test_session_link_matches_legacy_escape_and_greedy_semantics(): cases = [ "no links", "[](#session-id)", "[x](#session-)", "[x](#session-id)", "[[x](#session-id)", r"[x\](#session-id)", r"[x\]more](#session-id)", r"[x\](#session-first)\](#session-last)", r"[x\](#session-first)\](#session-)", r"[x\](#session-first)\](#session-unclosed", "[bad] then [good](#session-id)", "[x](#session-id[has]brackets)", "[x](#session-first) [y](#session-second)", ] rng = random.Random(6503) tokens = ["[", "]", "\\", "a", "(", ")", "(#session-id)", "(#session-)"] cases += ["".join(rng.choices(tokens, k=12)) for _ in range(2000)] cases += ["[" + "".join(label) + "](#session-id)" for n in range(5) for label in itertools.product("a[]\\", repeat=n)] for row in cases: expected = _SESSION_LINK.search(row) assert _session_link_from_row(row) == (expected.group(0) if expected else ""), row @pytest.mark.parametrize("row,expected", [ ("- **[Chat](#session-id)** (id: `id`, model: qwen, 2 msgs, last active today)", "- [Chat](#session-id) (last active today)"), ("- [Chat](#session-id) (irrelevant) (model: qwen) (last active later)", "- [Chat](#session-id)"), ("- [Chat](#session-id) (outer (last active today))", "- [Chat](#session-id) (last active today)"), ("- plain row", "- plain row"), ]) def test_session_summary_preserves_link_and_metadata(row, expected): assert _session_list_summary_from_tool_output("Chats:\n" + row) == "Chats:\n" + expected def test_gemma_delimiters_match_legacy_parse_and_strip(): cases = [ "ordinary prose", "<|tool_call|>call:web_search{query: 'news'}<|tool_call|>", "before call:read-file {\npath: 'README.md'\n} after", "call:x{a}call:y{b}", "call:x{call:y{b}", "call:x{unclosed", "}call:x{unclosed", ] for text in cases: actual = [(name, "{" + body + "}") for name, body in _iter_named_blocks( text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE )] assert actual == _GEMMA.findall(text) assert _strip_delimited(text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE) == _GEMMA.sub("", text) raw = '<|tool_call|>call:web_search{"query":"news"}<|tool_call|>' assert [(b.tool_type, b.content) for b in parse_tool_blocks(raw)] == [("web_search", "news")] assert strip_tool_blocks(raw) == "" @pytest.mark.parametrize("reference,opener,closer,tokens", [ (_QWEN_FUNCTION, _QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE, ["", "", "a", "\n", "\t", " "]), (_QWEN_PARAMETER, _QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE, ["", "", "a", "\n", "\t", " "]), ]) def test_qwen_delimiters_preserve_names_and_stripped_values(reference, opener, closer, tokens): rng = random.Random(6503) for _ in range(1000): text = "".join(rng.choices(tokens, k=16)) expected = [(name, body.strip()) for name, body in reference.findall(text)] actual = [(name, body.strip()) for name, body in _iter_named_blocks(text, opener, closer)] assert actual == expected def test_qwen_python_arguments_preserve_permissive_legacy_grammar(): cases = ["action='list', limit=5", "action = \"list\"", "1key=2", "aé=4", "key='mismatched\"", "broken word, okay=2", "a=\t, b=3", "a=\n'hi'", "a='hi\nthere'"] rng = random.Random(6503) tokens = ["key", "1", "中", "é", "=", "\n", " ", "\t", "'", '"', ",", "-", "_", "[]"] cases += ["".join(rng.choices(tokens, k=20)) for _ in range(2000)] for text in cases: assert list(_iter_qwen_python_args(text)) == _QWEN_ARGS.findall(text), text @pytest.mark.parametrize("text", [ "Done.", "Done.\nUpdated the document.\nDone.", "Done. undone.", "Done.\nDONE.\t", "Done.\u2003Done.\u2003", "Done. no terminal marker ", ]) def test_trailing_done_matches_legacy_cleanup(text): pattern = re.compile(r"\s*Done\.\s*$", re.I) expected = pattern.sub("", text).rstrip() if pattern.search(text) else text assert _strip_trailing_done(text) == expected @pytest.mark.parametrize("allow_empty", [False, True]) @pytest.mark.parametrize("replacement", ["", " "]) def test_angle_tag_cleanup_matches_legacy_flat_grammar(allow_empty, replacement): pattern = re.compile(r"<[^>]*>" if allow_empty else r"<[^>]+>") cases = ["beforeboldafter", "<>", "<<>>", "ab", "one", "a > b", "tail"] cases += ["".join(parts) for n in range(6) for parts in itertools.product("a<>\n", repeat=n)] for text in cases: assert strip_angle_tags(text, replacement, allow_empty=allow_empty) == pattern.sub(replacement, text) @pytest.mark.parametrize("ascii_only", [False, True]) def test_email_scanner_preserves_legacy_address_sets(ascii_only): pattern = re.compile( r"[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}" if ascii_only else r"[\w.+-]+@[\w.-]+\.\w+" ) cases = ["a@b.example", "a+b@b.example, c@d.example", "a@b@c.example", "a@b.c+d@e.f", "你好@例子.中国", "a@b.-.com", "'a'@example.com", "a@b.x", "a@.com", "a@b...", "a@b.c@d.example"] rng = random.Random(6503) tokens = ["a", "b", "é", "中", "@", ".", "-", "+", " ", "_", "'", "!", "/"] cases += ["".join(rng.choices(tokens, k=30)) for _ in range(3000)] for text in cases: assert list(iter_email_addresses(text, ascii_only=ascii_only)) == pattern.findall(text), text def test_prefixed_workspace_and_url_detectors_match_legacy_searches(): cases = [ "", "/workspace/a.py", "FILE:///WORKSPACE/report.HTML", "/workspace/input/a.png", "/workspace/input/a.png/workspace/out.png", "http://example.test/a.pdf", "http://x/http://example.test/pdf/view", "https://google.com/search?q=x", "http://news.google.co.uk/sorry/index", "https://x.bing.com/html", "http://bad/http://duckduckgo.com/?q=x", ] rng = random.Random(6503) tokens = ["/workspace/", "input/", "file://", "http://", "https://", ".py", ".pdf", ".html", "/pdf/", "google.", "bing.com/", "search", "x", " ", "'", "/"] cases += ["".join(rng.choices(tokens, k=18)) for _ in range(4000)] for text in cases: assert _mentions_workspace_script(text) == bool(_WORKSPACE_SCRIPT.search(text)), text assert _mentions_pdf_url(text) == bool(_PDF_URL.search(text)), text assert _mentions_workspace_artifact(text) == bool(_WORKSPACE_ARTIFACT.search(text)), text assert _mentions_workspace_output(text) == bool(_WORKSPACE_OUTPUT.search(text)), text assert contains_search_engine_navigation(text) == bool(_SEARCH_ENGINE_URL.search(text)), text def test_declared_workspace_artifacts_keeps_legacy_greedy_paths(): pattern = re.compile(r"/workspace/[^\s,,、;;`\"'<>]+\.[A-Za-z0-9]{1,12}", re.I) cases = [ "create /workspace/report.csv", "read /workspace/input.csv and create /workspace/out.json", "create /workspace/a.csv/workspace/b.json", "create /workspace/noext /workspace/out.txt", ] for text in cases: expected = [] for match in pattern.finditer(text): path = match.group(0).rstrip(".!?))]}") if path.startswith("/workspace/fixtures/") or path in expected: continue before = text[max(0, match.start() - 240):match.start()] before = pattern.sub("[workspace file]", before) clause = re.split(r"[.;!?\n]", before)[-1] if re.search( r"\b(?:from|using|inspect|read|open|analy[sz]e|transcribe|extract\s+(?:text\s+)?from|" r"input(?:\s+file)?(?:\s+is)?|source(?:\s+file)?(?:\s+is)?)\s*(?::|=)?\s*$", clause, re.I, ) or re.search(r"\b(?:read_file|inspect_media|extract_text|transcribe_media|pdf_extract)\b", clause, re.I): continue if not re.search( r"\b(?:create|write|save|export|render|generate|produce|output|deliver|store|convert|make)\b|" r"\b(?:write_file|output_path)\b", clause, re.I, ): continue expected.append(path) assert declared_workspace_artifacts(text) == tuple(expected) def test_intent_and_detail_detectors_preserve_short_legacy_language(): cases = [ "plain answer", "\n\nfunction = manage_notes", "mcp__email__read_email", "< function >", "web_search: cats", "prefix\n private_browser : url", "when does it end", "when\nwill it end", "first save event", "sequence please", "I can now carefully inspect the result.", "Answer. Now let me verify this.", "接下来我会查看结果", "。 让我先检查", ] rng = random.Random(6503) tokens = ["\n", " ", ".", "!", "function", "parameter", "=", "web_search", ":", "when", "first", "end", "event", "let me", "inspect", "x"] cases += ["".join(rng.choices(tokens, k=24)) for _ in range(3000)] for text in cases: assert _looks_like_ody_qwen_leaked_tool_text(text) == bool(_LEAKED_TOOL_TEXT.search(text)), text assert contains_detailed_sequence_request(text, include_first=False) == bool(_VIDEO_DETAIL_BASE.search(text)), text assert contains_detailed_sequence_request(text) == bool(_VIDEO_DETAIL_EXTENDED.search(text)), text for text in ( "The result is partial. Now let me inspect the rest.", "\n" * 20 + "let me check the source", "。\n 接下来我会查看结果", "A complete factual answer.", ): assert isinstance(_looks_like_agent_reasoning_preamble(text), bool) def test_suffix_and_split_helpers_preserve_legacy_results(): panel_cases = [ ("open calendar again!!!", ("ui_control", "open_panel calendar")), ("open calendar" + " " * 50 + "again?", ("ui_control", "open_panel calendar")), ("open calendar....", ("ui_control", "open_panel calendar")), ("open calendar again x", ("ui_control", "open_panel calendar")), ("open notes day view again.", ("ui_control", "open_panel notes")), ] for text, expected in panel_cases: assert _parse_explicit_open_panel_request(text) == expected values = ["one, two and three", " one ,two and three ", "candy and x", "a,and,b", ""] for value in values: expected_parts = [ re.sub(r"\s+", " ", part).strip(" .") for part in re.split(r"\s*,\s*|\s+\band\b\s+", value) ] expected = [{"text": part, "done": False} for part in expected_parts if part] assert _split_note_items(value) == expected for value in ["Alice ", "Alice <", "Alice <>", "Alice (a@example.com)", "Alice ((a@example.com)", "Alice > x "]: normalized = re.sub(r"\s+", " ", value).strip() expected_sender = re.sub(r"\s*\([^)]*@[^)]*\)\s*$", "", normalized).strip() expected_sender = re.sub(r"\s*<[^>]*>\s*$", "", expected_sender).strip() expected_sender = expected_sender or value.strip() expected_account = re.sub(r"\s*<[^>]+>\s*$", "", normalized).strip() or normalized assert _email_sender_name(value) == expected_sender assert _email_account_label(value) == expected_account for value in ["Reply body\nWant me to send it?", "Reply\n\n SHOULD I continue?", "Want me now", "x\nnot a prompt"]: expected = re.split(r"\n\s*(?:Want me|Would you like|Should I)\b", value, flags=re.I, maxsplit=1)[0] assert _split_before_assistant_prompt(value) == expected for value in ["a \n b\t\n", " x", "a \r\n", "\t\n\n", ""]: assert _strip_horizontal_space_before_lf(value) == re.sub(r"[ \t]+\n", "\n", value) for value in ["pwd && ls", "cat x | grep y", "echo x; printf y", "pwd " + " " * 20 + "&& ls", "pwd || rm x"]: legacy_parts = re.split(r"\s*(?:&&|\|\||;|\|)\s*", value.strip()) allowed = re.compile( r"^(?:pwd|ls|find|rg|grep|git\s+(?:status|diff|log|show|branch)|sed(?!\s+-i\b)|" r"head|tail|cat|stat|file|wc|sort|uniq|cut|ip|ipconfig|getent|nslookup|dig|arp|" r"hostname|uname|whoami|echo|printf|test|true|false|:)\b", re.I, ) expected = bool(legacy_parts) and all(part.strip() for part in legacy_parts) and all( allowed.match(part.strip()) for part in legacy_parts ) assert _read_only_shell_command(value) == bool(expected) @pytest.mark.parametrize("prefix,target_pattern", [ ("#session-", r"[^)]+"), ("#note-", r"[^)]+"), ("#event-", r"[0-9a-fA-F-]{8,64}"), ("", r"[^)]+"), ]) def test_forward_markdown_and_angle_scanners_match_legacy(prefix, target_pattern): reference = re.compile(r"\[([^\]]+)\]\(" + re.escape(prefix) + "(" + target_pattern + r")\)") validator = None if target_pattern == r"[^)]+" else re.compile(target_pattern) cases = ["[title](#session-id)", "[[title](#session-id)", "[](x)", "[a](x)[b](y)", "[a](#event-12345678)", "[a](broken"] rng = random.Random(6503) tokens = ["[", "]", "(", ")", "#session-", "#note-", "#event-", "a", "1", "-", " "] cases += ["".join(rng.choices(tokens, k=28)) for _ in range(3000)] for text in cases: expected = reference.findall(text) actual = [(label, target) for _start, _end, label, target in iter_markdown_links( text, target_prefix=prefix, target_re=validator )] assert actual == expected, text generic = re.compile(r"\[([^\]]+)\]\([^)]+\)") for text in cases: assert replace_markdown_links_with_labels(text) == generic.sub(r"\1", text), text angle = re.compile(r"<([^>]+)>") for text in cases + ["", "<", "<>", "", "<<<"]: assert [content for _start, _end, content in iter_angle_contents(text)] == angle.findall(text) def test_numbered_row_identifier_scan_matches_legacy_rows(): pattern = re.compile(r"^\s*\d+\.\s+.+?\s+\(([^)\n]+)\)\s+[—-]", re.M) cases = [ "1. Task (abc) — due", " 2. Label ((nested) - tail", "1. no id", "1. a (bad) x (good) — tail", "\n\n3. item (id-3) - tail", ] rng = random.Random(6503) tokens = ["1. ", "x", " ", "(", ")", " -", " —", "\n"] cases += ["".join(rng.choices(tokens, k=20)) for _ in range(2000)] for text in cases: assert _numbered_row_parenthesized_ids(text) == pattern.findall(text), text def test_model_and_request_extractors_match_legacy_grammars(): youtube = re.compile( r"\b(?:youtube|youtu\.be|yt|video\s+comments?|comments?\s+on\s+(?:the\s+)?video|" r"transcript\s+(?:of|for)|(?:latest|newest|recent)\s+(?:\d+\s+)?(?:videos?|uploads?)|" r"official\s+.+\s+channel)\b", re.I, ) destructive = re.compile(r"\b(delete|remove|archive|trash|send|reply|unsubscribe|mark\s+.*read)\b", re.I) terse = re.compile( r"\s*(?:(?:send|sned|share|give|show)?\s*(?:me\s+)?(?:the\s+)?(?:links?|urls?|sources?)" r"(?:\s+(?:for|to|from)\s+(?:those|that|them|these|it|this|the\s+(?:sites?|websites?|resources?|sources?)))?" r"|(?:for|to|from)\s+(?:those|that|them|these|it|this|the\s+(?:sites?|websites?|resources?|sources?)))" r"\s*(?:please|pls)?[.!?]?\s*", ) cases = ["official project channel", "official x channel", "mark all as read", "mark\nread", " send me the links please ", "for those sites", "plain text"] rng = random.Random(6503) tokens = ["official", "channel", "mark", "read", "send", "links", "for", "those", "sites", " ", "\n", "x"] cases += ["".join(rng.choices(tokens, k=24)) for _ in range(3000)] for text in cases: assert _looks_like_youtube_tool_turn(text) == bool(youtube.search(text)), text assert _looks_like_destructive_request(text) == bool(destructive.search(text)), text assert _is_terse_link_request(text) == bool(terse.fullmatch(text.lower())), text summary_re = re.compile( r"\bsummary\s*(?:\*\*)?\s*:?\s*(.+?)(?:\n\s*(?:-|\\*\\*|If you|Want me|This is|$))", re.I | re.S, ) summary_cases = ["Summary: useful\n- next", "summary **: x\n\nWant me to continue", "summary: no end"] for text in summary_cases: match = summary_re.search(text) assert _contextual_summary_fragment(text) == (match.group(1) if match else "") product_re = re.compile( r"\b(?:find|look\s+for|shop\s+for|search\s+for)\s+(?:me\s+)?(?:the\s+)?(?:best\s+)?" r"(?P.+?)\s*[?.!]*$", re.I, ) for text in ["find best camera???", "look for me the best shoes", "shop for ???", "plain"]: normalized = re.sub(r"\s+", " ", text).strip() match = product_re.search(normalized) expected = "" if match: expected = match.group("query").strip(" \t\r\n.,!?;:") expected = re.sub( r"\s+(?:on|at|from)\s+(?:the\s+)?[A-Za-z0-9&.' -]{1,60}$", "", expected, flags=re.I ).strip() expected = expected[:120] if 0 < len(expected.split()) <= 12 else "" assert _private_browser_product_query(text) == expected title_re = re.compile(r"]*)?>([\s\S]*?)", re.I) for text in ["x", "a<b", "x"]: match = title_re.search(text) assert first_tag_content(text, "title", allow_attributes=True) == (match.group(1) if match else None) def test_summary_capture_preserves_exact_head_whitespace_and_newline_grammar(): legacy = re.compile( r"\bsummary\s*(?:\*\*)?\s*:?\s*(.+?)(?:\n\s*(?:-|\\*\\*|If you|Want me|This is|$))", re.I | re.S, ) cases = [ "Summary: useful\nordinary next line\n- next", "Summary: useful\nx\n", "summary\n\n", "summary \n", "summary: \n", "summary **:\n\n", "summary:\nX\n", "summary summary: X\n", "summary: no newline", ] rng = random.Random(6509) tokens = ["summary", "Summary", ":", "**", "\\", " ", "\t", "\n", "x", "-", "If you"] cases += ["".join(rng.choices(tokens, k=24)) for _ in range(5000)] for text in cases: match = legacy.search(text) assert _contextual_summary_fragment(text) == (match.group(1) if match else ""), text def test_repeated_listing_words_preserve_legacy_request_grammar(): legacy = re.compile( r"\s*(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)\s+" r"(?:[\w .:/-]+\s+)?(?:stories|articles|posts|headlines|pages)" r"(?:\s+on\s+[\w .:/-]+)?[.!?]?\s*", re.I, ) for text in ("top Straße stories", "top İ stories", "lİst stories", "top storİes", "top café stories", "top stories on Straße", "top stories on İ"): assert _page_listing_request(text) == bool(legacy.fullmatch(text)), text for word in ("stories", "articles", "posts", "headlines", "pages"): for count in (1, 2, 8, 32): for suffix in ("X", "@", "\tX", "on x", "on \t", "on ", "\n"): for separator in (" ", " on ", "\t", " \t"): text = "top " + (word + separator) * count + suffix assert _page_listing_request(text) == bool(legacy.fullmatch(text)), text def test_repeated_navigation_schemes_preserve_legacy_url_search(): for count in (1, 2, 8, 32): for suffix in ("X", "google.com/search", "bing.com/html", "duckduckgo.com/?q=x"): text = "http://" * count + suffix assert contains_search_engine_navigation(text) == bool(_SEARCH_ENGINE_URL.search(text)), text def test_tool_listing_rows_match_legacy_grammars(): research = re.compile(r"^-\s+\[(.*?)\]\(#research-([^)]+)\)(.*)$") skills = re.compile(r"^-\s+\*\*(.*?)\*\*(?:\s+\((.*?)\)|\s+\[(draft)\])?(?::\s*(.*))?$") calendar = re.compile(r"^\s*-\s+(.+?):\s+\[(.*?)\]\(#event-([^)]+)\)(.*)$") draft = re.compile(r"\bTo:\s*.+\bSubject:\s*.+\n---", re.I | re.S) cases = [ "- [Title](#research-id) tail", "- **name** (published): description", "- **name** [draft]", " - when: [title](#event-id) tail", "To: a\nSubject: b\n---", "To:Subject:x\n---", "plain", ] rng = random.Random(6504) tokens = ["-", " ", "\t", "[", "]", "(", ")", "*", ":", "#research-", "#event-", "draft", "x"] cases += ["".join(rng.choices(tokens, k=28)) for _ in range(5000)] for text in cases: match = research.match(text) assert _research_listing_row(text) == (match.groups() if match else None), text match = skills.match(text) assert _skill_listing_row(text) == (match.groups() if match else None), text match = calendar.match(text) assert _calendar_listing_row(text) == (match.groups() if match else None), text assert _contains_email_draft_headers(text) == bool(draft.search(text)), text def test_terse_email_followup_matches_legacy_grammar(): legacy = re.compile( r"^\s*(?:and|so|well|still|then|okay|ok|did you find it(?: yet)?|what did you find)\s*[?.!]*\s*$", re.I, ) rng = random.Random(6508) tokens = ["and", "so", "well", "did you find it", " yet", "what did you find", " ", "\t", "?", ".", "!", "x"] cases = ["and?", " did you find it yet ! ", "what did you find", "and X"] cases += ["".join(rng.choices(tokens, k=20)) for _ in range(3000)] for text in cases: assert _is_terse_email_lookup_followup(text) == bool(legacy.match(text)), text def test_preview_request_and_source_scans_match_legacy_grammars(): page = re.compile( r"\s*(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)\s+" r"(?:[\w .:/-]+\s+)?(?:stories|articles|posts|headlines|pages)" r"(?:\s+on\s+[\w .:/-]+)?[.!?]?\s*", re.I, ) prior = re.compile( r"\s*(?:(?:where|what)\s+did\s+you\s+(?:get|find)\s+(?:that|this)\s+from[?., ]*" r"(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link[.!? ]*" r"|(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link(?:\s+for\s+that)?[.!? ]*" r"|what(?:['’]?s|\s+is)\s+(?:the\s+)?source(?:\s+link)?[.!? ]*)\s*", re.I, ) terminal = re.compile( r"(?:^|[.!?;,\n])\s*(?:(?:pls|please)\s+)?(?:sources?|citations?|links?)" r"\s*(?:pls|please)?\s*[.!?]*$", re.I, ) rows = re.compile(r"^\[\d+\]\s+(.+?)\s*\n\s*(https?://\S+)", re.M) cases = [ "top stories", "show me the latest stories on example.com", "give me the source link", "what's the source?", "x. please links pls!!", "[1] Title\nhttps://example.test/x", "plain", ] rng = random.Random(6505) tokens = ["top", "show", " me", " the", " stories", " on", "link", "source", "please", " ", "\t", ".", "!", "\n", "[1]", "http://x"] cases += ["".join(rng.choices(tokens, k=22)) for _ in range(5000)] for text in cases: assert _page_listing_request(text) == bool(page.fullmatch(text)), text assert _prior_web_source_request(text) == bool(prior.fullmatch(text)), text assert _terminal_source_link_clause(text) == bool(terminal.search(text)), text assert _web_source_rows(text) == rows.findall(text), text def test_staged_command_payload_parsers_match_legacy_grammars(): specs = [ ( re.compile(r"\bnote\s+titled\s+(.+?)\s+so\s+its\s+content\s+is\s+['\"]([^'\"]+)['\"]", re.I), r"\bnote\s+titled(?=\s)", r"\s+(.+?)\s+so\s+its\s+content\s+is\s+['\"]([^'\"]+)['\"]", ), ( re.compile(r"\b(?:delete|trash|remove|archive|mark(?:\s+as)?\s+(?:read|unread)|mark\s+(?:read|unread))\b\s+(?:all|every|the)?\s*(?:my\s+)?(.+?)\s+(?:emails?|mail|messages?)\b", re.I), r"\b(?:delete|trash|remove|archive|mark(?:\s+as)?\s+(?:read|unread)|mark\s+(?:read|unread))\b(?=\s)", r"\s+(?:all|every|the)?\s*(?:my\s+)?(.+?)\s+(?:emails?|mail|messages?)\b", ), ( re.compile(r"\b(?:make|create|add)\s+(?:a\s+)?checklist\s+(?:called|titled|named)\s+(.+?)\s+with\s+(.+?)\s*$", re.I), r"\b(?:make|create|add)\s+(?:a\s+)?checklist\s+(?:called|titled|named)(?=\s)", r"\s+(.+?)\s+with\s+(.+?)$", ), ( re.compile(r"\b(?:change|update|set|retag)\b\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b", re.I), r"\b(?:change|update|set|retag)\b(?=\s)", r"\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b", ), ( re.compile(r"\b(?:chang(?:e|es|ed|ing)|updat(?:e|es|ed|ing))\s+(.+?)\s+to\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)", re.I), r"\b(?:chang(?:e|es|ed|ing)|updat(?:e|es|ed|ing))(?=\s)", r"\s+(.+?)\s+to\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)", ), ( re.compile(r"\breplace\s+(.+?)\s+with\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)", re.I), r"\breplace(?=\s)", r"\s+(.+?)\s+with\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)", ), ] pipeline = re.compile( r"\bpipeline\s+using\s+([^\s,]+)\s+to\s+(.+?),\s*then\s+([^\s,]+)\s+to\s+(.+?)(?:[.!?]\s*)?$", re.I, ) session = re.compile(r"\b(?:find|search(?:\s+for)?|show)\s+(?:the\s+)?(.+?)\s+(?:chat|session|conversation)\b", re.I) checklist = re.compile(r"\b(?:what(?:'s| is)?|show|tell\s+me)\b.*?\b(?:left|remaining)\b.*?\b(?:on|in)\s+(?:the\s+)?(.+?)\s+checklist\b", re.I) rng = random.Random(6506) tokens = ["note", " titled", " so its content is ", "'x'", "delete", " all", " emails", "create checklist called", " with", "change", " tag to ", "replace", " to", " pipeline using ", " then", "find", " chat", "left", " on", " checklist", " ", "\t", ".", "x"] cases = ["update note titled a so its content is 'b'", "delete all all emails", "create checklist called a with b", "change the trip tag to work", "replace old with new", "pipeline using a to x, then b to y", "find the chat", "what is left on the trip checklist"] cases += ["".join(rng.choices(tokens, k=25)).strip() for _ in range(5000)] for text in cases: for legacy, prefix, remainder in specs: match = legacy.search(text) assert _captures_after_first_prefix(text, prefix, remainder) == (match.groups() if match else None), (legacy.pattern, text) match = pipeline.search(text) assert _pipeline_request_parts(text) == (match.groups() if match else None), text match = session.search(text) assert _session_find_query_capture(text) == (match.group(1) if match else None), text match = checklist.search(text) assert _remaining_checklist_name(text) == (match.group(1) if match else None), text def test_turn_contract_scans_match_legacy_grammars(): cancelled = re.compile( r"^(?:never\s*mind|scratch\s+that)\s*[,;:—–-]?\s*(?=(?:open|show|list|read|search|find|check|switch|go)\b)", re.I, ) prohibition = re.compile( r"^\s*(?:read[- ]only(?:\s+and)?\s+)?(?:do\s+not|don['’]?t|never)\s+" r"(?:add|create|make|write|draft|edit|change|update|delete|remove|send|reply|run|execute|download|serve|open|save|schedule|transcribe|inspect)\b" r"[^.;\n]*[.!?]*\s*$", re.I, ) budget = re.compile(r"\bunder\s+[¥$€£]?\s*\d+(?:[.,]\d+)?(?:\s*yen)?\b", re.I) terminal_specs = [ r"[,.;?]\s*(?:(?:only|just)\s+)?(?:(?:list|show)\s+(?:me\s+)?)?a\s+few(?:\s+(?:task\s+)?(?:names?|items?|results?|entries?))?(?:\s+and\s+(?:whether|if)\s+[^.;\n]+)?", r"[.;]\s*read[- ]only(?:\s+(?:please|pls|plz))?\s*,?\s*(?:(?:and\s+)?(?:do\s+not|don['’]?t|dont)\s+(?:change|edit|modify)(?:\s+or\s+send)?\s+(?:anything|data))?", r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?(?:do\s+not|don['’]?t|dont)\s+(?:touch|change|edit|modify)(?:\s+(?:anything|data|them))?(?:\s+yet)?", r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+changes?", r"[,;]\s*(?:keep\s+(?:them|it)\s+)?short\s+lines?\s*,?", ] rng = random.Random(6507) tokens = ["never", " mind", "scratch", " that", "open", "do not", " touch", " anything", "no changes", "read-only", "a few", " and whether", "short lines", "under", "$", "123", "yen", "open my calendar", "and", " what is that", " ", "\t", ".", "!", ",", ";", "x"] cases = ["never mind, open calendar", "do not edit anything.", "under $ 20 yen", "open my calendar", "open calendar and what is next", "x but "] cases += ["".join(rng.choices(tokens, k=22)) for _ in range(5000)] for text in cases: assert _strip_cancelled_request_lead(text) == cancelled.sub("", text), text assert _is_pure_action_prohibition(text) == bool(prohibition.fullmatch(text)), text assert _mentions_under_budget(text) == bool(budget.search(text)), text assert _strip_terminal_but(text) == re.sub(r"\s+but\s*$", "", text, flags=re.I), text for core in terminal_specs: legacy = re.search(core + r"[.!?]*\s*$", text, re.I) current = _terminal_clause_match(text, core) assert (current.start() if current else None) == (legacy.start() if legacy else None), (core, text) match = _WARM_RECALL.fullmatch(text) assert _warm_recall_parts(text) == ((match.group("target"), "") if match else None), text match = _WARM_RECALL_WITH_FOLLOWUP.fullmatch(text) assert _warm_recall_parts(text, with_followup=True) == ( (match.group("target"), match.group("followup")) if match else None ), text @pytest.mark.parametrize("program", [ r''' from src.agent_loop import _contextual_summary_fragment assert _contextual_summary_fragment("Summary: x\n" + "x\n" * 100_000) == "x" assert _contextual_summary_fragment("Summary:" + "\n" * 100_000) == "\n" ''', r''' from src.clean_agent_preview import _page_listing_request for text in ("top " + "stories " * 30_000 + "X", "top " + "stories on " * 30_000 + "@"): assert not _page_listing_request(text) ''', r''' from src.text_scanning import contains_search_engine_navigation assert not contains_search_engine_navigation("http://" * 100_000 + "X") assert contains_search_engine_navigation("http://" * 100_000 + "bing.com/search") ''', r''' from src.agent_loop import _is_terse_email_lookup_followup _is_terse_email_lookup_followup("and" + " " * 100_000 + "X") ''', r''' from src.turn_contract import (_is_pure_action_prohibition, _mentions_under_budget, _strip_cancelled_request_lead, _terminal_clause_match, _warm_recall_parts) space = " " * 100_000 _strip_cancelled_request_lead("never" + space + "mind" + space + "X") _terminal_clause_match(", a few and whether " + space + "X;", r"[,.;?]\s*a\s+few(?:\s+and\s+whether\s+[^.;\n]+)?") _is_pure_action_prohibition("do not add " + space + "X;") _mentions_under_budget("under" + space + "$" + space + "X") _warm_recall_parts("open" + space + "my" + space + "calendar" + space + "X") _warm_recall_parts("open" + space + "my" + space + "calendar" + space + "and" + space + "what " + "x" * 181, with_followup=True) ''', r''' from src.agent_loop import (_captures_after_first_prefix, _pipeline_request_parts, _remaining_checklist_name, _session_find_query_capture) evil = ("delete all x " * 30_000) + "z" _captures_after_first_prefix(evil, r"\bdelete\b(?=\s)", r"\s+(?:all)?\s*(.+?)\s+emails?\b") _pipeline_request_parts(("pipeline using m to x " * 30_000) + "z") _session_find_query_capture(("find x " * 30_000) + "z") _remaining_checklist_name("what " + ("left on " * 30_000) + "z") ''', r''' from src.clean_agent_preview import (_page_listing_request, _prior_web_source_request, _terminal_source_link_clause, _web_source_rows) _page_listing_request("top stories on " + " " * 100_000 + "\nX") _prior_web_source_request("give me link" + " " * 100_000 + "X") _terminal_source_link_clause("links" + " " * 100_000 + "X") _web_source_rows("[1] " + " " * 100_000) ''', r''' from src.agent_loop import (_calendar_listing_row, _contains_email_draft_headers, _research_listing_row, _skill_listing_row) for text in ( "- [" + "](#research-" * 30_000, "- **" + " **" * 30_000, "- when: [" + "](#event-" * 30_000, "To: x " + "Subject: x " * 30_000, ): _research_listing_row(text) _skill_listing_row(text) _calendar_listing_row(text) _contains_email_draft_headers(text) ''', r''' from src.agent_loop import _session_link_from_row, _session_list_summary_from_tool_output, _strip_trailing_done evil = "Done. " + "\t" * 100_000 + "x" assert _strip_trailing_done(evil) == evil for row in ("[" + "\\" * 100_000, "[" * 100_000, "[a" + "\\](#session-" * 20_000 + ")"): _session_link_from_row(row) for row in ("[x](#session-id) (" + "msgs" * 100_000, "[x](#session-id) (" + "(last active today)" * 20_000): _session_list_summary_from_tool_output("Chats:\n- " + row) ''', r''' from src.tool_parsing import * from src.tool_parsing import (_iter_named_blocks, _strip_delimited, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE) text = "}<|tool_call|>" + "<|tool_call|>call:web_search{" * 20_000 assert list(_iter_named_blocks(text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE)) == [] assert _strip_delimited(text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE) == text # Public entry points also stay responsive, including fallback parsers. text = "}<|tool_call|>" + "<|tool_call|>call:web_search{" * 3000 assert parse_tool_blocks(text) == [] strip_tool_blocks(text) ''', r''' from src.tool_parsing import (_iter_named_blocks, _iter_qwen_python_args, _QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE, _QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE, parse_tool_blocks) for opener, closer, text in ( (_QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE, "" * 20_000), (_QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE, "" * 20_000), (_QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE, "a" + "\t" * 100_000 + "x"), ): assert list(_iter_named_blocks(text, opener, closer)) == [] assert list(_iter_qwen_python_args("a" * 100_000)) == [] assert parse_tool_blocks("" * 3000) == [] assert parse_tool_blocks("manage_notes(" + "a" * 100_000 + ")") ''', r''' from src.tool_parsing import strip_angle_tags for allow_empty in (False, True): for replacement in ("", " "): for text in ("<" * 200_000, ">" + "<" * 200_000): assert strip_angle_tags(text, replacement, allow_empty=allow_empty) == text assert strip_angle_tags("<" * 200_000 + ">tail", replacement, allow_empty=allow_empty) == replacement + "tail" ''', r''' from src.tool_parsing import iter_email_addresses for ascii_only in (False, True): for text in ("+" * 100_000, "a@" + "a" * 100_000, "a@" + "." * 100_000, "a@" * 20_000): assert list(iter_email_addresses(text, ascii_only=ascii_only)) == [] ''', r''' from src.agent_loop import _mentions_pdf_url, _mentions_workspace_script from src.clean_agent_preview import declared_workspace_artifacts from src.text_scanning import contains_search_engine_navigation from src.turn_contract import _mentions_workspace_artifact, _mentions_workspace_output workspace = "/workspace/" * 30_000 + "x" assert not _mentions_workspace_script(workspace) assert not _mentions_workspace_artifact(workspace) assert not _mentions_workspace_output(workspace) assert declared_workspace_artifacts("create " + workspace) == () assert not _mentions_pdf_url("http://" * 30_000 + "x") assert not contains_search_engine_navigation("http://google." + "..google." * 30_000 + "x") ''', r''' from src.agent_loop import _looks_like_agent_reasoning_preamble, _looks_like_ody_qwen_leaked_tool_text from src.text_scanning import contains_detailed_sequence_request for text in ("\n" * 100_000 + "x", ("when " * 20_000) + "x"): _looks_like_agent_reasoning_preamble(text) _looks_like_ody_qwen_leaked_tool_text(text) contains_detailed_sequence_request(text) ''', r''' from src.agent_loop import (_email_account_label, _email_sender_name, _parse_explicit_open_panel_request, _read_only_shell_command, _split_before_assistant_prompt, _split_note_items, _strip_horizontal_space_before_lf) space = " " * 100_000 _parse_explicit_open_panel_request("open calendar" + space + "x") _split_note_items(space + "x") _email_sender_name("<" * 100_000 + "x") _email_account_label("<" * 100_000 + "x") _split_before_assistant_prompt("\n" * 100_000 + "x") _strip_horizontal_space_before_lf(" " * 100_000 + "x") _read_only_shell_command(space + "x") ''', r''' from src.agent_loop import _numbered_row_parenthesized_ids from src.text_scanning import iter_angle_contents, iter_markdown_links, replace_markdown_links_with_labels for text in ("<" * 100_000, "[" * 100_000, ("[x](#session-" * 20_000) + "missing"): list(iter_angle_contents(text)) list(iter_markdown_links(text, target_prefix="#session-")) replace_markdown_links_with_labels(text) _numbered_row_parenthesized_ids("1. x " + "(" * 100_000 + "x") ''', r''' from src.agent_loop import (_contextual_summary_fragment, _is_terse_link_request, _looks_like_destructive_request, _looks_like_youtube_tool_turn, _private_browser_product_query) from src.text_scanning import first_tag_content for text in (("official " * 30_000) + "x", ("mark " * 30_000) + "x", " " * 100_000 + "x", ("summary: x " * 20_000) + "x", "find best " + "?" * 100_000 + "x", "