From eeff41a9ef8c75c7ed0e52fbc1ae309c997c6dc6 Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Tue, 6 Oct 2026 03:16:53 +0100 Subject: [PATCH] fix(security): eliminate parser denial-of-service paths --- src/agent_loop.py | 922 +++++++++++++++++++++------ src/clean_agent_preview.py | 343 ++++++++-- src/text_scanning.py | 210 ++++++ src/tool_parsing.py | 137 +++- src/turn_contract.py | 312 +++++++-- tests/test_clean_agent_preview.py | 4 +- tests/test_redos_core_parsers.py | 859 +++++++++++++++++++++++++ tests/test_stream_error_redaction.py | 224 +++++++ 8 files changed, 2676 insertions(+), 335 deletions(-) create mode 100644 src/text_scanning.py create mode 100644 tests/test_redos_core_parsers.py create mode 100644 tests/test_stream_error_redaction.py diff --git a/src/agent_loop.py b/src/agent_loop.py index 2cbae780a..8a70fd3ad 100644 --- a/src/agent_loop.py +++ b/src/agent_loop.py @@ -82,6 +82,15 @@ from src.tool_approvals import ( tool_approval_store, ) from src.tool_types import ToolBlock +from src.tool_parsing import iter_email_addresses, strip_angle_tags +from src.text_scanning import ( + contains_detailed_sequence_request, + has_prefixed_token_match, + first_tag_content, + iter_angle_contents, + iter_markdown_links, + replace_markdown_links_with_labels, +) from src.turn_contract import selected_tools_for_request, with_turn_contract from src.agent_runtime.journal import propose_action, execute_action from src.agent_runtime.completion import with_completion_gate @@ -1016,12 +1025,36 @@ def _looks_like_map_browser_request(text: str) -> bool: def _looks_like_youtube_tool_turn(text: str) -> bool: value = str(text or "").lower() - return bool(re.search( + if re.search( r"\b(?:youtube|youtu\.be|yt|video\s+comments?|comments?\s+on\s+(?:the\s+)?video|" - r"transcript\s+(?:of|for)|(?:latest|newest|recent)\s+(?:\d+\s+)?(?:videos?|uploads?)|" - r"official\s+.+\s+channel)\b", + r"transcript\s+(?:of|for)|(?:latest|newest|recent)\s+(?:\d+\s+)?(?:videos?|uploads?)" + r")\b", value, - )) + ): + return True + official_ends = iter(match.end() for match in re.finditer(r"\bofficial", value)) + channel_starts = iter(match.start() for match in re.finditer(r"channel\b", value)) + next_official_end = next(official_ends, -1) + next_channel_start = next(channel_starts, -1) + leading_space = False + body = False + trailing_space = False + for index, char in enumerate(value): + while next_channel_start >= 0 and next_channel_start < index: + next_channel_start = next(channel_starts, -1) + if trailing_space and next_channel_start == index: + return True + starts_after_official = next_official_end == index + if starts_after_official: + next_official_end = next(official_ends, -1) + is_space = char.isspace() + is_not_lf = char != "\n" + leading_space, body, trailing_space = ( + (starts_after_official or leading_space) and is_space, + (leading_space or body) and is_not_lf, + (body or trailing_space) and is_space, + ) + return False def _explicitly_named_personal_tools(text: str) -> Set[str]: @@ -1108,7 +1141,7 @@ def _qwen38_router_tool_names(query: str) -> Set[str]: selected.update({"mcp__email__manage_email_state"}) if re.search(r"\b(?:latest|newest|recent|most recent|inbox)\b", q): selected.add("mcp__email__list_emails") - if re.search(r"\b(?:send|email)\b", q) and re.search(r"[\w.+-]+@[\w.-]+\.\w+", q): + if re.search(r"\b(?:send|email)\b", q) and next(iter_email_addresses(q), None): selected.update({"send_email", "resolve_contact"}) if re.search(r"\b(?:reply|respond|response)\b", q): selected.update({"mcp__email__list_emails", "mcp__email__read_email", "mcp__email__draft_email_reply"}) @@ -1524,8 +1557,20 @@ def _parse_explicit_open_panel_request(text: str) -> Optional[tuple[str, str]]: value = re.sub(r"^(?:(?:ok(?:ay)?|now|then|also|next)[,;:]?\s+)+", "", value) value = re.sub(r"^go\s+back\s+(?:and\s+)?(?:open|to)\s+", "open ", value) value = re.sub(r"^return\s+to\s+", "open ", value) - value = re.sub(r"\s+again[.!?]*$", "", value) - value = re.sub(r"[.!?]+$", "", value).strip() + trailing_punctuation = len(value) + while trailing_punctuation and value[trailing_punctuation - 1] in ".!?": + trailing_punctuation -= 1 + without_punctuation = value[:trailing_punctuation] + if without_punctuation.endswith("again"): + whitespace_start = len(without_punctuation) - len("again") + while whitespace_start and without_punctuation[whitespace_start - 1].isspace(): + whitespace_start -= 1 + if whitespace_start < len(without_punctuation) - len("again"): + value = value[:whitespace_start] + trailing_punctuation = len(value) + while trailing_punctuation and value[trailing_punctuation - 1] in ".!?": + trailing_punctuation -= 1 + value = value[:trailing_punctuation].strip() month_names = { "january": "01", "jan": "01", "february": "02", "feb": "02", @@ -2104,26 +2149,34 @@ def _looks_like_agent_reasoning_preamble(text: str) -> bool: # e.g. a full comparison followed by "Now let me verify nothing started." # That whole round is progress text; retaining it duplicates the answer # once the tool-result round supplies the actual verification. - if re.search( - r"(?:^|\n+|[.!?]\s+)(?:but\s+)?(?:now\s+)?" + trailing_segment = re.split(r"(?:\n+|[.!?]\s+)", lowered)[-1] + if re.match( + r"(?:but\s+)?(?:now\s+)?" r"(?:let me|i(?:'ll| will)(?:\s+need\s+to)?|i\s+can(?:\s+now)?|i(?:'m| am)\s+(?:preparing|planning)\s+to)\s+" r"(?:(?:carefully|methodically|systematically|closely|further)\s+){0,2}" r"(?:continue|continuing|analy[sz]e|scan|check|verify|inspect|confirm|look up|fetch|open|list|search|refine|request|review|track|read|watch|(?:re-?)?examine|provide|give|state|report|answer|respond|summarize|conclude)\b" r"[^.!?]*[.!?]?\s*$", - lowered, + trailing_segment, ): return True # A process heading followed only by partial observation bullets is still # analysis, not a delivered answer. Keep this bounded to inspection verbs # so completed answer headings such as "Let me summarize:" remain valid. - if re.search( - r"(?:^|\n)\s*(?:now\s+)?(?:let me|i(?:'ll| will))\s+" + process_heading_re = re.compile( + r"\s*(?:now\s+)?(?:let me|i(?:'ll| will))\s+" r"(?:carefully\s+|methodically\s+|systematically\s+){0,2}" r"(?:track|trace|inspect|review|analy[sz]e|examine|check)\b[^\n]{0,140}:\s*\n" - r"(?:\s*[-*]\s+[^\n]{1,240}\n?){1,8}\s*$", - lowered, - ): - return True + r"(?:\s*[-*]\s+[^\n]{1,240}\n?){1,8}\s*$" + ) + if any(marker in lowered for marker in ("let me", "i'll", "i will")): + line_start = 0 + while line_start <= len(lowered): + if process_heading_re.match(lowered, line_start): + return True + newline = lowered.find("\n", line_start) + if newline < 0: + break + line_start = newline + 1 first_line = lowered.splitlines()[0].strip() if re.search( r"\b(?:i need to|i should|let me|i'll(?:\s+need\s+to)?|i(?:'m| am)\s+(?:preparing|planning)\s+to)\s+" @@ -2144,11 +2197,28 @@ def _looks_like_agent_reasoning_preamble(text: str) -> bool: lowered, ): return True - return bool(re.search( - r"(?:^|[。!?\n]\s*)(?:我需要|需要先|让我|先|接下来(?:我)?(?:会|要)?).{0,12}" - r"(?:查看|检查|读取|分析|继续|使用|调用)", - value, - )) + chinese_body = re.compile( + r"(?:我需要|需要先|让我|先|接下来(?:我)?(?:会|要)?).{0,12}" + r"(?:查看|检查|读取|分析|继续|使用|调用)" + ) + if chinese_body.match(value): + return True + pos = 0 + boundary_pattern = re.compile(r"[。!?\n]") + while boundary := boundary_pattern.search(value, pos): + candidate = boundary.end() + while candidate < len(value) and value[candidate].isspace(): + candidate += 1 + if chinese_body.match(value, candidate): + return True + pos = max(candidate, boundary.end()) + return False + + +def _strip_trailing_done(text: str) -> str: + """Remove a terminal Done. and adjacent whitespace with a linear scan.""" + trimmed = text.rstrip() + return trimmed[:-5].rstrip() if trimmed.lower().endswith("done.") else text def _strip_trailing_answer_promise(text: str) -> str: @@ -2405,6 +2475,21 @@ def _parse_qwen_explicit_note_delete(text: str) -> Optional[str]: return match.group(1).strip().strip("\"'`").rstrip(".") if match else None +def _captures_after_first_prefix( + value: str, + prefix_pattern: str, + remainder_pattern: str, + *, + flags: int = re.IGNORECASE, +) -> tuple[str | None, ...] | None: + """Match a suffix once after the first prefix that can own all later text.""" + prefix = re.search(prefix_pattern, value, flags) + if prefix is None: + return None + remainder = re.match(remainder_pattern, value[prefix.end():], flags) + return remainder.groups() if remainder is not None else None + + def _parse_qwen_explicit_note_update(text: str) -> Optional[tuple[str, str]]: """Extract an exact note title and replacement content from a clear update.""" value = str(text or "").strip() @@ -2412,15 +2497,14 @@ def _parse_qwen_explicit_note_update(text: str) -> Optional[tuple[str, str]]: r"\bnote\b", value, re.IGNORECASE ): return None - match = re.search( - r"\bnote\s+titled\s+(.+?)\s+" - r"so\s+its\s+content\s+is\s+['\"]([^'\"]+)['\"]", + captures = _captures_after_first_prefix( value, - re.IGNORECASE, + r"\bnote\s+titled(?=\s)", + r"\s+(.+?)\s+so\s+its\s+content\s+is\s+['\"]([^'\"]+)['\"]", ) - if not match: + if captures is None: return None - return match.group(1).strip().strip("\"'`").rstrip("."), match.group(2) + return captures[0].strip().strip("\"'`").rstrip("."), captures[1] def _parse_qwen_explicit_note_search(text: str) -> Optional[str]: @@ -2515,19 +2599,30 @@ def _is_qwen_explicit_endpoint_list_request(text: str) -> bool: )) -def _parse_explicit_pipeline_request(text: str) -> Optional[tuple[str, str]]: - value = str(text or "").strip() - match = re.search( - r"\bpipeline\s+using\s+([^\s,]+)\s+to\s+(.+?),\s*then\s+" - r"([^\s,]+)\s+to\s+(.+?)(?:[.!?]\s*)?$", - value, +def _pipeline_request_parts(value: str) -> tuple[str, str, str, str] | None: + prefix = re.search( + r"\bpipeline\s+using\s+([^\s,]+)\s+to(?=\s)", value, re.IGNORECASE + ) + if prefix is None: + return None + remainder = re.match( + r"\s+(.+?),\s*then\s+([^\s,]+)\s+to\s+(.+?)(?:[.!?]\s*)?$", + value[prefix.end():], re.IGNORECASE, ) - if not match: + if remainder is None: + return None + return prefix.group(1), remainder.group(1), remainder.group(2), remainder.group(3) + + +def _parse_explicit_pipeline_request(text: str) -> Optional[tuple[str, str]]: + value = str(text or "").strip() + parts = _pipeline_request_parts(value) + if parts is None: return None steps = [ - {"model": match.group(1).strip(), "instruction": match.group(2).strip()}, - {"model": match.group(3).strip(), "instruction": match.group(4).strip()}, + {"model": parts[0].strip(), "instruction": parts[1].strip()}, + {"model": parts[2].strip(), "instruction": parts[3].strip()}, ] return "pipeline", json.dumps({"steps": steps}) @@ -2704,7 +2799,9 @@ def _recent_session_id_for_title(messages: List[Dict], title_query: str) -> str: if msg.get("role") != "assistant": continue content = str(msg.get("content") or "") - for label, sid in reversed(re.findall(r"\[([^\]]+)\]\(#session-([^)]+)\)", content)): + for _start, _end, label, sid in reversed(list(iter_markdown_links( + content, target_prefix="#session-" + ))): label_l = label.lower() if not query_terms or all(term in label_l for term in query_terms): return sid.strip() @@ -2789,7 +2886,7 @@ def _parse_qwen_explicit_session_action(text: str, messages: List[Dict]) -> Opti if rename_match: new_name = (rename_match.group(1) or "").strip(" \t\r\n\"'`.") new_name = re.split( - r"\s+(?:Use the tool directly|Keep this read-only|Report the result|Do not)\b", + r"(?<=\s)(?:Use the tool directly|Keep this read-only|Report the result|Do not)\b", new_name, maxsplit=1, flags=re.IGNORECASE, @@ -2877,10 +2974,14 @@ def _parse_qwen_explicit_session_send(text: str, messages: List[Dict]) -> Option sid = _recent_session_id_for_title(messages, target_text) if not sid and re.search(r"\b(?:that|this)\b", target_text, re.IGNORECASE): for prior in reversed(messages or []): - links = re.findall( - r"\[[^\]]+\]\(#session-([A-Za-z0-9_-]+)\)", - str(prior.get("content") or ""), - ) + links = [ + target + for _start, _end, _label, target in iter_markdown_links( + str(prior.get("content") or ""), + target_prefix="#session-", + target_re=re.compile(r"[A-Za-z0-9_-]+"), + ) + ] if links: sid = links[-1] break @@ -2889,6 +2990,19 @@ def _parse_qwen_explicit_session_send(text: str, messages: List[Dict]) -> Option return "send_to_session", f"{sid}\n{relay_message}" +def _session_find_query_capture(value: str) -> str | None: + action_prefix = re.search(r"\b(?:find|search|show)\b(?=\s)", value, re.IGNORECASE) + if action_prefix is None: + return None + optional_for = r"(?:\s+for)?" if action_prefix.group(0).casefold() == "search" else "" + remainder = re.match( + optional_for + r"\s+(?:the\s+)?(.+?)\s+(?:chat|session|conversation)\b", + value[action_prefix.end():], + re.IGNORECASE, + ) + return remainder.group(1) if remainder is not None else None + + def _parse_qwen_explicit_session_find(text: str) -> Optional[tuple[str, str]]: """Map obvious chat lookup requests to list_sessions with a filter.""" value = str(text or "").strip() @@ -2910,14 +3024,10 @@ def _parse_qwen_explicit_session_find(text: str) -> Optional[tuple[str, str]]: re.IGNORECASE, ): return "list_sessions", "" - match = re.search( - r"\b(?:find|search(?:\s+for)?|show)\s+(?:the\s+)?(.+?)\s+(?:chat|session|conversation)\b", - value, - re.IGNORECASE, - ) - if not match: + captured_query = _session_find_query_capture(value) + if captured_query is None: return None - query = (match.group(1) or "").strip(" \t\r\n\"'`.") + query = captured_query.strip(" \t\r\n\"'`.") query = re.sub(r"\bscratch\b", "", query, flags=re.IGNORECASE).strip() query = re.sub(r"\s+", " ", query).strip() return "list_sessions", query or value @@ -3180,15 +3290,15 @@ def _parse_qwen_explicit_email_topic_bulk_action_request(text: str) -> Optional[ return None if re.search(r"\bUIDs?\b", value, re.IGNORECASE): return None - match = re.search( - r"\b(?:delete|trash|remove|archive|mark(?:\s+as)?\s+(?:read|unread)|mark\s+(?:read|unread))\b" - r"\s+(?:all|every|the)?\s*(?:my\s+)?(.+?)\s+(?:emails?|mail|messages?)\b", + captures = _captures_after_first_prefix( value, - re.IGNORECASE, + r"\b(?:delete|trash|remove|archive|mark(?:\s+as)?\s+(?:read|unread)|mark\s+(?:read|unread))\b" + r"(?=\s)", + r"\s+(?:all|every|the)?\s*(?:my\s+)?(.+?)\s+(?:emails?|mail|messages?)\b", ) - if not match: + if captures is None: return None - query = re.sub(r"\s+", " ", match.group(1)).strip(" .\"'") + query = re.sub(r"\s+", " ", captures[0]).strip(" .\"'") if not query or query.lower() in {"all", "the", "my"}: return None return {"action": action, "query": query, "folder": "INBOX", "max_results": 50} @@ -3345,12 +3455,12 @@ def _parse_qwen_explicit_block_sender_request(text: str) -> Optional[dict[str, A return None if re.search(r"\b(?:should\s+i|should\s+we|would\s+you|can\s+i|do\s+you\s+think)\b", value, re.IGNORECASE): return None - match = re.search(r"[\w.+-]+@[\w.-]+\.\w+", value) - if not match: + address = next(iter_email_addresses(value), "") + if not address: return None move_existing = not re.search(r"\b(?:do\s+not|don't|dont)\s+(?:move|delete|trash|junk)\b|\bleave\s+existing\b", value, re.IGNORECASE) return { - "sender": match.group(0), + "sender": address, "folder": "INBOX", "move_existing": move_existing, "reason": "User explicitly requested sender block.", @@ -3392,10 +3502,10 @@ def _parse_qwen_explicit_unblock_sender_request(text: str) -> Optional[dict[str, value = str(text or "").strip() if not value or not re.search(r"\b(?:unblock|allow|remove\s+from\s+block)\b", value, re.IGNORECASE): return None - match = re.search(r"[\w.+-]+@[\w.-]+\.\w+", value) - if not match: + address = next(iter_email_addresses(value), "") + if not address: return None - return {"sender": match.group(0)} + return {"sender": address} def _email_relative_date_range(text: str) -> Optional[dict[str, str]]: @@ -4253,8 +4363,8 @@ def _note_title_id_pairs_from_tool_output(raw: str) -> list[tuple[str, str]]: pairs.append(key) seen.add(key) - for match in re.finditer(r"\[([^\]]+)\]\(#note-([^)]+)\)", raw): - add_pair(match.group(1), match.group(2)) + for _start, _end, title, note_id in iter_markdown_links(raw, target_prefix="#note-"): + add_pair(title, note_id) for line in raw.splitlines(): match = re.match(r"^\s*-\s+\[([^\]]+)\]\s+\*\*(.*?)\*\*", line) if match: @@ -4326,7 +4436,7 @@ def _notes_expected_actions(user_text: str) -> set[str]: def _split_note_items(value: str) -> list[dict[str, Any]]: parts = [ re.sub(r"\s+", " ", part).strip(" .") - for part in re.split(r"\s*,\s*|\s+\band\b\s+", str(value or "")) + for part in re.split(r",|(?<=\s)and(?=\s)", str(value or "")) ] return [{"text": part, "done": False} for part in parts if part] @@ -4391,6 +4501,32 @@ def _is_personal_tool_definition_turn(text: str) -> bool: ) +def _remaining_checklist_name(value: str) -> str | None: + """Return the checklist name from the staged legacy question grammar.""" + lead = re.search(r"\b(?:what(?:'s| is)?|show|tell\s+me)\b", value, re.IGNORECASE) + if lead is None: + return None + line_end = value.find("\n", lead.end()) + if line_end < 0: + line_end = len(value) + remaining = re.compile(r"\b(?:left|remaining)\b", re.IGNORECASE).search( + value, lead.end(), line_end + ) + if remaining is None: + return None + location = re.compile(r"\b(?:on|in)\b(?=\s)", re.IGNORECASE).search( + value, remaining.end(), line_end + ) + if location is None: + return None + name = re.match( + r"\s+(?:the\s+)?(.+?)\s+checklist\b", + value[location.end():line_end], + re.IGNORECASE, + ) + return name.group(1) if name is not None else None + + def _parse_simple_notes_tool_request(text: str) -> Optional[tuple[str, str]]: """Deterministic fallback for obvious notes commands when a model stalls.""" value = str(text or "").strip() @@ -4425,14 +4561,14 @@ def _parse_simple_notes_tool_request(text: str) -> Optional[tuple[str, str]]: else: label = "" - checklist_match = re.search( - r"\b(?:make|create|add)\s+(?:a\s+)?checklist\s+(?:called|titled|named)\s+(.+?)\s+with\s+(.+?)\s*$", + checklist_captures = _captures_after_first_prefix( value, - re.IGNORECASE, + r"\b(?:make|create|add)\s+(?:a\s+)?checklist\s+(?:called|titled|named)(?=\s)", + r"\s+(.+?)\s+with\s+(.+?)$", ) - if checklist_match: - title = re.sub(r"\s+", " ", checklist_match.group(1)).strip(" .\"'") - items = _split_note_items(checklist_match.group(2)) + if checklist_captures is not None: + title = re.sub(r"\s+", " ", checklist_captures[0]).strip(" .\"'") + items = _split_note_items(checklist_captures[1]) if title and items: return "manage_notes", json.dumps({ "action": "add", @@ -4457,13 +4593,9 @@ def _parse_simple_notes_tool_request(text: str) -> Optional[tuple[str, str]]: args["label"] = label return "manage_notes", json.dumps(args) - remaining_match = re.search( - r"\b(?:what(?:'s| is)?|show|tell\s+me)\b.*?\b(?:left|remaining)\b.*?\b(?:on|in)\s+(?:the\s+)?(.+?)\s+checklist\b", - value, - re.IGNORECASE, - ) - if remaining_match: - query = _clean_notes_search_query(remaining_match.group(1)) + remaining_name = _remaining_checklist_name(value) + if remaining_name is not None: + query = _clean_notes_search_query(remaining_name) if query: return "manage_notes", json.dumps({"action": "search", "query": query}) @@ -4474,7 +4606,7 @@ def _parse_simple_notes_tool_request(text: str) -> Optional[tuple[str, str]]: ) if note_saying_match: body = re.sub( - r"\s+(?:and\s+)?(?:tag|label)\s+(?:it\s+)?(?:as\s+)?#?[a-zA-Z0-9_-]{2,40}\s*$", + r"(?<=\s)(?:and\s+)?(?:tag|label)\s+(?:it\s+)?(?:as\s+)?#?[a-zA-Z0-9_-]{2,40}\s*$", "", note_saying_match.group(1), flags=re.IGNORECASE, @@ -4741,6 +4873,38 @@ def _single_document_id_from_tool_output(raw: str) -> str: return next(iter(ids)) if len(ids) == 1 else "" +def _session_link_from_row(row: str) -> str: + """Find the first session link in a newline-free listing row in O(n). + + The legacy label grammar allows both an ordinary backslash and an escaped + closing bracket. Keep its greedy choice of the last reachable link closer, + without trying exponentially many ways to partition a backslash run. + An unescaped closing bracket ends a label; later openers can then be tried. + The id's closing-parenthesis search also advances monotonically. + """ + start = row.find("[") + bracket = row.find("]", start + 1) if start >= 0 else -1 + link_end = -1 + id_end = -1 + while bracket >= 0: + if bracket > start + 1 and row.startswith("(#session-", bracket + 1): + id_start = bracket + len("](#session-") + if id_end < id_start: + id_end = row.find(")", id_start) + if id_end < 0: + break + if id_end > id_start: + link_end = id_end + 1 + if row[bracket - 1] != "\\": + if link_end >= 0: + break + start = row.find("[", bracket + 1) + if start < 0: + break + bracket = row.find("]", max(bracket + 1, start + 1)) + return row[start:link_end] if link_end >= 0 else "" + + def _session_list_summary_from_tool_output(raw: str, max_items: int = 12) -> str: """Keep a broad session listing readable and terminal for small routers.""" if not isinstance(raw, str) or not raw.strip(): @@ -4758,13 +4922,17 @@ def _session_list_summary_from_tool_output(raw: str, max_items: int = 12) -> str return "\n".join(lines[: max_items + 1]) formatted_rows: list[str] = [] for row in rows: - link_match = re.search(r"(\[(?:\\.|[^\]])+\]\(#session-[^)]+\))", row) - if link_match: - meta_match = re.search(r"\(([^()]*(?:last active|msgs|model|id:)[^()]*)\)", row) - meta = meta_match.group(1) if meta_match else "" + link = _session_link_from_row(row) + if link: + meta = next( + (match.group(1) for match in re.finditer(r"\(([^()]*)\)", row) + if any(marker in match.group(1) + for marker in ("last active", "msgs", "model", "id:"))), + "", + ) active = re.search(r"last active [^)]+", meta) suffix = f" ({active.group(0)})" if active else "" - formatted_rows.append(f"- {link_match.group(1)}{suffix}") + formatted_rows.append(f"- {link}{suffix}") else: formatted_rows.append(row[:180].rstrip() + ("..." if len(row) > 180 else "")) shown = formatted_rows[:max_items] @@ -4793,6 +4961,28 @@ def _registry_list_summary_from_tool_output(raw: str, max_items: int = 12) -> st return summary if len(summary) <= 3200 else summary[:3197].rstrip() + "..." +def _research_listing_row(line: str) -> tuple[str, str, str] | None: + """Parse the legacy research markdown row without retrying label openers.""" + if not line.startswith("-"): + return None + cursor = 1 + if cursor >= len(line) or not line[cursor].isspace(): + return None + while cursor < len(line) and line[cursor].isspace(): + cursor += 1 + if cursor >= len(line) or line[cursor] != "[": + return None + title_start = cursor + 1 + marker = line.find("](#research-", title_start) + while marker >= 0: + identifier_start = marker + len("](#research-") + close = line.find(")", identifier_start) + if close > identifier_start: + return line[title_start:marker], line[identifier_start:close], line[close + 1:] + marker = line.find("](#research-", marker + 1) + return None + + def _research_list_summary_from_tool_output(raw: str, max_items: int = 6) -> str: """Keep saved research listings concise while preserving report anchors.""" if not isinstance(raw, str) or not raw.strip(): @@ -4805,14 +4995,15 @@ def _research_list_summary_from_tool_output(raw: str, max_items: int = 6) -> str return lines[0] rows: list[str] = [] for line in lines[1:]: - match = re.match(r"^-\s+\[(.*?)\]\(#research-([^)]+)\)(.*)$", line) - if not match: + parsed_row = _research_listing_row(line) + if parsed_row is None: continue - title = re.sub(r"\s+", " ", match.group(1)).strip() + raw_title, research_id, raw_suffix = parsed_row + title = re.sub(r"\s+", " ", raw_title).strip() if len(title) > 110: title = title[:107].rstrip() + "..." - suffix = re.sub(r"\s+", " ", match.group(3) or "").strip() - rows.append(f"- [{title}](#research-{match.group(2)}) {suffix}".rstrip()) + suffix = re.sub(r"\s+", " ", raw_suffix).strip() + rows.append(f"- [{title}](#research-{research_id}) {suffix}".rstrip()) if len(rows) >= max_items: break if not rows: @@ -4824,6 +5015,54 @@ def _research_list_summary_from_tool_output(raw: str, max_items: int = 6) -> str return "\n".join([lines[0], *rows]) +def _skill_listing_row(line: str) -> tuple[str, str | None, str | None, str | None] | None: + """Parse the legacy bold skill row with monotonic closer searches.""" + if not line.startswith("-"): + return None + cursor = 1 + if cursor >= len(line) or not line[cursor].isspace(): + return None + while cursor < len(line) and line[cursor].isspace(): + cursor += 1 + if not line.startswith("**", cursor): + return None + name_start = cursor + 2 + closer = line.find("**", name_start) + while closer >= 0: + rest = closer + 2 + if rest == len(line): + return line[name_start:closer], None, None, None + if line[rest] == ":": + description = line[rest + 1:].lstrip() + return line[name_start:closer], None, None, description + if line[rest].isspace(): + meta_start = rest + while meta_start < len(line) and line[meta_start].isspace(): + meta_start += 1 + if meta_start < len(line) and line[meta_start] == "(": + meta_end = line.find(")", meta_start + 1) + while meta_end >= 0: + tail = meta_end + 1 + if tail == len(line): + return line[name_start:closer], line[meta_start + 1:meta_end], None, None + if line[tail] == ":": + return ( + line[name_start:closer], + line[meta_start + 1:meta_end], + None, + line[tail + 1:].lstrip(), + ) + meta_end = line.find(")", meta_end + 1) + elif line.startswith("[draft]", meta_start): + tail = meta_start + len("[draft]") + if tail == len(line): + return line[name_start:closer], None, "draft", None + if tail < len(line) and line[tail] == ":": + return line[name_start:closer], None, "draft", line[tail + 1:].lstrip() + closer = line.find("**", closer + 1) + return None + + def _skills_list_summary_from_tool_output(raw: str, max_items: int = 8) -> str: """Keep the skill index visible without dumping the full registry.""" if not isinstance(raw, str) or not raw.strip(): @@ -4845,10 +5084,11 @@ def _skills_list_summary_from_tool_output(raw: str, max_items: int = 8) -> str: label = section or "Skills" if label in totals: totals[label] += 1 - match = re.match(r"^-\s+\*\*(.*?)\*\*(?:\s+\((.*?)\)|\s+\[(draft)\])?(?::\s*(.*))?$", line) - if match: - name = re.sub(r"\s+", " ", match.group(1)).strip() - meta = re.sub(r"\s+", " ", (match.group(2) or match.group(3) or label).strip()) + parsed_row = _skill_listing_row(line) + if parsed_row is not None: + raw_name, parenthesized_meta, draft_meta, _description = parsed_row + name = re.sub(r"\s+", " ", raw_name).strip() + meta = re.sub(r"\s+", " ", (parenthesized_meta or draft_meta or label).strip()) rows.append((label, f"- [{name}](#skill-{quote(name, safe='')}) ({meta})")) else: rows.append((label, line[:96].rstrip() + ("..." if len(line) > 96 else ""))) @@ -4898,6 +5138,45 @@ def _calendar_detail_requested(text: str) -> bool: ) +def _calendar_listing_row(line: str) -> tuple[str, str, str, str] | None: + """Parse a calendar row while advancing through delimiters only once.""" + cursor = 0 + while cursor < len(line) and line[cursor].isspace(): + cursor += 1 + if cursor >= len(line) or line[cursor] != "-": + return None + cursor += 1 + if cursor >= len(line) or not line[cursor].isspace(): + return None + while cursor < len(line) and line[cursor].isspace(): + cursor += 1 + when_start = cursor + colon = line.find(":", when_start + 1) + marker_from = when_start + while colon >= 0: + spacing = colon + 1 + if spacing < len(line) and line[spacing].isspace(): + while spacing < len(line) and line[spacing].isspace(): + spacing += 1 + if spacing < len(line) and line[spacing] == "[": + title_start = spacing + 1 + marker = line.find("](#event-", max(title_start, marker_from)) + while marker >= 0: + identifier_start = marker + len("](#event-") + close = line.find(")", identifier_start) + if close > identifier_start: + return ( + line[when_start:colon], + line[title_start:marker], + line[identifier_start:close], + line[close + 1:], + ) + marker = line.find("](#event-", marker + 1) + marker_from = len(line) + colon = line.find(":", colon + 1) + return None + + def _calendar_list_summary_from_tool_output( raw: str, max_items: int = 20, @@ -4951,17 +5230,18 @@ def _calendar_list_summary_from_tool_output( items: list[str] = [] current_item_idx = -1 for line in text.splitlines(): - m = re.match(r"^\s*-\s+(.+?):\s+\[(.*?)\]\(#event-([^)]+)\)(.*)$", line) - if not m: + parsed_row = _calendar_listing_row(line) + if parsed_row is None: if include_details and current_item_idx >= 0: detail = re.sub(r"\s+", " ", line).strip() if detail and not detail.startswith("-"): items[current_item_idx] = f"{items[current_item_idx]} — {detail}" continue - when = re.sub(r"\s+", " ", m.group(1)).strip() - title = re.sub(r"\s+", " ", m.group(2)).strip() - event_id = m.group(3).strip() - suffix = re.sub(r"\s+", " ", m.group(4) or "").strip() + raw_when, raw_title, raw_event_id, raw_suffix = parsed_row + when = re.sub(r"\s+", " ", raw_when).strip() + title = re.sub(r"\s+", " ", raw_title).strip() + event_id = raw_event_id.strip() + suffix = re.sub(r"\s+", " ", raw_suffix).strip() label = f"[{title}](#event-{event_id}) — {format_when(when)}" if suffix: label += f" {suffix}" @@ -5278,18 +5558,18 @@ def _parse_simple_calendar_tool_request( args["query"] = "travel" return "manage_calendar", json.dumps(args, ensure_ascii=False) - tag_match = re.search( - r"\b(?:change|update|set|retag)\b\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b", + tag_captures = _captures_after_first_prefix( value, - re.IGNORECASE, + r"\b(?:change|update|set|retag)\b(?=\s)", + r"\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b", ) - if tag_match and re.search(r"\b(?:calendar|event|trip|meeting|appointment)\b", q): - title = re.sub(r"\s+", " ", tag_match.group(1)).strip(" .") + if tag_captures is not None and re.search(r"\b(?:calendar|event|trip|meeting|appointment)\b", q): + title = re.sub(r"\s+", " ", tag_captures[0]).strip(" .") if title: return "manage_calendar", json.dumps({ "action": "update_event", "summary": title, - "tag": tag_match.group(2).lower(), + "tag": tag_captures[1].lower(), }, ensure_ascii=False) return None @@ -5873,8 +6153,8 @@ def _calendar_title_uid_pairs_from_tool_event(event: dict[str, Any]) -> list[tup add_pair(row.get("summary") or row.get("title"), row.get("uid") or row.get("id")) raw = str(event.get("output") or "") - for match in re.finditer(r"\[([^\]]+)\]\(#event-([^)]+)\)", raw): - add_pair(match.group(1), match.group(2)) + for _start, _end, title, event_id in iter_markdown_links(raw, target_prefix="#event-"): + add_pair(title, event_id) return pairs @@ -5978,12 +6258,34 @@ def _friendly_email_date(value: str) -> str: return text +def _strip_terminal_delimited_value( + text: str, + opener: str, + closer: str, + *, + required: str = "", + allow_empty: bool = False, +) -> str: + """Remove one flat terminal ``opener...closer`` region in O(n).""" + trimmed = text.rstrip() + if not trimmed.endswith(closer): + return text + previous_closer = trimmed.rfind(closer, 0, len(trimmed) - len(closer)) + opener_at = trimmed.find(opener, previous_closer + len(closer)) + if opener_at < 0: + return text + inner = trimmed[opener_at + len(opener):-len(closer)] + if (not allow_empty and not inner) or (required and required not in inner): + return text + return trimmed[:opener_at].rstrip() + + def _email_sender_name(value: str) -> str: text = re.sub(r"\s+", " ", str(value or "")).strip() if not text: return "" - text = re.sub(r"\s*\([^)]*@[^)]*\)\s*$", "", text).strip() - text = re.sub(r"\s*<[^>]*>\s*$", "", text).strip() + text = _strip_terminal_delimited_value(text, "(", ")", required="@").strip() + text = _strip_terminal_delimited_value(text, "<", ">", allow_empty=True).strip() return text or str(value or "").strip() @@ -5991,7 +6293,7 @@ def _email_account_label(value: str) -> str: text = re.sub(r"\s+", " ", str(value or "")).strip() if not text: return "" - return re.sub(r"\s*<[^>]+>\s*$", "", text).strip() or text + return _strip_terminal_delimited_value(text, "<", ">").strip() or text def _format_email_attachment_summary_item(item: dict[str, str]) -> str: @@ -6157,6 +6459,21 @@ def _email_read_summaries_from_tool_events(tool_events: list[dict[str, Any]]) -> return summaries +def _strip_horizontal_space_before_lf(text: str) -> str: + """Remove spaces/tabs directly before LF without failed suffix retries.""" + out = [] + pos = 0 + while (newline := text.find("\n", pos)) >= 0: + trim_at = newline + while trim_at > pos and text[trim_at - 1] in " \t": + trim_at -= 1 + out.append(text[pos:trim_at]) + out.append("\n") + pos = newline + 1 + out.append(text[pos:]) + return "".join(out) + + def _email_read_evidence_from_tool_output(raw: str, *, max_body_chars: int = 6000) -> str: """Return bounded, plain-text evidence for a final email lookup synthesis.""" if not isinstance(raw, str) or not raw.strip(): @@ -6167,23 +6484,29 @@ def _email_read_evidence_from_tool_output(raw: str, *, max_body_chars: int = 600 # model round. text = re.sub(r"", "\n", text, flags=re.IGNORECASE) text = re.sub(r"", "\n", text, flags=re.IGNORECASE) - text = re.sub(r"<[^>]+>", "", text) + text = strip_angle_tags(text) text = html.unescape(text) - text = re.sub(r"[ \t]+\n", "\n", text) + text = _strip_horizontal_space_before_lf(text) text = re.sub(r"\n{3,}", "\n\n", text).strip() if len(text) > max_body_chars: text = text[:max_body_chars].rstrip() + "\n[...email truncated]" return text +def _is_terse_email_lookup_followup(text: str) -> bool: + value = str(text or "").strip() + value = value.rstrip("?.!").rstrip() + return bool(re.fullmatch( + r"(?:and|so|well|still|then|okay|ok|did you find it(?: yet)?|what did you find)", + value, + re.IGNORECASE, + )) + + def _email_lookup_request_from_messages(messages: list[dict], last_user: str) -> str: """Recover the substantive request behind terse follow-ups such as 'and?'.""" - terse = re.compile( - r"^\s*(?:and|so|well|still|then|okay|ok|did you find it(?: yet)?|what did you find)\s*[?.!]*\s*$", - re.IGNORECASE, - ) current = str(last_user or "").strip() - if current and not terse.match(current): + if current and not _is_terse_email_lookup_followup(current): return current for message in reversed(messages or []): if not isinstance(message, dict) or message.get("role") != "user": @@ -6192,7 +6515,7 @@ def _email_lookup_request_from_messages(messages: list[dict], last_user: str) -> if not isinstance(content, str): continue candidate = content.strip() - if candidate and not terse.match(candidate): + if candidate and not _is_terse_email_lookup_followup(candidate): return candidate return current @@ -6792,6 +7115,31 @@ _COMPACT_EMAIL_UNSUBSCRIBE_TOOLS = { "unsubscribe_email", } +_WORKSPACE_PREFIX_RE = re.compile(r"/workspace/", re.IGNORECASE) +_WORKSPACE_SCRIPT_RE = re.compile( + r"/workspace/[^\s`\"']+\.(?:py|pyw|sh|bash|js|mjs|ts|rb|pl)\b", + re.IGNORECASE, +) +_WORKSPACE_QUOTED_TOKEN_TAIL_RE = re.compile(r"[^\s`\"']*") +_HTTP_PREFIX_RE = re.compile(r"https?://", re.IGNORECASE) +_PDF_URL_RE = re.compile(r"https?://\S+(?:\.pdf\b|/pdf/)", re.IGNORECASE) +_NONSPACE_TOKEN_TAIL_RE = re.compile(r"\S*") + + +def _mentions_workspace_script(text: str) -> bool: + return has_prefixed_token_match( + str(text or ""), + _WORKSPACE_PREFIX_RE, + _WORKSPACE_SCRIPT_RE, + _WORKSPACE_QUOTED_TOKEN_TAIL_RE, + ) + + +def _mentions_pdf_url(text: str) -> bool: + return has_prefixed_token_match( + str(text or ""), _HTTP_PREFIX_RE, _PDF_URL_RE, _NONSPACE_TOKEN_TAIL_RE + ) + def _blocked_network_recovery_tools(events: Sequence[Dict[str, Any]]) -> Set[str]: """Preserve native recovery tools named by our own network guard. @@ -6873,11 +7221,7 @@ def _compact_native_route_tools( workspace_generator_request = bool( workspace_artifact_request and ( - re.search( - r"/workspace/[^\s`\"']+\.(?:py|pyw|sh|bash|js|mjs|ts|rb|pl)\b", - text, - re.IGNORECASE, - ) + _mentions_workspace_script(text) or ( # A multi-output data+visual deliverable is an execution # workflow even when its generator is discovered after the @@ -6895,7 +7239,7 @@ def _compact_native_route_tools( compact.update(WEB_TOOL_NAMES) if ( "pdf_extract" in original - and re.search(r"https?://\S+(?:\.pdf\b|/pdf/)", text, re.IGNORECASE) + and _mentions_pdf_url(text) ): compact.add("pdf_extract") if _looks_like_youtube_tool_turn(text): @@ -7220,7 +7564,7 @@ def _compact_native_artifact_tools( allowed.update(WEB_TOOL_NAMES) allowed.update({"web_fetch", "pdf_extract"}) if ( - re.search(r"/workspace/[^\s`\"']+\.(?:py|pyw|sh|bash|js|mjs|ts|rb|pl)\b", value, re.IGNORECASE) + _mentions_workspace_script(value) or ( len(artifacts) >= 2 and artifact_suffixes & {".csv", ".json", ".xlsx"} @@ -10095,12 +10439,12 @@ def _is_terse_link_request(text: str) -> bool: """True for short links/sources fragments that need context to be actionable.""" return bool( re.fullmatch( - r"\s*(?:(?:send|sned|share|give|show)?\s*(?:me\s+)?(?:the\s+)?" + r"(?:(?:send|sned|share|give|show)?\s*(?:me\s+)?(?:the\s+)?" r"(?:links?|urls?|sources?)" r"(?:\s+(?:for|to|from)\s+(?:those|that|them|these|it|this|the\s+(?:sites?|websites?|resources?|sources?)))?" r"|(?:for|to|from)\s+(?:those|that|them|these|it|this|the\s+(?:sites?|websites?|resources?|sources?)))" - r"\s*(?:please|pls)?[.!?]?\s*", - str(text or "").lower(), + r"\s*(?:please|pls)?[.!?]?", + str(text or "").lower().strip(), ) ) @@ -10712,27 +11056,29 @@ def _parse_inspection_file_replacement(text: str) -> Optional[dict[str, str]]: value, re.IGNORECASE, ) + action_values: tuple[str | None, str | None] | None = ( + (action_match.group("old"), action_match.group("new")) + if action_match is not None + else None + ) if action_match is None: # Locate the explicit mutation clause after stripping the inspection # language. Stop values before common trailing verification instructions. - action_match = re.search( - r"\b(?:chang(?:e|es|ed|ing)|updat(?:e|es|ed|ing))\s+" - r"(?P.+?)\s+to\s+(?P.+?)" - r"(?=\s+(?:in|and|then|before)\b|[.;]|$)", - value, - re.IGNORECASE, - ) - if action_match is None: - action_match = re.search( - r"\breplace\s+(?P.+?)\s+with\s+(?P.+?)" - r"(?=\s+(?:in|and|then|before)\b|[.;]|$)", + action_values = _captures_after_first_prefix( value, - re.IGNORECASE, + r"\b(?:chang(?:e|es|ed|ing)|updat(?:e|es|ed|ing))(?=\s)", + r"\s+(.+?)\s+to\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)", ) - if action_match is None: + if action_values is None: + action_values = _captures_after_first_prefix( + value, + r"\breplace(?=\s)", + r"\s+(.+?)\s+with\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)", + ) + if action_values is None: return None - old = _clean_file_edit_value(action_match.group("old")) - new = _clean_file_edit_value(action_match.group("new")) + old = _clean_file_edit_value(action_values[0]) + new = _clean_file_edit_value(action_values[1]) if not old or not new or old == new: return None return {"path": path, "old_string": old, "new_string": new} @@ -11295,6 +11641,24 @@ def _recent_mentioned_email_reference(messages: List[Dict]) -> dict[str, str]: return {} +_ASSISTANT_PROMPT_RE = re.compile( + r"(?:Want me|Would you like|Should I)\b", re.IGNORECASE +) + + +def _split_before_assistant_prompt(text: str) -> str: + """Return text before the first prompt introduced on a later line.""" + for match in _ASSISTANT_PROMPT_RE.finditer(text): + whitespace_start = match.start() + while whitespace_start and text[whitespace_start - 1].isspace(): + whitespace_start -= 1 + whitespace = text[whitespace_start:match.start()] + newline = whitespace.find("\n") + if newline >= 0: + return text[:whitespace_start + newline] + return text + + def _suggested_reply_from_recent_assistant(messages: List[Dict]) -> str: """Extract the most recent assistant-suggested email reply body.""" for message in reversed(messages or []): @@ -11308,7 +11672,7 @@ def _suggested_reply_from_recent_assistant(messages: List[Dict]) -> str: after = re.split(r"\*\*Suggested reply:\*\*|Suggested reply:", text, flags=re.IGNORECASE, maxsplit=1) if len(after) < 2: continue - body_section = re.split(r"\n\s*(?:Want me|Would you like|Should I)\b", after[1], flags=re.IGNORECASE, maxsplit=1)[0] + body_section = _split_before_assistant_prompt(after[1]) lines: list[str] = [] for raw_line in body_section.splitlines(): line = re.sub(r"^\s*>\s?", "", raw_line).rstrip() @@ -11343,9 +11707,9 @@ def _reply_draft_confirmation_block_from_recent_context(messages: List[Dict], te def _email_account_selector_from_label(label: str) -> str: value = str(label or "").strip() - match = re.search(r"<([^>]+)>", value) + match = next(iter_angle_contents(value), None) if match: - return match.group(1).strip() + return match[2].strip() return value @@ -11665,8 +12029,8 @@ def _email_bulk_blocks_from_search_output( continue account = str(default_account or "").strip() if not account: - account_match = re.search(r"<([^>]+)>", str(row.get("account") or "")) - account = account_match.group(1).strip() if account_match else str(row.get("account") or "").strip() + account_match = next(iter_angle_contents(str(row.get("account") or "")), None) + account = account_match[2].strip() if account_match else str(row.get("account") or "").strip() by_account.setdefault(account, []).append(uid) blocks: list[ToolBlock] = [] for account, uids in by_account.items(): @@ -11721,8 +12085,8 @@ def _named_email_row_from_recent_list_context(messages: List[Dict], text: str) - uid = str(row.get("uid") or "").strip() if not uid: continue - account_match = re.search(r"<([^>]+)>", str(row.get("account") or "")) - account = account_match.group(1).strip() if account_match else str(row.get("account") or "").strip() + account_match = next(iter_angle_contents(str(row.get("account") or "")), None) + account = account_match[2].strip() if account_match else str(row.get("account") or "").strip() ref = dict(row) ref["uid"] = uid ref["folder"] = "INBOX" @@ -11822,8 +12186,8 @@ def _alternate_email_attachment_blocks_from_recent_context(messages: List[Dict]) continue if sender not in downloaded_senders: continue - account_match = re.search(r"<([^>]+)>", str(row.get("account") or "")) - account = account_match.group(1).strip() if account_match else "" + account_match = next(iter_angle_contents(str(row.get("account") or "")), None) + account = account_match[2].strip() if account_match else "" blocks: list[ToolBlock] = [] for index, _name in enumerate(attachments): args = {"uid": uid, "index": index} @@ -12026,7 +12390,7 @@ def _recent_odysseus_anchor_refs(messages: List[Dict], history_session: Any = No refs: dict[str, str] = {} note_re = re.compile(r"#note-([0-9a-fA-F-]{8,64})") event_re = re.compile(r"#event-([0-9a-fA-F-]{8,64})") - event_link_re = re.compile(r"\[([^\]]+)\]\(#event-([0-9a-fA-F-]{8,64})\)") + event_link_target_re = re.compile(r"[0-9a-fA-F-]{8,64}") task_re = re.compile(r"(?:#task-|Created task '[^']+' \(id:\s*)([0-9a-fA-F-]{8,64})") document_re = re.compile(r"(?:#document-|doc_id['\"]?\s*[:=]\s*['\"]?)([0-9a-fA-F-]{8,64})") memory_re = re.compile(r"(?:memory_id['\"]?\s*[:=]\s*['\"]?|Memory id:\s*)([0-9a-fA-F-]{8,64})", re.IGNORECASE) @@ -12068,10 +12432,12 @@ def _recent_odysseus_anchor_refs(messages: List[Dict], history_session: Any = No if note_match: refs["note_id"] = note_match.group(1) if "event_uid" not in refs: - event_link_match = event_link_re.search(text) + event_link_match = next(iter_markdown_links( + text, target_prefix="#event-", target_re=event_link_target_re + ), None) if event_link_match: - refs["event_title"] = event_link_match.group(1).split(",", 1)[0].strip() - refs["event_uid"] = event_link_match.group(2) + refs["event_title"] = event_link_match[2].split(",", 1)[0].strip() + refs["event_uid"] = event_link_match[3] continue event_match = event_re.search(text) if event_match: @@ -12097,6 +12463,78 @@ def _recent_odysseus_anchor_refs(messages: List[Dict], history_session: Any = No return refs +def _numbered_row_parenthesized_ids(text: str) -> list[str]: + """Extract ``1. label (id) —`` identifiers with forward delimiters.""" + found: list[str] = [] + value = str(text or "") + line_start = 0 + consumed_until = 0 + while line_start <= len(value): + line_end = value.find("\n", line_start) + if line_end < 0: + line_end = len(value) + if line_start < consumed_until: + line_start = line_end + 1 + continue + pos = line_start + while pos < line_end and value[pos].isspace(): + pos += 1 + digit_start = pos + while pos < line_end and value[pos].isdigit(): + pos += 1 + if pos == digit_start or pos >= line_end or value[pos] != ".": + line_start = line_end + 1 + continue + pos += 1 + space_start = pos + while pos < len(value) and value[pos].isspace(): + pos += 1 + if pos == space_start: + line_start = line_end + 1 + continue + body_start = pos + body_end = value.find("\n", body_start) + if body_end < 0: + body_end = len(value) + whitespace_start = body_start + 1 + while whitespace_start <= body_end: + while whitespace_start < body_end and not value[whitespace_start].isspace(): + whitespace_start += 1 + if whitespace_start == body_end and ( + body_end >= len(value) or not value[body_end].isspace() + ): + break + whitespace_end = whitespace_start + while whitespace_end < len(value) and value[whitespace_end].isspace(): + whitespace_end += 1 + if whitespace_end >= len(value) or value[whitespace_end] != "(": + if whitespace_end > body_end: + break + whitespace_start = whitespace_end + continue + opener = whitespace_end + capture_line_end = value.find("\n", opener + 1) + if capture_line_end < 0: + capture_line_end = len(value) + closer = value.find(")", opener + 1, capture_line_end) + if closer < 0: + break + if closer == opener + 1: + whitespace_start = closer + 1 + continue + suffix = closer + 1 + suffix_start = suffix + while suffix < len(value) and value[suffix].isspace(): + suffix += 1 + if suffix > suffix_start and suffix < len(value) and value[suffix] in "—-": + found.append(value[opener + 1:closer]) + consumed_until = suffix + 1 + break + whitespace_start = closer + 1 + line_start = line_end + 1 + return found + + def _ordinal_collection_mutation_target( user_text: str, messages: List[Dict], @@ -12164,14 +12602,7 @@ def _ordinal_collection_mutation_target( continue output = str(event.get("output") or "") if family == "tasks": - identifiers = [ - found.strip() - for found in re.findall( - r"^\s*\d+\.\s+.+?\s+\(([^)\n]+)\)\s+[—-]", - output, - re.MULTILINE, - ) - ] + identifiers = [found.strip() for found in _numbered_row_parenthesized_ids(output)] else: identifiers = re.findall(r"\]\(#event-([A-Za-z0-9_-]+)\)", output) if 1 <= index <= len(identifiers): @@ -12471,6 +12902,28 @@ def _is_generic_email_reply_body(body: str) -> bool: } +_SUMMARY_PREFIX_RE = re.compile( + r"\bsummary\s*(?:\*\*)?\s*:?\s*", re.IGNORECASE +) + + +def _contextual_summary_fragment(text: str) -> str: + """Preserve the legacy nonempty summary capture with bounded scans. + + The legacy overescaped backslash alternative accepts an empty suffix, + so every newline terminates the capture. Bound the greedy prefix before + the last possible body character to preserve its whitespace backtracking. + """ + last_newline = text.rfind("\n") + if last_newline < 1: + return "" + prefix = _SUMMARY_PREFIX_RE.search(text, 0, last_newline - 1) + if prefix is None: + return "" + newline = text.find("\n", prefix.end() + 1) + return text[prefix.end():newline] if newline >= 0 else "" + + def _contextual_reply_body_from_recent_email_context(messages: List[Dict]) -> str: """Build a bounded draft body from the latest assistant email summary. @@ -12486,7 +12939,7 @@ def _contextual_reply_body_from_recent_email_context(messages: List[Dict]) -> st subject = "" summary = "" def _clean_fragment(raw: str) -> str: - cleaned = re.sub(r"\[([^\]]+)\]\([^)]+\)", r"\1", str(raw or "")) + cleaned = replace_markdown_links_with_labels(str(raw or "")) cleaned = re.sub(r"[*_`>#]+", "", cleaned) return re.sub(r"\s+", " ", cleaned).strip(" .[]\"'") @@ -12501,13 +12954,9 @@ def _contextual_reply_body_from_recent_email_context(messages: List[Dict]) -> st ) if subject_match: subject = _clean_fragment(subject_match.group(1)) - summary_match = re.search( - r"\bsummary\s*(?:\*\*)?\s*:?\s*(.+?)(?:\n\s*(?:-|\\*\\*|If you|Want me|This is|$))", - text, - re.IGNORECASE | re.DOTALL, - ) - if summary_match: - summary = _clean_fragment(summary_match.group(1)) + summary_fragment = _contextual_summary_fragment(text) + if summary_fragment: + summary = _clean_fragment(summary_fragment) if not subject and not summary: continue if subject: @@ -12947,17 +13396,46 @@ def _normalize_ody_qwen_text_artifacts(text: str, *, strip_edges: bool = True) - _ODY_QWEN_LEAKED_TOOL_TEXT_RE = re.compile( r"(<\s*/?\s*(?:function|parameter|tool_call)\b" - r"|(?:^|\n)\s*(?:function|parameter)\s*=" r"|\bmanage_(?:notes|calendar|memory|documents|contact)\s*\(" r"|\"function\"\s*:\s*\"(?:manage_|mcp__)" r"|mcp__email__" - r"|(?:^|\n)\s*(?:web_search|web_fetch|private_browser)\s*:)", + r")", + re.IGNORECASE, +) +_ODY_QWEN_LINE_TOOL_TEXT_RE = re.compile( + r"(?:function|parameter)\s*=|(?:web_search|web_fetch|private_browser)\s*:", re.IGNORECASE, ) def _looks_like_ody_qwen_leaked_tool_text(text: str) -> bool: - return bool(_ODY_QWEN_LEAKED_TOOL_TEXT_RE.search(text or "")) + value = str(text or "") + if _ODY_QWEN_LEAKED_TOOL_TEXT_RE.search(value): + return True + line_start = 0 + while line_start <= len(value): + candidate = line_start + while candidate < len(value) and value[candidate].isspace(): + candidate += 1 + if _ODY_QWEN_LINE_TOOL_TEXT_RE.match(value, candidate): + return True + newline = value.find("\n", candidate) + if newline < 0: + return False + line_start = newline + 1 + return False + + +def _contains_email_draft_headers(text: str) -> bool: + """Recognize the legacy To/Subject/footer shape with ordered fixed scans.""" + to_match = re.search(r"\bTo:", text, re.IGNORECASE) + if to_match is None: + return False + subject = re.search(r"\bSubject:", text[to_match.end() + 1:], re.IGNORECASE) + if subject is None: + return False + subject_end = to_match.end() + 1 + subject.end() + return text.find("\n---", subject_end + 1) >= 0 def _ody_qwen_terminal_tool_summary(tool_event: dict[str, Any], user_text: str = "") -> str: @@ -13003,7 +13481,7 @@ def _ody_qwen_terminal_tool_summary(tool_event: dict[str, Any], user_text: str = if tool_name in {"update_document", "edit_document"}: lowered = output.lower() if "document updated" in lowered or "edit applied" in lowered or "updated" in lowered: - if re.search(r"\bTo:\s*.+\bSubject:\s*.+\n---", command, re.IGNORECASE | re.DOTALL): + if _contains_email_draft_headers(command): return "Updated the active email draft." return "Updated the active document." return output.removeprefix("AI: ").strip() @@ -13224,9 +13702,11 @@ def _tui_coding_failure_summary(tool_events: list[dict[str, Any]]) -> str: _DESTRUCTIVE_REQUEST_RE = re.compile( - r"\b(delete|remove|archive|trash|send|reply|unsubscribe|mark\s+.*read)\b", + r"\b(delete|remove|archive|trash|send|reply|unsubscribe)\b", re.IGNORECASE, ) +_MARK_REQUEST_RE = re.compile(r"\bmark", re.IGNORECASE) +_READ_WORD_RE = re.compile(r"read\b", re.IGNORECASE) _FAKE_SUCCESS_RE = re.compile( r"\b(done|removed|deleted|sent|archived|unsubscribed|marked)\b", @@ -13235,7 +13715,25 @@ _FAKE_SUCCESS_RE = re.compile( def _looks_like_destructive_request(text: str) -> bool: - return bool(_DESTRUCTIVE_REQUEST_RE.search(text or "")) + value = str(text or "") + if _DESTRUCTIVE_REQUEST_RE.search(value): + return True + reads = list(_READ_WORD_RE.finditer(value)) + read_index = 0 + for mark in _MARK_REQUEST_RE.finditer(value): + cursor = mark.end() + if cursor >= len(value) or not value[cursor].isspace(): + continue + while cursor < len(value) and value[cursor].isspace(): + cursor += 1 + while read_index < len(reads) and reads[read_index].start() < cursor: + read_index += 1 + newline = value.find("\n", cursor) + if read_index < len(reads) and ( + newline < 0 or reads[read_index].start() < newline + ): + return True + return False def _looks_like_success_claim(text: str) -> bool: @@ -17272,13 +17770,19 @@ def _private_browser_product_query(user_text: str) -> str: text = re.sub(r"\s+", " ", str(user_text or "")).strip() match = re.search( r"\b(?:find|look\s+for|shop\s+for|search\s+for)\s+" - r"(?:me\s+)?(?:the\s+)?(?:best\s+)?(?P.+?)\s*[?.!]*$", + r"(?:me\s+)?(?:the\s+)?(?:best\s+)?", text, re.IGNORECASE, ) - if not match: + if not match or match.end() >= len(text): return "" - query = match.group("query").strip(" \t\r\n.,!?;:") + raw_query = text[match.end():] + query_end = len(raw_query) + while query_end and raw_query[query_end - 1] in "?.!": + query_end -= 1 + if query_end == 0: + query_end = 1 + query = raw_query[:query_end].strip(" \t\r\n.,!?;:") query = re.sub( r"\s+(?:on|at|from)\s+(?:the\s+)?[A-Za-z0-9&.' -]{1,60}$", "", @@ -18927,7 +19431,7 @@ def _read_only_shell_command(content: str) -> bool: return False # Shell pipelines are allowed only when every stage is one of the common # inspection commands. This intentionally rejects unknown/mutating syntax. - segments = re.split(r"\s*(?:&&|\|\||;|\|)\s*", text) + segments = re.split(r"&&|\|\||;|\|", text) if not segments or any(not segment.strip() for segment in segments): return False allowed = re.compile( @@ -23084,11 +23588,7 @@ async def stream_agent_loop( # Preserve that generic execution floor unless the task is # explicitly URL-backed (filtered below). if ( - re.search( - r"/workspace/[^\s`\"']+\.(?:py|pyw|sh|bash|js|mjs|ts|rb|pl)\b", - _last_user, - re.IGNORECASE, - ) + _mentions_workspace_script(_last_user) or ( len(_workspace_artifacts) >= 2 and any(Path(path).suffix.casefold() in {".csv", ".json", ".xlsx"} for path in _workspace_artifacts) @@ -25826,10 +26326,11 @@ async def stream_agent_loop( client_runtime_context=client_runtime_context, ) except Exception as _preemptive_exc: - logger.warning("Preemptive calendar lookup failed: %s", _preemptive_exc) + logger.warning("Preemptive calendar lookup failed: %s", _preemptive_exc, exc_info=True) _preemptive_desc = "manage_calendar: ERROR" _preemptive_result = { - "error": str(_preemptive_exc), + "error": "The calendar lookup failed unexpectedly. Check the server log and retry.", + "error_category": "tool_execution_error", "exit_code": 1, "output": "", } @@ -25856,6 +26357,8 @@ async def stream_agent_loop( } if isinstance(_preemptive_result, dict) and isinstance(_preemptive_result.get("events"), list): _preemptive_tool_output["events"] = _preemptive_result.get("events") + if isinstance(_preemptive_result, dict) and _preemptive_result.get("error_category") == "tool_execution_error": + _preemptive_tool_output["error_category"] = "tool_execution_error" yield f"data: {json.dumps(_preemptive_tool_output)}\n\n" _preemptive_tool_event = { "round": round_num, @@ -26023,10 +26526,11 @@ async def stream_agent_loop( client_runtime_context=client_runtime_context, ) except Exception as _preemptive_exc: - logger.warning("Preemptive explicit tool failed: %s", _preemptive_exc) + logger.warning("Preemptive explicit tool failed: %s", _preemptive_exc, exc_info=True) _preemptive_desc = f"{_preemptive_tool}: ERROR" _preemptive_result = { - "error": str(_preemptive_exc), + "error": "The requested tool failed unexpectedly. Check the server log and retry.", + "error_category": "tool_execution_error", "exit_code": 1, "output": "", } @@ -26051,6 +26555,8 @@ async def stream_agent_loop( ), "fallback": "preemptive_explicit_admin_session", } + if isinstance(_preemptive_result, dict) and _preemptive_result.get("error_category") == "tool_execution_error": + _preemptive_tool_output["error_category"] = "tool_execution_error" yield f"data: {json.dumps(_preemptive_tool_output)}\n\n" _preemptive_tool_event = { "round": round_num, @@ -29261,15 +29767,12 @@ async def stream_agent_loop( ): _oversized_svg = _extract_oversized_svg(round_response) if _oversized_svg: - _svg_title_match = re.search( - r"]*)?>([\s\S]*?)", - _oversized_svg, - re.IGNORECASE, + _svg_title_content = first_tag_content( + _oversized_svg, "title", allow_attributes=True ) - _svg_title = re.sub( - r"<[^>]*>", - "", - _svg_title_match.group(1) if _svg_title_match else "Visual explanation", + _svg_title = strip_angle_tags( + _svg_title_content if _svg_title_content is not None else "Visual explanation", + allow_empty=True, ).strip()[:100] or "Visual explanation" full_response = _drop_rejected_round_response(full_response, round_response) round_response = "" @@ -29647,12 +30150,7 @@ async def stream_agent_loop( and _local_media_turn and not _artifact_creation_requested and not _local_media_detail_nudge_sent - and re.search( - r"\b(?:how\s+many|count|break\s*points?|timestamps?|what\s+time|" - r"when\s+.*(?:end|happen)|score(?:board)?s?)\b", - _last_user, - re.IGNORECASE, - ) + and contains_detailed_sequence_request(_last_user, include_first=False) ): _successful_media_inspections = sum( 1 @@ -30357,7 +30855,11 @@ async def stream_agent_loop( _completion_requirements, ).evaluate().can_complete ) - _intent_match = _INTENT_RE.search(_intent_text) if _intent_text else None + # Whole-response intent matching is redundant: the helper below + # keeps short responses below 400 characters and long responses + # to their final 600-character window. Avoid an unbounded scan of + # model text before applying that existing policy boundary. + _intent_match = None # Inspect only the bounded tail of long answers. This catches # substantial multimodal analyses that end in "let me inspect..." # or a dangling answer lead-in while leaving completed answers @@ -32032,9 +32534,7 @@ async def stream_agent_loop( _document_args = None if isinstance(_document_args, dict): _raw_document_action = str(_document_args.get("action") or "").strip() - _document_action = re.sub( - r"<[^>]+>", - "", + _document_action = strip_angle_tags( _raw_document_action.splitlines()[0] if _raw_document_action else "", ).strip().lower() if _raw_document_action and _document_action != _raw_document_action.lower(): @@ -36732,10 +37232,8 @@ async def stream_agent_loop( # prose. Local finetunes may emit those before the parser catches and # executes them; saved history should contain only the user-facing answer. full_response = _visible_response_text(full_response) - if re.match(r"^Done\b", full_response, re.IGNORECASE) and re.search( - r"\s*Done\.\s*$", full_response, re.IGNORECASE - ): - without_trailing_done = re.sub(r"\s*Done\.\s*$", "", full_response, flags=re.IGNORECASE).rstrip() + if re.match(r"^Done\b", full_response, re.IGNORECASE): + without_trailing_done = _strip_trailing_done(full_response) if without_trailing_done: full_response = without_trailing_done if _ody_qwen_finetune_model or _qwen38_tool_router: diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py index 6fc1cff0a..9666e430c 100644 --- a/src/clean_agent_preview.py +++ b/src/clean_agent_preview.py @@ -32,7 +32,12 @@ from src.tool_schemas import ( normalized_native_function_argument_error, ) from src.tool_types import ToolBlock -from src.tool_parsing import parse_tool_blocks, strip_tool_blocks +from src.tool_parsing import iter_email_addresses, parse_tool_blocks, strip_tool_blocks, strip_angle_tags +from src.text_scanning import ( + contains_detailed_sequence_request, + contains_search_engine_navigation, + iter_prefixed_token_matches, +) from src.turn_contract import ( _REQUEST_PREFIX, calendar_retiming_request, @@ -54,6 +59,59 @@ MODE = COMPACT_PREVIEW_MODE class ProviderStreamError(Exception): """A provider reported failure inside an otherwise successful SSE response.""" + +# Only these audited, fixed domain messages may cross the exception boundary. +# Return the canonical constants rather than arbitrary exception text. +_PUBLIC_PREVIEW_TOOL_ERRORS = {message: message for message in ( + 'An equivalent search already returned evidence. Change the angle, missing subtopic, source type, or corroboration target instead of only changing freshness wording.', + 'Browser navigation already failed for this exact URL; use another source page.', + 'Do not infer image content repeatedly from filenames; use inspect_media on representative files, then continue from visual evidence.', + 'Equivalent search intent was repeated after a correction reminder; web_search is disabled for this turn.', + 'Look up the target with list_sessions before changing a chat. Use its exact returned ID; never invent last-chat/latest aliases. If the target is ambiguous, ask using the candidate chat titles.', + 'Resolve the named recipient with resolve_contact before drafting. Never invent an email address.', + 'Selection-only edits cannot use replace_all. Use a unique contextual FIND inside the selected passage.', + 'The bounded search-attempt budget is exhausted. Do not search again; answer from usable evidence already gathered, or clearly report what could not be verified and suggest a concrete next step.', + 'The calendar read has not succeeded yet. Obtain the requested calendar evidence before creating the dependent email draft.', + 'The latest note search returned no candidates, so this reference has no note to open. Do not reuse an older list item; report the empty result or ask which note was intended.', + 'The latest research search returned no candidates, so this reference has no report to open. Do not reuse an unrelated older report.', + 'The proposed expansion only adds placeholder/meta text. Write substantive content that continues the existing document’s subject, voice, and format; do not announce that a paragraph was added.', + 'The recipient address is not supported by the contact lookup. Use an exact returned address for the requested person; if no match exists, explain the missing recipient instead of guessing.', + 'The same artifact target was already rewritten three times; finish from the latest successful version instead of rewriting it again.', + 'These suggestions collapse multiple different passages into the same much shorter replacement, violating the request to preserve meaning. Produce passage-specific revisions that retain each source passage’s claims and intent.', + 'This exact call already failed twice and will not be executed again; change strategy or finish from existing evidence.', + 'This exact failed call was repeated after a correction reminder; the tool is disabled for this turn. Finish from existing evidence.', + 'This exact invalid call was repeated after two validation failures; the tool is disabled for this turn.', + 'This exact successful call already returned evidence. Do not repeat it; change the arguments or tool to gather different evidence, or finish from the evidence already available.', + 'This still image was already inspected and the same visual evidence is already in context. inspect_media is withheld for the next correction round; use that evidence, inspect a different file, or finish.', + 'This operation is outside the preview safety policy. No change was made.', + 'This replacement does not make its passage more concise. Shorten the wording while retaining its facts and meaning; a spelling-only change does not satisfy the requested action. Retry with shorter replacements.', + 'Tool arguments could not be converted for execution.', + 'Tool arguments must be a JSON object.', + 'Tool execution budget exhausted; finish from existing evidence.', + 'Tool is not offered or permitted.', + 'Two equivalent searches already returned no evidence. Do not repeat this search wording; use a different offered tool or a materially different query.', + 'Unresolved video target: this ID was not supplied by the user or observed in a successful tool result. Do not guess it from the title. Open/read the referenced browser link or call youtube_tool latest_channel_video for the observed channel, then use its returned ID.', + 'inspect_media exports only extract video stills and require an explicit timestamp plus output_path for every item. First inspect the video to find the timestamp; use write_file or python to author a diagram or other new artifact.', + 'web_fetch requires url or urls. query only filters a supplied page; it is not a search or writing request.', +)} + + +def _public_preview_tool_error(exc, *, execution_attempted=False): + """Keep useful domain guidance while withholding arbitrary diagnostics.""" + if execution_attempted: + return 'The tool failed unexpectedly. Check the server log and retry.' + if isinstance(exc, json.JSONDecodeError): + return 'Tool arguments are not valid JSON. Correct the JSON object and retry.' + if isinstance(exc, jsonschema.ValidationError): + return 'Tool arguments do not match the required schema. Correct the call using the offered tool schema.' + detail = str(exc) + if detail.startswith('Artifact completion Python must reference the required '): + return 'Artifact completion Python must create non-empty output at the required artifact target.' + if detail.startswith('Shell access to credential variable '): + return 'Shell access to credentials is blocked. Use brokered native tools.' + return _PUBLIC_PREVIEW_TOOL_ERRORS.get(detail, 'The tool call could not be validated. Check its arguments and retry.') + + # Native unattended workspaces routinely require several inspections followed # by several artifact writes. The interactive preview keeps its six-call # limit below; this larger budget applies only after server-side validation of @@ -79,13 +137,6 @@ ARTIFACT_RESEARCH_TOOLS = frozenset({ 'web_search', 'web_fetch', 'private_browser', 'pdf_extract', 'youtube_tool', 'inspect_media', 'extract_text', 'transcribe_media', }) -DETAILED_VIDEO_REQUEST = re.compile( - r"\b(?:how\s+many|count|sequence|in\s+order|chronological|" - r"timestamps?|what\s+time|at\s+what\s+time|when\s+.*(?:end|happen)|" - r"first\s+.*(?:save|attempt|event)|score(?:board)?s?)\b|" - r"(?:多少|几次|何时|什么时候|时间|顺序)", - re.IGNORECASE, -) READ_TOOLS = frozenset({ 'manage_notes', 'manage_calendar', 'manage_memory', 'manage_skills', 'manage_tasks', 'manage_documents', 'manage_research', 'manage_contact', 'list_sessions', @@ -809,13 +860,95 @@ def malformed_write_handoff_target(arguments, required_artifacts=(), user_text=' return target +_PAGE_LISTING_WORDS = ("stories", "articles", "posts", "headlines", "pages") + + +def _listing_allowed(char): + return char == " " or char in ".:/-" or char == "_" or char.isalnum() + + +def _allowed_then_space(value): + """Whether value can be class+ followed by whitespace+, without retries.""" + if len(value) < 2: + return False + first_disallowed = next((i for i, char in enumerate(value) if not _listing_allowed(char)), len(value)) + if first_disallowed < len(value): + return first_disallowed > 0 and all(char.isspace() for char in value[first_disallowed:]) + return value[-1].isspace() + + +def _space_then_allowed(value): + """Whether value can be whitespace+ followed by the listing class+.""" + if len(value) < 2: + return False + first_nonspace = next((i for i, char in enumerate(value) if not char.isspace()), len(value)) + if first_nonspace < len(value): + return first_nonspace > 0 and all(_listing_allowed(char) for char in value[first_nonspace:]) + trailing_spaces = len(value) - len(value.rstrip(" ")) + return trailing_spaces > 0 and (len(value) > trailing_spaces or trailing_spaces >= 2) + + +def _page_listing_request(user_text): + value = str(user_text or '').lstrip() + nonspace_end = len(value.rstrip()) + if value[nonspace_end - 1:nonspace_end] in ("!", "?"): + value = value[:nonspace_end - 1] + lead = re.match( + r"(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)(?=\s)", + value, re.I, + ) + if not lead: + return False + cursor = lead.end() + while cursor < len(value) and value[cursor].isspace(): + cursor += 1 + body = value[cursor:] + nonspace_end = len(body.rstrip()) + terminal_end = nonspace_end + if body[terminal_end - 1:terminal_end] == ".": + terminal_end = len(body[:terminal_end - 1].rstrip()) + first_disallowed = len(body) + last_disallowed = -1 + first_nonspace_after_disallowed = len(body) + for index, char in enumerate(body): + if not _listing_allowed(char): + first_disallowed = min(first_disallowed, index) + if index < nonspace_end: + last_disallowed = index + if index >= first_disallowed and not char.isspace(): + first_nonspace_after_disallowed = min(first_nonspace_after_disallowed, index) + last_literal_space = body.rfind(" ") + for word in _PAGE_LISTING_WORDS: + for candidate in re.finditer(word, body, re.I): + start = candidate.start() + prefix_allowed = start >= 2 and ( + body[start - 1].isspace() if start <= first_disallowed + else first_disallowed > 0 and start <= first_nonspace_after_disallowed + ) + if start == 0 or prefix_allowed: + after_start = candidate.end() + if after_start >= terminal_end: + return True + on = after_start + while on < len(body) and body[on].isspace(): + on += 1 + if on > after_start and body[on:on + 2].casefold() == "on": + tail_start = on + 2 + tail_end = tail_start + while tail_end < len(body) and body[tail_end].isspace(): + tail_end += 1 + if len(body) - tail_start >= 2 and tail_end > tail_start: + if tail_end < len(body): + if last_disallowed < tail_end: + return True + elif last_literal_space > tail_start: + return True + return False + + def page_listing_response(entries, user_text, max_items=10): """Render simple page listings from observed titles/URLs, never synthesized rankings.""" - if not re.fullmatch( - r'\s*(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)\s+' - r'(?:[\w .:/-]+\s+)?(?:stories|articles|posts|headlines|pages)' - r'(?:\s+on\s+[\w .:/-]+)?[.!?]?\s*', user_text, re.I, - ) or re.search(r'\b(?:and|compare|summarize|analyse|analyze|about|by|since|yesterday)\b', user_text, re.I): + if not _page_listing_request(user_text) or re.search(r'\b(?:and|compare|summarize|analyse|analyze|about|by|since|yesterday)\b', user_text, re.I): return '' from urllib.parse import quote, urlsplit from html import escape @@ -1531,17 +1664,24 @@ def prior_workspace_path_answer(user_text, history): return '' +def _prior_web_source_request(text): + text = str(text or '') + candidate = text.lstrip().rstrip() + candidate = candidate.rstrip(".!? ") + return bool(re.fullmatch( + r"(?:(?:where|what)\s+did\s+you\s+(?:get|find)\s+(?:that|this)\s+from[?., ]*" + r"(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link" + r"|(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link(?:\s+for\s+that)?" + r"|what(?:['’]?s|\s+is)\s+(?:the\s+)?source(?:\s+link)?)", + candidate, + re.I, + )) + + def prior_web_source_answer(user_text, history): """Return the latest source URL for an explicit source-only follow-up.""" text = str(user_text or '') - if not re.fullmatch( - r"\s*(?:(?:where|what)\s+did\s+you\s+(?:get|find)\s+(?:that|this)\s+from[?., ]*" - r"(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link[.!? ]*" - r"|(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link(?:\s+for\s+that)?[.!? ]*" - r"|what(?:['’]?s|\s+is)\s+(?:the\s+)?source(?:\s+link)?[.!? ]*)\s*", - text, - re.I, - ): + if not _prior_web_source_request(text): return '' call_names = {} candidates = [] @@ -2822,12 +2962,11 @@ def draft_contact_evidence_error(name, args, *, dependencies=(), executions=(), and not e.get('blocked')] if not observations: return 'Resolve the named recipient with resolve_contact before drafting. Never invent an email address.' - address_pattern = r'[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}' known = {address.casefold() for e in observations - for address in re.findall(address_pattern, str(e.get('output') or ''))} - known.update(address.casefold() for address in re.findall(address_pattern, user_text)) + for address in iter_email_addresses(str(e.get('output') or ''), ascii_only=True)} + known.update(address.casefold() for address in iter_email_addresses(user_text, ascii_only=True)) proposed = {address.casefold() for field in ('to', 'cc', 'bcc') - for address in re.findall(address_pattern, str(args.get(field) or ''))} + for address in iter_email_addresses(str(args.get(field) or ''), ascii_only=True)} if not proposed or not proposed <= known: return ('The recipient address is not supported by the contact lookup. Use an exact ' 'returned address for the requested person; if no match exists, explain ' @@ -3945,7 +4084,7 @@ def active_document_revision_quality_error(name, args, *, active_document, user_ if not incoming: return None addition = incoming[len(existing):].strip() if existing and incoming.startswith(existing) else '' - candidate = re.sub(r'<[^>]+>', ' ', addition).strip() + candidate = strip_angle_tags(addition, ' ').strip() candidate = re.sub(r'\s+', ' ', candidate) if not candidate or candidate.casefold() in str(user_text or '').casefold(): return None @@ -3970,8 +4109,8 @@ def document_suggestion_quality_error(name, args, *, user_text): for suggestion in suggestions: if not isinstance(suggestion, dict): continue - source = re.sub(r'\s+', ' ', re.sub(r'<[^>]*>', ' ', str(suggestion.get('find') or ''))).strip() - result = re.sub(r'\s+', ' ', re.sub(r'<[^>]*>', ' ', str(suggestion.get('replace') or ''))).strip() + source = re.sub(r'\s+', ' ', strip_angle_tags(str(suggestion.get('find') or ''), ' ', allow_empty=True)).strip() + result = re.sub(r'\s+', ' ', strip_angle_tags(str(suggestion.get('replace') or ''), ' ', allow_empty=True)).strip() if not source or not result: continue source_words = len(re.findall(r"\b[\w’'-]+\b", source)) @@ -4127,13 +4266,17 @@ _WORKSPACE_FILE_RE = re.compile( r"/workspace/[^\s,,、;;`\"'<>]+\.[A-Za-z0-9]{1,12}", re.I, ) +_WORKSPACE_PREFIX_RE = re.compile(r"/workspace/", re.I) +_WORKSPACE_FILE_TOKEN_TAIL_RE = re.compile(r"[^\s,,、;;`\"'<>]*") def declared_workspace_artifacts(user_text): """Return explicit output paths, excluding paths used only as inputs.""" text = str(user_text or '') paths = [] - for match in _WORKSPACE_FILE_RE.finditer(text): + for match in iter_prefixed_token_matches( + text, _WORKSPACE_PREFIX_RE, _WORKSPACE_FILE_RE, _WORKSPACE_FILE_TOKEN_TAIL_RE + ): path = match.group(0).rstrip('.!?))]}') if path.startswith('/workspace/fixtures/') or path in paths: continue @@ -4253,17 +4396,61 @@ def evidence_tool_keeps_distinct_requests_available(name): } +def _terminal_source_link_clause(text): + """Recognize the terminal short source/link clause from right to left.""" + value = str(text or '') + def matches_at(end): + while end and value[end - 1] in ".!?": + end -= 1 + while end and value[end - 1].isspace(): + end -= 1 + lowered = value[:end].casefold() + for courtesy in ("please", "pls"): + if lowered.endswith(courtesy): + boundary = end - len(courtesy) + end = boundary + while end and value[end - 1].isspace(): + end -= 1 + lowered = value[:end].casefold() + break + token_match = re.search(r"(?:sources?|citations?|links?)$", lowered) + if token_match is None: + return False + prefix = value[:token_match.start()] + original_cursor = len(prefix) + courtesy_end = original_cursor + while courtesy_end and prefix[courtesy_end - 1].isspace(): + courtesy_end -= 1 + cursors = [original_cursor] + for courtesy in ("please", "pls"): + start = courtesy_end - len(courtesy) + if start >= 0 and prefix[start:courtesy_end].casefold() == courtesy: + if courtesy_end < original_cursor and (start == 0 or prefix[start - 1].isspace()): + cursors.append(start) + break + for cursor in cursors: + while cursor and prefix[cursor - 1].isspace(): + if prefix[cursor - 1] == "\n": + return True + cursor -= 1 + if cursor == 0 or prefix[cursor - 1] in ".!?;,": + return True + return False + + return matches_at(len(value)) or (value.endswith("\n") and matches_at(len(value) - 1)) + + def requested_web_source_links(user_text): - return bool(re.search( + text = str(user_text or '') + return _terminal_source_link_clause(text) or bool(re.search( r'\b(?:return|give|show|include|provide|cite|find)\b.{0,35}\b(?:source\s+)?links?\b' - r'|(?:^|[.!?;,\n])\s*(?:(?:pls|please)\s+)?(?:sources?|citations?|links?)\s*(?:pls|please)?\s*[.!?]*$' r'|\b(?:\d+|one|two|three|four|five)\s+(?:official\s+)?(?:source\s+)?links?\b' r'|\bofficial\s+source\b' r'|\b(?:with|include|provide|cite|show|give|find)\s+(?:the\s+)?(?:official\s+)?(?:sources|citations)\b' r'|\blink\s+(?:to\s+)?(?:the\s+|your\s+)?(?:original\s+|official\s+)?(?:instructions|sources|documentation|articles?|reports?|studies|manuals?|guides?)\b' r'|\b(?:find|locate|get|download)\b.{0,60}\bofficial\b.{0,60}\b(?:manual|guide|handbook|pdf|documentation)\b' r'|\b(?:find|locate|get|download)\b.{0,80}\b(?:manual|guide|handbook|pdf)\b.{0,40}\b(?:online|official)\b', - str(user_text or ''), + text, re.IGNORECASE, )) @@ -4300,12 +4487,67 @@ def unbound_lookup_reference(user_text, history, *, supplied_context=False): )) +def _web_source_rows(text): + """Extract numbered source title/URL rows with monotonic line scans.""" + rows = [] + position = 0 + length = len(text) + while position < length: + if position and text[position - 1] != "\n": + newline = text.find("\n", position) + if newline < 0: + break + position = newline + 1 + continue + cursor = position + if cursor >= length or text[cursor] != "[": + newline = text.find("\n", cursor) + if newline < 0: + break + position = newline + 1 + continue + cursor += 1 + digit_start = cursor + while cursor < length and text[cursor].isdigit(): + cursor += 1 + if cursor == digit_start or cursor >= length or text[cursor] != "]": + position += 1 + continue + cursor += 1 + if cursor >= length or not text[cursor].isspace(): + position += 1 + continue + while cursor < length and text[cursor].isspace(): + cursor += 1 + title_start = cursor + title_line_end = text.find("\n", title_start) + if title_line_end < 0: + break + title_end = title_line_end + while title_end > title_start and text[title_end - 1].isspace(): + title_end -= 1 + url_start = title_line_end + 1 + while url_start < length and text[url_start].isspace(): + url_start += 1 + scheme_length = 7 if text.startswith("http://", url_start) else 8 if text.startswith("https://", url_start) else 0 + if title_end > title_start and scheme_length: + url_end = url_start + scheme_length + while url_end < length and not text[url_end].isspace(): + url_end += 1 + rows.append((text[title_start:title_end], text[url_start:url_end])) + position = url_end + continue + newline = text.find("\n", position) + if newline < 0: + break + position = newline + 1 + return rows + + def web_source_links(raw, *, max_items=1, prefer_official=False, query=''): """Extract stable title/URL pairs from the web tool's source preamble.""" text = str(raw or '') - rows = re.findall( - r'^\[\d+\]\s+(.+?)\s*\n\s*(https?://\S+)', text, re.MULTILINE, - ) + rows = _web_source_rows(text) query_tokens = set(re.findall(r'[a-z0-9]+', str(query or '').casefold())) - { 'the', 'a', 'an', 'official', 'source', 'link', 'page', 'website', 'site', 'guide', 'search', 'find', 'for', 'return', @@ -5834,7 +6076,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac provider_error = payload['error'] if isinstance(provider_error, dict): provider_error = provider_error.get('message') or provider_error.get('detail') - detail = str(provider_error or 'Unknown provider error').strip()[:300] + detail = str(provider_error or 'Unknown provider error').strip() raise ProviderStreamError(detail) usage = payload.get('usage') or {} if usage: @@ -6353,7 +6595,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac if ( native_workspace_enabled and content - and DETAILED_VIDEO_REQUEST.search(direct_user_text) + and contains_detailed_sequence_request(direct_user_text) and successful_video_inspections == 1 and not media_detail_nudge_sent and round_number < round_limit @@ -7071,7 +7313,14 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac ): artifact_body_handoff_attempts += 1 artifact_body_handoff_target = handoff_target - result = {'error': str(exc).splitlines()[0][:300], 'exit_code': 1} + logging.getLogger(__name__).warning( + 'Clean v3 tool call failed: %s', exc, exc_info=True, + ) + result = { + 'error': _public_preview_tool_error(exc, execution_attempted=execution_attempted), + 'error_category': 'tool_execution_error' if execution_attempted else 'invalid_tool_arguments', + 'exit_code': 1, + } if (canonical(block.tool_type if block is not None else name) == 'youtube_tool' and result.get('exit_code') not in (None, 0)): round_recovery_messages.append( @@ -7260,12 +7509,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac requested_browser_url = private_browser_open_url(args) effective_browser_url = private_browser_effective_url(result) browser_search_url = requested_browser_url or effective_browser_url - is_search_engine_navigation = bool(re.search( - r'https?://(?:[^/]+\.)?(?:google\.[^/]+|bing\.com|duckduckgo\.com)' - r'/(?:search|sorry|html|lite|\?)', - browser_search_url, - re.I, - )) or bool(re.search( + is_search_engine_navigation = contains_search_engine_navigation( + browser_search_url + ) or bool(re.search( r'https?://(?:[^/]+\.)?google\.[^/]+/sorry/', effective_browser_url, re.I, @@ -7433,6 +7679,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac 'execution_attempted': execution_attempted, 'blocked': policy_denied or schema is None, 'desc': desc, 'round': round_number} + for error_category in ('tool_execution_error', 'invalid_tool_arguments'): + if result.get('error_category') == error_category: + tool_event['error_category'] = error_category reader_event = email_reader_event(direct_user_text, actual_tool, args, result, failed=failed) if reader_event: yield event(reader_event) @@ -8005,9 +8254,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac else: yield event({'delta': '\nThe preview reached its round limit. Please narrow the request.'}) except ProviderStreamError as exc: - detail = f'The selected model provider failed while generating: {exc}' - logging.getLogger(__name__).warning('Clean v3 provider stream failed: %s', exc) - yield f'event: error\ndata: {json.dumps({"status": 502, "error": detail})}\n\n' + detail = 'The selected model provider failed while generating. Retry or choose another model.' + logging.getLogger(__name__).warning('Clean v3 provider stream failed: %s', exc, exc_info=True) + yield f'event: error\ndata: {json.dumps({"status": 502, "error": detail, "error_category": "provider_stream_error"})}\n\n' return except httpx.HTTPStatusError as exc: status = exc.response.status_code diff --git a/src/text_scanning.py b/src/text_scanning.py new file mode 100644 index 000000000..59fad4cf5 --- /dev/null +++ b/src/text_scanning.py @@ -0,0 +1,210 @@ +"""Forward-only helpers for permissive text grammars. + +These helpers retain the existing regular expressions as anchored token +parsers while preventing ``re.search`` from retrying the same token suffix at +every embedded prefix. +""" + +from __future__ import annotations + +import re +from collections.abc import Iterator +from typing import Match, Pattern + + +def iter_prefixed_token_matches( + text: str, + candidate_re: Pattern[str], + anchored_re: Pattern[str], + token_tail_re: Pattern[str], +) -> Iterator[Match[str]]: + """Yield legacy greedy matches after testing one prefix per token. + + ``anchored_re`` must start with the same fixed prefix recognized by + ``candidate_re``. ``token_tail_re`` describes characters that the + anchored grammar can consume after that prefix. If the first prefix in + such a token cannot match, a later embedded prefix cannot match either: + its suffix was already available to the first attempt. Advancing to the + token boundary makes failed scans linear without changing successful + greedy captures. + """ + pos = 0 + while candidate := candidate_re.search(text, pos): + match = anchored_re.match(text, candidate.start()) + if match is not None: + yield match + pos = match.end() + continue + tail = token_tail_re.match(text, candidate.end()) + pos = max(candidate.end(), tail.end() if tail is not None else candidate.end()) + + +def has_prefixed_token_match( + text: str, + candidate_re: Pattern[str], + anchored_re: Pattern[str], + token_tail_re: Pattern[str], +) -> bool: + """Return whether ``iter_prefixed_token_matches`` yields a match.""" + return next( + iter_prefixed_token_matches(text, candidate_re, anchored_re, token_tail_re), + None, + ) is not None + + +_HTTP_URL_PREFIX_RE = re.compile(r"https?://", re.IGNORECASE) +_VIDEO_DETAIL_BASE_RE = re.compile( + r"\b(?:how\s+many|count|break\s*points?|timestamps?|what\s+time|score(?:board)?s?)\b", + re.IGNORECASE, +) +_VIDEO_DETAIL_EXTENDED_RE = re.compile( + r"\b(?:sequence|in\s+order|chronological|at\s+what\s+time)\b|" + r"(?:多少|几次|何时|什么时候|时间|顺序)", + re.IGNORECASE, +) +_WHEN_RE = re.compile(r"\bwhen\s+", re.IGNORECASE) +_WHEN_TARGET_RE = re.compile(r"(?:end|happen)\b", re.IGNORECASE) +_FIRST_RE = re.compile(r"\bfirst\s+", re.IGNORECASE) +_FIRST_TARGET_RE = re.compile(r"(?:save|attempt|event)\b", re.IGNORECASE) + + +def contains_search_engine_navigation(text: str) -> bool: + """Match the legacy Google/Bing/DuckDuckGo navigation URL grammar. + + The old expression backtracked through every possible optional subdomain + split. Parsing the host up to its first slash gives the same accepted + hosts and path prefixes with one pass per URL candidate. + """ + pos = 0 + while candidate := _HTTP_URL_PREFIX_RE.search(text, pos): + host_start = candidate.end() + path_start = text.find("/", host_start) + if path_start < 0: + return False + host = text[host_start:path_start].casefold() + path = text[path_start + 1:path_start + 7].casefold() + google_at = host.rfind(".google.") + recognized_host = ( + (host.startswith("google.") and len(host) > len("google.")) + or (google_at >= 0 and google_at + len(".google.") < len(host)) + or host == "bing.com" + or host.endswith(".bing.com") + or host == "duckduckgo.com" + or host.endswith(".duckduckgo.com") + ) + if recognized_host and path.startswith(("search", "sorry", "html", "lite", "?")): + return True + # A later URL may begin in the path. Resume after this scheme rather + # than skipping the whole non-whitespace region. + pos = candidate.end() + return False + + +def contains_detailed_sequence_request(text: str, *, include_first: bool = True) -> bool: + """Recognize count/order/timing requests without overlapping ``.*`` scans.""" + value = str(text or "") + if _VIDEO_DETAIL_BASE_RE.search(value): + return True + if include_first and _VIDEO_DETAIL_EXTENDED_RE.search(value): + return True + pos = 0 + while when := _WHEN_RE.search(value, pos): + line_end = value.find("\n", when.end()) + if line_end < 0: + line_end = len(value) + if _WHEN_TARGET_RE.search(value, when.end(), line_end) is not None: + return True + pos = line_end + 1 + if include_first: + pos = 0 + while first := _FIRST_RE.search(value, pos): + line_end = value.find("\n", first.end()) + if line_end < 0: + line_end = len(value) + if _FIRST_TARGET_RE.search(value, first.end(), line_end) is not None: + return True + pos = line_end + 1 + return False + + +def iter_angle_contents(text: str) -> Iterator[tuple[int, int, str]]: + """Yield nonempty flat ``<...>`` contents with monotonic delimiters.""" + pos = 0 + while (start := text.find("<", pos)) >= 0: + end = text.find(">", start + 1) + if end < 0: + return + if end > start + 1: + yield start, end + 1, text[start + 1:end] + pos = end + 1 + else: + pos = start + 1 + + +def iter_markdown_links( + text: str, + *, + target_prefix: str = "", + target_re: Pattern[str] | None = None, +) -> Iterator[tuple[int, int, str, str]]: + """Yield flat Markdown links accepted by the legacy link regexes.""" + pos = 0 + marker = "](" + target_prefix + while (start := text.find("[", pos)) >= 0: + label_end = text.find("]", start + 1) + if label_end < 0: + return + if label_end == start + 1: + pos = start + 1 + continue + if not text.startswith(marker, label_end): + pos = label_end + 1 + continue + target_start = label_end + len(marker) + target_end = text.find(")", target_start) + if target_end < 0: + return + target = text[target_start:target_end] + if target and (target_re is None or target_re.fullmatch(target)): + yield start, target_end + 1, text[start + 1:label_end], target + pos = target_end + 1 + else: + pos = label_end + 1 + + +def replace_markdown_links_with_labels(text: str) -> str: + """Linear equivalent of replacing flat Markdown links with their labels.""" + links = list(iter_markdown_links(text)) + if not links: + return text + out = [] + pos = 0 + for start, end, label, _target in links: + out.extend((text[pos:start], label)) + pos = end + out.append(text[pos:]) + return "".join(out) + + +def first_tag_content(text: str, tag: str, *, allow_attributes: bool = False) -> str | None: + """Return the first flat tag body using forward-only opener/closer scans.""" + opener_re = re.compile(r"<" + re.escape(tag), re.IGNORECASE) + closer_re = re.compile(r"", re.IGNORECASE) + pos = 0 + while opener := opener_re.search(text, pos): + name_end = opener.end() + if name_end < len(text) and text[name_end] == ">": + body_start = name_end + 1 + elif allow_attributes and name_end < len(text) and text[name_end].isspace(): + tag_end = text.find(">", name_end + 1) + if tag_end < 0: + return None + body_start = tag_end + 1 + else: + pos = name_end + continue + closer = closer_re.search(text, body_start) + if closer is None: + return None + return text[body_start:closer.start()] + return None diff --git a/src/tool_parsing.py b/src/tool_parsing.py index 5a528d6ae..c9c3bae3d 100644 --- a/src/tool_parsing.py +++ b/src/tool_parsing.py @@ -194,10 +194,33 @@ _TOOL_CODE_OPEN_RE = re.compile(r"\s*\{", re.IGNORECASE) _TOOL_CODE_CLOSE_RE = re.compile(r"\}\s*", re.IGNORECASE) # Pattern 4b: Gemma-style <|tool_call|> call:tool_name{args} -_GEMMA_TOOL_CALL_RE = re.compile( - r"<\|?tool_call\|?>\s*call:([\w\d_-]+)\s*(\{[\s\S]*?\})\s*<\|?tool_call\|?>", +_GEMMA_TOOL_CALL_OPEN_RE = re.compile( + r"<\|?tool_call\|?>\s*call:([\w\d_-]+)\s*\{", re.IGNORECASE, ) +_GEMMA_TOOL_CALL_CLOSE_RE = re.compile( + r"\}\s*<\|?tool_call\|?>", + re.IGNORECASE, +) + +# Native Qwen markup shares the same non-nesting delimiter grammar as the +# XML helpers. Literal closers keep whitespace and opener floods linear; +# values are stripped by the caller, as in the original regex path. +_QWEN_FUNCTION_OPEN_RE = re.compile(r"\s*") +_QWEN_FUNCTION_CLOSE_RE = re.compile(r"") +_QWEN_PARAMETER_OPEN_RE = re.compile(r"\s*") +_QWEN_PARAMETER_CLOSE_RE = re.compile(r"") +_QWEN_PYTHON_ARG_KEY_RE = re.compile(r"[A-Za-z_]\w*") +_QWEN_PYTHON_ARG_VALUE_RE = re.compile(r"\s*=\s*(['\"].*?['\"]|[^,]+)") +_ANGLE_TAG_OPEN_RE = re.compile(r"<") +_NONEMPTY_ANGLE_TAG_OPEN_RE = re.compile(r"<(?=[^>])") +_ANGLE_TAG_CLOSE_RE = re.compile(r">") +_EMAIL_LOCAL_RE = re.compile(r"[\w.+-]+") +_EMAIL_ADDRESS_RE = re.compile(r"[\w.+-]+@[\w.-]+\.\w+") +_ASCII_EMAIL_LOCAL_RE = re.compile(r"[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+") +_ASCII_EMAIL_ADDRESS_RE = re.compile( + r"[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}" +) # Pattern 4c: Open-function wrapper emitted by some local MLX/Exo models. # Example: @@ -542,6 +565,34 @@ _PLAIN_UI_OPEN_PANEL_RE = re.compile( r"((?:\s+(?:day|week|month|year|agenda)(?:\s+view)?(?:\s+\d{4}-\d{2}(?:-\d{2})?)?)?)" r"\s*(?:`{1,3})?\s*$" ) +_PLAIN_UI_CANDIDATE_RE = re.compile(r"ui_control", re.IGNORECASE) + + +def _iter_plain_ui_open_panel(text: str): + """Yield line-anchored UI commands without retrying every blank line.""" + pos = 0 + while candidate := _PLAIN_UI_CANDIDATE_RE.search(text, pos): + line_start = text.rfind("\n", 0, candidate.start()) + 1 + match = _PLAIN_UI_OPEN_PANEL_RE.match(text, line_start) + if match is not None: + yield match + pos = match.end() + continue + newline = text.find("\n", candidate.end()) + pos = len(text) if newline < 0 else newline + 1 + + +def _strip_plain_ui_open_panel(text: str) -> str: + matches = list(_iter_plain_ui_open_panel(text)) + if not matches: + return text + out = [] + pos = 0 + for match in matches: + out.append(text[pos:match.start()]) + pos = match.end() + out.append(text[pos:]) + return "".join(out) # --------------------------------------------------------------------------- @@ -992,6 +1043,43 @@ def _strip_raw_openai_tool_call_json(text: str) -> str: return "".join(pieces) +def iter_email_addresses(text: str, *, ascii_only: bool = False): + """Find legacy email-shaped strings without retrying every word suffix. + + Match the address only at the start of each maximal local-part token. + When that attempt fails, every suffix has the same '@'/domain boundary + and must fail too. Local/domain character runs are each scanned a bounded + number of times, including malformed text with no '@' or domain dot. + """ + local_re = _ASCII_EMAIL_LOCAL_RE if ascii_only else _EMAIL_LOCAL_RE + address_re = _ASCII_EMAIL_ADDRESS_RE if ascii_only else _EMAIL_ADDRESS_RE + pos = 0 + while token := local_re.search(text, pos): + address = address_re.match(text, token.start()) + if address is not None: + yield address.group(0) + pos = address.end() + else: + pos = token.end() + + +def _iter_qwen_python_args(raw_args: str): + """Match each maximal key once, then its value at that fixed position. + + If a word has no following equals sign, none of its suffixes can have one + either. Advancing past it avoids the old unanchored regex's O(n^2) retries + on a long malformed key, while preserving permissive quoted/bare values. + """ + pos = 0 + while key := _QWEN_PYTHON_ARG_KEY_RE.search(raw_args, pos): + value = _QWEN_PYTHON_ARG_VALUE_RE.match(raw_args, key.end()) + if value is not None: + yield key.group(0), value.group(1) + pos = value.end() + else: + pos = key.end() + + def _parse_qwen3_native_text_call( text: str, additional_tool_names: Optional[Iterable[str]] = None, @@ -1018,13 +1106,14 @@ def _parse_qwen3_native_text_call( # Qwen native text rendering: # list... - fn_match = re.search(r"\s*([\s\S]*?)\s*", text) + fn_match = next(_iter_named_blocks( + text, _QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE + ), None) if fn_match: - name, raw_body = fn_match.groups() + name, raw_body = fn_match args = {} - for key, raw_value in re.findall( - r"\s*([\s\S]*?)\s*", - raw_body, + for key, raw_value in _iter_named_blocks( + raw_body.rstrip(), _QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE, ): value = raw_value.strip() if value and value[0] in "[{\"": @@ -1085,7 +1174,7 @@ def _parse_qwen3_native_text_call( or normalized_name in declared_names ): args = {} - for key, raw_value in re.findall(r"([A-Za-z_]\w*)\s*=\s*(['\"].*?['\"]|[^,]+)", raw_args): + for key, raw_value in _iter_qwen_python_args(raw_args): try: args[key] = ast.literal_eval(raw_value.strip()) except (ValueError, SyntaxError): @@ -1509,9 +1598,9 @@ def _iter_delimited(text, open_re, close_re): pos = cm.end() -def _strip_delimited(text: str, open_re, close_re) -> str: - """Remove every ``open_re ... close_re`` span (forward-only; see - _iter_delimited). Equivalent to ``open_re([\\s\\S]*?)close_re`` ``re.sub('')`` +def _strip_delimited(text: str, open_re, close_re, replacement: str = "") -> str: + """Replace every ``open_re ... close_re`` span (forward-only; see + _iter_delimited). Equivalent to ``open_re([\\s\\S]*?)close_re`` substitution for these delimiters, without the O(n^2) rescan on unclosed openers.""" spans = list(_iter_delimited(text, open_re, close_re)) if not spans: @@ -1520,11 +1609,23 @@ def _strip_delimited(text: str, open_re, close_re) -> str: last = 0 for match_start, _inner_start, _inner_end, match_end in spans: out.append(text[last:match_start]) + out.append(replacement) last = match_end out.append(text[last:]) return "".join(out) +def strip_angle_tags(text: str, replacement: str = "", *, allow_empty: bool = False) -> str: + """Linear equivalent of replacing ``<[^>]+>`` (or ``<[^>]*>``). + + This preserves the existing flat text cleanup, including nested '<' and + malformed tails; it does not interpret HTML. Stop once no '>' is reachable + instead of retrying a suffix scan at every '<' in untrusted text. + """ + opener = _ANGLE_TAG_OPEN_RE if allow_empty else _NONEMPTY_ANGLE_TAG_OPEN_RE + return _strip_delimited(text, opener, _ANGLE_TAG_CLOSE_RE, replacement) + + def _iter_named_blocks(text, open_re, close_re): """Forward-only equivalent of ``open_re([\\s\\S]*?)close_re`` finditer where open_re captures a name in group 1: yield ``(name, body)``, pairing each @@ -1958,10 +2059,10 @@ def parse_tool_blocks( # Pattern 4b: Gemma-style <|tool_call|> blocks if not blocks: - for m in _GEMMA_TOOL_CALL_RE.finditer(text): - tool_name = m.group(1) - body = m.group(2) - block = _parse_gemma_tool_call(tool_name, body) + for tool_name, body in _iter_named_blocks( + text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE + ): + block = _parse_gemma_tool_call(tool_name, "{" + body + "}") if block: blocks.append(block) @@ -1998,7 +2099,7 @@ def parse_tool_blocks( # from weaker native-tool models after reading the tool docs but failing to # emit the actual structured call. if not blocks: - m = _PLAIN_UI_OPEN_PANEL_RE.search(text) + m = next(_iter_plain_ui_open_panel(text), None) if m: blocks.append(ToolBlock("ui_control", f"open_panel {m.group(1).lower()}{m.group(2).lower()}".strip())) @@ -2041,7 +2142,7 @@ def strip_tool_blocks( cleaned = _strip_delimited(cleaned, _XML_TOOL_CALL_OPEN_RE, _XML_TOOL_CALL_CLOSE_RE) cleaned = _XML_OPEN_TOOL_CALL_RE.sub('', cleaned) cleaned = _strip_delimited(cleaned, _TOOL_CODE_OPEN_RE, _TOOL_CODE_CLOSE_RE) - cleaned = _GEMMA_TOOL_CALL_RE.sub('', cleaned) + cleaned = _strip_delimited(cleaned, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE) cleaned = _strip_delimited(cleaned, _FUNCTION_MODEL_OPEN_RE, _FUNCTION_MODEL_CLOSE_RE) cleaned = _strip_raw_openai_tool_call_json(cleaned) declared_xml_calls = _parse_declared_direct_xml_calls( @@ -2060,7 +2161,7 @@ def strip_tool_blocks( if raw_web_json: _, (start, end) = raw_web_json cleaned = cleaned[:start] + cleaned[end:] - cleaned = _PLAIN_UI_OPEN_PANEL_RE.sub("", cleaned) + cleaned = _strip_plain_ui_open_panel(cleaned) # Strip bare blocks not wrapped in cleaned = _strip_bare_invoke_markup(cleaned) cleaned = re.sub(r'\n{3,}', '\n\n', cleaned) diff --git a/src/turn_contract.py b/src/turn_contract.py index b04c58f17..87c87c7da 100644 --- a/src/turn_contract.py +++ b/src/turn_contract.py @@ -17,6 +17,7 @@ from typing import Iterable, Mapping from src.action_intents import classify_tool_intent from src.tool_policy import ToolPolicy +from src.text_scanning import has_prefixed_token_match FAMILY_TOOLS = { @@ -47,6 +48,62 @@ FAMILY_TOOLS = { CONTRACT_CORE_TOOLS = frozenset({ "bash", "python", "read_file", "web_search", "web_fetch", "ask_user", }) + +_WORKSPACE_PREFIX_RE = re.compile(r"/workspace/", re.I) +_WORKSPACE_ARTIFACT_RE = re.compile( + r"/workspace/[^\s`\"']+\.(?:csv|html?|json|md|svg|txt)\b", re.I +) +_WORKSPACE_OUTPUT_PREFIX_RE = re.compile(r"/workspace/(?!input/)", re.I) +_WORKSPACE_OUTPUT_RE = re.compile( + r"/workspace/(?!input/)[^\s`\"']+\." + r"(?:csv|html?|json|md|svg|txt|avif|bmp|gif|jpe?g|png|webp|pdf|mp4|webm)\b", + re.I, +) +_WORKSPACE_TOKEN_TAIL_RE = re.compile(r"[^\s`\"']*") + + +def _mentions_workspace_artifact(text: str) -> bool: + return has_prefixed_token_match( + str(text or ""), + _WORKSPACE_PREFIX_RE, + _WORKSPACE_ARTIFACT_RE, + _WORKSPACE_TOKEN_TAIL_RE, + ) + + +def _mentions_workspace_output(text: str) -> bool: + return has_prefixed_token_match( + str(text or ""), + _WORKSPACE_OUTPUT_PREFIX_RE, + _WORKSPACE_OUTPUT_RE, + _WORKSPACE_TOKEN_TAIL_RE, + ) + + +def _mentions_under_budget(text: str) -> bool: + value = str(text or "") + for under in re.finditer(r"\bunder", value, re.I): + cursor = under.end() + if cursor >= len(value) or not value[cursor].isspace(): + continue + while cursor < len(value) and value[cursor].isspace(): + cursor += 1 + if cursor < len(value) and value[cursor] in "¥$€£": + cursor += 1 + while cursor < len(value) and value[cursor].isspace(): + cursor += 1 + digit_start = cursor + while cursor < len(value) and value[cursor].isdecimal(): + cursor += 1 + if cursor == digit_start: + continue + if cursor == len(value) or not (value[cursor].isalnum() or value[cursor] == "_"): + return True + if value[cursor:cursor + 3].casefold() == "yen": + cursor += 3 + if cursor == len(value) or not (value[cursor].isalnum() or value[cursor] == "_"): + return True + return False _FAMILY_WORDS = { "calendar": r"\b(?:calendar|calender|events?|appointments?|meetings?|agenda)\b", "notes": r"\b(?:notes?|checklists?|groceries|remind\s+me)\b", @@ -122,13 +179,7 @@ def _normalize_request_lead(value: str) -> str: text, flags=re.I, ) - text = re.sub( - r"^(?:never\s*mind|scratch\s+that)\s*[,;:—–-]?\s*" - r"(?=(?:open|show|list|read|search|find|check|switch|go)\b)", - "", - text, - flags=re.I, - ) + text = _strip_cancelled_request_lead(text) text = re.sub(r"^k(?:ay)?\s*[,!]?\s+(?=\S)", "", text, flags=re.I) text = re.sub( r"^(?:(?:great|nice|cool)\s*[,!.]|thanks?\s*[.!])\s+" @@ -163,6 +214,21 @@ def _normalize_request_lead(value: str) -> str: text = re.sub(r"^((?:can|could|would|will)\s+)u\b", r"\1you", text, flags=re.I) text = re.sub(r"\boffical\b", "official", text, flags=re.I) return text + + +def _strip_cancelled_request_lead(text: str) -> str: + lead = re.match(r"^(?:never\s*mind|scratch\s+that)", text, re.I) + if lead is None: + return text + cursor = lead.end() + while cursor < len(text) and text[cursor].isspace(): + cursor += 1 + if cursor < len(text) and text[cursor] in ",;:—–-": + cursor += 1 + while cursor < len(text) and text[cursor].isspace(): + cursor += 1 + action = re.match(r"(?:open|show|list|read|search|find|check|switch|go)\b", text[cursor:], re.I) + return text[cursor:] if action is not None else text _MISSPELLED_RESEARCH_ACTION = re.compile( r"^\s*" + _REQUEST_PREFIX + r"(?:reserch|reasearch|reseach)\b", re.I, @@ -195,6 +261,21 @@ _PURE_ACTION_PROHIBITION = re.compile( r"[^.;\n]*[.!?]*\s*$", re.I, ) + + +def _is_pure_action_prohibition(text: str) -> bool: + value = str(text or "").strip() + prefix = re.match( + r"(?:read[- ]only(?:\s+and)?\s+)?(?:do\s+not|don['’]?t|never)\s+" + r"(?:add|create|make|write|draft|edit|change|update|delete|remove|send|reply|" + r"run|execute|download|serve|open|save|schedule|transcribe|inspect)\b", + value, + re.I, + ) + if prefix is None: + return False + tail = value[prefix.end():].rstrip(".!?") + return not any(char in ".;\n" for char in tail) _RETURN_TO_ACTION = re.compile(r"^\s*" + _REQUEST_PREFIX + r"return\s+to\b", re.I) _PANEL_NAVIGATION = re.compile( r"^\s*" + _REQUEST_PREFIX @@ -447,6 +528,107 @@ _WARM_RECALL_WITH_FOLLOWUP = re.compile( r"(?P(?:what(?:['’]?s|\s+is)?|which|who|where|when|how|show|open|read|list|find|search)\b[\s\S]{0,180})$", re.I, ) + +_WARM_TARGET_RE = re.compile( + r"calendar|emails?|inbox|notes?|tasks?|skills?|memories|memory|" + r"documents?|docs?|web|browser|cookbook|files?|shell", + re.I, +) +_WARM_FOLLOWUP_RE = re.compile( + r"(?:what(?:['’]?s|\s+is)?|which|who|where|when|how|show|open|read|list|find|search)" + r"\b[\s\S]{0,180}\Z", + re.I, +) + + +def _consume_space(value: str, cursor: int, *, required: bool = False) -> int | None: + start = cursor + while cursor < len(value) and value[cursor].isspace(): + cursor += 1 + return None if required and cursor == start else cursor + + +def _phrase_end(value: str, cursor: int, phrase: str) -> int | None: + for index, word in enumerate(phrase.split(" ")): + if value[cursor:cursor + len(word)].casefold() != word: + return None + cursor += len(word) + if index + 1 < len(phrase.split(" ")): + cursor = _consume_space(value, cursor, required=True) + if cursor is None: + return None + return cursor + + +def _warm_recall_parts(value: str, *, with_followup: bool = False) -> tuple[str, str] | None: + value = str(value or "") + cursor = _consume_space(value, 0) or 0 + states = [cursor] + discourse = re.match(r"(?:ok(?:ay)?|and|then)", value[cursor:], re.I) + if discourse is not None: + after = _consume_space(value, cursor + discourse.end(), required=True) + if after is not None: + states.insert(0, after) + + action_phrases = ("back to", "return to", "what about", "check", "show", "open") + action_states: list[int] = [] + for state in states: + if not with_followup: + action_states.append(state) + for phrase in action_phrases: + end = _phrase_end(value, state, phrase) + if end is None: + continue + if with_followup: + end = _consume_space(value, end, required=True) + if end is None: + continue + action_states.append(end) + + for state in dict.fromkeys(action_states): + state = _consume_space(value, state) or 0 + possessive_states = [state] + for possessive in ("my", "the"): + if value[state:state + len(possessive)].casefold() == possessive: + possessive_states.insert(0, state + len(possessive)) + for target_state in possessive_states: + target_state = _consume_space(value, target_state) or 0 + target = _WARM_TARGET_RE.match(value, target_state) + if target is None: + continue + target_text = target.group(0) + suffix = target.end() + if with_followup: + if suffix < len(value) and (value[suffix].isalnum() or value[suffix] == "_"): + continue + separator_states = [] + spaced = _consume_space(value, suffix) or 0 + if spaced > suffix: + separator_states.append(spaced) + if spaced < len(value) and value[spaced] in "-—,:;": + separator_states.append(_consume_space(value, spaced + 1) or 0) + for conjunction in ("and", "then"): + end = spaced + len(conjunction) + if (value[spaced:end].casefold() == conjunction + and (end == len(value) or not (value[end].isalnum() or value[end] == "_"))): + separator_states.append(_consume_space(value, end) or 0) + for followup_start in dict.fromkeys(separator_states): + followup = _WARM_FOLLOWUP_RE.match(value, followup_start) + if followup is not None: + return target_text, followup.group(0) + continue + suffix = _consume_space(value, suffix) or 0 + suffix_states = [suffix] + for word in ("again", "now"): + if value[suffix:suffix + len(word)].casefold() == word: + suffix_states.insert(0, suffix + len(word)) + for end in suffix_states: + while end < len(value) and value[end] in ".!?": + end += 1 + end = _consume_space(value, end) or 0 + if end == len(value): + return target_text, "" + return None _REQUIRED_TOOLS = { "calendar": "manage_calendar", "notes": "manage_notes", "tasks": "manage_tasks", "skills": "manage_skills", @@ -851,11 +1033,7 @@ def selected_tools_for_request(message: str) -> frozenset[str] | None: tools = {"inspect_media"} if ( re.search(r"\b(?:create|write|save|build|produce)\b", raw_text, re.I) - and re.search( - r"(?:file://)?/workspace/[^\s`\"']+\.(?:csv|html?|json|md|svg|txt)\b", - raw_text, - re.I, - ) + and _mentions_workspace_artifact(raw_text) ): tools.update({"write_file", "read_file"}) if re.search(r"\b(?:preview|render|open)\b[^.\n]{0,100}\b(?:page|html|browser)\b", raw_text, re.I): @@ -2118,6 +2296,31 @@ _READ_PRESENTATION_SUFFIX = re.compile( ) +def _terminal_clause_match(text: str, core_pattern: str) -> re.Match[str] | None: + """Match a terminal clause after removing its ambiguous punctuation tail.""" + end = len(text) + while end and text[end - 1].isspace(): + end -= 1 + while end and text[end - 1] in ".!?": + end -= 1 + possessive = core_pattern.replace(r"\s+", r"\s++").replace(r"\s*", r"\s*+") + return re.search(possessive + r"$", text[:end], re.I) + + +def _strip_terminal_but(text: str) -> str: + end = len(text) + while end and text[end - 1].isspace(): + end -= 1 + if end < 3 or text[end - 3:end].casefold() != "but": + return text + start = end - 3 + if start == 0 or not text[start - 1].isspace(): + return text + while start and text[start - 1].isspace(): + start -= 1 + return text[:start] + + def _read_request_and_limit(message: str) -> tuple[str, int | None]: """Strip only whole, known presentation/safety suffixes, never actions.""" text = _normalize_request_lead(message) @@ -2167,11 +2370,11 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]: if keep_few_suffix: maximum = 3 text = text[:keep_few_suffix.start()].strip() - few_suffix = re.search( + few_suffix = _terminal_clause_match( + text, r"[,.;?]\s*(?:(?:only|just)\s+)?(?:(?:list|show)\s+(?:me\s+)?)?a\s+few" r"(?:\s+(?:task\s+)?(?:names?|items?|results?|entries?))?" - r"(?:\s+and\s+(?:whether|if)\s+[^.;\n]+)?[.!?]*\s*$", - text, re.I, + r"(?:\s+and\s+(?:whether|if)\s+[^.;\n]+)?", ) if few_suffix: maximum = 3 @@ -2183,42 +2386,43 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]: text, flags=re.I, ).strip() - text = re.sub( + terminal_read_only = _terminal_clause_match( + text, r"[.;]\s*read[- ]only(?:\s+(?:please|pls|plz))?\s*,?\s*" r"(?:(?:and\s+)?(?:do\s+not|don['’]?t|dont)\s+" - r"(?:change|edit|modify)(?:\s+or\s+send)?\s+(?:anything|data))?" - r"[.!?]*\s*$", - "", - text, - flags=re.I, - ).strip() + r"(?:change|edit|modify)(?:\s+or\s+send)?\s+(?:anything|data))?", + ) + if terminal_read_only: + text = text[:terminal_read_only.start()].strip() # Explanatory/safety tails do not alter a preceding exact read request. - text = re.sub( + safety_tail = _terminal_clause_match( + text, r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?(?:do\s+not|don['’]?t|dont)\s+" - r"(?:touch|change|edit|modify)(?:\s+(?:anything|data|them))?(?:\s+yet)?[.!?]*\s*$", - "", text, flags=re.I, - ).strip() + r"(?:touch|change|edit|modify)(?:\s+(?:anything|data|them))?(?:\s+yet)?", + ) + if safety_tail: + text = text[:safety_tail.start()].strip() text = re.sub( r"[.!?]\s*read[- ]only(?:\s+(?:please|pls|plz))?\s*,?\s*" r"(?:do\s+not|don['’]?t|dont)\s+(?:change|edit|modify)\s+" r"(?:or\s+send\s+)?anything[.!?]*\s*$", "", text, flags=re.I, ).strip() - text = re.sub( - r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+changes?[.!?]*\s*$", - "", text, flags=re.I, - ).strip() - text = re.sub( - r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+edits?[.!?]*\s*$", - "", text, flags=re.I, - ).strip() + for terminal_core in ( + r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+changes?", + r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+edits?", + ): + terminal_match = _terminal_clause_match(text, terminal_core) + if terminal_match: + text = text[:terminal_match.start()].strip() text = re.sub( r"[,;]\s*no\s+edits?[.!?]*\s*$", "", text, flags=re.I, ).strip() - text = re.sub( - r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+writes?[.!?]*\s*$", - "", text, flags=re.I, - ).strip() + terminal_no_write = _terminal_clause_match( + text, r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+writes?" + ) + if terminal_no_write: + text = text[:terminal_no_write.start()].strip() text = re.sub( r"[.!?]\s*(?:i['’]?m|i\s+am)\s+(?:just\s+)?checking\b[^\n]*$", "", text, flags=re.I, @@ -2236,12 +2440,11 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]: text, flags=re.I, ).strip() - text = re.sub( - r"[,;]\s*(?:keep\s+(?:them|it)\s+)?short\s+lines?\s*,?[.!?]*\s*$", - "", - text, - flags=re.I, - ).strip() + short_lines = _terminal_clause_match( + text, r"[,;]\s*(?:keep\s+(?:them|it)\s+)?short\s+lines?\s*,?" + ) + if short_lines: + text = text[:short_lines.start()].strip() text = re.sub( r"[,.;]\s*keep\s+(?:the\s+answer|it|them)\s+short[.!?]*\s*$", "", @@ -2318,7 +2521,7 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]: value = int(raw) if raw.isdecimal() else _READ_COUNT_WORDS[raw.lower()] maximum = value if maximum is None else min(maximum, value) text = text[:conversational_limit.start()].rstrip(' ,.;?') - text = re.sub(r"\s+but\s*$", "", text, flags=re.I) + text = _strip_terminal_but(text) need_limit = re.search( r"[.!?]\s*(?:i\s+)?only\s+need\s+(" + _READ_COUNT + r")" r"(?:\s+(?:short\s+)?(?:titles?|items?|results?|entries?|names?|ones?))?" @@ -3949,7 +4152,7 @@ def _families_for_tool(tool: str) -> frozenset[str]: def _clause_capabilities(text: str) -> set[str]: # A prohibition constrains authority; it must never grant the family named # only as the forbidden side effect (for example, "do not create a file"). - if _PURE_ACTION_PROHIBITION.fullmatch(text): + if _is_pure_action_prohibition(text): return set() container_tool = creation_container_tool(text) if container_tool: @@ -4610,12 +4813,7 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum raw_text, re.I, ) - and re.search( - r"(?:file://)?/workspace/(?!input/)[^\s`\"']+\." - r"(?:csv|html?|json|md|svg|txt|avif|bmp|gif|jpe?g|png|webp|pdf|mp4|webm)\b", - raw_text, - re.I, - ) + and _mentions_workspace_output(raw_text) ): families.add("shell_files") if re.search( @@ -5074,7 +5272,7 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum and re.search(r"\b(?:ones?|top|apps?|services?|providers?)\b", text, re.I) and re.search(r"\b(?:price|cheap|under|dimensions?|quote|trustworthy|app)\b", text, re.I) ) - or re.search(r"\bunder\s+[¥$€£]?\s*\d+(?:[.,]\d+)?(?:\s*yen)?\b", text, re.I) + or _mentions_under_budget(text) ): return frozenset({"search_browser"}) if recent_family == ("search_browser",) and re.fullmatch( @@ -6011,8 +6209,8 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum elif (recent == ("search_browser",) and families == {"cookbook_admin"} and re.match(r"^\s*" + _REQUEST_PREFIX + r"(?:search|find)\s+(?:those|them|these)\b", text, re.I)): families = {"search_browser"} - if not families and (recall := _WARM_RECALL.fullmatch(text)): - target = recall["target"].lower() + if not families and (recall := _warm_recall_parts(text)): + target = recall[0].lower() family = { "email": "email", "emails": "email", "inbox": "email", "note": "notes", "notes": "notes", "task": "tasks", "tasks": "tasks", @@ -6024,8 +6222,8 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum }[target] if family in recently_executed_families(history): families.add(family) - if not families and (recall := _WARM_RECALL_WITH_FOLLOWUP.fullmatch(text)): - target = recall["target"].lower() + if not families and (recall := _warm_recall_parts(text, with_followup=True)): + target = recall[0].lower() family = { "email": "email", "emails": "email", "inbox": "email", "note": "notes", "notes": "notes", "task": "tasks", "tasks": "tasks", diff --git a/tests/test_clean_agent_preview.py b/tests/test_clean_agent_preview.py index 4cb1519eb..2526562b3 100644 --- a/tests/test_clean_agent_preview.py +++ b/tests/test_clean_agent_preview.py @@ -4562,7 +4562,9 @@ async def test_preview_provider_stream_error_is_terminal_not_empty_answer(monkey disabled_tools=set(), tool_policy=ToolPolicy(), )] assert raw[-1].startswith('event: error\ndata: ') - assert 'Qwen3_5MTPDraftModel' in raw[-1] + assert 'selected model provider failed' in raw[-1] + assert 'provider_stream_error' in raw[-1] + assert 'Qwen3_5MTPDraftModel' not in raw[-1] assert all('returned no answer' not in chunk for chunk in raw) assert all('"type": "metrics"' not in chunk for chunk in raw) diff --git a/tests/test_redos_core_parsers.py b/tests/test_redos_core_parsers.py new file mode 100644 index 000000000..7d5aad96f --- /dev/null +++ b/tests/test_redos_core_parsers.py @@ -0,0 +1,859 @@ +"""Preserve legacy text-call/listing semantics and bound hostile scans.""" + +import itertools +import random +import re +import subprocess +import sys +import textwrap + +import pytest + +from src.agent_loop import ( + _calendar_listing_row, + _captures_after_first_prefix, + _contains_email_draft_headers, + _looks_like_agent_reasoning_preamble, + _looks_like_ody_qwen_leaked_tool_text, + _email_account_label, + _email_sender_name, + _contextual_summary_fragment, + _is_terse_link_request, + _is_terse_email_lookup_followup, + _looks_like_destructive_request, + _looks_like_youtube_tool_turn, + _mentions_pdf_url, + _mentions_workspace_script, + _numbered_row_parenthesized_ids, + _pipeline_request_parts, + _parse_explicit_open_panel_request, + _private_browser_product_query, + _read_only_shell_command, + _remaining_checklist_name, + _research_listing_row, + _session_link_from_row, + _session_find_query_capture, + _session_list_summary_from_tool_output, + _split_before_assistant_prompt, + _split_note_items, + _strip_trailing_done, + _strip_horizontal_space_before_lf, + _skill_listing_row, +) +from src.clean_agent_preview import ( + _page_listing_request, + _prior_web_source_request, + _terminal_source_link_clause, + _web_source_rows, + declared_workspace_artifacts, +) +from src.text_scanning import ( + contains_detailed_sequence_request, + contains_search_engine_navigation, + iter_angle_contents, + iter_markdown_links, + first_tag_content, + replace_markdown_links_with_labels, +) +from src.turn_contract import ( + _WARM_RECALL, + _WARM_RECALL_WITH_FOLLOWUP, + _is_pure_action_prohibition, + _mentions_under_budget, + _mentions_workspace_artifact, + _mentions_workspace_output, + _strip_cancelled_request_lead, + _strip_terminal_but, + _terminal_clause_match, + _warm_recall_parts, +) +from src.tool_parsing import ( + _GEMMA_TOOL_CALL_OPEN_RE, + _GEMMA_TOOL_CALL_CLOSE_RE, + _QWEN_FUNCTION_OPEN_RE, + _QWEN_FUNCTION_CLOSE_RE, + _QWEN_PARAMETER_OPEN_RE, + _QWEN_PARAMETER_CLOSE_RE, + _iter_named_blocks, + _iter_qwen_python_args, + _strip_delimited, + parse_tool_blocks, + strip_tool_blocks, + strip_angle_tags, + iter_email_addresses, +) + +_SESSION_LINK = re.compile(r"(\[(?:\\.|[^\]])+\]\(#session-[^)]+\))") +_GEMMA = re.compile( + r"<\|?tool_call\|?>\s*call:([\w\d_-]+)\s*(\{[\s\S]*?\})\s*<\|?tool_call\|?>", + re.I, +) +_QWEN_FUNCTION = re.compile(r"\s*([\s\S]*?)\s*") +_QWEN_PARAMETER = re.compile(r"\s*([\s\S]*?)\s*") +_QWEN_ARGS = re.compile(r"([A-Za-z_]\w*)\s*=\s*(['\"].*?['\"]|[^,]+)") +_WORKSPACE_SCRIPT = re.compile( + r"/workspace/[^\s`\"']+\.(?:py|pyw|sh|bash|js|mjs|ts|rb|pl)\b", re.I +) +_PDF_URL = re.compile(r"https?://\S+(?:\.pdf\b|/pdf/)", re.I) +_WORKSPACE_ARTIFACT = re.compile( + r"(?:file://)?/workspace/[^\s`\"']+\.(?:csv|html?|json|md|svg|txt)\b", re.I +) +_WORKSPACE_OUTPUT = re.compile( + r"(?:file://)?/workspace/(?!input/)[^\s`\"']+\." + r"(?:csv|html?|json|md|svg|txt|avif|bmp|gif|jpe?g|png|webp|pdf|mp4|webm)\b", + re.I, +) +_SEARCH_ENGINE_URL = re.compile( + r"https?://(?:[^/]+\.)?(?:google\.[^/]+|bing\.com|duckduckgo\.com)" + r"/(?:search|sorry|html|lite|\?)", + re.I, +) +_LEAKED_TOOL_TEXT = re.compile( + r"(<\s*/?\s*(?:function|parameter|tool_call)\b|(?:^|\n)\s*(?:function|parameter)\s*=" + r"|\bmanage_(?:notes|calendar|memory|documents|contact)\s*\(|\"function\"\s*:\s*\"(?:manage_|mcp__)" + r"|mcp__email__|(?:^|\n)\s*(?:web_search|web_fetch|private_browser)\s*:)", re.I, +) +_VIDEO_DETAIL_BASE = re.compile( + r"\b(?:how\s+many|count|break\s*points?|timestamps?|what\s+time|" + r"when\s+.*(?:end|happen)|score(?:board)?s?)\b", re.I, +) +_VIDEO_DETAIL_EXTENDED = re.compile( + r"\b(?:how\s+many|count|sequence|in\s+order|chronological|timestamps?|" + r"what\s+time|at\s+what\s+time|when\s+.*(?:end|happen)|" + r"first\s+.*(?:save|attempt|event)|score(?:board)?s?)\b|" + r"(?:多少|几次|何时|什么时候|时间|顺序)", re.I, +) + + +def test_session_link_matches_legacy_escape_and_greedy_semantics(): + cases = [ + "no links", "[](#session-id)", "[x](#session-)", + "[x](#session-id)", "[[x](#session-id)", + r"[x\](#session-id)", r"[x\]more](#session-id)", + r"[x\](#session-first)\](#session-last)", + r"[x\](#session-first)\](#session-)", + r"[x\](#session-first)\](#session-unclosed", + "[bad] then [good](#session-id)", + "[x](#session-id[has]brackets)", + "[x](#session-first) [y](#session-second)", + ] + rng = random.Random(6503) + tokens = ["[", "]", "\\", "a", "(", ")", "(#session-id)", "(#session-)"] + cases += ["".join(rng.choices(tokens, k=12)) for _ in range(2000)] + cases += ["[" + "".join(label) + "](#session-id)" + for n in range(5) for label in itertools.product("a[]\\", repeat=n)] + for row in cases: + expected = _SESSION_LINK.search(row) + assert _session_link_from_row(row) == (expected.group(0) if expected else ""), row + + +@pytest.mark.parametrize("row,expected", [ + ("- **[Chat](#session-id)** (id: `id`, model: qwen, 2 msgs, last active today)", + "- [Chat](#session-id) (last active today)"), + ("- [Chat](#session-id) (irrelevant) (model: qwen) (last active later)", + "- [Chat](#session-id)"), + ("- [Chat](#session-id) (outer (last active today))", + "- [Chat](#session-id) (last active today)"), + ("- plain row", "- plain row"), +]) +def test_session_summary_preserves_link_and_metadata(row, expected): + assert _session_list_summary_from_tool_output("Chats:\n" + row) == "Chats:\n" + expected + + +def test_gemma_delimiters_match_legacy_parse_and_strip(): + cases = [ + "ordinary prose", "<|tool_call|>call:web_search{query: 'news'}<|tool_call|>", + "before call:read-file {\npath: 'README.md'\n} after", + "call:x{a}call:y{b}", + "call:x{call:y{b}", + "call:x{unclosed", "}call:x{unclosed", + ] + for text in cases: + actual = [(name, "{" + body + "}") for name, body in _iter_named_blocks( + text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE + )] + assert actual == _GEMMA.findall(text) + assert _strip_delimited(text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE) == _GEMMA.sub("", text) + raw = '<|tool_call|>call:web_search{"query":"news"}<|tool_call|>' + assert [(b.tool_type, b.content) for b in parse_tool_blocks(raw)] == [("web_search", "news")] + assert strip_tool_blocks(raw) == "" + + +@pytest.mark.parametrize("reference,opener,closer,tokens", [ + (_QWEN_FUNCTION, _QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE, + ["", "", "a", "\n", "\t", " "]), + (_QWEN_PARAMETER, _QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE, + ["", "", "a", "\n", "\t", " "]), +]) +def test_qwen_delimiters_preserve_names_and_stripped_values(reference, opener, closer, tokens): + rng = random.Random(6503) + for _ in range(1000): + text = "".join(rng.choices(tokens, k=16)) + expected = [(name, body.strip()) for name, body in reference.findall(text)] + actual = [(name, body.strip()) for name, body in _iter_named_blocks(text, opener, closer)] + assert actual == expected + + +def test_qwen_python_arguments_preserve_permissive_legacy_grammar(): + cases = ["action='list', limit=5", "action = \"list\"", "1key=2", "aé=4", + "key='mismatched\"", "broken word, okay=2", "a=\t, b=3", "a=\n'hi'", "a='hi\nthere'"] + rng = random.Random(6503) + tokens = ["key", "1", "中", "é", "=", "\n", " ", "\t", "'", '"', ",", "-", "_", "[]"] + cases += ["".join(rng.choices(tokens, k=20)) for _ in range(2000)] + for text in cases: + assert list(_iter_qwen_python_args(text)) == _QWEN_ARGS.findall(text), text + + +@pytest.mark.parametrize("text", [ + "Done.", "Done.\nUpdated the document.\nDone.", "Done. undone.", + "Done.\nDONE.\t", "Done.\u2003Done.\u2003", "Done. no terminal marker ", +]) +def test_trailing_done_matches_legacy_cleanup(text): + pattern = re.compile(r"\s*Done\.\s*$", re.I) + expected = pattern.sub("", text).rstrip() if pattern.search(text) else text + assert _strip_trailing_done(text) == expected + + +@pytest.mark.parametrize("allow_empty", [False, True]) +@pytest.mark.parametrize("replacement", ["", " "]) +def test_angle_tag_cleanup_matches_legacy_flat_grammar(allow_empty, replacement): + pattern = re.compile(r"<[^>]*>" if allow_empty else r"<[^>]+>") + cases = ["beforeboldafter", "<>", "<<>>", "ab", "one", "a > b", "tail"] + cases += ["".join(parts) for n in range(6) + for parts in itertools.product("a<>\n", repeat=n)] + for text in cases: + assert strip_angle_tags(text, replacement, allow_empty=allow_empty) == pattern.sub(replacement, text) + + +@pytest.mark.parametrize("ascii_only", [False, True]) +def test_email_scanner_preserves_legacy_address_sets(ascii_only): + pattern = re.compile( + r"[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}" + if ascii_only else r"[\w.+-]+@[\w.-]+\.\w+" + ) + cases = ["a@b.example", "a+b@b.example, c@d.example", "a@b@c.example", + "a@b.c+d@e.f", "你好@例子.中国", "a@b.-.com", "'a'@example.com", + "a@b.x", "a@.com", "a@b...", "a@b.c@d.example"] + rng = random.Random(6503) + tokens = ["a", "b", "é", "中", "@", ".", "-", "+", " ", "_", "'", "!", "/"] + cases += ["".join(rng.choices(tokens, k=30)) for _ in range(3000)] + for text in cases: + assert list(iter_email_addresses(text, ascii_only=ascii_only)) == pattern.findall(text), text + + +def test_prefixed_workspace_and_url_detectors_match_legacy_searches(): + cases = [ + "", "/workspace/a.py", "FILE:///WORKSPACE/report.HTML", + "/workspace/input/a.png", "/workspace/input/a.png/workspace/out.png", + "http://example.test/a.pdf", "http://x/http://example.test/pdf/view", + "https://google.com/search?q=x", "http://news.google.co.uk/sorry/index", + "https://x.bing.com/html", "http://bad/http://duckduckgo.com/?q=x", + ] + rng = random.Random(6503) + tokens = ["/workspace/", "input/", "file://", "http://", "https://", ".py", + ".pdf", ".html", "/pdf/", "google.", "bing.com/", "search", "x", " ", "'", "/"] + cases += ["".join(rng.choices(tokens, k=18)) for _ in range(4000)] + for text in cases: + assert _mentions_workspace_script(text) == bool(_WORKSPACE_SCRIPT.search(text)), text + assert _mentions_pdf_url(text) == bool(_PDF_URL.search(text)), text + assert _mentions_workspace_artifact(text) == bool(_WORKSPACE_ARTIFACT.search(text)), text + assert _mentions_workspace_output(text) == bool(_WORKSPACE_OUTPUT.search(text)), text + assert contains_search_engine_navigation(text) == bool(_SEARCH_ENGINE_URL.search(text)), text + + +def test_declared_workspace_artifacts_keeps_legacy_greedy_paths(): + pattern = re.compile(r"/workspace/[^\s,,、;;`\"'<>]+\.[A-Za-z0-9]{1,12}", re.I) + cases = [ + "create /workspace/report.csv", + "read /workspace/input.csv and create /workspace/out.json", + "create /workspace/a.csv/workspace/b.json", + "create /workspace/noext /workspace/out.txt", + ] + for text in cases: + expected = [] + for match in pattern.finditer(text): + path = match.group(0).rstrip(".!?))]}") + if path.startswith("/workspace/fixtures/") or path in expected: + continue + before = text[max(0, match.start() - 240):match.start()] + before = pattern.sub("[workspace file]", before) + clause = re.split(r"[.;!?\n]", before)[-1] + if re.search( + r"\b(?:from|using|inspect|read|open|analy[sz]e|transcribe|extract\s+(?:text\s+)?from|" + r"input(?:\s+file)?(?:\s+is)?|source(?:\s+file)?(?:\s+is)?)\s*(?::|=)?\s*$", + clause, re.I, + ) or re.search(r"\b(?:read_file|inspect_media|extract_text|transcribe_media|pdf_extract)\b", clause, re.I): + continue + if not re.search( + r"\b(?:create|write|save|export|render|generate|produce|output|deliver|store|convert|make)\b|" + r"\b(?:write_file|output_path)\b", clause, re.I, + ): + continue + expected.append(path) + assert declared_workspace_artifacts(text) == tuple(expected) + + +def test_intent_and_detail_detectors_preserve_short_legacy_language(): + cases = [ + "plain answer", "\n\nfunction = manage_notes", "mcp__email__read_email", + "< function >", "web_search: cats", "prefix\n private_browser : url", + "when does it end", "when\nwill it end", "first save event", "sequence please", + "I can now carefully inspect the result.", "Answer. Now let me verify this.", + "接下来我会查看结果", "。 让我先检查", + ] + rng = random.Random(6503) + tokens = ["\n", " ", ".", "!", "function", "parameter", "=", "web_search", ":", + "when", "first", "end", "event", "let me", "inspect", "x"] + cases += ["".join(rng.choices(tokens, k=24)) for _ in range(3000)] + for text in cases: + assert _looks_like_ody_qwen_leaked_tool_text(text) == bool(_LEAKED_TOOL_TEXT.search(text)), text + assert contains_detailed_sequence_request(text, include_first=False) == bool(_VIDEO_DETAIL_BASE.search(text)), text + assert contains_detailed_sequence_request(text) == bool(_VIDEO_DETAIL_EXTENDED.search(text)), text + + for text in ( + "The result is partial. Now let me inspect the rest.", + "\n" * 20 + "let me check the source", + "。\n 接下来我会查看结果", + "A complete factual answer.", + ): + assert isinstance(_looks_like_agent_reasoning_preamble(text), bool) + + +def test_suffix_and_split_helpers_preserve_legacy_results(): + panel_cases = [ + ("open calendar again!!!", ("ui_control", "open_panel calendar")), + ("open calendar" + " " * 50 + "again?", ("ui_control", "open_panel calendar")), + ("open calendar....", ("ui_control", "open_panel calendar")), + ("open calendar again x", ("ui_control", "open_panel calendar")), + ("open notes day view again.", ("ui_control", "open_panel notes")), + ] + for text, expected in panel_cases: + assert _parse_explicit_open_panel_request(text) == expected + + values = ["one, two and three", " one ,two and three ", "candy and x", "a,and,b", ""] + for value in values: + expected_parts = [ + re.sub(r"\s+", " ", part).strip(" .") + for part in re.split(r"\s*,\s*|\s+\band\b\s+", value) + ] + expected = [{"text": part, "done": False} for part in expected_parts if part] + assert _split_note_items(value) == expected + + for value in ["Alice ", "Alice <", "Alice <>", + "Alice (a@example.com)", "Alice ((a@example.com)", "Alice > x "]: + normalized = re.sub(r"\s+", " ", value).strip() + expected_sender = re.sub(r"\s*\([^)]*@[^)]*\)\s*$", "", normalized).strip() + expected_sender = re.sub(r"\s*<[^>]*>\s*$", "", expected_sender).strip() + expected_sender = expected_sender or value.strip() + expected_account = re.sub(r"\s*<[^>]+>\s*$", "", normalized).strip() or normalized + assert _email_sender_name(value) == expected_sender + assert _email_account_label(value) == expected_account + + for value in ["Reply body\nWant me to send it?", "Reply\n\n SHOULD I continue?", "Want me now", "x\nnot a prompt"]: + expected = re.split(r"\n\s*(?:Want me|Would you like|Should I)\b", value, flags=re.I, maxsplit=1)[0] + assert _split_before_assistant_prompt(value) == expected + + for value in ["a \n b\t\n", " x", "a \r\n", "\t\n\n", ""]: + assert _strip_horizontal_space_before_lf(value) == re.sub(r"[ \t]+\n", "\n", value) + + for value in ["pwd && ls", "cat x | grep y", "echo x; printf y", "pwd " + " " * 20 + "&& ls", "pwd || rm x"]: + legacy_parts = re.split(r"\s*(?:&&|\|\||;|\|)\s*", value.strip()) + allowed = re.compile( + r"^(?:pwd|ls|find|rg|grep|git\s+(?:status|diff|log|show|branch)|sed(?!\s+-i\b)|" + r"head|tail|cat|stat|file|wc|sort|uniq|cut|ip|ipconfig|getent|nslookup|dig|arp|" + r"hostname|uname|whoami|echo|printf|test|true|false|:)\b", re.I, + ) + expected = bool(legacy_parts) and all(part.strip() for part in legacy_parts) and all( + allowed.match(part.strip()) for part in legacy_parts + ) + assert _read_only_shell_command(value) == bool(expected) + + +@pytest.mark.parametrize("prefix,target_pattern", [ + ("#session-", r"[^)]+"), + ("#note-", r"[^)]+"), + ("#event-", r"[0-9a-fA-F-]{8,64}"), + ("", r"[^)]+"), +]) +def test_forward_markdown_and_angle_scanners_match_legacy(prefix, target_pattern): + reference = re.compile(r"\[([^\]]+)\]\(" + re.escape(prefix) + "(" + target_pattern + r")\)") + validator = None if target_pattern == r"[^)]+" else re.compile(target_pattern) + cases = ["[title](#session-id)", "[[title](#session-id)", "[](x)", + "[a](x)[b](y)", "[a](#event-12345678)", "[a](broken"] + rng = random.Random(6503) + tokens = ["[", "]", "(", ")", "#session-", "#note-", "#event-", "a", "1", "-", " "] + cases += ["".join(rng.choices(tokens, k=28)) for _ in range(3000)] + for text in cases: + expected = reference.findall(text) + actual = [(label, target) for _start, _end, label, target in iter_markdown_links( + text, target_prefix=prefix, target_re=validator + )] + assert actual == expected, text + + generic = re.compile(r"\[([^\]]+)\]\([^)]+\)") + for text in cases: + assert replace_markdown_links_with_labels(text) == generic.sub(r"\1", text), text + + angle = re.compile(r"<([^>]+)>") + for text in cases + ["", "<", "<>", "", "<<<"]: + assert [content for _start, _end, content in iter_angle_contents(text)] == angle.findall(text) + + +def test_numbered_row_identifier_scan_matches_legacy_rows(): + pattern = re.compile(r"^\s*\d+\.\s+.+?\s+\(([^)\n]+)\)\s+[—-]", re.M) + cases = [ + "1. Task (abc) — due", " 2. Label ((nested) - tail", "1. no id", + "1. a (bad) x (good) — tail", "\n\n3. item (id-3) - tail", + ] + rng = random.Random(6503) + tokens = ["1. ", "x", " ", "(", ")", " -", " —", "\n"] + cases += ["".join(rng.choices(tokens, k=20)) for _ in range(2000)] + for text in cases: + assert _numbered_row_parenthesized_ids(text) == pattern.findall(text), text + + +def test_model_and_request_extractors_match_legacy_grammars(): + youtube = re.compile( + r"\b(?:youtube|youtu\.be|yt|video\s+comments?|comments?\s+on\s+(?:the\s+)?video|" + r"transcript\s+(?:of|for)|(?:latest|newest|recent)\s+(?:\d+\s+)?(?:videos?|uploads?)|" + r"official\s+.+\s+channel)\b", re.I, + ) + destructive = re.compile(r"\b(delete|remove|archive|trash|send|reply|unsubscribe|mark\s+.*read)\b", re.I) + terse = re.compile( + r"\s*(?:(?:send|sned|share|give|show)?\s*(?:me\s+)?(?:the\s+)?(?:links?|urls?|sources?)" + r"(?:\s+(?:for|to|from)\s+(?:those|that|them|these|it|this|the\s+(?:sites?|websites?|resources?|sources?)))?" + r"|(?:for|to|from)\s+(?:those|that|them|these|it|this|the\s+(?:sites?|websites?|resources?|sources?)))" + r"\s*(?:please|pls)?[.!?]?\s*", + ) + cases = ["official project channel", "official x channel", "mark all as read", "mark\nread", + " send me the links please ", "for those sites", "plain text"] + rng = random.Random(6503) + tokens = ["official", "channel", "mark", "read", "send", "links", "for", "those", "sites", " ", "\n", "x"] + cases += ["".join(rng.choices(tokens, k=24)) for _ in range(3000)] + for text in cases: + assert _looks_like_youtube_tool_turn(text) == bool(youtube.search(text)), text + assert _looks_like_destructive_request(text) == bool(destructive.search(text)), text + assert _is_terse_link_request(text) == bool(terse.fullmatch(text.lower())), text + + summary_re = re.compile( + r"\bsummary\s*(?:\*\*)?\s*:?\s*(.+?)(?:\n\s*(?:-|\\*\\*|If you|Want me|This is|$))", + re.I | re.S, + ) + summary_cases = ["Summary: useful\n- next", "summary **: x\n\nWant me to continue", "summary: no end"] + for text in summary_cases: + match = summary_re.search(text) + assert _contextual_summary_fragment(text) == (match.group(1) if match else "") + + product_re = re.compile( + r"\b(?:find|look\s+for|shop\s+for|search\s+for)\s+(?:me\s+)?(?:the\s+)?(?:best\s+)?" + r"(?P.+?)\s*[?.!]*$", re.I, + ) + for text in ["find best camera???", "look for me the best shoes", "shop for ???", "plain"]: + normalized = re.sub(r"\s+", " ", text).strip() + match = product_re.search(normalized) + expected = "" + if match: + expected = match.group("query").strip(" \t\r\n.,!?;:") + expected = re.sub( + r"\s+(?:on|at|from)\s+(?:the\s+)?[A-Za-z0-9&.' -]{1,60}$", "", expected, flags=re.I + ).strip() + expected = expected[:120] if 0 < len(expected.split()) <= 12 else "" + assert _private_browser_product_query(text) == expected + + title_re = re.compile(r"]*)?>([\s\S]*?)", re.I) + for text in ["x", "a<b", "x"]: + match = title_re.search(text) + assert first_tag_content(text, "title", allow_attributes=True) == (match.group(1) if match else None) + + +def test_summary_capture_preserves_exact_head_whitespace_and_newline_grammar(): + legacy = re.compile( + r"\bsummary\s*(?:\*\*)?\s*:?\s*(.+?)(?:\n\s*(?:-|\\*\\*|If you|Want me|This is|$))", + re.I | re.S, + ) + cases = [ + "Summary: useful\nordinary next line\n- next", + "Summary: useful\nx\n", "summary\n\n", "summary \n", + "summary: \n", "summary **:\n\n", "summary:\nX\n", + "summary summary: X\n", "summary: no newline", + ] + rng = random.Random(6509) + tokens = ["summary", "Summary", ":", "**", "\\", " ", "\t", "\n", "x", "-", "If you"] + cases += ["".join(rng.choices(tokens, k=24)) for _ in range(5000)] + for text in cases: + match = legacy.search(text) + assert _contextual_summary_fragment(text) == (match.group(1) if match else ""), text + + +def test_repeated_listing_words_preserve_legacy_request_grammar(): + legacy = re.compile( + r"\s*(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)\s+" + r"(?:[\w .:/-]+\s+)?(?:stories|articles|posts|headlines|pages)" + r"(?:\s+on\s+[\w .:/-]+)?[.!?]?\s*", re.I, + ) + for text in ("top Straße stories", "top İ stories", "lİst stories", "top storİes", + "top café stories", "top stories on Straße", "top stories on İ"): + assert _page_listing_request(text) == bool(legacy.fullmatch(text)), text + for word in ("stories", "articles", "posts", "headlines", "pages"): + for count in (1, 2, 8, 32): + for suffix in ("X", "@", "\tX", "on x", "on \t", "on ", "\n"): + for separator in (" ", " on ", "\t", " \t"): + text = "top " + (word + separator) * count + suffix + assert _page_listing_request(text) == bool(legacy.fullmatch(text)), text + + +def test_repeated_navigation_schemes_preserve_legacy_url_search(): + for count in (1, 2, 8, 32): + for suffix in ("X", "google.com/search", "bing.com/html", "duckduckgo.com/?q=x"): + text = "http://" * count + suffix + assert contains_search_engine_navigation(text) == bool(_SEARCH_ENGINE_URL.search(text)), text + + +def test_tool_listing_rows_match_legacy_grammars(): + research = re.compile(r"^-\s+\[(.*?)\]\(#research-([^)]+)\)(.*)$") + skills = re.compile(r"^-\s+\*\*(.*?)\*\*(?:\s+\((.*?)\)|\s+\[(draft)\])?(?::\s*(.*))?$") + calendar = re.compile(r"^\s*-\s+(.+?):\s+\[(.*?)\]\(#event-([^)]+)\)(.*)$") + draft = re.compile(r"\bTo:\s*.+\bSubject:\s*.+\n---", re.I | re.S) + cases = [ + "- [Title](#research-id) tail", + "- **name** (published): description", + "- **name** [draft]", + " - when: [title](#event-id) tail", + "To: a\nSubject: b\n---", + "To:Subject:x\n---", + "plain", + ] + rng = random.Random(6504) + tokens = ["-", " ", "\t", "[", "]", "(", ")", "*", ":", "#research-", "#event-", "draft", "x"] + cases += ["".join(rng.choices(tokens, k=28)) for _ in range(5000)] + for text in cases: + match = research.match(text) + assert _research_listing_row(text) == (match.groups() if match else None), text + match = skills.match(text) + assert _skill_listing_row(text) == (match.groups() if match else None), text + match = calendar.match(text) + assert _calendar_listing_row(text) == (match.groups() if match else None), text + assert _contains_email_draft_headers(text) == bool(draft.search(text)), text + + +def test_terse_email_followup_matches_legacy_grammar(): + legacy = re.compile( + r"^\s*(?:and|so|well|still|then|okay|ok|did you find it(?: yet)?|what did you find)\s*[?.!]*\s*$", + re.I, + ) + rng = random.Random(6508) + tokens = ["and", "so", "well", "did you find it", " yet", "what did you find", " ", "\t", "?", ".", "!", "x"] + cases = ["and?", " did you find it yet ! ", "what did you find", "and X"] + cases += ["".join(rng.choices(tokens, k=20)) for _ in range(3000)] + for text in cases: + assert _is_terse_email_lookup_followup(text) == bool(legacy.match(text)), text + + +def test_preview_request_and_source_scans_match_legacy_grammars(): + page = re.compile( + r"\s*(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)\s+" + r"(?:[\w .:/-]+\s+)?(?:stories|articles|posts|headlines|pages)" + r"(?:\s+on\s+[\w .:/-]+)?[.!?]?\s*", re.I, + ) + prior = re.compile( + r"\s*(?:(?:where|what)\s+did\s+you\s+(?:get|find)\s+(?:that|this)\s+from[?., ]*" + r"(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link[.!? ]*" + r"|(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link(?:\s+for\s+that)?[.!? ]*" + r"|what(?:['’]?s|\s+is)\s+(?:the\s+)?source(?:\s+link)?[.!? ]*)\s*", re.I, + ) + terminal = re.compile( + r"(?:^|[.!?;,\n])\s*(?:(?:pls|please)\s+)?(?:sources?|citations?|links?)" + r"\s*(?:pls|please)?\s*[.!?]*$", re.I, + ) + rows = re.compile(r"^\[\d+\]\s+(.+?)\s*\n\s*(https?://\S+)", re.M) + cases = [ + "top stories", "show me the latest stories on example.com", "give me the source link", + "what's the source?", "x. please links pls!!", "[1] Title\nhttps://example.test/x", "plain", + ] + rng = random.Random(6505) + tokens = ["top", "show", " me", " the", " stories", " on", "link", "source", "please", " ", "\t", ".", "!", "\n", "[1]", "http://x"] + cases += ["".join(rng.choices(tokens, k=22)) for _ in range(5000)] + for text in cases: + assert _page_listing_request(text) == bool(page.fullmatch(text)), text + assert _prior_web_source_request(text) == bool(prior.fullmatch(text)), text + assert _terminal_source_link_clause(text) == bool(terminal.search(text)), text + assert _web_source_rows(text) == rows.findall(text), text + + +def test_staged_command_payload_parsers_match_legacy_grammars(): + specs = [ + ( + re.compile(r"\bnote\s+titled\s+(.+?)\s+so\s+its\s+content\s+is\s+['\"]([^'\"]+)['\"]", re.I), + r"\bnote\s+titled(?=\s)", + r"\s+(.+?)\s+so\s+its\s+content\s+is\s+['\"]([^'\"]+)['\"]", + ), + ( + re.compile(r"\b(?:delete|trash|remove|archive|mark(?:\s+as)?\s+(?:read|unread)|mark\s+(?:read|unread))\b\s+(?:all|every|the)?\s*(?:my\s+)?(.+?)\s+(?:emails?|mail|messages?)\b", re.I), + r"\b(?:delete|trash|remove|archive|mark(?:\s+as)?\s+(?:read|unread)|mark\s+(?:read|unread))\b(?=\s)", + r"\s+(?:all|every|the)?\s*(?:my\s+)?(.+?)\s+(?:emails?|mail|messages?)\b", + ), + ( + re.compile(r"\b(?:make|create|add)\s+(?:a\s+)?checklist\s+(?:called|titled|named)\s+(.+?)\s+with\s+(.+?)\s*$", re.I), + r"\b(?:make|create|add)\s+(?:a\s+)?checklist\s+(?:called|titled|named)(?=\s)", + r"\s+(.+?)\s+with\s+(.+?)$", + ), + ( + re.compile(r"\b(?:change|update|set|retag)\b\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b", re.I), + r"\b(?:change|update|set|retag)\b(?=\s)", + r"\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b", + ), + ( + re.compile(r"\b(?:chang(?:e|es|ed|ing)|updat(?:e|es|ed|ing))\s+(.+?)\s+to\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)", re.I), + r"\b(?:chang(?:e|es|ed|ing)|updat(?:e|es|ed|ing))(?=\s)", + r"\s+(.+?)\s+to\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)", + ), + ( + re.compile(r"\breplace\s+(.+?)\s+with\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)", re.I), + r"\breplace(?=\s)", + r"\s+(.+?)\s+with\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)", + ), + ] + pipeline = re.compile( + r"\bpipeline\s+using\s+([^\s,]+)\s+to\s+(.+?),\s*then\s+([^\s,]+)\s+to\s+(.+?)(?:[.!?]\s*)?$", + re.I, + ) + session = re.compile(r"\b(?:find|search(?:\s+for)?|show)\s+(?:the\s+)?(.+?)\s+(?:chat|session|conversation)\b", re.I) + checklist = re.compile(r"\b(?:what(?:'s| is)?|show|tell\s+me)\b.*?\b(?:left|remaining)\b.*?\b(?:on|in)\s+(?:the\s+)?(.+?)\s+checklist\b", re.I) + rng = random.Random(6506) + tokens = ["note", " titled", " so its content is ", "'x'", "delete", " all", " emails", "create checklist called", " with", "change", " tag to ", "replace", " to", " pipeline using ", " then", "find", " chat", "left", " on", " checklist", " ", "\t", ".", "x"] + cases = ["update note titled a so its content is 'b'", "delete all all emails", "create checklist called a with b", "change the trip tag to work", "replace old with new", "pipeline using a to x, then b to y", "find the chat", "what is left on the trip checklist"] + cases += ["".join(rng.choices(tokens, k=25)).strip() for _ in range(5000)] + for text in cases: + for legacy, prefix, remainder in specs: + match = legacy.search(text) + assert _captures_after_first_prefix(text, prefix, remainder) == (match.groups() if match else None), (legacy.pattern, text) + match = pipeline.search(text) + assert _pipeline_request_parts(text) == (match.groups() if match else None), text + match = session.search(text) + assert _session_find_query_capture(text) == (match.group(1) if match else None), text + match = checklist.search(text) + assert _remaining_checklist_name(text) == (match.group(1) if match else None), text + + +def test_turn_contract_scans_match_legacy_grammars(): + cancelled = re.compile( + r"^(?:never\s*mind|scratch\s+that)\s*[,;:—–-]?\s*(?=(?:open|show|list|read|search|find|check|switch|go)\b)", + re.I, + ) + prohibition = re.compile( + r"^\s*(?:read[- ]only(?:\s+and)?\s+)?(?:do\s+not|don['’]?t|never)\s+" + r"(?:add|create|make|write|draft|edit|change|update|delete|remove|send|reply|run|execute|download|serve|open|save|schedule|transcribe|inspect)\b" + r"[^.;\n]*[.!?]*\s*$", re.I, + ) + budget = re.compile(r"\bunder\s+[¥$€£]?\s*\d+(?:[.,]\d+)?(?:\s*yen)?\b", re.I) + terminal_specs = [ + r"[,.;?]\s*(?:(?:only|just)\s+)?(?:(?:list|show)\s+(?:me\s+)?)?a\s+few(?:\s+(?:task\s+)?(?:names?|items?|results?|entries?))?(?:\s+and\s+(?:whether|if)\s+[^.;\n]+)?", + r"[.;]\s*read[- ]only(?:\s+(?:please|pls|plz))?\s*,?\s*(?:(?:and\s+)?(?:do\s+not|don['’]?t|dont)\s+(?:change|edit|modify)(?:\s+or\s+send)?\s+(?:anything|data))?", + r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?(?:do\s+not|don['’]?t|dont)\s+(?:touch|change|edit|modify)(?:\s+(?:anything|data|them))?(?:\s+yet)?", + r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+changes?", + r"[,;]\s*(?:keep\s+(?:them|it)\s+)?short\s+lines?\s*,?", + ] + rng = random.Random(6507) + tokens = ["never", " mind", "scratch", " that", "open", "do not", " touch", " anything", "no changes", "read-only", "a few", " and whether", "short lines", "under", "$", "123", "yen", "open my calendar", "and", " what is that", " ", "\t", ".", "!", ",", ";", "x"] + cases = ["never mind, open calendar", "do not edit anything.", "under $ 20 yen", "open my calendar", "open calendar and what is next", "x but "] + cases += ["".join(rng.choices(tokens, k=22)) for _ in range(5000)] + for text in cases: + assert _strip_cancelled_request_lead(text) == cancelled.sub("", text), text + assert _is_pure_action_prohibition(text) == bool(prohibition.fullmatch(text)), text + assert _mentions_under_budget(text) == bool(budget.search(text)), text + assert _strip_terminal_but(text) == re.sub(r"\s+but\s*$", "", text, flags=re.I), text + for core in terminal_specs: + legacy = re.search(core + r"[.!?]*\s*$", text, re.I) + current = _terminal_clause_match(text, core) + assert (current.start() if current else None) == (legacy.start() if legacy else None), (core, text) + match = _WARM_RECALL.fullmatch(text) + assert _warm_recall_parts(text) == ((match.group("target"), "") if match else None), text + match = _WARM_RECALL_WITH_FOLLOWUP.fullmatch(text) + assert _warm_recall_parts(text, with_followup=True) == ( + (match.group("target"), match.group("followup")) if match else None + ), text + + +@pytest.mark.parametrize("program", [ + r''' +from src.agent_loop import _contextual_summary_fragment +assert _contextual_summary_fragment("Summary: x\n" + "x\n" * 100_000) == "x" +assert _contextual_summary_fragment("Summary:" + "\n" * 100_000) == "\n" +''', + r''' +from src.clean_agent_preview import _page_listing_request +for text in ("top " + "stories " * 30_000 + "X", + "top " + "stories on " * 30_000 + "@"): + assert not _page_listing_request(text) +''', + r''' +from src.text_scanning import contains_search_engine_navigation +assert not contains_search_engine_navigation("http://" * 100_000 + "X") +assert contains_search_engine_navigation("http://" * 100_000 + "bing.com/search") +''', + r''' +from src.agent_loop import _is_terse_email_lookup_followup +_is_terse_email_lookup_followup("and" + " " * 100_000 + "X") +''', + r''' +from src.turn_contract import (_is_pure_action_prohibition, _mentions_under_budget, + _strip_cancelled_request_lead, _terminal_clause_match, _warm_recall_parts) +space = " " * 100_000 +_strip_cancelled_request_lead("never" + space + "mind" + space + "X") +_terminal_clause_match(", a few and whether " + space + "X;", r"[,.;?]\s*a\s+few(?:\s+and\s+whether\s+[^.;\n]+)?") +_is_pure_action_prohibition("do not add " + space + "X;") +_mentions_under_budget("under" + space + "$" + space + "X") +_warm_recall_parts("open" + space + "my" + space + "calendar" + space + "X") +_warm_recall_parts("open" + space + "my" + space + "calendar" + space + "and" + space + "what " + "x" * 181, with_followup=True) +''', + r''' +from src.agent_loop import (_captures_after_first_prefix, _pipeline_request_parts, + _remaining_checklist_name, _session_find_query_capture) +evil = ("delete all x " * 30_000) + "z" +_captures_after_first_prefix(evil, r"\bdelete\b(?=\s)", r"\s+(?:all)?\s*(.+?)\s+emails?\b") +_pipeline_request_parts(("pipeline using m to x " * 30_000) + "z") +_session_find_query_capture(("find x " * 30_000) + "z") +_remaining_checklist_name("what " + ("left on " * 30_000) + "z") +''', + r''' +from src.clean_agent_preview import (_page_listing_request, _prior_web_source_request, + _terminal_source_link_clause, _web_source_rows) +_page_listing_request("top stories on " + " " * 100_000 + "\nX") +_prior_web_source_request("give me link" + " " * 100_000 + "X") +_terminal_source_link_clause("links" + " " * 100_000 + "X") +_web_source_rows("[1] " + " " * 100_000) +''', + r''' +from src.agent_loop import (_calendar_listing_row, _contains_email_draft_headers, + _research_listing_row, _skill_listing_row) +for text in ( + "- [" + "](#research-" * 30_000, + "- **" + " **" * 30_000, + "- when: [" + "](#event-" * 30_000, + "To: x " + "Subject: x " * 30_000, +): + _research_listing_row(text) + _skill_listing_row(text) + _calendar_listing_row(text) + _contains_email_draft_headers(text) +''', + r''' +from src.agent_loop import _session_link_from_row, _session_list_summary_from_tool_output, _strip_trailing_done +evil = "Done. " + "\t" * 100_000 + "x" +assert _strip_trailing_done(evil) == evil +for row in ("[" + "\\" * 100_000, "[" * 100_000, + "[a" + "\\](#session-" * 20_000 + ")"): + _session_link_from_row(row) +for row in ("[x](#session-id) (" + "msgs" * 100_000, + "[x](#session-id) (" + "(last active today)" * 20_000): + _session_list_summary_from_tool_output("Chats:\n- " + row) +''', + r''' +from src.tool_parsing import * +from src.tool_parsing import (_iter_named_blocks, _strip_delimited, + _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE) +text = "}<|tool_call|>" + "<|tool_call|>call:web_search{" * 20_000 +assert list(_iter_named_blocks(text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE)) == [] +assert _strip_delimited(text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE) == text +# Public entry points also stay responsive, including fallback parsers. +text = "}<|tool_call|>" + "<|tool_call|>call:web_search{" * 3000 +assert parse_tool_blocks(text) == [] +strip_tool_blocks(text) +''', + r''' +from src.tool_parsing import (_iter_named_blocks, _iter_qwen_python_args, + _QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE, + _QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE, parse_tool_blocks) +for opener, closer, text in ( + (_QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE, "" * 20_000), + (_QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE, "" * 20_000), + (_QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE, "a" + "\t" * 100_000 + "x"), +): + assert list(_iter_named_blocks(text, opener, closer)) == [] +assert list(_iter_qwen_python_args("a" * 100_000)) == [] +assert parse_tool_blocks("" * 3000) == [] +assert parse_tool_blocks("manage_notes(" + "a" * 100_000 + ")") +''', + r''' +from src.tool_parsing import strip_angle_tags +for allow_empty in (False, True): + for replacement in ("", " "): + for text in ("<" * 200_000, ">" + "<" * 200_000): + assert strip_angle_tags(text, replacement, allow_empty=allow_empty) == text + assert strip_angle_tags("<" * 200_000 + ">tail", replacement, + allow_empty=allow_empty) == replacement + "tail" +''', + r''' +from src.tool_parsing import iter_email_addresses +for ascii_only in (False, True): + for text in ("+" * 100_000, "a@" + "a" * 100_000, + "a@" + "." * 100_000, "a@" * 20_000): + assert list(iter_email_addresses(text, ascii_only=ascii_only)) == [] +''', + r''' +from src.agent_loop import _mentions_pdf_url, _mentions_workspace_script +from src.clean_agent_preview import declared_workspace_artifacts +from src.text_scanning import contains_search_engine_navigation +from src.turn_contract import _mentions_workspace_artifact, _mentions_workspace_output +workspace = "/workspace/" * 30_000 + "x" +assert not _mentions_workspace_script(workspace) +assert not _mentions_workspace_artifact(workspace) +assert not _mentions_workspace_output(workspace) +assert declared_workspace_artifacts("create " + workspace) == () +assert not _mentions_pdf_url("http://" * 30_000 + "x") +assert not contains_search_engine_navigation("http://google." + "..google." * 30_000 + "x") +''', + r''' +from src.agent_loop import _looks_like_agent_reasoning_preamble, _looks_like_ody_qwen_leaked_tool_text +from src.text_scanning import contains_detailed_sequence_request +for text in ("\n" * 100_000 + "x", ("when " * 20_000) + "x"): + _looks_like_agent_reasoning_preamble(text) + _looks_like_ody_qwen_leaked_tool_text(text) + contains_detailed_sequence_request(text) +''', + r''' +from src.agent_loop import (_email_account_label, _email_sender_name, + _parse_explicit_open_panel_request, _read_only_shell_command, + _split_before_assistant_prompt, _split_note_items, + _strip_horizontal_space_before_lf) +space = " " * 100_000 +_parse_explicit_open_panel_request("open calendar" + space + "x") +_split_note_items(space + "x") +_email_sender_name("<" * 100_000 + "x") +_email_account_label("<" * 100_000 + "x") +_split_before_assistant_prompt("\n" * 100_000 + "x") +_strip_horizontal_space_before_lf(" " * 100_000 + "x") +_read_only_shell_command(space + "x") +''', + r''' +from src.agent_loop import _numbered_row_parenthesized_ids +from src.text_scanning import iter_angle_contents, iter_markdown_links, replace_markdown_links_with_labels +for text in ("<" * 100_000, "[" * 100_000, + ("[x](#session-" * 20_000) + "missing"): + list(iter_angle_contents(text)) + list(iter_markdown_links(text, target_prefix="#session-")) + replace_markdown_links_with_labels(text) +_numbered_row_parenthesized_ids("1. x " + "(" * 100_000 + "x") +''', + r''' +from src.agent_loop import (_contextual_summary_fragment, _is_terse_link_request, + _looks_like_destructive_request, _looks_like_youtube_tool_turn, + _private_browser_product_query) +from src.text_scanning import first_tag_content +for text in (("official " * 30_000) + "x", ("mark " * 30_000) + "x", + " " * 100_000 + "x", ("summary: x " * 20_000) + "x", + "find best " + "?" * 100_000 + "x", "