diff --git a/src/agent_loop.py b/src/agent_loop.py
index 2cbae780a..8a70fd3ad 100644
--- a/src/agent_loop.py
+++ b/src/agent_loop.py
@@ -82,6 +82,15 @@ from src.tool_approvals import (
tool_approval_store,
)
from src.tool_types import ToolBlock
+from src.tool_parsing import iter_email_addresses, strip_angle_tags
+from src.text_scanning import (
+ contains_detailed_sequence_request,
+ has_prefixed_token_match,
+ first_tag_content,
+ iter_angle_contents,
+ iter_markdown_links,
+ replace_markdown_links_with_labels,
+)
from src.turn_contract import selected_tools_for_request, with_turn_contract
from src.agent_runtime.journal import propose_action, execute_action
from src.agent_runtime.completion import with_completion_gate
@@ -1016,12 +1025,36 @@ def _looks_like_map_browser_request(text: str) -> bool:
def _looks_like_youtube_tool_turn(text: str) -> bool:
value = str(text or "").lower()
- return bool(re.search(
+ if re.search(
r"\b(?:youtube|youtu\.be|yt|video\s+comments?|comments?\s+on\s+(?:the\s+)?video|"
- r"transcript\s+(?:of|for)|(?:latest|newest|recent)\s+(?:\d+\s+)?(?:videos?|uploads?)|"
- r"official\s+.+\s+channel)\b",
+ r"transcript\s+(?:of|for)|(?:latest|newest|recent)\s+(?:\d+\s+)?(?:videos?|uploads?)"
+ r")\b",
value,
- ))
+ ):
+ return True
+ official_ends = iter(match.end() for match in re.finditer(r"\bofficial", value))
+ channel_starts = iter(match.start() for match in re.finditer(r"channel\b", value))
+ next_official_end = next(official_ends, -1)
+ next_channel_start = next(channel_starts, -1)
+ leading_space = False
+ body = False
+ trailing_space = False
+ for index, char in enumerate(value):
+ while next_channel_start >= 0 and next_channel_start < index:
+ next_channel_start = next(channel_starts, -1)
+ if trailing_space and next_channel_start == index:
+ return True
+ starts_after_official = next_official_end == index
+ if starts_after_official:
+ next_official_end = next(official_ends, -1)
+ is_space = char.isspace()
+ is_not_lf = char != "\n"
+ leading_space, body, trailing_space = (
+ (starts_after_official or leading_space) and is_space,
+ (leading_space or body) and is_not_lf,
+ (body or trailing_space) and is_space,
+ )
+ return False
def _explicitly_named_personal_tools(text: str) -> Set[str]:
@@ -1108,7 +1141,7 @@ def _qwen38_router_tool_names(query: str) -> Set[str]:
selected.update({"mcp__email__manage_email_state"})
if re.search(r"\b(?:latest|newest|recent|most recent|inbox)\b", q):
selected.add("mcp__email__list_emails")
- if re.search(r"\b(?:send|email)\b", q) and re.search(r"[\w.+-]+@[\w.-]+\.\w+", q):
+ if re.search(r"\b(?:send|email)\b", q) and next(iter_email_addresses(q), None):
selected.update({"send_email", "resolve_contact"})
if re.search(r"\b(?:reply|respond|response)\b", q):
selected.update({"mcp__email__list_emails", "mcp__email__read_email", "mcp__email__draft_email_reply"})
@@ -1524,8 +1557,20 @@ def _parse_explicit_open_panel_request(text: str) -> Optional[tuple[str, str]]:
value = re.sub(r"^(?:(?:ok(?:ay)?|now|then|also|next)[,;:]?\s+)+", "", value)
value = re.sub(r"^go\s+back\s+(?:and\s+)?(?:open|to)\s+", "open ", value)
value = re.sub(r"^return\s+to\s+", "open ", value)
- value = re.sub(r"\s+again[.!?]*$", "", value)
- value = re.sub(r"[.!?]+$", "", value).strip()
+ trailing_punctuation = len(value)
+ while trailing_punctuation and value[trailing_punctuation - 1] in ".!?":
+ trailing_punctuation -= 1
+ without_punctuation = value[:trailing_punctuation]
+ if without_punctuation.endswith("again"):
+ whitespace_start = len(without_punctuation) - len("again")
+ while whitespace_start and without_punctuation[whitespace_start - 1].isspace():
+ whitespace_start -= 1
+ if whitespace_start < len(without_punctuation) - len("again"):
+ value = value[:whitespace_start]
+ trailing_punctuation = len(value)
+ while trailing_punctuation and value[trailing_punctuation - 1] in ".!?":
+ trailing_punctuation -= 1
+ value = value[:trailing_punctuation].strip()
month_names = {
"january": "01", "jan": "01",
"february": "02", "feb": "02",
@@ -2104,26 +2149,34 @@ def _looks_like_agent_reasoning_preamble(text: str) -> bool:
# e.g. a full comparison followed by "Now let me verify nothing started."
# That whole round is progress text; retaining it duplicates the answer
# once the tool-result round supplies the actual verification.
- if re.search(
- r"(?:^|\n+|[.!?]\s+)(?:but\s+)?(?:now\s+)?"
+ trailing_segment = re.split(r"(?:\n+|[.!?]\s+)", lowered)[-1]
+ if re.match(
+ r"(?:but\s+)?(?:now\s+)?"
r"(?:let me|i(?:'ll| will)(?:\s+need\s+to)?|i\s+can(?:\s+now)?|i(?:'m| am)\s+(?:preparing|planning)\s+to)\s+"
r"(?:(?:carefully|methodically|systematically|closely|further)\s+){0,2}"
r"(?:continue|continuing|analy[sz]e|scan|check|verify|inspect|confirm|look up|fetch|open|list|search|refine|request|review|track|read|watch|(?:re-?)?examine|provide|give|state|report|answer|respond|summarize|conclude)\b"
r"[^.!?]*[.!?]?\s*$",
- lowered,
+ trailing_segment,
):
return True
# A process heading followed only by partial observation bullets is still
# analysis, not a delivered answer. Keep this bounded to inspection verbs
# so completed answer headings such as "Let me summarize:" remain valid.
- if re.search(
- r"(?:^|\n)\s*(?:now\s+)?(?:let me|i(?:'ll| will))\s+"
+ process_heading_re = re.compile(
+ r"\s*(?:now\s+)?(?:let me|i(?:'ll| will))\s+"
r"(?:carefully\s+|methodically\s+|systematically\s+){0,2}"
r"(?:track|trace|inspect|review|analy[sz]e|examine|check)\b[^\n]{0,140}:\s*\n"
- r"(?:\s*[-*]\s+[^\n]{1,240}\n?){1,8}\s*$",
- lowered,
- ):
- return True
+ r"(?:\s*[-*]\s+[^\n]{1,240}\n?){1,8}\s*$"
+ )
+ if any(marker in lowered for marker in ("let me", "i'll", "i will")):
+ line_start = 0
+ while line_start <= len(lowered):
+ if process_heading_re.match(lowered, line_start):
+ return True
+ newline = lowered.find("\n", line_start)
+ if newline < 0:
+ break
+ line_start = newline + 1
first_line = lowered.splitlines()[0].strip()
if re.search(
r"\b(?:i need to|i should|let me|i'll(?:\s+need\s+to)?|i(?:'m| am)\s+(?:preparing|planning)\s+to)\s+"
@@ -2144,11 +2197,28 @@ def _looks_like_agent_reasoning_preamble(text: str) -> bool:
lowered,
):
return True
- return bool(re.search(
- r"(?:^|[。!?\n]\s*)(?:我需要|需要先|让我|先|接下来(?:我)?(?:会|要)?).{0,12}"
- r"(?:查看|检查|读取|分析|继续|使用|调用)",
- value,
- ))
+ chinese_body = re.compile(
+ r"(?:我需要|需要先|让我|先|接下来(?:我)?(?:会|要)?).{0,12}"
+ r"(?:查看|检查|读取|分析|继续|使用|调用)"
+ )
+ if chinese_body.match(value):
+ return True
+ pos = 0
+ boundary_pattern = re.compile(r"[。!?\n]")
+ while boundary := boundary_pattern.search(value, pos):
+ candidate = boundary.end()
+ while candidate < len(value) and value[candidate].isspace():
+ candidate += 1
+ if chinese_body.match(value, candidate):
+ return True
+ pos = max(candidate, boundary.end())
+ return False
+
+
+def _strip_trailing_done(text: str) -> str:
+ """Remove a terminal Done. and adjacent whitespace with a linear scan."""
+ trimmed = text.rstrip()
+ return trimmed[:-5].rstrip() if trimmed.lower().endswith("done.") else text
def _strip_trailing_answer_promise(text: str) -> str:
@@ -2405,6 +2475,21 @@ def _parse_qwen_explicit_note_delete(text: str) -> Optional[str]:
return match.group(1).strip().strip("\"'`").rstrip(".") if match else None
+def _captures_after_first_prefix(
+ value: str,
+ prefix_pattern: str,
+ remainder_pattern: str,
+ *,
+ flags: int = re.IGNORECASE,
+) -> tuple[str | None, ...] | None:
+ """Match a suffix once after the first prefix that can own all later text."""
+ prefix = re.search(prefix_pattern, value, flags)
+ if prefix is None:
+ return None
+ remainder = re.match(remainder_pattern, value[prefix.end():], flags)
+ return remainder.groups() if remainder is not None else None
+
+
def _parse_qwen_explicit_note_update(text: str) -> Optional[tuple[str, str]]:
"""Extract an exact note title and replacement content from a clear update."""
value = str(text or "").strip()
@@ -2412,15 +2497,14 @@ def _parse_qwen_explicit_note_update(text: str) -> Optional[tuple[str, str]]:
r"\bnote\b", value, re.IGNORECASE
):
return None
- match = re.search(
- r"\bnote\s+titled\s+(.+?)\s+"
- r"so\s+its\s+content\s+is\s+['\"]([^'\"]+)['\"]",
+ captures = _captures_after_first_prefix(
value,
- re.IGNORECASE,
+ r"\bnote\s+titled(?=\s)",
+ r"\s+(.+?)\s+so\s+its\s+content\s+is\s+['\"]([^'\"]+)['\"]",
)
- if not match:
+ if captures is None:
return None
- return match.group(1).strip().strip("\"'`").rstrip("."), match.group(2)
+ return captures[0].strip().strip("\"'`").rstrip("."), captures[1]
def _parse_qwen_explicit_note_search(text: str) -> Optional[str]:
@@ -2515,19 +2599,30 @@ def _is_qwen_explicit_endpoint_list_request(text: str) -> bool:
))
-def _parse_explicit_pipeline_request(text: str) -> Optional[tuple[str, str]]:
- value = str(text or "").strip()
- match = re.search(
- r"\bpipeline\s+using\s+([^\s,]+)\s+to\s+(.+?),\s*then\s+"
- r"([^\s,]+)\s+to\s+(.+?)(?:[.!?]\s*)?$",
- value,
+def _pipeline_request_parts(value: str) -> tuple[str, str, str, str] | None:
+ prefix = re.search(
+ r"\bpipeline\s+using\s+([^\s,]+)\s+to(?=\s)", value, re.IGNORECASE
+ )
+ if prefix is None:
+ return None
+ remainder = re.match(
+ r"\s+(.+?),\s*then\s+([^\s,]+)\s+to\s+(.+?)(?:[.!?]\s*)?$",
+ value[prefix.end():],
re.IGNORECASE,
)
- if not match:
+ if remainder is None:
+ return None
+ return prefix.group(1), remainder.group(1), remainder.group(2), remainder.group(3)
+
+
+def _parse_explicit_pipeline_request(text: str) -> Optional[tuple[str, str]]:
+ value = str(text or "").strip()
+ parts = _pipeline_request_parts(value)
+ if parts is None:
return None
steps = [
- {"model": match.group(1).strip(), "instruction": match.group(2).strip()},
- {"model": match.group(3).strip(), "instruction": match.group(4).strip()},
+ {"model": parts[0].strip(), "instruction": parts[1].strip()},
+ {"model": parts[2].strip(), "instruction": parts[3].strip()},
]
return "pipeline", json.dumps({"steps": steps})
@@ -2704,7 +2799,9 @@ def _recent_session_id_for_title(messages: List[Dict], title_query: str) -> str:
if msg.get("role") != "assistant":
continue
content = str(msg.get("content") or "")
- for label, sid in reversed(re.findall(r"\[([^\]]+)\]\(#session-([^)]+)\)", content)):
+ for _start, _end, label, sid in reversed(list(iter_markdown_links(
+ content, target_prefix="#session-"
+ ))):
label_l = label.lower()
if not query_terms or all(term in label_l for term in query_terms):
return sid.strip()
@@ -2789,7 +2886,7 @@ def _parse_qwen_explicit_session_action(text: str, messages: List[Dict]) -> Opti
if rename_match:
new_name = (rename_match.group(1) or "").strip(" \t\r\n\"'`.")
new_name = re.split(
- r"\s+(?:Use the tool directly|Keep this read-only|Report the result|Do not)\b",
+ r"(?<=\s)(?:Use the tool directly|Keep this read-only|Report the result|Do not)\b",
new_name,
maxsplit=1,
flags=re.IGNORECASE,
@@ -2877,10 +2974,14 @@ def _parse_qwen_explicit_session_send(text: str, messages: List[Dict]) -> Option
sid = _recent_session_id_for_title(messages, target_text)
if not sid and re.search(r"\b(?:that|this)\b", target_text, re.IGNORECASE):
for prior in reversed(messages or []):
- links = re.findall(
- r"\[[^\]]+\]\(#session-([A-Za-z0-9_-]+)\)",
- str(prior.get("content") or ""),
- )
+ links = [
+ target
+ for _start, _end, _label, target in iter_markdown_links(
+ str(prior.get("content") or ""),
+ target_prefix="#session-",
+ target_re=re.compile(r"[A-Za-z0-9_-]+"),
+ )
+ ]
if links:
sid = links[-1]
break
@@ -2889,6 +2990,19 @@ def _parse_qwen_explicit_session_send(text: str, messages: List[Dict]) -> Option
return "send_to_session", f"{sid}\n{relay_message}"
+def _session_find_query_capture(value: str) -> str | None:
+ action_prefix = re.search(r"\b(?:find|search|show)\b(?=\s)", value, re.IGNORECASE)
+ if action_prefix is None:
+ return None
+ optional_for = r"(?:\s+for)?" if action_prefix.group(0).casefold() == "search" else ""
+ remainder = re.match(
+ optional_for + r"\s+(?:the\s+)?(.+?)\s+(?:chat|session|conversation)\b",
+ value[action_prefix.end():],
+ re.IGNORECASE,
+ )
+ return remainder.group(1) if remainder is not None else None
+
+
def _parse_qwen_explicit_session_find(text: str) -> Optional[tuple[str, str]]:
"""Map obvious chat lookup requests to list_sessions with a filter."""
value = str(text or "").strip()
@@ -2910,14 +3024,10 @@ def _parse_qwen_explicit_session_find(text: str) -> Optional[tuple[str, str]]:
re.IGNORECASE,
):
return "list_sessions", ""
- match = re.search(
- r"\b(?:find|search(?:\s+for)?|show)\s+(?:the\s+)?(.+?)\s+(?:chat|session|conversation)\b",
- value,
- re.IGNORECASE,
- )
- if not match:
+ captured_query = _session_find_query_capture(value)
+ if captured_query is None:
return None
- query = (match.group(1) or "").strip(" \t\r\n\"'`.")
+ query = captured_query.strip(" \t\r\n\"'`.")
query = re.sub(r"\bscratch\b", "", query, flags=re.IGNORECASE).strip()
query = re.sub(r"\s+", " ", query).strip()
return "list_sessions", query or value
@@ -3180,15 +3290,15 @@ def _parse_qwen_explicit_email_topic_bulk_action_request(text: str) -> Optional[
return None
if re.search(r"\bUIDs?\b", value, re.IGNORECASE):
return None
- match = re.search(
- r"\b(?:delete|trash|remove|archive|mark(?:\s+as)?\s+(?:read|unread)|mark\s+(?:read|unread))\b"
- r"\s+(?:all|every|the)?\s*(?:my\s+)?(.+?)\s+(?:emails?|mail|messages?)\b",
+ captures = _captures_after_first_prefix(
value,
- re.IGNORECASE,
+ r"\b(?:delete|trash|remove|archive|mark(?:\s+as)?\s+(?:read|unread)|mark\s+(?:read|unread))\b"
+ r"(?=\s)",
+ r"\s+(?:all|every|the)?\s*(?:my\s+)?(.+?)\s+(?:emails?|mail|messages?)\b",
)
- if not match:
+ if captures is None:
return None
- query = re.sub(r"\s+", " ", match.group(1)).strip(" .\"'")
+ query = re.sub(r"\s+", " ", captures[0]).strip(" .\"'")
if not query or query.lower() in {"all", "the", "my"}:
return None
return {"action": action, "query": query, "folder": "INBOX", "max_results": 50}
@@ -3345,12 +3455,12 @@ def _parse_qwen_explicit_block_sender_request(text: str) -> Optional[dict[str, A
return None
if re.search(r"\b(?:should\s+i|should\s+we|would\s+you|can\s+i|do\s+you\s+think)\b", value, re.IGNORECASE):
return None
- match = re.search(r"[\w.+-]+@[\w.-]+\.\w+", value)
- if not match:
+ address = next(iter_email_addresses(value), "")
+ if not address:
return None
move_existing = not re.search(r"\b(?:do\s+not|don't|dont)\s+(?:move|delete|trash|junk)\b|\bleave\s+existing\b", value, re.IGNORECASE)
return {
- "sender": match.group(0),
+ "sender": address,
"folder": "INBOX",
"move_existing": move_existing,
"reason": "User explicitly requested sender block.",
@@ -3392,10 +3502,10 @@ def _parse_qwen_explicit_unblock_sender_request(text: str) -> Optional[dict[str,
value = str(text or "").strip()
if not value or not re.search(r"\b(?:unblock|allow|remove\s+from\s+block)\b", value, re.IGNORECASE):
return None
- match = re.search(r"[\w.+-]+@[\w.-]+\.\w+", value)
- if not match:
+ address = next(iter_email_addresses(value), "")
+ if not address:
return None
- return {"sender": match.group(0)}
+ return {"sender": address}
def _email_relative_date_range(text: str) -> Optional[dict[str, str]]:
@@ -4253,8 +4363,8 @@ def _note_title_id_pairs_from_tool_output(raw: str) -> list[tuple[str, str]]:
pairs.append(key)
seen.add(key)
- for match in re.finditer(r"\[([^\]]+)\]\(#note-([^)]+)\)", raw):
- add_pair(match.group(1), match.group(2))
+ for _start, _end, title, note_id in iter_markdown_links(raw, target_prefix="#note-"):
+ add_pair(title, note_id)
for line in raw.splitlines():
match = re.match(r"^\s*-\s+\[([^\]]+)\]\s+\*\*(.*?)\*\*", line)
if match:
@@ -4326,7 +4436,7 @@ def _notes_expected_actions(user_text: str) -> set[str]:
def _split_note_items(value: str) -> list[dict[str, Any]]:
parts = [
re.sub(r"\s+", " ", part).strip(" .")
- for part in re.split(r"\s*,\s*|\s+\band\b\s+", str(value or ""))
+ for part in re.split(r",|(?<=\s)and(?=\s)", str(value or ""))
]
return [{"text": part, "done": False} for part in parts if part]
@@ -4391,6 +4501,32 @@ def _is_personal_tool_definition_turn(text: str) -> bool:
)
+def _remaining_checklist_name(value: str) -> str | None:
+ """Return the checklist name from the staged legacy question grammar."""
+ lead = re.search(r"\b(?:what(?:'s| is)?|show|tell\s+me)\b", value, re.IGNORECASE)
+ if lead is None:
+ return None
+ line_end = value.find("\n", lead.end())
+ if line_end < 0:
+ line_end = len(value)
+ remaining = re.compile(r"\b(?:left|remaining)\b", re.IGNORECASE).search(
+ value, lead.end(), line_end
+ )
+ if remaining is None:
+ return None
+ location = re.compile(r"\b(?:on|in)\b(?=\s)", re.IGNORECASE).search(
+ value, remaining.end(), line_end
+ )
+ if location is None:
+ return None
+ name = re.match(
+ r"\s+(?:the\s+)?(.+?)\s+checklist\b",
+ value[location.end():line_end],
+ re.IGNORECASE,
+ )
+ return name.group(1) if name is not None else None
+
+
def _parse_simple_notes_tool_request(text: str) -> Optional[tuple[str, str]]:
"""Deterministic fallback for obvious notes commands when a model stalls."""
value = str(text or "").strip()
@@ -4425,14 +4561,14 @@ def _parse_simple_notes_tool_request(text: str) -> Optional[tuple[str, str]]:
else:
label = ""
- checklist_match = re.search(
- r"\b(?:make|create|add)\s+(?:a\s+)?checklist\s+(?:called|titled|named)\s+(.+?)\s+with\s+(.+?)\s*$",
+ checklist_captures = _captures_after_first_prefix(
value,
- re.IGNORECASE,
+ r"\b(?:make|create|add)\s+(?:a\s+)?checklist\s+(?:called|titled|named)(?=\s)",
+ r"\s+(.+?)\s+with\s+(.+?)$",
)
- if checklist_match:
- title = re.sub(r"\s+", " ", checklist_match.group(1)).strip(" .\"'")
- items = _split_note_items(checklist_match.group(2))
+ if checklist_captures is not None:
+ title = re.sub(r"\s+", " ", checklist_captures[0]).strip(" .\"'")
+ items = _split_note_items(checklist_captures[1])
if title and items:
return "manage_notes", json.dumps({
"action": "add",
@@ -4457,13 +4593,9 @@ def _parse_simple_notes_tool_request(text: str) -> Optional[tuple[str, str]]:
args["label"] = label
return "manage_notes", json.dumps(args)
- remaining_match = re.search(
- r"\b(?:what(?:'s| is)?|show|tell\s+me)\b.*?\b(?:left|remaining)\b.*?\b(?:on|in)\s+(?:the\s+)?(.+?)\s+checklist\b",
- value,
- re.IGNORECASE,
- )
- if remaining_match:
- query = _clean_notes_search_query(remaining_match.group(1))
+ remaining_name = _remaining_checklist_name(value)
+ if remaining_name is not None:
+ query = _clean_notes_search_query(remaining_name)
if query:
return "manage_notes", json.dumps({"action": "search", "query": query})
@@ -4474,7 +4606,7 @@ def _parse_simple_notes_tool_request(text: str) -> Optional[tuple[str, str]]:
)
if note_saying_match:
body = re.sub(
- r"\s+(?:and\s+)?(?:tag|label)\s+(?:it\s+)?(?:as\s+)?#?[a-zA-Z0-9_-]{2,40}\s*$",
+ r"(?<=\s)(?:and\s+)?(?:tag|label)\s+(?:it\s+)?(?:as\s+)?#?[a-zA-Z0-9_-]{2,40}\s*$",
"",
note_saying_match.group(1),
flags=re.IGNORECASE,
@@ -4741,6 +4873,38 @@ def _single_document_id_from_tool_output(raw: str) -> str:
return next(iter(ids)) if len(ids) == 1 else ""
+def _session_link_from_row(row: str) -> str:
+ """Find the first session link in a newline-free listing row in O(n).
+
+ The legacy label grammar allows both an ordinary backslash and an escaped
+ closing bracket. Keep its greedy choice of the last reachable link closer,
+ without trying exponentially many ways to partition a backslash run.
+ An unescaped closing bracket ends a label; later openers can then be tried.
+ The id's closing-parenthesis search also advances monotonically.
+ """
+ start = row.find("[")
+ bracket = row.find("]", start + 1) if start >= 0 else -1
+ link_end = -1
+ id_end = -1
+ while bracket >= 0:
+ if bracket > start + 1 and row.startswith("(#session-", bracket + 1):
+ id_start = bracket + len("](#session-")
+ if id_end < id_start:
+ id_end = row.find(")", id_start)
+ if id_end < 0:
+ break
+ if id_end > id_start:
+ link_end = id_end + 1
+ if row[bracket - 1] != "\\":
+ if link_end >= 0:
+ break
+ start = row.find("[", bracket + 1)
+ if start < 0:
+ break
+ bracket = row.find("]", max(bracket + 1, start + 1))
+ return row[start:link_end] if link_end >= 0 else ""
+
+
def _session_list_summary_from_tool_output(raw: str, max_items: int = 12) -> str:
"""Keep a broad session listing readable and terminal for small routers."""
if not isinstance(raw, str) or not raw.strip():
@@ -4758,13 +4922,17 @@ def _session_list_summary_from_tool_output(raw: str, max_items: int = 12) -> str
return "\n".join(lines[: max_items + 1])
formatted_rows: list[str] = []
for row in rows:
- link_match = re.search(r"(\[(?:\\.|[^\]])+\]\(#session-[^)]+\))", row)
- if link_match:
- meta_match = re.search(r"\(([^()]*(?:last active|msgs|model|id:)[^()]*)\)", row)
- meta = meta_match.group(1) if meta_match else ""
+ link = _session_link_from_row(row)
+ if link:
+ meta = next(
+ (match.group(1) for match in re.finditer(r"\(([^()]*)\)", row)
+ if any(marker in match.group(1)
+ for marker in ("last active", "msgs", "model", "id:"))),
+ "",
+ )
active = re.search(r"last active [^)]+", meta)
suffix = f" ({active.group(0)})" if active else ""
- formatted_rows.append(f"- {link_match.group(1)}{suffix}")
+ formatted_rows.append(f"- {link}{suffix}")
else:
formatted_rows.append(row[:180].rstrip() + ("..." if len(row) > 180 else ""))
shown = formatted_rows[:max_items]
@@ -4793,6 +4961,28 @@ def _registry_list_summary_from_tool_output(raw: str, max_items: int = 12) -> st
return summary if len(summary) <= 3200 else summary[:3197].rstrip() + "..."
+def _research_listing_row(line: str) -> tuple[str, str, str] | None:
+ """Parse the legacy research markdown row without retrying label openers."""
+ if not line.startswith("-"):
+ return None
+ cursor = 1
+ if cursor >= len(line) or not line[cursor].isspace():
+ return None
+ while cursor < len(line) and line[cursor].isspace():
+ cursor += 1
+ if cursor >= len(line) or line[cursor] != "[":
+ return None
+ title_start = cursor + 1
+ marker = line.find("](#research-", title_start)
+ while marker >= 0:
+ identifier_start = marker + len("](#research-")
+ close = line.find(")", identifier_start)
+ if close > identifier_start:
+ return line[title_start:marker], line[identifier_start:close], line[close + 1:]
+ marker = line.find("](#research-", marker + 1)
+ return None
+
+
def _research_list_summary_from_tool_output(raw: str, max_items: int = 6) -> str:
"""Keep saved research listings concise while preserving report anchors."""
if not isinstance(raw, str) or not raw.strip():
@@ -4805,14 +4995,15 @@ def _research_list_summary_from_tool_output(raw: str, max_items: int = 6) -> str
return lines[0]
rows: list[str] = []
for line in lines[1:]:
- match = re.match(r"^-\s+\[(.*?)\]\(#research-([^)]+)\)(.*)$", line)
- if not match:
+ parsed_row = _research_listing_row(line)
+ if parsed_row is None:
continue
- title = re.sub(r"\s+", " ", match.group(1)).strip()
+ raw_title, research_id, raw_suffix = parsed_row
+ title = re.sub(r"\s+", " ", raw_title).strip()
if len(title) > 110:
title = title[:107].rstrip() + "..."
- suffix = re.sub(r"\s+", " ", match.group(3) or "").strip()
- rows.append(f"- [{title}](#research-{match.group(2)}) {suffix}".rstrip())
+ suffix = re.sub(r"\s+", " ", raw_suffix).strip()
+ rows.append(f"- [{title}](#research-{research_id}) {suffix}".rstrip())
if len(rows) >= max_items:
break
if not rows:
@@ -4824,6 +5015,54 @@ def _research_list_summary_from_tool_output(raw: str, max_items: int = 6) -> str
return "\n".join([lines[0], *rows])
+def _skill_listing_row(line: str) -> tuple[str, str | None, str | None, str | None] | None:
+ """Parse the legacy bold skill row with monotonic closer searches."""
+ if not line.startswith("-"):
+ return None
+ cursor = 1
+ if cursor >= len(line) or not line[cursor].isspace():
+ return None
+ while cursor < len(line) and line[cursor].isspace():
+ cursor += 1
+ if not line.startswith("**", cursor):
+ return None
+ name_start = cursor + 2
+ closer = line.find("**", name_start)
+ while closer >= 0:
+ rest = closer + 2
+ if rest == len(line):
+ return line[name_start:closer], None, None, None
+ if line[rest] == ":":
+ description = line[rest + 1:].lstrip()
+ return line[name_start:closer], None, None, description
+ if line[rest].isspace():
+ meta_start = rest
+ while meta_start < len(line) and line[meta_start].isspace():
+ meta_start += 1
+ if meta_start < len(line) and line[meta_start] == "(":
+ meta_end = line.find(")", meta_start + 1)
+ while meta_end >= 0:
+ tail = meta_end + 1
+ if tail == len(line):
+ return line[name_start:closer], line[meta_start + 1:meta_end], None, None
+ if line[tail] == ":":
+ return (
+ line[name_start:closer],
+ line[meta_start + 1:meta_end],
+ None,
+ line[tail + 1:].lstrip(),
+ )
+ meta_end = line.find(")", meta_end + 1)
+ elif line.startswith("[draft]", meta_start):
+ tail = meta_start + len("[draft]")
+ if tail == len(line):
+ return line[name_start:closer], None, "draft", None
+ if tail < len(line) and line[tail] == ":":
+ return line[name_start:closer], None, "draft", line[tail + 1:].lstrip()
+ closer = line.find("**", closer + 1)
+ return None
+
+
def _skills_list_summary_from_tool_output(raw: str, max_items: int = 8) -> str:
"""Keep the skill index visible without dumping the full registry."""
if not isinstance(raw, str) or not raw.strip():
@@ -4845,10 +5084,11 @@ def _skills_list_summary_from_tool_output(raw: str, max_items: int = 8) -> str:
label = section or "Skills"
if label in totals:
totals[label] += 1
- match = re.match(r"^-\s+\*\*(.*?)\*\*(?:\s+\((.*?)\)|\s+\[(draft)\])?(?::\s*(.*))?$", line)
- if match:
- name = re.sub(r"\s+", " ", match.group(1)).strip()
- meta = re.sub(r"\s+", " ", (match.group(2) or match.group(3) or label).strip())
+ parsed_row = _skill_listing_row(line)
+ if parsed_row is not None:
+ raw_name, parenthesized_meta, draft_meta, _description = parsed_row
+ name = re.sub(r"\s+", " ", raw_name).strip()
+ meta = re.sub(r"\s+", " ", (parenthesized_meta or draft_meta or label).strip())
rows.append((label, f"- [{name}](#skill-{quote(name, safe='')}) ({meta})"))
else:
rows.append((label, line[:96].rstrip() + ("..." if len(line) > 96 else "")))
@@ -4898,6 +5138,45 @@ def _calendar_detail_requested(text: str) -> bool:
)
+def _calendar_listing_row(line: str) -> tuple[str, str, str, str] | None:
+ """Parse a calendar row while advancing through delimiters only once."""
+ cursor = 0
+ while cursor < len(line) and line[cursor].isspace():
+ cursor += 1
+ if cursor >= len(line) or line[cursor] != "-":
+ return None
+ cursor += 1
+ if cursor >= len(line) or not line[cursor].isspace():
+ return None
+ while cursor < len(line) and line[cursor].isspace():
+ cursor += 1
+ when_start = cursor
+ colon = line.find(":", when_start + 1)
+ marker_from = when_start
+ while colon >= 0:
+ spacing = colon + 1
+ if spacing < len(line) and line[spacing].isspace():
+ while spacing < len(line) and line[spacing].isspace():
+ spacing += 1
+ if spacing < len(line) and line[spacing] == "[":
+ title_start = spacing + 1
+ marker = line.find("](#event-", max(title_start, marker_from))
+ while marker >= 0:
+ identifier_start = marker + len("](#event-")
+ close = line.find(")", identifier_start)
+ if close > identifier_start:
+ return (
+ line[when_start:colon],
+ line[title_start:marker],
+ line[identifier_start:close],
+ line[close + 1:],
+ )
+ marker = line.find("](#event-", marker + 1)
+ marker_from = len(line)
+ colon = line.find(":", colon + 1)
+ return None
+
+
def _calendar_list_summary_from_tool_output(
raw: str,
max_items: int = 20,
@@ -4951,17 +5230,18 @@ def _calendar_list_summary_from_tool_output(
items: list[str] = []
current_item_idx = -1
for line in text.splitlines():
- m = re.match(r"^\s*-\s+(.+?):\s+\[(.*?)\]\(#event-([^)]+)\)(.*)$", line)
- if not m:
+ parsed_row = _calendar_listing_row(line)
+ if parsed_row is None:
if include_details and current_item_idx >= 0:
detail = re.sub(r"\s+", " ", line).strip()
if detail and not detail.startswith("-"):
items[current_item_idx] = f"{items[current_item_idx]} — {detail}"
continue
- when = re.sub(r"\s+", " ", m.group(1)).strip()
- title = re.sub(r"\s+", " ", m.group(2)).strip()
- event_id = m.group(3).strip()
- suffix = re.sub(r"\s+", " ", m.group(4) or "").strip()
+ raw_when, raw_title, raw_event_id, raw_suffix = parsed_row
+ when = re.sub(r"\s+", " ", raw_when).strip()
+ title = re.sub(r"\s+", " ", raw_title).strip()
+ event_id = raw_event_id.strip()
+ suffix = re.sub(r"\s+", " ", raw_suffix).strip()
label = f"[{title}](#event-{event_id}) — {format_when(when)}"
if suffix:
label += f" {suffix}"
@@ -5278,18 +5558,18 @@ def _parse_simple_calendar_tool_request(
args["query"] = "travel"
return "manage_calendar", json.dumps(args, ensure_ascii=False)
- tag_match = re.search(
- r"\b(?:change|update|set|retag)\b\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b",
+ tag_captures = _captures_after_first_prefix(
value,
- re.IGNORECASE,
+ r"\b(?:change|update|set|retag)\b(?=\s)",
+ r"\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b",
)
- if tag_match and re.search(r"\b(?:calendar|event|trip|meeting|appointment)\b", q):
- title = re.sub(r"\s+", " ", tag_match.group(1)).strip(" .")
+ if tag_captures is not None and re.search(r"\b(?:calendar|event|trip|meeting|appointment)\b", q):
+ title = re.sub(r"\s+", " ", tag_captures[0]).strip(" .")
if title:
return "manage_calendar", json.dumps({
"action": "update_event",
"summary": title,
- "tag": tag_match.group(2).lower(),
+ "tag": tag_captures[1].lower(),
}, ensure_ascii=False)
return None
@@ -5873,8 +6153,8 @@ def _calendar_title_uid_pairs_from_tool_event(event: dict[str, Any]) -> list[tup
add_pair(row.get("summary") or row.get("title"), row.get("uid") or row.get("id"))
raw = str(event.get("output") or "")
- for match in re.finditer(r"\[([^\]]+)\]\(#event-([^)]+)\)", raw):
- add_pair(match.group(1), match.group(2))
+ for _start, _end, title, event_id in iter_markdown_links(raw, target_prefix="#event-"):
+ add_pair(title, event_id)
return pairs
@@ -5978,12 +6258,34 @@ def _friendly_email_date(value: str) -> str:
return text
+def _strip_terminal_delimited_value(
+ text: str,
+ opener: str,
+ closer: str,
+ *,
+ required: str = "",
+ allow_empty: bool = False,
+) -> str:
+ """Remove one flat terminal ``opener...closer`` region in O(n)."""
+ trimmed = text.rstrip()
+ if not trimmed.endswith(closer):
+ return text
+ previous_closer = trimmed.rfind(closer, 0, len(trimmed) - len(closer))
+ opener_at = trimmed.find(opener, previous_closer + len(closer))
+ if opener_at < 0:
+ return text
+ inner = trimmed[opener_at + len(opener):-len(closer)]
+ if (not allow_empty and not inner) or (required and required not in inner):
+ return text
+ return trimmed[:opener_at].rstrip()
+
+
def _email_sender_name(value: str) -> str:
text = re.sub(r"\s+", " ", str(value or "")).strip()
if not text:
return ""
- text = re.sub(r"\s*\([^)]*@[^)]*\)\s*$", "", text).strip()
- text = re.sub(r"\s*<[^>]*>\s*$", "", text).strip()
+ text = _strip_terminal_delimited_value(text, "(", ")", required="@").strip()
+ text = _strip_terminal_delimited_value(text, "<", ">", allow_empty=True).strip()
return text or str(value or "").strip()
@@ -5991,7 +6293,7 @@ def _email_account_label(value: str) -> str:
text = re.sub(r"\s+", " ", str(value or "")).strip()
if not text:
return ""
- return re.sub(r"\s*<[^>]+>\s*$", "", text).strip() or text
+ return _strip_terminal_delimited_value(text, "<", ">").strip() or text
def _format_email_attachment_summary_item(item: dict[str, str]) -> str:
@@ -6157,6 +6459,21 @@ def _email_read_summaries_from_tool_events(tool_events: list[dict[str, Any]]) ->
return summaries
+def _strip_horizontal_space_before_lf(text: str) -> str:
+ """Remove spaces/tabs directly before LF without failed suffix retries."""
+ out = []
+ pos = 0
+ while (newline := text.find("\n", pos)) >= 0:
+ trim_at = newline
+ while trim_at > pos and text[trim_at - 1] in " \t":
+ trim_at -= 1
+ out.append(text[pos:trim_at])
+ out.append("\n")
+ pos = newline + 1
+ out.append(text[pos:])
+ return "".join(out)
+
+
def _email_read_evidence_from_tool_output(raw: str, *, max_body_chars: int = 6000) -> str:
"""Return bounded, plain-text evidence for a final email lookup synthesis."""
if not isinstance(raw, str) or not raw.strip():
@@ -6167,23 +6484,29 @@ def _email_read_evidence_from_tool_output(raw: str, *, max_body_chars: int = 600
# model round.
text = re.sub(r"
", "\n", text, flags=re.IGNORECASE)
text = re.sub(r"(?:p|div|li|tr|h[1-6])\s*>", "\n", text, flags=re.IGNORECASE)
- text = re.sub(r"<[^>]+>", "", text)
+ text = strip_angle_tags(text)
text = html.unescape(text)
- text = re.sub(r"[ \t]+\n", "\n", text)
+ text = _strip_horizontal_space_before_lf(text)
text = re.sub(r"\n{3,}", "\n\n", text).strip()
if len(text) > max_body_chars:
text = text[:max_body_chars].rstrip() + "\n[...email truncated]"
return text
+def _is_terse_email_lookup_followup(text: str) -> bool:
+ value = str(text or "").strip()
+ value = value.rstrip("?.!").rstrip()
+ return bool(re.fullmatch(
+ r"(?:and|so|well|still|then|okay|ok|did you find it(?: yet)?|what did you find)",
+ value,
+ re.IGNORECASE,
+ ))
+
+
def _email_lookup_request_from_messages(messages: list[dict], last_user: str) -> str:
"""Recover the substantive request behind terse follow-ups such as 'and?'."""
- terse = re.compile(
- r"^\s*(?:and|so|well|still|then|okay|ok|did you find it(?: yet)?|what did you find)\s*[?.!]*\s*$",
- re.IGNORECASE,
- )
current = str(last_user or "").strip()
- if current and not terse.match(current):
+ if current and not _is_terse_email_lookup_followup(current):
return current
for message in reversed(messages or []):
if not isinstance(message, dict) or message.get("role") != "user":
@@ -6192,7 +6515,7 @@ def _email_lookup_request_from_messages(messages: list[dict], last_user: str) ->
if not isinstance(content, str):
continue
candidate = content.strip()
- if candidate and not terse.match(candidate):
+ if candidate and not _is_terse_email_lookup_followup(candidate):
return candidate
return current
@@ -6792,6 +7115,31 @@ _COMPACT_EMAIL_UNSUBSCRIBE_TOOLS = {
"unsubscribe_email",
}
+_WORKSPACE_PREFIX_RE = re.compile(r"/workspace/", re.IGNORECASE)
+_WORKSPACE_SCRIPT_RE = re.compile(
+ r"/workspace/[^\s`\"']+\.(?:py|pyw|sh|bash|js|mjs|ts|rb|pl)\b",
+ re.IGNORECASE,
+)
+_WORKSPACE_QUOTED_TOKEN_TAIL_RE = re.compile(r"[^\s`\"']*")
+_HTTP_PREFIX_RE = re.compile(r"https?://", re.IGNORECASE)
+_PDF_URL_RE = re.compile(r"https?://\S+(?:\.pdf\b|/pdf/)", re.IGNORECASE)
+_NONSPACE_TOKEN_TAIL_RE = re.compile(r"\S*")
+
+
+def _mentions_workspace_script(text: str) -> bool:
+ return has_prefixed_token_match(
+ str(text or ""),
+ _WORKSPACE_PREFIX_RE,
+ _WORKSPACE_SCRIPT_RE,
+ _WORKSPACE_QUOTED_TOKEN_TAIL_RE,
+ )
+
+
+def _mentions_pdf_url(text: str) -> bool:
+ return has_prefixed_token_match(
+ str(text or ""), _HTTP_PREFIX_RE, _PDF_URL_RE, _NONSPACE_TOKEN_TAIL_RE
+ )
+
def _blocked_network_recovery_tools(events: Sequence[Dict[str, Any]]) -> Set[str]:
"""Preserve native recovery tools named by our own network guard.
@@ -6873,11 +7221,7 @@ def _compact_native_route_tools(
workspace_generator_request = bool(
workspace_artifact_request
and (
- re.search(
- r"/workspace/[^\s`\"']+\.(?:py|pyw|sh|bash|js|mjs|ts|rb|pl)\b",
- text,
- re.IGNORECASE,
- )
+ _mentions_workspace_script(text)
or (
# A multi-output data+visual deliverable is an execution
# workflow even when its generator is discovered after the
@@ -6895,7 +7239,7 @@ def _compact_native_route_tools(
compact.update(WEB_TOOL_NAMES)
if (
"pdf_extract" in original
- and re.search(r"https?://\S+(?:\.pdf\b|/pdf/)", text, re.IGNORECASE)
+ and _mentions_pdf_url(text)
):
compact.add("pdf_extract")
if _looks_like_youtube_tool_turn(text):
@@ -7220,7 +7564,7 @@ def _compact_native_artifact_tools(
allowed.update(WEB_TOOL_NAMES)
allowed.update({"web_fetch", "pdf_extract"})
if (
- re.search(r"/workspace/[^\s`\"']+\.(?:py|pyw|sh|bash|js|mjs|ts|rb|pl)\b", value, re.IGNORECASE)
+ _mentions_workspace_script(value)
or (
len(artifacts) >= 2
and artifact_suffixes & {".csv", ".json", ".xlsx"}
@@ -10095,12 +10439,12 @@ def _is_terse_link_request(text: str) -> bool:
"""True for short links/sources fragments that need context to be actionable."""
return bool(
re.fullmatch(
- r"\s*(?:(?:send|sned|share|give|show)?\s*(?:me\s+)?(?:the\s+)?"
+ r"(?:(?:send|sned|share|give|show)?\s*(?:me\s+)?(?:the\s+)?"
r"(?:links?|urls?|sources?)"
r"(?:\s+(?:for|to|from)\s+(?:those|that|them|these|it|this|the\s+(?:sites?|websites?|resources?|sources?)))?"
r"|(?:for|to|from)\s+(?:those|that|them|these|it|this|the\s+(?:sites?|websites?|resources?|sources?)))"
- r"\s*(?:please|pls)?[.!?]?\s*",
- str(text or "").lower(),
+ r"\s*(?:please|pls)?[.!?]?",
+ str(text or "").lower().strip(),
)
)
@@ -10712,27 +11056,29 @@ def _parse_inspection_file_replacement(text: str) -> Optional[dict[str, str]]:
value,
re.IGNORECASE,
)
+ action_values: tuple[str | None, str | None] | None = (
+ (action_match.group("old"), action_match.group("new"))
+ if action_match is not None
+ else None
+ )
if action_match is None:
# Locate the explicit mutation clause after stripping the inspection
# language. Stop values before common trailing verification instructions.
- action_match = re.search(
- r"\b(?:chang(?:e|es|ed|ing)|updat(?:e|es|ed|ing))\s+"
- r"(?P.+?)\s+to\s+(?P.+?)"
- r"(?=\s+(?:in|and|then|before)\b|[.;]|$)",
- value,
- re.IGNORECASE,
- )
- if action_match is None:
- action_match = re.search(
- r"\breplace\s+(?P.+?)\s+with\s+(?P.+?)"
- r"(?=\s+(?:in|and|then|before)\b|[.;]|$)",
+ action_values = _captures_after_first_prefix(
value,
- re.IGNORECASE,
+ r"\b(?:chang(?:e|es|ed|ing)|updat(?:e|es|ed|ing))(?=\s)",
+ r"\s+(.+?)\s+to\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)",
)
- if action_match is None:
+ if action_values is None:
+ action_values = _captures_after_first_prefix(
+ value,
+ r"\breplace(?=\s)",
+ r"\s+(.+?)\s+with\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)",
+ )
+ if action_values is None:
return None
- old = _clean_file_edit_value(action_match.group("old"))
- new = _clean_file_edit_value(action_match.group("new"))
+ old = _clean_file_edit_value(action_values[0])
+ new = _clean_file_edit_value(action_values[1])
if not old or not new or old == new:
return None
return {"path": path, "old_string": old, "new_string": new}
@@ -11295,6 +11641,24 @@ def _recent_mentioned_email_reference(messages: List[Dict]) -> dict[str, str]:
return {}
+_ASSISTANT_PROMPT_RE = re.compile(
+ r"(?:Want me|Would you like|Should I)\b", re.IGNORECASE
+)
+
+
+def _split_before_assistant_prompt(text: str) -> str:
+ """Return text before the first prompt introduced on a later line."""
+ for match in _ASSISTANT_PROMPT_RE.finditer(text):
+ whitespace_start = match.start()
+ while whitespace_start and text[whitespace_start - 1].isspace():
+ whitespace_start -= 1
+ whitespace = text[whitespace_start:match.start()]
+ newline = whitespace.find("\n")
+ if newline >= 0:
+ return text[:whitespace_start + newline]
+ return text
+
+
def _suggested_reply_from_recent_assistant(messages: List[Dict]) -> str:
"""Extract the most recent assistant-suggested email reply body."""
for message in reversed(messages or []):
@@ -11308,7 +11672,7 @@ def _suggested_reply_from_recent_assistant(messages: List[Dict]) -> str:
after = re.split(r"\*\*Suggested reply:\*\*|Suggested reply:", text, flags=re.IGNORECASE, maxsplit=1)
if len(after) < 2:
continue
- body_section = re.split(r"\n\s*(?:Want me|Would you like|Should I)\b", after[1], flags=re.IGNORECASE, maxsplit=1)[0]
+ body_section = _split_before_assistant_prompt(after[1])
lines: list[str] = []
for raw_line in body_section.splitlines():
line = re.sub(r"^\s*>\s?", "", raw_line).rstrip()
@@ -11343,9 +11707,9 @@ def _reply_draft_confirmation_block_from_recent_context(messages: List[Dict], te
def _email_account_selector_from_label(label: str) -> str:
value = str(label or "").strip()
- match = re.search(r"<([^>]+)>", value)
+ match = next(iter_angle_contents(value), None)
if match:
- return match.group(1).strip()
+ return match[2].strip()
return value
@@ -11665,8 +12029,8 @@ def _email_bulk_blocks_from_search_output(
continue
account = str(default_account or "").strip()
if not account:
- account_match = re.search(r"<([^>]+)>", str(row.get("account") or ""))
- account = account_match.group(1).strip() if account_match else str(row.get("account") or "").strip()
+ account_match = next(iter_angle_contents(str(row.get("account") or "")), None)
+ account = account_match[2].strip() if account_match else str(row.get("account") or "").strip()
by_account.setdefault(account, []).append(uid)
blocks: list[ToolBlock] = []
for account, uids in by_account.items():
@@ -11721,8 +12085,8 @@ def _named_email_row_from_recent_list_context(messages: List[Dict], text: str) -
uid = str(row.get("uid") or "").strip()
if not uid:
continue
- account_match = re.search(r"<([^>]+)>", str(row.get("account") or ""))
- account = account_match.group(1).strip() if account_match else str(row.get("account") or "").strip()
+ account_match = next(iter_angle_contents(str(row.get("account") or "")), None)
+ account = account_match[2].strip() if account_match else str(row.get("account") or "").strip()
ref = dict(row)
ref["uid"] = uid
ref["folder"] = "INBOX"
@@ -11822,8 +12186,8 @@ def _alternate_email_attachment_blocks_from_recent_context(messages: List[Dict])
continue
if sender not in downloaded_senders:
continue
- account_match = re.search(r"<([^>]+)>", str(row.get("account") or ""))
- account = account_match.group(1).strip() if account_match else ""
+ account_match = next(iter_angle_contents(str(row.get("account") or "")), None)
+ account = account_match[2].strip() if account_match else ""
blocks: list[ToolBlock] = []
for index, _name in enumerate(attachments):
args = {"uid": uid, "index": index}
@@ -12026,7 +12390,7 @@ def _recent_odysseus_anchor_refs(messages: List[Dict], history_session: Any = No
refs: dict[str, str] = {}
note_re = re.compile(r"#note-([0-9a-fA-F-]{8,64})")
event_re = re.compile(r"#event-([0-9a-fA-F-]{8,64})")
- event_link_re = re.compile(r"\[([^\]]+)\]\(#event-([0-9a-fA-F-]{8,64})\)")
+ event_link_target_re = re.compile(r"[0-9a-fA-F-]{8,64}")
task_re = re.compile(r"(?:#task-|Created task '[^']+' \(id:\s*)([0-9a-fA-F-]{8,64})")
document_re = re.compile(r"(?:#document-|doc_id['\"]?\s*[:=]\s*['\"]?)([0-9a-fA-F-]{8,64})")
memory_re = re.compile(r"(?:memory_id['\"]?\s*[:=]\s*['\"]?|Memory id:\s*)([0-9a-fA-F-]{8,64})", re.IGNORECASE)
@@ -12068,10 +12432,12 @@ def _recent_odysseus_anchor_refs(messages: List[Dict], history_session: Any = No
if note_match:
refs["note_id"] = note_match.group(1)
if "event_uid" not in refs:
- event_link_match = event_link_re.search(text)
+ event_link_match = next(iter_markdown_links(
+ text, target_prefix="#event-", target_re=event_link_target_re
+ ), None)
if event_link_match:
- refs["event_title"] = event_link_match.group(1).split(",", 1)[0].strip()
- refs["event_uid"] = event_link_match.group(2)
+ refs["event_title"] = event_link_match[2].split(",", 1)[0].strip()
+ refs["event_uid"] = event_link_match[3]
continue
event_match = event_re.search(text)
if event_match:
@@ -12097,6 +12463,78 @@ def _recent_odysseus_anchor_refs(messages: List[Dict], history_session: Any = No
return refs
+def _numbered_row_parenthesized_ids(text: str) -> list[str]:
+ """Extract ``1. label (id) —`` identifiers with forward delimiters."""
+ found: list[str] = []
+ value = str(text or "")
+ line_start = 0
+ consumed_until = 0
+ while line_start <= len(value):
+ line_end = value.find("\n", line_start)
+ if line_end < 0:
+ line_end = len(value)
+ if line_start < consumed_until:
+ line_start = line_end + 1
+ continue
+ pos = line_start
+ while pos < line_end and value[pos].isspace():
+ pos += 1
+ digit_start = pos
+ while pos < line_end and value[pos].isdigit():
+ pos += 1
+ if pos == digit_start or pos >= line_end or value[pos] != ".":
+ line_start = line_end + 1
+ continue
+ pos += 1
+ space_start = pos
+ while pos < len(value) and value[pos].isspace():
+ pos += 1
+ if pos == space_start:
+ line_start = line_end + 1
+ continue
+ body_start = pos
+ body_end = value.find("\n", body_start)
+ if body_end < 0:
+ body_end = len(value)
+ whitespace_start = body_start + 1
+ while whitespace_start <= body_end:
+ while whitespace_start < body_end and not value[whitespace_start].isspace():
+ whitespace_start += 1
+ if whitespace_start == body_end and (
+ body_end >= len(value) or not value[body_end].isspace()
+ ):
+ break
+ whitespace_end = whitespace_start
+ while whitespace_end < len(value) and value[whitespace_end].isspace():
+ whitespace_end += 1
+ if whitespace_end >= len(value) or value[whitespace_end] != "(":
+ if whitespace_end > body_end:
+ break
+ whitespace_start = whitespace_end
+ continue
+ opener = whitespace_end
+ capture_line_end = value.find("\n", opener + 1)
+ if capture_line_end < 0:
+ capture_line_end = len(value)
+ closer = value.find(")", opener + 1, capture_line_end)
+ if closer < 0:
+ break
+ if closer == opener + 1:
+ whitespace_start = closer + 1
+ continue
+ suffix = closer + 1
+ suffix_start = suffix
+ while suffix < len(value) and value[suffix].isspace():
+ suffix += 1
+ if suffix > suffix_start and suffix < len(value) and value[suffix] in "—-":
+ found.append(value[opener + 1:closer])
+ consumed_until = suffix + 1
+ break
+ whitespace_start = closer + 1
+ line_start = line_end + 1
+ return found
+
+
def _ordinal_collection_mutation_target(
user_text: str,
messages: List[Dict],
@@ -12164,14 +12602,7 @@ def _ordinal_collection_mutation_target(
continue
output = str(event.get("output") or "")
if family == "tasks":
- identifiers = [
- found.strip()
- for found in re.findall(
- r"^\s*\d+\.\s+.+?\s+\(([^)\n]+)\)\s+[—-]",
- output,
- re.MULTILINE,
- )
- ]
+ identifiers = [found.strip() for found in _numbered_row_parenthesized_ids(output)]
else:
identifiers = re.findall(r"\]\(#event-([A-Za-z0-9_-]+)\)", output)
if 1 <= index <= len(identifiers):
@@ -12471,6 +12902,28 @@ def _is_generic_email_reply_body(body: str) -> bool:
}
+_SUMMARY_PREFIX_RE = re.compile(
+ r"\bsummary\s*(?:\*\*)?\s*:?\s*", re.IGNORECASE
+)
+
+
+def _contextual_summary_fragment(text: str) -> str:
+ """Preserve the legacy nonempty summary capture with bounded scans.
+
+ The legacy overescaped backslash alternative accepts an empty suffix,
+ so every newline terminates the capture. Bound the greedy prefix before
+ the last possible body character to preserve its whitespace backtracking.
+ """
+ last_newline = text.rfind("\n")
+ if last_newline < 1:
+ return ""
+ prefix = _SUMMARY_PREFIX_RE.search(text, 0, last_newline - 1)
+ if prefix is None:
+ return ""
+ newline = text.find("\n", prefix.end() + 1)
+ return text[prefix.end():newline] if newline >= 0 else ""
+
+
def _contextual_reply_body_from_recent_email_context(messages: List[Dict]) -> str:
"""Build a bounded draft body from the latest assistant email summary.
@@ -12486,7 +12939,7 @@ def _contextual_reply_body_from_recent_email_context(messages: List[Dict]) -> st
subject = ""
summary = ""
def _clean_fragment(raw: str) -> str:
- cleaned = re.sub(r"\[([^\]]+)\]\([^)]+\)", r"\1", str(raw or ""))
+ cleaned = replace_markdown_links_with_labels(str(raw or ""))
cleaned = re.sub(r"[*_`>#]+", "", cleaned)
return re.sub(r"\s+", " ", cleaned).strip(" .[]\"'")
@@ -12501,13 +12954,9 @@ def _contextual_reply_body_from_recent_email_context(messages: List[Dict]) -> st
)
if subject_match:
subject = _clean_fragment(subject_match.group(1))
- summary_match = re.search(
- r"\bsummary\s*(?:\*\*)?\s*:?\s*(.+?)(?:\n\s*(?:-|\\*\\*|If you|Want me|This is|$))",
- text,
- re.IGNORECASE | re.DOTALL,
- )
- if summary_match:
- summary = _clean_fragment(summary_match.group(1))
+ summary_fragment = _contextual_summary_fragment(text)
+ if summary_fragment:
+ summary = _clean_fragment(summary_fragment)
if not subject and not summary:
continue
if subject:
@@ -12947,17 +13396,46 @@ def _normalize_ody_qwen_text_artifacts(text: str, *, strip_edges: bool = True) -
_ODY_QWEN_LEAKED_TOOL_TEXT_RE = re.compile(
r"(<\s*/?\s*(?:function|parameter|tool_call)\b"
- r"|(?:^|\n)\s*(?:function|parameter)\s*="
r"|\bmanage_(?:notes|calendar|memory|documents|contact)\s*\("
r"|\"function\"\s*:\s*\"(?:manage_|mcp__)"
r"|mcp__email__"
- r"|(?:^|\n)\s*(?:web_search|web_fetch|private_browser)\s*:)",
+ r")",
+ re.IGNORECASE,
+)
+_ODY_QWEN_LINE_TOOL_TEXT_RE = re.compile(
+ r"(?:function|parameter)\s*=|(?:web_search|web_fetch|private_browser)\s*:",
re.IGNORECASE,
)
def _looks_like_ody_qwen_leaked_tool_text(text: str) -> bool:
- return bool(_ODY_QWEN_LEAKED_TOOL_TEXT_RE.search(text or ""))
+ value = str(text or "")
+ if _ODY_QWEN_LEAKED_TOOL_TEXT_RE.search(value):
+ return True
+ line_start = 0
+ while line_start <= len(value):
+ candidate = line_start
+ while candidate < len(value) and value[candidate].isspace():
+ candidate += 1
+ if _ODY_QWEN_LINE_TOOL_TEXT_RE.match(value, candidate):
+ return True
+ newline = value.find("\n", candidate)
+ if newline < 0:
+ return False
+ line_start = newline + 1
+ return False
+
+
+def _contains_email_draft_headers(text: str) -> bool:
+ """Recognize the legacy To/Subject/footer shape with ordered fixed scans."""
+ to_match = re.search(r"\bTo:", text, re.IGNORECASE)
+ if to_match is None:
+ return False
+ subject = re.search(r"\bSubject:", text[to_match.end() + 1:], re.IGNORECASE)
+ if subject is None:
+ return False
+ subject_end = to_match.end() + 1 + subject.end()
+ return text.find("\n---", subject_end + 1) >= 0
def _ody_qwen_terminal_tool_summary(tool_event: dict[str, Any], user_text: str = "") -> str:
@@ -13003,7 +13481,7 @@ def _ody_qwen_terminal_tool_summary(tool_event: dict[str, Any], user_text: str =
if tool_name in {"update_document", "edit_document"}:
lowered = output.lower()
if "document updated" in lowered or "edit applied" in lowered or "updated" in lowered:
- if re.search(r"\bTo:\s*.+\bSubject:\s*.+\n---", command, re.IGNORECASE | re.DOTALL):
+ if _contains_email_draft_headers(command):
return "Updated the active email draft."
return "Updated the active document."
return output.removeprefix("AI: ").strip()
@@ -13224,9 +13702,11 @@ def _tui_coding_failure_summary(tool_events: list[dict[str, Any]]) -> str:
_DESTRUCTIVE_REQUEST_RE = re.compile(
- r"\b(delete|remove|archive|trash|send|reply|unsubscribe|mark\s+.*read)\b",
+ r"\b(delete|remove|archive|trash|send|reply|unsubscribe)\b",
re.IGNORECASE,
)
+_MARK_REQUEST_RE = re.compile(r"\bmark", re.IGNORECASE)
+_READ_WORD_RE = re.compile(r"read\b", re.IGNORECASE)
_FAKE_SUCCESS_RE = re.compile(
r"\b(done|removed|deleted|sent|archived|unsubscribed|marked)\b",
@@ -13235,7 +13715,25 @@ _FAKE_SUCCESS_RE = re.compile(
def _looks_like_destructive_request(text: str) -> bool:
- return bool(_DESTRUCTIVE_REQUEST_RE.search(text or ""))
+ value = str(text or "")
+ if _DESTRUCTIVE_REQUEST_RE.search(value):
+ return True
+ reads = list(_READ_WORD_RE.finditer(value))
+ read_index = 0
+ for mark in _MARK_REQUEST_RE.finditer(value):
+ cursor = mark.end()
+ if cursor >= len(value) or not value[cursor].isspace():
+ continue
+ while cursor < len(value) and value[cursor].isspace():
+ cursor += 1
+ while read_index < len(reads) and reads[read_index].start() < cursor:
+ read_index += 1
+ newline = value.find("\n", cursor)
+ if read_index < len(reads) and (
+ newline < 0 or reads[read_index].start() < newline
+ ):
+ return True
+ return False
def _looks_like_success_claim(text: str) -> bool:
@@ -17272,13 +17770,19 @@ def _private_browser_product_query(user_text: str) -> str:
text = re.sub(r"\s+", " ", str(user_text or "")).strip()
match = re.search(
r"\b(?:find|look\s+for|shop\s+for|search\s+for)\s+"
- r"(?:me\s+)?(?:the\s+)?(?:best\s+)?(?P.+?)\s*[?.!]*$",
+ r"(?:me\s+)?(?:the\s+)?(?:best\s+)?",
text,
re.IGNORECASE,
)
- if not match:
+ if not match or match.end() >= len(text):
return ""
- query = match.group("query").strip(" \t\r\n.,!?;:")
+ raw_query = text[match.end():]
+ query_end = len(raw_query)
+ while query_end and raw_query[query_end - 1] in "?.!":
+ query_end -= 1
+ if query_end == 0:
+ query_end = 1
+ query = raw_query[:query_end].strip(" \t\r\n.,!?;:")
query = re.sub(
r"\s+(?:on|at|from)\s+(?:the\s+)?[A-Za-z0-9&.' -]{1,60}$",
"",
@@ -18927,7 +19431,7 @@ def _read_only_shell_command(content: str) -> bool:
return False
# Shell pipelines are allowed only when every stage is one of the common
# inspection commands. This intentionally rejects unknown/mutating syntax.
- segments = re.split(r"\s*(?:&&|\|\||;|\|)\s*", text)
+ segments = re.split(r"&&|\|\||;|\|", text)
if not segments or any(not segment.strip() for segment in segments):
return False
allowed = re.compile(
@@ -23084,11 +23588,7 @@ async def stream_agent_loop(
# Preserve that generic execution floor unless the task is
# explicitly URL-backed (filtered below).
if (
- re.search(
- r"/workspace/[^\s`\"']+\.(?:py|pyw|sh|bash|js|mjs|ts|rb|pl)\b",
- _last_user,
- re.IGNORECASE,
- )
+ _mentions_workspace_script(_last_user)
or (
len(_workspace_artifacts) >= 2
and any(Path(path).suffix.casefold() in {".csv", ".json", ".xlsx"} for path in _workspace_artifacts)
@@ -25826,10 +26326,11 @@ async def stream_agent_loop(
client_runtime_context=client_runtime_context,
)
except Exception as _preemptive_exc:
- logger.warning("Preemptive calendar lookup failed: %s", _preemptive_exc)
+ logger.warning("Preemptive calendar lookup failed: %s", _preemptive_exc, exc_info=True)
_preemptive_desc = "manage_calendar: ERROR"
_preemptive_result = {
- "error": str(_preemptive_exc),
+ "error": "The calendar lookup failed unexpectedly. Check the server log and retry.",
+ "error_category": "tool_execution_error",
"exit_code": 1,
"output": "",
}
@@ -25856,6 +26357,8 @@ async def stream_agent_loop(
}
if isinstance(_preemptive_result, dict) and isinstance(_preemptive_result.get("events"), list):
_preemptive_tool_output["events"] = _preemptive_result.get("events")
+ if isinstance(_preemptive_result, dict) and _preemptive_result.get("error_category") == "tool_execution_error":
+ _preemptive_tool_output["error_category"] = "tool_execution_error"
yield f"data: {json.dumps(_preemptive_tool_output)}\n\n"
_preemptive_tool_event = {
"round": round_num,
@@ -26023,10 +26526,11 @@ async def stream_agent_loop(
client_runtime_context=client_runtime_context,
)
except Exception as _preemptive_exc:
- logger.warning("Preemptive explicit tool failed: %s", _preemptive_exc)
+ logger.warning("Preemptive explicit tool failed: %s", _preemptive_exc, exc_info=True)
_preemptive_desc = f"{_preemptive_tool}: ERROR"
_preemptive_result = {
- "error": str(_preemptive_exc),
+ "error": "The requested tool failed unexpectedly. Check the server log and retry.",
+ "error_category": "tool_execution_error",
"exit_code": 1,
"output": "",
}
@@ -26051,6 +26555,8 @@ async def stream_agent_loop(
),
"fallback": "preemptive_explicit_admin_session",
}
+ if isinstance(_preemptive_result, dict) and _preemptive_result.get("error_category") == "tool_execution_error":
+ _preemptive_tool_output["error_category"] = "tool_execution_error"
yield f"data: {json.dumps(_preemptive_tool_output)}\n\n"
_preemptive_tool_event = {
"round": round_num,
@@ -29261,15 +29767,12 @@ async def stream_agent_loop(
):
_oversized_svg = _extract_oversized_svg(round_response)
if _oversized_svg:
- _svg_title_match = re.search(
- r"]*)?>([\s\S]*?)",
- _oversized_svg,
- re.IGNORECASE,
+ _svg_title_content = first_tag_content(
+ _oversized_svg, "title", allow_attributes=True
)
- _svg_title = re.sub(
- r"<[^>]*>",
- "",
- _svg_title_match.group(1) if _svg_title_match else "Visual explanation",
+ _svg_title = strip_angle_tags(
+ _svg_title_content if _svg_title_content is not None else "Visual explanation",
+ allow_empty=True,
).strip()[:100] or "Visual explanation"
full_response = _drop_rejected_round_response(full_response, round_response)
round_response = ""
@@ -29647,12 +30150,7 @@ async def stream_agent_loop(
and _local_media_turn
and not _artifact_creation_requested
and not _local_media_detail_nudge_sent
- and re.search(
- r"\b(?:how\s+many|count|break\s*points?|timestamps?|what\s+time|"
- r"when\s+.*(?:end|happen)|score(?:board)?s?)\b",
- _last_user,
- re.IGNORECASE,
- )
+ and contains_detailed_sequence_request(_last_user, include_first=False)
):
_successful_media_inspections = sum(
1
@@ -30357,7 +30855,11 @@ async def stream_agent_loop(
_completion_requirements,
).evaluate().can_complete
)
- _intent_match = _INTENT_RE.search(_intent_text) if _intent_text else None
+ # Whole-response intent matching is redundant: the helper below
+ # keeps short responses below 400 characters and long responses
+ # to their final 600-character window. Avoid an unbounded scan of
+ # model text before applying that existing policy boundary.
+ _intent_match = None
# Inspect only the bounded tail of long answers. This catches
# substantial multimodal analyses that end in "let me inspect..."
# or a dangling answer lead-in while leaving completed answers
@@ -32032,9 +32534,7 @@ async def stream_agent_loop(
_document_args = None
if isinstance(_document_args, dict):
_raw_document_action = str(_document_args.get("action") or "").strip()
- _document_action = re.sub(
- r"<[^>]+>",
- "",
+ _document_action = strip_angle_tags(
_raw_document_action.splitlines()[0] if _raw_document_action else "",
).strip().lower()
if _raw_document_action and _document_action != _raw_document_action.lower():
@@ -36732,10 +37232,8 @@ async def stream_agent_loop(
# prose. Local finetunes may emit those before the parser catches and
# executes them; saved history should contain only the user-facing answer.
full_response = _visible_response_text(full_response)
- if re.match(r"^Done\b", full_response, re.IGNORECASE) and re.search(
- r"\s*Done\.\s*$", full_response, re.IGNORECASE
- ):
- without_trailing_done = re.sub(r"\s*Done\.\s*$", "", full_response, flags=re.IGNORECASE).rstrip()
+ if re.match(r"^Done\b", full_response, re.IGNORECASE):
+ without_trailing_done = _strip_trailing_done(full_response)
if without_trailing_done:
full_response = without_trailing_done
if _ody_qwen_finetune_model or _qwen38_tool_router:
diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py
index 6fc1cff0a..9666e430c 100644
--- a/src/clean_agent_preview.py
+++ b/src/clean_agent_preview.py
@@ -32,7 +32,12 @@ from src.tool_schemas import (
normalized_native_function_argument_error,
)
from src.tool_types import ToolBlock
-from src.tool_parsing import parse_tool_blocks, strip_tool_blocks
+from src.tool_parsing import iter_email_addresses, parse_tool_blocks, strip_tool_blocks, strip_angle_tags
+from src.text_scanning import (
+ contains_detailed_sequence_request,
+ contains_search_engine_navigation,
+ iter_prefixed_token_matches,
+)
from src.turn_contract import (
_REQUEST_PREFIX,
calendar_retiming_request,
@@ -54,6 +59,59 @@ MODE = COMPACT_PREVIEW_MODE
class ProviderStreamError(Exception):
"""A provider reported failure inside an otherwise successful SSE response."""
+
+# Only these audited, fixed domain messages may cross the exception boundary.
+# Return the canonical constants rather than arbitrary exception text.
+_PUBLIC_PREVIEW_TOOL_ERRORS = {message: message for message in (
+ 'An equivalent search already returned evidence. Change the angle, missing subtopic, source type, or corroboration target instead of only changing freshness wording.',
+ 'Browser navigation already failed for this exact URL; use another source page.',
+ 'Do not infer image content repeatedly from filenames; use inspect_media on representative files, then continue from visual evidence.',
+ 'Equivalent search intent was repeated after a correction reminder; web_search is disabled for this turn.',
+ 'Look up the target with list_sessions before changing a chat. Use its exact returned ID; never invent last-chat/latest aliases. If the target is ambiguous, ask using the candidate chat titles.',
+ 'Resolve the named recipient with resolve_contact before drafting. Never invent an email address.',
+ 'Selection-only edits cannot use replace_all. Use a unique contextual FIND inside the selected passage.',
+ 'The bounded search-attempt budget is exhausted. Do not search again; answer from usable evidence already gathered, or clearly report what could not be verified and suggest a concrete next step.',
+ 'The calendar read has not succeeded yet. Obtain the requested calendar evidence before creating the dependent email draft.',
+ 'The latest note search returned no candidates, so this reference has no note to open. Do not reuse an older list item; report the empty result or ask which note was intended.',
+ 'The latest research search returned no candidates, so this reference has no report to open. Do not reuse an unrelated older report.',
+ 'The proposed expansion only adds placeholder/meta text. Write substantive content that continues the existing document’s subject, voice, and format; do not announce that a paragraph was added.',
+ 'The recipient address is not supported by the contact lookup. Use an exact returned address for the requested person; if no match exists, explain the missing recipient instead of guessing.',
+ 'The same artifact target was already rewritten three times; finish from the latest successful version instead of rewriting it again.',
+ 'These suggestions collapse multiple different passages into the same much shorter replacement, violating the request to preserve meaning. Produce passage-specific revisions that retain each source passage’s claims and intent.',
+ 'This exact call already failed twice and will not be executed again; change strategy or finish from existing evidence.',
+ 'This exact failed call was repeated after a correction reminder; the tool is disabled for this turn. Finish from existing evidence.',
+ 'This exact invalid call was repeated after two validation failures; the tool is disabled for this turn.',
+ 'This exact successful call already returned evidence. Do not repeat it; change the arguments or tool to gather different evidence, or finish from the evidence already available.',
+ 'This still image was already inspected and the same visual evidence is already in context. inspect_media is withheld for the next correction round; use that evidence, inspect a different file, or finish.',
+ 'This operation is outside the preview safety policy. No change was made.',
+ 'This replacement does not make its passage more concise. Shorten the wording while retaining its facts and meaning; a spelling-only change does not satisfy the requested action. Retry with shorter replacements.',
+ 'Tool arguments could not be converted for execution.',
+ 'Tool arguments must be a JSON object.',
+ 'Tool execution budget exhausted; finish from existing evidence.',
+ 'Tool is not offered or permitted.',
+ 'Two equivalent searches already returned no evidence. Do not repeat this search wording; use a different offered tool or a materially different query.',
+ 'Unresolved video target: this ID was not supplied by the user or observed in a successful tool result. Do not guess it from the title. Open/read the referenced browser link or call youtube_tool latest_channel_video for the observed channel, then use its returned ID.',
+ 'inspect_media exports only extract video stills and require an explicit timestamp plus output_path for every item. First inspect the video to find the timestamp; use write_file or python to author a diagram or other new artifact.',
+ 'web_fetch requires url or urls. query only filters a supplied page; it is not a search or writing request.',
+)}
+
+
+def _public_preview_tool_error(exc, *, execution_attempted=False):
+ """Keep useful domain guidance while withholding arbitrary diagnostics."""
+ if execution_attempted:
+ return 'The tool failed unexpectedly. Check the server log and retry.'
+ if isinstance(exc, json.JSONDecodeError):
+ return 'Tool arguments are not valid JSON. Correct the JSON object and retry.'
+ if isinstance(exc, jsonschema.ValidationError):
+ return 'Tool arguments do not match the required schema. Correct the call using the offered tool schema.'
+ detail = str(exc)
+ if detail.startswith('Artifact completion Python must reference the required '):
+ return 'Artifact completion Python must create non-empty output at the required artifact target.'
+ if detail.startswith('Shell access to credential variable '):
+ return 'Shell access to credentials is blocked. Use brokered native tools.'
+ return _PUBLIC_PREVIEW_TOOL_ERRORS.get(detail, 'The tool call could not be validated. Check its arguments and retry.')
+
+
# Native unattended workspaces routinely require several inspections followed
# by several artifact writes. The interactive preview keeps its six-call
# limit below; this larger budget applies only after server-side validation of
@@ -79,13 +137,6 @@ ARTIFACT_RESEARCH_TOOLS = frozenset({
'web_search', 'web_fetch', 'private_browser', 'pdf_extract', 'youtube_tool',
'inspect_media', 'extract_text', 'transcribe_media',
})
-DETAILED_VIDEO_REQUEST = re.compile(
- r"\b(?:how\s+many|count|sequence|in\s+order|chronological|"
- r"timestamps?|what\s+time|at\s+what\s+time|when\s+.*(?:end|happen)|"
- r"first\s+.*(?:save|attempt|event)|score(?:board)?s?)\b|"
- r"(?:多少|几次|何时|什么时候|时间|顺序)",
- re.IGNORECASE,
-)
READ_TOOLS = frozenset({
'manage_notes', 'manage_calendar', 'manage_memory', 'manage_skills', 'manage_tasks',
'manage_documents', 'manage_research', 'manage_contact', 'list_sessions',
@@ -809,13 +860,95 @@ def malformed_write_handoff_target(arguments, required_artifacts=(), user_text='
return target
+_PAGE_LISTING_WORDS = ("stories", "articles", "posts", "headlines", "pages")
+
+
+def _listing_allowed(char):
+ return char == " " or char in ".:/-" or char == "_" or char.isalnum()
+
+
+def _allowed_then_space(value):
+ """Whether value can be class+ followed by whitespace+, without retries."""
+ if len(value) < 2:
+ return False
+ first_disallowed = next((i for i, char in enumerate(value) if not _listing_allowed(char)), len(value))
+ if first_disallowed < len(value):
+ return first_disallowed > 0 and all(char.isspace() for char in value[first_disallowed:])
+ return value[-1].isspace()
+
+
+def _space_then_allowed(value):
+ """Whether value can be whitespace+ followed by the listing class+."""
+ if len(value) < 2:
+ return False
+ first_nonspace = next((i for i, char in enumerate(value) if not char.isspace()), len(value))
+ if first_nonspace < len(value):
+ return first_nonspace > 0 and all(_listing_allowed(char) for char in value[first_nonspace:])
+ trailing_spaces = len(value) - len(value.rstrip(" "))
+ return trailing_spaces > 0 and (len(value) > trailing_spaces or trailing_spaces >= 2)
+
+
+def _page_listing_request(user_text):
+ value = str(user_text or '').lstrip()
+ nonspace_end = len(value.rstrip())
+ if value[nonspace_end - 1:nonspace_end] in ("!", "?"):
+ value = value[:nonspace_end - 1]
+ lead = re.match(
+ r"(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)(?=\s)",
+ value, re.I,
+ )
+ if not lead:
+ return False
+ cursor = lead.end()
+ while cursor < len(value) and value[cursor].isspace():
+ cursor += 1
+ body = value[cursor:]
+ nonspace_end = len(body.rstrip())
+ terminal_end = nonspace_end
+ if body[terminal_end - 1:terminal_end] == ".":
+ terminal_end = len(body[:terminal_end - 1].rstrip())
+ first_disallowed = len(body)
+ last_disallowed = -1
+ first_nonspace_after_disallowed = len(body)
+ for index, char in enumerate(body):
+ if not _listing_allowed(char):
+ first_disallowed = min(first_disallowed, index)
+ if index < nonspace_end:
+ last_disallowed = index
+ if index >= first_disallowed and not char.isspace():
+ first_nonspace_after_disallowed = min(first_nonspace_after_disallowed, index)
+ last_literal_space = body.rfind(" ")
+ for word in _PAGE_LISTING_WORDS:
+ for candidate in re.finditer(word, body, re.I):
+ start = candidate.start()
+ prefix_allowed = start >= 2 and (
+ body[start - 1].isspace() if start <= first_disallowed
+ else first_disallowed > 0 and start <= first_nonspace_after_disallowed
+ )
+ if start == 0 or prefix_allowed:
+ after_start = candidate.end()
+ if after_start >= terminal_end:
+ return True
+ on = after_start
+ while on < len(body) and body[on].isspace():
+ on += 1
+ if on > after_start and body[on:on + 2].casefold() == "on":
+ tail_start = on + 2
+ tail_end = tail_start
+ while tail_end < len(body) and body[tail_end].isspace():
+ tail_end += 1
+ if len(body) - tail_start >= 2 and tail_end > tail_start:
+ if tail_end < len(body):
+ if last_disallowed < tail_end:
+ return True
+ elif last_literal_space > tail_start:
+ return True
+ return False
+
+
def page_listing_response(entries, user_text, max_items=10):
"""Render simple page listings from observed titles/URLs, never synthesized rankings."""
- if not re.fullmatch(
- r'\s*(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)\s+'
- r'(?:[\w .:/-]+\s+)?(?:stories|articles|posts|headlines|pages)'
- r'(?:\s+on\s+[\w .:/-]+)?[.!?]?\s*', user_text, re.I,
- ) or re.search(r'\b(?:and|compare|summarize|analyse|analyze|about|by|since|yesterday)\b', user_text, re.I):
+ if not _page_listing_request(user_text) or re.search(r'\b(?:and|compare|summarize|analyse|analyze|about|by|since|yesterday)\b', user_text, re.I):
return ''
from urllib.parse import quote, urlsplit
from html import escape
@@ -1531,17 +1664,24 @@ def prior_workspace_path_answer(user_text, history):
return ''
+def _prior_web_source_request(text):
+ text = str(text or '')
+ candidate = text.lstrip().rstrip()
+ candidate = candidate.rstrip(".!? ")
+ return bool(re.fullmatch(
+ r"(?:(?:where|what)\s+did\s+you\s+(?:get|find)\s+(?:that|this)\s+from[?., ]*"
+ r"(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link"
+ r"|(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link(?:\s+for\s+that)?"
+ r"|what(?:['’]?s|\s+is)\s+(?:the\s+)?source(?:\s+link)?)",
+ candidate,
+ re.I,
+ ))
+
+
def prior_web_source_answer(user_text, history):
"""Return the latest source URL for an explicit source-only follow-up."""
text = str(user_text or '')
- if not re.fullmatch(
- r"\s*(?:(?:where|what)\s+did\s+you\s+(?:get|find)\s+(?:that|this)\s+from[?., ]*"
- r"(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link[.!? ]*"
- r"|(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link(?:\s+for\s+that)?[.!? ]*"
- r"|what(?:['’]?s|\s+is)\s+(?:the\s+)?source(?:\s+link)?[.!? ]*)\s*",
- text,
- re.I,
- ):
+ if not _prior_web_source_request(text):
return ''
call_names = {}
candidates = []
@@ -2822,12 +2962,11 @@ def draft_contact_evidence_error(name, args, *, dependencies=(), executions=(),
and not e.get('blocked')]
if not observations:
return 'Resolve the named recipient with resolve_contact before drafting. Never invent an email address.'
- address_pattern = r'[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}'
known = {address.casefold() for e in observations
- for address in re.findall(address_pattern, str(e.get('output') or ''))}
- known.update(address.casefold() for address in re.findall(address_pattern, user_text))
+ for address in iter_email_addresses(str(e.get('output') or ''), ascii_only=True)}
+ known.update(address.casefold() for address in iter_email_addresses(user_text, ascii_only=True))
proposed = {address.casefold() for field in ('to', 'cc', 'bcc')
- for address in re.findall(address_pattern, str(args.get(field) or ''))}
+ for address in iter_email_addresses(str(args.get(field) or ''), ascii_only=True)}
if not proposed or not proposed <= known:
return ('The recipient address is not supported by the contact lookup. Use an exact '
'returned address for the requested person; if no match exists, explain '
@@ -3945,7 +4084,7 @@ def active_document_revision_quality_error(name, args, *, active_document, user_
if not incoming:
return None
addition = incoming[len(existing):].strip() if existing and incoming.startswith(existing) else ''
- candidate = re.sub(r'<[^>]+>', ' ', addition).strip()
+ candidate = strip_angle_tags(addition, ' ').strip()
candidate = re.sub(r'\s+', ' ', candidate)
if not candidate or candidate.casefold() in str(user_text or '').casefold():
return None
@@ -3970,8 +4109,8 @@ def document_suggestion_quality_error(name, args, *, user_text):
for suggestion in suggestions:
if not isinstance(suggestion, dict):
continue
- source = re.sub(r'\s+', ' ', re.sub(r'<[^>]*>', ' ', str(suggestion.get('find') or ''))).strip()
- result = re.sub(r'\s+', ' ', re.sub(r'<[^>]*>', ' ', str(suggestion.get('replace') or ''))).strip()
+ source = re.sub(r'\s+', ' ', strip_angle_tags(str(suggestion.get('find') or ''), ' ', allow_empty=True)).strip()
+ result = re.sub(r'\s+', ' ', strip_angle_tags(str(suggestion.get('replace') or ''), ' ', allow_empty=True)).strip()
if not source or not result:
continue
source_words = len(re.findall(r"\b[\w’'-]+\b", source))
@@ -4127,13 +4266,17 @@ _WORKSPACE_FILE_RE = re.compile(
r"/workspace/[^\s,,、;;`\"'<>]+\.[A-Za-z0-9]{1,12}",
re.I,
)
+_WORKSPACE_PREFIX_RE = re.compile(r"/workspace/", re.I)
+_WORKSPACE_FILE_TOKEN_TAIL_RE = re.compile(r"[^\s,,、;;`\"'<>]*")
def declared_workspace_artifacts(user_text):
"""Return explicit output paths, excluding paths used only as inputs."""
text = str(user_text or '')
paths = []
- for match in _WORKSPACE_FILE_RE.finditer(text):
+ for match in iter_prefixed_token_matches(
+ text, _WORKSPACE_PREFIX_RE, _WORKSPACE_FILE_RE, _WORKSPACE_FILE_TOKEN_TAIL_RE
+ ):
path = match.group(0).rstrip('.!?))]}')
if path.startswith('/workspace/fixtures/') or path in paths:
continue
@@ -4253,17 +4396,61 @@ def evidence_tool_keeps_distinct_requests_available(name):
}
+def _terminal_source_link_clause(text):
+ """Recognize the terminal short source/link clause from right to left."""
+ value = str(text or '')
+ def matches_at(end):
+ while end and value[end - 1] in ".!?":
+ end -= 1
+ while end and value[end - 1].isspace():
+ end -= 1
+ lowered = value[:end].casefold()
+ for courtesy in ("please", "pls"):
+ if lowered.endswith(courtesy):
+ boundary = end - len(courtesy)
+ end = boundary
+ while end and value[end - 1].isspace():
+ end -= 1
+ lowered = value[:end].casefold()
+ break
+ token_match = re.search(r"(?:sources?|citations?|links?)$", lowered)
+ if token_match is None:
+ return False
+ prefix = value[:token_match.start()]
+ original_cursor = len(prefix)
+ courtesy_end = original_cursor
+ while courtesy_end and prefix[courtesy_end - 1].isspace():
+ courtesy_end -= 1
+ cursors = [original_cursor]
+ for courtesy in ("please", "pls"):
+ start = courtesy_end - len(courtesy)
+ if start >= 0 and prefix[start:courtesy_end].casefold() == courtesy:
+ if courtesy_end < original_cursor and (start == 0 or prefix[start - 1].isspace()):
+ cursors.append(start)
+ break
+ for cursor in cursors:
+ while cursor and prefix[cursor - 1].isspace():
+ if prefix[cursor - 1] == "\n":
+ return True
+ cursor -= 1
+ if cursor == 0 or prefix[cursor - 1] in ".!?;,":
+ return True
+ return False
+
+ return matches_at(len(value)) or (value.endswith("\n") and matches_at(len(value) - 1))
+
+
def requested_web_source_links(user_text):
- return bool(re.search(
+ text = str(user_text or '')
+ return _terminal_source_link_clause(text) or bool(re.search(
r'\b(?:return|give|show|include|provide|cite|find)\b.{0,35}\b(?:source\s+)?links?\b'
- r'|(?:^|[.!?;,\n])\s*(?:(?:pls|please)\s+)?(?:sources?|citations?|links?)\s*(?:pls|please)?\s*[.!?]*$'
r'|\b(?:\d+|one|two|three|four|five)\s+(?:official\s+)?(?:source\s+)?links?\b'
r'|\bofficial\s+source\b'
r'|\b(?:with|include|provide|cite|show|give|find)\s+(?:the\s+)?(?:official\s+)?(?:sources|citations)\b'
r'|\blink\s+(?:to\s+)?(?:the\s+|your\s+)?(?:original\s+|official\s+)?(?:instructions|sources|documentation|articles?|reports?|studies|manuals?|guides?)\b'
r'|\b(?:find|locate|get|download)\b.{0,60}\bofficial\b.{0,60}\b(?:manual|guide|handbook|pdf|documentation)\b'
r'|\b(?:find|locate|get|download)\b.{0,80}\b(?:manual|guide|handbook|pdf)\b.{0,40}\b(?:online|official)\b',
- str(user_text or ''),
+ text,
re.IGNORECASE,
))
@@ -4300,12 +4487,67 @@ def unbound_lookup_reference(user_text, history, *, supplied_context=False):
))
+def _web_source_rows(text):
+ """Extract numbered source title/URL rows with monotonic line scans."""
+ rows = []
+ position = 0
+ length = len(text)
+ while position < length:
+ if position and text[position - 1] != "\n":
+ newline = text.find("\n", position)
+ if newline < 0:
+ break
+ position = newline + 1
+ continue
+ cursor = position
+ if cursor >= length or text[cursor] != "[":
+ newline = text.find("\n", cursor)
+ if newline < 0:
+ break
+ position = newline + 1
+ continue
+ cursor += 1
+ digit_start = cursor
+ while cursor < length and text[cursor].isdigit():
+ cursor += 1
+ if cursor == digit_start or cursor >= length or text[cursor] != "]":
+ position += 1
+ continue
+ cursor += 1
+ if cursor >= length or not text[cursor].isspace():
+ position += 1
+ continue
+ while cursor < length and text[cursor].isspace():
+ cursor += 1
+ title_start = cursor
+ title_line_end = text.find("\n", title_start)
+ if title_line_end < 0:
+ break
+ title_end = title_line_end
+ while title_end > title_start and text[title_end - 1].isspace():
+ title_end -= 1
+ url_start = title_line_end + 1
+ while url_start < length and text[url_start].isspace():
+ url_start += 1
+ scheme_length = 7 if text.startswith("http://", url_start) else 8 if text.startswith("https://", url_start) else 0
+ if title_end > title_start and scheme_length:
+ url_end = url_start + scheme_length
+ while url_end < length and not text[url_end].isspace():
+ url_end += 1
+ rows.append((text[title_start:title_end], text[url_start:url_end]))
+ position = url_end
+ continue
+ newline = text.find("\n", position)
+ if newline < 0:
+ break
+ position = newline + 1
+ return rows
+
+
def web_source_links(raw, *, max_items=1, prefer_official=False, query=''):
"""Extract stable title/URL pairs from the web tool's source preamble."""
text = str(raw or '')
- rows = re.findall(
- r'^\[\d+\]\s+(.+?)\s*\n\s*(https?://\S+)', text, re.MULTILINE,
- )
+ rows = _web_source_rows(text)
query_tokens = set(re.findall(r'[a-z0-9]+', str(query or '').casefold())) - {
'the', 'a', 'an', 'official', 'source', 'link', 'page', 'website',
'site', 'guide', 'search', 'find', 'for', 'return',
@@ -5834,7 +6076,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
provider_error = payload['error']
if isinstance(provider_error, dict):
provider_error = provider_error.get('message') or provider_error.get('detail')
- detail = str(provider_error or 'Unknown provider error').strip()[:300]
+ detail = str(provider_error or 'Unknown provider error').strip()
raise ProviderStreamError(detail)
usage = payload.get('usage') or {}
if usage:
@@ -6353,7 +6595,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
if (
native_workspace_enabled
and content
- and DETAILED_VIDEO_REQUEST.search(direct_user_text)
+ and contains_detailed_sequence_request(direct_user_text)
and successful_video_inspections == 1
and not media_detail_nudge_sent
and round_number < round_limit
@@ -7071,7 +7313,14 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
):
artifact_body_handoff_attempts += 1
artifact_body_handoff_target = handoff_target
- result = {'error': str(exc).splitlines()[0][:300], 'exit_code': 1}
+ logging.getLogger(__name__).warning(
+ 'Clean v3 tool call failed: %s', exc, exc_info=True,
+ )
+ result = {
+ 'error': _public_preview_tool_error(exc, execution_attempted=execution_attempted),
+ 'error_category': 'tool_execution_error' if execution_attempted else 'invalid_tool_arguments',
+ 'exit_code': 1,
+ }
if (canonical(block.tool_type if block is not None else name) == 'youtube_tool'
and result.get('exit_code') not in (None, 0)):
round_recovery_messages.append(
@@ -7260,12 +7509,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
requested_browser_url = private_browser_open_url(args)
effective_browser_url = private_browser_effective_url(result)
browser_search_url = requested_browser_url or effective_browser_url
- is_search_engine_navigation = bool(re.search(
- r'https?://(?:[^/]+\.)?(?:google\.[^/]+|bing\.com|duckduckgo\.com)'
- r'/(?:search|sorry|html|lite|\?)',
- browser_search_url,
- re.I,
- )) or bool(re.search(
+ is_search_engine_navigation = contains_search_engine_navigation(
+ browser_search_url
+ ) or bool(re.search(
r'https?://(?:[^/]+\.)?google\.[^/]+/sorry/',
effective_browser_url,
re.I,
@@ -7433,6 +7679,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
'execution_attempted': execution_attempted,
'blocked': policy_denied or schema is None,
'desc': desc, 'round': round_number}
+ for error_category in ('tool_execution_error', 'invalid_tool_arguments'):
+ if result.get('error_category') == error_category:
+ tool_event['error_category'] = error_category
reader_event = email_reader_event(direct_user_text, actual_tool, args, result, failed=failed)
if reader_event:
yield event(reader_event)
@@ -8005,9 +8254,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
else:
yield event({'delta': '\nThe preview reached its round limit. Please narrow the request.'})
except ProviderStreamError as exc:
- detail = f'The selected model provider failed while generating: {exc}'
- logging.getLogger(__name__).warning('Clean v3 provider stream failed: %s', exc)
- yield f'event: error\ndata: {json.dumps({"status": 502, "error": detail})}\n\n'
+ detail = 'The selected model provider failed while generating. Retry or choose another model.'
+ logging.getLogger(__name__).warning('Clean v3 provider stream failed: %s', exc, exc_info=True)
+ yield f'event: error\ndata: {json.dumps({"status": 502, "error": detail, "error_category": "provider_stream_error"})}\n\n'
return
except httpx.HTTPStatusError as exc:
status = exc.response.status_code
diff --git a/src/text_scanning.py b/src/text_scanning.py
new file mode 100644
index 000000000..59fad4cf5
--- /dev/null
+++ b/src/text_scanning.py
@@ -0,0 +1,210 @@
+"""Forward-only helpers for permissive text grammars.
+
+These helpers retain the existing regular expressions as anchored token
+parsers while preventing ``re.search`` from retrying the same token suffix at
+every embedded prefix.
+"""
+
+from __future__ import annotations
+
+import re
+from collections.abc import Iterator
+from typing import Match, Pattern
+
+
+def iter_prefixed_token_matches(
+ text: str,
+ candidate_re: Pattern[str],
+ anchored_re: Pattern[str],
+ token_tail_re: Pattern[str],
+) -> Iterator[Match[str]]:
+ """Yield legacy greedy matches after testing one prefix per token.
+
+ ``anchored_re`` must start with the same fixed prefix recognized by
+ ``candidate_re``. ``token_tail_re`` describes characters that the
+ anchored grammar can consume after that prefix. If the first prefix in
+ such a token cannot match, a later embedded prefix cannot match either:
+ its suffix was already available to the first attempt. Advancing to the
+ token boundary makes failed scans linear without changing successful
+ greedy captures.
+ """
+ pos = 0
+ while candidate := candidate_re.search(text, pos):
+ match = anchored_re.match(text, candidate.start())
+ if match is not None:
+ yield match
+ pos = match.end()
+ continue
+ tail = token_tail_re.match(text, candidate.end())
+ pos = max(candidate.end(), tail.end() if tail is not None else candidate.end())
+
+
+def has_prefixed_token_match(
+ text: str,
+ candidate_re: Pattern[str],
+ anchored_re: Pattern[str],
+ token_tail_re: Pattern[str],
+) -> bool:
+ """Return whether ``iter_prefixed_token_matches`` yields a match."""
+ return next(
+ iter_prefixed_token_matches(text, candidate_re, anchored_re, token_tail_re),
+ None,
+ ) is not None
+
+
+_HTTP_URL_PREFIX_RE = re.compile(r"https?://", re.IGNORECASE)
+_VIDEO_DETAIL_BASE_RE = re.compile(
+ r"\b(?:how\s+many|count|break\s*points?|timestamps?|what\s+time|score(?:board)?s?)\b",
+ re.IGNORECASE,
+)
+_VIDEO_DETAIL_EXTENDED_RE = re.compile(
+ r"\b(?:sequence|in\s+order|chronological|at\s+what\s+time)\b|"
+ r"(?:多少|几次|何时|什么时候|时间|顺序)",
+ re.IGNORECASE,
+)
+_WHEN_RE = re.compile(r"\bwhen\s+", re.IGNORECASE)
+_WHEN_TARGET_RE = re.compile(r"(?:end|happen)\b", re.IGNORECASE)
+_FIRST_RE = re.compile(r"\bfirst\s+", re.IGNORECASE)
+_FIRST_TARGET_RE = re.compile(r"(?:save|attempt|event)\b", re.IGNORECASE)
+
+
+def contains_search_engine_navigation(text: str) -> bool:
+ """Match the legacy Google/Bing/DuckDuckGo navigation URL grammar.
+
+ The old expression backtracked through every possible optional subdomain
+ split. Parsing the host up to its first slash gives the same accepted
+ hosts and path prefixes with one pass per URL candidate.
+ """
+ pos = 0
+ while candidate := _HTTP_URL_PREFIX_RE.search(text, pos):
+ host_start = candidate.end()
+ path_start = text.find("/", host_start)
+ if path_start < 0:
+ return False
+ host = text[host_start:path_start].casefold()
+ path = text[path_start + 1:path_start + 7].casefold()
+ google_at = host.rfind(".google.")
+ recognized_host = (
+ (host.startswith("google.") and len(host) > len("google."))
+ or (google_at >= 0 and google_at + len(".google.") < len(host))
+ or host == "bing.com"
+ or host.endswith(".bing.com")
+ or host == "duckduckgo.com"
+ or host.endswith(".duckduckgo.com")
+ )
+ if recognized_host and path.startswith(("search", "sorry", "html", "lite", "?")):
+ return True
+ # A later URL may begin in the path. Resume after this scheme rather
+ # than skipping the whole non-whitespace region.
+ pos = candidate.end()
+ return False
+
+
+def contains_detailed_sequence_request(text: str, *, include_first: bool = True) -> bool:
+ """Recognize count/order/timing requests without overlapping ``.*`` scans."""
+ value = str(text or "")
+ if _VIDEO_DETAIL_BASE_RE.search(value):
+ return True
+ if include_first and _VIDEO_DETAIL_EXTENDED_RE.search(value):
+ return True
+ pos = 0
+ while when := _WHEN_RE.search(value, pos):
+ line_end = value.find("\n", when.end())
+ if line_end < 0:
+ line_end = len(value)
+ if _WHEN_TARGET_RE.search(value, when.end(), line_end) is not None:
+ return True
+ pos = line_end + 1
+ if include_first:
+ pos = 0
+ while first := _FIRST_RE.search(value, pos):
+ line_end = value.find("\n", first.end())
+ if line_end < 0:
+ line_end = len(value)
+ if _FIRST_TARGET_RE.search(value, first.end(), line_end) is not None:
+ return True
+ pos = line_end + 1
+ return False
+
+
+def iter_angle_contents(text: str) -> Iterator[tuple[int, int, str]]:
+ """Yield nonempty flat ``<...>`` contents with monotonic delimiters."""
+ pos = 0
+ while (start := text.find("<", pos)) >= 0:
+ end = text.find(">", start + 1)
+ if end < 0:
+ return
+ if end > start + 1:
+ yield start, end + 1, text[start + 1:end]
+ pos = end + 1
+ else:
+ pos = start + 1
+
+
+def iter_markdown_links(
+ text: str,
+ *,
+ target_prefix: str = "",
+ target_re: Pattern[str] | None = None,
+) -> Iterator[tuple[int, int, str, str]]:
+ """Yield flat Markdown links accepted by the legacy link regexes."""
+ pos = 0
+ marker = "](" + target_prefix
+ while (start := text.find("[", pos)) >= 0:
+ label_end = text.find("]", start + 1)
+ if label_end < 0:
+ return
+ if label_end == start + 1:
+ pos = start + 1
+ continue
+ if not text.startswith(marker, label_end):
+ pos = label_end + 1
+ continue
+ target_start = label_end + len(marker)
+ target_end = text.find(")", target_start)
+ if target_end < 0:
+ return
+ target = text[target_start:target_end]
+ if target and (target_re is None or target_re.fullmatch(target)):
+ yield start, target_end + 1, text[start + 1:label_end], target
+ pos = target_end + 1
+ else:
+ pos = label_end + 1
+
+
+def replace_markdown_links_with_labels(text: str) -> str:
+ """Linear equivalent of replacing flat Markdown links with their labels."""
+ links = list(iter_markdown_links(text))
+ if not links:
+ return text
+ out = []
+ pos = 0
+ for start, end, label, _target in links:
+ out.extend((text[pos:start], label))
+ pos = end
+ out.append(text[pos:])
+ return "".join(out)
+
+
+def first_tag_content(text: str, tag: str, *, allow_attributes: bool = False) -> str | None:
+ """Return the first flat tag body using forward-only opener/closer scans."""
+ opener_re = re.compile(r"<" + re.escape(tag), re.IGNORECASE)
+ closer_re = re.compile(r"" + re.escape(tag) + r">", re.IGNORECASE)
+ pos = 0
+ while opener := opener_re.search(text, pos):
+ name_end = opener.end()
+ if name_end < len(text) and text[name_end] == ">":
+ body_start = name_end + 1
+ elif allow_attributes and name_end < len(text) and text[name_end].isspace():
+ tag_end = text.find(">", name_end + 1)
+ if tag_end < 0:
+ return None
+ body_start = tag_end + 1
+ else:
+ pos = name_end
+ continue
+ closer = closer_re.search(text, body_start)
+ if closer is None:
+ return None
+ return text[body_start:closer.start()]
+ return None
diff --git a/src/tool_parsing.py b/src/tool_parsing.py
index 5a528d6ae..c9c3bae3d 100644
--- a/src/tool_parsing.py
+++ b/src/tool_parsing.py
@@ -194,10 +194,33 @@ _TOOL_CODE_OPEN_RE = re.compile(r"\s*\{", re.IGNORECASE)
_TOOL_CODE_CLOSE_RE = re.compile(r"\}\s*", re.IGNORECASE)
# Pattern 4b: Gemma-style <|tool_call|> call:tool_name{args}
-_GEMMA_TOOL_CALL_RE = re.compile(
- r"<\|?tool_call\|?>\s*call:([\w\d_-]+)\s*(\{[\s\S]*?\})\s*<\|?tool_call\|?>",
+_GEMMA_TOOL_CALL_OPEN_RE = re.compile(
+ r"<\|?tool_call\|?>\s*call:([\w\d_-]+)\s*\{",
re.IGNORECASE,
)
+_GEMMA_TOOL_CALL_CLOSE_RE = re.compile(
+ r"\}\s*<\|?tool_call\|?>",
+ re.IGNORECASE,
+)
+
+# Native Qwen markup shares the same non-nesting delimiter grammar as the
+# XML helpers. Literal closers keep whitespace and opener floods linear;
+# values are stripped by the caller, as in the original regex path.
+_QWEN_FUNCTION_OPEN_RE = re.compile(r"\s*")
+_QWEN_FUNCTION_CLOSE_RE = re.compile(r"")
+_QWEN_PARAMETER_OPEN_RE = re.compile(r"\s*")
+_QWEN_PARAMETER_CLOSE_RE = re.compile(r"")
+_QWEN_PYTHON_ARG_KEY_RE = re.compile(r"[A-Za-z_]\w*")
+_QWEN_PYTHON_ARG_VALUE_RE = re.compile(r"\s*=\s*(['\"].*?['\"]|[^,]+)")
+_ANGLE_TAG_OPEN_RE = re.compile(r"<")
+_NONEMPTY_ANGLE_TAG_OPEN_RE = re.compile(r"<(?=[^>])")
+_ANGLE_TAG_CLOSE_RE = re.compile(r">")
+_EMAIL_LOCAL_RE = re.compile(r"[\w.+-]+")
+_EMAIL_ADDRESS_RE = re.compile(r"[\w.+-]+@[\w.-]+\.\w+")
+_ASCII_EMAIL_LOCAL_RE = re.compile(r"[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+")
+_ASCII_EMAIL_ADDRESS_RE = re.compile(
+ r"[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}"
+)
# Pattern 4c: Open-function wrapper emitted by some local MLX/Exo models.
# Example:
@@ -542,6 +565,34 @@ _PLAIN_UI_OPEN_PANEL_RE = re.compile(
r"((?:\s+(?:day|week|month|year|agenda)(?:\s+view)?(?:\s+\d{4}-\d{2}(?:-\d{2})?)?)?)"
r"\s*(?:`{1,3})?\s*$"
)
+_PLAIN_UI_CANDIDATE_RE = re.compile(r"ui_control", re.IGNORECASE)
+
+
+def _iter_plain_ui_open_panel(text: str):
+ """Yield line-anchored UI commands without retrying every blank line."""
+ pos = 0
+ while candidate := _PLAIN_UI_CANDIDATE_RE.search(text, pos):
+ line_start = text.rfind("\n", 0, candidate.start()) + 1
+ match = _PLAIN_UI_OPEN_PANEL_RE.match(text, line_start)
+ if match is not None:
+ yield match
+ pos = match.end()
+ continue
+ newline = text.find("\n", candidate.end())
+ pos = len(text) if newline < 0 else newline + 1
+
+
+def _strip_plain_ui_open_panel(text: str) -> str:
+ matches = list(_iter_plain_ui_open_panel(text))
+ if not matches:
+ return text
+ out = []
+ pos = 0
+ for match in matches:
+ out.append(text[pos:match.start()])
+ pos = match.end()
+ out.append(text[pos:])
+ return "".join(out)
# ---------------------------------------------------------------------------
@@ -992,6 +1043,43 @@ def _strip_raw_openai_tool_call_json(text: str) -> str:
return "".join(pieces)
+def iter_email_addresses(text: str, *, ascii_only: bool = False):
+ """Find legacy email-shaped strings without retrying every word suffix.
+
+ Match the address only at the start of each maximal local-part token.
+ When that attempt fails, every suffix has the same '@'/domain boundary
+ and must fail too. Local/domain character runs are each scanned a bounded
+ number of times, including malformed text with no '@' or domain dot.
+ """
+ local_re = _ASCII_EMAIL_LOCAL_RE if ascii_only else _EMAIL_LOCAL_RE
+ address_re = _ASCII_EMAIL_ADDRESS_RE if ascii_only else _EMAIL_ADDRESS_RE
+ pos = 0
+ while token := local_re.search(text, pos):
+ address = address_re.match(text, token.start())
+ if address is not None:
+ yield address.group(0)
+ pos = address.end()
+ else:
+ pos = token.end()
+
+
+def _iter_qwen_python_args(raw_args: str):
+ """Match each maximal key once, then its value at that fixed position.
+
+ If a word has no following equals sign, none of its suffixes can have one
+ either. Advancing past it avoids the old unanchored regex's O(n^2) retries
+ on a long malformed key, while preserving permissive quoted/bare values.
+ """
+ pos = 0
+ while key := _QWEN_PYTHON_ARG_KEY_RE.search(raw_args, pos):
+ value = _QWEN_PYTHON_ARG_VALUE_RE.match(raw_args, key.end())
+ if value is not None:
+ yield key.group(0), value.group(1)
+ pos = value.end()
+ else:
+ pos = key.end()
+
+
def _parse_qwen3_native_text_call(
text: str,
additional_tool_names: Optional[Iterable[str]] = None,
@@ -1018,13 +1106,14 @@ def _parse_qwen3_native_text_call(
# Qwen native text rendering:
# list...
- fn_match = re.search(r"\s*([\s\S]*?)\s*", text)
+ fn_match = next(_iter_named_blocks(
+ text, _QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE
+ ), None)
if fn_match:
- name, raw_body = fn_match.groups()
+ name, raw_body = fn_match
args = {}
- for key, raw_value in re.findall(
- r"\s*([\s\S]*?)\s*",
- raw_body,
+ for key, raw_value in _iter_named_blocks(
+ raw_body.rstrip(), _QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE,
):
value = raw_value.strip()
if value and value[0] in "[{\"":
@@ -1085,7 +1174,7 @@ def _parse_qwen3_native_text_call(
or normalized_name in declared_names
):
args = {}
- for key, raw_value in re.findall(r"([A-Za-z_]\w*)\s*=\s*(['\"].*?['\"]|[^,]+)", raw_args):
+ for key, raw_value in _iter_qwen_python_args(raw_args):
try:
args[key] = ast.literal_eval(raw_value.strip())
except (ValueError, SyntaxError):
@@ -1509,9 +1598,9 @@ def _iter_delimited(text, open_re, close_re):
pos = cm.end()
-def _strip_delimited(text: str, open_re, close_re) -> str:
- """Remove every ``open_re ... close_re`` span (forward-only; see
- _iter_delimited). Equivalent to ``open_re([\\s\\S]*?)close_re`` ``re.sub('')``
+def _strip_delimited(text: str, open_re, close_re, replacement: str = "") -> str:
+ """Replace every ``open_re ... close_re`` span (forward-only; see
+ _iter_delimited). Equivalent to ``open_re([\\s\\S]*?)close_re`` substitution
for these delimiters, without the O(n^2) rescan on unclosed openers."""
spans = list(_iter_delimited(text, open_re, close_re))
if not spans:
@@ -1520,11 +1609,23 @@ def _strip_delimited(text: str, open_re, close_re) -> str:
last = 0
for match_start, _inner_start, _inner_end, match_end in spans:
out.append(text[last:match_start])
+ out.append(replacement)
last = match_end
out.append(text[last:])
return "".join(out)
+def strip_angle_tags(text: str, replacement: str = "", *, allow_empty: bool = False) -> str:
+ """Linear equivalent of replacing ``<[^>]+>`` (or ``<[^>]*>``).
+
+ This preserves the existing flat text cleanup, including nested '<' and
+ malformed tails; it does not interpret HTML. Stop once no '>' is reachable
+ instead of retrying a suffix scan at every '<' in untrusted text.
+ """
+ opener = _ANGLE_TAG_OPEN_RE if allow_empty else _NONEMPTY_ANGLE_TAG_OPEN_RE
+ return _strip_delimited(text, opener, _ANGLE_TAG_CLOSE_RE, replacement)
+
+
def _iter_named_blocks(text, open_re, close_re):
"""Forward-only equivalent of ``open_re([\\s\\S]*?)close_re`` finditer where
open_re captures a name in group 1: yield ``(name, body)``, pairing each
@@ -1958,10 +2059,10 @@ def parse_tool_blocks(
# Pattern 4b: Gemma-style <|tool_call|> blocks
if not blocks:
- for m in _GEMMA_TOOL_CALL_RE.finditer(text):
- tool_name = m.group(1)
- body = m.group(2)
- block = _parse_gemma_tool_call(tool_name, body)
+ for tool_name, body in _iter_named_blocks(
+ text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE
+ ):
+ block = _parse_gemma_tool_call(tool_name, "{" + body + "}")
if block:
blocks.append(block)
@@ -1998,7 +2099,7 @@ def parse_tool_blocks(
# from weaker native-tool models after reading the tool docs but failing to
# emit the actual structured call.
if not blocks:
- m = _PLAIN_UI_OPEN_PANEL_RE.search(text)
+ m = next(_iter_plain_ui_open_panel(text), None)
if m:
blocks.append(ToolBlock("ui_control", f"open_panel {m.group(1).lower()}{m.group(2).lower()}".strip()))
@@ -2041,7 +2142,7 @@ def strip_tool_blocks(
cleaned = _strip_delimited(cleaned, _XML_TOOL_CALL_OPEN_RE, _XML_TOOL_CALL_CLOSE_RE)
cleaned = _XML_OPEN_TOOL_CALL_RE.sub('', cleaned)
cleaned = _strip_delimited(cleaned, _TOOL_CODE_OPEN_RE, _TOOL_CODE_CLOSE_RE)
- cleaned = _GEMMA_TOOL_CALL_RE.sub('', cleaned)
+ cleaned = _strip_delimited(cleaned, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE)
cleaned = _strip_delimited(cleaned, _FUNCTION_MODEL_OPEN_RE, _FUNCTION_MODEL_CLOSE_RE)
cleaned = _strip_raw_openai_tool_call_json(cleaned)
declared_xml_calls = _parse_declared_direct_xml_calls(
@@ -2060,7 +2161,7 @@ def strip_tool_blocks(
if raw_web_json:
_, (start, end) = raw_web_json
cleaned = cleaned[:start] + cleaned[end:]
- cleaned = _PLAIN_UI_OPEN_PANEL_RE.sub("", cleaned)
+ cleaned = _strip_plain_ui_open_panel(cleaned)
# Strip bare blocks not wrapped in
cleaned = _strip_bare_invoke_markup(cleaned)
cleaned = re.sub(r'\n{3,}', '\n\n', cleaned)
diff --git a/src/turn_contract.py b/src/turn_contract.py
index b04c58f17..87c87c7da 100644
--- a/src/turn_contract.py
+++ b/src/turn_contract.py
@@ -17,6 +17,7 @@ from typing import Iterable, Mapping
from src.action_intents import classify_tool_intent
from src.tool_policy import ToolPolicy
+from src.text_scanning import has_prefixed_token_match
FAMILY_TOOLS = {
@@ -47,6 +48,62 @@ FAMILY_TOOLS = {
CONTRACT_CORE_TOOLS = frozenset({
"bash", "python", "read_file", "web_search", "web_fetch", "ask_user",
})
+
+_WORKSPACE_PREFIX_RE = re.compile(r"/workspace/", re.I)
+_WORKSPACE_ARTIFACT_RE = re.compile(
+ r"/workspace/[^\s`\"']+\.(?:csv|html?|json|md|svg|txt)\b", re.I
+)
+_WORKSPACE_OUTPUT_PREFIX_RE = re.compile(r"/workspace/(?!input/)", re.I)
+_WORKSPACE_OUTPUT_RE = re.compile(
+ r"/workspace/(?!input/)[^\s`\"']+\."
+ r"(?:csv|html?|json|md|svg|txt|avif|bmp|gif|jpe?g|png|webp|pdf|mp4|webm)\b",
+ re.I,
+)
+_WORKSPACE_TOKEN_TAIL_RE = re.compile(r"[^\s`\"']*")
+
+
+def _mentions_workspace_artifact(text: str) -> bool:
+ return has_prefixed_token_match(
+ str(text or ""),
+ _WORKSPACE_PREFIX_RE,
+ _WORKSPACE_ARTIFACT_RE,
+ _WORKSPACE_TOKEN_TAIL_RE,
+ )
+
+
+def _mentions_workspace_output(text: str) -> bool:
+ return has_prefixed_token_match(
+ str(text or ""),
+ _WORKSPACE_OUTPUT_PREFIX_RE,
+ _WORKSPACE_OUTPUT_RE,
+ _WORKSPACE_TOKEN_TAIL_RE,
+ )
+
+
+def _mentions_under_budget(text: str) -> bool:
+ value = str(text or "")
+ for under in re.finditer(r"\bunder", value, re.I):
+ cursor = under.end()
+ if cursor >= len(value) or not value[cursor].isspace():
+ continue
+ while cursor < len(value) and value[cursor].isspace():
+ cursor += 1
+ if cursor < len(value) and value[cursor] in "¥$€£":
+ cursor += 1
+ while cursor < len(value) and value[cursor].isspace():
+ cursor += 1
+ digit_start = cursor
+ while cursor < len(value) and value[cursor].isdecimal():
+ cursor += 1
+ if cursor == digit_start:
+ continue
+ if cursor == len(value) or not (value[cursor].isalnum() or value[cursor] == "_"):
+ return True
+ if value[cursor:cursor + 3].casefold() == "yen":
+ cursor += 3
+ if cursor == len(value) or not (value[cursor].isalnum() or value[cursor] == "_"):
+ return True
+ return False
_FAMILY_WORDS = {
"calendar": r"\b(?:calendar|calender|events?|appointments?|meetings?|agenda)\b",
"notes": r"\b(?:notes?|checklists?|groceries|remind\s+me)\b",
@@ -122,13 +179,7 @@ def _normalize_request_lead(value: str) -> str:
text,
flags=re.I,
)
- text = re.sub(
- r"^(?:never\s*mind|scratch\s+that)\s*[,;:—–-]?\s*"
- r"(?=(?:open|show|list|read|search|find|check|switch|go)\b)",
- "",
- text,
- flags=re.I,
- )
+ text = _strip_cancelled_request_lead(text)
text = re.sub(r"^k(?:ay)?\s*[,!]?\s+(?=\S)", "", text, flags=re.I)
text = re.sub(
r"^(?:(?:great|nice|cool)\s*[,!.]|thanks?\s*[.!])\s+"
@@ -163,6 +214,21 @@ def _normalize_request_lead(value: str) -> str:
text = re.sub(r"^((?:can|could|would|will)\s+)u\b", r"\1you", text, flags=re.I)
text = re.sub(r"\boffical\b", "official", text, flags=re.I)
return text
+
+
+def _strip_cancelled_request_lead(text: str) -> str:
+ lead = re.match(r"^(?:never\s*mind|scratch\s+that)", text, re.I)
+ if lead is None:
+ return text
+ cursor = lead.end()
+ while cursor < len(text) and text[cursor].isspace():
+ cursor += 1
+ if cursor < len(text) and text[cursor] in ",;:—–-":
+ cursor += 1
+ while cursor < len(text) and text[cursor].isspace():
+ cursor += 1
+ action = re.match(r"(?:open|show|list|read|search|find|check|switch|go)\b", text[cursor:], re.I)
+ return text[cursor:] if action is not None else text
_MISSPELLED_RESEARCH_ACTION = re.compile(
r"^\s*" + _REQUEST_PREFIX + r"(?:reserch|reasearch|reseach)\b",
re.I,
@@ -195,6 +261,21 @@ _PURE_ACTION_PROHIBITION = re.compile(
r"[^.;\n]*[.!?]*\s*$",
re.I,
)
+
+
+def _is_pure_action_prohibition(text: str) -> bool:
+ value = str(text or "").strip()
+ prefix = re.match(
+ r"(?:read[- ]only(?:\s+and)?\s+)?(?:do\s+not|don['’]?t|never)\s+"
+ r"(?:add|create|make|write|draft|edit|change|update|delete|remove|send|reply|"
+ r"run|execute|download|serve|open|save|schedule|transcribe|inspect)\b",
+ value,
+ re.I,
+ )
+ if prefix is None:
+ return False
+ tail = value[prefix.end():].rstrip(".!?")
+ return not any(char in ".;\n" for char in tail)
_RETURN_TO_ACTION = re.compile(r"^\s*" + _REQUEST_PREFIX + r"return\s+to\b", re.I)
_PANEL_NAVIGATION = re.compile(
r"^\s*" + _REQUEST_PREFIX
@@ -447,6 +528,107 @@ _WARM_RECALL_WITH_FOLLOWUP = re.compile(
r"(?P(?:what(?:['’]?s|\s+is)?|which|who|where|when|how|show|open|read|list|find|search)\b[\s\S]{0,180})$",
re.I,
)
+
+_WARM_TARGET_RE = re.compile(
+ r"calendar|emails?|inbox|notes?|tasks?|skills?|memories|memory|"
+ r"documents?|docs?|web|browser|cookbook|files?|shell",
+ re.I,
+)
+_WARM_FOLLOWUP_RE = re.compile(
+ r"(?:what(?:['’]?s|\s+is)?|which|who|where|when|how|show|open|read|list|find|search)"
+ r"\b[\s\S]{0,180}\Z",
+ re.I,
+)
+
+
+def _consume_space(value: str, cursor: int, *, required: bool = False) -> int | None:
+ start = cursor
+ while cursor < len(value) and value[cursor].isspace():
+ cursor += 1
+ return None if required and cursor == start else cursor
+
+
+def _phrase_end(value: str, cursor: int, phrase: str) -> int | None:
+ for index, word in enumerate(phrase.split(" ")):
+ if value[cursor:cursor + len(word)].casefold() != word:
+ return None
+ cursor += len(word)
+ if index + 1 < len(phrase.split(" ")):
+ cursor = _consume_space(value, cursor, required=True)
+ if cursor is None:
+ return None
+ return cursor
+
+
+def _warm_recall_parts(value: str, *, with_followup: bool = False) -> tuple[str, str] | None:
+ value = str(value or "")
+ cursor = _consume_space(value, 0) or 0
+ states = [cursor]
+ discourse = re.match(r"(?:ok(?:ay)?|and|then)", value[cursor:], re.I)
+ if discourse is not None:
+ after = _consume_space(value, cursor + discourse.end(), required=True)
+ if after is not None:
+ states.insert(0, after)
+
+ action_phrases = ("back to", "return to", "what about", "check", "show", "open")
+ action_states: list[int] = []
+ for state in states:
+ if not with_followup:
+ action_states.append(state)
+ for phrase in action_phrases:
+ end = _phrase_end(value, state, phrase)
+ if end is None:
+ continue
+ if with_followup:
+ end = _consume_space(value, end, required=True)
+ if end is None:
+ continue
+ action_states.append(end)
+
+ for state in dict.fromkeys(action_states):
+ state = _consume_space(value, state) or 0
+ possessive_states = [state]
+ for possessive in ("my", "the"):
+ if value[state:state + len(possessive)].casefold() == possessive:
+ possessive_states.insert(0, state + len(possessive))
+ for target_state in possessive_states:
+ target_state = _consume_space(value, target_state) or 0
+ target = _WARM_TARGET_RE.match(value, target_state)
+ if target is None:
+ continue
+ target_text = target.group(0)
+ suffix = target.end()
+ if with_followup:
+ if suffix < len(value) and (value[suffix].isalnum() or value[suffix] == "_"):
+ continue
+ separator_states = []
+ spaced = _consume_space(value, suffix) or 0
+ if spaced > suffix:
+ separator_states.append(spaced)
+ if spaced < len(value) and value[spaced] in "-—,:;":
+ separator_states.append(_consume_space(value, spaced + 1) or 0)
+ for conjunction in ("and", "then"):
+ end = spaced + len(conjunction)
+ if (value[spaced:end].casefold() == conjunction
+ and (end == len(value) or not (value[end].isalnum() or value[end] == "_"))):
+ separator_states.append(_consume_space(value, end) or 0)
+ for followup_start in dict.fromkeys(separator_states):
+ followup = _WARM_FOLLOWUP_RE.match(value, followup_start)
+ if followup is not None:
+ return target_text, followup.group(0)
+ continue
+ suffix = _consume_space(value, suffix) or 0
+ suffix_states = [suffix]
+ for word in ("again", "now"):
+ if value[suffix:suffix + len(word)].casefold() == word:
+ suffix_states.insert(0, suffix + len(word))
+ for end in suffix_states:
+ while end < len(value) and value[end] in ".!?":
+ end += 1
+ end = _consume_space(value, end) or 0
+ if end == len(value):
+ return target_text, ""
+ return None
_REQUIRED_TOOLS = {
"calendar": "manage_calendar", "notes": "manage_notes",
"tasks": "manage_tasks", "skills": "manage_skills",
@@ -851,11 +1033,7 @@ def selected_tools_for_request(message: str) -> frozenset[str] | None:
tools = {"inspect_media"}
if (
re.search(r"\b(?:create|write|save|build|produce)\b", raw_text, re.I)
- and re.search(
- r"(?:file://)?/workspace/[^\s`\"']+\.(?:csv|html?|json|md|svg|txt)\b",
- raw_text,
- re.I,
- )
+ and _mentions_workspace_artifact(raw_text)
):
tools.update({"write_file", "read_file"})
if re.search(r"\b(?:preview|render|open)\b[^.\n]{0,100}\b(?:page|html|browser)\b", raw_text, re.I):
@@ -2118,6 +2296,31 @@ _READ_PRESENTATION_SUFFIX = re.compile(
)
+def _terminal_clause_match(text: str, core_pattern: str) -> re.Match[str] | None:
+ """Match a terminal clause after removing its ambiguous punctuation tail."""
+ end = len(text)
+ while end and text[end - 1].isspace():
+ end -= 1
+ while end and text[end - 1] in ".!?":
+ end -= 1
+ possessive = core_pattern.replace(r"\s+", r"\s++").replace(r"\s*", r"\s*+")
+ return re.search(possessive + r"$", text[:end], re.I)
+
+
+def _strip_terminal_but(text: str) -> str:
+ end = len(text)
+ while end and text[end - 1].isspace():
+ end -= 1
+ if end < 3 or text[end - 3:end].casefold() != "but":
+ return text
+ start = end - 3
+ if start == 0 or not text[start - 1].isspace():
+ return text
+ while start and text[start - 1].isspace():
+ start -= 1
+ return text[:start]
+
+
def _read_request_and_limit(message: str) -> tuple[str, int | None]:
"""Strip only whole, known presentation/safety suffixes, never actions."""
text = _normalize_request_lead(message)
@@ -2167,11 +2370,11 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
if keep_few_suffix:
maximum = 3
text = text[:keep_few_suffix.start()].strip()
- few_suffix = re.search(
+ few_suffix = _terminal_clause_match(
+ text,
r"[,.;?]\s*(?:(?:only|just)\s+)?(?:(?:list|show)\s+(?:me\s+)?)?a\s+few"
r"(?:\s+(?:task\s+)?(?:names?|items?|results?|entries?))?"
- r"(?:\s+and\s+(?:whether|if)\s+[^.;\n]+)?[.!?]*\s*$",
- text, re.I,
+ r"(?:\s+and\s+(?:whether|if)\s+[^.;\n]+)?",
)
if few_suffix:
maximum = 3
@@ -2183,42 +2386,43 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
text,
flags=re.I,
).strip()
- text = re.sub(
+ terminal_read_only = _terminal_clause_match(
+ text,
r"[.;]\s*read[- ]only(?:\s+(?:please|pls|plz))?\s*,?\s*"
r"(?:(?:and\s+)?(?:do\s+not|don['’]?t|dont)\s+"
- r"(?:change|edit|modify)(?:\s+or\s+send)?\s+(?:anything|data))?"
- r"[.!?]*\s*$",
- "",
- text,
- flags=re.I,
- ).strip()
+ r"(?:change|edit|modify)(?:\s+or\s+send)?\s+(?:anything|data))?",
+ )
+ if terminal_read_only:
+ text = text[:terminal_read_only.start()].strip()
# Explanatory/safety tails do not alter a preceding exact read request.
- text = re.sub(
+ safety_tail = _terminal_clause_match(
+ text,
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?(?:do\s+not|don['’]?t|dont)\s+"
- r"(?:touch|change|edit|modify)(?:\s+(?:anything|data|them))?(?:\s+yet)?[.!?]*\s*$",
- "", text, flags=re.I,
- ).strip()
+ r"(?:touch|change|edit|modify)(?:\s+(?:anything|data|them))?(?:\s+yet)?",
+ )
+ if safety_tail:
+ text = text[:safety_tail.start()].strip()
text = re.sub(
r"[.!?]\s*read[- ]only(?:\s+(?:please|pls|plz))?\s*,?\s*"
r"(?:do\s+not|don['’]?t|dont)\s+(?:change|edit|modify)\s+"
r"(?:or\s+send\s+)?anything[.!?]*\s*$",
"", text, flags=re.I,
).strip()
- text = re.sub(
- r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+changes?[.!?]*\s*$",
- "", text, flags=re.I,
- ).strip()
- text = re.sub(
- r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+edits?[.!?]*\s*$",
- "", text, flags=re.I,
- ).strip()
+ for terminal_core in (
+ r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+changes?",
+ r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+edits?",
+ ):
+ terminal_match = _terminal_clause_match(text, terminal_core)
+ if terminal_match:
+ text = text[:terminal_match.start()].strip()
text = re.sub(
r"[,;]\s*no\s+edits?[.!?]*\s*$", "", text, flags=re.I,
).strip()
- text = re.sub(
- r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+writes?[.!?]*\s*$",
- "", text, flags=re.I,
- ).strip()
+ terminal_no_write = _terminal_clause_match(
+ text, r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+writes?"
+ )
+ if terminal_no_write:
+ text = text[:terminal_no_write.start()].strip()
text = re.sub(
r"[.!?]\s*(?:i['’]?m|i\s+am)\s+(?:just\s+)?checking\b[^\n]*$",
"", text, flags=re.I,
@@ -2236,12 +2440,11 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
text,
flags=re.I,
).strip()
- text = re.sub(
- r"[,;]\s*(?:keep\s+(?:them|it)\s+)?short\s+lines?\s*,?[.!?]*\s*$",
- "",
- text,
- flags=re.I,
- ).strip()
+ short_lines = _terminal_clause_match(
+ text, r"[,;]\s*(?:keep\s+(?:them|it)\s+)?short\s+lines?\s*,?"
+ )
+ if short_lines:
+ text = text[:short_lines.start()].strip()
text = re.sub(
r"[,.;]\s*keep\s+(?:the\s+answer|it|them)\s+short[.!?]*\s*$",
"",
@@ -2318,7 +2521,7 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
value = int(raw) if raw.isdecimal() else _READ_COUNT_WORDS[raw.lower()]
maximum = value if maximum is None else min(maximum, value)
text = text[:conversational_limit.start()].rstrip(' ,.;?')
- text = re.sub(r"\s+but\s*$", "", text, flags=re.I)
+ text = _strip_terminal_but(text)
need_limit = re.search(
r"[.!?]\s*(?:i\s+)?only\s+need\s+(" + _READ_COUNT + r")"
r"(?:\s+(?:short\s+)?(?:titles?|items?|results?|entries?|names?|ones?))?"
@@ -3949,7 +4152,7 @@ def _families_for_tool(tool: str) -> frozenset[str]:
def _clause_capabilities(text: str) -> set[str]:
# A prohibition constrains authority; it must never grant the family named
# only as the forbidden side effect (for example, "do not create a file").
- if _PURE_ACTION_PROHIBITION.fullmatch(text):
+ if _is_pure_action_prohibition(text):
return set()
container_tool = creation_container_tool(text)
if container_tool:
@@ -4610,12 +4813,7 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum
raw_text,
re.I,
)
- and re.search(
- r"(?:file://)?/workspace/(?!input/)[^\s`\"']+\."
- r"(?:csv|html?|json|md|svg|txt|avif|bmp|gif|jpe?g|png|webp|pdf|mp4|webm)\b",
- raw_text,
- re.I,
- )
+ and _mentions_workspace_output(raw_text)
):
families.add("shell_files")
if re.search(
@@ -5074,7 +5272,7 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum
and re.search(r"\b(?:ones?|top|apps?|services?|providers?)\b", text, re.I)
and re.search(r"\b(?:price|cheap|under|dimensions?|quote|trustworthy|app)\b", text, re.I)
)
- or re.search(r"\bunder\s+[¥$€£]?\s*\d+(?:[.,]\d+)?(?:\s*yen)?\b", text, re.I)
+ or _mentions_under_budget(text)
):
return frozenset({"search_browser"})
if recent_family == ("search_browser",) and re.fullmatch(
@@ -6011,8 +6209,8 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum
elif (recent == ("search_browser",) and families == {"cookbook_admin"}
and re.match(r"^\s*" + _REQUEST_PREFIX + r"(?:search|find)\s+(?:those|them|these)\b", text, re.I)):
families = {"search_browser"}
- if not families and (recall := _WARM_RECALL.fullmatch(text)):
- target = recall["target"].lower()
+ if not families and (recall := _warm_recall_parts(text)):
+ target = recall[0].lower()
family = {
"email": "email", "emails": "email", "inbox": "email",
"note": "notes", "notes": "notes", "task": "tasks", "tasks": "tasks",
@@ -6024,8 +6222,8 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum
}[target]
if family in recently_executed_families(history):
families.add(family)
- if not families and (recall := _WARM_RECALL_WITH_FOLLOWUP.fullmatch(text)):
- target = recall["target"].lower()
+ if not families and (recall := _warm_recall_parts(text, with_followup=True)):
+ target = recall[0].lower()
family = {
"email": "email", "emails": "email", "inbox": "email",
"note": "notes", "notes": "notes", "task": "tasks", "tasks": "tasks",
diff --git a/tests/test_clean_agent_preview.py b/tests/test_clean_agent_preview.py
index 4cb1519eb..2526562b3 100644
--- a/tests/test_clean_agent_preview.py
+++ b/tests/test_clean_agent_preview.py
@@ -4562,7 +4562,9 @@ async def test_preview_provider_stream_error_is_terminal_not_empty_answer(monkey
disabled_tools=set(), tool_policy=ToolPolicy(),
)]
assert raw[-1].startswith('event: error\ndata: ')
- assert 'Qwen3_5MTPDraftModel' in raw[-1]
+ assert 'selected model provider failed' in raw[-1]
+ assert 'provider_stream_error' in raw[-1]
+ assert 'Qwen3_5MTPDraftModel' not in raw[-1]
assert all('returned no answer' not in chunk for chunk in raw)
assert all('"type": "metrics"' not in chunk for chunk in raw)
diff --git a/tests/test_redos_core_parsers.py b/tests/test_redos_core_parsers.py
new file mode 100644
index 000000000..7d5aad96f
--- /dev/null
+++ b/tests/test_redos_core_parsers.py
@@ -0,0 +1,859 @@
+"""Preserve legacy text-call/listing semantics and bound hostile scans."""
+
+import itertools
+import random
+import re
+import subprocess
+import sys
+import textwrap
+
+import pytest
+
+from src.agent_loop import (
+ _calendar_listing_row,
+ _captures_after_first_prefix,
+ _contains_email_draft_headers,
+ _looks_like_agent_reasoning_preamble,
+ _looks_like_ody_qwen_leaked_tool_text,
+ _email_account_label,
+ _email_sender_name,
+ _contextual_summary_fragment,
+ _is_terse_link_request,
+ _is_terse_email_lookup_followup,
+ _looks_like_destructive_request,
+ _looks_like_youtube_tool_turn,
+ _mentions_pdf_url,
+ _mentions_workspace_script,
+ _numbered_row_parenthesized_ids,
+ _pipeline_request_parts,
+ _parse_explicit_open_panel_request,
+ _private_browser_product_query,
+ _read_only_shell_command,
+ _remaining_checklist_name,
+ _research_listing_row,
+ _session_link_from_row,
+ _session_find_query_capture,
+ _session_list_summary_from_tool_output,
+ _split_before_assistant_prompt,
+ _split_note_items,
+ _strip_trailing_done,
+ _strip_horizontal_space_before_lf,
+ _skill_listing_row,
+)
+from src.clean_agent_preview import (
+ _page_listing_request,
+ _prior_web_source_request,
+ _terminal_source_link_clause,
+ _web_source_rows,
+ declared_workspace_artifacts,
+)
+from src.text_scanning import (
+ contains_detailed_sequence_request,
+ contains_search_engine_navigation,
+ iter_angle_contents,
+ iter_markdown_links,
+ first_tag_content,
+ replace_markdown_links_with_labels,
+)
+from src.turn_contract import (
+ _WARM_RECALL,
+ _WARM_RECALL_WITH_FOLLOWUP,
+ _is_pure_action_prohibition,
+ _mentions_under_budget,
+ _mentions_workspace_artifact,
+ _mentions_workspace_output,
+ _strip_cancelled_request_lead,
+ _strip_terminal_but,
+ _terminal_clause_match,
+ _warm_recall_parts,
+)
+from src.tool_parsing import (
+ _GEMMA_TOOL_CALL_OPEN_RE,
+ _GEMMA_TOOL_CALL_CLOSE_RE,
+ _QWEN_FUNCTION_OPEN_RE,
+ _QWEN_FUNCTION_CLOSE_RE,
+ _QWEN_PARAMETER_OPEN_RE,
+ _QWEN_PARAMETER_CLOSE_RE,
+ _iter_named_blocks,
+ _iter_qwen_python_args,
+ _strip_delimited,
+ parse_tool_blocks,
+ strip_tool_blocks,
+ strip_angle_tags,
+ iter_email_addresses,
+)
+
+_SESSION_LINK = re.compile(r"(\[(?:\\.|[^\]])+\]\(#session-[^)]+\))")
+_GEMMA = re.compile(
+ r"<\|?tool_call\|?>\s*call:([\w\d_-]+)\s*(\{[\s\S]*?\})\s*<\|?tool_call\|?>",
+ re.I,
+)
+_QWEN_FUNCTION = re.compile(r"\s*([\s\S]*?)\s*")
+_QWEN_PARAMETER = re.compile(r"\s*([\s\S]*?)\s*")
+_QWEN_ARGS = re.compile(r"([A-Za-z_]\w*)\s*=\s*(['\"].*?['\"]|[^,]+)")
+_WORKSPACE_SCRIPT = re.compile(
+ r"/workspace/[^\s`\"']+\.(?:py|pyw|sh|bash|js|mjs|ts|rb|pl)\b", re.I
+)
+_PDF_URL = re.compile(r"https?://\S+(?:\.pdf\b|/pdf/)", re.I)
+_WORKSPACE_ARTIFACT = re.compile(
+ r"(?:file://)?/workspace/[^\s`\"']+\.(?:csv|html?|json|md|svg|txt)\b", re.I
+)
+_WORKSPACE_OUTPUT = re.compile(
+ r"(?:file://)?/workspace/(?!input/)[^\s`\"']+\."
+ r"(?:csv|html?|json|md|svg|txt|avif|bmp|gif|jpe?g|png|webp|pdf|mp4|webm)\b",
+ re.I,
+)
+_SEARCH_ENGINE_URL = re.compile(
+ r"https?://(?:[^/]+\.)?(?:google\.[^/]+|bing\.com|duckduckgo\.com)"
+ r"/(?:search|sorry|html|lite|\?)",
+ re.I,
+)
+_LEAKED_TOOL_TEXT = re.compile(
+ r"(<\s*/?\s*(?:function|parameter|tool_call)\b|(?:^|\n)\s*(?:function|parameter)\s*="
+ r"|\bmanage_(?:notes|calendar|memory|documents|contact)\s*\(|\"function\"\s*:\s*\"(?:manage_|mcp__)"
+ r"|mcp__email__|(?:^|\n)\s*(?:web_search|web_fetch|private_browser)\s*:)", re.I,
+)
+_VIDEO_DETAIL_BASE = re.compile(
+ r"\b(?:how\s+many|count|break\s*points?|timestamps?|what\s+time|"
+ r"when\s+.*(?:end|happen)|score(?:board)?s?)\b", re.I,
+)
+_VIDEO_DETAIL_EXTENDED = re.compile(
+ r"\b(?:how\s+many|count|sequence|in\s+order|chronological|timestamps?|"
+ r"what\s+time|at\s+what\s+time|when\s+.*(?:end|happen)|"
+ r"first\s+.*(?:save|attempt|event)|score(?:board)?s?)\b|"
+ r"(?:多少|几次|何时|什么时候|时间|顺序)", re.I,
+)
+
+
+def test_session_link_matches_legacy_escape_and_greedy_semantics():
+ cases = [
+ "no links", "[](#session-id)", "[x](#session-)",
+ "[x](#session-id)", "[[x](#session-id)",
+ r"[x\](#session-id)", r"[x\]more](#session-id)",
+ r"[x\](#session-first)\](#session-last)",
+ r"[x\](#session-first)\](#session-)",
+ r"[x\](#session-first)\](#session-unclosed",
+ "[bad] then [good](#session-id)",
+ "[x](#session-id[has]brackets)",
+ "[x](#session-first) [y](#session-second)",
+ ]
+ rng = random.Random(6503)
+ tokens = ["[", "]", "\\", "a", "(", ")", "(#session-id)", "(#session-)"]
+ cases += ["".join(rng.choices(tokens, k=12)) for _ in range(2000)]
+ cases += ["[" + "".join(label) + "](#session-id)"
+ for n in range(5) for label in itertools.product("a[]\\", repeat=n)]
+ for row in cases:
+ expected = _SESSION_LINK.search(row)
+ assert _session_link_from_row(row) == (expected.group(0) if expected else ""), row
+
+
+@pytest.mark.parametrize("row,expected", [
+ ("- **[Chat](#session-id)** (id: `id`, model: qwen, 2 msgs, last active today)",
+ "- [Chat](#session-id) (last active today)"),
+ ("- [Chat](#session-id) (irrelevant) (model: qwen) (last active later)",
+ "- [Chat](#session-id)"),
+ ("- [Chat](#session-id) (outer (last active today))",
+ "- [Chat](#session-id) (last active today)"),
+ ("- plain row", "- plain row"),
+])
+def test_session_summary_preserves_link_and_metadata(row, expected):
+ assert _session_list_summary_from_tool_output("Chats:\n" + row) == "Chats:\n" + expected
+
+
+def test_gemma_delimiters_match_legacy_parse_and_strip():
+ cases = [
+ "ordinary prose", "<|tool_call|>call:web_search{query: 'news'}<|tool_call|>",
+ "before call:read-file {\npath: 'README.md'\n} after",
+ "call:x{a}call:y{b}",
+ "call:x{call:y{b}",
+ "call:x{unclosed", "}call:x{unclosed",
+ ]
+ for text in cases:
+ actual = [(name, "{" + body + "}") for name, body in _iter_named_blocks(
+ text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE
+ )]
+ assert actual == _GEMMA.findall(text)
+ assert _strip_delimited(text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE) == _GEMMA.sub("", text)
+ raw = '<|tool_call|>call:web_search{"query":"news"}<|tool_call|>'
+ assert [(b.tool_type, b.content) for b in parse_tool_blocks(raw)] == [("web_search", "news")]
+ assert strip_tool_blocks(raw) == ""
+
+
+@pytest.mark.parametrize("reference,opener,closer,tokens", [
+ (_QWEN_FUNCTION, _QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE,
+ ["", "", "a", "\n", "\t", " "]),
+ (_QWEN_PARAMETER, _QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE,
+ ["", "", "a", "\n", "\t", " "]),
+])
+def test_qwen_delimiters_preserve_names_and_stripped_values(reference, opener, closer, tokens):
+ rng = random.Random(6503)
+ for _ in range(1000):
+ text = "".join(rng.choices(tokens, k=16))
+ expected = [(name, body.strip()) for name, body in reference.findall(text)]
+ actual = [(name, body.strip()) for name, body in _iter_named_blocks(text, opener, closer)]
+ assert actual == expected
+
+
+def test_qwen_python_arguments_preserve_permissive_legacy_grammar():
+ cases = ["action='list', limit=5", "action = \"list\"", "1key=2", "aé=4",
+ "key='mismatched\"", "broken word, okay=2", "a=\t, b=3", "a=\n'hi'", "a='hi\nthere'"]
+ rng = random.Random(6503)
+ tokens = ["key", "1", "中", "é", "=", "\n", " ", "\t", "'", '"', ",", "-", "_", "[]"]
+ cases += ["".join(rng.choices(tokens, k=20)) for _ in range(2000)]
+ for text in cases:
+ assert list(_iter_qwen_python_args(text)) == _QWEN_ARGS.findall(text), text
+
+
+@pytest.mark.parametrize("text", [
+ "Done.", "Done.\nUpdated the document.\nDone.", "Done. undone.",
+ "Done.\nDONE.\t", "Done.\u2003Done.\u2003", "Done. no terminal marker ",
+])
+def test_trailing_done_matches_legacy_cleanup(text):
+ pattern = re.compile(r"\s*Done\.\s*$", re.I)
+ expected = pattern.sub("", text).rstrip() if pattern.search(text) else text
+ assert _strip_trailing_done(text) == expected
+
+
+@pytest.mark.parametrize("allow_empty", [False, True])
+@pytest.mark.parametrize("replacement", ["", " "])
+def test_angle_tag_cleanup_matches_legacy_flat_grammar(allow_empty, replacement):
+ pattern = re.compile(r"<[^>]*>" if allow_empty else r"<[^>]+>")
+ cases = ["beforeboldafter", "<>", "<<>>", "ab", "one", "a > b", "tail"]
+ cases += ["".join(parts) for n in range(6)
+ for parts in itertools.product("a<>\n", repeat=n)]
+ for text in cases:
+ assert strip_angle_tags(text, replacement, allow_empty=allow_empty) == pattern.sub(replacement, text)
+
+
+@pytest.mark.parametrize("ascii_only", [False, True])
+def test_email_scanner_preserves_legacy_address_sets(ascii_only):
+ pattern = re.compile(
+ r"[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}"
+ if ascii_only else r"[\w.+-]+@[\w.-]+\.\w+"
+ )
+ cases = ["a@b.example", "a+b@b.example, c@d.example", "a@b@c.example",
+ "a@b.c+d@e.f", "你好@例子.中国", "a@b.-.com", "'a'@example.com",
+ "a@b.x", "a@.com", "a@b...", "a@b.c@d.example"]
+ rng = random.Random(6503)
+ tokens = ["a", "b", "é", "中", "@", ".", "-", "+", " ", "_", "'", "!", "/"]
+ cases += ["".join(rng.choices(tokens, k=30)) for _ in range(3000)]
+ for text in cases:
+ assert list(iter_email_addresses(text, ascii_only=ascii_only)) == pattern.findall(text), text
+
+
+def test_prefixed_workspace_and_url_detectors_match_legacy_searches():
+ cases = [
+ "", "/workspace/a.py", "FILE:///WORKSPACE/report.HTML",
+ "/workspace/input/a.png", "/workspace/input/a.png/workspace/out.png",
+ "http://example.test/a.pdf", "http://x/http://example.test/pdf/view",
+ "https://google.com/search?q=x", "http://news.google.co.uk/sorry/index",
+ "https://x.bing.com/html", "http://bad/http://duckduckgo.com/?q=x",
+ ]
+ rng = random.Random(6503)
+ tokens = ["/workspace/", "input/", "file://", "http://", "https://", ".py",
+ ".pdf", ".html", "/pdf/", "google.", "bing.com/", "search", "x", " ", "'", "/"]
+ cases += ["".join(rng.choices(tokens, k=18)) for _ in range(4000)]
+ for text in cases:
+ assert _mentions_workspace_script(text) == bool(_WORKSPACE_SCRIPT.search(text)), text
+ assert _mentions_pdf_url(text) == bool(_PDF_URL.search(text)), text
+ assert _mentions_workspace_artifact(text) == bool(_WORKSPACE_ARTIFACT.search(text)), text
+ assert _mentions_workspace_output(text) == bool(_WORKSPACE_OUTPUT.search(text)), text
+ assert contains_search_engine_navigation(text) == bool(_SEARCH_ENGINE_URL.search(text)), text
+
+
+def test_declared_workspace_artifacts_keeps_legacy_greedy_paths():
+ pattern = re.compile(r"/workspace/[^\s,,、;;`\"'<>]+\.[A-Za-z0-9]{1,12}", re.I)
+ cases = [
+ "create /workspace/report.csv",
+ "read /workspace/input.csv and create /workspace/out.json",
+ "create /workspace/a.csv/workspace/b.json",
+ "create /workspace/noext /workspace/out.txt",
+ ]
+ for text in cases:
+ expected = []
+ for match in pattern.finditer(text):
+ path = match.group(0).rstrip(".!?))]}")
+ if path.startswith("/workspace/fixtures/") or path in expected:
+ continue
+ before = text[max(0, match.start() - 240):match.start()]
+ before = pattern.sub("[workspace file]", before)
+ clause = re.split(r"[.;!?\n]", before)[-1]
+ if re.search(
+ r"\b(?:from|using|inspect|read|open|analy[sz]e|transcribe|extract\s+(?:text\s+)?from|"
+ r"input(?:\s+file)?(?:\s+is)?|source(?:\s+file)?(?:\s+is)?)\s*(?::|=)?\s*$",
+ clause, re.I,
+ ) or re.search(r"\b(?:read_file|inspect_media|extract_text|transcribe_media|pdf_extract)\b", clause, re.I):
+ continue
+ if not re.search(
+ r"\b(?:create|write|save|export|render|generate|produce|output|deliver|store|convert|make)\b|"
+ r"\b(?:write_file|output_path)\b", clause, re.I,
+ ):
+ continue
+ expected.append(path)
+ assert declared_workspace_artifacts(text) == tuple(expected)
+
+
+def test_intent_and_detail_detectors_preserve_short_legacy_language():
+ cases = [
+ "plain answer", "\n\nfunction = manage_notes", "mcp__email__read_email",
+ "< function >", "web_search: cats", "prefix\n private_browser : url",
+ "when does it end", "when\nwill it end", "first save event", "sequence please",
+ "I can now carefully inspect the result.", "Answer. Now let me verify this.",
+ "接下来我会查看结果", "。 让我先检查",
+ ]
+ rng = random.Random(6503)
+ tokens = ["\n", " ", ".", "!", "function", "parameter", "=", "web_search", ":",
+ "when", "first", "end", "event", "let me", "inspect", "x"]
+ cases += ["".join(rng.choices(tokens, k=24)) for _ in range(3000)]
+ for text in cases:
+ assert _looks_like_ody_qwen_leaked_tool_text(text) == bool(_LEAKED_TOOL_TEXT.search(text)), text
+ assert contains_detailed_sequence_request(text, include_first=False) == bool(_VIDEO_DETAIL_BASE.search(text)), text
+ assert contains_detailed_sequence_request(text) == bool(_VIDEO_DETAIL_EXTENDED.search(text)), text
+
+ for text in (
+ "The result is partial. Now let me inspect the rest.",
+ "\n" * 20 + "let me check the source",
+ "。\n 接下来我会查看结果",
+ "A complete factual answer.",
+ ):
+ assert isinstance(_looks_like_agent_reasoning_preamble(text), bool)
+
+
+def test_suffix_and_split_helpers_preserve_legacy_results():
+ panel_cases = [
+ ("open calendar again!!!", ("ui_control", "open_panel calendar")),
+ ("open calendar" + " " * 50 + "again?", ("ui_control", "open_panel calendar")),
+ ("open calendar....", ("ui_control", "open_panel calendar")),
+ ("open calendar again x", ("ui_control", "open_panel calendar")),
+ ("open notes day view again.", ("ui_control", "open_panel notes")),
+ ]
+ for text, expected in panel_cases:
+ assert _parse_explicit_open_panel_request(text) == expected
+
+ values = ["one, two and three", " one ,two and three ", "candy and x", "a,and,b", ""]
+ for value in values:
+ expected_parts = [
+ re.sub(r"\s+", " ", part).strip(" .")
+ for part in re.split(r"\s*,\s*|\s+\band\b\s+", value)
+ ]
+ expected = [{"text": part, "done": False} for part in expected_parts if part]
+ assert _split_note_items(value) == expected
+
+ for value in ["Alice ", "Alice <", "Alice <>",
+ "Alice (a@example.com)", "Alice ((a@example.com)", "Alice > x "]:
+ normalized = re.sub(r"\s+", " ", value).strip()
+ expected_sender = re.sub(r"\s*\([^)]*@[^)]*\)\s*$", "", normalized).strip()
+ expected_sender = re.sub(r"\s*<[^>]*>\s*$", "", expected_sender).strip()
+ expected_sender = expected_sender or value.strip()
+ expected_account = re.sub(r"\s*<[^>]+>\s*$", "", normalized).strip() or normalized
+ assert _email_sender_name(value) == expected_sender
+ assert _email_account_label(value) == expected_account
+
+ for value in ["Reply body\nWant me to send it?", "Reply\n\n SHOULD I continue?", "Want me now", "x\nnot a prompt"]:
+ expected = re.split(r"\n\s*(?:Want me|Would you like|Should I)\b", value, flags=re.I, maxsplit=1)[0]
+ assert _split_before_assistant_prompt(value) == expected
+
+ for value in ["a \n b\t\n", " x", "a \r\n", "\t\n\n", ""]:
+ assert _strip_horizontal_space_before_lf(value) == re.sub(r"[ \t]+\n", "\n", value)
+
+ for value in ["pwd && ls", "cat x | grep y", "echo x; printf y", "pwd " + " " * 20 + "&& ls", "pwd || rm x"]:
+ legacy_parts = re.split(r"\s*(?:&&|\|\||;|\|)\s*", value.strip())
+ allowed = re.compile(
+ r"^(?:pwd|ls|find|rg|grep|git\s+(?:status|diff|log|show|branch)|sed(?!\s+-i\b)|"
+ r"head|tail|cat|stat|file|wc|sort|uniq|cut|ip|ipconfig|getent|nslookup|dig|arp|"
+ r"hostname|uname|whoami|echo|printf|test|true|false|:)\b", re.I,
+ )
+ expected = bool(legacy_parts) and all(part.strip() for part in legacy_parts) and all(
+ allowed.match(part.strip()) for part in legacy_parts
+ )
+ assert _read_only_shell_command(value) == bool(expected)
+
+
+@pytest.mark.parametrize("prefix,target_pattern", [
+ ("#session-", r"[^)]+"),
+ ("#note-", r"[^)]+"),
+ ("#event-", r"[0-9a-fA-F-]{8,64}"),
+ ("", r"[^)]+"),
+])
+def test_forward_markdown_and_angle_scanners_match_legacy(prefix, target_pattern):
+ reference = re.compile(r"\[([^\]]+)\]\(" + re.escape(prefix) + "(" + target_pattern + r")\)")
+ validator = None if target_pattern == r"[^)]+" else re.compile(target_pattern)
+ cases = ["[title](#session-id)", "[[title](#session-id)", "[](x)",
+ "[a](x)[b](y)", "[a](#event-12345678)", "[a](broken"]
+ rng = random.Random(6503)
+ tokens = ["[", "]", "(", ")", "#session-", "#note-", "#event-", "a", "1", "-", " "]
+ cases += ["".join(rng.choices(tokens, k=28)) for _ in range(3000)]
+ for text in cases:
+ expected = reference.findall(text)
+ actual = [(label, target) for _start, _end, label, target in iter_markdown_links(
+ text, target_prefix=prefix, target_re=validator
+ )]
+ assert actual == expected, text
+
+ generic = re.compile(r"\[([^\]]+)\]\([^)]+\)")
+ for text in cases:
+ assert replace_markdown_links_with_labels(text) == generic.sub(r"\1", text), text
+
+ angle = re.compile(r"<([^>]+)>")
+ for text in cases + ["", "<", "<>", "", "<<<"]:
+ assert [content for _start, _end, content in iter_angle_contents(text)] == angle.findall(text)
+
+
+def test_numbered_row_identifier_scan_matches_legacy_rows():
+ pattern = re.compile(r"^\s*\d+\.\s+.+?\s+\(([^)\n]+)\)\s+[—-]", re.M)
+ cases = [
+ "1. Task (abc) — due", " 2. Label ((nested) - tail", "1. no id",
+ "1. a (bad) x (good) — tail", "\n\n3. item (id-3) - tail",
+ ]
+ rng = random.Random(6503)
+ tokens = ["1. ", "x", " ", "(", ")", " -", " —", "\n"]
+ cases += ["".join(rng.choices(tokens, k=20)) for _ in range(2000)]
+ for text in cases:
+ assert _numbered_row_parenthesized_ids(text) == pattern.findall(text), text
+
+
+def test_model_and_request_extractors_match_legacy_grammars():
+ youtube = re.compile(
+ r"\b(?:youtube|youtu\.be|yt|video\s+comments?|comments?\s+on\s+(?:the\s+)?video|"
+ r"transcript\s+(?:of|for)|(?:latest|newest|recent)\s+(?:\d+\s+)?(?:videos?|uploads?)|"
+ r"official\s+.+\s+channel)\b", re.I,
+ )
+ destructive = re.compile(r"\b(delete|remove|archive|trash|send|reply|unsubscribe|mark\s+.*read)\b", re.I)
+ terse = re.compile(
+ r"\s*(?:(?:send|sned|share|give|show)?\s*(?:me\s+)?(?:the\s+)?(?:links?|urls?|sources?)"
+ r"(?:\s+(?:for|to|from)\s+(?:those|that|them|these|it|this|the\s+(?:sites?|websites?|resources?|sources?)))?"
+ r"|(?:for|to|from)\s+(?:those|that|them|these|it|this|the\s+(?:sites?|websites?|resources?|sources?)))"
+ r"\s*(?:please|pls)?[.!?]?\s*",
+ )
+ cases = ["official project channel", "official x channel", "mark all as read", "mark\nread",
+ " send me the links please ", "for those sites", "plain text"]
+ rng = random.Random(6503)
+ tokens = ["official", "channel", "mark", "read", "send", "links", "for", "those", "sites", " ", "\n", "x"]
+ cases += ["".join(rng.choices(tokens, k=24)) for _ in range(3000)]
+ for text in cases:
+ assert _looks_like_youtube_tool_turn(text) == bool(youtube.search(text)), text
+ assert _looks_like_destructive_request(text) == bool(destructive.search(text)), text
+ assert _is_terse_link_request(text) == bool(terse.fullmatch(text.lower())), text
+
+ summary_re = re.compile(
+ r"\bsummary\s*(?:\*\*)?\s*:?\s*(.+?)(?:\n\s*(?:-|\\*\\*|If you|Want me|This is|$))",
+ re.I | re.S,
+ )
+ summary_cases = ["Summary: useful\n- next", "summary **: x\n\nWant me to continue", "summary: no end"]
+ for text in summary_cases:
+ match = summary_re.search(text)
+ assert _contextual_summary_fragment(text) == (match.group(1) if match else "")
+
+ product_re = re.compile(
+ r"\b(?:find|look\s+for|shop\s+for|search\s+for)\s+(?:me\s+)?(?:the\s+)?(?:best\s+)?"
+ r"(?P.+?)\s*[?.!]*$", re.I,
+ )
+ for text in ["find best camera???", "look for me the best shoes", "shop for ???", "plain"]:
+ normalized = re.sub(r"\s+", " ", text).strip()
+ match = product_re.search(normalized)
+ expected = ""
+ if match:
+ expected = match.group("query").strip(" \t\r\n.,!?;:")
+ expected = re.sub(
+ r"\s+(?:on|at|from)\s+(?:the\s+)?[A-Za-z0-9&.' -]{1,60}$", "", expected, flags=re.I
+ ).strip()
+ expected = expected[:120] if 0 < len(expected.split()) <= 12 else ""
+ assert _private_browser_product_query(text) == expected
+
+ title_re = re.compile(r"]*)?>([\s\S]*?)", re.I)
+ for text in ["x", "a", "x"]:
+ match = title_re.search(text)
+ assert first_tag_content(text, "title", allow_attributes=True) == (match.group(1) if match else None)
+
+
+def test_summary_capture_preserves_exact_head_whitespace_and_newline_grammar():
+ legacy = re.compile(
+ r"\bsummary\s*(?:\*\*)?\s*:?\s*(.+?)(?:\n\s*(?:-|\\*\\*|If you|Want me|This is|$))",
+ re.I | re.S,
+ )
+ cases = [
+ "Summary: useful\nordinary next line\n- next",
+ "Summary: useful\nx\n", "summary\n\n", "summary \n",
+ "summary: \n", "summary **:\n\n", "summary:\nX\n",
+ "summary summary: X\n", "summary: no newline",
+ ]
+ rng = random.Random(6509)
+ tokens = ["summary", "Summary", ":", "**", "\\", " ", "\t", "\n", "x", "-", "If you"]
+ cases += ["".join(rng.choices(tokens, k=24)) for _ in range(5000)]
+ for text in cases:
+ match = legacy.search(text)
+ assert _contextual_summary_fragment(text) == (match.group(1) if match else ""), text
+
+
+def test_repeated_listing_words_preserve_legacy_request_grammar():
+ legacy = re.compile(
+ r"\s*(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)\s+"
+ r"(?:[\w .:/-]+\s+)?(?:stories|articles|posts|headlines|pages)"
+ r"(?:\s+on\s+[\w .:/-]+)?[.!?]?\s*", re.I,
+ )
+ for text in ("top Straße stories", "top İ stories", "lİst stories", "top storİes",
+ "top café stories", "top stories on Straße", "top stories on İ"):
+ assert _page_listing_request(text) == bool(legacy.fullmatch(text)), text
+ for word in ("stories", "articles", "posts", "headlines", "pages"):
+ for count in (1, 2, 8, 32):
+ for suffix in ("X", "@", "\tX", "on x", "on \t", "on ", "\n"):
+ for separator in (" ", " on ", "\t", " \t"):
+ text = "top " + (word + separator) * count + suffix
+ assert _page_listing_request(text) == bool(legacy.fullmatch(text)), text
+
+
+def test_repeated_navigation_schemes_preserve_legacy_url_search():
+ for count in (1, 2, 8, 32):
+ for suffix in ("X", "google.com/search", "bing.com/html", "duckduckgo.com/?q=x"):
+ text = "http://" * count + suffix
+ assert contains_search_engine_navigation(text) == bool(_SEARCH_ENGINE_URL.search(text)), text
+
+
+def test_tool_listing_rows_match_legacy_grammars():
+ research = re.compile(r"^-\s+\[(.*?)\]\(#research-([^)]+)\)(.*)$")
+ skills = re.compile(r"^-\s+\*\*(.*?)\*\*(?:\s+\((.*?)\)|\s+\[(draft)\])?(?::\s*(.*))?$")
+ calendar = re.compile(r"^\s*-\s+(.+?):\s+\[(.*?)\]\(#event-([^)]+)\)(.*)$")
+ draft = re.compile(r"\bTo:\s*.+\bSubject:\s*.+\n---", re.I | re.S)
+ cases = [
+ "- [Title](#research-id) tail",
+ "- **name** (published): description",
+ "- **name** [draft]",
+ " - when: [title](#event-id) tail",
+ "To: a\nSubject: b\n---",
+ "To:Subject:x\n---",
+ "plain",
+ ]
+ rng = random.Random(6504)
+ tokens = ["-", " ", "\t", "[", "]", "(", ")", "*", ":", "#research-", "#event-", "draft", "x"]
+ cases += ["".join(rng.choices(tokens, k=28)) for _ in range(5000)]
+ for text in cases:
+ match = research.match(text)
+ assert _research_listing_row(text) == (match.groups() if match else None), text
+ match = skills.match(text)
+ assert _skill_listing_row(text) == (match.groups() if match else None), text
+ match = calendar.match(text)
+ assert _calendar_listing_row(text) == (match.groups() if match else None), text
+ assert _contains_email_draft_headers(text) == bool(draft.search(text)), text
+
+
+def test_terse_email_followup_matches_legacy_grammar():
+ legacy = re.compile(
+ r"^\s*(?:and|so|well|still|then|okay|ok|did you find it(?: yet)?|what did you find)\s*[?.!]*\s*$",
+ re.I,
+ )
+ rng = random.Random(6508)
+ tokens = ["and", "so", "well", "did you find it", " yet", "what did you find", " ", "\t", "?", ".", "!", "x"]
+ cases = ["and?", " did you find it yet ! ", "what did you find", "and X"]
+ cases += ["".join(rng.choices(tokens, k=20)) for _ in range(3000)]
+ for text in cases:
+ assert _is_terse_email_lookup_followup(text) == bool(legacy.match(text)), text
+
+
+def test_preview_request_and_source_scans_match_legacy_grammars():
+ page = re.compile(
+ r"\s*(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)\s+"
+ r"(?:[\w .:/-]+\s+)?(?:stories|articles|posts|headlines|pages)"
+ r"(?:\s+on\s+[\w .:/-]+)?[.!?]?\s*", re.I,
+ )
+ prior = re.compile(
+ r"\s*(?:(?:where|what)\s+did\s+you\s+(?:get|find)\s+(?:that|this)\s+from[?., ]*"
+ r"(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link[.!? ]*"
+ r"|(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link(?:\s+for\s+that)?[.!? ]*"
+ r"|what(?:['’]?s|\s+is)\s+(?:the\s+)?source(?:\s+link)?[.!? ]*)\s*", re.I,
+ )
+ terminal = re.compile(
+ r"(?:^|[.!?;,\n])\s*(?:(?:pls|please)\s+)?(?:sources?|citations?|links?)"
+ r"\s*(?:pls|please)?\s*[.!?]*$", re.I,
+ )
+ rows = re.compile(r"^\[\d+\]\s+(.+?)\s*\n\s*(https?://\S+)", re.M)
+ cases = [
+ "top stories", "show me the latest stories on example.com", "give me the source link",
+ "what's the source?", "x. please links pls!!", "[1] Title\nhttps://example.test/x", "plain",
+ ]
+ rng = random.Random(6505)
+ tokens = ["top", "show", " me", " the", " stories", " on", "link", "source", "please", " ", "\t", ".", "!", "\n", "[1]", "http://x"]
+ cases += ["".join(rng.choices(tokens, k=22)) for _ in range(5000)]
+ for text in cases:
+ assert _page_listing_request(text) == bool(page.fullmatch(text)), text
+ assert _prior_web_source_request(text) == bool(prior.fullmatch(text)), text
+ assert _terminal_source_link_clause(text) == bool(terminal.search(text)), text
+ assert _web_source_rows(text) == rows.findall(text), text
+
+
+def test_staged_command_payload_parsers_match_legacy_grammars():
+ specs = [
+ (
+ re.compile(r"\bnote\s+titled\s+(.+?)\s+so\s+its\s+content\s+is\s+['\"]([^'\"]+)['\"]", re.I),
+ r"\bnote\s+titled(?=\s)",
+ r"\s+(.+?)\s+so\s+its\s+content\s+is\s+['\"]([^'\"]+)['\"]",
+ ),
+ (
+ re.compile(r"\b(?:delete|trash|remove|archive|mark(?:\s+as)?\s+(?:read|unread)|mark\s+(?:read|unread))\b\s+(?:all|every|the)?\s*(?:my\s+)?(.+?)\s+(?:emails?|mail|messages?)\b", re.I),
+ r"\b(?:delete|trash|remove|archive|mark(?:\s+as)?\s+(?:read|unread)|mark\s+(?:read|unread))\b(?=\s)",
+ r"\s+(?:all|every|the)?\s*(?:my\s+)?(.+?)\s+(?:emails?|mail|messages?)\b",
+ ),
+ (
+ re.compile(r"\b(?:make|create|add)\s+(?:a\s+)?checklist\s+(?:called|titled|named)\s+(.+?)\s+with\s+(.+?)\s*$", re.I),
+ r"\b(?:make|create|add)\s+(?:a\s+)?checklist\s+(?:called|titled|named)(?=\s)",
+ r"\s+(.+?)\s+with\s+(.+?)$",
+ ),
+ (
+ re.compile(r"\b(?:change|update|set|retag)\b\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b", re.I),
+ r"\b(?:change|update|set|retag)\b(?=\s)",
+ r"\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b",
+ ),
+ (
+ re.compile(r"\b(?:chang(?:e|es|ed|ing)|updat(?:e|es|ed|ing))\s+(.+?)\s+to\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)", re.I),
+ r"\b(?:chang(?:e|es|ed|ing)|updat(?:e|es|ed|ing))(?=\s)",
+ r"\s+(.+?)\s+to\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)",
+ ),
+ (
+ re.compile(r"\breplace\s+(.+?)\s+with\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)", re.I),
+ r"\breplace(?=\s)",
+ r"\s+(.+?)\s+with\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)",
+ ),
+ ]
+ pipeline = re.compile(
+ r"\bpipeline\s+using\s+([^\s,]+)\s+to\s+(.+?),\s*then\s+([^\s,]+)\s+to\s+(.+?)(?:[.!?]\s*)?$",
+ re.I,
+ )
+ session = re.compile(r"\b(?:find|search(?:\s+for)?|show)\s+(?:the\s+)?(.+?)\s+(?:chat|session|conversation)\b", re.I)
+ checklist = re.compile(r"\b(?:what(?:'s| is)?|show|tell\s+me)\b.*?\b(?:left|remaining)\b.*?\b(?:on|in)\s+(?:the\s+)?(.+?)\s+checklist\b", re.I)
+ rng = random.Random(6506)
+ tokens = ["note", " titled", " so its content is ", "'x'", "delete", " all", " emails", "create checklist called", " with", "change", " tag to ", "replace", " to", " pipeline using ", " then", "find", " chat", "left", " on", " checklist", " ", "\t", ".", "x"]
+ cases = ["update note titled a so its content is 'b'", "delete all all emails", "create checklist called a with b", "change the trip tag to work", "replace old with new", "pipeline using a to x, then b to y", "find the chat", "what is left on the trip checklist"]
+ cases += ["".join(rng.choices(tokens, k=25)).strip() for _ in range(5000)]
+ for text in cases:
+ for legacy, prefix, remainder in specs:
+ match = legacy.search(text)
+ assert _captures_after_first_prefix(text, prefix, remainder) == (match.groups() if match else None), (legacy.pattern, text)
+ match = pipeline.search(text)
+ assert _pipeline_request_parts(text) == (match.groups() if match else None), text
+ match = session.search(text)
+ assert _session_find_query_capture(text) == (match.group(1) if match else None), text
+ match = checklist.search(text)
+ assert _remaining_checklist_name(text) == (match.group(1) if match else None), text
+
+
+def test_turn_contract_scans_match_legacy_grammars():
+ cancelled = re.compile(
+ r"^(?:never\s*mind|scratch\s+that)\s*[,;:—–-]?\s*(?=(?:open|show|list|read|search|find|check|switch|go)\b)",
+ re.I,
+ )
+ prohibition = re.compile(
+ r"^\s*(?:read[- ]only(?:\s+and)?\s+)?(?:do\s+not|don['’]?t|never)\s+"
+ r"(?:add|create|make|write|draft|edit|change|update|delete|remove|send|reply|run|execute|download|serve|open|save|schedule|transcribe|inspect)\b"
+ r"[^.;\n]*[.!?]*\s*$", re.I,
+ )
+ budget = re.compile(r"\bunder\s+[¥$€£]?\s*\d+(?:[.,]\d+)?(?:\s*yen)?\b", re.I)
+ terminal_specs = [
+ r"[,.;?]\s*(?:(?:only|just)\s+)?(?:(?:list|show)\s+(?:me\s+)?)?a\s+few(?:\s+(?:task\s+)?(?:names?|items?|results?|entries?))?(?:\s+and\s+(?:whether|if)\s+[^.;\n]+)?",
+ r"[.;]\s*read[- ]only(?:\s+(?:please|pls|plz))?\s*,?\s*(?:(?:and\s+)?(?:do\s+not|don['’]?t|dont)\s+(?:change|edit|modify)(?:\s+or\s+send)?\s+(?:anything|data))?",
+ r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?(?:do\s+not|don['’]?t|dont)\s+(?:touch|change|edit|modify)(?:\s+(?:anything|data|them))?(?:\s+yet)?",
+ r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+changes?",
+ r"[,;]\s*(?:keep\s+(?:them|it)\s+)?short\s+lines?\s*,?",
+ ]
+ rng = random.Random(6507)
+ tokens = ["never", " mind", "scratch", " that", "open", "do not", " touch", " anything", "no changes", "read-only", "a few", " and whether", "short lines", "under", "$", "123", "yen", "open my calendar", "and", " what is that", " ", "\t", ".", "!", ",", ";", "x"]
+ cases = ["never mind, open calendar", "do not edit anything.", "under $ 20 yen", "open my calendar", "open calendar and what is next", "x but "]
+ cases += ["".join(rng.choices(tokens, k=22)) for _ in range(5000)]
+ for text in cases:
+ assert _strip_cancelled_request_lead(text) == cancelled.sub("", text), text
+ assert _is_pure_action_prohibition(text) == bool(prohibition.fullmatch(text)), text
+ assert _mentions_under_budget(text) == bool(budget.search(text)), text
+ assert _strip_terminal_but(text) == re.sub(r"\s+but\s*$", "", text, flags=re.I), text
+ for core in terminal_specs:
+ legacy = re.search(core + r"[.!?]*\s*$", text, re.I)
+ current = _terminal_clause_match(text, core)
+ assert (current.start() if current else None) == (legacy.start() if legacy else None), (core, text)
+ match = _WARM_RECALL.fullmatch(text)
+ assert _warm_recall_parts(text) == ((match.group("target"), "") if match else None), text
+ match = _WARM_RECALL_WITH_FOLLOWUP.fullmatch(text)
+ assert _warm_recall_parts(text, with_followup=True) == (
+ (match.group("target"), match.group("followup")) if match else None
+ ), text
+
+
+@pytest.mark.parametrize("program", [
+ r'''
+from src.agent_loop import _contextual_summary_fragment
+assert _contextual_summary_fragment("Summary: x\n" + "x\n" * 100_000) == "x"
+assert _contextual_summary_fragment("Summary:" + "\n" * 100_000) == "\n"
+''',
+ r'''
+from src.clean_agent_preview import _page_listing_request
+for text in ("top " + "stories " * 30_000 + "X",
+ "top " + "stories on " * 30_000 + "@"):
+ assert not _page_listing_request(text)
+''',
+ r'''
+from src.text_scanning import contains_search_engine_navigation
+assert not contains_search_engine_navigation("http://" * 100_000 + "X")
+assert contains_search_engine_navigation("http://" * 100_000 + "bing.com/search")
+''',
+ r'''
+from src.agent_loop import _is_terse_email_lookup_followup
+_is_terse_email_lookup_followup("and" + " " * 100_000 + "X")
+''',
+ r'''
+from src.turn_contract import (_is_pure_action_prohibition, _mentions_under_budget,
+ _strip_cancelled_request_lead, _terminal_clause_match, _warm_recall_parts)
+space = " " * 100_000
+_strip_cancelled_request_lead("never" + space + "mind" + space + "X")
+_terminal_clause_match(", a few and whether " + space + "X;", r"[,.;?]\s*a\s+few(?:\s+and\s+whether\s+[^.;\n]+)?")
+_is_pure_action_prohibition("do not add " + space + "X;")
+_mentions_under_budget("under" + space + "$" + space + "X")
+_warm_recall_parts("open" + space + "my" + space + "calendar" + space + "X")
+_warm_recall_parts("open" + space + "my" + space + "calendar" + space + "and" + space + "what " + "x" * 181, with_followup=True)
+''',
+ r'''
+from src.agent_loop import (_captures_after_first_prefix, _pipeline_request_parts,
+ _remaining_checklist_name, _session_find_query_capture)
+evil = ("delete all x " * 30_000) + "z"
+_captures_after_first_prefix(evil, r"\bdelete\b(?=\s)", r"\s+(?:all)?\s*(.+?)\s+emails?\b")
+_pipeline_request_parts(("pipeline using m to x " * 30_000) + "z")
+_session_find_query_capture(("find x " * 30_000) + "z")
+_remaining_checklist_name("what " + ("left on " * 30_000) + "z")
+''',
+ r'''
+from src.clean_agent_preview import (_page_listing_request, _prior_web_source_request,
+ _terminal_source_link_clause, _web_source_rows)
+_page_listing_request("top stories on " + " " * 100_000 + "\nX")
+_prior_web_source_request("give me link" + " " * 100_000 + "X")
+_terminal_source_link_clause("links" + " " * 100_000 + "X")
+_web_source_rows("[1] " + " " * 100_000)
+''',
+ r'''
+from src.agent_loop import (_calendar_listing_row, _contains_email_draft_headers,
+ _research_listing_row, _skill_listing_row)
+for text in (
+ "- [" + "](#research-" * 30_000,
+ "- **" + " **" * 30_000,
+ "- when: [" + "](#event-" * 30_000,
+ "To: x " + "Subject: x " * 30_000,
+):
+ _research_listing_row(text)
+ _skill_listing_row(text)
+ _calendar_listing_row(text)
+ _contains_email_draft_headers(text)
+''',
+ r'''
+from src.agent_loop import _session_link_from_row, _session_list_summary_from_tool_output, _strip_trailing_done
+evil = "Done. " + "\t" * 100_000 + "x"
+assert _strip_trailing_done(evil) == evil
+for row in ("[" + "\\" * 100_000, "[" * 100_000,
+ "[a" + "\\](#session-" * 20_000 + ")"):
+ _session_link_from_row(row)
+for row in ("[x](#session-id) (" + "msgs" * 100_000,
+ "[x](#session-id) (" + "(last active today)" * 20_000):
+ _session_list_summary_from_tool_output("Chats:\n- " + row)
+''',
+ r'''
+from src.tool_parsing import *
+from src.tool_parsing import (_iter_named_blocks, _strip_delimited,
+ _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE)
+text = "}<|tool_call|>" + "<|tool_call|>call:web_search{" * 20_000
+assert list(_iter_named_blocks(text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE)) == []
+assert _strip_delimited(text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE) == text
+# Public entry points also stay responsive, including fallback parsers.
+text = "}<|tool_call|>" + "<|tool_call|>call:web_search{" * 3000
+assert parse_tool_blocks(text) == []
+strip_tool_blocks(text)
+''',
+ r'''
+from src.tool_parsing import (_iter_named_blocks, _iter_qwen_python_args,
+ _QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE,
+ _QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE, parse_tool_blocks)
+for opener, closer, text in (
+ (_QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE, "" * 20_000),
+ (_QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE, "" * 20_000),
+ (_QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE, "a" + "\t" * 100_000 + "x"),
+):
+ assert list(_iter_named_blocks(text, opener, closer)) == []
+assert list(_iter_qwen_python_args("a" * 100_000)) == []
+assert parse_tool_blocks("" * 3000) == []
+assert parse_tool_blocks("manage_notes(" + "a" * 100_000 + ")")
+''',
+ r'''
+from src.tool_parsing import strip_angle_tags
+for allow_empty in (False, True):
+ for replacement in ("", " "):
+ for text in ("<" * 200_000, ">" + "<" * 200_000):
+ assert strip_angle_tags(text, replacement, allow_empty=allow_empty) == text
+ assert strip_angle_tags("<" * 200_000 + ">tail", replacement,
+ allow_empty=allow_empty) == replacement + "tail"
+''',
+ r'''
+from src.tool_parsing import iter_email_addresses
+for ascii_only in (False, True):
+ for text in ("+" * 100_000, "a@" + "a" * 100_000,
+ "a@" + "." * 100_000, "a@" * 20_000):
+ assert list(iter_email_addresses(text, ascii_only=ascii_only)) == []
+''',
+ r'''
+from src.agent_loop import _mentions_pdf_url, _mentions_workspace_script
+from src.clean_agent_preview import declared_workspace_artifacts
+from src.text_scanning import contains_search_engine_navigation
+from src.turn_contract import _mentions_workspace_artifact, _mentions_workspace_output
+workspace = "/workspace/" * 30_000 + "x"
+assert not _mentions_workspace_script(workspace)
+assert not _mentions_workspace_artifact(workspace)
+assert not _mentions_workspace_output(workspace)
+assert declared_workspace_artifacts("create " + workspace) == ()
+assert not _mentions_pdf_url("http://" * 30_000 + "x")
+assert not contains_search_engine_navigation("http://google." + "..google." * 30_000 + "x")
+''',
+ r'''
+from src.agent_loop import _looks_like_agent_reasoning_preamble, _looks_like_ody_qwen_leaked_tool_text
+from src.text_scanning import contains_detailed_sequence_request
+for text in ("\n" * 100_000 + "x", ("when " * 20_000) + "x"):
+ _looks_like_agent_reasoning_preamble(text)
+ _looks_like_ody_qwen_leaked_tool_text(text)
+ contains_detailed_sequence_request(text)
+''',
+ r'''
+from src.agent_loop import (_email_account_label, _email_sender_name,
+ _parse_explicit_open_panel_request, _read_only_shell_command,
+ _split_before_assistant_prompt, _split_note_items,
+ _strip_horizontal_space_before_lf)
+space = " " * 100_000
+_parse_explicit_open_panel_request("open calendar" + space + "x")
+_split_note_items(space + "x")
+_email_sender_name("<" * 100_000 + "x")
+_email_account_label("<" * 100_000 + "x")
+_split_before_assistant_prompt("\n" * 100_000 + "x")
+_strip_horizontal_space_before_lf(" " * 100_000 + "x")
+_read_only_shell_command(space + "x")
+''',
+ r'''
+from src.agent_loop import _numbered_row_parenthesized_ids
+from src.text_scanning import iter_angle_contents, iter_markdown_links, replace_markdown_links_with_labels
+for text in ("<" * 100_000, "[" * 100_000,
+ ("[x](#session-" * 20_000) + "missing"):
+ list(iter_angle_contents(text))
+ list(iter_markdown_links(text, target_prefix="#session-"))
+ replace_markdown_links_with_labels(text)
+_numbered_row_parenthesized_ids("1. x " + "(" * 100_000 + "x")
+''',
+ r'''
+from src.agent_loop import (_contextual_summary_fragment, _is_terse_link_request,
+ _looks_like_destructive_request, _looks_like_youtube_tool_turn,
+ _private_browser_product_query)
+from src.text_scanning import first_tag_content
+for text in (("official " * 30_000) + "x", ("mark " * 30_000) + "x",
+ " " * 100_000 + "x", ("summary: x " * 20_000) + "x",
+ "find best " + "?" * 100_000 + "x", "