fix(security): eliminate parser denial-of-service paths

This commit is contained in:
Alexandre Teixeira
2026-10-06 03:16:53 +01:00
parent e0a45aeae7
commit eeff41a9ef
8 changed files with 2676 additions and 335 deletions
+710 -212
View File
File diff suppressed because it is too large Load Diff
+296 -47
View File
@@ -32,7 +32,12 @@ from src.tool_schemas import (
normalized_native_function_argument_error,
)
from src.tool_types import ToolBlock
from src.tool_parsing import parse_tool_blocks, strip_tool_blocks
from src.tool_parsing import iter_email_addresses, parse_tool_blocks, strip_tool_blocks, strip_angle_tags
from src.text_scanning import (
contains_detailed_sequence_request,
contains_search_engine_navigation,
iter_prefixed_token_matches,
)
from src.turn_contract import (
_REQUEST_PREFIX,
calendar_retiming_request,
@@ -54,6 +59,59 @@ MODE = COMPACT_PREVIEW_MODE
class ProviderStreamError(Exception):
"""A provider reported failure inside an otherwise successful SSE response."""
# Only these audited, fixed domain messages may cross the exception boundary.
# Return the canonical constants rather than arbitrary exception text.
_PUBLIC_PREVIEW_TOOL_ERRORS = {message: message for message in (
'An equivalent search already returned evidence. Change the angle, missing subtopic, source type, or corroboration target instead of only changing freshness wording.',
'Browser navigation already failed for this exact URL; use another source page.',
'Do not infer image content repeatedly from filenames; use inspect_media on representative files, then continue from visual evidence.',
'Equivalent search intent was repeated after a correction reminder; web_search is disabled for this turn.',
'Look up the target with list_sessions before changing a chat. Use its exact returned ID; never invent last-chat/latest aliases. If the target is ambiguous, ask using the candidate chat titles.',
'Resolve the named recipient with resolve_contact before drafting. Never invent an email address.',
'Selection-only edits cannot use replace_all. Use a unique contextual FIND inside the selected passage.',
'The bounded search-attempt budget is exhausted. Do not search again; answer from usable evidence already gathered, or clearly report what could not be verified and suggest a concrete next step.',
'The calendar read has not succeeded yet. Obtain the requested calendar evidence before creating the dependent email draft.',
'The latest note search returned no candidates, so this reference has no note to open. Do not reuse an older list item; report the empty result or ask which note was intended.',
'The latest research search returned no candidates, so this reference has no report to open. Do not reuse an unrelated older report.',
'The proposed expansion only adds placeholder/meta text. Write substantive content that continues the existing document’s subject, voice, and format; do not announce that a paragraph was added.',
'The recipient address is not supported by the contact lookup. Use an exact returned address for the requested person; if no match exists, explain the missing recipient instead of guessing.',
'The same artifact target was already rewritten three times; finish from the latest successful version instead of rewriting it again.',
'These suggestions collapse multiple different passages into the same much shorter replacement, violating the request to preserve meaning. Produce passage-specific revisions that retain each source passage’s claims and intent.',
'This exact call already failed twice and will not be executed again; change strategy or finish from existing evidence.',
'This exact failed call was repeated after a correction reminder; the tool is disabled for this turn. Finish from existing evidence.',
'This exact invalid call was repeated after two validation failures; the tool is disabled for this turn.',
'This exact successful call already returned evidence. Do not repeat it; change the arguments or tool to gather different evidence, or finish from the evidence already available.',
'This still image was already inspected and the same visual evidence is already in context. inspect_media is withheld for the next correction round; use that evidence, inspect a different file, or finish.',
'This operation is outside the preview safety policy. No change was made.',
'This replacement does not make its passage more concise. Shorten the wording while retaining its facts and meaning; a spelling-only change does not satisfy the requested action. Retry with shorter replacements.',
'Tool arguments could not be converted for execution.',
'Tool arguments must be a JSON object.',
'Tool execution budget exhausted; finish from existing evidence.',
'Tool is not offered or permitted.',
'Two equivalent searches already returned no evidence. Do not repeat this search wording; use a different offered tool or a materially different query.',
'Unresolved video target: this ID was not supplied by the user or observed in a successful tool result. Do not guess it from the title. Open/read the referenced browser link or call youtube_tool latest_channel_video for the observed channel, then use its returned ID.',
'inspect_media exports only extract video stills and require an explicit timestamp plus output_path for every item. First inspect the video to find the timestamp; use write_file or python to author a diagram or other new artifact.',
'web_fetch requires url or urls. query only filters a supplied page; it is not a search or writing request.',
)}
def _public_preview_tool_error(exc, *, execution_attempted=False):
"""Keep useful domain guidance while withholding arbitrary diagnostics."""
if execution_attempted:
return 'The tool failed unexpectedly. Check the server log and retry.'
if isinstance(exc, json.JSONDecodeError):
return 'Tool arguments are not valid JSON. Correct the JSON object and retry.'
if isinstance(exc, jsonschema.ValidationError):
return 'Tool arguments do not match the required schema. Correct the call using the offered tool schema.'
detail = str(exc)
if detail.startswith('Artifact completion Python must reference the required '):
return 'Artifact completion Python must create non-empty output at the required artifact target.'
if detail.startswith('Shell access to credential variable '):
return 'Shell access to credentials is blocked. Use brokered native tools.'
return _PUBLIC_PREVIEW_TOOL_ERRORS.get(detail, 'The tool call could not be validated. Check its arguments and retry.')
# Native unattended workspaces routinely require several inspections followed
# by several artifact writes. The interactive preview keeps its six-call
# limit below; this larger budget applies only after server-side validation of
@@ -79,13 +137,6 @@ ARTIFACT_RESEARCH_TOOLS = frozenset({
'web_search', 'web_fetch', 'private_browser', 'pdf_extract', 'youtube_tool',
'inspect_media', 'extract_text', 'transcribe_media',
})
DETAILED_VIDEO_REQUEST = re.compile(
r"\b(?:how\s+many|count|sequence|in\s+order|chronological|"
r"timestamps?|what\s+time|at\s+what\s+time|when\s+.*(?:end|happen)|"
r"first\s+.*(?:save|attempt|event)|score(?:board)?s?)\b|"
r"(?:多少|几次|何时|什么时候|时间|顺序)",
re.IGNORECASE,
)
READ_TOOLS = frozenset({
'manage_notes', 'manage_calendar', 'manage_memory', 'manage_skills', 'manage_tasks',
'manage_documents', 'manage_research', 'manage_contact', 'list_sessions',
@@ -809,13 +860,95 @@ def malformed_write_handoff_target(arguments, required_artifacts=(), user_text='
return target
_PAGE_LISTING_WORDS = ("stories", "articles", "posts", "headlines", "pages")
def _listing_allowed(char):
return char == " " or char in ".:/-" or char == "_" or char.isalnum()
def _allowed_then_space(value):
"""Whether value can be class+ followed by whitespace+, without retries."""
if len(value) < 2:
return False
first_disallowed = next((i for i, char in enumerate(value) if not _listing_allowed(char)), len(value))
if first_disallowed < len(value):
return first_disallowed > 0 and all(char.isspace() for char in value[first_disallowed:])
return value[-1].isspace()
def _space_then_allowed(value):
"""Whether value can be whitespace+ followed by the listing class+."""
if len(value) < 2:
return False
first_nonspace = next((i for i, char in enumerate(value) if not char.isspace()), len(value))
if first_nonspace < len(value):
return first_nonspace > 0 and all(_listing_allowed(char) for char in value[first_nonspace:])
trailing_spaces = len(value) - len(value.rstrip(" "))
return trailing_spaces > 0 and (len(value) > trailing_spaces or trailing_spaces >= 2)
def _page_listing_request(user_text):
value = str(user_text or '').lstrip()
nonspace_end = len(value.rstrip())
if value[nonspace_end - 1:nonspace_end] in ("!", "?"):
value = value[:nonspace_end - 1]
lead = re.match(
r"(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)(?=\s)",
value, re.I,
)
if not lead:
return False
cursor = lead.end()
while cursor < len(value) and value[cursor].isspace():
cursor += 1
body = value[cursor:]
nonspace_end = len(body.rstrip())
terminal_end = nonspace_end
if body[terminal_end - 1:terminal_end] == ".":
terminal_end = len(body[:terminal_end - 1].rstrip())
first_disallowed = len(body)
last_disallowed = -1
first_nonspace_after_disallowed = len(body)
for index, char in enumerate(body):
if not _listing_allowed(char):
first_disallowed = min(first_disallowed, index)
if index < nonspace_end:
last_disallowed = index
if index >= first_disallowed and not char.isspace():
first_nonspace_after_disallowed = min(first_nonspace_after_disallowed, index)
last_literal_space = body.rfind(" ")
for word in _PAGE_LISTING_WORDS:
for candidate in re.finditer(word, body, re.I):
start = candidate.start()
prefix_allowed = start >= 2 and (
body[start - 1].isspace() if start <= first_disallowed
else first_disallowed > 0 and start <= first_nonspace_after_disallowed
)
if start == 0 or prefix_allowed:
after_start = candidate.end()
if after_start >= terminal_end:
return True
on = after_start
while on < len(body) and body[on].isspace():
on += 1
if on > after_start and body[on:on + 2].casefold() == "on":
tail_start = on + 2
tail_end = tail_start
while tail_end < len(body) and body[tail_end].isspace():
tail_end += 1
if len(body) - tail_start >= 2 and tail_end > tail_start:
if tail_end < len(body):
if last_disallowed < tail_end:
return True
elif last_literal_space > tail_start:
return True
return False
def page_listing_response(entries, user_text, max_items=10):
"""Render simple page listings from observed titles/URLs, never synthesized rankings."""
if not re.fullmatch(
r'\s*(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)\s+'
r'(?:[\w .:/-]+\s+)?(?:stories|articles|posts|headlines|pages)'
r'(?:\s+on\s+[\w .:/-]+)?[.!?]?\s*', user_text, re.I,
) or re.search(r'\b(?:and|compare|summarize|analyse|analyze|about|by|since|yesterday)\b', user_text, re.I):
if not _page_listing_request(user_text) or re.search(r'\b(?:and|compare|summarize|analyse|analyze|about|by|since|yesterday)\b', user_text, re.I):
return ''
from urllib.parse import quote, urlsplit
from html import escape
@@ -1531,17 +1664,24 @@ def prior_workspace_path_answer(user_text, history):
return ''
def _prior_web_source_request(text):
text = str(text or '')
candidate = text.lstrip().rstrip()
candidate = candidate.rstrip(".!? ")
return bool(re.fullmatch(
r"(?:(?:where|what)\s+did\s+you\s+(?:get|find)\s+(?:that|this)\s+from[?., ]*"
r"(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link"
r"|(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link(?:\s+for\s+that)?"
r"|what(?:['’]?s|\s+is)\s+(?:the\s+)?source(?:\s+link)?)",
candidate,
re.I,
))
def prior_web_source_answer(user_text, history):
"""Return the latest source URL for an explicit source-only follow-up."""
text = str(user_text or '')
if not re.fullmatch(
r"\s*(?:(?:where|what)\s+did\s+you\s+(?:get|find)\s+(?:that|this)\s+from[?., ]*"
r"(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link[.!? ]*"
r"|(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link(?:\s+for\s+that)?[.!? ]*"
r"|what(?:['’]?s|\s+is)\s+(?:the\s+)?source(?:\s+link)?[.!? ]*)\s*",
text,
re.I,
):
if not _prior_web_source_request(text):
return ''
call_names = {}
candidates = []
@@ -2822,12 +2962,11 @@ def draft_contact_evidence_error(name, args, *, dependencies=(), executions=(),
and not e.get('blocked')]
if not observations:
return 'Resolve the named recipient with resolve_contact before drafting. Never invent an email address.'
address_pattern = r'[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}'
known = {address.casefold() for e in observations
for address in re.findall(address_pattern, str(e.get('output') or ''))}
known.update(address.casefold() for address in re.findall(address_pattern, user_text))
for address in iter_email_addresses(str(e.get('output') or ''), ascii_only=True)}
known.update(address.casefold() for address in iter_email_addresses(user_text, ascii_only=True))
proposed = {address.casefold() for field in ('to', 'cc', 'bcc')
for address in re.findall(address_pattern, str(args.get(field) or ''))}
for address in iter_email_addresses(str(args.get(field) or ''), ascii_only=True)}
if not proposed or not proposed <= known:
return ('The recipient address is not supported by the contact lookup. Use an exact '
'returned address for the requested person; if no match exists, explain '
@@ -3945,7 +4084,7 @@ def active_document_revision_quality_error(name, args, *, active_document, user_
if not incoming:
return None
addition = incoming[len(existing):].strip() if existing and incoming.startswith(existing) else ''
candidate = re.sub(r'<[^>]+>', ' ', addition).strip()
candidate = strip_angle_tags(addition, ' ').strip()
candidate = re.sub(r'\s+', ' ', candidate)
if not candidate or candidate.casefold() in str(user_text or '').casefold():
return None
@@ -3970,8 +4109,8 @@ def document_suggestion_quality_error(name, args, *, user_text):
for suggestion in suggestions:
if not isinstance(suggestion, dict):
continue
source = re.sub(r'\s+', ' ', re.sub(r'<[^>]*>', ' ', str(suggestion.get('find') or ''))).strip()
result = re.sub(r'\s+', ' ', re.sub(r'<[^>]*>', ' ', str(suggestion.get('replace') or ''))).strip()
source = re.sub(r'\s+', ' ', strip_angle_tags(str(suggestion.get('find') or ''), ' ', allow_empty=True)).strip()
result = re.sub(r'\s+', ' ', strip_angle_tags(str(suggestion.get('replace') or ''), ' ', allow_empty=True)).strip()
if not source or not result:
continue
source_words = len(re.findall(r"\b[\w’'-]+\b", source))
@@ -4127,13 +4266,17 @@ _WORKSPACE_FILE_RE = re.compile(
r"/workspace/[^\s,,、;;`\"'<>]+\.[A-Za-z0-9]{1,12}",
re.I,
)
_WORKSPACE_PREFIX_RE = re.compile(r"/workspace/", re.I)
_WORKSPACE_FILE_TOKEN_TAIL_RE = re.compile(r"[^\s,,、;;`\"'<>]*")
def declared_workspace_artifacts(user_text):
"""Return explicit output paths, excluding paths used only as inputs."""
text = str(user_text or '')
paths = []
for match in _WORKSPACE_FILE_RE.finditer(text):
for match in iter_prefixed_token_matches(
text, _WORKSPACE_PREFIX_RE, _WORKSPACE_FILE_RE, _WORKSPACE_FILE_TOKEN_TAIL_RE
):
path = match.group(0).rstrip('.!?))]}')
if path.startswith('/workspace/fixtures/') or path in paths:
continue
@@ -4253,17 +4396,61 @@ def evidence_tool_keeps_distinct_requests_available(name):
}
def _terminal_source_link_clause(text):
"""Recognize the terminal short source/link clause from right to left."""
value = str(text or '')
def matches_at(end):
while end and value[end - 1] in ".!?":
end -= 1
while end and value[end - 1].isspace():
end -= 1
lowered = value[:end].casefold()
for courtesy in ("please", "pls"):
if lowered.endswith(courtesy):
boundary = end - len(courtesy)
end = boundary
while end and value[end - 1].isspace():
end -= 1
lowered = value[:end].casefold()
break
token_match = re.search(r"(?:sources?|citations?|links?)$", lowered)
if token_match is None:
return False
prefix = value[:token_match.start()]
original_cursor = len(prefix)
courtesy_end = original_cursor
while courtesy_end and prefix[courtesy_end - 1].isspace():
courtesy_end -= 1
cursors = [original_cursor]
for courtesy in ("please", "pls"):
start = courtesy_end - len(courtesy)
if start >= 0 and prefix[start:courtesy_end].casefold() == courtesy:
if courtesy_end < original_cursor and (start == 0 or prefix[start - 1].isspace()):
cursors.append(start)
break
for cursor in cursors:
while cursor and prefix[cursor - 1].isspace():
if prefix[cursor - 1] == "\n":
return True
cursor -= 1
if cursor == 0 or prefix[cursor - 1] in ".!?;,":
return True
return False
return matches_at(len(value)) or (value.endswith("\n") and matches_at(len(value) - 1))
def requested_web_source_links(user_text):
return bool(re.search(
text = str(user_text or '')
return _terminal_source_link_clause(text) or bool(re.search(
r'\b(?:return|give|show|include|provide|cite|find)\b.{0,35}\b(?:source\s+)?links?\b'
r'|(?:^|[.!?;,\n])\s*(?:(?:pls|please)\s+)?(?:sources?|citations?|links?)\s*(?:pls|please)?\s*[.!?]*$'
r'|\b(?:\d+|one|two|three|four|five)\s+(?:official\s+)?(?:source\s+)?links?\b'
r'|\bofficial\s+source\b'
r'|\b(?:with|include|provide|cite|show|give|find)\s+(?:the\s+)?(?:official\s+)?(?:sources|citations)\b'
r'|\blink\s+(?:to\s+)?(?:the\s+|your\s+)?(?:original\s+|official\s+)?(?:instructions|sources|documentation|articles?|reports?|studies|manuals?|guides?)\b'
r'|\b(?:find|locate|get|download)\b.{0,60}\bofficial\b.{0,60}\b(?:manual|guide|handbook|pdf|documentation)\b'
r'|\b(?:find|locate|get|download)\b.{0,80}\b(?:manual|guide|handbook|pdf)\b.{0,40}\b(?:online|official)\b',
str(user_text or ''),
text,
re.IGNORECASE,
))
@@ -4300,12 +4487,67 @@ def unbound_lookup_reference(user_text, history, *, supplied_context=False):
))
def _web_source_rows(text):
"""Extract numbered source title/URL rows with monotonic line scans."""
rows = []
position = 0
length = len(text)
while position < length:
if position and text[position - 1] != "\n":
newline = text.find("\n", position)
if newline < 0:
break
position = newline + 1
continue
cursor = position
if cursor >= length or text[cursor] != "[":
newline = text.find("\n", cursor)
if newline < 0:
break
position = newline + 1
continue
cursor += 1
digit_start = cursor
while cursor < length and text[cursor].isdigit():
cursor += 1
if cursor == digit_start or cursor >= length or text[cursor] != "]":
position += 1
continue
cursor += 1
if cursor >= length or not text[cursor].isspace():
position += 1
continue
while cursor < length and text[cursor].isspace():
cursor += 1
title_start = cursor
title_line_end = text.find("\n", title_start)
if title_line_end < 0:
break
title_end = title_line_end
while title_end > title_start and text[title_end - 1].isspace():
title_end -= 1
url_start = title_line_end + 1
while url_start < length and text[url_start].isspace():
url_start += 1
scheme_length = 7 if text.startswith("http://", url_start) else 8 if text.startswith("https://", url_start) else 0
if title_end > title_start and scheme_length:
url_end = url_start + scheme_length
while url_end < length and not text[url_end].isspace():
url_end += 1
rows.append((text[title_start:title_end], text[url_start:url_end]))
position = url_end
continue
newline = text.find("\n", position)
if newline < 0:
break
position = newline + 1
return rows
def web_source_links(raw, *, max_items=1, prefer_official=False, query=''):
"""Extract stable title/URL pairs from the web tool's source preamble."""
text = str(raw or '')
rows = re.findall(
r'^\[\d+\]\s+(.+?)\s*\n\s*(https?://\S+)', text, re.MULTILINE,
)
rows = _web_source_rows(text)
query_tokens = set(re.findall(r'[a-z0-9]+', str(query or '').casefold())) - {
'the', 'a', 'an', 'official', 'source', 'link', 'page', 'website',
'site', 'guide', 'search', 'find', 'for', 'return',
@@ -5834,7 +6076,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
provider_error = payload['error']
if isinstance(provider_error, dict):
provider_error = provider_error.get('message') or provider_error.get('detail')
detail = str(provider_error or 'Unknown provider error').strip()[:300]
detail = str(provider_error or 'Unknown provider error').strip()
raise ProviderStreamError(detail)
usage = payload.get('usage') or {}
if usage:
@@ -6353,7 +6595,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
if (
native_workspace_enabled
and content
and DETAILED_VIDEO_REQUEST.search(direct_user_text)
and contains_detailed_sequence_request(direct_user_text)
and successful_video_inspections == 1
and not media_detail_nudge_sent
and round_number < round_limit
@@ -7071,7 +7313,14 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
):
artifact_body_handoff_attempts += 1
artifact_body_handoff_target = handoff_target
result = {'error': str(exc).splitlines()[0][:300], 'exit_code': 1}
logging.getLogger(__name__).warning(
'Clean v3 tool call failed: %s', exc, exc_info=True,
)
result = {
'error': _public_preview_tool_error(exc, execution_attempted=execution_attempted),
'error_category': 'tool_execution_error' if execution_attempted else 'invalid_tool_arguments',
'exit_code': 1,
}
if (canonical(block.tool_type if block is not None else name) == 'youtube_tool'
and result.get('exit_code') not in (None, 0)):
round_recovery_messages.append(
@@ -7260,12 +7509,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
requested_browser_url = private_browser_open_url(args)
effective_browser_url = private_browser_effective_url(result)
browser_search_url = requested_browser_url or effective_browser_url
is_search_engine_navigation = bool(re.search(
r'https?://(?:[^/]+\.)?(?:google\.[^/]+|bing\.com|duckduckgo\.com)'
r'/(?:search|sorry|html|lite|\?)',
browser_search_url,
re.I,
)) or bool(re.search(
is_search_engine_navigation = contains_search_engine_navigation(
browser_search_url
) or bool(re.search(
r'https?://(?:[^/]+\.)?google\.[^/]+/sorry/',
effective_browser_url,
re.I,
@@ -7433,6 +7679,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
'execution_attempted': execution_attempted,
'blocked': policy_denied or schema is None,
'desc': desc, 'round': round_number}
for error_category in ('tool_execution_error', 'invalid_tool_arguments'):
if result.get('error_category') == error_category:
tool_event['error_category'] = error_category
reader_event = email_reader_event(direct_user_text, actual_tool, args, result, failed=failed)
if reader_event:
yield event(reader_event)
@@ -8005,9 +8254,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
else:
yield event({'delta': '\nThe preview reached its round limit. Please narrow the request.'})
except ProviderStreamError as exc:
detail = f'The selected model provider failed while generating: {exc}'
logging.getLogger(__name__).warning('Clean v3 provider stream failed: %s', exc)
yield f'event: error\ndata: {json.dumps({"status": 502, "error": detail})}\n\n'
detail = 'The selected model provider failed while generating. Retry or choose another model.'
logging.getLogger(__name__).warning('Clean v3 provider stream failed: %s', exc, exc_info=True)
yield f'event: error\ndata: {json.dumps({"status": 502, "error": detail, "error_category": "provider_stream_error"})}\n\n'
return
except httpx.HTTPStatusError as exc:
status = exc.response.status_code
+210
View File
@@ -0,0 +1,210 @@
"""Forward-only helpers for permissive text grammars.
These helpers retain the existing regular expressions as anchored token
parsers while preventing ``re.search`` from retrying the same token suffix at
every embedded prefix.
"""
from __future__ import annotations
import re
from collections.abc import Iterator
from typing import Match, Pattern
def iter_prefixed_token_matches(
text: str,
candidate_re: Pattern[str],
anchored_re: Pattern[str],
token_tail_re: Pattern[str],
) -> Iterator[Match[str]]:
"""Yield legacy greedy matches after testing one prefix per token.
``anchored_re`` must start with the same fixed prefix recognized by
``candidate_re``. ``token_tail_re`` describes characters that the
anchored grammar can consume after that prefix. If the first prefix in
such a token cannot match, a later embedded prefix cannot match either:
its suffix was already available to the first attempt. Advancing to the
token boundary makes failed scans linear without changing successful
greedy captures.
"""
pos = 0
while candidate := candidate_re.search(text, pos):
match = anchored_re.match(text, candidate.start())
if match is not None:
yield match
pos = match.end()
continue
tail = token_tail_re.match(text, candidate.end())
pos = max(candidate.end(), tail.end() if tail is not None else candidate.end())
def has_prefixed_token_match(
text: str,
candidate_re: Pattern[str],
anchored_re: Pattern[str],
token_tail_re: Pattern[str],
) -> bool:
"""Return whether ``iter_prefixed_token_matches`` yields a match."""
return next(
iter_prefixed_token_matches(text, candidate_re, anchored_re, token_tail_re),
None,
) is not None
_HTTP_URL_PREFIX_RE = re.compile(r"https?://", re.IGNORECASE)
_VIDEO_DETAIL_BASE_RE = re.compile(
r"\b(?:how\s+many|count|break\s*points?|timestamps?|what\s+time|score(?:board)?s?)\b",
re.IGNORECASE,
)
_VIDEO_DETAIL_EXTENDED_RE = re.compile(
r"\b(?:sequence|in\s+order|chronological|at\s+what\s+time)\b|"
r"(?:多少|几次|何时|什么时候|时间|顺序)",
re.IGNORECASE,
)
_WHEN_RE = re.compile(r"\bwhen\s+", re.IGNORECASE)
_WHEN_TARGET_RE = re.compile(r"(?:end|happen)\b", re.IGNORECASE)
_FIRST_RE = re.compile(r"\bfirst\s+", re.IGNORECASE)
_FIRST_TARGET_RE = re.compile(r"(?:save|attempt|event)\b", re.IGNORECASE)
def contains_search_engine_navigation(text: str) -> bool:
"""Match the legacy Google/Bing/DuckDuckGo navigation URL grammar.
The old expression backtracked through every possible optional subdomain
split. Parsing the host up to its first slash gives the same accepted
hosts and path prefixes with one pass per URL candidate.
"""
pos = 0
while candidate := _HTTP_URL_PREFIX_RE.search(text, pos):
host_start = candidate.end()
path_start = text.find("/", host_start)
if path_start < 0:
return False
host = text[host_start:path_start].casefold()
path = text[path_start + 1:path_start + 7].casefold()
google_at = host.rfind(".google.")
recognized_host = (
(host.startswith("google.") and len(host) > len("google."))
or (google_at >= 0 and google_at + len(".google.") < len(host))
or host == "bing.com"
or host.endswith(".bing.com")
or host == "duckduckgo.com"
or host.endswith(".duckduckgo.com")
)
if recognized_host and path.startswith(("search", "sorry", "html", "lite", "?")):
return True
# A later URL may begin in the path. Resume after this scheme rather
# than skipping the whole non-whitespace region.
pos = candidate.end()
return False
def contains_detailed_sequence_request(text: str, *, include_first: bool = True) -> bool:
"""Recognize count/order/timing requests without overlapping ``.*`` scans."""
value = str(text or "")
if _VIDEO_DETAIL_BASE_RE.search(value):
return True
if include_first and _VIDEO_DETAIL_EXTENDED_RE.search(value):
return True
pos = 0
while when := _WHEN_RE.search(value, pos):
line_end = value.find("\n", when.end())
if line_end < 0:
line_end = len(value)
if _WHEN_TARGET_RE.search(value, when.end(), line_end) is not None:
return True
pos = line_end + 1
if include_first:
pos = 0
while first := _FIRST_RE.search(value, pos):
line_end = value.find("\n", first.end())
if line_end < 0:
line_end = len(value)
if _FIRST_TARGET_RE.search(value, first.end(), line_end) is not None:
return True
pos = line_end + 1
return False
def iter_angle_contents(text: str) -> Iterator[tuple[int, int, str]]:
"""Yield nonempty flat ``<...>`` contents with monotonic delimiters."""
pos = 0
while (start := text.find("<", pos)) >= 0:
end = text.find(">", start + 1)
if end < 0:
return
if end > start + 1:
yield start, end + 1, text[start + 1:end]
pos = end + 1
else:
pos = start + 1
def iter_markdown_links(
text: str,
*,
target_prefix: str = "",
target_re: Pattern[str] | None = None,
) -> Iterator[tuple[int, int, str, str]]:
"""Yield flat Markdown links accepted by the legacy link regexes."""
pos = 0
marker = "](" + target_prefix
while (start := text.find("[", pos)) >= 0:
label_end = text.find("]", start + 1)
if label_end < 0:
return
if label_end == start + 1:
pos = start + 1
continue
if not text.startswith(marker, label_end):
pos = label_end + 1
continue
target_start = label_end + len(marker)
target_end = text.find(")", target_start)
if target_end < 0:
return
target = text[target_start:target_end]
if target and (target_re is None or target_re.fullmatch(target)):
yield start, target_end + 1, text[start + 1:label_end], target
pos = target_end + 1
else:
pos = label_end + 1
def replace_markdown_links_with_labels(text: str) -> str:
"""Linear equivalent of replacing flat Markdown links with their labels."""
links = list(iter_markdown_links(text))
if not links:
return text
out = []
pos = 0
for start, end, label, _target in links:
out.extend((text[pos:start], label))
pos = end
out.append(text[pos:])
return "".join(out)
def first_tag_content(text: str, tag: str, *, allow_attributes: bool = False) -> str | None:
"""Return the first flat tag body using forward-only opener/closer scans."""
opener_re = re.compile(r"<" + re.escape(tag), re.IGNORECASE)
closer_re = re.compile(r"</" + re.escape(tag) + r">", re.IGNORECASE)
pos = 0
while opener := opener_re.search(text, pos):
name_end = opener.end()
if name_end < len(text) and text[name_end] == ">":
body_start = name_end + 1
elif allow_attributes and name_end < len(text) and text[name_end].isspace():
tag_end = text.find(">", name_end + 1)
if tag_end < 0:
return None
body_start = tag_end + 1
else:
pos = name_end
continue
closer = closer_re.search(text, body_start)
if closer is None:
return None
return text[body_start:closer.start()]
return None
+119 -18
View File
@@ -194,10 +194,33 @@ _TOOL_CODE_OPEN_RE = re.compile(r"<tool_code>\s*\{", re.IGNORECASE)
_TOOL_CODE_CLOSE_RE = re.compile(r"\}\s*</tool_code>", re.IGNORECASE)
# Pattern 4b: Gemma-style <|tool_call|> call:tool_name{args} <tool_call|>
_GEMMA_TOOL_CALL_RE = re.compile(
r"<\|?tool_call\|?>\s*call:([\w\d_-]+)\s*(\{[\s\S]*?\})\s*<\|?tool_call\|?>",
_GEMMA_TOOL_CALL_OPEN_RE = re.compile(
r"<\|?tool_call\|?>\s*call:([\w\d_-]+)\s*\{",
re.IGNORECASE,
)
_GEMMA_TOOL_CALL_CLOSE_RE = re.compile(
r"\}\s*<\|?tool_call\|?>",
re.IGNORECASE,
)
# Native Qwen markup shares the same non-nesting delimiter grammar as the
# XML helpers. Literal closers keep whitespace and opener floods linear;
# values are stripped by the caller, as in the original regex path.
_QWEN_FUNCTION_OPEN_RE = re.compile(r"<function=([A-Za-z_][\w:.-]*)>\s*")
_QWEN_FUNCTION_CLOSE_RE = re.compile(r"</function>")
_QWEN_PARAMETER_OPEN_RE = re.compile(r"<parameter=([A-Za-z_]\w*)>\s*")
_QWEN_PARAMETER_CLOSE_RE = re.compile(r"</parameter>")
_QWEN_PYTHON_ARG_KEY_RE = re.compile(r"[A-Za-z_]\w*")
_QWEN_PYTHON_ARG_VALUE_RE = re.compile(r"\s*=\s*(['\"].*?['\"]|[^,]+)")
_ANGLE_TAG_OPEN_RE = re.compile(r"<")
_NONEMPTY_ANGLE_TAG_OPEN_RE = re.compile(r"<(?=[^>])")
_ANGLE_TAG_CLOSE_RE = re.compile(r">")
_EMAIL_LOCAL_RE = re.compile(r"[\w.+-]+")
_EMAIL_ADDRESS_RE = re.compile(r"[\w.+-]+@[\w.-]+\.\w+")
_ASCII_EMAIL_LOCAL_RE = re.compile(r"[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+")
_ASCII_EMAIL_ADDRESS_RE = re.compile(
r"[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}"
)
# Pattern 4c: Open-function wrapper emitted by some local MLX/Exo models.
# Example:
@@ -542,6 +565,34 @@ _PLAIN_UI_OPEN_PANEL_RE = re.compile(
r"((?:\s+(?:day|week|month|year|agenda)(?:\s+view)?(?:\s+\d{4}-\d{2}(?:-\d{2})?)?)?)"
r"\s*(?:`{1,3})?\s*$"
)
_PLAIN_UI_CANDIDATE_RE = re.compile(r"ui_control", re.IGNORECASE)
def _iter_plain_ui_open_panel(text: str):
"""Yield line-anchored UI commands without retrying every blank line."""
pos = 0
while candidate := _PLAIN_UI_CANDIDATE_RE.search(text, pos):
line_start = text.rfind("\n", 0, candidate.start()) + 1
match = _PLAIN_UI_OPEN_PANEL_RE.match(text, line_start)
if match is not None:
yield match
pos = match.end()
continue
newline = text.find("\n", candidate.end())
pos = len(text) if newline < 0 else newline + 1
def _strip_plain_ui_open_panel(text: str) -> str:
matches = list(_iter_plain_ui_open_panel(text))
if not matches:
return text
out = []
pos = 0
for match in matches:
out.append(text[pos:match.start()])
pos = match.end()
out.append(text[pos:])
return "".join(out)
# ---------------------------------------------------------------------------
@@ -992,6 +1043,43 @@ def _strip_raw_openai_tool_call_json(text: str) -> str:
return "".join(pieces)
def iter_email_addresses(text: str, *, ascii_only: bool = False):
"""Find legacy email-shaped strings without retrying every word suffix.
Match the address only at the start of each maximal local-part token.
When that attempt fails, every suffix has the same '@'/domain boundary
and must fail too. Local/domain character runs are each scanned a bounded
number of times, including malformed text with no '@' or domain dot.
"""
local_re = _ASCII_EMAIL_LOCAL_RE if ascii_only else _EMAIL_LOCAL_RE
address_re = _ASCII_EMAIL_ADDRESS_RE if ascii_only else _EMAIL_ADDRESS_RE
pos = 0
while token := local_re.search(text, pos):
address = address_re.match(text, token.start())
if address is not None:
yield address.group(0)
pos = address.end()
else:
pos = token.end()
def _iter_qwen_python_args(raw_args: str):
"""Match each maximal key once, then its value at that fixed position.
If a word has no following equals sign, none of its suffixes can have one
either. Advancing past it avoids the old unanchored regex's O(n^2) retries
on a long malformed key, while preserving permissive quoted/bare values.
"""
pos = 0
while key := _QWEN_PYTHON_ARG_KEY_RE.search(raw_args, pos):
value = _QWEN_PYTHON_ARG_VALUE_RE.match(raw_args, key.end())
if value is not None:
yield key.group(0), value.group(1)
pos = value.end()
else:
pos = key.end()
def _parse_qwen3_native_text_call(
text: str,
additional_tool_names: Optional[Iterable[str]] = None,
@@ -1018,13 +1106,14 @@ def _parse_qwen3_native_text_call(
# Qwen native text rendering:
# <tool_call><function=manage_notes><parameter=action>list</parameter>...
fn_match = re.search(r"<function=([A-Za-z_][\w:.-]*)>\s*([\s\S]*?)\s*</function>", text)
fn_match = next(_iter_named_blocks(
text, _QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE
), None)
if fn_match:
name, raw_body = fn_match.groups()
name, raw_body = fn_match
args = {}
for key, raw_value in re.findall(
r"<parameter=([A-Za-z_]\w*)>\s*([\s\S]*?)\s*</parameter>",
raw_body,
for key, raw_value in _iter_named_blocks(
raw_body.rstrip(), _QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE,
):
value = raw_value.strip()
if value and value[0] in "[{\"":
@@ -1085,7 +1174,7 @@ def _parse_qwen3_native_text_call(
or normalized_name in declared_names
):
args = {}
for key, raw_value in re.findall(r"([A-Za-z_]\w*)\s*=\s*(['\"].*?['\"]|[^,]+)", raw_args):
for key, raw_value in _iter_qwen_python_args(raw_args):
try:
args[key] = ast.literal_eval(raw_value.strip())
except (ValueError, SyntaxError):
@@ -1509,9 +1598,9 @@ def _iter_delimited(text, open_re, close_re):
pos = cm.end()
def _strip_delimited(text: str, open_re, close_re) -> str:
"""Remove every ``open_re ... close_re`` span (forward-only; see
_iter_delimited). Equivalent to ``open_re([\\s\\S]*?)close_re`` ``re.sub('')``
def _strip_delimited(text: str, open_re, close_re, replacement: str = "") -> str:
"""Replace every ``open_re ... close_re`` span (forward-only; see
_iter_delimited). Equivalent to ``open_re([\\s\\S]*?)close_re`` substitution
for these delimiters, without the O(n^2) rescan on unclosed openers."""
spans = list(_iter_delimited(text, open_re, close_re))
if not spans:
@@ -1520,11 +1609,23 @@ def _strip_delimited(text: str, open_re, close_re) -> str:
last = 0
for match_start, _inner_start, _inner_end, match_end in spans:
out.append(text[last:match_start])
out.append(replacement)
last = match_end
out.append(text[last:])
return "".join(out)
def strip_angle_tags(text: str, replacement: str = "", *, allow_empty: bool = False) -> str:
"""Linear equivalent of replacing ``<[^>]+>`` (or ``<[^>]*>``).
This preserves the existing flat text cleanup, including nested '<' and
malformed tails; it does not interpret HTML. Stop once no '>' is reachable
instead of retrying a suffix scan at every '<' in untrusted text.
"""
opener = _ANGLE_TAG_OPEN_RE if allow_empty else _NONEMPTY_ANGLE_TAG_OPEN_RE
return _strip_delimited(text, opener, _ANGLE_TAG_CLOSE_RE, replacement)
def _iter_named_blocks(text, open_re, close_re):
"""Forward-only equivalent of ``open_re([\\s\\S]*?)close_re`` finditer where
open_re captures a name in group 1: yield ``(name, body)``, pairing each
@@ -1958,10 +2059,10 @@ def parse_tool_blocks(
# Pattern 4b: Gemma-style <|tool_call|> blocks
if not blocks:
for m in _GEMMA_TOOL_CALL_RE.finditer(text):
tool_name = m.group(1)
body = m.group(2)
block = _parse_gemma_tool_call(tool_name, body)
for tool_name, body in _iter_named_blocks(
text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE
):
block = _parse_gemma_tool_call(tool_name, "{" + body + "}")
if block:
blocks.append(block)
@@ -1998,7 +2099,7 @@ def parse_tool_blocks(
# from weaker native-tool models after reading the tool docs but failing to
# emit the actual structured call.
if not blocks:
m = _PLAIN_UI_OPEN_PANEL_RE.search(text)
m = next(_iter_plain_ui_open_panel(text), None)
if m:
blocks.append(ToolBlock("ui_control", f"open_panel {m.group(1).lower()}{m.group(2).lower()}".strip()))
@@ -2041,7 +2142,7 @@ def strip_tool_blocks(
cleaned = _strip_delimited(cleaned, _XML_TOOL_CALL_OPEN_RE, _XML_TOOL_CALL_CLOSE_RE)
cleaned = _XML_OPEN_TOOL_CALL_RE.sub('', cleaned)
cleaned = _strip_delimited(cleaned, _TOOL_CODE_OPEN_RE, _TOOL_CODE_CLOSE_RE)
cleaned = _GEMMA_TOOL_CALL_RE.sub('', cleaned)
cleaned = _strip_delimited(cleaned, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE)
cleaned = _strip_delimited(cleaned, _FUNCTION_MODEL_OPEN_RE, _FUNCTION_MODEL_CLOSE_RE)
cleaned = _strip_raw_openai_tool_call_json(cleaned)
declared_xml_calls = _parse_declared_direct_xml_calls(
@@ -2060,7 +2161,7 @@ def strip_tool_blocks(
if raw_web_json:
_, (start, end) = raw_web_json
cleaned = cleaned[:start] + cleaned[end:]
cleaned = _PLAIN_UI_OPEN_PANEL_RE.sub("", cleaned)
cleaned = _strip_plain_ui_open_panel(cleaned)
# Strip bare <invoke> blocks not wrapped in <tool_call>
cleaned = _strip_bare_invoke_markup(cleaned)
cleaned = re.sub(r'\n{3,}', '\n\n', cleaned)
+255 -57
View File
@@ -17,6 +17,7 @@ from typing import Iterable, Mapping
from src.action_intents import classify_tool_intent
from src.tool_policy import ToolPolicy
from src.text_scanning import has_prefixed_token_match
FAMILY_TOOLS = {
@@ -47,6 +48,62 @@ FAMILY_TOOLS = {
CONTRACT_CORE_TOOLS = frozenset({
"bash", "python", "read_file", "web_search", "web_fetch", "ask_user",
})
_WORKSPACE_PREFIX_RE = re.compile(r"/workspace/", re.I)
_WORKSPACE_ARTIFACT_RE = re.compile(
r"/workspace/[^\s`\"']+\.(?:csv|html?|json|md|svg|txt)\b", re.I
)
_WORKSPACE_OUTPUT_PREFIX_RE = re.compile(r"/workspace/(?!input/)", re.I)
_WORKSPACE_OUTPUT_RE = re.compile(
r"/workspace/(?!input/)[^\s`\"']+\."
r"(?:csv|html?|json|md|svg|txt|avif|bmp|gif|jpe?g|png|webp|pdf|mp4|webm)\b",
re.I,
)
_WORKSPACE_TOKEN_TAIL_RE = re.compile(r"[^\s`\"']*")
def _mentions_workspace_artifact(text: str) -> bool:
return has_prefixed_token_match(
str(text or ""),
_WORKSPACE_PREFIX_RE,
_WORKSPACE_ARTIFACT_RE,
_WORKSPACE_TOKEN_TAIL_RE,
)
def _mentions_workspace_output(text: str) -> bool:
return has_prefixed_token_match(
str(text or ""),
_WORKSPACE_OUTPUT_PREFIX_RE,
_WORKSPACE_OUTPUT_RE,
_WORKSPACE_TOKEN_TAIL_RE,
)
def _mentions_under_budget(text: str) -> bool:
value = str(text or "")
for under in re.finditer(r"\bunder", value, re.I):
cursor = under.end()
if cursor >= len(value) or not value[cursor].isspace():
continue
while cursor < len(value) and value[cursor].isspace():
cursor += 1
if cursor < len(value) and value[cursor] in "¥$€£":
cursor += 1
while cursor < len(value) and value[cursor].isspace():
cursor += 1
digit_start = cursor
while cursor < len(value) and value[cursor].isdecimal():
cursor += 1
if cursor == digit_start:
continue
if cursor == len(value) or not (value[cursor].isalnum() or value[cursor] == "_"):
return True
if value[cursor:cursor + 3].casefold() == "yen":
cursor += 3
if cursor == len(value) or not (value[cursor].isalnum() or value[cursor] == "_"):
return True
return False
_FAMILY_WORDS = {
"calendar": r"\b(?:calendar|calender|events?|appointments?|meetings?|agenda)\b",
"notes": r"\b(?:notes?|checklists?|groceries|remind\s+me)\b",
@@ -122,13 +179,7 @@ def _normalize_request_lead(value: str) -> str:
text,
flags=re.I,
)
text = re.sub(
r"^(?:never\s*mind|scratch\s+that)\s*[,;:—–-]?\s*"
r"(?=(?:open|show|list|read|search|find|check|switch|go)\b)",
"",
text,
flags=re.I,
)
text = _strip_cancelled_request_lead(text)
text = re.sub(r"^k(?:ay)?\s*[,!]?\s+(?=\S)", "", text, flags=re.I)
text = re.sub(
r"^(?:(?:great|nice|cool)\s*[,!.]|thanks?\s*[.!])\s+"
@@ -163,6 +214,21 @@ def _normalize_request_lead(value: str) -> str:
text = re.sub(r"^((?:can|could|would|will)\s+)u\b", r"\1you", text, flags=re.I)
text = re.sub(r"\boffical\b", "official", text, flags=re.I)
return text
def _strip_cancelled_request_lead(text: str) -> str:
lead = re.match(r"^(?:never\s*mind|scratch\s+that)", text, re.I)
if lead is None:
return text
cursor = lead.end()
while cursor < len(text) and text[cursor].isspace():
cursor += 1
if cursor < len(text) and text[cursor] in ",;:—–-":
cursor += 1
while cursor < len(text) and text[cursor].isspace():
cursor += 1
action = re.match(r"(?:open|show|list|read|search|find|check|switch|go)\b", text[cursor:], re.I)
return text[cursor:] if action is not None else text
_MISSPELLED_RESEARCH_ACTION = re.compile(
r"^\s*" + _REQUEST_PREFIX + r"(?:reserch|reasearch|reseach)\b",
re.I,
@@ -195,6 +261,21 @@ _PURE_ACTION_PROHIBITION = re.compile(
r"[^.;\n]*[.!?]*\s*$",
re.I,
)
def _is_pure_action_prohibition(text: str) -> bool:
value = str(text or "").strip()
prefix = re.match(
r"(?:read[- ]only(?:\s+and)?\s+)?(?:do\s+not|don['’]?t|never)\s+"
r"(?:add|create|make|write|draft|edit|change|update|delete|remove|send|reply|"
r"run|execute|download|serve|open|save|schedule|transcribe|inspect)\b",
value,
re.I,
)
if prefix is None:
return False
tail = value[prefix.end():].rstrip(".!?")
return not any(char in ".;\n" for char in tail)
_RETURN_TO_ACTION = re.compile(r"^\s*" + _REQUEST_PREFIX + r"return\s+to\b", re.I)
_PANEL_NAVIGATION = re.compile(
r"^\s*" + _REQUEST_PREFIX
@@ -447,6 +528,107 @@ _WARM_RECALL_WITH_FOLLOWUP = re.compile(
r"(?P<followup>(?:what(?:['’]?s|\s+is)?|which|who|where|when|how|show|open|read|list|find|search)\b[\s\S]{0,180})$",
re.I,
)
_WARM_TARGET_RE = re.compile(
r"calendar|emails?|inbox|notes?|tasks?|skills?|memories|memory|"
r"documents?|docs?|web|browser|cookbook|files?|shell",
re.I,
)
_WARM_FOLLOWUP_RE = re.compile(
r"(?:what(?:['’]?s|\s+is)?|which|who|where|when|how|show|open|read|list|find|search)"
r"\b[\s\S]{0,180}\Z",
re.I,
)
def _consume_space(value: str, cursor: int, *, required: bool = False) -> int | None:
start = cursor
while cursor < len(value) and value[cursor].isspace():
cursor += 1
return None if required and cursor == start else cursor
def _phrase_end(value: str, cursor: int, phrase: str) -> int | None:
for index, word in enumerate(phrase.split(" ")):
if value[cursor:cursor + len(word)].casefold() != word:
return None
cursor += len(word)
if index + 1 < len(phrase.split(" ")):
cursor = _consume_space(value, cursor, required=True)
if cursor is None:
return None
return cursor
def _warm_recall_parts(value: str, *, with_followup: bool = False) -> tuple[str, str] | None:
value = str(value or "")
cursor = _consume_space(value, 0) or 0
states = [cursor]
discourse = re.match(r"(?:ok(?:ay)?|and|then)", value[cursor:], re.I)
if discourse is not None:
after = _consume_space(value, cursor + discourse.end(), required=True)
if after is not None:
states.insert(0, after)
action_phrases = ("back to", "return to", "what about", "check", "show", "open")
action_states: list[int] = []
for state in states:
if not with_followup:
action_states.append(state)
for phrase in action_phrases:
end = _phrase_end(value, state, phrase)
if end is None:
continue
if with_followup:
end = _consume_space(value, end, required=True)
if end is None:
continue
action_states.append(end)
for state in dict.fromkeys(action_states):
state = _consume_space(value, state) or 0
possessive_states = [state]
for possessive in ("my", "the"):
if value[state:state + len(possessive)].casefold() == possessive:
possessive_states.insert(0, state + len(possessive))
for target_state in possessive_states:
target_state = _consume_space(value, target_state) or 0
target = _WARM_TARGET_RE.match(value, target_state)
if target is None:
continue
target_text = target.group(0)
suffix = target.end()
if with_followup:
if suffix < len(value) and (value[suffix].isalnum() or value[suffix] == "_"):
continue
separator_states = []
spaced = _consume_space(value, suffix) or 0
if spaced > suffix:
separator_states.append(spaced)
if spaced < len(value) and value[spaced] in "-—,:;":
separator_states.append(_consume_space(value, spaced + 1) or 0)
for conjunction in ("and", "then"):
end = spaced + len(conjunction)
if (value[spaced:end].casefold() == conjunction
and (end == len(value) or not (value[end].isalnum() or value[end] == "_"))):
separator_states.append(_consume_space(value, end) or 0)
for followup_start in dict.fromkeys(separator_states):
followup = _WARM_FOLLOWUP_RE.match(value, followup_start)
if followup is not None:
return target_text, followup.group(0)
continue
suffix = _consume_space(value, suffix) or 0
suffix_states = [suffix]
for word in ("again", "now"):
if value[suffix:suffix + len(word)].casefold() == word:
suffix_states.insert(0, suffix + len(word))
for end in suffix_states:
while end < len(value) and value[end] in ".!?":
end += 1
end = _consume_space(value, end) or 0
if end == len(value):
return target_text, ""
return None
_REQUIRED_TOOLS = {
"calendar": "manage_calendar", "notes": "manage_notes",
"tasks": "manage_tasks", "skills": "manage_skills",
@@ -851,11 +1033,7 @@ def selected_tools_for_request(message: str) -> frozenset[str] | None:
tools = {"inspect_media"}
if (
re.search(r"\b(?:create|write|save|build|produce)\b", raw_text, re.I)
and re.search(
r"(?:file://)?/workspace/[^\s`\"']+\.(?:csv|html?|json|md|svg|txt)\b",
raw_text,
re.I,
)
and _mentions_workspace_artifact(raw_text)
):
tools.update({"write_file", "read_file"})
if re.search(r"\b(?:preview|render|open)\b[^.\n]{0,100}\b(?:page|html|browser)\b", raw_text, re.I):
@@ -2118,6 +2296,31 @@ _READ_PRESENTATION_SUFFIX = re.compile(
)
def _terminal_clause_match(text: str, core_pattern: str) -> re.Match[str] | None:
"""Match a terminal clause after removing its ambiguous punctuation tail."""
end = len(text)
while end and text[end - 1].isspace():
end -= 1
while end and text[end - 1] in ".!?":
end -= 1
possessive = core_pattern.replace(r"\s+", r"\s++").replace(r"\s*", r"\s*+")
return re.search(possessive + r"$", text[:end], re.I)
def _strip_terminal_but(text: str) -> str:
end = len(text)
while end and text[end - 1].isspace():
end -= 1
if end < 3 or text[end - 3:end].casefold() != "but":
return text
start = end - 3
if start == 0 or not text[start - 1].isspace():
return text
while start and text[start - 1].isspace():
start -= 1
return text[:start]
def _read_request_and_limit(message: str) -> tuple[str, int | None]:
"""Strip only whole, known presentation/safety suffixes, never actions."""
text = _normalize_request_lead(message)
@@ -2167,11 +2370,11 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
if keep_few_suffix:
maximum = 3
text = text[:keep_few_suffix.start()].strip()
few_suffix = re.search(
few_suffix = _terminal_clause_match(
text,
r"[,.;?]\s*(?:(?:only|just)\s+)?(?:(?:list|show)\s+(?:me\s+)?)?a\s+few"
r"(?:\s+(?:task\s+)?(?:names?|items?|results?|entries?))?"
r"(?:\s+and\s+(?:whether|if)\s+[^.;\n]+)?[.!?]*\s*$",
text, re.I,
r"(?:\s+and\s+(?:whether|if)\s+[^.;\n]+)?",
)
if few_suffix:
maximum = 3
@@ -2183,42 +2386,43 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
text,
flags=re.I,
).strip()
text = re.sub(
terminal_read_only = _terminal_clause_match(
text,
r"[.;]\s*read[- ]only(?:\s+(?:please|pls|plz))?\s*,?\s*"
r"(?:(?:and\s+)?(?:do\s+not|don['’]?t|dont)\s+"
r"(?:change|edit|modify)(?:\s+or\s+send)?\s+(?:anything|data))?"
r"[.!?]*\s*$",
"",
text,
flags=re.I,
).strip()
r"(?:change|edit|modify)(?:\s+or\s+send)?\s+(?:anything|data))?",
)
if terminal_read_only:
text = text[:terminal_read_only.start()].strip()
# Explanatory/safety tails do not alter a preceding exact read request.
text = re.sub(
safety_tail = _terminal_clause_match(
text,
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?(?:do\s+not|don['’]?t|dont)\s+"
r"(?:touch|change|edit|modify)(?:\s+(?:anything|data|them))?(?:\s+yet)?[.!?]*\s*$",
"", text, flags=re.I,
).strip()
r"(?:touch|change|edit|modify)(?:\s+(?:anything|data|them))?(?:\s+yet)?",
)
if safety_tail:
text = text[:safety_tail.start()].strip()
text = re.sub(
r"[.!?]\s*read[- ]only(?:\s+(?:please|pls|plz))?\s*,?\s*"
r"(?:do\s+not|don['’]?t|dont)\s+(?:change|edit|modify)\s+"
r"(?:or\s+send\s+)?anything[.!?]*\s*$",
"", text, flags=re.I,
).strip()
text = re.sub(
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+changes?[.!?]*\s*$",
"", text, flags=re.I,
).strip()
text = re.sub(
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+edits?[.!?]*\s*$",
"", text, flags=re.I,
).strip()
for terminal_core in (
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+changes?",
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+edits?",
):
terminal_match = _terminal_clause_match(text, terminal_core)
if terminal_match:
text = text[:terminal_match.start()].strip()
text = re.sub(
r"[,;]\s*no\s+edits?[.!?]*\s*$", "", text, flags=re.I,
).strip()
text = re.sub(
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+writes?[.!?]*\s*$",
"", text, flags=re.I,
).strip()
terminal_no_write = _terminal_clause_match(
text, r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+writes?"
)
if terminal_no_write:
text = text[:terminal_no_write.start()].strip()
text = re.sub(
r"[.!?]\s*(?:i['’]?m|i\s+am)\s+(?:just\s+)?checking\b[^\n]*$",
"", text, flags=re.I,
@@ -2236,12 +2440,11 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
text,
flags=re.I,
).strip()
text = re.sub(
r"[,;]\s*(?:keep\s+(?:them|it)\s+)?short\s+lines?\s*,?[.!?]*\s*$",
"",
text,
flags=re.I,
).strip()
short_lines = _terminal_clause_match(
text, r"[,;]\s*(?:keep\s+(?:them|it)\s+)?short\s+lines?\s*,?"
)
if short_lines:
text = text[:short_lines.start()].strip()
text = re.sub(
r"[,.;]\s*keep\s+(?:the\s+answer|it|them)\s+short[.!?]*\s*$",
"",
@@ -2318,7 +2521,7 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
value = int(raw) if raw.isdecimal() else _READ_COUNT_WORDS[raw.lower()]
maximum = value if maximum is None else min(maximum, value)
text = text[:conversational_limit.start()].rstrip(' ,.;?')
text = re.sub(r"\s+but\s*$", "", text, flags=re.I)
text = _strip_terminal_but(text)
need_limit = re.search(
r"[.!?]\s*(?:i\s+)?only\s+need\s+(" + _READ_COUNT + r")"
r"(?:\s+(?:short\s+)?(?:titles?|items?|results?|entries?|names?|ones?))?"
@@ -3949,7 +4152,7 @@ def _families_for_tool(tool: str) -> frozenset[str]:
def _clause_capabilities(text: str) -> set[str]:
# A prohibition constrains authority; it must never grant the family named
# only as the forbidden side effect (for example, "do not create a file").
if _PURE_ACTION_PROHIBITION.fullmatch(text):
if _is_pure_action_prohibition(text):
return set()
container_tool = creation_container_tool(text)
if container_tool:
@@ -4610,12 +4813,7 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum
raw_text,
re.I,
)
and re.search(
r"(?:file://)?/workspace/(?!input/)[^\s`\"']+\."
r"(?:csv|html?|json|md|svg|txt|avif|bmp|gif|jpe?g|png|webp|pdf|mp4|webm)\b",
raw_text,
re.I,
)
and _mentions_workspace_output(raw_text)
):
families.add("shell_files")
if re.search(
@@ -5074,7 +5272,7 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum
and re.search(r"\b(?:ones?|top|apps?|services?|providers?)\b", text, re.I)
and re.search(r"\b(?:price|cheap|under|dimensions?|quote|trustworthy|app)\b", text, re.I)
)
or re.search(r"\bunder\s+[¥$€£]?\s*\d+(?:[.,]\d+)?(?:\s*yen)?\b", text, re.I)
or _mentions_under_budget(text)
):
return frozenset({"search_browser"})
if recent_family == ("search_browser",) and re.fullmatch(
@@ -6011,8 +6209,8 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum
elif (recent == ("search_browser",) and families == {"cookbook_admin"}
and re.match(r"^\s*" + _REQUEST_PREFIX + r"(?:search|find)\s+(?:those|them|these)\b", text, re.I)):
families = {"search_browser"}
if not families and (recall := _WARM_RECALL.fullmatch(text)):
target = recall["target"].lower()
if not families and (recall := _warm_recall_parts(text)):
target = recall[0].lower()
family = {
"email": "email", "emails": "email", "inbox": "email",
"note": "notes", "notes": "notes", "task": "tasks", "tasks": "tasks",
@@ -6024,8 +6222,8 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum
}[target]
if family in recently_executed_families(history):
families.add(family)
if not families and (recall := _WARM_RECALL_WITH_FOLLOWUP.fullmatch(text)):
target = recall["target"].lower()
if not families and (recall := _warm_recall_parts(text, with_followup=True)):
target = recall[0].lower()
family = {
"email": "email", "emails": "email", "inbox": "email",
"note": "notes", "notes": "notes", "task": "tasks", "tasks": "tasks",
+3 -1
View File
@@ -4562,7 +4562,9 @@ async def test_preview_provider_stream_error_is_terminal_not_empty_answer(monkey
disabled_tools=set(), tool_policy=ToolPolicy(),
)]
assert raw[-1].startswith('event: error\ndata: ')
assert 'Qwen3_5MTPDraftModel' in raw[-1]
assert 'selected model provider failed' in raw[-1]
assert 'provider_stream_error' in raw[-1]
assert 'Qwen3_5MTPDraftModel' not in raw[-1]
assert all('returned no answer' not in chunk for chunk in raw)
assert all('"type": "metrics"' not in chunk for chunk in raw)
+859
View File
@@ -0,0 +1,859 @@
"""Preserve legacy text-call/listing semantics and bound hostile scans."""
import itertools
import random
import re
import subprocess
import sys
import textwrap
import pytest
from src.agent_loop import (
_calendar_listing_row,
_captures_after_first_prefix,
_contains_email_draft_headers,
_looks_like_agent_reasoning_preamble,
_looks_like_ody_qwen_leaked_tool_text,
_email_account_label,
_email_sender_name,
_contextual_summary_fragment,
_is_terse_link_request,
_is_terse_email_lookup_followup,
_looks_like_destructive_request,
_looks_like_youtube_tool_turn,
_mentions_pdf_url,
_mentions_workspace_script,
_numbered_row_parenthesized_ids,
_pipeline_request_parts,
_parse_explicit_open_panel_request,
_private_browser_product_query,
_read_only_shell_command,
_remaining_checklist_name,
_research_listing_row,
_session_link_from_row,
_session_find_query_capture,
_session_list_summary_from_tool_output,
_split_before_assistant_prompt,
_split_note_items,
_strip_trailing_done,
_strip_horizontal_space_before_lf,
_skill_listing_row,
)
from src.clean_agent_preview import (
_page_listing_request,
_prior_web_source_request,
_terminal_source_link_clause,
_web_source_rows,
declared_workspace_artifacts,
)
from src.text_scanning import (
contains_detailed_sequence_request,
contains_search_engine_navigation,
iter_angle_contents,
iter_markdown_links,
first_tag_content,
replace_markdown_links_with_labels,
)
from src.turn_contract import (
_WARM_RECALL,
_WARM_RECALL_WITH_FOLLOWUP,
_is_pure_action_prohibition,
_mentions_under_budget,
_mentions_workspace_artifact,
_mentions_workspace_output,
_strip_cancelled_request_lead,
_strip_terminal_but,
_terminal_clause_match,
_warm_recall_parts,
)
from src.tool_parsing import (
_GEMMA_TOOL_CALL_OPEN_RE,
_GEMMA_TOOL_CALL_CLOSE_RE,
_QWEN_FUNCTION_OPEN_RE,
_QWEN_FUNCTION_CLOSE_RE,
_QWEN_PARAMETER_OPEN_RE,
_QWEN_PARAMETER_CLOSE_RE,
_iter_named_blocks,
_iter_qwen_python_args,
_strip_delimited,
parse_tool_blocks,
strip_tool_blocks,
strip_angle_tags,
iter_email_addresses,
)
_SESSION_LINK = re.compile(r"(\[(?:\\.|[^\]])+\]\(#session-[^)]+\))")
_GEMMA = re.compile(
r"<\|?tool_call\|?>\s*call:([\w\d_-]+)\s*(\{[\s\S]*?\})\s*<\|?tool_call\|?>",
re.I,
)
_QWEN_FUNCTION = re.compile(r"<function=([A-Za-z_][\w:.-]*)>\s*([\s\S]*?)\s*</function>")
_QWEN_PARAMETER = re.compile(r"<parameter=([A-Za-z_]\w*)>\s*([\s\S]*?)\s*</parameter>")
_QWEN_ARGS = re.compile(r"([A-Za-z_]\w*)\s*=\s*(['\"].*?['\"]|[^,]+)")
_WORKSPACE_SCRIPT = re.compile(
r"/workspace/[^\s`\"']+\.(?:py|pyw|sh|bash|js|mjs|ts|rb|pl)\b", re.I
)
_PDF_URL = re.compile(r"https?://\S+(?:\.pdf\b|/pdf/)", re.I)
_WORKSPACE_ARTIFACT = re.compile(
r"(?:file://)?/workspace/[^\s`\"']+\.(?:csv|html?|json|md|svg|txt)\b", re.I
)
_WORKSPACE_OUTPUT = re.compile(
r"(?:file://)?/workspace/(?!input/)[^\s`\"']+\."
r"(?:csv|html?|json|md|svg|txt|avif|bmp|gif|jpe?g|png|webp|pdf|mp4|webm)\b",
re.I,
)
_SEARCH_ENGINE_URL = re.compile(
r"https?://(?:[^/]+\.)?(?:google\.[^/]+|bing\.com|duckduckgo\.com)"
r"/(?:search|sorry|html|lite|\?)",
re.I,
)
_LEAKED_TOOL_TEXT = re.compile(
r"(<\s*/?\s*(?:function|parameter|tool_call)\b|(?:^|\n)\s*(?:function|parameter)\s*="
r"|\bmanage_(?:notes|calendar|memory|documents|contact)\s*\(|\"function\"\s*:\s*\"(?:manage_|mcp__)"
r"|mcp__email__|(?:^|\n)\s*(?:web_search|web_fetch|private_browser)\s*:)", re.I,
)
_VIDEO_DETAIL_BASE = re.compile(
r"\b(?:how\s+many|count|break\s*points?|timestamps?|what\s+time|"
r"when\s+.*(?:end|happen)|score(?:board)?s?)\b", re.I,
)
_VIDEO_DETAIL_EXTENDED = re.compile(
r"\b(?:how\s+many|count|sequence|in\s+order|chronological|timestamps?|"
r"what\s+time|at\s+what\s+time|when\s+.*(?:end|happen)|"
r"first\s+.*(?:save|attempt|event)|score(?:board)?s?)\b|"
r"(?:多少|几次|何时|什么时候|时间|顺序)", re.I,
)
def test_session_link_matches_legacy_escape_and_greedy_semantics():
cases = [
"no links", "[](#session-id)", "[x](#session-)",
"[x](#session-id)", "[[x](#session-id)",
r"[x\](#session-id)", r"[x\]more](#session-id)",
r"[x\](#session-first)\](#session-last)",
r"[x\](#session-first)\](#session-)",
r"[x\](#session-first)\](#session-unclosed",
"[bad] then [good](#session-id)",
"[x](#session-id[has]brackets)",
"[x](#session-first) [y](#session-second)",
]
rng = random.Random(6503)
tokens = ["[", "]", "\\", "a", "(", ")", "(#session-id)", "(#session-)"]
cases += ["".join(rng.choices(tokens, k=12)) for _ in range(2000)]
cases += ["[" + "".join(label) + "](#session-id)"
for n in range(5) for label in itertools.product("a[]\\", repeat=n)]
for row in cases:
expected = _SESSION_LINK.search(row)
assert _session_link_from_row(row) == (expected.group(0) if expected else ""), row
@pytest.mark.parametrize("row,expected", [
("- **[Chat](#session-id)** (id: `id`, model: qwen, 2 msgs, last active today)",
"- [Chat](#session-id) (last active today)"),
("- [Chat](#session-id) (irrelevant) (model: qwen) (last active later)",
"- [Chat](#session-id)"),
("- [Chat](#session-id) (outer (last active today))",
"- [Chat](#session-id) (last active today)"),
("- plain row", "- plain row"),
])
def test_session_summary_preserves_link_and_metadata(row, expected):
assert _session_list_summary_from_tool_output("Chats:\n" + row) == "Chats:\n" + expected
def test_gemma_delimiters_match_legacy_parse_and_strip():
cases = [
"ordinary prose", "<|tool_call|>call:web_search{query: 'news'}<|tool_call|>",
"before<tool_call> call:read-file {\npath: 'README.md'\n} <tool_call>after",
"<tool_call>call:x{a}<tool_call>call:y{b}<tool_call>",
"<tool_call>call:x{<tool_call>call:y{b}<tool_call>",
"<tool_call>call:x{unclosed", "}<tool_call><tool_call>call:x{unclosed",
]
for text in cases:
actual = [(name, "{" + body + "}") for name, body in _iter_named_blocks(
text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE
)]
assert actual == _GEMMA.findall(text)
assert _strip_delimited(text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE) == _GEMMA.sub("", text)
raw = '<|tool_call|>call:web_search{"query":"news"}<|tool_call|>'
assert [(b.tool_type, b.content) for b in parse_tool_blocks(raw)] == [("web_search", "news")]
assert strip_tool_blocks(raw) == ""
@pytest.mark.parametrize("reference,opener,closer,tokens", [
(_QWEN_FUNCTION, _QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE,
["<function=manage_notes>", "</function>", "a", "\n", "\t", " "]),
(_QWEN_PARAMETER, _QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE,
["<parameter=query>", "</parameter>", "a", "\n", "\t", " "]),
])
def test_qwen_delimiters_preserve_names_and_stripped_values(reference, opener, closer, tokens):
rng = random.Random(6503)
for _ in range(1000):
text = "".join(rng.choices(tokens, k=16))
expected = [(name, body.strip()) for name, body in reference.findall(text)]
actual = [(name, body.strip()) for name, body in _iter_named_blocks(text, opener, closer)]
assert actual == expected
def test_qwen_python_arguments_preserve_permissive_legacy_grammar():
cases = ["action='list', limit=5", "action = \"list\"", "1key=2", "aé=4",
"key='mismatched\"", "broken word, okay=2", "a=\t, b=3", "a=\n'hi'", "a='hi\nthere'"]
rng = random.Random(6503)
tokens = ["key", "1", "中", "é", "=", "\n", " ", "\t", "'", '"', ",", "-", "_", "[]"]
cases += ["".join(rng.choices(tokens, k=20)) for _ in range(2000)]
for text in cases:
assert list(_iter_qwen_python_args(text)) == _QWEN_ARGS.findall(text), text
@pytest.mark.parametrize("text", [
"Done.", "Done.\nUpdated the document.\nDone.", "Done. undone.",
"Done.\nDONE.\t", "Done.\u2003Done.\u2003", "Done. no terminal marker ",
])
def test_trailing_done_matches_legacy_cleanup(text):
pattern = re.compile(r"\s*Done\.\s*$", re.I)
expected = pattern.sub("", text).rstrip() if pattern.search(text) else text
assert _strip_trailing_done(text) == expected
@pytest.mark.parametrize("allow_empty", [False, True])
@pytest.mark.parametrize("replacement", ["", " "])
def test_angle_tag_cleanup_matches_legacy_flat_grammar(allow_empty, replacement):
pattern = re.compile(r"<[^>]*>" if allow_empty else r"<[^>]+>")
cases = ["before<b>bold</b>after", "<>", "<<>>", "a<x\ny>b", "<broken",
"<a><b>one</b></a>", "a > b", "<x title='>'>tail"]
cases += ["".join(parts) for n in range(6)
for parts in itertools.product("a<>\n", repeat=n)]
for text in cases:
assert strip_angle_tags(text, replacement, allow_empty=allow_empty) == pattern.sub(replacement, text)
@pytest.mark.parametrize("ascii_only", [False, True])
def test_email_scanner_preserves_legacy_address_sets(ascii_only):
pattern = re.compile(
r"[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}"
if ascii_only else r"[\w.+-]+@[\w.-]+\.\w+"
)
cases = ["a@b.example", "a+b@b.example, c@d.example", "a@b@c.example",
"a@b.c+d@e.f", "你好@例子.中国", "a@b.-.com", "'a'@example.com",
"a@b.x", "a@.com", "a@b...", "a@b.c@d.example"]
rng = random.Random(6503)
tokens = ["a", "b", "é", "中", "@", ".", "-", "+", " ", "_", "'", "!", "/"]
cases += ["".join(rng.choices(tokens, k=30)) for _ in range(3000)]
for text in cases:
assert list(iter_email_addresses(text, ascii_only=ascii_only)) == pattern.findall(text), text
def test_prefixed_workspace_and_url_detectors_match_legacy_searches():
cases = [
"", "/workspace/a.py", "FILE:///WORKSPACE/report.HTML",
"/workspace/input/a.png", "/workspace/input/a.png/workspace/out.png",
"http://example.test/a.pdf", "http://x/http://example.test/pdf/view",
"https://google.com/search?q=x", "http://news.google.co.uk/sorry/index",
"https://x.bing.com/html", "http://bad/http://duckduckgo.com/?q=x",
]
rng = random.Random(6503)
tokens = ["/workspace/", "input/", "file://", "http://", "https://", ".py",
".pdf", ".html", "/pdf/", "google.", "bing.com/", "search", "x", " ", "'", "/"]
cases += ["".join(rng.choices(tokens, k=18)) for _ in range(4000)]
for text in cases:
assert _mentions_workspace_script(text) == bool(_WORKSPACE_SCRIPT.search(text)), text
assert _mentions_pdf_url(text) == bool(_PDF_URL.search(text)), text
assert _mentions_workspace_artifact(text) == bool(_WORKSPACE_ARTIFACT.search(text)), text
assert _mentions_workspace_output(text) == bool(_WORKSPACE_OUTPUT.search(text)), text
assert contains_search_engine_navigation(text) == bool(_SEARCH_ENGINE_URL.search(text)), text
def test_declared_workspace_artifacts_keeps_legacy_greedy_paths():
pattern = re.compile(r"/workspace/[^\s,,、;;`\"'<>]+\.[A-Za-z0-9]{1,12}", re.I)
cases = [
"create /workspace/report.csv",
"read /workspace/input.csv and create /workspace/out.json",
"create /workspace/a.csv/workspace/b.json",
"create /workspace/noext /workspace/out.txt",
]
for text in cases:
expected = []
for match in pattern.finditer(text):
path = match.group(0).rstrip(".!?))]}")
if path.startswith("/workspace/fixtures/") or path in expected:
continue
before = text[max(0, match.start() - 240):match.start()]
before = pattern.sub("[workspace file]", before)
clause = re.split(r"[.;!?\n]", before)[-1]
if re.search(
r"\b(?:from|using|inspect|read|open|analy[sz]e|transcribe|extract\s+(?:text\s+)?from|"
r"input(?:\s+file)?(?:\s+is)?|source(?:\s+file)?(?:\s+is)?)\s*(?::|=)?\s*$",
clause, re.I,
) or re.search(r"\b(?:read_file|inspect_media|extract_text|transcribe_media|pdf_extract)\b", clause, re.I):
continue
if not re.search(
r"\b(?:create|write|save|export|render|generate|produce|output|deliver|store|convert|make)\b|"
r"\b(?:write_file|output_path)\b", clause, re.I,
):
continue
expected.append(path)
assert declared_workspace_artifacts(text) == tuple(expected)
def test_intent_and_detail_detectors_preserve_short_legacy_language():
cases = [
"plain answer", "\n\nfunction = manage_notes", "mcp__email__read_email",
"< function >", "web_search: cats", "prefix\n private_browser : url",
"when does it end", "when\nwill it end", "first save event", "sequence please",
"I can now carefully inspect the result.", "Answer. Now let me verify this.",
"接下来我会查看结果", "。 让我先检查",
]
rng = random.Random(6503)
tokens = ["\n", " ", ".", "!", "function", "parameter", "=", "web_search", ":",
"when", "first", "end", "event", "let me", "inspect", "x"]
cases += ["".join(rng.choices(tokens, k=24)) for _ in range(3000)]
for text in cases:
assert _looks_like_ody_qwen_leaked_tool_text(text) == bool(_LEAKED_TOOL_TEXT.search(text)), text
assert contains_detailed_sequence_request(text, include_first=False) == bool(_VIDEO_DETAIL_BASE.search(text)), text
assert contains_detailed_sequence_request(text) == bool(_VIDEO_DETAIL_EXTENDED.search(text)), text
for text in (
"The result is partial. Now let me inspect the rest.",
"\n" * 20 + "let me check the source",
"。\n 接下来我会查看结果",
"A complete factual answer.",
):
assert isinstance(_looks_like_agent_reasoning_preamble(text), bool)
def test_suffix_and_split_helpers_preserve_legacy_results():
panel_cases = [
("open calendar again!!!", ("ui_control", "open_panel calendar")),
("open calendar" + " " * 50 + "again?", ("ui_control", "open_panel calendar")),
("open calendar....", ("ui_control", "open_panel calendar")),
("open calendar again x", ("ui_control", "open_panel calendar")),
("open notes day view again.", ("ui_control", "open_panel notes")),
]
for text, expected in panel_cases:
assert _parse_explicit_open_panel_request(text) == expected
values = ["one, two and three", " one ,two and three ", "candy and x", "a,and,b", ""]
for value in values:
expected_parts = [
re.sub(r"\s+", " ", part).strip(" .")
for part in re.split(r"\s*,\s*|\s+\band\b\s+", value)
]
expected = [{"text": part, "done": False} for part in expected_parts if part]
assert _split_note_items(value) == expected
for value in ["Alice <a@example.com>", "Alice <<a@example.com>", "Alice <>",
"Alice (a@example.com)", "Alice ((a@example.com)", "Alice > x <a>"]:
normalized = re.sub(r"\s+", " ", value).strip()
expected_sender = re.sub(r"\s*\([^)]*@[^)]*\)\s*$", "", normalized).strip()
expected_sender = re.sub(r"\s*<[^>]*>\s*$", "", expected_sender).strip()
expected_sender = expected_sender or value.strip()
expected_account = re.sub(r"\s*<[^>]+>\s*$", "", normalized).strip() or normalized
assert _email_sender_name(value) == expected_sender
assert _email_account_label(value) == expected_account
for value in ["Reply body\nWant me to send it?", "Reply\n\n SHOULD I continue?", "Want me now", "x\nnot a prompt"]:
expected = re.split(r"\n\s*(?:Want me|Would you like|Should I)\b", value, flags=re.I, maxsplit=1)[0]
assert _split_before_assistant_prompt(value) == expected
for value in ["a \n b\t\n", " x", "a \r\n", "\t\n\n", ""]:
assert _strip_horizontal_space_before_lf(value) == re.sub(r"[ \t]+\n", "\n", value)
for value in ["pwd && ls", "cat x | grep y", "echo x; printf y", "pwd " + " " * 20 + "&& ls", "pwd || rm x"]:
legacy_parts = re.split(r"\s*(?:&&|\|\||;|\|)\s*", value.strip())
allowed = re.compile(
r"^(?:pwd|ls|find|rg|grep|git\s+(?:status|diff|log|show|branch)|sed(?!\s+-i\b)|"
r"head|tail|cat|stat|file|wc|sort|uniq|cut|ip|ipconfig|getent|nslookup|dig|arp|"
r"hostname|uname|whoami|echo|printf|test|true|false|:)\b", re.I,
)
expected = bool(legacy_parts) and all(part.strip() for part in legacy_parts) and all(
allowed.match(part.strip()) for part in legacy_parts
)
assert _read_only_shell_command(value) == bool(expected)
@pytest.mark.parametrize("prefix,target_pattern", [
("#session-", r"[^)]+"),
("#note-", r"[^)]+"),
("#event-", r"[0-9a-fA-F-]{8,64}"),
("", r"[^)]+"),
])
def test_forward_markdown_and_angle_scanners_match_legacy(prefix, target_pattern):
reference = re.compile(r"\[([^\]]+)\]\(" + re.escape(prefix) + "(" + target_pattern + r")\)")
validator = None if target_pattern == r"[^)]+" else re.compile(target_pattern)
cases = ["[title](#session-id)", "[[title](#session-id)", "[](x)",
"[a](x)[b](y)", "[a](#event-12345678)", "[a](broken"]
rng = random.Random(6503)
tokens = ["[", "]", "(", ")", "#session-", "#note-", "#event-", "a", "1", "-", " "]
cases += ["".join(rng.choices(tokens, k=28)) for _ in range(3000)]
for text in cases:
expected = reference.findall(text)
actual = [(label, target) for _start, _end, label, target in iter_markdown_links(
text, target_prefix=prefix, target_re=validator
)]
assert actual == expected, text
generic = re.compile(r"\[([^\]]+)\]\([^)]+\)")
for text in cases:
assert replace_markdown_links_with_labels(text) == generic.sub(r"\1", text), text
angle = re.compile(r"<([^>]+)>")
for text in cases + ["<x>", "<<x>", "<>", "<a><b>", "<<<"]:
assert [content for _start, _end, content in iter_angle_contents(text)] == angle.findall(text)
def test_numbered_row_identifier_scan_matches_legacy_rows():
pattern = re.compile(r"^\s*\d+\.\s+.+?\s+\(([^)\n]+)\)\s+[—-]", re.M)
cases = [
"1. Task (abc) — due", " 2. Label ((nested) - tail", "1. no id",
"1. a (bad) x (good) — tail", "\n\n3. item (id-3) - tail",
]
rng = random.Random(6503)
tokens = ["1. ", "x", " ", "(", ")", " -", " —", "\n"]
cases += ["".join(rng.choices(tokens, k=20)) for _ in range(2000)]
for text in cases:
assert _numbered_row_parenthesized_ids(text) == pattern.findall(text), text
def test_model_and_request_extractors_match_legacy_grammars():
youtube = re.compile(
r"\b(?:youtube|youtu\.be|yt|video\s+comments?|comments?\s+on\s+(?:the\s+)?video|"
r"transcript\s+(?:of|for)|(?:latest|newest|recent)\s+(?:\d+\s+)?(?:videos?|uploads?)|"
r"official\s+.+\s+channel)\b", re.I,
)
destructive = re.compile(r"\b(delete|remove|archive|trash|send|reply|unsubscribe|mark\s+.*read)\b", re.I)
terse = re.compile(
r"\s*(?:(?:send|sned|share|give|show)?\s*(?:me\s+)?(?:the\s+)?(?:links?|urls?|sources?)"
r"(?:\s+(?:for|to|from)\s+(?:those|that|them|these|it|this|the\s+(?:sites?|websites?|resources?|sources?)))?"
r"|(?:for|to|from)\s+(?:those|that|them|these|it|this|the\s+(?:sites?|websites?|resources?|sources?)))"
r"\s*(?:please|pls)?[.!?]?\s*",
)
cases = ["official project channel", "official x channel", "mark all as read", "mark\nread",
" send me the links please ", "for those sites", "plain text"]
rng = random.Random(6503)
tokens = ["official", "channel", "mark", "read", "send", "links", "for", "those", "sites", " ", "\n", "x"]
cases += ["".join(rng.choices(tokens, k=24)) for _ in range(3000)]
for text in cases:
assert _looks_like_youtube_tool_turn(text) == bool(youtube.search(text)), text
assert _looks_like_destructive_request(text) == bool(destructive.search(text)), text
assert _is_terse_link_request(text) == bool(terse.fullmatch(text.lower())), text
summary_re = re.compile(
r"\bsummary\s*(?:\*\*)?\s*:?\s*(.+?)(?:\n\s*(?:-|\\*\\*|If you|Want me|This is|$))",
re.I | re.S,
)
summary_cases = ["Summary: useful\n- next", "summary **: x\n\nWant me to continue", "summary: no end"]
for text in summary_cases:
match = summary_re.search(text)
assert _contextual_summary_fragment(text) == (match.group(1) if match else "")
product_re = re.compile(
r"\b(?:find|look\s+for|shop\s+for|search\s+for)\s+(?:me\s+)?(?:the\s+)?(?:best\s+)?"
r"(?P<query>.+?)\s*[?.!]*$", re.I,
)
for text in ["find best camera???", "look for me the best shoes", "shop for ???", "plain"]:
normalized = re.sub(r"\s+", " ", text).strip()
match = product_re.search(normalized)
expected = ""
if match:
expected = match.group("query").strip(" \t\r\n.,!?;:")
expected = re.sub(
r"\s+(?:on|at|from)\s+(?:the\s+)?[A-Za-z0-9&.' -]{1,60}$", "", expected, flags=re.I
).strip()
expected = expected[:120] if 0 < len(expected.split()) <= 12 else ""
assert _private_browser_product_query(text) == expected
title_re = re.compile(r"<title(?:\s[^>]*)?>([\s\S]*?)</title>", re.I)
for text in ["<title>x</title>", "<TITLE class='x'>a<b</title>", "<title", "<titlex>x</titlex>"]:
match = title_re.search(text)
assert first_tag_content(text, "title", allow_attributes=True) == (match.group(1) if match else None)
def test_summary_capture_preserves_exact_head_whitespace_and_newline_grammar():
legacy = re.compile(
r"\bsummary\s*(?:\*\*)?\s*:?\s*(.+?)(?:\n\s*(?:-|\\*\\*|If you|Want me|This is|$))",
re.I | re.S,
)
cases = [
"Summary: useful\nordinary next line\n- next",
"Summary: useful\nx\n", "summary\n\n", "summary \n",
"summary: \n", "summary **:\n\n", "summary:\nX\n",
"summary summary: X\n", "summary: no newline",
]
rng = random.Random(6509)
tokens = ["summary", "Summary", ":", "**", "\\", " ", "\t", "\n", "x", "-", "If you"]
cases += ["".join(rng.choices(tokens, k=24)) for _ in range(5000)]
for text in cases:
match = legacy.search(text)
assert _contextual_summary_fragment(text) == (match.group(1) if match else ""), text
def test_repeated_listing_words_preserve_legacy_request_grammar():
legacy = re.compile(
r"\s*(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)\s+"
r"(?:[\w .:/-]+\s+)?(?:stories|articles|posts|headlines|pages)"
r"(?:\s+on\s+[\w .:/-]+)?[.!?]?\s*", re.I,
)
for text in ("top Straße stories", "top İ stories", "lİst stories", "top storİes",
"top café stories", "top stories on Straße", "top stories on İ"):
assert _page_listing_request(text) == bool(legacy.fullmatch(text)), text
for word in ("stories", "articles", "posts", "headlines", "pages"):
for count in (1, 2, 8, 32):
for suffix in ("X", "@", "\tX", "on x", "on \t", "on ", "\n"):
for separator in (" ", " on ", "\t", " \t"):
text = "top " + (word + separator) * count + suffix
assert _page_listing_request(text) == bool(legacy.fullmatch(text)), text
def test_repeated_navigation_schemes_preserve_legacy_url_search():
for count in (1, 2, 8, 32):
for suffix in ("X", "google.com/search", "bing.com/html", "duckduckgo.com/?q=x"):
text = "http://" * count + suffix
assert contains_search_engine_navigation(text) == bool(_SEARCH_ENGINE_URL.search(text)), text
def test_tool_listing_rows_match_legacy_grammars():
research = re.compile(r"^-\s+\[(.*?)\]\(#research-([^)]+)\)(.*)$")
skills = re.compile(r"^-\s+\*\*(.*?)\*\*(?:\s+\((.*?)\)|\s+\[(draft)\])?(?::\s*(.*))?$")
calendar = re.compile(r"^\s*-\s+(.+?):\s+\[(.*?)\]\(#event-([^)]+)\)(.*)$")
draft = re.compile(r"\bTo:\s*.+\bSubject:\s*.+\n---", re.I | re.S)
cases = [
"- [Title](#research-id) tail",
"- **name** (published): description",
"- **name** [draft]",
" - when: [title](#event-id) tail",
"To: a\nSubject: b\n---",
"To:Subject:x\n---",
"plain",
]
rng = random.Random(6504)
tokens = ["-", " ", "\t", "[", "]", "(", ")", "*", ":", "#research-", "#event-", "draft", "x"]
cases += ["".join(rng.choices(tokens, k=28)) for _ in range(5000)]
for text in cases:
match = research.match(text)
assert _research_listing_row(text) == (match.groups() if match else None), text
match = skills.match(text)
assert _skill_listing_row(text) == (match.groups() if match else None), text
match = calendar.match(text)
assert _calendar_listing_row(text) == (match.groups() if match else None), text
assert _contains_email_draft_headers(text) == bool(draft.search(text)), text
def test_terse_email_followup_matches_legacy_grammar():
legacy = re.compile(
r"^\s*(?:and|so|well|still|then|okay|ok|did you find it(?: yet)?|what did you find)\s*[?.!]*\s*$",
re.I,
)
rng = random.Random(6508)
tokens = ["and", "so", "well", "did you find it", " yet", "what did you find", " ", "\t", "?", ".", "!", "x"]
cases = ["and?", " did you find it yet ! ", "what did you find", "and X"]
cases += ["".join(rng.choices(tokens, k=20)) for _ in range(3000)]
for text in cases:
assert _is_terse_email_lookup_followup(text) == bool(legacy.match(text)), text
def test_preview_request_and_source_scans_match_legacy_grammars():
page = re.compile(
r"\s*(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)\s+"
r"(?:[\w .:/-]+\s+)?(?:stories|articles|posts|headlines|pages)"
r"(?:\s+on\s+[\w .:/-]+)?[.!?]?\s*", re.I,
)
prior = re.compile(
r"\s*(?:(?:where|what)\s+did\s+you\s+(?:get|find)\s+(?:that|this)\s+from[?., ]*"
r"(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link[.!? ]*"
r"|(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link(?:\s+for\s+that)?[.!? ]*"
r"|what(?:['’]?s|\s+is)\s+(?:the\s+)?source(?:\s+link)?[.!? ]*)\s*", re.I,
)
terminal = re.compile(
r"(?:^|[.!?;,\n])\s*(?:(?:pls|please)\s+)?(?:sources?|citations?|links?)"
r"\s*(?:pls|please)?\s*[.!?]*$", re.I,
)
rows = re.compile(r"^\[\d+\]\s+(.+?)\s*\n\s*(https?://\S+)", re.M)
cases = [
"top stories", "show me the latest stories on example.com", "give me the source link",
"what's the source?", "x. please links pls!!", "[1] Title\nhttps://example.test/x", "plain",
]
rng = random.Random(6505)
tokens = ["top", "show", " me", " the", " stories", " on", "link", "source", "please", " ", "\t", ".", "!", "\n", "[1]", "http://x"]
cases += ["".join(rng.choices(tokens, k=22)) for _ in range(5000)]
for text in cases:
assert _page_listing_request(text) == bool(page.fullmatch(text)), text
assert _prior_web_source_request(text) == bool(prior.fullmatch(text)), text
assert _terminal_source_link_clause(text) == bool(terminal.search(text)), text
assert _web_source_rows(text) == rows.findall(text), text
def test_staged_command_payload_parsers_match_legacy_grammars():
specs = [
(
re.compile(r"\bnote\s+titled\s+(.+?)\s+so\s+its\s+content\s+is\s+['\"]([^'\"]+)['\"]", re.I),
r"\bnote\s+titled(?=\s)",
r"\s+(.+?)\s+so\s+its\s+content\s+is\s+['\"]([^'\"]+)['\"]",
),
(
re.compile(r"\b(?:delete|trash|remove|archive|mark(?:\s+as)?\s+(?:read|unread)|mark\s+(?:read|unread))\b\s+(?:all|every|the)?\s*(?:my\s+)?(.+?)\s+(?:emails?|mail|messages?)\b", re.I),
r"\b(?:delete|trash|remove|archive|mark(?:\s+as)?\s+(?:read|unread)|mark\s+(?:read|unread))\b(?=\s)",
r"\s+(?:all|every|the)?\s*(?:my\s+)?(.+?)\s+(?:emails?|mail|messages?)\b",
),
(
re.compile(r"\b(?:make|create|add)\s+(?:a\s+)?checklist\s+(?:called|titled|named)\s+(.+?)\s+with\s+(.+?)\s*$", re.I),
r"\b(?:make|create|add)\s+(?:a\s+)?checklist\s+(?:called|titled|named)(?=\s)",
r"\s+(.+?)\s+with\s+(.+?)$",
),
(
re.compile(r"\b(?:change|update|set|retag)\b\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b", re.I),
r"\b(?:change|update|set|retag)\b(?=\s)",
r"\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b",
),
(
re.compile(r"\b(?:chang(?:e|es|ed|ing)|updat(?:e|es|ed|ing))\s+(.+?)\s+to\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)", re.I),
r"\b(?:chang(?:e|es|ed|ing)|updat(?:e|es|ed|ing))(?=\s)",
r"\s+(.+?)\s+to\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)",
),
(
re.compile(r"\breplace\s+(.+?)\s+with\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)", re.I),
r"\breplace(?=\s)",
r"\s+(.+?)\s+with\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)",
),
]
pipeline = re.compile(
r"\bpipeline\s+using\s+([^\s,]+)\s+to\s+(.+?),\s*then\s+([^\s,]+)\s+to\s+(.+?)(?:[.!?]\s*)?$",
re.I,
)
session = re.compile(r"\b(?:find|search(?:\s+for)?|show)\s+(?:the\s+)?(.+?)\s+(?:chat|session|conversation)\b", re.I)
checklist = re.compile(r"\b(?:what(?:'s| is)?|show|tell\s+me)\b.*?\b(?:left|remaining)\b.*?\b(?:on|in)\s+(?:the\s+)?(.+?)\s+checklist\b", re.I)
rng = random.Random(6506)
tokens = ["note", " titled", " so its content is ", "'x'", "delete", " all", " emails", "create checklist called", " with", "change", " tag to ", "replace", " to", " pipeline using ", " then", "find", " chat", "left", " on", " checklist", " ", "\t", ".", "x"]
cases = ["update note titled a so its content is 'b'", "delete all all emails", "create checklist called a with b", "change the trip tag to work", "replace old with new", "pipeline using a to x, then b to y", "find the chat", "what is left on the trip checklist"]
cases += ["".join(rng.choices(tokens, k=25)).strip() for _ in range(5000)]
for text in cases:
for legacy, prefix, remainder in specs:
match = legacy.search(text)
assert _captures_after_first_prefix(text, prefix, remainder) == (match.groups() if match else None), (legacy.pattern, text)
match = pipeline.search(text)
assert _pipeline_request_parts(text) == (match.groups() if match else None), text
match = session.search(text)
assert _session_find_query_capture(text) == (match.group(1) if match else None), text
match = checklist.search(text)
assert _remaining_checklist_name(text) == (match.group(1) if match else None), text
def test_turn_contract_scans_match_legacy_grammars():
cancelled = re.compile(
r"^(?:never\s*mind|scratch\s+that)\s*[,;:—–-]?\s*(?=(?:open|show|list|read|search|find|check|switch|go)\b)",
re.I,
)
prohibition = re.compile(
r"^\s*(?:read[- ]only(?:\s+and)?\s+)?(?:do\s+not|don['’]?t|never)\s+"
r"(?:add|create|make|write|draft|edit|change|update|delete|remove|send|reply|run|execute|download|serve|open|save|schedule|transcribe|inspect)\b"
r"[^.;\n]*[.!?]*\s*$", re.I,
)
budget = re.compile(r"\bunder\s+[¥$€£]?\s*\d+(?:[.,]\d+)?(?:\s*yen)?\b", re.I)
terminal_specs = [
r"[,.;?]\s*(?:(?:only|just)\s+)?(?:(?:list|show)\s+(?:me\s+)?)?a\s+few(?:\s+(?:task\s+)?(?:names?|items?|results?|entries?))?(?:\s+and\s+(?:whether|if)\s+[^.;\n]+)?",
r"[.;]\s*read[- ]only(?:\s+(?:please|pls|plz))?\s*,?\s*(?:(?:and\s+)?(?:do\s+not|don['’]?t|dont)\s+(?:change|edit|modify)(?:\s+or\s+send)?\s+(?:anything|data))?",
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?(?:do\s+not|don['’]?t|dont)\s+(?:touch|change|edit|modify)(?:\s+(?:anything|data|them))?(?:\s+yet)?",
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+changes?",
r"[,;]\s*(?:keep\s+(?:them|it)\s+)?short\s+lines?\s*,?",
]
rng = random.Random(6507)
tokens = ["never", " mind", "scratch", " that", "open", "do not", " touch", " anything", "no changes", "read-only", "a few", " and whether", "short lines", "under", "$", "123", "yen", "open my calendar", "and", " what is that", " ", "\t", ".", "!", ",", ";", "x"]
cases = ["never mind, open calendar", "do not edit anything.", "under $ 20 yen", "open my calendar", "open calendar and what is next", "x but "]
cases += ["".join(rng.choices(tokens, k=22)) for _ in range(5000)]
for text in cases:
assert _strip_cancelled_request_lead(text) == cancelled.sub("", text), text
assert _is_pure_action_prohibition(text) == bool(prohibition.fullmatch(text)), text
assert _mentions_under_budget(text) == bool(budget.search(text)), text
assert _strip_terminal_but(text) == re.sub(r"\s+but\s*$", "", text, flags=re.I), text
for core in terminal_specs:
legacy = re.search(core + r"[.!?]*\s*$", text, re.I)
current = _terminal_clause_match(text, core)
assert (current.start() if current else None) == (legacy.start() if legacy else None), (core, text)
match = _WARM_RECALL.fullmatch(text)
assert _warm_recall_parts(text) == ((match.group("target"), "") if match else None), text
match = _WARM_RECALL_WITH_FOLLOWUP.fullmatch(text)
assert _warm_recall_parts(text, with_followup=True) == (
(match.group("target"), match.group("followup")) if match else None
), text
@pytest.mark.parametrize("program", [
r'''
from src.agent_loop import _contextual_summary_fragment
assert _contextual_summary_fragment("Summary: x\n" + "x\n" * 100_000) == "x"
assert _contextual_summary_fragment("Summary:" + "\n" * 100_000) == "\n"
''',
r'''
from src.clean_agent_preview import _page_listing_request
for text in ("top " + "stories " * 30_000 + "X",
"top " + "stories on " * 30_000 + "@"):
assert not _page_listing_request(text)
''',
r'''
from src.text_scanning import contains_search_engine_navigation
assert not contains_search_engine_navigation("http://" * 100_000 + "X")
assert contains_search_engine_navigation("http://" * 100_000 + "bing.com/search")
''',
r'''
from src.agent_loop import _is_terse_email_lookup_followup
_is_terse_email_lookup_followup("and" + " " * 100_000 + "X")
''',
r'''
from src.turn_contract import (_is_pure_action_prohibition, _mentions_under_budget,
_strip_cancelled_request_lead, _terminal_clause_match, _warm_recall_parts)
space = " " * 100_000
_strip_cancelled_request_lead("never" + space + "mind" + space + "X")
_terminal_clause_match(", a few and whether " + space + "X;", r"[,.;?]\s*a\s+few(?:\s+and\s+whether\s+[^.;\n]+)?")
_is_pure_action_prohibition("do not add " + space + "X;")
_mentions_under_budget("under" + space + "$" + space + "X")
_warm_recall_parts("open" + space + "my" + space + "calendar" + space + "X")
_warm_recall_parts("open" + space + "my" + space + "calendar" + space + "and" + space + "what " + "x" * 181, with_followup=True)
''',
r'''
from src.agent_loop import (_captures_after_first_prefix, _pipeline_request_parts,
_remaining_checklist_name, _session_find_query_capture)
evil = ("delete all x " * 30_000) + "z"
_captures_after_first_prefix(evil, r"\bdelete\b(?=\s)", r"\s+(?:all)?\s*(.+?)\s+emails?\b")
_pipeline_request_parts(("pipeline using m to x " * 30_000) + "z")
_session_find_query_capture(("find x " * 30_000) + "z")
_remaining_checklist_name("what " + ("left on " * 30_000) + "z")
''',
r'''
from src.clean_agent_preview import (_page_listing_request, _prior_web_source_request,
_terminal_source_link_clause, _web_source_rows)
_page_listing_request("top stories on " + " " * 100_000 + "\nX")
_prior_web_source_request("give me link" + " " * 100_000 + "X")
_terminal_source_link_clause("links" + " " * 100_000 + "X")
_web_source_rows("[1] " + " " * 100_000)
''',
r'''
from src.agent_loop import (_calendar_listing_row, _contains_email_draft_headers,
_research_listing_row, _skill_listing_row)
for text in (
"- [" + "](#research-" * 30_000,
"- **" + " **" * 30_000,
"- when: [" + "](#event-" * 30_000,
"To: x " + "Subject: x " * 30_000,
):
_research_listing_row(text)
_skill_listing_row(text)
_calendar_listing_row(text)
_contains_email_draft_headers(text)
''',
r'''
from src.agent_loop import _session_link_from_row, _session_list_summary_from_tool_output, _strip_trailing_done
evil = "Done. " + "\t" * 100_000 + "x"
assert _strip_trailing_done(evil) == evil
for row in ("[" + "\\" * 100_000, "[" * 100_000,
"[a" + "\\](#session-" * 20_000 + ")"):
_session_link_from_row(row)
for row in ("[x](#session-id) (" + "msgs" * 100_000,
"[x](#session-id) (" + "(last active today)" * 20_000):
_session_list_summary_from_tool_output("Chats:\n- " + row)
''',
r'''
from src.tool_parsing import *
from src.tool_parsing import (_iter_named_blocks, _strip_delimited,
_GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE)
text = "}<|tool_call|>" + "<|tool_call|>call:web_search{" * 20_000
assert list(_iter_named_blocks(text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE)) == []
assert _strip_delimited(text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE) == text
# Public entry points also stay responsive, including fallback parsers.
text = "}<|tool_call|>" + "<|tool_call|>call:web_search{" * 3000
assert parse_tool_blocks(text) == []
strip_tool_blocks(text)
''',
r'''
from src.tool_parsing import (_iter_named_blocks, _iter_qwen_python_args,
_QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE,
_QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE, parse_tool_blocks)
for opener, closer, text in (
(_QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE, "<function=manage_notes>" * 20_000),
(_QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE, "<parameter=action>" * 20_000),
(_QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE, "<function=manage_notes>a" + "\t" * 100_000 + "x"),
):
assert list(_iter_named_blocks(text, opener, closer)) == []
assert list(_iter_qwen_python_args("a" * 100_000)) == []
assert parse_tool_blocks("<function=manage_notes>" * 3000) == []
assert parse_tool_blocks("manage_notes(" + "a" * 100_000 + ")")
''',
r'''
from src.tool_parsing import strip_angle_tags
for allow_empty in (False, True):
for replacement in ("", " "):
for text in ("<" * 200_000, ">" + "<" * 200_000):
assert strip_angle_tags(text, replacement, allow_empty=allow_empty) == text
assert strip_angle_tags("<" * 200_000 + ">tail", replacement,
allow_empty=allow_empty) == replacement + "tail"
''',
r'''
from src.tool_parsing import iter_email_addresses
for ascii_only in (False, True):
for text in ("+" * 100_000, "a@" + "a" * 100_000,
"a@" + "." * 100_000, "a@" * 20_000):
assert list(iter_email_addresses(text, ascii_only=ascii_only)) == []
''',
r'''
from src.agent_loop import _mentions_pdf_url, _mentions_workspace_script
from src.clean_agent_preview import declared_workspace_artifacts
from src.text_scanning import contains_search_engine_navigation
from src.turn_contract import _mentions_workspace_artifact, _mentions_workspace_output
workspace = "/workspace/" * 30_000 + "x"
assert not _mentions_workspace_script(workspace)
assert not _mentions_workspace_artifact(workspace)
assert not _mentions_workspace_output(workspace)
assert declared_workspace_artifacts("create " + workspace) == ()
assert not _mentions_pdf_url("http://" * 30_000 + "x")
assert not contains_search_engine_navigation("http://google." + "..google." * 30_000 + "x")
''',
r'''
from src.agent_loop import _looks_like_agent_reasoning_preamble, _looks_like_ody_qwen_leaked_tool_text
from src.text_scanning import contains_detailed_sequence_request
for text in ("\n" * 100_000 + "x", ("when " * 20_000) + "x"):
_looks_like_agent_reasoning_preamble(text)
_looks_like_ody_qwen_leaked_tool_text(text)
contains_detailed_sequence_request(text)
''',
r'''
from src.agent_loop import (_email_account_label, _email_sender_name,
_parse_explicit_open_panel_request, _read_only_shell_command,
_split_before_assistant_prompt, _split_note_items,
_strip_horizontal_space_before_lf)
space = " " * 100_000
_parse_explicit_open_panel_request("open calendar" + space + "x")
_split_note_items(space + "x")
_email_sender_name("<" * 100_000 + "x")
_email_account_label("<" * 100_000 + "x")
_split_before_assistant_prompt("\n" * 100_000 + "x")
_strip_horizontal_space_before_lf(" " * 100_000 + "x")
_read_only_shell_command(space + "x")
''',
r'''
from src.agent_loop import _numbered_row_parenthesized_ids
from src.text_scanning import iter_angle_contents, iter_markdown_links, replace_markdown_links_with_labels
for text in ("<" * 100_000, "[" * 100_000,
("[x](#session-" * 20_000) + "missing"):
list(iter_angle_contents(text))
list(iter_markdown_links(text, target_prefix="#session-"))
replace_markdown_links_with_labels(text)
_numbered_row_parenthesized_ids("1. x " + "(" * 100_000 + "x")
''',
r'''
from src.agent_loop import (_contextual_summary_fragment, _is_terse_link_request,
_looks_like_destructive_request, _looks_like_youtube_tool_turn,
_private_browser_product_query)
from src.text_scanning import first_tag_content
for text in (("official " * 30_000) + "x", ("mark " * 30_000) + "x",
" " * 100_000 + "x", ("summary: x " * 20_000) + "x",
"find best " + "?" * 100_000 + "x", "<title" * 30_000):
_looks_like_youtube_tool_turn(text)
_looks_like_destructive_request(text)
_is_terse_link_request(text)
_contextual_summary_fragment(text)
_private_browser_product_query(text)
first_tag_content(text, "title", allow_attributes=True)
''',
])
def test_hostile_parsers_complete_with_a_short_process_deadline(program):
# A regression cannot wedge pytest: the subprocess is killed at the limit.
# This includes import overhead; fixed scans themselves take milliseconds.
subprocess.run([sys.executable, "-c", textwrap.dedent(program)],
check=True, timeout=8, capture_output=True, text=True)
+224
View File
@@ -0,0 +1,224 @@
"""Exception details stay in server logs across both chat SSE transports."""
import json
import logging
from uuid import uuid4
import jsonschema
import pytest
from starlette.responses import StreamingResponse
from src import agent_runs
from src.tool_policy import ToolPolicy
from src.tool_schemas import FUNCTION_TOOL_SCHEMAS
from src.turn_contract import resolve_full_inventory_contract
from tests.runtime_evidence_helpers import authoritative_executor
SENSITIVE = (
"TAKEOVER_SECRET_827_828 /srv/private/credentials.json "
"https://internal.example/debug?token=PRIVATE_TOKEN "
'{"request_body":"PRIVATE_BODY"}'
)
async def _client_chunks(generator, detached):
session = "redaction-" + uuid4().hex
run = None
try:
if detached:
run = agent_runs.start(session, generator)
await run.task
generator = agent_runs.subscribe(session, run)
response = StreamingResponse(generator, media_type="text/event-stream")
return [chunk async for chunk in response.body_iterator]
finally:
if run is not None:
if run.evict_task:
run.evict_task.cancel()
agent_runs._RUNS.pop(session, None)
def _events(chunks):
return [
json.loads(chunk.split("data: ", 1)[1])
for chunk in chunks if "data: " in chunk and "[DONE]" not in chunk
]
@pytest.mark.asyncio
@pytest.mark.parametrize("detached", [False, True], ids=["direct-827", "detached-828"])
@pytest.mark.parametrize("boundary", ["calendar", "explicit"])
async def test_native_preemptive_exception_detail_stays_server_side(
monkeypatch, caplog, detached, boundary,
):
import src.agent_loop as module
monkeypatch.setattr(module, "get_setting", lambda key, default=None: default)
monkeypatch.setattr(module, "get_mcp_manager", lambda: None)
monkeypatch.setattr(module, "blocked_tools_for_owner", lambda owner: set())
monkeypatch.setattr(module, "estimate_tokens", lambda *args, **kwargs: 10)
monkeypatch.setattr(module, "_agent_route_tool_mode", lambda *args, **kwargs: (True, False, False))
monkeypatch.setattr(module, "_build_system_prompt", lambda messages, *args, **kwargs: (list(messages), []))
monkeypatch.setattr(module, "_required_safe_read_operation", lambda contract: None)
async def execute(*args, **kwargs):
raise RuntimeError(SENSITIVE)
async def stream(*args, **kwargs):
yield 'data: {"delta":"The requested tool failed."}\n\n'
yield "data: [DONE]\n\n"
monkeypatch.setattr(module, "execute_tool_block", authoritative_executor(execute))
monkeypatch.setattr(module, "stream_llm_with_fallback", stream)
tool = "manage_calendar" if boundary == "calendar" else "list_sessions"
if boundary == "calendar":
instruction = "What meetings do I have tomorrow?"
fallback = "preemptive_calendar_lookup"
else:
monkeypatch.setattr(module, "_parse_simple_calendar_tool_request", lambda *args: None)
instruction = "List all chats."
fallback = "preemptive_explicit_admin_session"
caplog.set_level(logging.WARNING)
chunks = await _client_chunks(module.stream_agent_loop(
"http://model.test/v1", "test-model",
[{"role": "user", "content": instruction}],
relevant_tools={tool}, owner="fixture", max_rounds=1, _is_teacher_run=True,
), detached)
tool_event = next((e for e in _events(chunks) if e.get("type") == "tool_output"), None)
assert tool_event is not None, _events(chunks)
assert tool_event["fallback"] == fallback
assert tool_event["exit_code"] == 1
assert tool_event["output"]
assert "TAKEOVER_SECRET" not in "".join(chunks)
assert "/srv/private" not in "".join(chunks)
assert "PRIVATE_TOKEN" not in "".join(chunks)
assert "PRIVATE_BODY" not in "".join(chunks)
assert tool_event["error_category"] == "tool_execution_error"
assert SENSITIVE in caplog.text
def _preview_provider(monkeypatch, payloads):
import src.clean_agent_preview as module
replies = iter(payloads)
class Response:
async def __aenter__(self): return self
async def __aexit__(self, *args): pass
def raise_for_status(self): pass
async def aiter_lines(self):
yield "data: " + json.dumps(next(replies))
yield "data: [DONE]"
class Client:
def __init__(self, **kwargs): pass
async def __aenter__(self): return self
async def __aexit__(self, *args): pass
def stream(self, *args, **kwargs): return Response()
monkeypatch.setattr(module.httpx, "AsyncClient", Client)
return module
def _preview_generator(module, schemas=()):
policy = ToolPolicy()
contract = resolve_full_inventory_contract(schemas=list(schemas), policy=policy)
return module.stream_preview(
endpoint_url="http://model.test", model="test", headers={},
messages=[{"role": "user", "content": "List my notes."}],
turn_contract=contract, session_id="fixture-redaction", owner="fixture",
disabled_tools=set(), tool_policy=policy,
)
@pytest.mark.asyncio
@pytest.mark.parametrize("detached", [False, True], ids=["direct-827", "detached-828"])
@pytest.mark.parametrize("shape", ["string", "message", "detail"])
async def test_preview_provider_detail_stays_server_side(monkeypatch, caplog, detached, shape):
error = SENSITIVE if shape == "string" else {shape: SENSITIVE}
module = _preview_provider(monkeypatch, [{"error": error}])
caplog.set_level(logging.WARNING)
chunks = await _client_chunks(_preview_generator(module), detached)
assert chunks[-1].startswith("event: error\n")
payload = _events(chunks)[-1]
assert payload["status"] == 502
assert "provider" in payload["error"]
assert SENSITIVE not in "".join(chunks)
assert "/srv/private" not in "".join(chunks)
assert "PRIVATE_TOKEN" not in "".join(chunks)
assert payload["error_category"] == "provider_stream_error"
assert SENSITIVE in caplog.text
@pytest.mark.asyncio
@pytest.mark.parametrize("detached", [False, True], ids=["direct-827", "detached-828"])
@pytest.mark.parametrize("failure", ["execution", "schema", "json", "arguments"])
async def test_preview_tool_exception_detail_stays_server_side(
monkeypatch, caplog, detached, failure,
):
module = _preview_provider(monkeypatch, [
{"choices": [{"delta": {"tool_calls": [{"index": 0, "id": "call-1", "function": {
"name": "manage_notes", "arguments": '{"action":"list"}',
}}]}}]},
{"choices": [{"delta": {"content": "The requested call failed."}}]},
] * 8)
async def execute(*args, **kwargs):
raise ValueError(SENSITIVE)
def fail(*args, **kwargs):
if failure == "schema":
raise jsonschema.ValidationError(SENSITIVE, instance={"private": SENSITIVE})
if failure == "json":
raise json.JSONDecodeError(SENSITIVE, SENSITIVE, 0)
raise ValueError(SENSITIVE)
if failure == "execution":
monkeypatch.setattr(module, "execute_tool_block", execute)
elif failure == "schema":
monkeypatch.setattr(module.jsonschema, "validate", fail)
else:
monkeypatch.setattr(module, "normalize_preview_call_args", fail)
schema = next(s for s in FUNCTION_TOOL_SCHEMAS if s["function"]["name"] == "manage_notes")
caplog.set_level(logging.WARNING)
chunks = await _client_chunks(_preview_generator(module, [schema]), detached)
tool_event = next(e for e in _events(chunks) if e.get("type") == "tool_output")
assert tool_event["error"] is True
assert tool_event["exit_code"] == 1
assert tool_event["output"]
assert "TAKEOVER_SECRET" not in "".join(chunks)
assert "/srv/private" not in "".join(chunks)
assert "PRIVATE_TOKEN" not in "".join(chunks)
assert "PRIVATE_BODY" not in "".join(chunks)
assert tool_event["error_category"] == (
"tool_execution_error" if failure == "execution" else "invalid_tool_arguments"
)
assert SENSITIVE in caplog.text
@pytest.mark.parametrize("message", [
"Tool is not offered or permitted.",
"Tool arguments must be a JSON object.",
"This operation is outside the preview safety policy. No change was made.",
"Resolve the named recipient with resolve_contact before drafting. Never invent an email address.",
"The calendar read has not succeeded yet. Obtain the requested calendar evidence before creating the dependent email draft.",
])
def test_curated_domain_errors_remain_useful(message):
from src.clean_agent_preview import _public_preview_tool_error
assert _public_preview_tool_error(ValueError(message)) == message
assert message not in _public_preview_tool_error(ValueError(message), execution_attempted=True)
@pytest.mark.parametrize("detail", [
SENSITIVE,
"Tool is not offered or permitted.\n" + SENSITIVE,
"Shell access to credential variable " + SENSITIVE,
"Artifact completion Python must reference the required " + SENSITIVE,
])
def test_untrusted_validation_detail_cannot_masquerade_as_curated_guidance(detail):
from src.clean_agent_preview import _public_preview_tool_error
public = _public_preview_tool_error(ValueError(detail))
assert public
assert "TAKEOVER_SECRET" not in public
assert "/srv/private" not in public
assert "PRIVATE_TOKEN" not in public