fix(security): eliminate parser denial-of-service paths

This commit is contained in:
Alexandre Teixeira
2026-10-06 03:16:53 +01:00
parent e0a45aeae7
commit eeff41a9ef
8 changed files with 2676 additions and 335 deletions
+710 -212
View File
File diff suppressed because it is too large Load Diff
+296 -47
View File
@@ -32,7 +32,12 @@ from src.tool_schemas import (
normalized_native_function_argument_error,
)
from src.tool_types import ToolBlock
from src.tool_parsing import parse_tool_blocks, strip_tool_blocks
from src.tool_parsing import iter_email_addresses, parse_tool_blocks, strip_tool_blocks, strip_angle_tags
from src.text_scanning import (
contains_detailed_sequence_request,
contains_search_engine_navigation,
iter_prefixed_token_matches,
)
from src.turn_contract import (
_REQUEST_PREFIX,
calendar_retiming_request,
@@ -54,6 +59,59 @@ MODE = COMPACT_PREVIEW_MODE
class ProviderStreamError(Exception):
"""A provider reported failure inside an otherwise successful SSE response."""
# Only these audited, fixed domain messages may cross the exception boundary.
# Return the canonical constants rather than arbitrary exception text.
_PUBLIC_PREVIEW_TOOL_ERRORS = {message: message for message in (
'An equivalent search already returned evidence. Change the angle, missing subtopic, source type, or corroboration target instead of only changing freshness wording.',
'Browser navigation already failed for this exact URL; use another source page.',
'Do not infer image content repeatedly from filenames; use inspect_media on representative files, then continue from visual evidence.',
'Equivalent search intent was repeated after a correction reminder; web_search is disabled for this turn.',
'Look up the target with list_sessions before changing a chat. Use its exact returned ID; never invent last-chat/latest aliases. If the target is ambiguous, ask using the candidate chat titles.',
'Resolve the named recipient with resolve_contact before drafting. Never invent an email address.',
'Selection-only edits cannot use replace_all. Use a unique contextual FIND inside the selected passage.',
'The bounded search-attempt budget is exhausted. Do not search again; answer from usable evidence already gathered, or clearly report what could not be verified and suggest a concrete next step.',
'The calendar read has not succeeded yet. Obtain the requested calendar evidence before creating the dependent email draft.',
'The latest note search returned no candidates, so this reference has no note to open. Do not reuse an older list item; report the empty result or ask which note was intended.',
'The latest research search returned no candidates, so this reference has no report to open. Do not reuse an unrelated older report.',
'The proposed expansion only adds placeholder/meta text. Write substantive content that continues the existing document’s subject, voice, and format; do not announce that a paragraph was added.',
'The recipient address is not supported by the contact lookup. Use an exact returned address for the requested person; if no match exists, explain the missing recipient instead of guessing.',
'The same artifact target was already rewritten three times; finish from the latest successful version instead of rewriting it again.',
'These suggestions collapse multiple different passages into the same much shorter replacement, violating the request to preserve meaning. Produce passage-specific revisions that retain each source passage’s claims and intent.',
'This exact call already failed twice and will not be executed again; change strategy or finish from existing evidence.',
'This exact failed call was repeated after a correction reminder; the tool is disabled for this turn. Finish from existing evidence.',
'This exact invalid call was repeated after two validation failures; the tool is disabled for this turn.',
'This exact successful call already returned evidence. Do not repeat it; change the arguments or tool to gather different evidence, or finish from the evidence already available.',
'This still image was already inspected and the same visual evidence is already in context. inspect_media is withheld for the next correction round; use that evidence, inspect a different file, or finish.',
'This operation is outside the preview safety policy. No change was made.',
'This replacement does not make its passage more concise. Shorten the wording while retaining its facts and meaning; a spelling-only change does not satisfy the requested action. Retry with shorter replacements.',
'Tool arguments could not be converted for execution.',
'Tool arguments must be a JSON object.',
'Tool execution budget exhausted; finish from existing evidence.',
'Tool is not offered or permitted.',
'Two equivalent searches already returned no evidence. Do not repeat this search wording; use a different offered tool or a materially different query.',
'Unresolved video target: this ID was not supplied by the user or observed in a successful tool result. Do not guess it from the title. Open/read the referenced browser link or call youtube_tool latest_channel_video for the observed channel, then use its returned ID.',
'inspect_media exports only extract video stills and require an explicit timestamp plus output_path for every item. First inspect the video to find the timestamp; use write_file or python to author a diagram or other new artifact.',
'web_fetch requires url or urls. query only filters a supplied page; it is not a search or writing request.',
)}
def _public_preview_tool_error(exc, *, execution_attempted=False):
"""Keep useful domain guidance while withholding arbitrary diagnostics."""
if execution_attempted:
return 'The tool failed unexpectedly. Check the server log and retry.'
if isinstance(exc, json.JSONDecodeError):
return 'Tool arguments are not valid JSON. Correct the JSON object and retry.'
if isinstance(exc, jsonschema.ValidationError):
return 'Tool arguments do not match the required schema. Correct the call using the offered tool schema.'
detail = str(exc)
if detail.startswith('Artifact completion Python must reference the required '):
return 'Artifact completion Python must create non-empty output at the required artifact target.'
if detail.startswith('Shell access to credential variable '):
return 'Shell access to credentials is blocked. Use brokered native tools.'
return _PUBLIC_PREVIEW_TOOL_ERRORS.get(detail, 'The tool call could not be validated. Check its arguments and retry.')
# Native unattended workspaces routinely require several inspections followed
# by several artifact writes. The interactive preview keeps its six-call
# limit below; this larger budget applies only after server-side validation of
@@ -79,13 +137,6 @@ ARTIFACT_RESEARCH_TOOLS = frozenset({
'web_search', 'web_fetch', 'private_browser', 'pdf_extract', 'youtube_tool',
'inspect_media', 'extract_text', 'transcribe_media',
})
DETAILED_VIDEO_REQUEST = re.compile(
r"\b(?:how\s+many|count|sequence|in\s+order|chronological|"
r"timestamps?|what\s+time|at\s+what\s+time|when\s+.*(?:end|happen)|"
r"first\s+.*(?:save|attempt|event)|score(?:board)?s?)\b|"
r"(?:多少|几次|何时|什么时候|时间|顺序)",
re.IGNORECASE,
)
READ_TOOLS = frozenset({
'manage_notes', 'manage_calendar', 'manage_memory', 'manage_skills', 'manage_tasks',
'manage_documents', 'manage_research', 'manage_contact', 'list_sessions',
@@ -809,13 +860,95 @@ def malformed_write_handoff_target(arguments, required_artifacts=(), user_text='
return target
_PAGE_LISTING_WORDS = ("stories", "articles", "posts", "headlines", "pages")
def _listing_allowed(char):
return char == " " or char in ".:/-" or char == "_" or char.isalnum()
def _allowed_then_space(value):
"""Whether value can be class+ followed by whitespace+, without retries."""
if len(value) < 2:
return False
first_disallowed = next((i for i, char in enumerate(value) if not _listing_allowed(char)), len(value))
if first_disallowed < len(value):
return first_disallowed > 0 and all(char.isspace() for char in value[first_disallowed:])
return value[-1].isspace()
def _space_then_allowed(value):
"""Whether value can be whitespace+ followed by the listing class+."""
if len(value) < 2:
return False
first_nonspace = next((i for i, char in enumerate(value) if not char.isspace()), len(value))
if first_nonspace < len(value):
return first_nonspace > 0 and all(_listing_allowed(char) for char in value[first_nonspace:])
trailing_spaces = len(value) - len(value.rstrip(" "))
return trailing_spaces > 0 and (len(value) > trailing_spaces or trailing_spaces >= 2)
def _page_listing_request(user_text):
value = str(user_text or '').lstrip()
nonspace_end = len(value.rstrip())
if value[nonspace_end - 1:nonspace_end] in ("!", "?"):
value = value[:nonspace_end - 1]
lead = re.match(
r"(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)(?=\s)",
value, re.I,
)
if not lead:
return False
cursor = lead.end()
while cursor < len(value) and value[cursor].isspace():
cursor += 1
body = value[cursor:]
nonspace_end = len(body.rstrip())
terminal_end = nonspace_end
if body[terminal_end - 1:terminal_end] == ".":
terminal_end = len(body[:terminal_end - 1].rstrip())
first_disallowed = len(body)
last_disallowed = -1
first_nonspace_after_disallowed = len(body)
for index, char in enumerate(body):
if not _listing_allowed(char):
first_disallowed = min(first_disallowed, index)
if index < nonspace_end:
last_disallowed = index
if index >= first_disallowed and not char.isspace():
first_nonspace_after_disallowed = min(first_nonspace_after_disallowed, index)
last_literal_space = body.rfind(" ")
for word in _PAGE_LISTING_WORDS:
for candidate in re.finditer(word, body, re.I):
start = candidate.start()
prefix_allowed = start >= 2 and (
body[start - 1].isspace() if start <= first_disallowed
else first_disallowed > 0 and start <= first_nonspace_after_disallowed
)
if start == 0 or prefix_allowed:
after_start = candidate.end()
if after_start >= terminal_end:
return True
on = after_start
while on < len(body) and body[on].isspace():
on += 1
if on > after_start and body[on:on + 2].casefold() == "on":
tail_start = on + 2
tail_end = tail_start
while tail_end < len(body) and body[tail_end].isspace():
tail_end += 1
if len(body) - tail_start >= 2 and tail_end > tail_start:
if tail_end < len(body):
if last_disallowed < tail_end:
return True
elif last_literal_space > tail_start:
return True
return False
def page_listing_response(entries, user_text, max_items=10):
"""Render simple page listings from observed titles/URLs, never synthesized rankings."""
if not re.fullmatch(
r'\s*(?:top|latest|recent|list(?: the)?|show(?: me)?(?: the)?)\s+'
r'(?:[\w .:/-]+\s+)?(?:stories|articles|posts|headlines|pages)'
r'(?:\s+on\s+[\w .:/-]+)?[.!?]?\s*', user_text, re.I,
) or re.search(r'\b(?:and|compare|summarize|analyse|analyze|about|by|since|yesterday)\b', user_text, re.I):
if not _page_listing_request(user_text) or re.search(r'\b(?:and|compare|summarize|analyse|analyze|about|by|since|yesterday)\b', user_text, re.I):
return ''
from urllib.parse import quote, urlsplit
from html import escape
@@ -1531,17 +1664,24 @@ def prior_workspace_path_answer(user_text, history):
return ''
def _prior_web_source_request(text):
text = str(text or '')
candidate = text.lstrip().rstrip()
candidate = candidate.rstrip(".!? ")
return bool(re.fullmatch(
r"(?:(?:where|what)\s+did\s+you\s+(?:get|find)\s+(?:that|this)\s+from[?., ]*"
r"(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link"
r"|(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link(?:\s+for\s+that)?"
r"|what(?:['’]?s|\s+is)\s+(?:the\s+)?source(?:\s+link)?)",
candidate,
re.I,
))
def prior_web_source_answer(user_text, history):
"""Return the latest source URL for an explicit source-only follow-up."""
text = str(user_text or '')
if not re.fullmatch(
r"\s*(?:(?:where|what)\s+did\s+you\s+(?:get|find)\s+(?:that|this)\s+from[?., ]*"
r"(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link[.!? ]*"
r"|(?:give|show|send)\s+me\s+(?:the\s+)?(?:source\s+)?link(?:\s+for\s+that)?[.!? ]*"
r"|what(?:['’]?s|\s+is)\s+(?:the\s+)?source(?:\s+link)?[.!? ]*)\s*",
text,
re.I,
):
if not _prior_web_source_request(text):
return ''
call_names = {}
candidates = []
@@ -2822,12 +2962,11 @@ def draft_contact_evidence_error(name, args, *, dependencies=(), executions=(),
and not e.get('blocked')]
if not observations:
return 'Resolve the named recipient with resolve_contact before drafting. Never invent an email address.'
address_pattern = r'[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}'
known = {address.casefold() for e in observations
for address in re.findall(address_pattern, str(e.get('output') or ''))}
known.update(address.casefold() for address in re.findall(address_pattern, user_text))
for address in iter_email_addresses(str(e.get('output') or ''), ascii_only=True)}
known.update(address.casefold() for address in iter_email_addresses(user_text, ascii_only=True))
proposed = {address.casefold() for field in ('to', 'cc', 'bcc')
for address in re.findall(address_pattern, str(args.get(field) or ''))}
for address in iter_email_addresses(str(args.get(field) or ''), ascii_only=True)}
if not proposed or not proposed <= known:
return ('The recipient address is not supported by the contact lookup. Use an exact '
'returned address for the requested person; if no match exists, explain '
@@ -3945,7 +4084,7 @@ def active_document_revision_quality_error(name, args, *, active_document, user_
if not incoming:
return None
addition = incoming[len(existing):].strip() if existing and incoming.startswith(existing) else ''
candidate = re.sub(r'<[^>]+>', ' ', addition).strip()
candidate = strip_angle_tags(addition, ' ').strip()
candidate = re.sub(r'\s+', ' ', candidate)
if not candidate or candidate.casefold() in str(user_text or '').casefold():
return None
@@ -3970,8 +4109,8 @@ def document_suggestion_quality_error(name, args, *, user_text):
for suggestion in suggestions:
if not isinstance(suggestion, dict):
continue
source = re.sub(r'\s+', ' ', re.sub(r'<[^>]*>', ' ', str(suggestion.get('find') or ''))).strip()
result = re.sub(r'\s+', ' ', re.sub(r'<[^>]*>', ' ', str(suggestion.get('replace') or ''))).strip()
source = re.sub(r'\s+', ' ', strip_angle_tags(str(suggestion.get('find') or ''), ' ', allow_empty=True)).strip()
result = re.sub(r'\s+', ' ', strip_angle_tags(str(suggestion.get('replace') or ''), ' ', allow_empty=True)).strip()
if not source or not result:
continue
source_words = len(re.findall(r"\b[\w’'-]+\b", source))
@@ -4127,13 +4266,17 @@ _WORKSPACE_FILE_RE = re.compile(
r"/workspace/[^\s,,、;;`\"'<>]+\.[A-Za-z0-9]{1,12}",
re.I,
)
_WORKSPACE_PREFIX_RE = re.compile(r"/workspace/", re.I)
_WORKSPACE_FILE_TOKEN_TAIL_RE = re.compile(r"[^\s,,、;;`\"'<>]*")
def declared_workspace_artifacts(user_text):
"""Return explicit output paths, excluding paths used only as inputs."""
text = str(user_text or '')
paths = []
for match in _WORKSPACE_FILE_RE.finditer(text):
for match in iter_prefixed_token_matches(
text, _WORKSPACE_PREFIX_RE, _WORKSPACE_FILE_RE, _WORKSPACE_FILE_TOKEN_TAIL_RE
):
path = match.group(0).rstrip('.!?))]}')
if path.startswith('/workspace/fixtures/') or path in paths:
continue
@@ -4253,17 +4396,61 @@ def evidence_tool_keeps_distinct_requests_available(name):
}
def _terminal_source_link_clause(text):
"""Recognize the terminal short source/link clause from right to left."""
value = str(text or '')
def matches_at(end):
while end and value[end - 1] in ".!?":
end -= 1
while end and value[end - 1].isspace():
end -= 1
lowered = value[:end].casefold()
for courtesy in ("please", "pls"):
if lowered.endswith(courtesy):
boundary = end - len(courtesy)
end = boundary
while end and value[end - 1].isspace():
end -= 1
lowered = value[:end].casefold()
break
token_match = re.search(r"(?:sources?|citations?|links?)$", lowered)
if token_match is None:
return False
prefix = value[:token_match.start()]
original_cursor = len(prefix)
courtesy_end = original_cursor
while courtesy_end and prefix[courtesy_end - 1].isspace():
courtesy_end -= 1
cursors = [original_cursor]
for courtesy in ("please", "pls"):
start = courtesy_end - len(courtesy)
if start >= 0 and prefix[start:courtesy_end].casefold() == courtesy:
if courtesy_end < original_cursor and (start == 0 or prefix[start - 1].isspace()):
cursors.append(start)
break
for cursor in cursors:
while cursor and prefix[cursor - 1].isspace():
if prefix[cursor - 1] == "\n":
return True
cursor -= 1
if cursor == 0 or prefix[cursor - 1] in ".!?;,":
return True
return False
return matches_at(len(value)) or (value.endswith("\n") and matches_at(len(value) - 1))
def requested_web_source_links(user_text):
return bool(re.search(
text = str(user_text or '')
return _terminal_source_link_clause(text) or bool(re.search(
r'\b(?:return|give|show|include|provide|cite|find)\b.{0,35}\b(?:source\s+)?links?\b'
r'|(?:^|[.!?;,\n])\s*(?:(?:pls|please)\s+)?(?:sources?|citations?|links?)\s*(?:pls|please)?\s*[.!?]*$'
r'|\b(?:\d+|one|two|three|four|five)\s+(?:official\s+)?(?:source\s+)?links?\b'
r'|\bofficial\s+source\b'
r'|\b(?:with|include|provide|cite|show|give|find)\s+(?:the\s+)?(?:official\s+)?(?:sources|citations)\b'
r'|\blink\s+(?:to\s+)?(?:the\s+|your\s+)?(?:original\s+|official\s+)?(?:instructions|sources|documentation|articles?|reports?|studies|manuals?|guides?)\b'
r'|\b(?:find|locate|get|download)\b.{0,60}\bofficial\b.{0,60}\b(?:manual|guide|handbook|pdf|documentation)\b'
r'|\b(?:find|locate|get|download)\b.{0,80}\b(?:manual|guide|handbook|pdf)\b.{0,40}\b(?:online|official)\b',
str(user_text or ''),
text,
re.IGNORECASE,
))
@@ -4300,12 +4487,67 @@ def unbound_lookup_reference(user_text, history, *, supplied_context=False):
))
def _web_source_rows(text):
"""Extract numbered source title/URL rows with monotonic line scans."""
rows = []
position = 0
length = len(text)
while position < length:
if position and text[position - 1] != "\n":
newline = text.find("\n", position)
if newline < 0:
break
position = newline + 1
continue
cursor = position
if cursor >= length or text[cursor] != "[":
newline = text.find("\n", cursor)
if newline < 0:
break
position = newline + 1
continue
cursor += 1
digit_start = cursor
while cursor < length and text[cursor].isdigit():
cursor += 1
if cursor == digit_start or cursor >= length or text[cursor] != "]":
position += 1
continue
cursor += 1
if cursor >= length or not text[cursor].isspace():
position += 1
continue
while cursor < length and text[cursor].isspace():
cursor += 1
title_start = cursor
title_line_end = text.find("\n", title_start)
if title_line_end < 0:
break
title_end = title_line_end
while title_end > title_start and text[title_end - 1].isspace():
title_end -= 1
url_start = title_line_end + 1
while url_start < length and text[url_start].isspace():
url_start += 1
scheme_length = 7 if text.startswith("http://", url_start) else 8 if text.startswith("https://", url_start) else 0
if title_end > title_start and scheme_length:
url_end = url_start + scheme_length
while url_end < length and not text[url_end].isspace():
url_end += 1
rows.append((text[title_start:title_end], text[url_start:url_end]))
position = url_end
continue
newline = text.find("\n", position)
if newline < 0:
break
position = newline + 1
return rows
def web_source_links(raw, *, max_items=1, prefer_official=False, query=''):
"""Extract stable title/URL pairs from the web tool's source preamble."""
text = str(raw or '')
rows = re.findall(
r'^\[\d+\]\s+(.+?)\s*\n\s*(https?://\S+)', text, re.MULTILINE,
)
rows = _web_source_rows(text)
query_tokens = set(re.findall(r'[a-z0-9]+', str(query or '').casefold())) - {
'the', 'a', 'an', 'official', 'source', 'link', 'page', 'website',
'site', 'guide', 'search', 'find', 'for', 'return',
@@ -5834,7 +6076,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
provider_error = payload['error']
if isinstance(provider_error, dict):
provider_error = provider_error.get('message') or provider_error.get('detail')
detail = str(provider_error or 'Unknown provider error').strip()[:300]
detail = str(provider_error or 'Unknown provider error').strip()
raise ProviderStreamError(detail)
usage = payload.get('usage') or {}
if usage:
@@ -6353,7 +6595,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
if (
native_workspace_enabled
and content
and DETAILED_VIDEO_REQUEST.search(direct_user_text)
and contains_detailed_sequence_request(direct_user_text)
and successful_video_inspections == 1
and not media_detail_nudge_sent
and round_number < round_limit
@@ -7071,7 +7313,14 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
):
artifact_body_handoff_attempts += 1
artifact_body_handoff_target = handoff_target
result = {'error': str(exc).splitlines()[0][:300], 'exit_code': 1}
logging.getLogger(__name__).warning(
'Clean v3 tool call failed: %s', exc, exc_info=True,
)
result = {
'error': _public_preview_tool_error(exc, execution_attempted=execution_attempted),
'error_category': 'tool_execution_error' if execution_attempted else 'invalid_tool_arguments',
'exit_code': 1,
}
if (canonical(block.tool_type if block is not None else name) == 'youtube_tool'
and result.get('exit_code') not in (None, 0)):
round_recovery_messages.append(
@@ -7260,12 +7509,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
requested_browser_url = private_browser_open_url(args)
effective_browser_url = private_browser_effective_url(result)
browser_search_url = requested_browser_url or effective_browser_url
is_search_engine_navigation = bool(re.search(
r'https?://(?:[^/]+\.)?(?:google\.[^/]+|bing\.com|duckduckgo\.com)'
r'/(?:search|sorry|html|lite|\?)',
browser_search_url,
re.I,
)) or bool(re.search(
is_search_engine_navigation = contains_search_engine_navigation(
browser_search_url
) or bool(re.search(
r'https?://(?:[^/]+\.)?google\.[^/]+/sorry/',
effective_browser_url,
re.I,
@@ -7433,6 +7679,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
'execution_attempted': execution_attempted,
'blocked': policy_denied or schema is None,
'desc': desc, 'round': round_number}
for error_category in ('tool_execution_error', 'invalid_tool_arguments'):
if result.get('error_category') == error_category:
tool_event['error_category'] = error_category
reader_event = email_reader_event(direct_user_text, actual_tool, args, result, failed=failed)
if reader_event:
yield event(reader_event)
@@ -8005,9 +8254,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
else:
yield event({'delta': '\nThe preview reached its round limit. Please narrow the request.'})
except ProviderStreamError as exc:
detail = f'The selected model provider failed while generating: {exc}'
logging.getLogger(__name__).warning('Clean v3 provider stream failed: %s', exc)
yield f'event: error\ndata: {json.dumps({"status": 502, "error": detail})}\n\n'
detail = 'The selected model provider failed while generating. Retry or choose another model.'
logging.getLogger(__name__).warning('Clean v3 provider stream failed: %s', exc, exc_info=True)
yield f'event: error\ndata: {json.dumps({"status": 502, "error": detail, "error_category": "provider_stream_error"})}\n\n'
return
except httpx.HTTPStatusError as exc:
status = exc.response.status_code
+210
View File
@@ -0,0 +1,210 @@
"""Forward-only helpers for permissive text grammars.
These helpers retain the existing regular expressions as anchored token
parsers while preventing ``re.search`` from retrying the same token suffix at
every embedded prefix.
"""
from __future__ import annotations
import re
from collections.abc import Iterator
from typing import Match, Pattern
def iter_prefixed_token_matches(
text: str,
candidate_re: Pattern[str],
anchored_re: Pattern[str],
token_tail_re: Pattern[str],
) -> Iterator[Match[str]]:
"""Yield legacy greedy matches after testing one prefix per token.
``anchored_re`` must start with the same fixed prefix recognized by
``candidate_re``. ``token_tail_re`` describes characters that the
anchored grammar can consume after that prefix. If the first prefix in
such a token cannot match, a later embedded prefix cannot match either:
its suffix was already available to the first attempt. Advancing to the
token boundary makes failed scans linear without changing successful
greedy captures.
"""
pos = 0
while candidate := candidate_re.search(text, pos):
match = anchored_re.match(text, candidate.start())
if match is not None:
yield match
pos = match.end()
continue
tail = token_tail_re.match(text, candidate.end())
pos = max(candidate.end(), tail.end() if tail is not None else candidate.end())
def has_prefixed_token_match(
text: str,
candidate_re: Pattern[str],
anchored_re: Pattern[str],
token_tail_re: Pattern[str],
) -> bool:
"""Return whether ``iter_prefixed_token_matches`` yields a match."""
return next(
iter_prefixed_token_matches(text, candidate_re, anchored_re, token_tail_re),
None,
) is not None
_HTTP_URL_PREFIX_RE = re.compile(r"https?://", re.IGNORECASE)
_VIDEO_DETAIL_BASE_RE = re.compile(
r"\b(?:how\s+many|count|break\s*points?|timestamps?|what\s+time|score(?:board)?s?)\b",
re.IGNORECASE,
)
_VIDEO_DETAIL_EXTENDED_RE = re.compile(
r"\b(?:sequence|in\s+order|chronological|at\s+what\s+time)\b|"
r"(?:多少|几次|何时|什么时候|时间|顺序)",
re.IGNORECASE,
)
_WHEN_RE = re.compile(r"\bwhen\s+", re.IGNORECASE)
_WHEN_TARGET_RE = re.compile(r"(?:end|happen)\b", re.IGNORECASE)
_FIRST_RE = re.compile(r"\bfirst\s+", re.IGNORECASE)
_FIRST_TARGET_RE = re.compile(r"(?:save|attempt|event)\b", re.IGNORECASE)
def contains_search_engine_navigation(text: str) -> bool:
"""Match the legacy Google/Bing/DuckDuckGo navigation URL grammar.
The old expression backtracked through every possible optional subdomain
split. Parsing the host up to its first slash gives the same accepted
hosts and path prefixes with one pass per URL candidate.
"""
pos = 0
while candidate := _HTTP_URL_PREFIX_RE.search(text, pos):
host_start = candidate.end()
path_start = text.find("/", host_start)
if path_start < 0:
return False
host = text[host_start:path_start].casefold()
path = text[path_start + 1:path_start + 7].casefold()
google_at = host.rfind(".google.")
recognized_host = (
(host.startswith("google.") and len(host) > len("google."))
or (google_at >= 0 and google_at + len(".google.") < len(host))
or host == "bing.com"
or host.endswith(".bing.com")
or host == "duckduckgo.com"
or host.endswith(".duckduckgo.com")
)
if recognized_host and path.startswith(("search", "sorry", "html", "lite", "?")):
return True
# A later URL may begin in the path. Resume after this scheme rather
# than skipping the whole non-whitespace region.
pos = candidate.end()
return False
def contains_detailed_sequence_request(text: str, *, include_first: bool = True) -> bool:
"""Recognize count/order/timing requests without overlapping ``.*`` scans."""
value = str(text or "")
if _VIDEO_DETAIL_BASE_RE.search(value):
return True
if include_first and _VIDEO_DETAIL_EXTENDED_RE.search(value):
return True
pos = 0
while when := _WHEN_RE.search(value, pos):
line_end = value.find("\n", when.end())
if line_end < 0:
line_end = len(value)
if _WHEN_TARGET_RE.search(value, when.end(), line_end) is not None:
return True
pos = line_end + 1
if include_first:
pos = 0
while first := _FIRST_RE.search(value, pos):
line_end = value.find("\n", first.end())
if line_end < 0:
line_end = len(value)
if _FIRST_TARGET_RE.search(value, first.end(), line_end) is not None:
return True
pos = line_end + 1
return False
def iter_angle_contents(text: str) -> Iterator[tuple[int, int, str]]:
"""Yield nonempty flat ``<...>`` contents with monotonic delimiters."""
pos = 0
while (start := text.find("<", pos)) >= 0:
end = text.find(">", start + 1)
if end < 0:
return
if end > start + 1:
yield start, end + 1, text[start + 1:end]
pos = end + 1
else:
pos = start + 1
def iter_markdown_links(
text: str,
*,
target_prefix: str = "",
target_re: Pattern[str] | None = None,
) -> Iterator[tuple[int, int, str, str]]:
"""Yield flat Markdown links accepted by the legacy link regexes."""
pos = 0
marker = "](" + target_prefix
while (start := text.find("[", pos)) >= 0:
label_end = text.find("]", start + 1)
if label_end < 0:
return
if label_end == start + 1:
pos = start + 1
continue
if not text.startswith(marker, label_end):
pos = label_end + 1
continue
target_start = label_end + len(marker)
target_end = text.find(")", target_start)
if target_end < 0:
return
target = text[target_start:target_end]
if target and (target_re is None or target_re.fullmatch(target)):
yield start, target_end + 1, text[start + 1:label_end], target
pos = target_end + 1
else:
pos = label_end + 1
def replace_markdown_links_with_labels(text: str) -> str:
"""Linear equivalent of replacing flat Markdown links with their labels."""
links = list(iter_markdown_links(text))
if not links:
return text
out = []
pos = 0
for start, end, label, _target in links:
out.extend((text[pos:start], label))
pos = end
out.append(text[pos:])
return "".join(out)
def first_tag_content(text: str, tag: str, *, allow_attributes: bool = False) -> str | None:
"""Return the first flat tag body using forward-only opener/closer scans."""
opener_re = re.compile(r"<" + re.escape(tag), re.IGNORECASE)
closer_re = re.compile(r"</" + re.escape(tag) + r">", re.IGNORECASE)
pos = 0
while opener := opener_re.search(text, pos):
name_end = opener.end()
if name_end < len(text) and text[name_end] == ">":
body_start = name_end + 1
elif allow_attributes and name_end < len(text) and text[name_end].isspace():
tag_end = text.find(">", name_end + 1)
if tag_end < 0:
return None
body_start = tag_end + 1
else:
pos = name_end
continue
closer = closer_re.search(text, body_start)
if closer is None:
return None
return text[body_start:closer.start()]
return None
+119 -18
View File
@@ -194,10 +194,33 @@ _TOOL_CODE_OPEN_RE = re.compile(r"<tool_code>\s*\{", re.IGNORECASE)
_TOOL_CODE_CLOSE_RE = re.compile(r"\}\s*</tool_code>", re.IGNORECASE)
# Pattern 4b: Gemma-style <|tool_call|> call:tool_name{args} <tool_call|>
_GEMMA_TOOL_CALL_RE = re.compile(
r"<\|?tool_call\|?>\s*call:([\w\d_-]+)\s*(\{[\s\S]*?\})\s*<\|?tool_call\|?>",
_GEMMA_TOOL_CALL_OPEN_RE = re.compile(
r"<\|?tool_call\|?>\s*call:([\w\d_-]+)\s*\{",
re.IGNORECASE,
)
_GEMMA_TOOL_CALL_CLOSE_RE = re.compile(
r"\}\s*<\|?tool_call\|?>",
re.IGNORECASE,
)
# Native Qwen markup shares the same non-nesting delimiter grammar as the
# XML helpers. Literal closers keep whitespace and opener floods linear;
# values are stripped by the caller, as in the original regex path.
_QWEN_FUNCTION_OPEN_RE = re.compile(r"<function=([A-Za-z_][\w:.-]*)>\s*")
_QWEN_FUNCTION_CLOSE_RE = re.compile(r"</function>")
_QWEN_PARAMETER_OPEN_RE = re.compile(r"<parameter=([A-Za-z_]\w*)>\s*")
_QWEN_PARAMETER_CLOSE_RE = re.compile(r"</parameter>")
_QWEN_PYTHON_ARG_KEY_RE = re.compile(r"[A-Za-z_]\w*")
_QWEN_PYTHON_ARG_VALUE_RE = re.compile(r"\s*=\s*(['\"].*?['\"]|[^,]+)")
_ANGLE_TAG_OPEN_RE = re.compile(r"<")
_NONEMPTY_ANGLE_TAG_OPEN_RE = re.compile(r"<(?=[^>])")
_ANGLE_TAG_CLOSE_RE = re.compile(r">")
_EMAIL_LOCAL_RE = re.compile(r"[\w.+-]+")
_EMAIL_ADDRESS_RE = re.compile(r"[\w.+-]+@[\w.-]+\.\w+")
_ASCII_EMAIL_LOCAL_RE = re.compile(r"[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+")
_ASCII_EMAIL_ADDRESS_RE = re.compile(
r"[A-Za-z0-9.!#$%&\x27*+/=?^_`{|}~-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}"
)
# Pattern 4c: Open-function wrapper emitted by some local MLX/Exo models.
# Example:
@@ -542,6 +565,34 @@ _PLAIN_UI_OPEN_PANEL_RE = re.compile(
r"((?:\s+(?:day|week|month|year|agenda)(?:\s+view)?(?:\s+\d{4}-\d{2}(?:-\d{2})?)?)?)"
r"\s*(?:`{1,3})?\s*$"
)
_PLAIN_UI_CANDIDATE_RE = re.compile(r"ui_control", re.IGNORECASE)
def _iter_plain_ui_open_panel(text: str):
"""Yield line-anchored UI commands without retrying every blank line."""
pos = 0
while candidate := _PLAIN_UI_CANDIDATE_RE.search(text, pos):
line_start = text.rfind("\n", 0, candidate.start()) + 1
match = _PLAIN_UI_OPEN_PANEL_RE.match(text, line_start)
if match is not None:
yield match
pos = match.end()
continue
newline = text.find("\n", candidate.end())
pos = len(text) if newline < 0 else newline + 1
def _strip_plain_ui_open_panel(text: str) -> str:
matches = list(_iter_plain_ui_open_panel(text))
if not matches:
return text
out = []
pos = 0
for match in matches:
out.append(text[pos:match.start()])
pos = match.end()
out.append(text[pos:])
return "".join(out)
# ---------------------------------------------------------------------------
@@ -992,6 +1043,43 @@ def _strip_raw_openai_tool_call_json(text: str) -> str:
return "".join(pieces)
def iter_email_addresses(text: str, *, ascii_only: bool = False):
"""Find legacy email-shaped strings without retrying every word suffix.
Match the address only at the start of each maximal local-part token.
When that attempt fails, every suffix has the same '@'/domain boundary
and must fail too. Local/domain character runs are each scanned a bounded
number of times, including malformed text with no '@' or domain dot.
"""
local_re = _ASCII_EMAIL_LOCAL_RE if ascii_only else _EMAIL_LOCAL_RE
address_re = _ASCII_EMAIL_ADDRESS_RE if ascii_only else _EMAIL_ADDRESS_RE
pos = 0
while token := local_re.search(text, pos):
address = address_re.match(text, token.start())
if address is not None:
yield address.group(0)
pos = address.end()
else:
pos = token.end()
def _iter_qwen_python_args(raw_args: str):
"""Match each maximal key once, then its value at that fixed position.
If a word has no following equals sign, none of its suffixes can have one
either. Advancing past it avoids the old unanchored regex's O(n^2) retries
on a long malformed key, while preserving permissive quoted/bare values.
"""
pos = 0
while key := _QWEN_PYTHON_ARG_KEY_RE.search(raw_args, pos):
value = _QWEN_PYTHON_ARG_VALUE_RE.match(raw_args, key.end())
if value is not None:
yield key.group(0), value.group(1)
pos = value.end()
else:
pos = key.end()
def _parse_qwen3_native_text_call(
text: str,
additional_tool_names: Optional[Iterable[str]] = None,
@@ -1018,13 +1106,14 @@ def _parse_qwen3_native_text_call(
# Qwen native text rendering:
# <tool_call><function=manage_notes><parameter=action>list</parameter>...
fn_match = re.search(r"<function=([A-Za-z_][\w:.-]*)>\s*([\s\S]*?)\s*</function>", text)
fn_match = next(_iter_named_blocks(
text, _QWEN_FUNCTION_OPEN_RE, _QWEN_FUNCTION_CLOSE_RE
), None)
if fn_match:
name, raw_body = fn_match.groups()
name, raw_body = fn_match
args = {}
for key, raw_value in re.findall(
r"<parameter=([A-Za-z_]\w*)>\s*([\s\S]*?)\s*</parameter>",
raw_body,
for key, raw_value in _iter_named_blocks(
raw_body.rstrip(), _QWEN_PARAMETER_OPEN_RE, _QWEN_PARAMETER_CLOSE_RE,
):
value = raw_value.strip()
if value and value[0] in "[{\"":
@@ -1085,7 +1174,7 @@ def _parse_qwen3_native_text_call(
or normalized_name in declared_names
):
args = {}
for key, raw_value in re.findall(r"([A-Za-z_]\w*)\s*=\s*(['\"].*?['\"]|[^,]+)", raw_args):
for key, raw_value in _iter_qwen_python_args(raw_args):
try:
args[key] = ast.literal_eval(raw_value.strip())
except (ValueError, SyntaxError):
@@ -1509,9 +1598,9 @@ def _iter_delimited(text, open_re, close_re):
pos = cm.end()
def _strip_delimited(text: str, open_re, close_re) -> str:
"""Remove every ``open_re ... close_re`` span (forward-only; see
_iter_delimited). Equivalent to ``open_re([\\s\\S]*?)close_re`` ``re.sub('')``
def _strip_delimited(text: str, open_re, close_re, replacement: str = "") -> str:
"""Replace every ``open_re ... close_re`` span (forward-only; see
_iter_delimited). Equivalent to ``open_re([\\s\\S]*?)close_re`` substitution
for these delimiters, without the O(n^2) rescan on unclosed openers."""
spans = list(_iter_delimited(text, open_re, close_re))
if not spans:
@@ -1520,11 +1609,23 @@ def _strip_delimited(text: str, open_re, close_re) -> str:
last = 0
for match_start, _inner_start, _inner_end, match_end in spans:
out.append(text[last:match_start])
out.append(replacement)
last = match_end
out.append(text[last:])
return "".join(out)
def strip_angle_tags(text: str, replacement: str = "", *, allow_empty: bool = False) -> str:
"""Linear equivalent of replacing ``<[^>]+>`` (or ``<[^>]*>``).
This preserves the existing flat text cleanup, including nested '<' and
malformed tails; it does not interpret HTML. Stop once no '>' is reachable
instead of retrying a suffix scan at every '<' in untrusted text.
"""
opener = _ANGLE_TAG_OPEN_RE if allow_empty else _NONEMPTY_ANGLE_TAG_OPEN_RE
return _strip_delimited(text, opener, _ANGLE_TAG_CLOSE_RE, replacement)
def _iter_named_blocks(text, open_re, close_re):
"""Forward-only equivalent of ``open_re([\\s\\S]*?)close_re`` finditer where
open_re captures a name in group 1: yield ``(name, body)``, pairing each
@@ -1958,10 +2059,10 @@ def parse_tool_blocks(
# Pattern 4b: Gemma-style <|tool_call|> blocks
if not blocks:
for m in _GEMMA_TOOL_CALL_RE.finditer(text):
tool_name = m.group(1)
body = m.group(2)
block = _parse_gemma_tool_call(tool_name, body)
for tool_name, body in _iter_named_blocks(
text, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE
):
block = _parse_gemma_tool_call(tool_name, "{" + body + "}")
if block:
blocks.append(block)
@@ -1998,7 +2099,7 @@ def parse_tool_blocks(
# from weaker native-tool models after reading the tool docs but failing to
# emit the actual structured call.
if not blocks:
m = _PLAIN_UI_OPEN_PANEL_RE.search(text)
m = next(_iter_plain_ui_open_panel(text), None)
if m:
blocks.append(ToolBlock("ui_control", f"open_panel {m.group(1).lower()}{m.group(2).lower()}".strip()))
@@ -2041,7 +2142,7 @@ def strip_tool_blocks(
cleaned = _strip_delimited(cleaned, _XML_TOOL_CALL_OPEN_RE, _XML_TOOL_CALL_CLOSE_RE)
cleaned = _XML_OPEN_TOOL_CALL_RE.sub('', cleaned)
cleaned = _strip_delimited(cleaned, _TOOL_CODE_OPEN_RE, _TOOL_CODE_CLOSE_RE)
cleaned = _GEMMA_TOOL_CALL_RE.sub('', cleaned)
cleaned = _strip_delimited(cleaned, _GEMMA_TOOL_CALL_OPEN_RE, _GEMMA_TOOL_CALL_CLOSE_RE)
cleaned = _strip_delimited(cleaned, _FUNCTION_MODEL_OPEN_RE, _FUNCTION_MODEL_CLOSE_RE)
cleaned = _strip_raw_openai_tool_call_json(cleaned)
declared_xml_calls = _parse_declared_direct_xml_calls(
@@ -2060,7 +2161,7 @@ def strip_tool_blocks(
if raw_web_json:
_, (start, end) = raw_web_json
cleaned = cleaned[:start] + cleaned[end:]
cleaned = _PLAIN_UI_OPEN_PANEL_RE.sub("", cleaned)
cleaned = _strip_plain_ui_open_panel(cleaned)
# Strip bare <invoke> blocks not wrapped in <tool_call>
cleaned = _strip_bare_invoke_markup(cleaned)
cleaned = re.sub(r'\n{3,}', '\n\n', cleaned)
+255 -57
View File
@@ -17,6 +17,7 @@ from typing import Iterable, Mapping
from src.action_intents import classify_tool_intent
from src.tool_policy import ToolPolicy
from src.text_scanning import has_prefixed_token_match
FAMILY_TOOLS = {
@@ -47,6 +48,62 @@ FAMILY_TOOLS = {
CONTRACT_CORE_TOOLS = frozenset({
"bash", "python", "read_file", "web_search", "web_fetch", "ask_user",
})
_WORKSPACE_PREFIX_RE = re.compile(r"/workspace/", re.I)
_WORKSPACE_ARTIFACT_RE = re.compile(
r"/workspace/[^\s`\"']+\.(?:csv|html?|json|md|svg|txt)\b", re.I
)
_WORKSPACE_OUTPUT_PREFIX_RE = re.compile(r"/workspace/(?!input/)", re.I)
_WORKSPACE_OUTPUT_RE = re.compile(
r"/workspace/(?!input/)[^\s`\"']+\."
r"(?:csv|html?|json|md|svg|txt|avif|bmp|gif|jpe?g|png|webp|pdf|mp4|webm)\b",
re.I,
)
_WORKSPACE_TOKEN_TAIL_RE = re.compile(r"[^\s`\"']*")
def _mentions_workspace_artifact(text: str) -> bool:
return has_prefixed_token_match(
str(text or ""),
_WORKSPACE_PREFIX_RE,
_WORKSPACE_ARTIFACT_RE,
_WORKSPACE_TOKEN_TAIL_RE,
)
def _mentions_workspace_output(text: str) -> bool:
return has_prefixed_token_match(
str(text or ""),
_WORKSPACE_OUTPUT_PREFIX_RE,
_WORKSPACE_OUTPUT_RE,
_WORKSPACE_TOKEN_TAIL_RE,
)
def _mentions_under_budget(text: str) -> bool:
value = str(text or "")
for under in re.finditer(r"\bunder", value, re.I):
cursor = under.end()
if cursor >= len(value) or not value[cursor].isspace():
continue
while cursor < len(value) and value[cursor].isspace():
cursor += 1
if cursor < len(value) and value[cursor] in "¥$€£":
cursor += 1
while cursor < len(value) and value[cursor].isspace():
cursor += 1
digit_start = cursor
while cursor < len(value) and value[cursor].isdecimal():
cursor += 1
if cursor == digit_start:
continue
if cursor == len(value) or not (value[cursor].isalnum() or value[cursor] == "_"):
return True
if value[cursor:cursor + 3].casefold() == "yen":
cursor += 3
if cursor == len(value) or not (value[cursor].isalnum() or value[cursor] == "_"):
return True
return False
_FAMILY_WORDS = {
"calendar": r"\b(?:calendar|calender|events?|appointments?|meetings?|agenda)\b",
"notes": r"\b(?:notes?|checklists?|groceries|remind\s+me)\b",
@@ -122,13 +179,7 @@ def _normalize_request_lead(value: str) -> str:
text,
flags=re.I,
)
text = re.sub(
r"^(?:never\s*mind|scratch\s+that)\s*[,;:—–-]?\s*"
r"(?=(?:open|show|list|read|search|find|check|switch|go)\b)",
"",
text,
flags=re.I,
)
text = _strip_cancelled_request_lead(text)
text = re.sub(r"^k(?:ay)?\s*[,!]?\s+(?=\S)", "", text, flags=re.I)
text = re.sub(
r"^(?:(?:great|nice|cool)\s*[,!.]|thanks?\s*[.!])\s+"
@@ -163,6 +214,21 @@ def _normalize_request_lead(value: str) -> str:
text = re.sub(r"^((?:can|could|would|will)\s+)u\b", r"\1you", text, flags=re.I)
text = re.sub(r"\boffical\b", "official", text, flags=re.I)
return text
def _strip_cancelled_request_lead(text: str) -> str:
lead = re.match(r"^(?:never\s*mind|scratch\s+that)", text, re.I)
if lead is None:
return text
cursor = lead.end()
while cursor < len(text) and text[cursor].isspace():
cursor += 1
if cursor < len(text) and text[cursor] in ",;:—–-":
cursor += 1
while cursor < len(text) and text[cursor].isspace():
cursor += 1
action = re.match(r"(?:open|show|list|read|search|find|check|switch|go)\b", text[cursor:], re.I)
return text[cursor:] if action is not None else text
_MISSPELLED_RESEARCH_ACTION = re.compile(
r"^\s*" + _REQUEST_PREFIX + r"(?:reserch|reasearch|reseach)\b",
re.I,
@@ -195,6 +261,21 @@ _PURE_ACTION_PROHIBITION = re.compile(
r"[^.;\n]*[.!?]*\s*$",
re.I,
)
def _is_pure_action_prohibition(text: str) -> bool:
value = str(text or "").strip()
prefix = re.match(
r"(?:read[- ]only(?:\s+and)?\s+)?(?:do\s+not|don['’]?t|never)\s+"
r"(?:add|create|make|write|draft|edit|change|update|delete|remove|send|reply|"
r"run|execute|download|serve|open|save|schedule|transcribe|inspect)\b",
value,
re.I,
)
if prefix is None:
return False
tail = value[prefix.end():].rstrip(".!?")
return not any(char in ".;\n" for char in tail)
_RETURN_TO_ACTION = re.compile(r"^\s*" + _REQUEST_PREFIX + r"return\s+to\b", re.I)
_PANEL_NAVIGATION = re.compile(
r"^\s*" + _REQUEST_PREFIX
@@ -447,6 +528,107 @@ _WARM_RECALL_WITH_FOLLOWUP = re.compile(
r"(?P<followup>(?:what(?:['’]?s|\s+is)?|which|who|where|when|how|show|open|read|list|find|search)\b[\s\S]{0,180})$",
re.I,
)
_WARM_TARGET_RE = re.compile(
r"calendar|emails?|inbox|notes?|tasks?|skills?|memories|memory|"
r"documents?|docs?|web|browser|cookbook|files?|shell",
re.I,
)
_WARM_FOLLOWUP_RE = re.compile(
r"(?:what(?:['’]?s|\s+is)?|which|who|where|when|how|show|open|read|list|find|search)"
r"\b[\s\S]{0,180}\Z",
re.I,
)
def _consume_space(value: str, cursor: int, *, required: bool = False) -> int | None:
start = cursor
while cursor < len(value) and value[cursor].isspace():
cursor += 1
return None if required and cursor == start else cursor
def _phrase_end(value: str, cursor: int, phrase: str) -> int | None:
for index, word in enumerate(phrase.split(" ")):
if value[cursor:cursor + len(word)].casefold() != word:
return None
cursor += len(word)
if index + 1 < len(phrase.split(" ")):
cursor = _consume_space(value, cursor, required=True)
if cursor is None:
return None
return cursor
def _warm_recall_parts(value: str, *, with_followup: bool = False) -> tuple[str, str] | None:
value = str(value or "")
cursor = _consume_space(value, 0) or 0
states = [cursor]
discourse = re.match(r"(?:ok(?:ay)?|and|then)", value[cursor:], re.I)
if discourse is not None:
after = _consume_space(value, cursor + discourse.end(), required=True)
if after is not None:
states.insert(0, after)
action_phrases = ("back to", "return to", "what about", "check", "show", "open")
action_states: list[int] = []
for state in states:
if not with_followup:
action_states.append(state)
for phrase in action_phrases:
end = _phrase_end(value, state, phrase)
if end is None:
continue
if with_followup:
end = _consume_space(value, end, required=True)
if end is None:
continue
action_states.append(end)
for state in dict.fromkeys(action_states):
state = _consume_space(value, state) or 0
possessive_states = [state]
for possessive in ("my", "the"):
if value[state:state + len(possessive)].casefold() == possessive:
possessive_states.insert(0, state + len(possessive))
for target_state in possessive_states:
target_state = _consume_space(value, target_state) or 0
target = _WARM_TARGET_RE.match(value, target_state)
if target is None:
continue
target_text = target.group(0)
suffix = target.end()
if with_followup:
if suffix < len(value) and (value[suffix].isalnum() or value[suffix] == "_"):
continue
separator_states = []
spaced = _consume_space(value, suffix) or 0
if spaced > suffix:
separator_states.append(spaced)
if spaced < len(value) and value[spaced] in "-—,:;":
separator_states.append(_consume_space(value, spaced + 1) or 0)
for conjunction in ("and", "then"):
end = spaced + len(conjunction)
if (value[spaced:end].casefold() == conjunction
and (end == len(value) or not (value[end].isalnum() or value[end] == "_"))):
separator_states.append(_consume_space(value, end) or 0)
for followup_start in dict.fromkeys(separator_states):
followup = _WARM_FOLLOWUP_RE.match(value, followup_start)
if followup is not None:
return target_text, followup.group(0)
continue
suffix = _consume_space(value, suffix) or 0
suffix_states = [suffix]
for word in ("again", "now"):
if value[suffix:suffix + len(word)].casefold() == word:
suffix_states.insert(0, suffix + len(word))
for end in suffix_states:
while end < len(value) and value[end] in ".!?":
end += 1
end = _consume_space(value, end) or 0
if end == len(value):
return target_text, ""
return None
_REQUIRED_TOOLS = {
"calendar": "manage_calendar", "notes": "manage_notes",
"tasks": "manage_tasks", "skills": "manage_skills",
@@ -851,11 +1033,7 @@ def selected_tools_for_request(message: str) -> frozenset[str] | None:
tools = {"inspect_media"}
if (
re.search(r"\b(?:create|write|save|build|produce)\b", raw_text, re.I)
and re.search(
r"(?:file://)?/workspace/[^\s`\"']+\.(?:csv|html?|json|md|svg|txt)\b",
raw_text,
re.I,
)
and _mentions_workspace_artifact(raw_text)
):
tools.update({"write_file", "read_file"})
if re.search(r"\b(?:preview|render|open)\b[^.\n]{0,100}\b(?:page|html|browser)\b", raw_text, re.I):
@@ -2118,6 +2296,31 @@ _READ_PRESENTATION_SUFFIX = re.compile(
)
def _terminal_clause_match(text: str, core_pattern: str) -> re.Match[str] | None:
"""Match a terminal clause after removing its ambiguous punctuation tail."""
end = len(text)
while end and text[end - 1].isspace():
end -= 1
while end and text[end - 1] in ".!?":
end -= 1
possessive = core_pattern.replace(r"\s+", r"\s++").replace(r"\s*", r"\s*+")
return re.search(possessive + r"$", text[:end], re.I)
def _strip_terminal_but(text: str) -> str:
end = len(text)
while end and text[end - 1].isspace():
end -= 1
if end < 3 or text[end - 3:end].casefold() != "but":
return text
start = end - 3
if start == 0 or not text[start - 1].isspace():
return text
while start and text[start - 1].isspace():
start -= 1
return text[:start]
def _read_request_and_limit(message: str) -> tuple[str, int | None]:
"""Strip only whole, known presentation/safety suffixes, never actions."""
text = _normalize_request_lead(message)
@@ -2167,11 +2370,11 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
if keep_few_suffix:
maximum = 3
text = text[:keep_few_suffix.start()].strip()
few_suffix = re.search(
few_suffix = _terminal_clause_match(
text,
r"[,.;?]\s*(?:(?:only|just)\s+)?(?:(?:list|show)\s+(?:me\s+)?)?a\s+few"
r"(?:\s+(?:task\s+)?(?:names?|items?|results?|entries?))?"
r"(?:\s+and\s+(?:whether|if)\s+[^.;\n]+)?[.!?]*\s*$",
text, re.I,
r"(?:\s+and\s+(?:whether|if)\s+[^.;\n]+)?",
)
if few_suffix:
maximum = 3
@@ -2183,42 +2386,43 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
text,
flags=re.I,
).strip()
text = re.sub(
terminal_read_only = _terminal_clause_match(
text,
r"[.;]\s*read[- ]only(?:\s+(?:please|pls|plz))?\s*,?\s*"
r"(?:(?:and\s+)?(?:do\s+not|don['’]?t|dont)\s+"
r"(?:change|edit|modify)(?:\s+or\s+send)?\s+(?:anything|data))?"
r"[.!?]*\s*$",
"",
text,
flags=re.I,
).strip()
r"(?:change|edit|modify)(?:\s+or\s+send)?\s+(?:anything|data))?",
)
if terminal_read_only:
text = text[:terminal_read_only.start()].strip()
# Explanatory/safety tails do not alter a preceding exact read request.
text = re.sub(
safety_tail = _terminal_clause_match(
text,
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?(?:do\s+not|don['’]?t|dont)\s+"
r"(?:touch|change|edit|modify)(?:\s+(?:anything|data|them))?(?:\s+yet)?[.!?]*\s*$",
"", text, flags=re.I,
).strip()
r"(?:touch|change|edit|modify)(?:\s+(?:anything|data|them))?(?:\s+yet)?",
)
if safety_tail:
text = text[:safety_tail.start()].strip()
text = re.sub(
r"[.!?]\s*read[- ]only(?:\s+(?:please|pls|plz))?\s*,?\s*"
r"(?:do\s+not|don['’]?t|dont)\s+(?:change|edit|modify)\s+"
r"(?:or\s+send\s+)?anything[.!?]*\s*$",
"", text, flags=re.I,
).strip()
text = re.sub(
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+changes?[.!?]*\s*$",
"", text, flags=re.I,
).strip()
text = re.sub(
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+edits?[.!?]*\s*$",
"", text, flags=re.I,
).strip()
for terminal_core in (
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+changes?",
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+edits?",
):
terminal_match = _terminal_clause_match(text, terminal_core)
if terminal_match:
text = text[:terminal_match.start()].strip()
text = re.sub(
r"[,;]\s*no\s+edits?[.!?]*\s*$", "", text, flags=re.I,
).strip()
text = re.sub(
r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+writes?[.!?]*\s*$",
"", text, flags=re.I,
).strip()
terminal_no_write = _terminal_clause_match(
text, r"(?:(?:[,;]\s*(?:and\s+)?|\s+and\s+))?no\s+writes?"
)
if terminal_no_write:
text = text[:terminal_no_write.start()].strip()
text = re.sub(
r"[.!?]\s*(?:i['’]?m|i\s+am)\s+(?:just\s+)?checking\b[^\n]*$",
"", text, flags=re.I,
@@ -2236,12 +2440,11 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
text,
flags=re.I,
).strip()
text = re.sub(
r"[,;]\s*(?:keep\s+(?:them|it)\s+)?short\s+lines?\s*,?[.!?]*\s*$",
"",
text,
flags=re.I,
).strip()
short_lines = _terminal_clause_match(
text, r"[,;]\s*(?:keep\s+(?:them|it)\s+)?short\s+lines?\s*,?"
)
if short_lines:
text = text[:short_lines.start()].strip()
text = re.sub(
r"[,.;]\s*keep\s+(?:the\s+answer|it|them)\s+short[.!?]*\s*$",
"",
@@ -2318,7 +2521,7 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
value = int(raw) if raw.isdecimal() else _READ_COUNT_WORDS[raw.lower()]
maximum = value if maximum is None else min(maximum, value)
text = text[:conversational_limit.start()].rstrip(' ,.;?')
text = re.sub(r"\s+but\s*$", "", text, flags=re.I)
text = _strip_terminal_but(text)
need_limit = re.search(
r"[.!?]\s*(?:i\s+)?only\s+need\s+(" + _READ_COUNT + r")"
r"(?:\s+(?:short\s+)?(?:titles?|items?|results?|entries?|names?|ones?))?"
@@ -3949,7 +4152,7 @@ def _families_for_tool(tool: str) -> frozenset[str]:
def _clause_capabilities(text: str) -> set[str]:
# A prohibition constrains authority; it must never grant the family named
# only as the forbidden side effect (for example, "do not create a file").
if _PURE_ACTION_PROHIBITION.fullmatch(text):
if _is_pure_action_prohibition(text):
return set()
container_tool = creation_container_tool(text)
if container_tool:
@@ -4610,12 +4813,7 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum
raw_text,
re.I,
)
and re.search(
r"(?:file://)?/workspace/(?!input/)[^\s`\"']+\."
r"(?:csv|html?|json|md|svg|txt|avif|bmp|gif|jpe?g|png|webp|pdf|mp4|webm)\b",
raw_text,
re.I,
)
and _mentions_workspace_output(raw_text)
):
families.add("shell_files")
if re.search(
@@ -5074,7 +5272,7 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum
and re.search(r"\b(?:ones?|top|apps?|services?|providers?)\b", text, re.I)
and re.search(r"\b(?:price|cheap|under|dimensions?|quote|trustworthy|app)\b", text, re.I)
)
or re.search(r"\bunder\s+[¥$€£]?\s*\d+(?:[.,]\d+)?(?:\s*yen)?\b", text, re.I)
or _mentions_under_budget(text)
):
return frozenset({"search_browser"})
if recent_family == ("search_browser",) and re.fullmatch(
@@ -6011,8 +6209,8 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum
elif (recent == ("search_browser",) and families == {"cookbook_admin"}
and re.match(r"^\s*" + _REQUEST_PREFIX + r"(?:search|find)\s+(?:those|them|these)\b", text, re.I)):
families = {"search_browser"}
if not families and (recall := _WARM_RECALL.fullmatch(text)):
target = recall["target"].lower()
if not families and (recall := _warm_recall_parts(text)):
target = recall[0].lower()
family = {
"email": "email", "emails": "email", "inbox": "email",
"note": "notes", "notes": "notes", "task": "tasks", "tasks": "tasks",
@@ -6024,8 +6222,8 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum
}[target]
if family in recently_executed_families(history):
families.add(family)
if not families and (recall := _WARM_RECALL_WITH_FOLLOWUP.fullmatch(text)):
target = recall["target"].lower()
if not families and (recall := _warm_recall_parts(text, with_followup=True)):
target = recall[0].lower()
family = {
"email": "email", "emails": "email", "inbox": "email",
"note": "notes", "notes": "notes", "task": "tasks", "tasks": "tasks",