fix(security): eliminate induced regex denial-of-service paths

This commit is contained in:
Alexandre Teixeira
2026-10-06 13:36:01 +01:00
parent 2e4ad7c383
commit ee48c9c51e
6 changed files with 476 additions and 47 deletions
+147 -28
View File
@@ -86,6 +86,8 @@ from src.tool_parsing import iter_email_addresses, strip_angle_tags
from src.text_scanning import (
contains_detailed_sequence_request,
has_prefixed_token_match,
space_delimited_fields,
terminal_dot_field,
first_tag_content,
iter_angle_contents,
iter_markdown_links,
@@ -2155,7 +2157,7 @@ def _looks_like_agent_reasoning_preamble(text: str) -> bool:
r"(?:let me|i(?:'ll| will)(?:\s+need\s+to)?|i\s+can(?:\s+now)?|i(?:'m| am)\s+(?:preparing|planning)\s+to)\s+"
r"(?:(?:carefully|methodically|systematically|closely|further)\s+){0,2}"
r"(?:continue|continuing|analy[sz]e|scan|check|verify|inspect|confirm|look up|fetch|open|list|search|refine|request|review|track|read|watch|(?:re-?)?examine|provide|give|state|report|answer|respond|summarize|conclude)\b"
r"[^.!?]*[.!?]?\s*$",
r"[^.!?]*+[.!?]?\s*$",
trailing_segment,
):
return True
@@ -2475,6 +2477,91 @@ def _parse_qwen_explicit_note_delete(text: str) -> Optional[str]:
return match.group(1).strip().strip("\"'`").rstrip(".") if match else None
def _space_field_leaders(patterns, flags=re.IGNORECASE):
return [(re.compile(pattern, flags), minimum) for pattern, minimum in patterns]
def _payload_fields(value, grammar, flags=re.IGNORECASE):
"""Scan the finite command payload grammars without whitespace partitions."""
simple = [(r"\s++", 1)]
the = [(r"\s++the\s++", 1), *simple]
if grammar == "note_update":
leaders = simple
separator = r"(?<!\s)(\s++)so\s++its\s++content\s++is(?=\s)"
quoted = re.compile(r"\s++['\"]([^'\"]+)['\"]", flags)
def tail(match):
result = quoted.match(value, match.end())
return result.groups() if result else None
elif grammar == "email_mutation":
leaders = [
(r"\s++(?:all|every|the)\s*+my\s++", 1),
(r"\s++(?:all|every|the)\s*+", 0),
(r"\s++my\s++", 1), *simple,
]
separator = r"(?<!\s)(\s++)(?:emails?|mail|messages?)\b"
tail = lambda match: ()
elif grammar in ("checklist", "replace", "change"):
leaders = simple
word = "to" if grammar == "change" else "with"
separator = rf"(?<!\s)(\s++){word}"
if grammar == "checklist":
last_newline = value.rfind("\n", 0, len(value) - int(value.endswith("\n")))
tail = lambda match: terminal_dot_field(value, match.end(), last_newline=last_newline)
else:
markers = list(re.finditer(r"(?<!\s)(\s++)(?:in|and|then|before)\b|[.;]|$", value, flags))
marker_index = 0
previous_start = -1
newline = value.find("\n")
def tail(match):
nonlocal marker_index, previous_start, newline
start = match.end()
if start >= len(value) or not value[start].isspace():
return None
cursor = start
while cursor < len(value) and value[cursor].isspace():
cursor += 1
if start < previous_start:
marker_index = 0
newline = value.find("\n")
previous_start = start
while marker_index < len(markers) and markers[marker_index].start() <= cursor:
marker_index += 1
if 0 <= newline < cursor:
newline = value.find("\n", cursor)
if marker_index < len(markers):
end = markers[marker_index].start()
if newline < 0 or newline >= end:
return (value[cursor:end],)
# Only after the greedy whitespace-led field fails may its
# leading run give a single dot character back to the value.
prior = marker_index - 1
if prior >= 0:
marker = markers[prior]
if marker.start() <= cursor and (
marker.start() == cursor or marker.end(1) == cursor
):
candidate = cursor - (2 if marker.group(1) else 1)
while candidate > start and value[candidate] == "\n":
candidate -= 1
if candidate > start:
return (value[candidate:candidate + 1],)
return None
elif grammar == "tag":
leaders = the
separator = r"(?<!\s)(\s++)tag\s++to(?=\s)"
tag = re.compile(r"\s++#?([a-z][a-z0-9_-]{1,30})\b", flags)
def tail(match):
result = tag.match(value, match.end())
return result.groups() if result else None
elif grammar == "remaining_checklist":
leaders = the
separator = r"(?<!\s)(\s++)checklist\b"
tail = lambda match: ()
else:
raise ValueError(f"Unknown payload grammar: {grammar}")
return space_delimited_fields(value, _space_field_leaders(leaders, flags), re.compile(separator, flags), tail)
def _captures_after_first_prefix(
value: str,
prefix_pattern: str,
@@ -2482,12 +2569,20 @@ def _captures_after_first_prefix(
*,
flags: int = re.IGNORECASE,
) -> tuple[str | None, ...] | None:
"""Match a suffix once after the first prefix that can own all later text."""
"""Match the staged production payload grammars with forward scans."""
prefix = re.search(prefix_pattern, value, flags)
if prefix is None:
return None
remainder = re.match(remainder_pattern, value[prefix.end():], flags)
return remainder.groups() if remainder is not None else None
grammars = {
r"\s+(.+?)\s+so\s+its\s+content\s+is\s+['\"]([^'\"]+)['\"]": "note_update",
r"\s+(?:all|every|the)?\s*(?:my\s+)?(.+?)\s+(?:emails?|mail|messages?)\b": "email_mutation",
r"\s+(.+?)\s+with\s+(.+?)$": "checklist",
r"\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b": "tag",
r"\s+(.+?)\s+to\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)": "change",
r"\s+(.+?)\s+with\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)": "replace",
r"\s+(?:all)?\s*(.+?)\s+emails?\b": "email_mutation",
}
return _payload_fields(value[prefix.end():], grammars[remainder_pattern], flags)
def _parse_qwen_explicit_note_update(text: str) -> Optional[tuple[str, str]]:
@@ -2605,14 +2700,14 @@ def _pipeline_request_parts(value: str) -> tuple[str, str, str, str] | None:
)
if prefix is None:
return None
remainder = re.match(
r"\s+(.+?),\s*then\s+([^\s,]+)\s+to\s+(.+?)(?:[.!?]\s*)?$",
value[prefix.end():],
re.IGNORECASE,
)
if remainder is None:
return None
return prefix.group(1), remainder.group(1), remainder.group(2), remainder.group(3)
rest = value[prefix.end():]
last_newline = rest.rfind("\n", 0, len(rest) - int(rest.endswith("\n")))
separator = re.compile(r"(),\s*+then\s++([^\s,]+)\s++to(?=\s)", re.I)
def tail(match):
final = terminal_dot_field(rest, match.end(), punctuation=True, last_newline=last_newline)
return (match.group(2), *final) if final is not None else None
fields = space_delimited_fields(rest, _space_field_leaders([(r"\s++", 1)]), separator, tail)
return (prefix.group(1), *fields) if fields is not None else None
def _parse_explicit_pipeline_request(text: str) -> Optional[tuple[str, str]]:
@@ -3018,7 +3113,7 @@ def _parse_qwen_explicit_session_find(text: str) -> Optional[tuple[str, str]]:
):
return None
if re.search(
r"\b(?:list|show|view)\b.{0,30}\b(?:my\s+)?(?:recent|latest|all)?\s*(?:chats?|sessions?|conversations?)\b"
r"\b(?:list|show|view)\b.{0,30}\b(?:my\s++)?(?:recent|latest|all)?\s*+(?:chats?|sessions?|conversations?)\b"
r"|\b(?:recent|latest|all)\s+(?:chats?|sessions?|conversations?)\b",
value,
re.IGNORECASE,
@@ -4519,12 +4614,8 @@ def _remaining_checklist_name(value: str) -> str | None:
)
if location is None:
return None
name = re.match(
r"\s+(?:the\s+)?(.+?)\s+checklist\b",
value[location.end():line_end],
re.IGNORECASE,
)
return name.group(1) if name is not None else None
fields = _payload_fields(value[location.end():line_end], "remaining_checklist")
return fields[0] if fields is not None else None
def _parse_simple_notes_tool_request(text: str) -> Optional[tuple[str, str]]:
@@ -4600,7 +4691,7 @@ def _parse_simple_notes_tool_request(text: str) -> Optional[tuple[str, str]]:
return "manage_notes", json.dumps({"action": "search", "query": query})
note_saying_match = re.search(
r"\b(?:create|add|make|save)\s+(?:a\s+)?note\s+(?:saying|that says|with)\s+(.+?)\s*$",
r"\b(?:create|add|make|save)\s++(?:a\s++)?note\s++(?:saying|that says|with)\s++(.+?)$",
value,
re.IGNORECASE,
)
@@ -12924,6 +13015,38 @@ def _contextual_summary_fragment(text: str) -> str:
return text[prefix.end():newline] if newline >= 0 else ""
def _contextual_email_subject(text):
"""Read the legacy Subject heading without partitioning newline runs."""
for keyword in re.finditer(r"\bsubject", text, re.I):
# Optional formatting is tried in the regex's original greedy order.
# There are only twelve prefix alternatives, each with disjoint spaces.
for bold in (True, False):
for colon in (True, False):
for quote in ('"', '*"', ''):
pattern = r"\s*+" + (r"\*\*\s*+" if bold else "")
pattern += (r":\s*+" if colon else "") + re.escape(quote)
prefix = re.compile(pattern).match(text, keyword.end())
if prefix is None:
continue
start = prefix.end()
if start < len(text) and text[start] != "\n":
# The first capture character is mandatory, even when
# it is itself a quote. Only later quotes terminate it.
newline = text.find("\n", start + 1)
closer = text.find('"', start + 1)
ends = [end for end in (newline, closer) if end >= 0]
end = min(ends) if ends else len(text)
return text[start:end]
# Greedy whitespace can give back its last non-LF dot
# character. This also preserves whitespace-only subjects.
candidate = start - 1
while candidate >= keyword.end() and text[candidate].isspace():
if text[candidate] != "\n":
return text[candidate:candidate + 1]
candidate -= 1
return None
def _contextual_reply_body_from_recent_email_context(messages: List[Dict]) -> str:
"""Build a bounded draft body from the latest assistant email summary.
@@ -12947,13 +13070,9 @@ def _contextual_reply_body_from_recent_email_context(messages: List[Dict]) -> st
if link_match:
subject = _clean_fragment(link_match.group(1))
if not subject:
subject_match = re.search(
r"\bsubject\s*(?:\*\*)?\s*:?\s*(?:\"|\*\")?(.+?)(?:\"|\n|$)",
text,
re.IGNORECASE,
)
if subject_match:
subject = _clean_fragment(subject_match.group(1))
subject_fragment = _contextual_email_subject(text)
if subject_fragment is not None:
subject = _clean_fragment(subject_fragment)
summary_fragment = _contextual_summary_fragment(text)
if summary_fragment:
summary = _clean_fragment(summary_fragment)
@@ -13395,7 +13514,7 @@ def _normalize_ody_qwen_text_artifacts(text: str, *, strip_edges: bool = True) -
_ODY_QWEN_LEAKED_TOOL_TEXT_RE = re.compile(
r"(<\s*/?\s*(?:function|parameter|tool_call)\b"
r"(<\s*+(?:/\s*+)?(?:function|parameter|tool_call)\b"
r"|\bmanage_(?:notes|calendar|memory|documents|contact)\s*\("
r"|\"function\"\s*:\s*\"(?:manage_|mcp__)"
r"|mcp__email__"
+76
View File
@@ -208,3 +208,79 @@ def first_tag_content(text: str, tag: str, *, allow_attributes: bool = False) ->
return None
return text[body_start:closer.start()]
return None
def space_delimited_fields(text, leaders, separator_re, tail):
"""Match a lazy dot field after a finite set of greedy leading grammars.
Separators consume a maximal whitespace run (group 1) and a fixed grammar.
Test each run once, rather than repartitioning it between a leading space
quantifier, a dot capture, and the separator. ``tail`` must also scan
monotonically or use only a fixed/bounded grammar.
"""
separators = list(separator_re.finditer(text))
for leader_re, minimum_space in leaders:
leader = leader_re.match(text)
if leader is None:
continue
start = leader.end()
newline = text.find("\n", start)
line_end = len(text) if newline < 0 else newline
for separator in separators:
end = separator.start(1)
if start < end <= line_end:
remainder = tail(separator)
if remainder is not None:
return (text[start:end], *remainder)
# A greedy leading run may give back a dot character when the field
# consists entirely of whitespace. Only the last non-LF character
# that leaves the mandatory separator can win; do not retry suffixes.
run_start = start
while run_start and text[run_start - 1].isspace():
run_start -= 1
earliest = run_start + minimum_space
for separator in separators:
if separator.end(1) != start:
continue
end = start - (1 if separator.group(1) else 0)
candidate = end - 1
while candidate >= earliest and text[candidate] == "\n":
candidate -= 1
if candidate >= earliest:
remainder = tail(separator)
if remainder is not None:
return (text[candidate:candidate + 1], *remainder)
return None
def terminal_dot_field(text, start, *, punctuation=False, last_newline=None):
"""Read a whitespace-led dot field ending at Python's dollar boundary."""
if start >= len(text) or not text[start].isspace():
return None
cursor = start
while cursor < len(text) and text[cursor].isspace():
cursor += 1
end = len(text) - (1 if text.endswith("\n") else 0)
if cursor >= end:
cursor = end - 1
while cursor > start and text[cursor] == "\n":
cursor -= 1
if cursor <= start or text[cursor] == "\n":
return None
# Newlines before the field can be leading whitespace; newlines inside
# the dot capture cannot be consumed. The last LF is a constant-time veto.
if last_newline is None:
last_newline = text.rfind("\n", 0, end)
if last_newline >= cursor:
return None
capture_end = end
if punctuation:
while capture_end > cursor and text[capture_end - 1].isspace():
capture_end -= 1
if capture_end > cursor and text[capture_end - 1] in ".!?":
capture_end -= 1
else:
capture_end = end
if capture_end == cursor:
capture_end = end
return (text[cursor:capture_end],)
+40 -6
View File
@@ -1515,6 +1515,45 @@ def _parse_tool_code_block(raw: str) -> Optional[ToolBlock]:
return ToolBlock(tool_name, content.strip())
return None
_GEMMA_FALLBACK_KEY_RE = re.compile(r"\w++")
_GEMMA_FALLBACK_COLON_RE = re.compile(r"\s*+:")
_GEMMA_FALLBACK_BOUNDARY_RE = re.compile(r"(?<!\s)(\s*+)(?:,\s*+\w++\s*+:|})")
def _iter_gemma_fallback_items(body):
"""Preserve permissive k:v recovery with one pass over keys/boundaries."""
boundaries = list(_GEMMA_FALLBACK_BOUNDARY_RE.finditer(body))
boundary_index = 0
pos = 0
newline = body.find("\n")
while key := _GEMMA_FALLBACK_KEY_RE.search(body, pos):
colon = _GEMMA_FALLBACK_COLON_RE.match(body, key.end())
if colon is None:
pos = key.end()
continue
start = colon.end()
while start < len(body) and body[start].isspace():
start += 1
if start < len(body) and body[start] in "\"'":
start += 1
while boundary_index < len(boundaries) and boundaries[boundary_index].end(1) < start:
boundary_index += 1
if boundary_index == len(boundaries):
return
boundary = boundaries[boundary_index]
end = max(start, boundary.start())
if 0 <= newline < start:
newline = body.find("\n", start)
if newline >= 0 and newline < end:
pos = key.end()
continue
capture_end = end
if capture_end > start and body[capture_end - 1] in "\"'":
capture_end -= 1
yield key.group(0), body[start:capture_end].strip()
pos = end
def _parse_gemma_tool_call(tool_name: str, body: str) -> Optional[ToolBlock]:
"""Parse a Gemma-style call:tool_name{...} block into a ToolBlock."""
tool_name = tool_name.strip().lower().replace("-", "_")
@@ -1539,12 +1578,7 @@ def _parse_gemma_tool_call(tool_name: str, body: str) -> Optional[ToolBlock]:
if not isinstance(params, dict):
params = {}
except Exception:
# Simple regex key-value extraction fallback
params = {}
for m in re.finditer(r'(\w+)\s*:\s*["\']?(.*?)["\']?(?=\s*,\s*\w+\s*:|\s*\})', body):
k = m.group(1)
v = m.group(2).strip()
params[k] = v
params = dict(_iter_gemma_fallback_items(body))
from src.tool_schemas import function_call_to_tool_block
return function_call_to_tool_block(tool_name, json.dumps(params))
+10 -7
View File
@@ -2304,6 +2304,9 @@ def _terminal_clause_match(text: str, core_pattern: str) -> re.Match[str] | None
while end and text[end - 1] in ".!?":
end -= 1
possessive = core_pattern.replace(r"\s+", r"\s++").replace(r"\s*", r"\s*+")
# An optional leading " and" branch must start at a whitespace boundary;
# otherwise search retries the whole possessive run at every suffix.
possessive = possessive.replace(r"|\s++and", r"|(?<!\s)\s++and")
return re.search(possessive + r"$", text[:end], re.I)
@@ -2403,7 +2406,7 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
if safety_tail:
text = text[:safety_tail.start()].strip()
text = re.sub(
r"[.!?]\s*read[- ]only(?:\s+(?:please|pls|plz))?\s*,?\s*"
r"[.!?]\s*+read[- ]only(?:\s++(?:please|pls|plz))?\s*+(?:,\s*+)?"
r"(?:do\s+not|don['’]?t|dont)\s+(?:change|edit|modify)\s+"
r"(?:or\s+send\s+)?anything[.!?]*\s*$",
"", text, flags=re.I,
@@ -2469,11 +2472,11 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
maximum = int(raw) if raw.isdecimal() else _READ_COUNT_WORDS[raw.lower()]
text = text[:approximate_limit.start()].strip()
natural_limit = re.search(
r"(?:[,.;?]|[—–-]|\s+but\s+|\s+)\s*(?:"
r"(?:i\s+)?only\s+(?:need|want|show(?:\s+me)?)?\s*(?:(?:the\s+)?first\s+)?"
r"|just\s+(?:(?:the\s+)?first\s+)?|(?:show\s+me\s+)?like\s+|no\s+more\s+than\s+"
r"|cap(?:\s+(?:it|them|the\s+(?:answer|list)))?\s+at\s+"
r"|(?:maybe\s+)?(?:first|same)\s+)"
r"(?:[,.;?]|[—–-]|(?<!\s)\s++but\s++|(?<!\s)\s++)\s*+(?:"
r"(?:i\s++)?only\s++(?:need|want|show(?:\s++me)?)?\s*+(?:(?:the\s++)?first\s++)?"
r"|just\s++(?:(?:the\s++)?first\s++)?|(?:show\s++me\s++)?like\s++|no\s++more\s++than\s++"
r"|cap(?:\s++(?:it|them|the\s++(?:answer|list)))?\s++at\s++"
r"|(?:maybe\s++)?(?:first|same)\s++)"
r"(" + _READ_COUNT + r")"
r"(?:\s+(?:short\s+)?(?:titles?|items?|results?|entries?|names?|ones?|bits?|things?))?"
r"(?:\s*(?:and|\+)\s+(?:their\s+)?(?:status(?:es)?|states?))?"
@@ -2511,7 +2514,7 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]:
maximum = int(raw) if raw.isdecimal() else _READ_COUNT_WORDS[raw.lower()]
text = text[:compact_limit.start()].strip()
conversational_limit = re.search(
r"(?:[,.;?]\s*|\s+)(?:maybe\s+)?(?:keep\s+it\s+to\s+|stick\s+to\s+|(?:first|top)\s+)"
r"(?:[,.;?]\s*|(?<!\s)\s++)(?:maybe\s+)?(?:keep\s+it\s+to\s+|stick\s+to\s+|(?:first|top)\s+)"
r"(" + _READ_COUNT + r")(?:\s+(?:short\s+)?(?:titles?|items?|results?|entries?|names?|ones?))?"
r"[.!?]*\s*$",
text, re.I,