From b1e6c7d3eb0087881db614a2eb06fbe63f0c94c7 Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Tue, 6 Oct 2026 13:36:01 +0100 Subject: [PATCH] fix(security): eliminate induced regex denial-of-service paths --- src/agent_loop.py | 175 +++++++++++++++++++++---- src/text_scanning.py | 76 +++++++++++ src/tool_parsing.py | 46 ++++++- src/turn_contract.py | 17 ++- tests/test_codeql_induced_redos.py | 197 +++++++++++++++++++++++++++++ website/configuration-reference.md | 12 +- 6 files changed, 476 insertions(+), 47 deletions(-) create mode 100644 tests/test_codeql_induced_redos.py diff --git a/src/agent_loop.py b/src/agent_loop.py index 8a70fd3ad..efc6b4e05 100644 --- a/src/agent_loop.py +++ b/src/agent_loop.py @@ -86,6 +86,8 @@ from src.tool_parsing import iter_email_addresses, strip_angle_tags from src.text_scanning import ( contains_detailed_sequence_request, has_prefixed_token_match, + space_delimited_fields, + terminal_dot_field, first_tag_content, iter_angle_contents, iter_markdown_links, @@ -2155,7 +2157,7 @@ def _looks_like_agent_reasoning_preamble(text: str) -> bool: r"(?:let me|i(?:'ll| will)(?:\s+need\s+to)?|i\s+can(?:\s+now)?|i(?:'m| am)\s+(?:preparing|planning)\s+to)\s+" r"(?:(?:carefully|methodically|systematically|closely|further)\s+){0,2}" r"(?:continue|continuing|analy[sz]e|scan|check|verify|inspect|confirm|look up|fetch|open|list|search|refine|request|review|track|read|watch|(?:re-?)?examine|provide|give|state|report|answer|respond|summarize|conclude)\b" - r"[^.!?]*[.!?]?\s*$", + r"[^.!?]*+[.!?]?\s*$", trailing_segment, ): return True @@ -2475,6 +2477,91 @@ def _parse_qwen_explicit_note_delete(text: str) -> Optional[str]: return match.group(1).strip().strip("\"'`").rstrip(".") if match else None +def _space_field_leaders(patterns, flags=re.IGNORECASE): + return [(re.compile(pattern, flags), minimum) for pattern, minimum in patterns] + + +def _payload_fields(value, grammar, flags=re.IGNORECASE): + """Scan the finite command payload grammars without whitespace partitions.""" + simple = [(r"\s++", 1)] + the = [(r"\s++the\s++", 1), *simple] + if grammar == "note_update": + leaders = simple + separator = r"(?= len(value) or not value[start].isspace(): + return None + cursor = start + while cursor < len(value) and value[cursor].isspace(): + cursor += 1 + if start < previous_start: + marker_index = 0 + newline = value.find("\n") + previous_start = start + while marker_index < len(markers) and markers[marker_index].start() <= cursor: + marker_index += 1 + if 0 <= newline < cursor: + newline = value.find("\n", cursor) + if marker_index < len(markers): + end = markers[marker_index].start() + if newline < 0 or newline >= end: + return (value[cursor:end],) + # Only after the greedy whitespace-led field fails may its + # leading run give a single dot character back to the value. + prior = marker_index - 1 + if prior >= 0: + marker = markers[prior] + if marker.start() <= cursor and ( + marker.start() == cursor or marker.end(1) == cursor + ): + candidate = cursor - (2 if marker.group(1) else 1) + while candidate > start and value[candidate] == "\n": + candidate -= 1 + if candidate > start: + return (value[candidate:candidate + 1],) + return None + elif grammar == "tag": + leaders = the + separator = r"(? tuple[str | None, ...] | None: - """Match a suffix once after the first prefix that can own all later text.""" + """Match the staged production payload grammars with forward scans.""" prefix = re.search(prefix_pattern, value, flags) if prefix is None: return None - remainder = re.match(remainder_pattern, value[prefix.end():], flags) - return remainder.groups() if remainder is not None else None + grammars = { + r"\s+(.+?)\s+so\s+its\s+content\s+is\s+['\"]([^'\"]+)['\"]": "note_update", + r"\s+(?:all|every|the)?\s*(?:my\s+)?(.+?)\s+(?:emails?|mail|messages?)\b": "email_mutation", + r"\s+(.+?)\s+with\s+(.+?)$": "checklist", + r"\s+(?:the\s+)?(.+?)\s+tag\s+to\s+#?([a-z][a-z0-9_-]{1,30})\b": "tag", + r"\s+(.+?)\s+to\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)": "change", + r"\s+(.+?)\s+with\s+(.+?)(?=\s+(?:in|and|then|before)\b|[.;]|$)": "replace", + r"\s+(?:all)?\s*(.+?)\s+emails?\b": "email_mutation", + } + return _payload_fields(value[prefix.end():], grammars[remainder_pattern], flags) def _parse_qwen_explicit_note_update(text: str) -> Optional[tuple[str, str]]: @@ -2605,14 +2700,14 @@ def _pipeline_request_parts(value: str) -> tuple[str, str, str, str] | None: ) if prefix is None: return None - remainder = re.match( - r"\s+(.+?),\s*then\s+([^\s,]+)\s+to\s+(.+?)(?:[.!?]\s*)?$", - value[prefix.end():], - re.IGNORECASE, - ) - if remainder is None: - return None - return prefix.group(1), remainder.group(1), remainder.group(2), remainder.group(3) + rest = value[prefix.end():] + last_newline = rest.rfind("\n", 0, len(rest) - int(rest.endswith("\n"))) + separator = re.compile(r"(),\s*+then\s++([^\s,]+)\s++to(?=\s)", re.I) + def tail(match): + final = terminal_dot_field(rest, match.end(), punctuation=True, last_newline=last_newline) + return (match.group(2), *final) if final is not None else None + fields = space_delimited_fields(rest, _space_field_leaders([(r"\s++", 1)]), separator, tail) + return (prefix.group(1), *fields) if fields is not None else None def _parse_explicit_pipeline_request(text: str) -> Optional[tuple[str, str]]: @@ -3018,7 +3113,7 @@ def _parse_qwen_explicit_session_find(text: str) -> Optional[tuple[str, str]]: ): return None if re.search( - r"\b(?:list|show|view)\b.{0,30}\b(?:my\s+)?(?:recent|latest|all)?\s*(?:chats?|sessions?|conversations?)\b" + r"\b(?:list|show|view)\b.{0,30}\b(?:my\s++)?(?:recent|latest|all)?\s*+(?:chats?|sessions?|conversations?)\b" r"|\b(?:recent|latest|all)\s+(?:chats?|sessions?|conversations?)\b", value, re.IGNORECASE, @@ -4519,12 +4614,8 @@ def _remaining_checklist_name(value: str) -> str | None: ) if location is None: return None - name = re.match( - r"\s+(?:the\s+)?(.+?)\s+checklist\b", - value[location.end():line_end], - re.IGNORECASE, - ) - return name.group(1) if name is not None else None + fields = _payload_fields(value[location.end():line_end], "remaining_checklist") + return fields[0] if fields is not None else None def _parse_simple_notes_tool_request(text: str) -> Optional[tuple[str, str]]: @@ -4600,7 +4691,7 @@ def _parse_simple_notes_tool_request(text: str) -> Optional[tuple[str, str]]: return "manage_notes", json.dumps({"action": "search", "query": query}) note_saying_match = re.search( - r"\b(?:create|add|make|save)\s+(?:a\s+)?note\s+(?:saying|that says|with)\s+(.+?)\s*$", + r"\b(?:create|add|make|save)\s++(?:a\s++)?note\s++(?:saying|that says|with)\s++(.+?)$", value, re.IGNORECASE, ) @@ -12924,6 +13015,38 @@ def _contextual_summary_fragment(text: str) -> str: return text[prefix.end():newline] if newline >= 0 else "" +def _contextual_email_subject(text): + """Read the legacy Subject heading without partitioning newline runs.""" + for keyword in re.finditer(r"\bsubject", text, re.I): + # Optional formatting is tried in the regex's original greedy order. + # There are only twelve prefix alternatives, each with disjoint spaces. + for bold in (True, False): + for colon in (True, False): + for quote in ('"', '*"', ''): + pattern = r"\s*+" + (r"\*\*\s*+" if bold else "") + pattern += (r":\s*+" if colon else "") + re.escape(quote) + prefix = re.compile(pattern).match(text, keyword.end()) + if prefix is None: + continue + start = prefix.end() + if start < len(text) and text[start] != "\n": + # The first capture character is mandatory, even when + # it is itself a quote. Only later quotes terminate it. + newline = text.find("\n", start + 1) + closer = text.find('"', start + 1) + ends = [end for end in (newline, closer) if end >= 0] + end = min(ends) if ends else len(text) + return text[start:end] + # Greedy whitespace can give back its last non-LF dot + # character. This also preserves whitespace-only subjects. + candidate = start - 1 + while candidate >= keyword.end() and text[candidate].isspace(): + if text[candidate] != "\n": + return text[candidate:candidate + 1] + candidate -= 1 + return None + + def _contextual_reply_body_from_recent_email_context(messages: List[Dict]) -> str: """Build a bounded draft body from the latest assistant email summary. @@ -12947,13 +13070,9 @@ def _contextual_reply_body_from_recent_email_context(messages: List[Dict]) -> st if link_match: subject = _clean_fragment(link_match.group(1)) if not subject: - subject_match = re.search( - r"\bsubject\s*(?:\*\*)?\s*:?\s*(?:\"|\*\")?(.+?)(?:\"|\n|$)", - text, - re.IGNORECASE, - ) - if subject_match: - subject = _clean_fragment(subject_match.group(1)) + subject_fragment = _contextual_email_subject(text) + if subject_fragment is not None: + subject = _clean_fragment(subject_fragment) summary_fragment = _contextual_summary_fragment(text) if summary_fragment: summary = _clean_fragment(summary_fragment) @@ -13395,7 +13514,7 @@ def _normalize_ody_qwen_text_artifacts(text: str, *, strip_edges: bool = True) - _ODY_QWEN_LEAKED_TOOL_TEXT_RE = re.compile( - r"(<\s*/?\s*(?:function|parameter|tool_call)\b" + r"(<\s*+(?:/\s*+)?(?:function|parameter|tool_call)\b" r"|\bmanage_(?:notes|calendar|memory|documents|contact)\s*\(" r"|\"function\"\s*:\s*\"(?:manage_|mcp__)" r"|mcp__email__" diff --git a/src/text_scanning.py b/src/text_scanning.py index 59fad4cf5..c18783eac 100644 --- a/src/text_scanning.py +++ b/src/text_scanning.py @@ -208,3 +208,79 @@ def first_tag_content(text: str, tag: str, *, allow_attributes: bool = False) -> return None return text[body_start:closer.start()] return None + + +def space_delimited_fields(text, leaders, separator_re, tail): + """Match a lazy dot field after a finite set of greedy leading grammars. + + Separators consume a maximal whitespace run (group 1) and a fixed grammar. + Test each run once, rather than repartitioning it between a leading space + quantifier, a dot capture, and the separator. ``tail`` must also scan + monotonically or use only a fixed/bounded grammar. + """ + separators = list(separator_re.finditer(text)) + for leader_re, minimum_space in leaders: + leader = leader_re.match(text) + if leader is None: + continue + start = leader.end() + newline = text.find("\n", start) + line_end = len(text) if newline < 0 else newline + for separator in separators: + end = separator.start(1) + if start < end <= line_end: + remainder = tail(separator) + if remainder is not None: + return (text[start:end], *remainder) + # A greedy leading run may give back a dot character when the field + # consists entirely of whitespace. Only the last non-LF character + # that leaves the mandatory separator can win; do not retry suffixes. + run_start = start + while run_start and text[run_start - 1].isspace(): + run_start -= 1 + earliest = run_start + minimum_space + for separator in separators: + if separator.end(1) != start: + continue + end = start - (1 if separator.group(1) else 0) + candidate = end - 1 + while candidate >= earliest and text[candidate] == "\n": + candidate -= 1 + if candidate >= earliest: + remainder = tail(separator) + if remainder is not None: + return (text[candidate:candidate + 1], *remainder) + return None + + +def terminal_dot_field(text, start, *, punctuation=False, last_newline=None): + """Read a whitespace-led dot field ending at Python's dollar boundary.""" + if start >= len(text) or not text[start].isspace(): + return None + cursor = start + while cursor < len(text) and text[cursor].isspace(): + cursor += 1 + end = len(text) - (1 if text.endswith("\n") else 0) + if cursor >= end: + cursor = end - 1 + while cursor > start and text[cursor] == "\n": + cursor -= 1 + if cursor <= start or text[cursor] == "\n": + return None + # Newlines before the field can be leading whitespace; newlines inside + # the dot capture cannot be consumed. The last LF is a constant-time veto. + if last_newline is None: + last_newline = text.rfind("\n", 0, end) + if last_newline >= cursor: + return None + capture_end = end + if punctuation: + while capture_end > cursor and text[capture_end - 1].isspace(): + capture_end -= 1 + if capture_end > cursor and text[capture_end - 1] in ".!?": + capture_end -= 1 + else: + capture_end = end + if capture_end == cursor: + capture_end = end + return (text[cursor:capture_end],) diff --git a/src/tool_parsing.py b/src/tool_parsing.py index c9c3bae3d..ef5a52034 100644 --- a/src/tool_parsing.py +++ b/src/tool_parsing.py @@ -1515,6 +1515,45 @@ def _parse_tool_code_block(raw: str) -> Optional[ToolBlock]: return ToolBlock(tool_name, content.strip()) return None +_GEMMA_FALLBACK_KEY_RE = re.compile(r"\w++") +_GEMMA_FALLBACK_COLON_RE = re.compile(r"\s*+:") +_GEMMA_FALLBACK_BOUNDARY_RE = re.compile(r"(?= 0 and newline < end: + pos = key.end() + continue + capture_end = end + if capture_end > start and body[capture_end - 1] in "\"'": + capture_end -= 1 + yield key.group(0), body[start:capture_end].strip() + pos = end + + def _parse_gemma_tool_call(tool_name: str, body: str) -> Optional[ToolBlock]: """Parse a Gemma-style call:tool_name{...} block into a ToolBlock.""" tool_name = tool_name.strip().lower().replace("-", "_") @@ -1539,12 +1578,7 @@ def _parse_gemma_tool_call(tool_name: str, body: str) -> Optional[ToolBlock]: if not isinstance(params, dict): params = {} except Exception: - # Simple regex key-value extraction fallback - params = {} - for m in re.finditer(r'(\w+)\s*:\s*["\']?(.*?)["\']?(?=\s*,\s*\w+\s*:|\s*\})', body): - k = m.group(1) - v = m.group(2).strip() - params[k] = v + params = dict(_iter_gemma_fallback_items(body)) from src.tool_schemas import function_call_to_tool_block return function_call_to_tool_block(tool_name, json.dumps(params)) diff --git a/src/turn_contract.py b/src/turn_contract.py index 87c87c7da..d45df1d26 100644 --- a/src/turn_contract.py +++ b/src/turn_contract.py @@ -2304,6 +2304,9 @@ def _terminal_clause_match(text: str, core_pattern: str) -> re.Match[str] | None while end and text[end - 1] in ".!?": end -= 1 possessive = core_pattern.replace(r"\s+", r"\s++").replace(r"\s*", r"\s*+") + # An optional leading " and" branch must start at a whitespace boundary; + # otherwise search retries the whole possessive run at every suffix. + possessive = possessive.replace(r"|\s++and", r"|(? tuple[str, int | None]: if safety_tail: text = text[:safety_tail.start()].strip() text = re.sub( - r"[.!?]\s*read[- ]only(?:\s+(?:please|pls|plz))?\s*,?\s*" + r"[.!?]\s*+read[- ]only(?:\s++(?:please|pls|plz))?\s*+(?:,\s*+)?" r"(?:do\s+not|don['’]?t|dont)\s+(?:change|edit|modify)\s+" r"(?:or\s+send\s+)?anything[.!?]*\s*$", "", text, flags=re.I, @@ -2469,11 +2472,11 @@ def _read_request_and_limit(message: str) -> tuple[str, int | None]: maximum = int(raw) if raw.isdecimal() else _READ_COUNT_WORDS[raw.lower()] text = text[:approximate_limit.start()].strip() natural_limit = re.search( - r"(?:[,.;?]|[—–-]|\s+but\s+|\s+)\s*(?:" - r"(?:i\s+)?only\s+(?:need|want|show(?:\s+me)?)?\s*(?:(?:the\s+)?first\s+)?" - r"|just\s+(?:(?:the\s+)?first\s+)?|(?:show\s+me\s+)?like\s+|no\s+more\s+than\s+" - r"|cap(?:\s+(?:it|them|the\s+(?:answer|list)))?\s+at\s+" - r"|(?:maybe\s+)?(?:first|same)\s+)" + r"(?:[,.;?]|[—–-]|(? tuple[str, int | None]: maximum = int(raw) if raw.isdecimal() else _READ_COUNT_WORDS[raw.lower()] text = text[:compact_limit.start()].strip() conversational_limit = re.search( - r"(?:[,.;?]\s*|\s+)(?:maybe\s+)?(?:keep\s+it\s+to\s+|stick\s+to\s+|(?:first|top)\s+)" + r"(?:[,.;?]\s*|(?