fix(security): eliminate parser denial-of-service paths

This commit is contained in:
Alexandre Teixeira
2026-10-06 03:16:53 +01:00
parent ab6f52d30c
commit da3800b662
8 changed files with 2676 additions and 335 deletions
+210
View File
@@ -0,0 +1,210 @@
"""Forward-only helpers for permissive text grammars.
These helpers retain the existing regular expressions as anchored token
parsers while preventing ``re.search`` from retrying the same token suffix at
every embedded prefix.
"""
from __future__ import annotations
import re
from collections.abc import Iterator
from typing import Match, Pattern
def iter_prefixed_token_matches(
text: str,
candidate_re: Pattern[str],
anchored_re: Pattern[str],
token_tail_re: Pattern[str],
) -> Iterator[Match[str]]:
"""Yield legacy greedy matches after testing one prefix per token.
``anchored_re`` must start with the same fixed prefix recognized by
``candidate_re``. ``token_tail_re`` describes characters that the
anchored grammar can consume after that prefix. If the first prefix in
such a token cannot match, a later embedded prefix cannot match either:
its suffix was already available to the first attempt. Advancing to the
token boundary makes failed scans linear without changing successful
greedy captures.
"""
pos = 0
while candidate := candidate_re.search(text, pos):
match = anchored_re.match(text, candidate.start())
if match is not None:
yield match
pos = match.end()
continue
tail = token_tail_re.match(text, candidate.end())
pos = max(candidate.end(), tail.end() if tail is not None else candidate.end())
def has_prefixed_token_match(
text: str,
candidate_re: Pattern[str],
anchored_re: Pattern[str],
token_tail_re: Pattern[str],
) -> bool:
"""Return whether ``iter_prefixed_token_matches`` yields a match."""
return next(
iter_prefixed_token_matches(text, candidate_re, anchored_re, token_tail_re),
None,
) is not None
_HTTP_URL_PREFIX_RE = re.compile(r"https?://", re.IGNORECASE)
_VIDEO_DETAIL_BASE_RE = re.compile(
r"\b(?:how\s+many|count|break\s*points?|timestamps?|what\s+time|score(?:board)?s?)\b",
re.IGNORECASE,
)
_VIDEO_DETAIL_EXTENDED_RE = re.compile(
r"\b(?:sequence|in\s+order|chronological|at\s+what\s+time)\b|"
r"(?:多少|几次|何时|什么时候|时间|顺序)",
re.IGNORECASE,
)
_WHEN_RE = re.compile(r"\bwhen\s+", re.IGNORECASE)
_WHEN_TARGET_RE = re.compile(r"(?:end|happen)\b", re.IGNORECASE)
_FIRST_RE = re.compile(r"\bfirst\s+", re.IGNORECASE)
_FIRST_TARGET_RE = re.compile(r"(?:save|attempt|event)\b", re.IGNORECASE)
def contains_search_engine_navigation(text: str) -> bool:
"""Match the legacy Google/Bing/DuckDuckGo navigation URL grammar.
The old expression backtracked through every possible optional subdomain
split. Parsing the host up to its first slash gives the same accepted
hosts and path prefixes with one pass per URL candidate.
"""
pos = 0
while candidate := _HTTP_URL_PREFIX_RE.search(text, pos):
host_start = candidate.end()
path_start = text.find("/", host_start)
if path_start < 0:
return False
host = text[host_start:path_start].casefold()
path = text[path_start + 1:path_start + 7].casefold()
google_at = host.rfind(".google.")
recognized_host = (
(host.startswith("google.") and len(host) > len("google."))
or (google_at >= 0 and google_at + len(".google.") < len(host))
or host == "bing.com"
or host.endswith(".bing.com")
or host == "duckduckgo.com"
or host.endswith(".duckduckgo.com")
)
if recognized_host and path.startswith(("search", "sorry", "html", "lite", "?")):
return True
# A later URL may begin in the path. Resume after this scheme rather
# than skipping the whole non-whitespace region.
pos = candidate.end()
return False
def contains_detailed_sequence_request(text: str, *, include_first: bool = True) -> bool:
"""Recognize count/order/timing requests without overlapping ``.*`` scans."""
value = str(text or "")
if _VIDEO_DETAIL_BASE_RE.search(value):
return True
if include_first and _VIDEO_DETAIL_EXTENDED_RE.search(value):
return True
pos = 0
while when := _WHEN_RE.search(value, pos):
line_end = value.find("\n", when.end())
if line_end < 0:
line_end = len(value)
if _WHEN_TARGET_RE.search(value, when.end(), line_end) is not None:
return True
pos = line_end + 1
if include_first:
pos = 0
while first := _FIRST_RE.search(value, pos):
line_end = value.find("\n", first.end())
if line_end < 0:
line_end = len(value)
if _FIRST_TARGET_RE.search(value, first.end(), line_end) is not None:
return True
pos = line_end + 1
return False
def iter_angle_contents(text: str) -> Iterator[tuple[int, int, str]]:
"""Yield nonempty flat ``<...>`` contents with monotonic delimiters."""
pos = 0
while (start := text.find("<", pos)) >= 0:
end = text.find(">", start + 1)
if end < 0:
return
if end > start + 1:
yield start, end + 1, text[start + 1:end]
pos = end + 1
else:
pos = start + 1
def iter_markdown_links(
text: str,
*,
target_prefix: str = "",
target_re: Pattern[str] | None = None,
) -> Iterator[tuple[int, int, str, str]]:
"""Yield flat Markdown links accepted by the legacy link regexes."""
pos = 0
marker = "](" + target_prefix
while (start := text.find("[", pos)) >= 0:
label_end = text.find("]", start + 1)
if label_end < 0:
return
if label_end == start + 1:
pos = start + 1
continue
if not text.startswith(marker, label_end):
pos = label_end + 1
continue
target_start = label_end + len(marker)
target_end = text.find(")", target_start)
if target_end < 0:
return
target = text[target_start:target_end]
if target and (target_re is None or target_re.fullmatch(target)):
yield start, target_end + 1, text[start + 1:label_end], target
pos = target_end + 1
else:
pos = label_end + 1
def replace_markdown_links_with_labels(text: str) -> str:
"""Linear equivalent of replacing flat Markdown links with their labels."""
links = list(iter_markdown_links(text))
if not links:
return text
out = []
pos = 0
for start, end, label, _target in links:
out.extend((text[pos:start], label))
pos = end
out.append(text[pos:])
return "".join(out)
def first_tag_content(text: str, tag: str, *, allow_attributes: bool = False) -> str | None:
"""Return the first flat tag body using forward-only opener/closer scans."""
opener_re = re.compile(r"<" + re.escape(tag), re.IGNORECASE)
closer_re = re.compile(r"</" + re.escape(tag) + r">", re.IGNORECASE)
pos = 0
while opener := opener_re.search(text, pos):
name_end = opener.end()
if name_end < len(text) and text[name_end] == ">":
body_start = name_end + 1
elif allow_attributes and name_end < len(text) and text[name_end].isspace():
tag_end = text.find(">", name_end + 1)
if tag_end < 0:
return None
body_start = tag_end + 1
else:
pos = name_end
continue
closer = closer_re.search(text, body_start)
if closer is None:
return None
return text[body_start:closer.start()]
return None