mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-11 09:22:21 +02:00
fix(security): eliminate parser denial-of-service paths
This commit is contained in:
@@ -0,0 +1,210 @@
|
||||
"""Forward-only helpers for permissive text grammars.
|
||||
|
||||
These helpers retain the existing regular expressions as anchored token
|
||||
parsers while preventing ``re.search`` from retrying the same token suffix at
|
||||
every embedded prefix.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from collections.abc import Iterator
|
||||
from typing import Match, Pattern
|
||||
|
||||
|
||||
def iter_prefixed_token_matches(
|
||||
text: str,
|
||||
candidate_re: Pattern[str],
|
||||
anchored_re: Pattern[str],
|
||||
token_tail_re: Pattern[str],
|
||||
) -> Iterator[Match[str]]:
|
||||
"""Yield legacy greedy matches after testing one prefix per token.
|
||||
|
||||
``anchored_re`` must start with the same fixed prefix recognized by
|
||||
``candidate_re``. ``token_tail_re`` describes characters that the
|
||||
anchored grammar can consume after that prefix. If the first prefix in
|
||||
such a token cannot match, a later embedded prefix cannot match either:
|
||||
its suffix was already available to the first attempt. Advancing to the
|
||||
token boundary makes failed scans linear without changing successful
|
||||
greedy captures.
|
||||
"""
|
||||
pos = 0
|
||||
while candidate := candidate_re.search(text, pos):
|
||||
match = anchored_re.match(text, candidate.start())
|
||||
if match is not None:
|
||||
yield match
|
||||
pos = match.end()
|
||||
continue
|
||||
tail = token_tail_re.match(text, candidate.end())
|
||||
pos = max(candidate.end(), tail.end() if tail is not None else candidate.end())
|
||||
|
||||
|
||||
def has_prefixed_token_match(
|
||||
text: str,
|
||||
candidate_re: Pattern[str],
|
||||
anchored_re: Pattern[str],
|
||||
token_tail_re: Pattern[str],
|
||||
) -> bool:
|
||||
"""Return whether ``iter_prefixed_token_matches`` yields a match."""
|
||||
return next(
|
||||
iter_prefixed_token_matches(text, candidate_re, anchored_re, token_tail_re),
|
||||
None,
|
||||
) is not None
|
||||
|
||||
|
||||
_HTTP_URL_PREFIX_RE = re.compile(r"https?://", re.IGNORECASE)
|
||||
_VIDEO_DETAIL_BASE_RE = re.compile(
|
||||
r"\b(?:how\s+many|count|break\s*points?|timestamps?|what\s+time|score(?:board)?s?)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_VIDEO_DETAIL_EXTENDED_RE = re.compile(
|
||||
r"\b(?:sequence|in\s+order|chronological|at\s+what\s+time)\b|"
|
||||
r"(?:多少|几次|何时|什么时候|时间|顺序)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_WHEN_RE = re.compile(r"\bwhen\s+", re.IGNORECASE)
|
||||
_WHEN_TARGET_RE = re.compile(r"(?:end|happen)\b", re.IGNORECASE)
|
||||
_FIRST_RE = re.compile(r"\bfirst\s+", re.IGNORECASE)
|
||||
_FIRST_TARGET_RE = re.compile(r"(?:save|attempt|event)\b", re.IGNORECASE)
|
||||
|
||||
|
||||
def contains_search_engine_navigation(text: str) -> bool:
|
||||
"""Match the legacy Google/Bing/DuckDuckGo navigation URL grammar.
|
||||
|
||||
The old expression backtracked through every possible optional subdomain
|
||||
split. Parsing the host up to its first slash gives the same accepted
|
||||
hosts and path prefixes with one pass per URL candidate.
|
||||
"""
|
||||
pos = 0
|
||||
while candidate := _HTTP_URL_PREFIX_RE.search(text, pos):
|
||||
host_start = candidate.end()
|
||||
path_start = text.find("/", host_start)
|
||||
if path_start < 0:
|
||||
return False
|
||||
host = text[host_start:path_start].casefold()
|
||||
path = text[path_start + 1:path_start + 7].casefold()
|
||||
google_at = host.rfind(".google.")
|
||||
recognized_host = (
|
||||
(host.startswith("google.") and len(host) > len("google."))
|
||||
or (google_at >= 0 and google_at + len(".google.") < len(host))
|
||||
or host == "bing.com"
|
||||
or host.endswith(".bing.com")
|
||||
or host == "duckduckgo.com"
|
||||
or host.endswith(".duckduckgo.com")
|
||||
)
|
||||
if recognized_host and path.startswith(("search", "sorry", "html", "lite", "?")):
|
||||
return True
|
||||
# A later URL may begin in the path. Resume after this scheme rather
|
||||
# than skipping the whole non-whitespace region.
|
||||
pos = candidate.end()
|
||||
return False
|
||||
|
||||
|
||||
def contains_detailed_sequence_request(text: str, *, include_first: bool = True) -> bool:
|
||||
"""Recognize count/order/timing requests without overlapping ``.*`` scans."""
|
||||
value = str(text or "")
|
||||
if _VIDEO_DETAIL_BASE_RE.search(value):
|
||||
return True
|
||||
if include_first and _VIDEO_DETAIL_EXTENDED_RE.search(value):
|
||||
return True
|
||||
pos = 0
|
||||
while when := _WHEN_RE.search(value, pos):
|
||||
line_end = value.find("\n", when.end())
|
||||
if line_end < 0:
|
||||
line_end = len(value)
|
||||
if _WHEN_TARGET_RE.search(value, when.end(), line_end) is not None:
|
||||
return True
|
||||
pos = line_end + 1
|
||||
if include_first:
|
||||
pos = 0
|
||||
while first := _FIRST_RE.search(value, pos):
|
||||
line_end = value.find("\n", first.end())
|
||||
if line_end < 0:
|
||||
line_end = len(value)
|
||||
if _FIRST_TARGET_RE.search(value, first.end(), line_end) is not None:
|
||||
return True
|
||||
pos = line_end + 1
|
||||
return False
|
||||
|
||||
|
||||
def iter_angle_contents(text: str) -> Iterator[tuple[int, int, str]]:
|
||||
"""Yield nonempty flat ``<...>`` contents with monotonic delimiters."""
|
||||
pos = 0
|
||||
while (start := text.find("<", pos)) >= 0:
|
||||
end = text.find(">", start + 1)
|
||||
if end < 0:
|
||||
return
|
||||
if end > start + 1:
|
||||
yield start, end + 1, text[start + 1:end]
|
||||
pos = end + 1
|
||||
else:
|
||||
pos = start + 1
|
||||
|
||||
|
||||
def iter_markdown_links(
|
||||
text: str,
|
||||
*,
|
||||
target_prefix: str = "",
|
||||
target_re: Pattern[str] | None = None,
|
||||
) -> Iterator[tuple[int, int, str, str]]:
|
||||
"""Yield flat Markdown links accepted by the legacy link regexes."""
|
||||
pos = 0
|
||||
marker = "](" + target_prefix
|
||||
while (start := text.find("[", pos)) >= 0:
|
||||
label_end = text.find("]", start + 1)
|
||||
if label_end < 0:
|
||||
return
|
||||
if label_end == start + 1:
|
||||
pos = start + 1
|
||||
continue
|
||||
if not text.startswith(marker, label_end):
|
||||
pos = label_end + 1
|
||||
continue
|
||||
target_start = label_end + len(marker)
|
||||
target_end = text.find(")", target_start)
|
||||
if target_end < 0:
|
||||
return
|
||||
target = text[target_start:target_end]
|
||||
if target and (target_re is None or target_re.fullmatch(target)):
|
||||
yield start, target_end + 1, text[start + 1:label_end], target
|
||||
pos = target_end + 1
|
||||
else:
|
||||
pos = label_end + 1
|
||||
|
||||
|
||||
def replace_markdown_links_with_labels(text: str) -> str:
|
||||
"""Linear equivalent of replacing flat Markdown links with their labels."""
|
||||
links = list(iter_markdown_links(text))
|
||||
if not links:
|
||||
return text
|
||||
out = []
|
||||
pos = 0
|
||||
for start, end, label, _target in links:
|
||||
out.extend((text[pos:start], label))
|
||||
pos = end
|
||||
out.append(text[pos:])
|
||||
return "".join(out)
|
||||
|
||||
|
||||
def first_tag_content(text: str, tag: str, *, allow_attributes: bool = False) -> str | None:
|
||||
"""Return the first flat tag body using forward-only opener/closer scans."""
|
||||
opener_re = re.compile(r"<" + re.escape(tag), re.IGNORECASE)
|
||||
closer_re = re.compile(r"</" + re.escape(tag) + r">", re.IGNORECASE)
|
||||
pos = 0
|
||||
while opener := opener_re.search(text, pos):
|
||||
name_end = opener.end()
|
||||
if name_end < len(text) and text[name_end] == ">":
|
||||
body_start = name_end + 1
|
||||
elif allow_attributes and name_end < len(text) and text[name_end].isspace():
|
||||
tag_end = text.find(">", name_end + 1)
|
||||
if tag_end < 0:
|
||||
return None
|
||||
body_start = tag_end + 1
|
||||
else:
|
||||
pos = name_end
|
||||
continue
|
||||
closer = closer_re.search(text, body_start)
|
||||
if closer is None:
|
||||
return None
|
||||
return text[body_start:closer.start()]
|
||||
return None
|
||||
Reference in New Issue
Block a user