mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-08 07:52:20 +02:00
287 lines
11 KiB
Python
287 lines
11 KiB
Python
"""Forward-only helpers for permissive text grammars.
|
|
|
|
These helpers retain the existing regular expressions as anchored token
|
|
parsers while preventing ``re.search`` from retrying the same token suffix at
|
|
every embedded prefix.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from collections.abc import Iterator
|
|
from typing import Match, Pattern
|
|
|
|
|
|
def iter_prefixed_token_matches(
|
|
text: str,
|
|
candidate_re: Pattern[str],
|
|
anchored_re: Pattern[str],
|
|
token_tail_re: Pattern[str],
|
|
) -> Iterator[Match[str]]:
|
|
"""Yield legacy greedy matches after testing one prefix per token.
|
|
|
|
``anchored_re`` must start with the same fixed prefix recognized by
|
|
``candidate_re``. ``token_tail_re`` describes characters that the
|
|
anchored grammar can consume after that prefix. If the first prefix in
|
|
such a token cannot match, a later embedded prefix cannot match either:
|
|
its suffix was already available to the first attempt. Advancing to the
|
|
token boundary makes failed scans linear without changing successful
|
|
greedy captures.
|
|
"""
|
|
pos = 0
|
|
while candidate := candidate_re.search(text, pos):
|
|
match = anchored_re.match(text, candidate.start())
|
|
if match is not None:
|
|
yield match
|
|
pos = match.end()
|
|
continue
|
|
tail = token_tail_re.match(text, candidate.end())
|
|
pos = max(candidate.end(), tail.end() if tail is not None else candidate.end())
|
|
|
|
|
|
def has_prefixed_token_match(
|
|
text: str,
|
|
candidate_re: Pattern[str],
|
|
anchored_re: Pattern[str],
|
|
token_tail_re: Pattern[str],
|
|
) -> bool:
|
|
"""Return whether ``iter_prefixed_token_matches`` yields a match."""
|
|
return next(
|
|
iter_prefixed_token_matches(text, candidate_re, anchored_re, token_tail_re),
|
|
None,
|
|
) is not None
|
|
|
|
|
|
_HTTP_URL_PREFIX_RE = re.compile(r"https?://", re.IGNORECASE)
|
|
_VIDEO_DETAIL_BASE_RE = re.compile(
|
|
r"\b(?:how\s+many|count|break\s*points?|timestamps?|what\s+time|score(?:board)?s?)\b",
|
|
re.IGNORECASE,
|
|
)
|
|
_VIDEO_DETAIL_EXTENDED_RE = re.compile(
|
|
r"\b(?:sequence|in\s+order|chronological|at\s+what\s+time)\b|"
|
|
r"(?:多少|几次|何时|什么时候|时间|顺序)",
|
|
re.IGNORECASE,
|
|
)
|
|
_WHEN_RE = re.compile(r"\bwhen\s+", re.IGNORECASE)
|
|
_WHEN_TARGET_RE = re.compile(r"(?:end|happen)\b", re.IGNORECASE)
|
|
_FIRST_RE = re.compile(r"\bfirst\s+", re.IGNORECASE)
|
|
_FIRST_TARGET_RE = re.compile(r"(?:save|attempt|event)\b", re.IGNORECASE)
|
|
|
|
|
|
def contains_search_engine_navigation(text: str) -> bool:
|
|
"""Match the legacy Google/Bing/DuckDuckGo navigation URL grammar.
|
|
|
|
The old expression backtracked through every possible optional subdomain
|
|
split. Parsing the host up to its first slash gives the same accepted
|
|
hosts and path prefixes with one pass per URL candidate.
|
|
"""
|
|
pos = 0
|
|
while candidate := _HTTP_URL_PREFIX_RE.search(text, pos):
|
|
host_start = candidate.end()
|
|
path_start = text.find("/", host_start)
|
|
if path_start < 0:
|
|
return False
|
|
host = text[host_start:path_start].casefold()
|
|
path = text[path_start + 1:path_start + 7].casefold()
|
|
google_at = host.rfind(".google.")
|
|
recognized_host = (
|
|
(host.startswith("google.") and len(host) > len("google."))
|
|
or (google_at >= 0 and google_at + len(".google.") < len(host))
|
|
or host == "bing.com"
|
|
or host.endswith(".bing.com")
|
|
or host == "duckduckgo.com"
|
|
or host.endswith(".duckduckgo.com")
|
|
)
|
|
if recognized_host and path.startswith(("search", "sorry", "html", "lite", "?")):
|
|
return True
|
|
# A later URL may begin in the path. Resume after this scheme rather
|
|
# than skipping the whole non-whitespace region.
|
|
pos = candidate.end()
|
|
return False
|
|
|
|
|
|
def contains_detailed_sequence_request(text: str, *, include_first: bool = True) -> bool:
|
|
"""Recognize count/order/timing requests without overlapping ``.*`` scans."""
|
|
value = str(text or "")
|
|
if _VIDEO_DETAIL_BASE_RE.search(value):
|
|
return True
|
|
if include_first and _VIDEO_DETAIL_EXTENDED_RE.search(value):
|
|
return True
|
|
pos = 0
|
|
while when := _WHEN_RE.search(value, pos):
|
|
line_end = value.find("\n", when.end())
|
|
if line_end < 0:
|
|
line_end = len(value)
|
|
if _WHEN_TARGET_RE.search(value, when.end(), line_end) is not None:
|
|
return True
|
|
pos = line_end + 1
|
|
if include_first:
|
|
pos = 0
|
|
while first := _FIRST_RE.search(value, pos):
|
|
line_end = value.find("\n", first.end())
|
|
if line_end < 0:
|
|
line_end = len(value)
|
|
if _FIRST_TARGET_RE.search(value, first.end(), line_end) is not None:
|
|
return True
|
|
pos = line_end + 1
|
|
return False
|
|
|
|
|
|
def iter_angle_contents(text: str) -> Iterator[tuple[int, int, str]]:
|
|
"""Yield nonempty flat ``<...>`` contents with monotonic delimiters."""
|
|
pos = 0
|
|
while (start := text.find("<", pos)) >= 0:
|
|
end = text.find(">", start + 1)
|
|
if end < 0:
|
|
return
|
|
if end > start + 1:
|
|
yield start, end + 1, text[start + 1:end]
|
|
pos = end + 1
|
|
else:
|
|
pos = start + 1
|
|
|
|
|
|
def iter_markdown_links(
|
|
text: str,
|
|
*,
|
|
target_prefix: str = "",
|
|
target_re: Pattern[str] | None = None,
|
|
) -> Iterator[tuple[int, int, str, str]]:
|
|
"""Yield flat Markdown links accepted by the legacy link regexes."""
|
|
pos = 0
|
|
marker = "](" + target_prefix
|
|
while (start := text.find("[", pos)) >= 0:
|
|
label_end = text.find("]", start + 1)
|
|
if label_end < 0:
|
|
return
|
|
if label_end == start + 1:
|
|
pos = start + 1
|
|
continue
|
|
if not text.startswith(marker, label_end):
|
|
pos = label_end + 1
|
|
continue
|
|
target_start = label_end + len(marker)
|
|
target_end = text.find(")", target_start)
|
|
if target_end < 0:
|
|
return
|
|
target = text[target_start:target_end]
|
|
if target and (target_re is None or target_re.fullmatch(target)):
|
|
yield start, target_end + 1, text[start + 1:label_end], target
|
|
pos = target_end + 1
|
|
else:
|
|
pos = label_end + 1
|
|
|
|
|
|
def replace_markdown_links_with_labels(text: str) -> str:
|
|
"""Linear equivalent of replacing flat Markdown links with their labels."""
|
|
links = list(iter_markdown_links(text))
|
|
if not links:
|
|
return text
|
|
out = []
|
|
pos = 0
|
|
for start, end, label, _target in links:
|
|
out.extend((text[pos:start], label))
|
|
pos = end
|
|
out.append(text[pos:])
|
|
return "".join(out)
|
|
|
|
|
|
def first_tag_content(text: str, tag: str, *, allow_attributes: bool = False) -> str | None:
|
|
"""Return the first flat tag body using forward-only opener/closer scans."""
|
|
opener_re = re.compile(r"<" + re.escape(tag), re.IGNORECASE)
|
|
closer_re = re.compile(r"</" + re.escape(tag) + r">", re.IGNORECASE)
|
|
pos = 0
|
|
while opener := opener_re.search(text, pos):
|
|
name_end = opener.end()
|
|
if name_end < len(text) and text[name_end] == ">":
|
|
body_start = name_end + 1
|
|
elif allow_attributes and name_end < len(text) and text[name_end].isspace():
|
|
tag_end = text.find(">", name_end + 1)
|
|
if tag_end < 0:
|
|
return None
|
|
body_start = tag_end + 1
|
|
else:
|
|
pos = name_end
|
|
continue
|
|
closer = closer_re.search(text, body_start)
|
|
if closer is None:
|
|
return None
|
|
return text[body_start:closer.start()]
|
|
return None
|
|
|
|
|
|
def space_delimited_fields(text, leaders, separator_re, tail):
|
|
"""Match a lazy dot field after a finite set of greedy leading grammars.
|
|
|
|
Separators consume a maximal whitespace run (group 1) and a fixed grammar.
|
|
Test each run once, rather than repartitioning it between a leading space
|
|
quantifier, a dot capture, and the separator. ``tail`` must also scan
|
|
monotonically or use only a fixed/bounded grammar.
|
|
"""
|
|
separators = list(separator_re.finditer(text))
|
|
for leader_re, minimum_space in leaders:
|
|
leader = leader_re.match(text)
|
|
if leader is None:
|
|
continue
|
|
start = leader.end()
|
|
newline = text.find("\n", start)
|
|
line_end = len(text) if newline < 0 else newline
|
|
for separator in separators:
|
|
end = separator.start(1)
|
|
if start < end <= line_end:
|
|
remainder = tail(separator)
|
|
if remainder is not None:
|
|
return (text[start:end], *remainder)
|
|
# A greedy leading run may give back a dot character when the field
|
|
# consists entirely of whitespace. Only the last non-LF character
|
|
# that leaves the mandatory separator can win; do not retry suffixes.
|
|
run_start = start
|
|
while run_start and text[run_start - 1].isspace():
|
|
run_start -= 1
|
|
earliest = run_start + minimum_space
|
|
for separator in separators:
|
|
if separator.end(1) != start:
|
|
continue
|
|
end = start - (1 if separator.group(1) else 0)
|
|
candidate = end - 1
|
|
while candidate >= earliest and text[candidate] == "\n":
|
|
candidate -= 1
|
|
if candidate >= earliest:
|
|
remainder = tail(separator)
|
|
if remainder is not None:
|
|
return (text[candidate:candidate + 1], *remainder)
|
|
return None
|
|
|
|
|
|
def terminal_dot_field(text, start, *, punctuation=False, last_newline=None):
|
|
"""Read a whitespace-led dot field ending at Python's dollar boundary."""
|
|
if start >= len(text) or not text[start].isspace():
|
|
return None
|
|
cursor = start
|
|
while cursor < len(text) and text[cursor].isspace():
|
|
cursor += 1
|
|
end = len(text) - (1 if text.endswith("\n") else 0)
|
|
if cursor >= end:
|
|
cursor = end - 1
|
|
while cursor > start and text[cursor] == "\n":
|
|
cursor -= 1
|
|
if cursor <= start or text[cursor] == "\n":
|
|
return None
|
|
# Newlines before the field can be leading whitespace; newlines inside
|
|
# the dot capture cannot be consumed. The last LF is a constant-time veto.
|
|
if last_newline is None:
|
|
last_newline = text.rfind("\n", 0, end)
|
|
if last_newline >= cursor:
|
|
return None
|
|
capture_end = end
|
|
if punctuation:
|
|
while capture_end > cursor and text[capture_end - 1].isspace():
|
|
capture_end -= 1
|
|
if capture_end > cursor and text[capture_end - 1] in ".!?":
|
|
capture_end -= 1
|
|
else:
|
|
capture_end = end
|
|
if capture_end == cursor:
|
|
capture_end = end
|
|
return (text[cursor:capture_end],)
|