"""Regression tests for the residual py/polynomial-redos fixes (PR #6503 lane).
Each rewritten matcher is pinned two ways:
* equivalence with the regex it replaced on ordinary inputs, and
* the CodeQL-reported adversarial shape completing within a loose budget
(the replaced regexes took seconds to tens of seconds on these sizes).
"""
import re
import time
from pathlib import Path
import pytest
from src.text_helpers import strip_closed_think_blocks
_BUDGET_S = 2.0
_ROOT = Path(__file__).resolve().parents[1]
def _fast(fn, *args):
started = time.perf_counter()
result = fn(*args)
elapsed = time.perf_counter() - started
assert elapsed < _BUDGET_S, f"{getattr(fn, '__name__', fn)} took {elapsed:.2f}s"
return result
# -- closed blocks (skills_routes x4, memory_extractor.audit_memories) --
_THINK_REF = re.compile(r"[\s\S]*?", re.I)
@pytest.mark.parametrize("text", [
"",
"plain",
"x{\"ok\": true}",
"abcde",
"anestedrest",
"orphan then unclosed",
"casekept",
"\nmulti\nline\n\n{}",
])
def test_strip_closed_think_blocks_matches_reference(text):
assert strip_closed_think_blocks(text) == _THINK_REF.sub("", text)
def test_strip_closed_think_blocks_unclosed_opener_flood_is_linear():
flood = "" * 40_000
assert _fast(strip_closed_think_blocks, flood) == flood
def test_lazy_think_regex_is_gone_from_llm_review_parsers():
for rel in ("routes/skills_routes.py", "services/memory/memory_extractor.py"):
source = (_ROOT / rel).read_text(encoding="utf-8")
assert r"[\s\S]*?" not in source, rel
# -- email compose markdown links (email_routes._md_to_email_html) -----------
_LINK_REF = re.compile(r"\[([^\]]+)\]\((https?://[^)\s]+)\)")
@pytest.mark.parametrize("text", [
"see [docs](https://example.com/a) and [b](http://x.y)",
"[a[b](http://x)",
"[]()[x](http://y)",
"[a](http://x [b c](http://y)",
"[a](http://)[b](https://z)",
"[a](ftp://x) [b](https://ok)",
"[unterminated(http://x)",
"[t](http://u) tail ] [",
])
def test_md_links_to_html_matches_reference(text):
from routes.email.email_routes import _md_links_to_html
assert _md_links_to_html(text) == _LINK_REF.sub(r'\1', text)
@pytest.mark.parametrize("flood", [
"[" + "[\\" * 40_000, # CodeQL's reported shape
"[a](http://x" * 8_000, # URL that never closes, repeated
])
def test_md_links_to_html_floods_are_linear(flood):
from routes.email.email_routes import _md_links_to_html
assert _fast(_md_links_to_html, flood) == flood
def test_md_to_email_html_still_renders_and_escapes_links():
from routes.email.email_routes import _md_to_email_html
html = _md_to_email_html("**hi** [site](https://example.com/x)\n