Unique substantive evidence.
This is the substantive body text that should be retained.
It is much longer than the tiny class-matched wrapper.
"""Content extraction behavior for the canonical services.search.content module.""" import httpx import pytest pytest.importorskip("bs4") from services.search import content as service_content class _FakeResponse: status_code = 200 headers = {"Content-Type": "text/html; charset=utf-8"} content = b"" def __init__(self, text: str): self.text = text def raise_for_status(self): return None class _FakeErrorResponse: """Mimics an httpx.Response that fails raise_for_status with a given status code.""" headers = {"Content-Type": "text/html; charset=utf-8"} content = b"" text = "" def __init__(self, status_code: int): self.status_code = status_code def raise_for_status(self): raise httpx.HTTPStatusError( f"{self.status_code} error", request=None, response=self ) @pytest.mark.parametrize('wrapper', ['main', 'article', 'div class="content"']) def test_extraction_does_not_repeat_nested_content_or_include_navigation(wrapper, tmp_path, monkeypatch): closing = wrapper.split()[0] html = (f'
<{wrapper}>' 'Unique substantive evidence.
This is the substantive body text that should be retained.
It is much longer than the tiny class-matched wrapper.