mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-10 00:42:20 +02:00
Squash Odysseus development history
This commit is contained in:
@@ -5,6 +5,7 @@ The old call sites used ``str.lstrip("\\n[PDF content]:")``, which treats the
|
||||
argument as a *set of characters* and keeps stripping leading characters that
|
||||
happen to be in that set — corrupting the start of the extracted document.
|
||||
"""
|
||||
from src import document_processor
|
||||
from src.document_processor import strip_pdf_content_marker, _PDF_CONTENT_MARKER
|
||||
|
||||
|
||||
@@ -28,3 +29,28 @@ def test_text_without_marker_is_only_stripped():
|
||||
|
||||
def test_handles_none():
|
||||
assert strip_pdf_content_marker(None) == ""
|
||||
|
||||
|
||||
def test_local_document_extraction_disables_embedded_image_analysis(monkeypatch, tmp_path):
|
||||
source = tmp_path / "brief.pdf"
|
||||
source.write_bytes(b"%PDF-1.4")
|
||||
observed = {}
|
||||
|
||||
def fake_process(path, owner=None, *, analyze_embedded_images=True):
|
||||
observed.update(
|
||||
path=path,
|
||||
owner=owner,
|
||||
analyze_embedded_images=analyze_embedded_images,
|
||||
)
|
||||
return "extracted"
|
||||
|
||||
monkeypatch.setattr(document_processor, "_process_pdf", fake_process)
|
||||
|
||||
result = document_processor.extract_local_document(str(source))
|
||||
|
||||
assert result == "extracted"
|
||||
assert observed == {
|
||||
"path": str(source),
|
||||
"owner": None,
|
||||
"analyze_embedded_images": False,
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user