mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 23:12:22 +02:00
Squash Odysseus development history
This commit is contained in:
+101
-6
@@ -5,6 +5,8 @@ import os
|
||||
import logging
|
||||
import mimetypes
|
||||
import base64
|
||||
import shutil
|
||||
import subprocess
|
||||
import tempfile
|
||||
from typing import List, Dict, Any
|
||||
|
||||
@@ -18,10 +20,12 @@ MIN_INLINE_ATTACHMENT_SLICE = 500
|
||||
|
||||
def _is_text_file(path: str) -> bool:
|
||||
"""Check if file has text extension."""
|
||||
return any(
|
||||
path.lower().endswith(ext)
|
||||
for ext in (".txt", ".py", ".html", ".htm", ".md", ".json", ".csv", ".log", ".js", ".nix")
|
||||
)
|
||||
return os.path.splitext(path.lower())[1] in {
|
||||
".bash", ".c", ".cpp", ".css", ".csv", ".go", ".h", ".htm",
|
||||
".html", ".java", ".js", ".json", ".jsx", ".log", ".md",
|
||||
".markdown", ".nix", ".php", ".py", ".rb", ".rs", ".sh",
|
||||
".sql", ".ts", ".tsx", ".txt", ".xml", ".yaml", ".yml",
|
||||
}
|
||||
|
||||
|
||||
def _process_text_file(path: str) -> str:
|
||||
@@ -109,7 +113,12 @@ def _process_text_file(path: str) -> str:
|
||||
return result
|
||||
|
||||
|
||||
def _process_pdf(path: str, owner: str | None = None) -> str:
|
||||
def _process_pdf(
|
||||
path: str,
|
||||
owner: str | None = None,
|
||||
*,
|
||||
analyze_embedded_images: bool = True,
|
||||
) -> str:
|
||||
"""Process PDF file with text extraction (pypdf). Uses VL model for image-heavy pages."""
|
||||
try:
|
||||
from pypdf import PdfReader
|
||||
@@ -126,7 +135,7 @@ def _process_pdf(path: str, owner: str | None = None) -> str:
|
||||
images = list(page.images)
|
||||
except Exception:
|
||||
images = []
|
||||
if images and len(page_text) < 50:
|
||||
if analyze_embedded_images and images and len(page_text) < 50:
|
||||
for img_index, img in enumerate(images[:3]): # cap at 3 images per page
|
||||
try:
|
||||
with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as tmp:
|
||||
@@ -278,6 +287,90 @@ def _process_office_document(
|
||||
return f"\n\n[Attached document: {display_name} — {exc}]"
|
||||
|
||||
|
||||
def _process_legacy_word_document(path: str, display_name: str) -> str:
|
||||
"""Extract readable text from an old binary Word ``.doc`` file."""
|
||||
commands: list[tuple[str, list[str]]] = []
|
||||
if shutil.which("antiword"):
|
||||
commands.append(("antiword", ["antiword", path]))
|
||||
if shutil.which("catdoc"):
|
||||
commands.append(("catdoc", ["catdoc", path]))
|
||||
if shutil.which("strings"):
|
||||
commands.extend((
|
||||
("strings", ["strings", "-n", "4", path]),
|
||||
("strings (UTF-16LE)", ["strings", "-e", "l", "-n", "4", path]),
|
||||
))
|
||||
|
||||
collected: list[str] = []
|
||||
seen: set[str] = set()
|
||||
used: list[str] = []
|
||||
for label, command in commands:
|
||||
try:
|
||||
result = subprocess.run(
|
||||
command,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
errors="replace",
|
||||
timeout=20,
|
||||
check=False,
|
||||
)
|
||||
except (OSError, subprocess.SubprocessError) as exc:
|
||||
logger.warning("Legacy Word extraction via %s failed for %s: %s", label, path, exc)
|
||||
continue
|
||||
text = (result.stdout or "").strip()
|
||||
if not text:
|
||||
continue
|
||||
used.append(label)
|
||||
for line in text.splitlines():
|
||||
line = line.strip()
|
||||
if line and line not in seen:
|
||||
seen.add(line)
|
||||
collected.append(line)
|
||||
if label in {"antiword", "catdoc"} and collected:
|
||||
break
|
||||
|
||||
title = os.path.splitext(os.path.basename(display_name or path))[0]
|
||||
body, marker = _truncate_inline("\n".join(collected))
|
||||
if body:
|
||||
method = used[0] if used else "best-effort extraction"
|
||||
return (
|
||||
f"\n\n[Legacy Word content — {title}; formatting omitted; "
|
||||
f"extracted with {method}]:\n{body}{marker}"
|
||||
)
|
||||
return (
|
||||
f"\n\n[Attached legacy Word document: {display_name} — no readable text "
|
||||
"could be extracted. Install antiword or LibreOffice for fuller support.]"
|
||||
)
|
||||
|
||||
|
||||
def extract_local_document(
|
||||
path: str,
|
||||
*,
|
||||
display_name: str | None = None,
|
||||
owner: str | None = None,
|
||||
analyze_embedded_images: bool = False,
|
||||
) -> str:
|
||||
"""Extract a local document into bounded model-readable text.
|
||||
|
||||
This side-effect-free entry point is shared by non-UI runtimes. It avoids
|
||||
creating session documents and defaults to text-only PDF extraction so a
|
||||
background task bridge cannot make an unexpected vision-model call.
|
||||
"""
|
||||
|
||||
name = display_name or os.path.basename(path)
|
||||
mime = mimetypes.guess_type(name)[0] or "application/octet-stream"
|
||||
if path.lower().endswith(".doc") or name.lower().endswith(".doc"):
|
||||
return _process_legacy_word_document(path, name)
|
||||
if mime == "application/pdf" or path.lower().endswith(".pdf"):
|
||||
return _process_pdf(
|
||||
path,
|
||||
owner=owner,
|
||||
analyze_embedded_images=analyze_embedded_images,
|
||||
)
|
||||
if mime.startswith("text/") or _is_text_file(path):
|
||||
return _process_text_file(path)
|
||||
return _process_office_document(path, name, owner=owner)
|
||||
|
||||
|
||||
# Marker that _process_pdf prepends to extracted text.
|
||||
_PDF_CONTENT_MARKER = "\n\n[PDF content]:"
|
||||
|
||||
@@ -570,6 +663,8 @@ def build_user_content(
|
||||
logger.warning(f"PDF auto-doc creation failed for {path}: {e}")
|
||||
if extracted_text is None:
|
||||
extracted_text = _process_pdf(path, owner=owner)
|
||||
elif path.lower().endswith(".doc") or display_name.lower().endswith(".doc"):
|
||||
extracted_text = _process_legacy_word_document(path, display_name)
|
||||
elif mime.startswith("text/") or _is_text_file(path):
|
||||
extracted_text = _process_text_file(path)
|
||||
else:
|
||||
|
||||
Reference in New Issue
Block a user