Squash Odysseus development history

This commit is contained in:
pewdiepie-archdaemon
2026-09-11 06:04:19 +00:00
parent c9dd68d890
commit 84aa9a91de
871 changed files with 265870 additions and 27854 deletions
+101 -6
View File
@@ -5,6 +5,8 @@ import os
import logging
import mimetypes
import base64
import shutil
import subprocess
import tempfile
from typing import List, Dict, Any
@@ -18,10 +20,12 @@ MIN_INLINE_ATTACHMENT_SLICE = 500
def _is_text_file(path: str) -> bool:
"""Check if file has text extension."""
return any(
path.lower().endswith(ext)
for ext in (".txt", ".py", ".html", ".htm", ".md", ".json", ".csv", ".log", ".js", ".nix")
)
return os.path.splitext(path.lower())[1] in {
".bash", ".c", ".cpp", ".css", ".csv", ".go", ".h", ".htm",
".html", ".java", ".js", ".json", ".jsx", ".log", ".md",
".markdown", ".nix", ".php", ".py", ".rb", ".rs", ".sh",
".sql", ".ts", ".tsx", ".txt", ".xml", ".yaml", ".yml",
}
def _process_text_file(path: str) -> str:
@@ -109,7 +113,12 @@ def _process_text_file(path: str) -> str:
return result
def _process_pdf(path: str, owner: str | None = None) -> str:
def _process_pdf(
path: str,
owner: str | None = None,
*,
analyze_embedded_images: bool = True,
) -> str:
"""Process PDF file with text extraction (pypdf). Uses VL model for image-heavy pages."""
try:
from pypdf import PdfReader
@@ -126,7 +135,7 @@ def _process_pdf(path: str, owner: str | None = None) -> str:
images = list(page.images)
except Exception:
images = []
if images and len(page_text) < 50:
if analyze_embedded_images and images and len(page_text) < 50:
for img_index, img in enumerate(images[:3]): # cap at 3 images per page
try:
with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as tmp:
@@ -278,6 +287,90 @@ def _process_office_document(
return f"\n\n[Attached document: {display_name} — {exc}]"
def _process_legacy_word_document(path: str, display_name: str) -> str:
"""Extract readable text from an old binary Word ``.doc`` file."""
commands: list[tuple[str, list[str]]] = []
if shutil.which("antiword"):
commands.append(("antiword", ["antiword", path]))
if shutil.which("catdoc"):
commands.append(("catdoc", ["catdoc", path]))
if shutil.which("strings"):
commands.extend((
("strings", ["strings", "-n", "4", path]),
("strings (UTF-16LE)", ["strings", "-e", "l", "-n", "4", path]),
))
collected: list[str] = []
seen: set[str] = set()
used: list[str] = []
for label, command in commands:
try:
result = subprocess.run(
command,
capture_output=True,
text=True,
errors="replace",
timeout=20,
check=False,
)
except (OSError, subprocess.SubprocessError) as exc:
logger.warning("Legacy Word extraction via %s failed for %s: %s", label, path, exc)
continue
text = (result.stdout or "").strip()
if not text:
continue
used.append(label)
for line in text.splitlines():
line = line.strip()
if line and line not in seen:
seen.add(line)
collected.append(line)
if label in {"antiword", "catdoc"} and collected:
break
title = os.path.splitext(os.path.basename(display_name or path))[0]
body, marker = _truncate_inline("\n".join(collected))
if body:
method = used[0] if used else "best-effort extraction"
return (
f"\n\n[Legacy Word content — {title}; formatting omitted; "
f"extracted with {method}]:\n{body}{marker}"
)
return (
f"\n\n[Attached legacy Word document: {display_name} — no readable text "
"could be extracted. Install antiword or LibreOffice for fuller support.]"
)
def extract_local_document(
path: str,
*,
display_name: str | None = None,
owner: str | None = None,
analyze_embedded_images: bool = False,
) -> str:
"""Extract a local document into bounded model-readable text.
This side-effect-free entry point is shared by non-UI runtimes. It avoids
creating session documents and defaults to text-only PDF extraction so a
background task bridge cannot make an unexpected vision-model call.
"""
name = display_name or os.path.basename(path)
mime = mimetypes.guess_type(name)[0] or "application/octet-stream"
if path.lower().endswith(".doc") or name.lower().endswith(".doc"):
return _process_legacy_word_document(path, name)
if mime == "application/pdf" or path.lower().endswith(".pdf"):
return _process_pdf(
path,
owner=owner,
analyze_embedded_images=analyze_embedded_images,
)
if mime.startswith("text/") or _is_text_file(path):
return _process_text_file(path)
return _process_office_document(path, name, owner=owner)
# Marker that _process_pdf prepends to extracted text.
_PDF_CONTENT_MARKER = "\n\n[PDF content]:"
@@ -570,6 +663,8 @@ def build_user_content(
logger.warning(f"PDF auto-doc creation failed for {path}: {e}")
if extracted_text is None:
extracted_text = _process_pdf(path, owner=owner)
elif path.lower().endswith(".doc") or display_name.lower().endswith(".doc"):
extracted_text = _process_legacy_word_document(path, display_name)
elif mime.startswith("text/") or _is_text_file(path):
extracted_text = _process_text_file(path)
else: