mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 15:02:20 +02:00
Snapshot current maintainer-preview application changes and regression fixtures for integration into lab. Excludes local runtime data, evaluation outputs and source backups. Focused Python regression selection: 140 passed; full suite not certified.
57 lines
2.9 KiB
Python
57 lines
2.9 KiB
Python
"""Bounded local previews of already-authorized email attachments."""
|
|
from pathlib import Path
|
|
|
|
|
|
def attachment_text(path, *, max_chars=12000):
|
|
path = Path(path)
|
|
if path.stat().st_size > 20 * 1024 * 1024:
|
|
return {'content_status': 'too_large', 'content_note': 'Attachment exceeds the 20 MB reading limit.'}
|
|
suffix = path.suffix.lower()
|
|
parts = []
|
|
truncated = False
|
|
try:
|
|
if suffix == '.pdf':
|
|
from pypdf import PdfReader
|
|
reader = PdfReader(path)
|
|
if reader.is_encrypted and not reader.decrypt(''):
|
|
return {'content_status': 'encrypted', 'content_note': 'PDF requires a password.'}
|
|
for number, page in enumerate(reader.pages):
|
|
if number >= 50 or sum(map(len, parts)) >= max_chars:
|
|
truncated = True
|
|
break
|
|
parts.append(f'Page {number + 1}:\n' + (page.extract_text() or ''))
|
|
if not any(part.split(':\n', 1)[-1].strip() for part in parts):
|
|
return {'content_status': 'needs_ocr', 'content_note': 'No embedded PDF text. Scanned pages require OCR; contents have not been read.'}
|
|
elif suffix in {'.txt', '.md', '.csv', '.tsv', '.json', '.xml', '.log'}:
|
|
with path.open(encoding='utf-8', errors='replace') as file:
|
|
parts.append(file.read(max_chars + 1))
|
|
elif suffix == '.docx':
|
|
from docx import Document
|
|
doc = Document(path)
|
|
parts.extend(p.text for p in doc.paragraphs)
|
|
for table in doc.tables:
|
|
parts.extend('\t'.join(cell.text for cell in row.cells) for row in table.rows)
|
|
elif suffix == '.xlsx':
|
|
from openpyxl import load_workbook
|
|
book = load_workbook(path, read_only=True, data_only=True, keep_links=False)
|
|
try:
|
|
for sheet in book:
|
|
parts.append(f'Sheet: {sheet.title}')
|
|
for index, row in enumerate(sheet.iter_rows(values_only=True)):
|
|
if index >= 1000 or sum(map(len, parts)) >= max_chars:
|
|
truncated = True
|
|
break
|
|
parts.append('\t'.join('' if cell is None else str(cell) for cell in row))
|
|
if truncated:
|
|
break
|
|
finally:
|
|
book.close()
|
|
else:
|
|
return {'content_status': 'unsupported', 'content_note': 'This attachment format has no inline text reader.'}
|
|
text = '\n'.join(parts).strip()
|
|
truncated |= len(text) > max_chars
|
|
return {'content': text[:max_chars], 'content_status': 'read' if text else 'empty',
|
|
'content_note': 'Preview truncated; remaining content was not read.' if truncated else ''}
|
|
except Exception as exc:
|
|
return {'content_status': 'failed', 'content_note': f'Attachment text extraction failed ({type(exc).__name__}); contents have not been read.'}
|