Files
odysseus/src/email_attachment_text.py
T
pewdiepie-archdaemon 2e8413a54a Preserve preview harness, editor, email and task improvements
Snapshot current maintainer-preview application changes and regression fixtures for integration into lab. Excludes local runtime data, evaluation outputs and source backups. Focused Python regression selection: 140 passed; full suite not certified.
2026-10-01 01:34:26 +00:00

57 lines
2.9 KiB
Python

"""Bounded local previews of already-authorized email attachments."""
from pathlib import Path
def attachment_text(path, *, max_chars=12000):
path = Path(path)
if path.stat().st_size > 20 * 1024 * 1024:
return {'content_status': 'too_large', 'content_note': 'Attachment exceeds the 20 MB reading limit.'}
suffix = path.suffix.lower()
parts = []
truncated = False
try:
if suffix == '.pdf':
from pypdf import PdfReader
reader = PdfReader(path)
if reader.is_encrypted and not reader.decrypt(''):
return {'content_status': 'encrypted', 'content_note': 'PDF requires a password.'}
for number, page in enumerate(reader.pages):
if number >= 50 or sum(map(len, parts)) >= max_chars:
truncated = True
break
parts.append(f'Page {number + 1}:\n' + (page.extract_text() or ''))
if not any(part.split(':\n', 1)[-1].strip() for part in parts):
return {'content_status': 'needs_ocr', 'content_note': 'No embedded PDF text. Scanned pages require OCR; contents have not been read.'}
elif suffix in {'.txt', '.md', '.csv', '.tsv', '.json', '.xml', '.log'}:
with path.open(encoding='utf-8', errors='replace') as file:
parts.append(file.read(max_chars + 1))
elif suffix == '.docx':
from docx import Document
doc = Document(path)
parts.extend(p.text for p in doc.paragraphs)
for table in doc.tables:
parts.extend('\t'.join(cell.text for cell in row.cells) for row in table.rows)
elif suffix == '.xlsx':
from openpyxl import load_workbook
book = load_workbook(path, read_only=True, data_only=True, keep_links=False)
try:
for sheet in book:
parts.append(f'Sheet: {sheet.title}')
for index, row in enumerate(sheet.iter_rows(values_only=True)):
if index >= 1000 or sum(map(len, parts)) >= max_chars:
truncated = True
break
parts.append('\t'.join('' if cell is None else str(cell) for cell in row))
if truncated:
break
finally:
book.close()
else:
return {'content_status': 'unsupported', 'content_note': 'This attachment format has no inline text reader.'}
text = '\n'.join(parts).strip()
truncated |= len(text) > max_chars
return {'content': text[:max_chars], 'content_status': 'read' if text else 'empty',
'content_note': 'Preview truncated; remaining content was not read.' if truncated else ''}
except Exception as exc:
return {'content_status': 'failed', 'content_note': f'Attachment text extraction failed ({type(exc).__name__}); contents have not been read.'}