diff --git a/CHANGELOG.md b/CHANGELOG.md index 2393725..278ec33 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,16 @@ All notable changes to Library Desk will be documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [1.4.6] - 2025-12-25 + +### Fixed + +- **Paperless Webhook Payload Format** - Updated model to match Paperless `include_document=true` format + - Paperless sends `id` instead of `document_id` + - Paperless sends full document data including `content`, `title`, `tags`, etc. + - Webhook now uses content from payload, skipping extra Paperless API call + - Added `extra = "ignore"` to handle additional Paperless fields + ## [1.4.5] - 2025-12-25 ### Added diff --git a/pyproject.toml b/pyproject.toml index bc0c42d..20adf33 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "library-desk" -version = "1.4.5" +version = "1.4.6" description = "Coordination service for The Library system - HybridRAG queries, document ingestion, entity extraction, and knowledge consolidation" readme = "README.md" requires-python = ">=3.12" diff --git a/src/models/document.py b/src/models/document.py index dcdbc33..23816e7 100644 --- a/src/models/document.py +++ b/src/models/document.py @@ -86,13 +86,22 @@ class DocumentUploadResponse(BaseModel): class PaperlessWebhookPayload(BaseModel): - """Payload from Paperless-ngx webhook.""" - document_id: int = Field(..., description="Paperless document ID") - event: str = Field(..., description="Event type (document_added, document_updated)") + """Payload from Paperless-ngx webhook (include_document=true format).""" + id: int = Field(..., description="Paperless document ID") title: Optional[str] = Field(None, description="Document title") + content: Optional[str] = Field(None, description="Document text content") correspondent: Optional[int] = Field(None, description="Correspondent ID") document_type: Optional[int] = Field(None, description="Document type ID") tags: List[int] = Field(default_factory=list, description="Tag IDs") + created: Optional[str] = Field(None, description="Created date") + modified: Optional[str] = Field(None, description="Modified datetime") + added: Optional[str] = Field(None, description="Added datetime") + original_file_name: Optional[str] = Field(None, description="Original filename") + owner: Optional[int] = Field(None, description="Owner user ID") + custom_fields: List[Dict[str, Any]] = Field(default_factory=list, description="Custom fields") + + class Config: + extra = "ignore" # Ignore extra fields from Paperless class WebhookResponse(BaseModel): diff --git a/src/routers/documents.py b/src/routers/documents.py index 004e53f..db65357 100644 --- a/src/routers/documents.py +++ b/src/routers/documents.py @@ -64,12 +64,12 @@ async def receive_webhook( """ from src.services.document_sync_service import DocumentSyncService - logger.info(f"Webhook received: document_id={payload.document_id}, event={payload.event}") + logger.info(f"Webhook received: document_id={payload.id}, title={payload.title}") settings = get_settings() if not settings.document_store_enabled: return WebhookResponse( - document_id=payload.document_id, + document_id=payload.id, status="skipped", indexed=False, message="Document store is disabled" @@ -86,21 +86,23 @@ async def receive_webhook( ) result = await sync_service.index_document( - document_id=payload.document_id, - user=user + document_id=payload.id, + user=user, + content=payload.content, # Use content from webhook payload + title=payload.title, ) return WebhookResponse( - document_id=payload.document_id, + document_id=payload.id, status="indexed" if result.success else "failed", indexed=result.success, message=result.error if not result.success else f"Indexed: {result.title}" ) except Exception as e: - logger.error(f"Webhook processing failed for document {payload.document_id}: {e}", exc_info=True) + logger.error(f"Webhook processing failed for document {payload.id}: {e}", exc_info=True) return WebhookResponse( - document_id=payload.document_id, + document_id=payload.id, status="error", indexed=False, message=str(e) diff --git a/src/services/document_sync_service.py b/src/services/document_sync_service.py index ece34a2..7eb2aa8 100644 --- a/src/services/document_sync_service.py +++ b/src/services/document_sync_service.py @@ -86,6 +86,8 @@ class DocumentSyncService: self, document_id: int, user: str, + content: Optional[str] = None, + title: Optional[str] = None, ) -> IndexResult: """ Index a single document from Paperless into vectors and graph. @@ -93,6 +95,8 @@ class DocumentSyncService: Args: document_id: Paperless document ID user: User identifier for multi-tenancy + content: Optional document content (if provided, skip Paperless API call) + title: Optional document title (if provided, skip Paperless API call) Returns: IndexResult with success status and details @@ -100,24 +104,36 @@ class DocumentSyncService: logger.info(f"Indexing document {document_id} for user {user}") try: - # Fetch document from Paperless - doc = await self.paperless.get_document(document_id) - if not doc: - return IndexResult( - success=False, - document_id=document_id, - error="Document not found in Paperless" - ) + # If content and title provided (from webhook), skip API call + if content is not None and title is not None: + doc_title = title + doc_content = content + original_filename = None + correspondent = None + document_type = None + tags = [] + else: + # Fetch document from Paperless + doc = await self.paperless.get_document(document_id) + if not doc: + return IndexResult( + success=False, + document_id=document_id, + error="Document not found in Paperless" + ) + doc_title = doc.title + doc_content = doc.content or "" + original_filename = doc.original_file_name + correspondent = doc.correspondent + document_type = doc.document_type + tags = doc.tags - title = doc.title - content = doc.content or "" - - if not content.strip(): + if not doc_content.strip(): logger.warning(f"Document {document_id} has no text content") return IndexResult( success=True, document_id=document_id, - title=title, + title=doc_title, chunks_created=0, error="No text content (possibly image/video only)" ) @@ -125,23 +141,23 @@ class DocumentSyncService: # Index vectors chunks_created = await self._index_vectors( document_id=document_id, - title=title, - content=content, + title=doc_title, + content=doc_content, user=user, metadata={ "paperless_id": document_id, - "original_filename": doc.original_file_name, - "correspondent": doc.correspondent, - "document_type": doc.document_type, - "tags": doc.tags, + "original_filename": original_filename, + "correspondent": correspondent, + "document_type": document_type, + "tags": tags, } ) # Index graph node await self._index_graph( document_id=document_id, - title=title, - content=content, + title=doc_title, + content=doc_content, user=user, ) @@ -156,7 +172,7 @@ class DocumentSyncService: return IndexResult( success=True, document_id=document_id, - title=title, + title=doc_title, chunks_created=chunks_created )