fix: update Paperless webhook payload to match include_document format
Build and Push / build (release) Successful in 30s

- Change model field from document_id to id (Paperless sends id)
- Add content, created, modified, added, original_file_name, owner fields
- Add extra="ignore" config to handle additional Paperless fields
- Update sync service to use content from webhook payload
- Skip Paperless API call when content already provided

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
This commit is contained in:
2025-12-25 14:38:40 +01:00
co-authored by Claude Opus 4.5
parent f4352841a2
commit f2b8c7d111
5 changed files with 70 additions and 33 deletions
+38 -22
View File
@@ -86,6 +86,8 @@ class DocumentSyncService:
self,
document_id: int,
user: str,
content: Optional[str] = None,
title: Optional[str] = None,
) -> IndexResult:
"""
Index a single document from Paperless into vectors and graph.
@@ -93,6 +95,8 @@ class DocumentSyncService:
Args:
document_id: Paperless document ID
user: User identifier for multi-tenancy
content: Optional document content (if provided, skip Paperless API call)
title: Optional document title (if provided, skip Paperless API call)
Returns:
IndexResult with success status and details
@@ -100,24 +104,36 @@ class DocumentSyncService:
logger.info(f"Indexing document {document_id} for user {user}")
try:
# Fetch document from Paperless
doc = await self.paperless.get_document(document_id)
if not doc:
return IndexResult(
success=False,
document_id=document_id,
error="Document not found in Paperless"
)
# If content and title provided (from webhook), skip API call
if content is not None and title is not None:
doc_title = title
doc_content = content
original_filename = None
correspondent = None
document_type = None
tags = []
else:
# Fetch document from Paperless
doc = await self.paperless.get_document(document_id)
if not doc:
return IndexResult(
success=False,
document_id=document_id,
error="Document not found in Paperless"
)
doc_title = doc.title
doc_content = doc.content or ""
original_filename = doc.original_file_name
correspondent = doc.correspondent
document_type = doc.document_type
tags = doc.tags
title = doc.title
content = doc.content or ""
if not content.strip():
if not doc_content.strip():
logger.warning(f"Document {document_id} has no text content")
return IndexResult(
success=True,
document_id=document_id,
title=title,
title=doc_title,
chunks_created=0,
error="No text content (possibly image/video only)"
)
@@ -125,23 +141,23 @@ class DocumentSyncService:
# Index vectors
chunks_created = await self._index_vectors(
document_id=document_id,
title=title,
content=content,
title=doc_title,
content=doc_content,
user=user,
metadata={
"paperless_id": document_id,
"original_filename": doc.original_file_name,
"correspondent": doc.correspondent,
"document_type": doc.document_type,
"tags": doc.tags,
"original_filename": original_filename,
"correspondent": correspondent,
"document_type": document_type,
"tags": tags,
}
)
# Index graph node
await self._index_graph(
document_id=document_id,
title=title,
content=content,
title=doc_title,
content=doc_content,
user=user,
)
@@ -156,7 +172,7 @@ class DocumentSyncService:
return IndexResult(
success=True,
document_id=document_id,
title=title,
title=doc_title,
chunks_created=chunks_created
)