fix: upsert document vectors before pruning stale chunks

DocumentSyncService._index_vectors ran delete_by_filter on the
document's existing chunks FIRST and only then embedded; if the
embedding pass failed (Ollama down) the Paperless document was left
with zero vectors until the next successful sync - the same
zero-vector hazard already fixed for wiki pages in
VectorService.update_from_page.

Chunk ids are now deterministic uuid5 (document_{id}_chunk_{i}) so
re-upserting overwrites in place; new points are upserted first, then
stale points (including legacy random-uuid4 ones) are pruned via
scroll + delete_by_ids, and only after a successful upsert.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01QbFZyDvYksazX6nYQYZ67L
This commit is contained in:
2026-07-14 15:19:03 +02:00
co-authored by Claude Fable 5
parent bc68d3b691
commit 9681a63757
3 changed files with 77 additions and 19 deletions
+31 -14
View File
@@ -196,18 +196,6 @@ class DocumentSyncService:
collection = get_qdrant_collection_name(user)
await self.qdrant.ensure_collection(collection)
# Delete existing chunks for this document (via the async wrapper)
try:
await self.qdrant.delete_by_filter(
collection_name=collection,
filter_conditions={
"doc_type": "document",
"paperless_id": document_id,
}
)
except Exception as e:
logger.debug(f"No existing chunks to delete: {e}")
# Chunk content
chunks = self._chunk_text(content)
if not chunks:
@@ -229,7 +217,11 @@ class DocumentSyncService:
)
continue
point_id = str(uuid.uuid4())
# Deterministic id: re-upserting the same document overwrites
# its previous chunks in place (enables delete-last below).
point_id = str(
uuid.uuid5(uuid.NAMESPACE_DNS, f"document_{document_id}_chunk_{i}")
)
content_hash = hashlib.md5(chunk.encode()).hexdigest()
points.append({
@@ -256,13 +248,38 @@ class DocumentSyncService:
f"(embedding failures); indexing the remaining {len(points)}"
)
# Upsert to Qdrant in one batch via the async wrapper
# Upsert BEFORE pruning stale chunks (same order as the wiki
# reindex fix in VectorService.update_from_page): the old
# delete-first order left the document with ZERO vectors until the
# next successful sync whenever the embedding pass failed after the
# delete (e.g. Ollama down). Deterministic uuid5 ids make the
# in-place overwrite safe.
if points:
await self.qdrant.upsert_points(
collection_name=collection,
points=points
)
# Prune chunks left over from a previous version of the document
# (indexes beyond the new count, or legacy random-uuid4 points).
# Only prune after a successful upsert - a fully failed embedding
# pass must not wipe the old vectors.
new_ids = {p["id"] for p in points}
existing = await self.qdrant.scroll_all_points(
collection_name=collection,
filter_conditions={
"doc_type": "document",
"paperless_id": document_id,
},
with_payload=False,
)
stale_ids = [pt["id"] for pt in existing if pt["id"] not in new_ids]
if stale_ids:
await self.qdrant.delete_by_ids(
collection_name=collection,
point_ids=stale_ids,
)
return len(points)
async def _index_graph(