fix: upsert document vectors before pruning stale chunks
DocumentSyncService._index_vectors ran delete_by_filter on the
document's existing chunks FIRST and only then embedded; if the
embedding pass failed (Ollama down) the Paperless document was left
with zero vectors until the next successful sync - the same
zero-vector hazard already fixed for wiki pages in
VectorService.update_from_page.
Chunk ids are now deterministic uuid5 (document_{id}_chunk_{i}) so
re-upserting overwrites in place; new points are upserted first, then
stale points (including legacy random-uuid4 ones) are pruned via
scroll + delete_by_ids, and only after a successful upsert.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01QbFZyDvYksazX6nYQYZ67L
This commit is contained in:
@@ -196,18 +196,6 @@ class DocumentSyncService:
|
||||
collection = get_qdrant_collection_name(user)
|
||||
await self.qdrant.ensure_collection(collection)
|
||||
|
||||
# Delete existing chunks for this document (via the async wrapper)
|
||||
try:
|
||||
await self.qdrant.delete_by_filter(
|
||||
collection_name=collection,
|
||||
filter_conditions={
|
||||
"doc_type": "document",
|
||||
"paperless_id": document_id,
|
||||
}
|
||||
)
|
||||
except Exception as e:
|
||||
logger.debug(f"No existing chunks to delete: {e}")
|
||||
|
||||
# Chunk content
|
||||
chunks = self._chunk_text(content)
|
||||
if not chunks:
|
||||
@@ -229,7 +217,11 @@ class DocumentSyncService:
|
||||
)
|
||||
continue
|
||||
|
||||
point_id = str(uuid.uuid4())
|
||||
# Deterministic id: re-upserting the same document overwrites
|
||||
# its previous chunks in place (enables delete-last below).
|
||||
point_id = str(
|
||||
uuid.uuid5(uuid.NAMESPACE_DNS, f"document_{document_id}_chunk_{i}")
|
||||
)
|
||||
content_hash = hashlib.md5(chunk.encode()).hexdigest()
|
||||
|
||||
points.append({
|
||||
@@ -256,13 +248,38 @@ class DocumentSyncService:
|
||||
f"(embedding failures); indexing the remaining {len(points)}"
|
||||
)
|
||||
|
||||
# Upsert to Qdrant in one batch via the async wrapper
|
||||
# Upsert BEFORE pruning stale chunks (same order as the wiki
|
||||
# reindex fix in VectorService.update_from_page): the old
|
||||
# delete-first order left the document with ZERO vectors until the
|
||||
# next successful sync whenever the embedding pass failed after the
|
||||
# delete (e.g. Ollama down). Deterministic uuid5 ids make the
|
||||
# in-place overwrite safe.
|
||||
if points:
|
||||
await self.qdrant.upsert_points(
|
||||
collection_name=collection,
|
||||
points=points
|
||||
)
|
||||
|
||||
# Prune chunks left over from a previous version of the document
|
||||
# (indexes beyond the new count, or legacy random-uuid4 points).
|
||||
# Only prune after a successful upsert - a fully failed embedding
|
||||
# pass must not wipe the old vectors.
|
||||
new_ids = {p["id"] for p in points}
|
||||
existing = await self.qdrant.scroll_all_points(
|
||||
collection_name=collection,
|
||||
filter_conditions={
|
||||
"doc_type": "document",
|
||||
"paperless_id": document_id,
|
||||
},
|
||||
with_payload=False,
|
||||
)
|
||||
stale_ids = [pt["id"] for pt in existing if pt["id"] not in new_ids]
|
||||
if stale_ids:
|
||||
await self.qdrant.delete_by_ids(
|
||||
collection_name=collection,
|
||||
point_ids=stale_ids,
|
||||
)
|
||||
|
||||
return len(points)
|
||||
|
||||
async def _index_graph(
|
||||
|
||||
Reference in New Issue
Block a user