feat: add maintenance system with index reconciliation
Complete maintenance subsystem for index health and cleanup: Endpoints: - GET /maintenance/health - lightweight (or detailed) health check - POST /maintenance/cleanup/all - full orphan cleanup - POST /maintenance/cleanup/vectors - purge orphan vector chunks - POST /maintenance/cleanup/graph - purge orphan graph nodes - POST /maintenance/reconcile-index - cleanup + reindex missing pages Bidirectional orphan detection: - find_documents_without_vectors() in GraphService - find_chunks_without_graph_nodes() in VectorService Redis integration: - Tracks last_cleanup timestamp for scheduler visibility Config additions: - Document store, volatile cache, and maintenance settings - VectorServiceDep and GraphServiceDep type aliases 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
This commit is contained in:
@@ -1261,3 +1261,353 @@ Feel free to expand it with more details!
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to create entity mentions: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
# ========== Cleanup Methods ==========
|
||||
|
||||
async def delete_document_node(
|
||||
self,
|
||||
document_id: str,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Delete a Document Store document node and all its relationships.
|
||||
|
||||
Args:
|
||||
document_id: Document UUID (Document Store)
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of nodes deleted (1 if successful, 0 if not found)
|
||||
"""
|
||||
user_doc_label = get_neo4j_user_label(user)
|
||||
|
||||
delete_query = f"""
|
||||
MATCH (d:{user_doc_label}:Document {{document_id: $document_id}})
|
||||
DETACH DELETE d
|
||||
RETURN count(d) as deleted_count
|
||||
"""
|
||||
|
||||
try:
|
||||
result = await self.neo4j.execute_query(
|
||||
delete_query,
|
||||
{"document_id": document_id}
|
||||
)
|
||||
|
||||
deleted_count = result[0]["deleted_count"] if result else 0
|
||||
|
||||
if deleted_count > 0:
|
||||
logger.info(f"Deleted Document node for document {document_id}")
|
||||
else:
|
||||
logger.warning(f"No Document node found for document {document_id}")
|
||||
|
||||
return deleted_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to delete document {document_id} from graph: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def delete_collection_node(
|
||||
self,
|
||||
collection_id: str,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Delete a DocumentCollection node and all contained documents.
|
||||
|
||||
Args:
|
||||
collection_id: Collection UUID
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of nodes deleted (collection + documents)
|
||||
"""
|
||||
user_doc_label = get_neo4j_user_label(user)
|
||||
|
||||
# Delete collection and all documents it contains
|
||||
delete_query = f"""
|
||||
MATCH (c:{user_doc_label}:DocumentCollection {{id: $collection_id}})
|
||||
OPTIONAL MATCH (c)-[:CONTAINS]->(d:Document)
|
||||
DETACH DELETE c, d
|
||||
RETURN count(c) + count(d) as deleted_count
|
||||
"""
|
||||
|
||||
try:
|
||||
result = await self.neo4j.execute_query(
|
||||
delete_query,
|
||||
{"collection_id": collection_id}
|
||||
)
|
||||
|
||||
deleted_count = result[0]["deleted_count"] if result else 0
|
||||
logger.info(f"Deleted collection {collection_id} with {deleted_count} total nodes")
|
||||
return deleted_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to delete collection {collection_id}: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def find_orphan_entities(
|
||||
self,
|
||||
user: str
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Find entities with no MENTIONS relationships (orphaned).
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
List of orphaned entities {id, name, type}
|
||||
"""
|
||||
from src.core.multi_tenancy import get_neo4j_user_base_label
|
||||
|
||||
user_base_label = get_neo4j_user_base_label(user)
|
||||
|
||||
query = f"""
|
||||
MATCH (e:{user_base_label})
|
||||
WHERE NOT e:Document
|
||||
AND NOT e:DocumentCollection
|
||||
AND NOT EXISTS {{ (d:Document)-[:MENTIONS]->(e) }}
|
||||
RETURN elementId(e) as id, e.name as name, labels(e) as labels
|
||||
"""
|
||||
|
||||
try:
|
||||
results = await self.neo4j.execute_query(query, {})
|
||||
|
||||
orphans = []
|
||||
for r in results:
|
||||
labels = r.get("labels", [])
|
||||
entity_type = next(
|
||||
(l for l in labels if l != user_base_label),
|
||||
"Unknown"
|
||||
)
|
||||
orphans.append({
|
||||
"id": r["id"],
|
||||
"name": r["name"],
|
||||
"type": entity_type
|
||||
})
|
||||
|
||||
logger.info(f"Found {len(orphans)} orphan entities for user {user}")
|
||||
return orphans
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to find orphan entities: {e}", exc_info=True)
|
||||
return []
|
||||
|
||||
async def purge_orphan_entities(
|
||||
self,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Delete all orphaned entities (entities with no MENTIONS relationships).
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of entities purged
|
||||
"""
|
||||
from src.core.multi_tenancy import get_neo4j_user_base_label
|
||||
|
||||
user_base_label = get_neo4j_user_base_label(user)
|
||||
|
||||
query = f"""
|
||||
MATCH (e:{user_base_label})
|
||||
WHERE NOT e:Document
|
||||
AND NOT e:DocumentCollection
|
||||
AND NOT EXISTS {{ (d:Document)-[:MENTIONS]->(e) }}
|
||||
DETACH DELETE e
|
||||
RETURN count(e) as purged_count
|
||||
"""
|
||||
|
||||
try:
|
||||
results = await self.neo4j.execute_query(query, {})
|
||||
purged_count = results[0]["purged_count"] if results else 0
|
||||
|
||||
logger.info(f"Purged {purged_count} orphan entities for user {user}")
|
||||
return purged_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to purge orphan entities: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def get_all_document_references(
|
||||
self,
|
||||
user: str
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Get all Document node references for orphan detection.
|
||||
|
||||
Returns page_id for wiki docs and document_id for Document Store docs.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
List of document references {page_id, document_id, doc_type, title}
|
||||
"""
|
||||
user_doc_label = get_neo4j_user_label(user)
|
||||
|
||||
query = f"""
|
||||
MATCH (d:{user_doc_label}:Document)
|
||||
RETURN d.page_id as page_id,
|
||||
d.document_id as document_id,
|
||||
COALESCE(d.doc_type, 'wiki') as doc_type,
|
||||
d.title as title
|
||||
"""
|
||||
|
||||
try:
|
||||
results = await self.neo4j.execute_query(query, {})
|
||||
|
||||
references = []
|
||||
for r in results:
|
||||
references.append({
|
||||
"page_id": r.get("page_id"),
|
||||
"document_id": r.get("document_id"),
|
||||
"doc_type": r.get("doc_type", "wiki"),
|
||||
"title": r.get("title")
|
||||
})
|
||||
|
||||
logger.info(f"Found {len(references)} document references for user {user}")
|
||||
return references
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to get document references: {e}", exc_info=True)
|
||||
return []
|
||||
|
||||
async def purge_stale_documents_by_ids(
|
||||
self,
|
||||
user: str,
|
||||
page_ids: List[int] = None,
|
||||
document_ids: List[str] = None
|
||||
) -> int:
|
||||
"""
|
||||
Delete specific stale Document nodes by their IDs.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
page_ids: List of wiki page IDs to delete
|
||||
document_ids: List of Document Store document IDs to delete
|
||||
|
||||
Returns:
|
||||
Number of documents purged
|
||||
"""
|
||||
user_doc_label = get_neo4j_user_label(user)
|
||||
total_purged = 0
|
||||
|
||||
try:
|
||||
# Purge by page_id (wiki docs)
|
||||
if page_ids:
|
||||
query = f"""
|
||||
MATCH (d:{user_doc_label}:Document)
|
||||
WHERE d.page_id IN $page_ids
|
||||
DETACH DELETE d
|
||||
RETURN count(d) as purged_count
|
||||
"""
|
||||
results = await self.neo4j.execute_query(query, {"page_ids": page_ids})
|
||||
count = results[0]["purged_count"] if results else 0
|
||||
total_purged += count
|
||||
logger.info(f"Purged {count} wiki Document nodes")
|
||||
|
||||
# Purge by document_id (Document Store docs)
|
||||
if document_ids:
|
||||
query = f"""
|
||||
MATCH (d:{user_doc_label}:Document)
|
||||
WHERE d.document_id IN $document_ids
|
||||
DETACH DELETE d
|
||||
RETURN count(d) as purged_count
|
||||
"""
|
||||
results = await self.neo4j.execute_query(query, {"document_ids": document_ids})
|
||||
count = results[0]["purged_count"] if results else 0
|
||||
total_purged += count
|
||||
logger.info(f"Purged {count} Document Store Document nodes")
|
||||
|
||||
return total_purged
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to purge stale documents: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def cleanup_broken_relationships(
|
||||
self,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Clean up broken FOUND relationships from SearchQuery nodes.
|
||||
|
||||
Removes relationships pointing to deleted documents.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of relationships cleaned
|
||||
"""
|
||||
query = """
|
||||
MATCH (sq:SearchQuery)-[r:FOUND]->(d)
|
||||
WHERE NOT EXISTS { (d) }
|
||||
DELETE r
|
||||
RETURN count(r) as cleaned_count
|
||||
"""
|
||||
|
||||
try:
|
||||
results = await self.neo4j.execute_query(query, {})
|
||||
cleaned_count = results[0]["cleaned_count"] if results else 0
|
||||
|
||||
if cleaned_count > 0:
|
||||
logger.info(f"Cleaned {cleaned_count} broken FOUND relationships")
|
||||
|
||||
return cleaned_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to cleanup broken relationships: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def find_documents_without_vectors(
|
||||
self,
|
||||
user: str,
|
||||
vector_references: List[Dict[str, Any]]
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Find Document nodes that have no corresponding vectors.
|
||||
|
||||
Used for bidirectional orphan detection - graph nodes without vector data.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
vector_references: List of vector refs from VectorService.get_all_chunk_references()
|
||||
|
||||
Returns:
|
||||
List of orphan documents {page_id, document_id, doc_type, title}
|
||||
"""
|
||||
# Get all graph document references
|
||||
graph_docs = await self.get_all_document_references(user)
|
||||
|
||||
if not graph_docs:
|
||||
return []
|
||||
|
||||
# Build sets of IDs that have vectors
|
||||
vector_page_ids = {
|
||||
ref.get("page_id") for ref in vector_references
|
||||
if ref.get("doc_type") == "wiki" and ref.get("page_id")
|
||||
}
|
||||
vector_doc_ids = {
|
||||
ref.get("document_id") for ref in vector_references
|
||||
if ref.get("doc_type") != "wiki" and ref.get("document_id")
|
||||
}
|
||||
|
||||
# Find graph docs with no vectors
|
||||
orphans = []
|
||||
for doc in graph_docs:
|
||||
doc_type = doc.get("doc_type", "wiki")
|
||||
|
||||
if doc_type == "wiki":
|
||||
page_id = doc.get("page_id")
|
||||
if page_id and page_id not in vector_page_ids:
|
||||
orphans.append(doc)
|
||||
else:
|
||||
document_id = doc.get("document_id")
|
||||
if document_id and document_id not in vector_doc_ids:
|
||||
orphans.append(doc)
|
||||
|
||||
logger.info(f"Found {len(orphans)} graph documents without vectors for user {user}")
|
||||
return orphans
|
||||
|
||||
@@ -356,3 +356,189 @@ class VectorService:
|
||||
collections=[],
|
||||
total=0
|
||||
)
|
||||
|
||||
# ========== Cleanup Methods ==========
|
||||
|
||||
async def delete_document_chunks(
|
||||
self,
|
||||
document_id: str,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Delete all chunks for a document (Document Store).
|
||||
|
||||
Args:
|
||||
document_id: Document UUID
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of chunks deleted
|
||||
"""
|
||||
collection_name = get_qdrant_collection_name(user)
|
||||
|
||||
try:
|
||||
deleted_count = await self.qdrant.delete_by_filter(
|
||||
collection_name=collection_name,
|
||||
filter_conditions={"document_id": document_id}
|
||||
)
|
||||
|
||||
logger.info(f"Deleted chunks for document {document_id}")
|
||||
return deleted_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to delete chunks for document {document_id}: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def delete_collection_chunks(
|
||||
self,
|
||||
collection_id: str,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Delete all chunks for a document collection.
|
||||
|
||||
Args:
|
||||
collection_id: Collection UUID
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of chunks deleted
|
||||
"""
|
||||
collection_name = get_qdrant_collection_name(user)
|
||||
|
||||
try:
|
||||
deleted_count = await self.qdrant.delete_by_filter(
|
||||
collection_name=collection_name,
|
||||
filter_conditions={"collection_id": collection_id}
|
||||
)
|
||||
|
||||
logger.info(f"Deleted chunks for collection {collection_id}")
|
||||
return deleted_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to delete chunks for collection {collection_id}: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def get_all_chunk_references(
|
||||
self,
|
||||
user: str
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Get all chunk references for orphan detection.
|
||||
|
||||
Returns list of {id, page_id, document_id} for all chunks.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
List of chunk references
|
||||
"""
|
||||
collection_name = get_qdrant_collection_name(user)
|
||||
|
||||
try:
|
||||
# Check if collection exists
|
||||
exists = await self.qdrant.collection_exists(collection_name)
|
||||
if not exists:
|
||||
return []
|
||||
|
||||
all_points = await self.qdrant.scroll_all_points(
|
||||
collection_name=collection_name,
|
||||
batch_size=100,
|
||||
with_payload=True
|
||||
)
|
||||
|
||||
references = []
|
||||
for point in all_points:
|
||||
payload = point.get("payload", {})
|
||||
references.append({
|
||||
"chunk_id": point["id"],
|
||||
"page_id": payload.get("page_id"),
|
||||
"document_id": payload.get("document_id"),
|
||||
"collection_id": payload.get("collection_id"),
|
||||
"doc_type": payload.get("doc_type", "wiki")
|
||||
})
|
||||
|
||||
logger.info(f"Found {len(references)} chunks for user {user}")
|
||||
return references
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to get chunk references: {e}", exc_info=True)
|
||||
return []
|
||||
|
||||
async def purge_chunks_by_ids(
|
||||
self,
|
||||
user: str,
|
||||
chunk_ids: List[str]
|
||||
) -> int:
|
||||
"""
|
||||
Delete specific chunks by their IDs.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
chunk_ids: List of chunk IDs to delete
|
||||
|
||||
Returns:
|
||||
Number of chunks deleted
|
||||
"""
|
||||
if not chunk_ids:
|
||||
return 0
|
||||
|
||||
collection_name = get_qdrant_collection_name(user)
|
||||
|
||||
try:
|
||||
deleted_count = await self.qdrant.delete_by_ids(
|
||||
collection_name=collection_name,
|
||||
point_ids=chunk_ids
|
||||
)
|
||||
|
||||
logger.info(f"Purged {deleted_count} orphan chunks for user {user}")
|
||||
return deleted_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to purge chunks: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
def find_chunks_without_graph_nodes(
|
||||
self,
|
||||
chunk_references: List[Dict[str, Any]],
|
||||
graph_references: List[Dict[str, Any]]
|
||||
) -> List[str]:
|
||||
"""
|
||||
Find vector chunks that have no corresponding graph Document node.
|
||||
|
||||
Used for bidirectional orphan detection - vectors without graph representation.
|
||||
|
||||
Args:
|
||||
chunk_references: List from get_all_chunk_references()
|
||||
graph_references: List from GraphService.get_all_document_references()
|
||||
|
||||
Returns:
|
||||
List of orphan chunk IDs
|
||||
"""
|
||||
# Build sets of IDs that have graph nodes
|
||||
graph_page_ids = {
|
||||
ref.get("page_id") for ref in graph_references
|
||||
if ref.get("doc_type") == "wiki" and ref.get("page_id")
|
||||
}
|
||||
graph_doc_ids = {
|
||||
ref.get("document_id") for ref in graph_references
|
||||
if ref.get("doc_type") != "wiki" and ref.get("document_id")
|
||||
}
|
||||
|
||||
# Find chunks with no graph node
|
||||
orphan_ids = []
|
||||
for chunk in chunk_references:
|
||||
doc_type = chunk.get("doc_type", "wiki")
|
||||
|
||||
if doc_type == "wiki":
|
||||
page_id = chunk.get("page_id")
|
||||
if page_id and page_id not in graph_page_ids:
|
||||
orphan_ids.append(chunk["chunk_id"])
|
||||
else:
|
||||
document_id = chunk.get("document_id")
|
||||
if document_id and document_id not in graph_doc_ids:
|
||||
orphan_ids.append(chunk["chunk_id"])
|
||||
|
||||
logger.info(f"Found {len(orphan_ids)} vector chunks without graph nodes")
|
||||
return orphan_ids
|
||||
|
||||
Reference in New Issue
Block a user