feat: implement check-updates, job-backed ingest status, and dedup scan
Replace the four stub endpoints with real implementations, all requiring
an explicit tenant user (Phase B rule):
- /ingest/check-updates: GraphService now records a SHA-256 content_hash
on every Document node at ingestion time; the endpoint compares those
stored hashes against current Wiki.js page content in one UNWIND Cypher
query per tenant and returns changed/new/deleted page lists (entity-stub
pages excluded, pre-hash-tracking documents flagged stored_hash_missing).
- /ingest/status/{job_id}: backed by the Redis JobManager; jobs are
tenant-scoped (foreign jobs 404). /ingest/page, /ingest/batch and
/ingest/all now create job records and return job_id.
- /ingest/repo-status/{repository}: wiki page count vs indexed Document
nodes under users/{tenant}/{repository} plus tenant job stats.
- /deduplicate/check: tenant-scoped Qdrant similarity scan; chunk pairs
above ~0.9 cosine from different pages grouped per page pair with best
score and page references (read-only).
Supporting changes: get_job_manager dependency (+ shutdown close),
scroll_all_points can return vectors, VectorService.find_duplicate_pairs,
src/core/hashing.compute_content_hash. 13 new offline unit tests.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01QbFZyDvYksazX6nYQYZ67L
This commit is contained in:
@@ -452,14 +452,20 @@ class GraphService:
|
||||
user_base_label = get_neo4j_user_base_label(user) # For entities
|
||||
user_doc_label = get_neo4j_user_label(user) # For documents
|
||||
|
||||
# Create/update Document node
|
||||
# Create/update Document node.
|
||||
# content_hash records the fingerprint of the ingested content so
|
||||
# /ingest/check-updates can detect changed pages without re-reading
|
||||
# the graph's source content.
|
||||
from src.core.hashing import compute_content_hash
|
||||
|
||||
doc_query = f"""
|
||||
MERGE (d:{user_doc_label}:Document {{page_id: $page_id}})
|
||||
SET d.title = $title,
|
||||
d.path = $path,
|
||||
d.tags = $tags,
|
||||
d.updated_at = datetime(),
|
||||
d.content_length = $content_length
|
||||
d.content_length = $content_length,
|
||||
d.content_hash = $content_hash
|
||||
RETURN d
|
||||
"""
|
||||
|
||||
@@ -468,7 +474,8 @@ class GraphService:
|
||||
"title": page.get("title"),
|
||||
"path": page.get("path"),
|
||||
"tags": tags,
|
||||
"content_length": len(content)
|
||||
"content_length": len(content),
|
||||
"content_hash": compute_content_hash(content)
|
||||
})
|
||||
|
||||
nodes_created = 1 # Document node
|
||||
|
||||
@@ -543,6 +543,129 @@ class VectorService:
|
||||
logger.error(f"Failed to purge chunks: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def find_duplicate_pairs(
|
||||
self,
|
||||
user: str,
|
||||
similarity_threshold: float = 0.9,
|
||||
max_chunks_scanned: int = 2000,
|
||||
max_pairs: int = 100
|
||||
) -> Dict[str, Any]:
|
||||
"""
|
||||
Tenant-scoped similarity scan for near-duplicate wiki pages.
|
||||
|
||||
Scrolls the tenant's own Qdrant collection (never another tenant's),
|
||||
then queries each chunk's vector against the same collection. Chunk
|
||||
pairs from DIFFERENT pages scoring above the threshold are grouped
|
||||
per page pair with the best score and the number of matching chunk
|
||||
pairs. Read-only: nothing is modified.
|
||||
|
||||
Args:
|
||||
user: Tenant user identifier
|
||||
similarity_threshold: Minimum cosine similarity (default 0.9)
|
||||
max_chunks_scanned: Safety cap on chunks used as probes
|
||||
max_pairs: Maximum page pairs returned (highest score first)
|
||||
|
||||
Returns:
|
||||
{
|
||||
"chunks_scanned": int,
|
||||
"duplicate_groups": [
|
||||
{
|
||||
"pages": [{page_id, path, title}, {page_id, path, title}],
|
||||
"max_similarity": float,
|
||||
"matching_chunk_pairs": int
|
||||
}, ...
|
||||
]
|
||||
}
|
||||
"""
|
||||
collection_name = get_qdrant_collection_name(user)
|
||||
|
||||
exists = await self.qdrant.collection_exists(collection_name)
|
||||
if not exists:
|
||||
return {"chunks_scanned": 0, "duplicate_groups": []}
|
||||
|
||||
points = await self.qdrant.scroll_all_points(
|
||||
collection_name=collection_name,
|
||||
batch_size=100,
|
||||
with_payload=True,
|
||||
with_vectors=True
|
||||
)
|
||||
|
||||
# Only wiki chunks participate (documents have their own dedup story)
|
||||
wiki_points = [
|
||||
p for p in points
|
||||
if p.get("vector") is not None
|
||||
and (p.get("payload") or {}).get("doc_type", "wiki") == "wiki"
|
||||
and (p.get("payload") or {}).get("page_id")
|
||||
][:max_chunks_scanned]
|
||||
|
||||
page_meta: Dict[int, Dict[str, Any]] = {}
|
||||
pair_stats: Dict[tuple, Dict[str, Any]] = {}
|
||||
seen_chunk_pairs = set()
|
||||
|
||||
for point in wiki_points:
|
||||
payload = point.get("payload") or {}
|
||||
page_id = payload.get("page_id")
|
||||
page_meta.setdefault(page_id, {
|
||||
"page_id": page_id,
|
||||
"path": payload.get("page_path", ""),
|
||||
"title": payload.get("page_title", "")
|
||||
})
|
||||
|
||||
hits = await self.qdrant.search_vectors(
|
||||
collection_name=collection_name,
|
||||
query_vector=point["vector"],
|
||||
limit=10,
|
||||
score_threshold=similarity_threshold
|
||||
)
|
||||
|
||||
for hit in hits:
|
||||
hit_payload = hit.get("payload") or {}
|
||||
hit_page_id = hit_payload.get("page_id")
|
||||
if not hit_page_id or hit_page_id == page_id:
|
||||
continue
|
||||
if hit_payload.get("doc_type", "wiki") != "wiki":
|
||||
continue
|
||||
|
||||
# Deduplicate the A->B / B->A chunk pair directions
|
||||
chunk_pair = tuple(sorted((point["id"], hit["id"])))
|
||||
if chunk_pair in seen_chunk_pairs:
|
||||
continue
|
||||
seen_chunk_pairs.add(chunk_pair)
|
||||
|
||||
page_meta.setdefault(hit_page_id, {
|
||||
"page_id": hit_page_id,
|
||||
"path": hit_payload.get("page_path", ""),
|
||||
"title": hit_payload.get("page_title", "")
|
||||
})
|
||||
|
||||
page_pair = tuple(sorted((page_id, hit_page_id)))
|
||||
stats = pair_stats.setdefault(page_pair, {
|
||||
"max_similarity": 0.0,
|
||||
"matching_chunk_pairs": 0
|
||||
})
|
||||
stats["max_similarity"] = max(stats["max_similarity"], hit["score"])
|
||||
stats["matching_chunk_pairs"] += 1
|
||||
|
||||
groups = [
|
||||
{
|
||||
"pages": [page_meta[a], page_meta[b]],
|
||||
"max_similarity": stats["max_similarity"],
|
||||
"matching_chunk_pairs": stats["matching_chunk_pairs"]
|
||||
}
|
||||
for (a, b), stats in pair_stats.items()
|
||||
]
|
||||
groups.sort(key=lambda g: g["max_similarity"], reverse=True)
|
||||
|
||||
logger.info(
|
||||
f"Duplicate scan for {user}: {len(wiki_points)} chunks scanned, "
|
||||
f"{len(groups)} page pairs above {similarity_threshold}"
|
||||
)
|
||||
|
||||
return {
|
||||
"chunks_scanned": len(wiki_points),
|
||||
"duplicate_groups": groups[:max_pairs]
|
||||
}
|
||||
|
||||
def find_chunks_without_graph_nodes(
|
||||
self,
|
||||
chunk_references: List[Dict[str, Any]],
|
||||
|
||||
Reference in New Issue
Block a user