Files
library-desk/src/routers/hybrid_rag.py
T
jpmschweitzerandClaude Opus 4.5 37f8e1819e
Build and Push / build (release) Successful in 28s
feat: refactor volatile cache to vector storage with HybridRAG integration
- Migrate volatile backend from Redis to Qdrant for semantic search
- Add natural language conversion for structured data embedding
- Simplify API: /volatile/search, /volatile/store, /{namespace}/{key}
- Integrate volatile into HybridRAG with priority boost in RRF fusion
- Add POST /maintenance/cleanup/volatile for expiry purging
- Update tests for new Qdrant-based architecture (37/37 pass)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-24 20:03:37 +01:00

121 lines
4.1 KiB
Python

"""
HybridRAG router for multi-source search API.
Provides endpoint for combining vector, graph, volatile cache, and web search
with RRF fusion and LLM re-ranking.
"""
from fastapi import APIRouter, HTTPException, Depends, Query
import logging
from src.models.hybrid_rag import HybridRAGRequest, HybridRAGResponse
from src.services.hybrid_rag_service import HybridRAGService
from src.core.dependencies import (
Neo4jDep, WikiJSDep, QdrantDep, OllamaDep,
SearXNGDep, ContentExtractorDep, verify_api_key, get_settings
)
from src.config import Settings
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/query", tags=["HybridRAG"])
# Dependency to get HybridRAG service
def get_hybrid_rag_service(
neo4j_client: Neo4jDep,
wiki_client: WikiJSDep,
qdrant_client: QdrantDep,
ollama_client: OllamaDep,
searxng_client: SearXNGDep,
content_extractor: ContentExtractorDep,
settings: Settings = Depends(get_settings)
) -> HybridRAGService:
"""Get HybridRAG service instance with all dependencies."""
from src.services.vector_service import VectorService
from src.services.graph_service import GraphService
from src.services.volatile_service import VolatileCacheService
# Create component services
vector_service = VectorService(qdrant_client, wiki_client, ollama_client)
graph_service = GraphService(neo4j_client, wiki_client)
volatile_service = VolatileCacheService(qdrant_client, ollama_client, settings)
# Create HybridRAG service
return HybridRAGService(
vector_service=vector_service,
graph_service=graph_service,
searxng_client=searxng_client,
ollama_client=ollama_client,
content_extractor=content_extractor,
settings=settings,
volatile_service=volatile_service
)
@router.post("/hybrid", response_model=HybridRAGResponse)
async def hybrid_search(
request: HybridRAGRequest,
user: str = Query(default="jpmschweitzer", description="User identifier for multi-tenancy"),
hybrid_rag_service: HybridRAGService = Depends(get_hybrid_rag_service),
api_key: str = Depends(verify_api_key)
):
"""
Execute HybridRAG query combining vector, graph, volatile cache, and web search.
**6-Phase Pipeline:**
1. **Query Enhancement**: Extract keywords/synonyms with LLM
2. **Parallel Retrieval**: Search vector (Qdrant), graph (Neo4j), volatile cache, web (SearXNG)
3. **RRF Fusion**: Merge results with Reciprocal Rank Fusion (volatile gets priority boost)
4. **Enrichment**: Add related documents via shared entities
5. **LLM Re-ranking**: Re-rank with configured model for relevance
6. **Context Formatting**: Format for LLM consumption
7. **Persistence**: Store for Librarian knowledge consolidation
**Example Request:**
```json
{
"query": "What's the weather in Rotterdam?",
"user": "jpmschweitzer",
"config": {
"vector_limit": 10,
"graph_limit": 10,
"web_limit": 5,
"volatile_limit": 5,
"enable_volatile": true,
"enable_reranking": true,
"final_result_count": 10
}
}
```
**Returns:**
- Ranked results from all sources (wiki, volatile, web)
- Extracted keywords/synonyms
- Related dossiers (via graph)
- Formatted context for LLM
- Performance timing breakdown
- Search ID for Librarian tracking
"""
try:
logger.info(f"HybridRAG request: '{request.query}' for user '{user}'")
response = await hybrid_rag_service.search(
query=request.query,
user=user,
config=request.config
)
logger.info(
f"HybridRAG completed: {response.total_results} results in {response.timing.total_ms:.0f}ms"
)
return response
except ValueError as e:
logger.error(f"Invalid request: {e}")
raise HTTPException(status_code=400, detail=str(e))
except Exception as e:
logger.error(f"HybridRAG search failed: {e}", exc_info=True)
raise HTTPException(status_code=500, detail="Search failed")