feat: add RAG search endpoint with content extraction
Build and Push / build (release) Successful in 1m2s

- Add /rag/search endpoint for web, news, and image search via SearXNG
- Add /content/extract and /content/extract/batch endpoints
- Add ContentExtractor client using Trafilatura for content extraction
- Enhance HybridRAG web search with full content extraction
- Add Redis caching for search results
- Add new configuration options for search and extraction timeouts

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
This commit is contained in:
2025-12-15 15:50:48 +01:00
co-authored by Claude Opus 4.5
parent 16b86a7c1b
commit 61863ff597
16 changed files with 1642 additions and 4 deletions
+132
View File
@@ -0,0 +1,132 @@
"""
Content extraction router for Library Desk API.
Endpoints for extracting main content from web URLs using Trafilatura.
"""
import time
from fastapi import APIRouter, HTTPException, Depends
import logging
from src.models.content import (
ContentExtractionRequest,
ContentExtractionResponse,
BatchContentExtractionRequest,
BatchContentExtractionResponse,
)
from src.clients.content_extractor import ContentExtractor
from src.core.dependencies import verify_api_key
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/content", tags=["Content Extraction"])
# Lazy import to avoid circular dependency
def get_content_extractor() -> ContentExtractor:
"""Get content extractor instance."""
from src.core.dependencies import get_content_extractor as _get_extractor
return _get_extractor()
@router.post("/extract", response_model=ContentExtractionResponse)
async def extract_content(
request: ContentExtractionRequest,
api_key: str = Depends(verify_api_key)
):
"""
Extract main content from a single URL.
Uses Trafilatura to fetch the URL and extract the main text content,
removing navigation, ads, and other boilerplate.
**Example Request:**
```json
{
"url": "https://example.com/article",
"include_metadata": true,
"max_length": 2000
}
```
**Returns:** Extracted content with optional metadata (title, author, date)
"""
start_time = time.time()
try:
extractor = get_content_extractor()
result = await extractor.extract(
url=request.url,
include_metadata=request.include_metadata,
max_length=request.max_length
)
extraction_time_ms = int((time.time() - start_time) * 1000)
return ContentExtractionResponse(
result=result,
extraction_time_ms=extraction_time_ms
)
except ValueError as e:
raise HTTPException(status_code=400, detail=str(e))
except Exception as e:
logger.error(f"Content extraction failed: {e}", exc_info=True)
raise HTTPException(status_code=500, detail="Content extraction failed")
@router.post("/extract/batch", response_model=BatchContentExtractionResponse)
async def extract_content_batch(
request: BatchContentExtractionRequest,
api_key: str = Depends(verify_api_key)
):
"""
Extract content from multiple URLs in parallel.
Processes up to 20 URLs concurrently with per-URL timeouts.
Failed extractions are included in results with success=false.
**Example Request:**
```json
{
"urls": [
"https://example.com/article1",
"https://example.com/article2"
],
"include_metadata": true,
"max_length": 2000
}
```
**Returns:** List of extraction results with success/failure counts
"""
start_time = time.time()
if not request.urls:
raise HTTPException(status_code=400, detail="URLs list cannot be empty")
try:
extractor = get_content_extractor()
results = await extractor.extract_batch(
urls=request.urls,
include_metadata=request.include_metadata,
max_length=request.max_length
)
extraction_time_ms = int((time.time() - start_time) * 1000)
successful = sum(1 for r in results if r.success)
failed = len(results) - successful
return BatchContentExtractionResponse(
results=results,
total_urls=len(request.urls),
successful=successful,
failed=failed,
extraction_time_ms=extraction_time_ms
)
except ValueError as e:
raise HTTPException(status_code=400, detail=str(e))
except Exception as e:
logger.error(f"Batch content extraction failed: {e}", exc_info=True)
raise HTTPException(status_code=500, detail="Batch extraction failed")
+87
View File
@@ -0,0 +1,87 @@
"""
RAG search router for Library Desk API.
Endpoints for web, news, and image search with content extraction.
"""
import httpx
from fastapi import APIRouter, HTTPException, Depends
import logging
from src.models.rag_search import RAGSearchRequest, RAGSearchResponse
from src.services.rag_search_service import RAGSearchService
from src.core.dependencies import verify_api_key
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/rag", tags=["RAG Search"])
# Lazy import to avoid circular dependency
def get_rag_search_service() -> RAGSearchService:
"""Get RAG search service instance."""
from src.core.dependencies import get_rag_search_service as _get_service
return _get_service()
@router.post("/search", response_model=RAGSearchResponse)
async def search(
request: RAGSearchRequest,
api_key: str = Depends(verify_api_key)
):
"""
Execute RAG-optimized web search with content extraction.
Searches via SearXNG and extracts full content from results using
Trafilatura. Results are cached in Redis for efficiency.
**Search Types:**
- `web`: General web search (default)
- `news`: News articles with recency filtering
- `images`: Image search results
**Example Request:**
```json
{
"query": "Python async programming best practices",
"search_type": "web",
"limit": 10,
"user": "default"
}
```
**Response includes:**
- Full extracted text content per result
- Original search snippets
- Source domain names
- Markdown sources summary for LLM consumption
**Error Codes:**
- 400: Invalid query (empty or too long)
- 502: Search provider (SearXNG) error
- 504: Search timeout
"""
try:
service = get_rag_search_service()
response = await service.search(
query=request.query,
search_type=request.search_type,
limit=request.limit,
user=request.user
)
return response
except ValueError as e:
raise HTTPException(status_code=400, detail=str(e))
except httpx.TimeoutException:
logger.error(f"Search timed out for query: {request.query}")
raise HTTPException(status_code=504, detail="Search timed out")
except httpx.HTTPError as e:
logger.error(f"Search provider error: {e}")
raise HTTPException(status_code=502, detail="Search provider error")
except Exception as e:
logger.error(f"RAG search failed: {e}", exc_info=True)
raise HTTPException(status_code=500, detail="Search failed")