feat: add RAG search endpoint with content extraction
Build and Push / build (release) Successful in 1m2s
Build and Push / build (release) Successful in 1m2s
- Add /rag/search endpoint for web, news, and image search via SearXNG - Add /content/extract and /content/extract/batch endpoints - Add ContentExtractor client using Trafilatura for content extraction - Enhance HybridRAG web search with full content extraction - Add Redis caching for search results - Add new configuration options for search and extraction timeouts 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,132 @@
|
||||
"""
|
||||
Content extraction router for Library Desk API.
|
||||
|
||||
Endpoints for extracting main content from web URLs using Trafilatura.
|
||||
"""
|
||||
|
||||
import time
|
||||
from fastapi import APIRouter, HTTPException, Depends
|
||||
import logging
|
||||
|
||||
from src.models.content import (
|
||||
ContentExtractionRequest,
|
||||
ContentExtractionResponse,
|
||||
BatchContentExtractionRequest,
|
||||
BatchContentExtractionResponse,
|
||||
)
|
||||
from src.clients.content_extractor import ContentExtractor
|
||||
from src.core.dependencies import verify_api_key
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/content", tags=["Content Extraction"])
|
||||
|
||||
|
||||
# Lazy import to avoid circular dependency
|
||||
def get_content_extractor() -> ContentExtractor:
|
||||
"""Get content extractor instance."""
|
||||
from src.core.dependencies import get_content_extractor as _get_extractor
|
||||
return _get_extractor()
|
||||
|
||||
|
||||
@router.post("/extract", response_model=ContentExtractionResponse)
|
||||
async def extract_content(
|
||||
request: ContentExtractionRequest,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Extract main content from a single URL.
|
||||
|
||||
Uses Trafilatura to fetch the URL and extract the main text content,
|
||||
removing navigation, ads, and other boilerplate.
|
||||
|
||||
**Example Request:**
|
||||
```json
|
||||
{
|
||||
"url": "https://example.com/article",
|
||||
"include_metadata": true,
|
||||
"max_length": 2000
|
||||
}
|
||||
```
|
||||
|
||||
**Returns:** Extracted content with optional metadata (title, author, date)
|
||||
"""
|
||||
start_time = time.time()
|
||||
|
||||
try:
|
||||
extractor = get_content_extractor()
|
||||
result = await extractor.extract(
|
||||
url=request.url,
|
||||
include_metadata=request.include_metadata,
|
||||
max_length=request.max_length
|
||||
)
|
||||
|
||||
extraction_time_ms = int((time.time() - start_time) * 1000)
|
||||
|
||||
return ContentExtractionResponse(
|
||||
result=result,
|
||||
extraction_time_ms=extraction_time_ms
|
||||
)
|
||||
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
except Exception as e:
|
||||
logger.error(f"Content extraction failed: {e}", exc_info=True)
|
||||
raise HTTPException(status_code=500, detail="Content extraction failed")
|
||||
|
||||
|
||||
@router.post("/extract/batch", response_model=BatchContentExtractionResponse)
|
||||
async def extract_content_batch(
|
||||
request: BatchContentExtractionRequest,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Extract content from multiple URLs in parallel.
|
||||
|
||||
Processes up to 20 URLs concurrently with per-URL timeouts.
|
||||
Failed extractions are included in results with success=false.
|
||||
|
||||
**Example Request:**
|
||||
```json
|
||||
{
|
||||
"urls": [
|
||||
"https://example.com/article1",
|
||||
"https://example.com/article2"
|
||||
],
|
||||
"include_metadata": true,
|
||||
"max_length": 2000
|
||||
}
|
||||
```
|
||||
|
||||
**Returns:** List of extraction results with success/failure counts
|
||||
"""
|
||||
start_time = time.time()
|
||||
|
||||
if not request.urls:
|
||||
raise HTTPException(status_code=400, detail="URLs list cannot be empty")
|
||||
|
||||
try:
|
||||
extractor = get_content_extractor()
|
||||
results = await extractor.extract_batch(
|
||||
urls=request.urls,
|
||||
include_metadata=request.include_metadata,
|
||||
max_length=request.max_length
|
||||
)
|
||||
|
||||
extraction_time_ms = int((time.time() - start_time) * 1000)
|
||||
successful = sum(1 for r in results if r.success)
|
||||
failed = len(results) - successful
|
||||
|
||||
return BatchContentExtractionResponse(
|
||||
results=results,
|
||||
total_urls=len(request.urls),
|
||||
successful=successful,
|
||||
failed=failed,
|
||||
extraction_time_ms=extraction_time_ms
|
||||
)
|
||||
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
except Exception as e:
|
||||
logger.error(f"Batch content extraction failed: {e}", exc_info=True)
|
||||
raise HTTPException(status_code=500, detail="Batch extraction failed")
|
||||
@@ -0,0 +1,87 @@
|
||||
"""
|
||||
RAG search router for Library Desk API.
|
||||
|
||||
Endpoints for web, news, and image search with content extraction.
|
||||
"""
|
||||
|
||||
import httpx
|
||||
from fastapi import APIRouter, HTTPException, Depends
|
||||
import logging
|
||||
|
||||
from src.models.rag_search import RAGSearchRequest, RAGSearchResponse
|
||||
from src.services.rag_search_service import RAGSearchService
|
||||
from src.core.dependencies import verify_api_key
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/rag", tags=["RAG Search"])
|
||||
|
||||
|
||||
# Lazy import to avoid circular dependency
|
||||
def get_rag_search_service() -> RAGSearchService:
|
||||
"""Get RAG search service instance."""
|
||||
from src.core.dependencies import get_rag_search_service as _get_service
|
||||
return _get_service()
|
||||
|
||||
|
||||
@router.post("/search", response_model=RAGSearchResponse)
|
||||
async def search(
|
||||
request: RAGSearchRequest,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Execute RAG-optimized web search with content extraction.
|
||||
|
||||
Searches via SearXNG and extracts full content from results using
|
||||
Trafilatura. Results are cached in Redis for efficiency.
|
||||
|
||||
**Search Types:**
|
||||
- `web`: General web search (default)
|
||||
- `news`: News articles with recency filtering
|
||||
- `images`: Image search results
|
||||
|
||||
**Example Request:**
|
||||
```json
|
||||
{
|
||||
"query": "Python async programming best practices",
|
||||
"search_type": "web",
|
||||
"limit": 10,
|
||||
"user": "default"
|
||||
}
|
||||
```
|
||||
|
||||
**Response includes:**
|
||||
- Full extracted text content per result
|
||||
- Original search snippets
|
||||
- Source domain names
|
||||
- Markdown sources summary for LLM consumption
|
||||
|
||||
**Error Codes:**
|
||||
- 400: Invalid query (empty or too long)
|
||||
- 502: Search provider (SearXNG) error
|
||||
- 504: Search timeout
|
||||
"""
|
||||
try:
|
||||
service = get_rag_search_service()
|
||||
response = await service.search(
|
||||
query=request.query,
|
||||
search_type=request.search_type,
|
||||
limit=request.limit,
|
||||
user=request.user
|
||||
)
|
||||
return response
|
||||
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
|
||||
except httpx.TimeoutException:
|
||||
logger.error(f"Search timed out for query: {request.query}")
|
||||
raise HTTPException(status_code=504, detail="Search timed out")
|
||||
|
||||
except httpx.HTTPError as e:
|
||||
logger.error(f"Search provider error: {e}")
|
||||
raise HTTPException(status_code=502, detail="Search provider error")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"RAG search failed: {e}", exc_info=True)
|
||||
raise HTTPException(status_code=500, detail="Search failed")
|
||||
Reference in New Issue
Block a user