feat: add RAG search endpoint with content extraction
Build and Push / build (release) Successful in 1m2s
Build and Push / build (release) Successful in 1m2s
- Add /rag/search endpoint for web, news, and image search via SearXNG - Add /content/extract and /content/extract/batch endpoints - Add ContentExtractor client using Trafilatura for content extraction - Enhance HybridRAG web search with full content extraction - Add Redis caching for search results - Add new configuration options for search and extraction timeouts 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,88 @@
|
||||
"""
|
||||
RAG search models for Library Desk.
|
||||
|
||||
Pydantic models for web/news/image search requests and responses.
|
||||
"""
|
||||
|
||||
from enum import Enum
|
||||
from typing import Optional, List
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from src.core.multi_tenancy import DEFAULT_USER
|
||||
|
||||
|
||||
class SearchType(str, Enum):
|
||||
"""Supported search types."""
|
||||
WEB = "web"
|
||||
NEWS = "news"
|
||||
IMAGES = "images"
|
||||
|
||||
|
||||
class RAGSearchRequest(BaseModel):
|
||||
"""Request for RAG search endpoint."""
|
||||
|
||||
query: str = Field(
|
||||
...,
|
||||
min_length=1,
|
||||
max_length=500,
|
||||
description="The search query"
|
||||
)
|
||||
search_type: SearchType = Field(
|
||||
default=SearchType.WEB,
|
||||
description="Type of search: web, news, or images"
|
||||
)
|
||||
limit: int = Field(
|
||||
default=10,
|
||||
ge=1,
|
||||
le=20,
|
||||
description="Maximum number of results (1-20)"
|
||||
)
|
||||
user: str = Field(
|
||||
default=DEFAULT_USER,
|
||||
description="User identifier for rate limiting/personalization"
|
||||
)
|
||||
|
||||
|
||||
class RAGSearchResult(BaseModel):
|
||||
"""A single search result with extracted content."""
|
||||
|
||||
title: str = Field(..., description="Title of the result")
|
||||
url: str = Field(..., description="URL of the source")
|
||||
content: str = Field(
|
||||
"",
|
||||
description="Full extracted text via Trafilatura (max ~2000 chars)"
|
||||
)
|
||||
snippet: str = Field(
|
||||
"",
|
||||
description="Original search engine snippet (150-300 chars)"
|
||||
)
|
||||
source: str = Field(..., description="Domain name of the source")
|
||||
published_date: Optional[str] = Field(
|
||||
None,
|
||||
description="Publication date in ISO format if available"
|
||||
)
|
||||
|
||||
|
||||
class RAGSearchResponse(BaseModel):
|
||||
"""Response from RAG search endpoint."""
|
||||
|
||||
query: str = Field(..., description="Echo of the original query")
|
||||
search_type: SearchType = Field(..., description="Type of search performed")
|
||||
results: List[RAGSearchResult] = Field(
|
||||
default_factory=list,
|
||||
description="List of search results with extracted content"
|
||||
)
|
||||
total_results: int = Field(
|
||||
...,
|
||||
ge=0,
|
||||
description="Number of results returned"
|
||||
)
|
||||
search_time_ms: int = Field(
|
||||
...,
|
||||
ge=0,
|
||||
description="Total time for search and content extraction"
|
||||
)
|
||||
sources_summary: str = Field(
|
||||
"",
|
||||
description="Markdown-formatted list of all source URLs"
|
||||
)
|
||||
Reference in New Issue
Block a user