Compare commits
@@ -0,0 +1,25 @@
|
||||
# Service URLs for local dev (pointing to your server)
|
||||
TEST_HOST=192.168.86.149
|
||||
WIKIJS_URL=http://192.168.86.149:8088
|
||||
NEO4J_URI=bolt://192.168.86.149:7687
|
||||
QDRANT_HOST=192.168.86.149
|
||||
QDRANT_PORT=6333
|
||||
OLLAMA_URL=http://192.168.86.149:11434
|
||||
SEARXNG_URL=http://192.168.86.149:8080
|
||||
REDIS_HOST=192.168.86.149
|
||||
PAPERLESS_URL=http://192.168.86.149:8091
|
||||
|
||||
OLLAMA_MODEL=mistral-nemo-large:latest
|
||||
OLLAMA_EMBEDDING_MODEL=nomic-embed-text
|
||||
|
||||
# Wiki.js auth
|
||||
WIKIJS_USERNAME=librarian@schweitz.net
|
||||
WIKIJS_PASSWORD=key_here
|
||||
# Wiki.js GraphQL API token (generate from Admin → API Access)
|
||||
WIKI_GRAPHQL_API=your_jwt_token_here
|
||||
|
||||
LIBRARY_API_KEY=key_here
|
||||
NEO4J_PASSWORD=key_here
|
||||
WIKIJS_DB_PASSWORD=key_here
|
||||
SCHEDULER_API_KEY=key_here
|
||||
PAPERLESS_TOKEN=key_here
|
||||
@@ -13,7 +13,7 @@ jobs:
|
||||
- name: Login to Gitea Registry
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
registry: git.schweitz.net
|
||||
registry: git.schweitz.internal
|
||||
username: ${{ secrets.REGISTRY_USER }}
|
||||
password: ${{ secrets.REGISTRY_PASSWORD }}
|
||||
|
||||
@@ -23,5 +23,11 @@ jobs:
|
||||
context: .
|
||||
push: true
|
||||
tags: |
|
||||
git.schweitz.net/jpmschweitzer/library-desk:latest
|
||||
git.schweitz.net/jpmschweitzer/library-desk:${{ github.ref_name }}
|
||||
git.schweitz.internal/jpmschweitzer/library-desk:latest
|
||||
git.schweitz.internal/jpmschweitzer/library-desk:${{ github.ref_name }}
|
||||
|
||||
- name: Trigger Watchtower update
|
||||
if: success()
|
||||
run: |
|
||||
curl -sf -H "Authorization: Bearer ${{ secrets.WATCHTOWER_TOKEN }}" \
|
||||
http://watchtower:8080/v1/update
|
||||
|
||||
@@ -0,0 +1,87 @@
|
||||
|
||||
# AGENTS.md
|
||||
|
||||
> **Start every session by reading this file.**
|
||||
> This file outlines the operational protocols, coding standards, and architectural decisions for this FastAPI project.
|
||||
|
||||
## 1. Agent Operational Protocols
|
||||
|
||||
### 🧠 Work Patterns (Plan-Act-Reflect)
|
||||
* **Plan:** Before writing code, briefly outline your plan. Identify which files you will touch and what the side effects might be.
|
||||
* **Act:** Execute the changes in small, atomic steps.
|
||||
* **Reflect:** After coding, verify your work. Did you break existing tests? Did you add new tests?
|
||||
|
||||
### 🛡️ Git Discipline
|
||||
* **NEVER commit to `main` or `master` directly.** Always create a feature branch: `feature/your-feature-name` or `fix/issue-description`.
|
||||
* **Commit Messages:** Use the [Conventional Commits](https://www.conventionalcommits.org/) format.
|
||||
* `feat: add user login endpoint`
|
||||
* `fix: resolve database connection timeout`
|
||||
* `refactor: split monolith dependency file`
|
||||
* **Atomic Commits:** Keep commits small. One logical change = one commit.
|
||||
|
||||
### 📝 Changelog Maintenance
|
||||
* **Update `CHANGELOG.md`** with every user-facing change.
|
||||
* Format: `## [Unreleased] - YYYY-MM-DD` followed by `### Added`, `### Changed`, or `### Fixed`.
|
||||
|
||||
### 🚀 Release Flow
|
||||
When changes are ready for deployment:
|
||||
|
||||
1. **Ask user if deploy cycle is desired **
|
||||
|
||||
2. **Update version** in `pyproject.toml`:
|
||||
- Bug fixes: bump patch version (1.8.3 → 1.8.4)
|
||||
- New features: bump minor version (1.8.4 → 1.9.0)
|
||||
|
||||
3. **Update CHANGELOG.md**:
|
||||
- Move items from `[Unreleased]` to new version section
|
||||
- Add release date: `## [1.8.4] - 2025-12-16`
|
||||
|
||||
4. **Commit and tag**:
|
||||
```bash
|
||||
git add -A
|
||||
git commit -m "fix: description of changes"
|
||||
git tag v1.8.4
|
||||
git push origin main --tags
|
||||
```
|
||||
|
||||
5. **CI/CD triggers automatically**:
|
||||
- Gitea CI builds Docker image on new tag
|
||||
- Watchtower pulls and deploys to production
|
||||
- Verify deployment: `curl http://192.168.86.149:8000/health`
|
||||
|
||||
---
|
||||
|
||||
### 🧪 Local Development Setup
|
||||
* **Always test locally first** before committing and deploying. The build-deploy loop is slow.
|
||||
* **Start the local server** with `./wakeup.sh` - logs are written to `logs/server.log` for easy tailing
|
||||
* **Auto-reload**: The wakeup script runs uvicorn in reload mode - code changes are picked up automatically without restart (except for requirements.txt changes)
|
||||
* **Test REST endpoints** against `http://localhost:8778` using curl or similar tools
|
||||
* **Only deploy** when a phase or feature is complete and tested locally
|
||||
* **Environment**: Copy `.env.example` to `.env` and configure for your local setup (Ollama, Redis, Neo4j, Qdrant, Wiki.js hosts)
|
||||
* **Running tests**: Always use the venv explicitly to avoid environment mismatches:
|
||||
```bash
|
||||
.venv/bin/python -m pytest tests/ # All tests
|
||||
.venv/bin/python -m pytest tests/ -v # Verbose output
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 2. FastAPI Architecture & Best Practices
|
||||
*Reference: [FastAPI Best Practices](https://github.com/zhanymkanov/fastapi-best-practices)*
|
||||
|
||||
### 📂 Project Structure (Directory-based, NOT File-type based)
|
||||
Do **not** group files by type (e.g., one huge `routers` folder). Group by **domain/module** inside a `src/` directory.
|
||||
|
||||
**Correct Structure:**
|
||||
```text
|
||||
src/
|
||||
├── auth/
|
||||
│ ├── router.py # Endpoints
|
||||
│ ├── schemas.py # Pydantic models
|
||||
│ ├── service.py # Business logic (CRUD, etc.)
|
||||
│ ├── dependencies.py# Module-specific dependencies
|
||||
│ └── config.py # Module-specific settings
|
||||
├── posts/
|
||||
│ ├── router.py
|
||||
│ └── ...
|
||||
└── main.py # App entry point
|
||||
+348
@@ -0,0 +1,348 @@
|
||||
# Changelog
|
||||
|
||||
All notable changes to Library Desk will be documented in this file.
|
||||
|
||||
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
||||
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
||||
|
||||
## [1.4.8] - 2025-12-25
|
||||
|
||||
### Added
|
||||
|
||||
- **Paperless Orphan Cleanup** - `POST /maintenance/cleanup/paperless` endpoint
|
||||
- Detects documents deleted from Paperless but still indexed in Library Desk
|
||||
- Removes orphaned vectors and graph nodes
|
||||
- Supports `dry_run=true` for preview mode
|
||||
|
||||
## [1.4.7] - 2025-12-25
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Paperless Custom Field Update** - Fixed 400 error when marking documents as indexed
|
||||
- Paperless API requires field ID (integer) not field name (string)
|
||||
- Now looks up `library_indexed` field ID before updating
|
||||
- Webhook params format: `doc_url` and `title` from Jinja templates
|
||||
|
||||
### Added
|
||||
|
||||
- **Webhook Debug Endpoint** - `POST /documents/webhook-capture` for development testing
|
||||
|
||||
## [1.4.6] - 2025-12-25
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Paperless Webhook Payload Format** - Updated model to match Paperless `include_document=true` format
|
||||
- Paperless sends `id` instead of `document_id`
|
||||
- Paperless sends full document data including `content`, `title`, `tags`, etc.
|
||||
- Webhook now uses content from payload, skipping extra Paperless API call
|
||||
- Added `extra = "ignore"` to handle additional Paperless fields
|
||||
|
||||
## [1.4.5] - 2025-12-25
|
||||
|
||||
### Added
|
||||
|
||||
- **Document Storage Integration** - Paperless-ngx integration for PDFs, images, and documents
|
||||
- Event-driven architecture via Paperless webhooks
|
||||
- `POST /documents/webhook` - Receive document events from Paperless workflows
|
||||
- `POST /documents/upload` - Upload files directly to Paperless
|
||||
- `POST /documents/upload-url` - Download and upload documents from URL
|
||||
- `POST /documents/search` - Semantic search across indexed documents
|
||||
- `GET /documents/health` - Paperless connectivity health check
|
||||
- **DocumentSyncService** - Indexes Paperless documents into vectors and graph
|
||||
- Fetches document content via Paperless API
|
||||
- Chunks text and generates embeddings for Qdrant
|
||||
- Creates Document nodes in Neo4j knowledge graph
|
||||
- Supports multi-tenancy via user parameter in webhook URL
|
||||
- **PaperlessClient** - REST API client for Paperless-ngx
|
||||
- Document retrieval, upload, and update operations
|
||||
- Health check support
|
||||
- **Paperless Workflow Configuration**
|
||||
- Production workflow: Document Added (NOT tagged llm-test) → webhook to Library Desk
|
||||
- Test workflow: Document Added (tagged llm-test) → webhook with test user
|
||||
|
||||
### Changed
|
||||
|
||||
- Updated `src/config.py` with Paperless configuration settings
|
||||
- Added `PaperlessDep` dependency injection for document endpoints
|
||||
|
||||
## [1.4.4] - 2025-12-24
|
||||
|
||||
### Added
|
||||
|
||||
- **Test Data Cleanup Endpoint** - `POST /maintenance/cleanup/test-data`
|
||||
- Purges LLM test data from wiki, graph, and vectors
|
||||
- Security-restricted to test user namespace only (`users/llm-tester/*`, `users/llm_tester/*`)
|
||||
- Supports `dry_run=true` (default) to preview before deleting
|
||||
- Scheduler task configured for weekly cleanup (Sunday 3:00 AM)
|
||||
|
||||
## [1.4.3] - 2025-12-24
|
||||
|
||||
### Changed
|
||||
|
||||
- **Volatile Cache System Refactored to Vector Storage**
|
||||
- Backend migrated from Redis to Qdrant for semantic search capability
|
||||
- Data converted to natural language for embedding and semantic retrieval
|
||||
- Collection naming: `volatile_{user}` for per-user isolation
|
||||
- TTL implemented via `ttl_expiry` timestamp in vector payload
|
||||
- Simplified endpoints:
|
||||
- `GET /volatile/search?q=...` - Semantic search across volatile data
|
||||
- `POST /volatile/store?namespace=...&key=...` - Store with query params
|
||||
- `GET /volatile/{namespace}/{key}` - Get specific record
|
||||
- `DELETE /volatile/{namespace}/{key}` - Delete record
|
||||
- Removed namespace-specific URL patterns (simpler API for LLM tool use)
|
||||
|
||||
### Added
|
||||
|
||||
- **HybridRAG Volatile Integration** - Volatile cache now included in multi-source search
|
||||
- Volatile results get priority boost in RRF fusion (current data ranks higher)
|
||||
- New config options: `enable_volatile`, `volatile_limit` (default 1), `volatile_threshold`
|
||||
- Timing breakdown includes `volatile_ms`
|
||||
- **Volatile Cleanup Endpoint** - `POST /maintenance/cleanup/volatile`
|
||||
- Purges expired records across all `volatile_*` collections
|
||||
- Scheduler task for every 10 minutes recommended
|
||||
- Returns per-collection cleanup counts
|
||||
- **Natural Language Conversion** - Structured data converted for embedding
|
||||
- Template-based conversion for each namespace (weather, news, financial, etc.)
|
||||
- Fallback for custom namespaces
|
||||
|
||||
## [1.4.2] - 2025-12-24
|
||||
|
||||
### Added
|
||||
|
||||
- **Volatile Cache System** - Ephemeral data storage with TTL
|
||||
- `GET /volatile/{namespace}/{key}` - Retrieve cached record
|
||||
- `POST /volatile/{namespace}/{key}` - Store/update record with TTL
|
||||
- `DELETE /volatile/{namespace}/{key}` - Remove record
|
||||
- `GET /volatile/{namespace}` - List keys in namespace
|
||||
- `DELETE /volatile/{namespace}` - Clear all records in namespace
|
||||
- `GET /volatile/stats` - Cache statistics by namespace
|
||||
- `GET /volatile/scheduled` - Records needing refresh (for scheduler)
|
||||
- `GET /volatile/namespaces` - List available namespaces with default TTLs
|
||||
- **Volatile Namespaces** - Predefined categories with appropriate TTLs:
|
||||
- `weather` (30min) - Weather conditions and forecasts
|
||||
- `news` (1hr) - Headlines and breaking news
|
||||
- `financial` (5min) - Stock prices, exchange rates
|
||||
- `transit` (5min) - Train/bus schedules, delays
|
||||
- `traffic` (10min) - Commute times, road conditions
|
||||
- `air_quality` (1hr) - Pollution, pollen counts
|
||||
- `sports` (1min) - Live scores, matches
|
||||
- `social` (10min) - Social notifications
|
||||
- `system` (1min) - Service health status
|
||||
- `context` (1hr) - Session state
|
||||
- `custom` (1hr) - User-defined data
|
||||
- **Refresh Schedule Support** - Optional cron expressions for scheduler integration
|
||||
|
||||
## [1.4.1] - 2025-12-24
|
||||
|
||||
### Fixed
|
||||
|
||||
- Wiki.js API token now optional - GraphQL API works without authentication
|
||||
- Container startup failure when `WIKI_GRAPHQL_API` env var not set
|
||||
|
||||
## [1.4.0] - 2025-12-24
|
||||
|
||||
### Added
|
||||
|
||||
- **Maintenance Router** - New `/maintenance` endpoints for system health and cleanup
|
||||
- `GET /maintenance/health` - Lightweight health check (detailed mode available)
|
||||
- `POST /maintenance/cleanup/all` - Full orphan cleanup (vectors + graph)
|
||||
- `POST /maintenance/cleanup/vectors` - Purge orphan vector chunks
|
||||
- `POST /maintenance/cleanup/graph` - Purge orphan graph nodes
|
||||
- `POST /maintenance/reconcile-index` - Combined cleanup + reindex missing pages
|
||||
- **Bidirectional Orphan Detection** - Cross-validate vectors and graph nodes
|
||||
- `find_documents_without_vectors()` - Graph nodes missing vector chunks
|
||||
- `find_chunks_without_graph_nodes()` - Vector chunks missing graph nodes
|
||||
- **Qdrant Client Methods** - Bulk operations for maintenance
|
||||
- `scroll_all_points()` - Iterate all points with pagination
|
||||
- `delete_by_ids()` - Batch delete by point IDs
|
||||
- **Graph Service Cleanup** - Node deletion methods
|
||||
- `delete_document_node()` - Remove document and relationships
|
||||
- `delete_collection_node()` - Remove collection and contained documents
|
||||
- `get_all_document_references()` - Get all document references for validation
|
||||
- **Redis Timestamp Tracking** - `last_cleanup` timestamp for scheduler integration
|
||||
- **Memory System Plan** - Documented three-tier architecture (volatile/documents/knowledge)
|
||||
|
||||
### Changed
|
||||
|
||||
- **Wiki.js Authentication** - Switched from username/password to API token
|
||||
- New `WIKI_GRAPHQL_API` environment variable for JWT token
|
||||
- Deprecated `WIKIJS_USERNAME` and `WIKIJS_PASSWORD` (kept for backwards compatibility)
|
||||
- **Service Dependencies** - Added `VectorServiceDep` and `GraphServiceDep` type aliases
|
||||
|
||||
### Fixed
|
||||
|
||||
- Wiki.js client now properly handles API token auth without login flow
|
||||
|
||||
## [1.3.3] - 2025-12-23
|
||||
|
||||
### Added
|
||||
|
||||
- Temperature parameter to `OllamaClient.generate_text()` for controlling output determinism
|
||||
- `TODO.md` tracking remaining stub endpoints to implement
|
||||
- Wired `/query/semantic` endpoint to VectorService
|
||||
- Wired `/query/graph` endpoint to GraphService
|
||||
|
||||
### Changed
|
||||
|
||||
- **Improved LLM prompts** based on llm-findings.md recommendations:
|
||||
- Keyword extraction: temperature 0.0, negative constraints
|
||||
- LLM re-ranking: temperature 0.0, explicit rules
|
||||
- Conflict detection: temperature 0.0, analysis steps (CoT)
|
||||
- Wiki page creation: temperature 0.3, anti-hallucination constraints
|
||||
- Page reconstruction: temperature 0.2, preservation constraints
|
||||
- Web results analysis: temperature 0.0, conservative approach
|
||||
- Test fixtures now use configurable host (TEST_HOST) instead of Docker hostnames
|
||||
|
||||
### Removed
|
||||
|
||||
- Dead code: unused `get_default_user()` function
|
||||
- Unused imports from routers (wiki.py, graph.py, hybrid_rag.py)
|
||||
- Stub endpoints shadowed by real implementations (/stats, /ingest/document, /ingest/batch)
|
||||
|
||||
## [1.3.2] - 2025-12-22
|
||||
|
||||
### Changed
|
||||
|
||||
- **Consolidated Ollama model configuration** - All LLM operations now use single `OLLAMA_MODEL` environment variable
|
||||
- Removed separate `reranker_model` setting
|
||||
- HybridRAG re-ranking, consolidation analysis, and wiki page writing all use the same model
|
||||
- Improves VRAM efficiency by keeping one model hot
|
||||
- Added `OLLAMA_EMBEDDING_MODEL` environment variable for embedding model (previously overloaded `OLLAMA_MODEL`)
|
||||
- Updated WikiPageWriter to accept settings instead of hardcoded model name
|
||||
|
||||
## [1.3.1] - 2025-12-16
|
||||
|
||||
### Fixed
|
||||
|
||||
- Smart create endpoint missing `content_extractor` dependency causing 500 errors on `POST /wiki/pages/smart-create`
|
||||
|
||||
## [1.3.0] - 2025-12-15
|
||||
|
||||
### Changed
|
||||
|
||||
- **Two-Stage RRF Architecture** - Major refactor to level the playing field between wiki and web results
|
||||
- Stage 1: Vector and graph results merged into single "wiki" ranking using mini-RRF
|
||||
- Stage 2: Final RRF between wiki (single source) and web (single source)
|
||||
- Wiki pages no longer get 2x advantage from appearing in both vector and graph searches
|
||||
- Multi-source confirmation still determines wiki internal ranking
|
||||
|
||||
- **Skip synonyms in graph search** - LLM-generated synonyms (e.g., "author") no longer match unrelated graph entities (e.g., "author2000")
|
||||
- Vector search still uses synonyms for semantic similarity
|
||||
- Graph search uses only core keywords for exact entity matching
|
||||
|
||||
### Added
|
||||
|
||||
- `VECTOR_SIMILARITY_THRESHOLD` config setting (default: 0.7) to filter weak vector matches
|
||||
- Deduplication in graph search to prevent same document appearing multiple times
|
||||
|
||||
### Fixed
|
||||
|
||||
- Graph search duplicate entity bug where same document could appear twice if entity linked multiple times
|
||||
|
||||
## [1.2.1] - 2025-12-15
|
||||
|
||||
### Fixed
|
||||
|
||||
- HybridRAG router missing `content_extractor` dependency causing 500 errors on `/query/hybrid` endpoint
|
||||
|
||||
## [1.2.0] - 2025-12-15
|
||||
|
||||
### Added
|
||||
|
||||
- **RAG Search Endpoint** (`POST /rag/search`)
|
||||
- Web, news, and image search via SearXNG
|
||||
- Full content extraction using Trafilatura (F1 score 0.958)
|
||||
- Redis caching with configurable TTL
|
||||
- Markdown sources summary for LLM consumption
|
||||
- Returns both extracted content and original snippets
|
||||
|
||||
- **Content Extraction Endpoints** (`/content/*`)
|
||||
- `POST /content/extract` - Extract content from a single URL
|
||||
- `POST /content/extract/batch` - Batch extraction (up to 20 URLs)
|
||||
- Reusable ContentExtractor client for use across the codebase
|
||||
|
||||
- **HybridRAG Content Extraction Enhancement**
|
||||
- Web search results now include full extracted content via Trafilatura
|
||||
- Falls back to original snippets if extraction fails
|
||||
- Improves context quality for LLM re-ranking and consumption
|
||||
|
||||
### Changed
|
||||
|
||||
- Added new configuration options:
|
||||
- `SEARCH_CACHE_TTL` - Search cache TTL in seconds (default: 300)
|
||||
- `SEARCH_TIMEOUT` - SearXNG timeout (default: 10s)
|
||||
- `CONTENT_EXTRACTION_TIMEOUT` - Per-URL extraction timeout (default: 5s)
|
||||
- `CONTENT_MAX_LENGTH` - Max extracted content length (default: 2000)
|
||||
- `SEARCH_DEFAULT_LIMIT` - Default search results (default: 10)
|
||||
|
||||
### Dependencies
|
||||
|
||||
- Added `trafilatura~=1.12.0` for content extraction
|
||||
|
||||
## [1.1.3] - 2025-12-14
|
||||
|
||||
### Added
|
||||
|
||||
- Watchtower update trigger in Gitea workflow after successful build
|
||||
|
||||
## [1.1.2] - 2025-12-14
|
||||
|
||||
### Fixed
|
||||
|
||||
- Updated registry login URL in Gitea workflow (git.schweitz.net → git.schweitz.internal)
|
||||
|
||||
## [1.1.1] - 2025-12-14
|
||||
|
||||
### Fixed
|
||||
|
||||
- Updated container registry tag URLs in Gitea workflow (git.schweitz.net → git.schweitz.internal)
|
||||
|
||||
### Added
|
||||
|
||||
- Tests for Smart Page Creation feature (`test_smart_create.py`)
|
||||
- Model validation tests for WikiSmartCreateRequest/Response
|
||||
- WikiService.smart_create_page method tests
|
||||
- Bidirectional entity linking utility tests
|
||||
- Endpoint validation tests
|
||||
|
||||
## [1.1.0] - 2025-12-11
|
||||
|
||||
### Added
|
||||
|
||||
- **Smart Page Creation Endpoint** (`POST /wiki/pages/smart-create`)
|
||||
- Combines HybridRAG research with LLM content generation
|
||||
- Searches existing wiki, knowledge graph, and web for topic context
|
||||
- Uses WikiPageWriter to synthesize findings into structured wiki content
|
||||
- Auto-generates page path from topic if not provided
|
||||
- Returns research summary with source counts
|
||||
|
||||
- **Bidirectional Entity Linking**
|
||||
- New shared utility (`entity_linking_utils.py`) for reusable entity linking
|
||||
- Forward links: Links entities mentioned in new pages to existing entity pages
|
||||
- Backward links: Updates existing pages that mention the new entity
|
||||
- Runs automatically in background after smart page creation
|
||||
|
||||
- **Version Management**
|
||||
- Added `pyproject.toml` with project metadata and version
|
||||
- Version is now read from `pyproject.toml` (single source of truth)
|
||||
- Health check endpoint returns current version
|
||||
- FastAPI docs show current version
|
||||
|
||||
### Changed
|
||||
|
||||
- Updated `config.py` to read version from `pyproject.toml`
|
||||
- Updated `main.py` to use centralized version
|
||||
|
||||
## [1.0.0] - 2025-12-10
|
||||
|
||||
### Added
|
||||
|
||||
- Initial release extracted from portainer-core
|
||||
- Wiki page management (`/wiki/pages` CRUD endpoints)
|
||||
- HybridRAG search (`/query/hybrid`) with vector, graph, and web search
|
||||
- Knowledge graph operations (`/graph/*`)
|
||||
- Vector search operations (`/vector/*`)
|
||||
- Knowledge consolidation from search results (`/consolidate/knowledge`)
|
||||
- Entity linking and extraction
|
||||
- Wiki.js change listener for auto-processing user edits
|
||||
- Multi-tenant architecture with user namespace isolation
|
||||
@@ -0,0 +1,17 @@
|
||||
# Claude Code Instructions
|
||||
|
||||
**MANDATORY: Read AGENTS.md instead of this file.**
|
||||
|
||||
This project uses a unified configuration file for all LLM coding agents.
|
||||
|
||||
## Instructions
|
||||
|
||||
1. **Read and follow AGENTS.md** - All project guidelines are located there
|
||||
2. **Do not modify this file** - Only update AGENTS.md
|
||||
3. **Do not create or modify other agent-specific files** - Use AGENTS.md as the single source of truth
|
||||
|
||||
This approach ensures consistent behavior across all LLM coding agents without managing separate configuration files.
|
||||
|
||||
---
|
||||
|
||||
If you need to update project guidelines, edit AGENTS.md, not this file.
|
||||
@@ -12,6 +12,7 @@ COPY requirements.txt .
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
# Copy application
|
||||
COPY pyproject.toml .
|
||||
COPY src/ ./src/
|
||||
COPY static/ ./static/
|
||||
|
||||
|
||||
@@ -455,6 +455,154 @@ LIBRARY_BATCH_SIZE=50
|
||||
LIBRARY_SYNC_ENABLED=true
|
||||
```
|
||||
|
||||
## Maintenance Tasks
|
||||
|
||||
### Index Reconciliation (Daily)
|
||||
|
||||
The `reconcile-index` endpoint performs full index maintenance:
|
||||
|
||||
1. **Cleanup Phase**: Remove orphaned data
|
||||
- Vector chunks without wiki source
|
||||
- Graph nodes without vectors (bidirectional)
|
||||
- Vectors without graph nodes (bidirectional)
|
||||
- Orphan entities (no MENTIONS relationships)
|
||||
- Broken relationships
|
||||
|
||||
2. **Reindex Phase**: Index missing pages
|
||||
- Wiki pages without vector embeddings
|
||||
- Wiki pages without graph Document nodes
|
||||
|
||||
**Scheduler Task: `library_reconcile_index`**
|
||||
|
||||
```yaml
|
||||
Task Name: library_reconcile_index
|
||||
Description: Daily index reconciliation - cleanup orphans + reindex missing pages
|
||||
Schedule: Daily at 04:00 (after library_sync at 03:30)
|
||||
Priority: 10 (system maintenance)
|
||||
Service: library
|
||||
Executor: POST /maintenance/reconcile-index
|
||||
Configuration:
|
||||
- LIBRARY_DESK_URL: http://library-desk:8089
|
||||
- LIBRARY_API_KEY: ${LIBRARY_API_KEY}
|
||||
Parameters:
|
||||
- user: jpmschweitzer
|
||||
- dry_run: false
|
||||
Outputs:
|
||||
- Vector orphans purged
|
||||
- Entity orphans purged
|
||||
- Missing pages reindexed
|
||||
```
|
||||
|
||||
### Maintenance Endpoints
|
||||
|
||||
| Endpoint | Method | Purpose |
|
||||
|----------|--------|---------|
|
||||
| `/maintenance/reconcile-index` | POST | **Recommended**: Full cleanup + reindex missing |
|
||||
| `/maintenance/cleanup/all` | POST | Cleanup only (orphan removal) |
|
||||
| `/maintenance/cleanup/vectors` | POST | Clean orphan vector chunks only |
|
||||
| `/maintenance/cleanup/graph` | POST | Clean orphan entities & stale docs only |
|
||||
| `/maintenance/health` | GET | Lightweight health check (for uptime monitoring) |
|
||||
| `/maintenance/health?detailed=true` | GET | Full analysis with orphan counts |
|
||||
| `/maintenance/reindex/{page_id}` | POST | Force re-index a specific page |
|
||||
|
||||
### Health Check Modes
|
||||
|
||||
**Lightweight (default)** - Use for frequent uptime checks (every 30s):
|
||||
```bash
|
||||
curl "http://library-desk:8089/maintenance/health?user=jpmschweitzer" \
|
||||
-H "Authorization: Bearer ${LIBRARY_API_KEY}"
|
||||
```
|
||||
|
||||
Returns only last cleanup timestamp and basic status (no database queries).
|
||||
|
||||
**Detailed** - Use for dashboards or before reconciliation:
|
||||
```bash
|
||||
curl "http://library-desk:8089/maintenance/health?user=jpmschweitzer&detailed=true" \
|
||||
-H "Authorization: Bearer ${LIBRARY_API_KEY}"
|
||||
```
|
||||
|
||||
Returns full orphan analysis (runs database queries).
|
||||
|
||||
### Example Reconcile Request
|
||||
|
||||
```bash
|
||||
curl -X POST "http://library-desk:8089/maintenance/reconcile-index?user=jpmschweitzer" \
|
||||
-H "Authorization: Bearer ${LIBRARY_API_KEY}"
|
||||
```
|
||||
|
||||
### Example Response
|
||||
|
||||
```json
|
||||
{
|
||||
"success": true,
|
||||
"cleanup": {
|
||||
"success": true,
|
||||
"vector_cleanup": {
|
||||
"wiki_chunks": {"orphans_found": 5, "orphans_purged": 5},
|
||||
"document_chunks": {"orphans_found": 0, "orphans_purged": 0},
|
||||
"chunks_without_graph": {"orphans_found": 2, "orphans_purged": 2},
|
||||
"total_chunks_scanned": 1250,
|
||||
"total_orphans_purged": 7
|
||||
},
|
||||
"graph_cleanup": {
|
||||
"orphan_entities": {"orphans_found": 3, "orphans_purged": 3},
|
||||
"stale_wiki_documents": {"orphans_found": 1, "orphans_purged": 1},
|
||||
"stale_store_documents": {"orphans_found": 0, "orphans_purged": 0},
|
||||
"docs_without_vectors": {"orphans_found": 0, "orphans_purged": 0},
|
||||
"broken_relationships_cleaned": 0
|
||||
},
|
||||
"total_duration_ms": 1523.5
|
||||
},
|
||||
"reindex_missing": {
|
||||
"pages_without_vectors": 2,
|
||||
"pages_without_graph": 1,
|
||||
"pages_reindexed": 2,
|
||||
"pages_failed": 0,
|
||||
"failed_page_ids": [],
|
||||
"duration_ms": 3421.2
|
||||
},
|
||||
"total_duration_ms": 4944.7
|
||||
}
|
||||
```
|
||||
|
||||
### Scheduler Integration Code
|
||||
|
||||
```python
|
||||
# scheduler/src/tasks/library_maintenance.py
|
||||
|
||||
async def library_reconcile_index_task(user: str = "jpmschweitzer"):
|
||||
"""Run daily Library Desk index reconciliation."""
|
||||
|
||||
async with httpx.AsyncClient() as client:
|
||||
# Run reconcile-index (cleanup + reindex missing)
|
||||
result = await client.post(
|
||||
f"{LIBRARY_DESK_URL}/maintenance/reconcile-index",
|
||||
params={"user": user, "dry_run": False},
|
||||
headers={"Authorization": f"Bearer {LIBRARY_API_KEY}"},
|
||||
timeout=600.0 # 10 minutes for large indexes
|
||||
)
|
||||
|
||||
data = result.json()
|
||||
|
||||
# Log summary
|
||||
cleanup = data["cleanup"]
|
||||
reindex = data["reindex_missing"]
|
||||
|
||||
logger.info(
|
||||
f"Reconcile complete: "
|
||||
f"{cleanup['vector_cleanup']['total_orphans_purged']} vector orphans, "
|
||||
f"{cleanup['graph_cleanup']['orphan_entities']['orphans_purged']} entity orphans, "
|
||||
f"{reindex['pages_reindexed']} pages reindexed"
|
||||
)
|
||||
|
||||
if reindex["pages_failed"] > 0:
|
||||
logger.warning(f"Failed to reindex pages: {reindex['failed_page_ids']}")
|
||||
|
||||
return data
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Next Steps
|
||||
|
||||
1. Implement ingestion endpoints in Library Desk
|
||||
|
||||
@@ -49,7 +49,8 @@ QDRANT_PORT=6333
|
||||
WIKIJS_URL=http://wiki:3000
|
||||
SEARXNG_URL=http://searxng:8080
|
||||
OLLAMA_URL=http://ollama:11434
|
||||
OLLAMA_MODEL=nomic-embed-text
|
||||
OLLAMA_MODEL=mistral-nemo-large:latest
|
||||
OLLAMA_EMBEDDING_MODEL=nomic-embed-text
|
||||
REDIS_HOST=redis-shared
|
||||
REDIS_PORT=6379
|
||||
REDIS_DB=2
|
||||
|
||||
@@ -0,0 +1,52 @@
|
||||
# TODO
|
||||
|
||||
Outstanding work items for Library Desk.
|
||||
|
||||
## Stub Endpoints to Implement
|
||||
|
||||
The following endpoints in `src/main.py` return stub responses and need real implementations:
|
||||
|
||||
### Ingestion Status Endpoints
|
||||
|
||||
#### `POST /ingest/check-updates`
|
||||
Check which documents need updating based on content hashes. Used by Scheduler to determine what changed since last sync.
|
||||
|
||||
**Implementation needed:**
|
||||
1. Query existing documents by path
|
||||
2. Compare content hashes
|
||||
3. Return list of updates needed
|
||||
|
||||
#### `GET /ingest/status/{document_id}`
|
||||
Get processing status for a document.
|
||||
|
||||
**Implementation needed:**
|
||||
- Status tracking system (Redis or database)
|
||||
- Track ingestion progress per document
|
||||
|
||||
#### `GET /ingest/repo-status/{repository}`
|
||||
Get indexing status for an entire repository.
|
||||
|
||||
**Implementation needed:**
|
||||
- Repository-level statistics
|
||||
- Track which documents from a repo are indexed
|
||||
|
||||
### Deduplication
|
||||
|
||||
#### `POST /deduplicate/check`
|
||||
Check for duplicate or highly similar documents using vector similarity and graph analysis.
|
||||
|
||||
**Implementation needed:**
|
||||
1. Get document embedding from Qdrant
|
||||
2. Find similar vectors above threshold
|
||||
3. Check graph relationships
|
||||
4. Return candidates with similarity scores
|
||||
|
||||
## System Statistics
|
||||
|
||||
#### `GET /stats`
|
||||
Get system statistics (wiki pages, neo4j nodes, qdrant vectors).
|
||||
|
||||
**Implementation needed:**
|
||||
- Query Neo4j for node count
|
||||
- Query Qdrant for vector count
|
||||
- Query Wiki.js for page count
|
||||
@@ -0,0 +1,509 @@
|
||||
# Phase 3: Document Storage System - Implementation Plan
|
||||
|
||||
## Overview
|
||||
|
||||
Document storage tier for Library Desk - storing and indexing PDFs, images, videos, and git documentation mirrors.
|
||||
|
||||
**User Decisions:**
|
||||
- Paperless-ngx container for OCR
|
||||
- Ebooks deferred to future phase
|
||||
- Video.js player deferred to after core implementation
|
||||
|
||||
| Phase | Status | Version |
|
||||
|-------|--------|---------|
|
||||
| Phase 1: Cleanup System | Complete | v1.4.0 |
|
||||
| Phase 2: Volatile Memory | Complete | v1.4.3 |
|
||||
| Phase 3: Document Storage | Planning | - |
|
||||
| Phase 4: Test Data Cleanup | Complete | v1.4.4 |
|
||||
|
||||
---
|
||||
|
||||
## Architecture
|
||||
|
||||
**Paperless-ngx as primary document store** (no SeaweedFS needed):
|
||||
|
||||
```
|
||||
┌─────────────────────────────────────────────────────────────────┐
|
||||
│ External Sources │
|
||||
│ ┌─────────┐ ┌────────────┐ ┌──────────────┐ │
|
||||
│ │ GitHub │ │ Direct │ │ Email/Folder │ │
|
||||
│ │ Docs │ │ Upload │ │ Ingestion │ │
|
||||
│ └────┬────┘ └─────┬──────┘ └──────┬───────┘ │
|
||||
└───────┼─────────────┼────────────────┼──────────────────────────┘
|
||||
│ │ │
|
||||
▼ ▼ ▼
|
||||
┌─────────────────────────────────────────────────────────────────┐
|
||||
│ Paperless-ngx │
|
||||
│ ┌───────────────────────────────────────────────────────────┐ │
|
||||
│ │ - Document storage (PDFs, images, videos) │ │
|
||||
│ │ - OCR via Tesseract (PDFs, images) │ │
|
||||
│ │ - Web UI for browsing/tagging │ │
|
||||
│ │ - REST API for integration │ │
|
||||
│ └─────────────────────────┬─────────────────────────────────┘ │
|
||||
└────────────────────────────┼────────────────────────────────────┘
|
||||
│ REST API (sync)
|
||||
▼
|
||||
┌─────────────────────────────────────────────────────────────────┐
|
||||
│ Library Desk │
|
||||
│ ┌───────────────────────────────────────────────────────────┐ │
|
||||
│ │ DocumentSyncService │ │
|
||||
│ │ - Polls Paperless for new/updated docs │ │
|
||||
│ │ - Extracts text + metadata via API │ │
|
||||
│ │ - Sends to vector/graph pipelines │ │
|
||||
│ └─────────────────────────┬─────────────────────────────────┘ │
|
||||
│ │ │
|
||||
│ ┌────────────────┼────────────────┐ │
|
||||
│ ▼ ▼ ▼ │
|
||||
│ ┌──────────┐ ┌──────────┐ ┌──────────┐ │
|
||||
│ │ Qdrant │ │ Neo4j │ │ Wiki.js │ │
|
||||
│ │ (vectors)│ │ (graph) │ │ (catalog)│ │
|
||||
│ └──────────┘ └──────────┘ └──────────┘ │
|
||||
└─────────────────────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
**File handling by type:**
|
||||
|
||||
| File Type | Paperless | Library Desk |
|
||||
|-----------|-----------|--------------|
|
||||
| PDFs | OCR → text | Index text → vectors/graph |
|
||||
| Images | OCR → text | Index text → vectors/graph |
|
||||
| Videos | Storage only | Index metadata → vectors/graph |
|
||||
|
||||
---
|
||||
|
||||
## Technology Stack
|
||||
|
||||
| Component | Purpose | Rationale |
|
||||
|-----------|---------|-----------|
|
||||
| **Paperless-ngx** | Document storage + OCR | All-in-one: storage, OCR, web UI, REST API |
|
||||
| **ClamAV** | Virus scanning | Host OS install, pyclamd integration, better isolation |
|
||||
| **PDF.js** | PDF viewer | Embeddable in Wiki.js (deferred) |
|
||||
|
||||
**Why Paperless-ngx as primary store:**
|
||||
- Eliminates need for separate blob storage (SeaweedFS/MinIO)
|
||||
- Built-in web UI for browsing and tagging
|
||||
- Tesseract OCR with 100+ language support
|
||||
- REST API for Library Desk integration
|
||||
- Handles videos as raw files (no OCR, but stored)
|
||||
- Email and folder watching for automatic ingestion
|
||||
- Active community, well-maintained
|
||||
|
||||
---
|
||||
|
||||
## Paperless-ngx API Deep Dive
|
||||
|
||||
### Authentication
|
||||
```
|
||||
POST /api/token/
|
||||
Body: {"username": "...", "password": "..."}
|
||||
Response: {"token": "..."}
|
||||
|
||||
Header: Authorization: Token <token>
|
||||
```
|
||||
|
||||
### Document Upload (for HybridRAG → Paperless)
|
||||
```
|
||||
POST /api/documents/post_document/
|
||||
Content-Type: multipart/form-data
|
||||
|
||||
Fields:
|
||||
- document (file, required)
|
||||
- title (string)
|
||||
- created (datetime)
|
||||
- correspondent (ID)
|
||||
- document_type (ID)
|
||||
- storage_path (ID)
|
||||
- tags (repeatable IDs)
|
||||
- custom_fields (JSON array)
|
||||
|
||||
Response: {"task_id": "uuid"}
|
||||
```
|
||||
|
||||
Track consumption: `GET /api/tasks/?task_id={uuid}` → returns document ID when complete
|
||||
|
||||
### Document Search
|
||||
```
|
||||
GET /api/documents/?query=search+terms # Full-text search
|
||||
GET /api/documents/?more_like_id=123 # Similarity search
|
||||
|
||||
Response includes __search_hit__:
|
||||
{
|
||||
"score": 0.95,
|
||||
"highlights": "<span>matched</span> text",
|
||||
"rank": 0
|
||||
}
|
||||
```
|
||||
|
||||
### Custom Field Filtering
|
||||
```
|
||||
GET /api/documents/?custom_field_query=field_name__operation=value
|
||||
|
||||
Operations:
|
||||
- exact, in, isnull, exists (all types)
|
||||
- icontains, istartswith, iendswith (text)
|
||||
- gt, gte, lt, lte, range (numeric/date)
|
||||
- contains (document links)
|
||||
```
|
||||
|
||||
### Bulk Operations
|
||||
```
|
||||
POST /api/documents/bulk_edit/
|
||||
{
|
||||
"documents": [1, 2, 3],
|
||||
"method": "add_tag|remove_tag|set_correspondent|set_document_type|merge|split|...",
|
||||
"parameters": {...}
|
||||
}
|
||||
```
|
||||
|
||||
### Webhooks (Push to Library Desk!)
|
||||
Paperless workflows can trigger webhooks on document events:
|
||||
|
||||
| Trigger | When | Available Data |
|
||||
|---------|------|----------------|
|
||||
| Consumption Started | Before OCR | file_path, source, filename |
|
||||
| Document Added | After OCR | content, tags, doc_type, correspondent, `{doc_url}` |
|
||||
| Document Updated | On change | Same as Added |
|
||||
| Scheduled | Time-based | Date offsets from document dates |
|
||||
|
||||
**Webhook Action**: POST to Library Desk endpoint with document data
|
||||
|
||||
### Organization Features
|
||||
|
||||
| Feature | Purpose | API Endpoint |
|
||||
|---------|---------|--------------|
|
||||
| Tags | Nested labels (5 levels deep) | `/api/tags/` |
|
||||
| Correspondents | Source/destination | `/api/correspondents/` |
|
||||
| Document Types | Classification | `/api/document_types/` |
|
||||
| Storage Paths | File organization | `/api/storage_paths/` |
|
||||
| Custom Fields | Extensible metadata | `/api/custom_fields/` |
|
||||
|
||||
### Custom Fields We Should Create
|
||||
| Field Name | Type | Purpose |
|
||||
|------------|------|---------|
|
||||
| `source_url` | URL | Original download URL (for HybridRAG uploads) |
|
||||
| `library_indexed` | Boolean | Sync status with Library Desk |
|
||||
| `library_doc_id` | Text | Library Desk document reference |
|
||||
| `collection` | Text | Logical grouping (e.g., "fastapi-docs") |
|
||||
|
||||
### External LLM Add-ons (Optional)
|
||||
Community tools exist for Ollama integration:
|
||||
- **[paperless-ai](https://github.com/clusterzx/paperless-ai)** - Auto-tagging, RAG chat
|
||||
- **[paperless-gpt](https://github.com/icereed/paperless-gpt)** - LLM-enhanced OCR, auto-titling
|
||||
|
||||
**Recommendation:** Skip these - Library Desk already has Ollama integration for:
|
||||
- Embedding (nomic-embed-text)
|
||||
- LLM analysis (mistral-nemo)
|
||||
- Entity extraction
|
||||
- HybridRAG
|
||||
|
||||
We'll do our own classification/tagging via Library Desk after sync.
|
||||
|
||||
---
|
||||
|
||||
## Virus Scanning Integration
|
||||
|
||||
**ClamAV daemon + pyclamd** (no third-party REST wrappers):
|
||||
|
||||
```
|
||||
┌─────────────────┐ ┌─────────────────┐ ┌─────────────────┐
|
||||
│ File Upload │────►│ Library Desk │────►│ ClamAV Daemon │
|
||||
│ (URL or file) │ │ (pyclamd) │ │ (clamd:3310) │
|
||||
└─────────────────┘ └────────┬────────┘ └─────────────────┘
|
||||
│
|
||||
┌────────────┴────────────┐
|
||||
▼ ▼
|
||||
┌──────────┐ ┌──────────┐
|
||||
│ Clean │ │ Infected │
|
||||
│ ✓ │ │ ✗ │
|
||||
└────┬─────┘ └────┬─────┘
|
||||
│ │
|
||||
▼ ▼
|
||||
Upload to Paperless Reject + Log
|
||||
```
|
||||
|
||||
### ClamAV Deployment (Host OS)
|
||||
|
||||
ClamAV runs on the host OS (not containerized) for better security isolation:
|
||||
|
||||
```bash
|
||||
# Installed via apt on Ubuntu/Debian
|
||||
# Config: /etc/clamav/clamd.conf
|
||||
# TCPSocket 3310
|
||||
# TCPAddr 0.0.0.0
|
||||
```
|
||||
|
||||
Benefits: scans outside container isolation, single virus DB, survives container restarts.
|
||||
|
||||
### Library Desk Integration
|
||||
```python
|
||||
# src/clients/clamav_client.py
|
||||
import pyclamd
|
||||
|
||||
class ClamAVClient:
|
||||
def __init__(self, host: str, port: int = 3310):
|
||||
self.cd = pyclamd.ClamdNetworkSocket(host, port)
|
||||
|
||||
async def scan_bytes(self, data: bytes) -> ScanResult:
|
||||
"""Scan file bytes, return clean/infected status."""
|
||||
result = self.cd.scan_stream(data)
|
||||
if result is None:
|
||||
return ScanResult(clean=True)
|
||||
return ScanResult(clean=False, virus_name=result['stream'][1])
|
||||
|
||||
def ping(self) -> bool:
|
||||
"""Health check."""
|
||||
return self.cd.ping()
|
||||
```
|
||||
|
||||
### Scan Points
|
||||
| Location | When | Action on Infected |
|
||||
|----------|------|-------------------|
|
||||
| `/documents/upload` | Before Paperless upload | Reject with 400, log threat |
|
||||
| HybridRAG web fetch | Before saving PDF | Skip file, log threat |
|
||||
| `/documents/webhook` | Optional re-scan | Quarantine in Paperless |
|
||||
|
||||
### Config Settings
|
||||
```python
|
||||
# src/config.py
|
||||
CLAMAV_HOST: str = "192.168.86.149" # Host OS IP (not container)
|
||||
CLAMAV_PORT: int = 3310
|
||||
CLAMAV_ENABLED: bool = True # Bypass for testing
|
||||
CLAMAV_TIMEOUT: int = 30 # seconds
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Integration Strategy
|
||||
|
||||
### Option A: Webhook Push (Preferred)
|
||||
```
|
||||
Paperless Workflow → POST webhook → Library Desk /documents/webhook
|
||||
```
|
||||
- Real-time indexing when documents added/updated
|
||||
- Configure in Paperless: Workflow → Document Added → Webhook Action
|
||||
- Library Desk receives document ID, fetches content via API
|
||||
|
||||
### Option B: Polling Pull (Fallback)
|
||||
```
|
||||
Scheduler → POST /documents/sync → Library Desk polls Paperless
|
||||
```
|
||||
- Periodic sync for missed webhooks or initial bulk import
|
||||
- Track `library_indexed` custom field to skip already-processed docs
|
||||
|
||||
### Option C: HybridRAG Upload (New!)
|
||||
```
|
||||
HybridRAG web search → finds PDF → POST to Paperless → webhook → indexed
|
||||
```
|
||||
- When HybridRAG finds a relevant PDF/document in web results
|
||||
- Download and upload to Paperless with `source_url` custom field
|
||||
- Paperless OCRs it, triggers webhook, Library Desk indexes
|
||||
|
||||
---
|
||||
|
||||
## Library Desk API Design
|
||||
|
||||
### Documents Router (`/documents`)
|
||||
|
||||
| Endpoint | Method | Purpose |
|
||||
|----------|--------|---------|
|
||||
| `/documents/webhook` | POST | Receive Paperless webhook (Document Added/Updated) |
|
||||
| `/documents/sync` | POST | Pull new/updated docs from Paperless → index |
|
||||
| `/documents/upload` | POST | Upload file to Paperless (for HybridRAG) |
|
||||
| `/documents/sync-from-git` | POST | Pull docs from Gitea → upload to Paperless → index |
|
||||
| `/documents/{document_id}` | GET | Get document metadata |
|
||||
| `/documents/{document_id}/text` | GET | Get extracted text |
|
||||
| `/documents/search` | POST | Semantic search across documents |
|
||||
| `/documents/collection/{name}` | GET | List documents in collection |
|
||||
| `/documents/collection/{name}/catalog` | POST | Generate wiki catalog page |
|
||||
|
||||
**Upload flow (HybridRAG → Paperless):**
|
||||
1. HybridRAG finds PDF in web results
|
||||
2. POST `/documents/upload` with URL or file
|
||||
3. Library Desk downloads, uploads to Paperless with metadata
|
||||
4. Returns task_id for async tracking
|
||||
5. Paperless webhook triggers indexing when OCR complete
|
||||
|
||||
### Viewers Router (`/viewers`) - Deferred
|
||||
|
||||
| Endpoint | Method | Purpose |
|
||||
|----------|--------|---------|
|
||||
| `/viewers/pdf/{document_id}` | GET | Serve PDF.js viewer |
|
||||
| `/viewers/image/{document_id}` | GET | Serve image lightbox |
|
||||
| `/viewers/video/{document_id}` | GET | Serve Video.js player |
|
||||
|
||||
---
|
||||
|
||||
## Data Flow: Document Processing Pipeline
|
||||
|
||||
```
|
||||
1. INTAKE (Paperless-ngx handles this)
|
||||
└─ Upload via Paperless UI, email, or folder watch
|
||||
└─ Paperless assigns document ID and stores file
|
||||
|
||||
2. OCR EXTRACTION (Paperless-ngx handles this)
|
||||
├─ PDFs → Tesseract → Plain text
|
||||
├─ Images → Tesseract → Plain text
|
||||
└─ Videos → Metadata only (no OCR)
|
||||
|
||||
3. SYNC TO LIBRARY DESK (scheduled or manual)
|
||||
└─ Poll Paperless API for new/updated documents
|
||||
└─ Fetch text content + metadata
|
||||
|
||||
4. TEXT CHUNKING
|
||||
└─ VectorService._chunk_text() (existing)
|
||||
|
||||
5. EMBEDDING
|
||||
└─ OllamaClient.embed() (existing)
|
||||
|
||||
6. VECTOR STORAGE (Qdrant)
|
||||
└─ Payload: {doc_type: "document", paperless_id, ...}
|
||||
|
||||
7. GRAPH STORAGE (Neo4j)
|
||||
└─ Document node + MENTIONS relationships
|
||||
|
||||
8. WIKI CATALOG (optional)
|
||||
└─ Auto-generate catalog page via ConsolidationService
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Git Docs Integration
|
||||
|
||||
Extends existing `scheduler/src/executors/doc_sync_executor.py`:
|
||||
|
||||
1. **Scheduler** syncs docs from GitHub → Gitea (existing)
|
||||
2. **Post-sync hook** calls `POST /documents/sync-from-git`
|
||||
3. **Library Desk** indexes docs into vectors/graph
|
||||
4. **Auto-generate** wiki catalog page for collection
|
||||
|
||||
---
|
||||
|
||||
## Wiki.js Viewer Integration
|
||||
|
||||
Since Wiki.js v2 requires disabled HTML sanitization for iframes:
|
||||
|
||||
```markdown
|
||||
<!-- In wiki catalog page -->
|
||||
## Document Preview
|
||||
|
||||
<iframe
|
||||
src="http://library-desk:8089/viewers/pdf/abc123"
|
||||
width="100%" height="600px">
|
||||
</iframe>
|
||||
```
|
||||
|
||||
**Wiki.js Settings Required:**
|
||||
- `Administration > Security > Allowed HTML Elements: iframe`
|
||||
- `Content Security Policy: frame-src http://library-desk:8089`
|
||||
|
||||
---
|
||||
|
||||
## Implementation Phases
|
||||
|
||||
### Phase 3.1: Infrastructure Setup
|
||||
- [ ] Deploy Paperless-ngx container (Docker Compose)
|
||||
- [x] ClamAV installed on host OS (port 3310)
|
||||
- [ ] Configure Paperless: storage path, OCR settings, API token
|
||||
- [ ] Create custom fields in Paperless: `source_url`, `library_indexed`, `library_doc_id`, `collection`
|
||||
- [ ] Create `src/clients/paperless_client.py`
|
||||
- [ ] Create `src/clients/clamav_client.py` (pyclamd wrapper)
|
||||
- [ ] Create `src/models/document.py`
|
||||
- [ ] Add config settings to `src/config.py` (PAPERLESS_*, CLAMAV_*)
|
||||
|
||||
### Phase 3.2: Webhook Integration (Push)
|
||||
- [ ] Create `src/routers/documents.py`
|
||||
- [ ] Implement `/documents/webhook` endpoint (receives Paperless events)
|
||||
- [ ] Configure Paperless Workflow: Document Added → Webhook → Library Desk
|
||||
- [ ] Create `src/services/document_sync_service.py`
|
||||
- [ ] Implement document indexing pipeline (fetch text → chunk → embed → graph)
|
||||
|
||||
### Phase 3.3: Polling Sync (Pull Fallback)
|
||||
- [ ] Implement `/documents/sync` endpoint
|
||||
- [ ] Poll Paperless for docs where `library_indexed=false`
|
||||
- [ ] Track sync state (last_sync timestamp in Redis)
|
||||
- [ ] Update `library_indexed` after successful indexing
|
||||
|
||||
### Phase 3.4: HybridRAG Upload Integration
|
||||
- [ ] Implement `/documents/upload` endpoint
|
||||
- [ ] Download file from URL
|
||||
- [ ] **Virus scan before upload** (reject if infected, log threat)
|
||||
- [ ] Upload clean files to Paperless with metadata
|
||||
- [ ] Set `source_url` custom field
|
||||
- [ ] Extend HybridRAG service to detect and upload relevant PDFs
|
||||
- [ ] Add `save_to_documents` option to HybridRAG config
|
||||
|
||||
### Phase 3.5: Indexing Pipeline
|
||||
- [ ] Extend VectorService for `doc_type: "document"`
|
||||
- [ ] Extend GraphService for Document nodes (link to Paperless ID)
|
||||
- [ ] Implement `/documents/search` endpoint
|
||||
- [ ] Add dependency injection
|
||||
|
||||
### Phase 3.6: Git Docs Integration
|
||||
- [ ] Create `src/clients/gitea_client.py`
|
||||
- [ ] Implement `/documents/sync-from-git` → bulk upload to Paperless
|
||||
- [ ] Create collection auto-cataloging (wiki pages)
|
||||
- [ ] Add scheduler task for periodic git sync
|
||||
|
||||
### Phase 3.7: Viewers (Deferred)
|
||||
*After core implementation is working*
|
||||
- [ ] Create `static/pdf-viewer.html` (PDF.js)
|
||||
- [ ] Create `static/image-viewer.html`
|
||||
- [ ] Create `static/video-player.html` (Video.js)
|
||||
- [ ] Create `src/routers/viewers.py`
|
||||
|
||||
### Phase 3.8: Maintenance & Testing
|
||||
- [ ] Extend cleanup for document orphans
|
||||
- [ ] Add document orphan detection (Paperless deleted but still in Qdrant/Neo4j)
|
||||
- [ ] Create `tests/test_document_sync.py`
|
||||
- [ ] Create `tests/test_paperless_client.py`
|
||||
|
||||
---
|
||||
|
||||
## Files to Create
|
||||
|
||||
| Path | Purpose |
|
||||
|------|---------|
|
||||
| `src/clients/paperless_client.py` | Paperless-ngx REST API client |
|
||||
| `src/clients/clamav_client.py` | ClamAV scanner (pyclamd wrapper) |
|
||||
| `src/clients/gitea_client.py` | Gitea repo access |
|
||||
| `src/models/document.py` | Document/Collection/ScanResult models |
|
||||
| `src/services/document_sync_service.py` | Sync orchestrator |
|
||||
| `src/routers/documents.py` | Document endpoints (webhook, sync, upload, search) |
|
||||
| `tests/test_document_sync.py` | Sync service tests |
|
||||
| `tests/test_paperless_client.py` | API client tests |
|
||||
| `tests/test_clamav_client.py` | Virus scanner tests |
|
||||
| `docker/docker-compose.documents.yml` | Paperless + ClamAV deployment |
|
||||
|
||||
**Deferred files (Phase 3.7):**
|
||||
|
||||
| Path | Purpose |
|
||||
|------|---------|
|
||||
| `src/routers/viewers.py` | Viewer endpoints |
|
||||
| `static/pdf-viewer.html` | PDF.js viewer |
|
||||
| `static/image-viewer.html` | Image lightbox |
|
||||
| `static/video-player.html` | Video.js player |
|
||||
|
||||
## Files to Modify
|
||||
|
||||
| Path | Changes |
|
||||
|------|---------|
|
||||
| `src/config.py` | `PAPERLESS_*`, `CLAMAV_*` settings |
|
||||
| `src/core/dependencies.py` | DocumentSyncService, PaperlessClient, ClamAVClient DI |
|
||||
| `src/main.py` | Register documents router |
|
||||
| `src/services/vector_service.py` | `doc_type: "document"` handling |
|
||||
| `src/services/graph_service.py` | Document node with Paperless ID |
|
||||
| `src/services/hybrid_rag_service.py` | Add `save_to_documents` option + virus scan |
|
||||
| `src/models/hybrid_rag.py` | Add `save_to_documents` config |
|
||||
| `src/routers/maintenance.py` | Document orphan cleanup, ClamAV health check |
|
||||
| `requirements.txt` | Add `pyclamd` |
|
||||
|
||||
## Paperless Custom Fields Setup
|
||||
|
||||
Create these in Paperless UI (Administration → Custom Fields):
|
||||
|
||||
| Field | Type | Purpose |
|
||||
|-------|------|---------|
|
||||
| `source_url` | URL | Original download URL |
|
||||
| `library_indexed` | Boolean | Sync status |
|
||||
| `library_doc_id` | Text | Library Desk reference |
|
||||
| `collection` | Text | Logical grouping |
|
||||
@@ -0,0 +1,58 @@
|
||||
# HybridRAG Architecture
|
||||
|
||||
## Overview
|
||||
|
||||
HybridRAG combines three search sources to provide comprehensive results:
|
||||
- **Vector search** (Qdrant) - Semantic similarity via embeddings
|
||||
- **Graph search** (Neo4j) - Entity relationships in knowledge graph
|
||||
- **Web search** (SearXNG) - External web results via Trafilatura extraction
|
||||
|
||||
## Two-Stage RRF Fusion (v1.3.0+)
|
||||
|
||||
To ensure fair ranking between wiki and web results, we use a two-stage Reciprocal Rank Fusion:
|
||||
|
||||
```
|
||||
Stage 1: Wiki Merge
|
||||
vector results ─┬─→ Mini-RRF ─→ Unified wiki ranking
|
||||
graph results ─┘
|
||||
|
||||
Stage 2: Final RRF
|
||||
wiki (merged) ─┬─→ Final RRF ─→ Combined results
|
||||
web results ─┘
|
||||
```
|
||||
|
||||
**Why two stages?**
|
||||
|
||||
Previously, wiki pages found by BOTH vector and graph received double RRF contribution, giving them an unfair 2x advantage over web results. The two-stage approach:
|
||||
1. Merges vector+graph into a single "wiki" source
|
||||
2. Wiki's internal ranking still benefits from multi-source confirmation
|
||||
3. Wiki and web compete as equals in final ranking
|
||||
|
||||
## Configuration
|
||||
|
||||
| Setting | Default | Description |
|
||||
|---------|---------|-------------|
|
||||
| `VECTOR_SIMILARITY_THRESHOLD` | 0.7 | Minimum similarity score for vector results |
|
||||
| `HYBRID_RAG_VECTOR_LIMIT` | 10 | Max vector results |
|
||||
| `HYBRID_RAG_GRAPH_LIMIT` | 10 | Max graph results |
|
||||
| `HYBRID_RAG_WEB_LIMIT` | 5 | Max web results |
|
||||
|
||||
## Known Limitations & Future Improvements
|
||||
|
||||
### Vector Search Noise
|
||||
|
||||
**Status:** Open for improvement if needed after observation period.
|
||||
|
||||
Vector search may return generic category/index pages (e.g., "Reference", "Projects", "Places") with high similarity scores (~0.86). These pages often have similar boilerplate content leading to uniform scores.
|
||||
|
||||
**Potential solutions if this becomes problematic:**
|
||||
1. **Raise threshold** - Increase `VECTOR_SIMILARITY_THRESHOLD` to 0.85+
|
||||
2. **Page-type filtering** - Exclude pages tagged as category/index/stub
|
||||
3. **Content length signal** - Penalize pages with minimal content
|
||||
4. **Duplicate score detection** - Flag results with suspiciously identical scores
|
||||
|
||||
The LLM re-ranking phase typically demotes these low-quality results, so this may not require immediate action.
|
||||
|
||||
### Graph Search
|
||||
|
||||
Graph search uses only core keywords (no LLM-generated synonyms) to avoid false matches like "author" → "author2000". This is intentional - vector search handles semantic similarity via embeddings.
|
||||
@@ -0,0 +1,332 @@
|
||||
# Memory Management System - Implementation Plan
|
||||
|
||||
## Overview
|
||||
|
||||
A three-tier memory architecture for Library Desk with intelligent orchestration:
|
||||
|
||||
| Tier | Storage | Purpose | TTL |
|
||||
|------|---------|---------|-----|
|
||||
| **Volatile** | Qdrant (vectors) | Weather, news, financial, ephemeral context | 5min - 2hr |
|
||||
| **Documents** | Paperless-ngx + ClamAV (host) | Git mirrors, PDFs, video, images | Permanent |
|
||||
| **Knowledge** | Wiki + Neo4j | Personal dossiers, research, summaries | Permanent |
|
||||
|
||||
**Implementation Priority**: Cleanup → Volatile → Documents → Test Data Cleanup
|
||||
|
||||
### Phase Status
|
||||
|
||||
| Phase | Status | Version |
|
||||
|-------|--------|---------|
|
||||
| Phase 1: Cleanup System | ✅ Complete | v1.4.0 |
|
||||
| Phase 2: Volatile Memory | ✅ Complete | v1.4.3 |
|
||||
| Phase 3: Document Storage | ✅ Planned | See [DOCUMENT_STORAGE_PLAN.md](DOCUMENT_STORAGE_PLAN.md) |
|
||||
| Phase 4: Test Data Cleanup | ✅ Complete | v1.4.4 |
|
||||
|
||||
---
|
||||
|
||||
## Phase 1: Cleanup System Completion ✅
|
||||
|
||||
### Current State
|
||||
- **COMPLETE** - All Phase 1 tasks implemented
|
||||
- Redis timestamp tracking for last cleanup
|
||||
- Bidirectional orphan detection between vectors and graph
|
||||
- Scheduler integration endpoints ready
|
||||
|
||||
### Tasks
|
||||
|
||||
#### 1.1 Add Scheduler Integration Points ✅
|
||||
**Files**: `src/routers/maintenance.py`
|
||||
|
||||
- [x] Add `last_cleanup` timestamp tracking in Redis
|
||||
- [x] Return cleanup stats in format scheduler can log
|
||||
- [x] Added `RedisDep` to cleanup endpoints
|
||||
|
||||
#### 1.2 Bidirectional Orphan Detection ✅
|
||||
**Files**: `src/services/graph_service.py`, `src/services/vector_service.py`
|
||||
|
||||
- [x] `find_documents_without_vectors()` - graph nodes with no vectors
|
||||
- [x] `find_chunks_without_graph_nodes()` - vectors with no graph node
|
||||
- [x] Updated maintenance endpoints to use bidirectional checks
|
||||
- [x] Added `chunks_without_graph` and `docs_without_vectors` to response models
|
||||
|
||||
#### 1.3 Scheduler Configuration ✅
|
||||
**Scheduler-side task definition:**
|
||||
```json
|
||||
{
|
||||
"task_name": "library_reconcile_index",
|
||||
"schedule": "0 4 * * *",
|
||||
"endpoint": "POST /maintenance/reconcile-index?user=jpmschweitzer",
|
||||
"description": "Daily index reconciliation - cleanup + reindex missing"
|
||||
}
|
||||
```
|
||||
|
||||
- [x] Documented in `LIBRARIAN_INTEGRATION.md`
|
||||
- [x] Added `reconcile-index` endpoint (cleanup + reindex missing)
|
||||
- [x] Lightweight health check mode for uptime monitoring
|
||||
- [x] Detailed health check mode for dashboards
|
||||
|
||||
---
|
||||
|
||||
## Phase 2: Volatile Memory System ✅
|
||||
|
||||
### Architecture (Final Implementation)
|
||||
|
||||
```
|
||||
┌─────────────────┐ ┌──────────────┐ ┌─────────────────┐
|
||||
│ Library-Desk │◄───│ Scheduler │───►│ External APIs │
|
||||
│ │ │ │ │ (weather, news) │
|
||||
│ VolatileCache │ │ Refresh │ └─────────────────┘
|
||||
│ Service │ │ Jobs │
|
||||
└────────┬────────┘ └──────────────┘
|
||||
│
|
||||
▼
|
||||
┌─────────────────┐
|
||||
│ Qdrant │
|
||||
│ (volatile_{user})│
|
||||
└─────────────────┘
|
||||
```
|
||||
|
||||
**Key design decisions:**
|
||||
- Vector storage in Qdrant (not Redis) for semantic search
|
||||
- Collection per user: `volatile_{user}`
|
||||
- TTL via `ttl_expiry` timestamp in payload
|
||||
- Natural language conversion for embedding structured data
|
||||
- Integrated into HybridRAG with priority boost
|
||||
|
||||
### Endpoints (Implemented)
|
||||
|
||||
| Endpoint | Method | Purpose |
|
||||
|----------|--------|---------|
|
||||
| `/volatile/search?q=...` | GET | Semantic search across volatile data |
|
||||
| `/volatile/store?namespace=...&key=...` | POST | Store/update record |
|
||||
| `/volatile/{namespace}/{key}` | GET | Retrieve specific record |
|
||||
| `/volatile/{namespace}/{key}` | DELETE | Remove record |
|
||||
| `/volatile/stats` | GET | Cache statistics |
|
||||
| `/volatile/scheduled` | GET | Records needing refresh |
|
||||
| `/volatile/namespaces` | GET | List available namespaces |
|
||||
| `/maintenance/cleanup/volatile` | POST | Purge expired records |
|
||||
|
||||
### Namespaces
|
||||
|
||||
| Namespace | Default TTL | Use Case |
|
||||
|-----------|-------------|----------|
|
||||
| weather | 30 min | Current conditions, forecasts |
|
||||
| news | 1 hour | Headlines, breaking news |
|
||||
| financial | 5 min | Stock prices, exchange rates |
|
||||
| transit | 5 min | Train/bus schedules, delays |
|
||||
| traffic | 10 min | Commute times, road conditions |
|
||||
| air_quality | 1 hour | Pollution, pollen counts |
|
||||
| sports | 1 min | Live scores, matches |
|
||||
| social | 10 min | Social notifications |
|
||||
| system | 1 min | Service health status |
|
||||
| context | 1 hour | Session state |
|
||||
| custom | 1 hour | User-defined data |
|
||||
|
||||
---
|
||||
|
||||
## Phase 3: Document Storage (Research + Implementation)
|
||||
|
||||
### Research Scope
|
||||
|
||||
Evaluate FOSS self-hosted options for:
|
||||
- Git repository mirroring
|
||||
- PDF/document storage with metadata
|
||||
- Image/video blob storage
|
||||
- Full-text search capability
|
||||
|
||||
**Constraints**:
|
||||
- Must be self-hosted, Docker-deployable
|
||||
- Performance is priority (can wrap complexity in API)
|
||||
- No cloud dependencies
|
||||
|
||||
**Candidates to evaluate**:
|
||||
1. MinIO (S3-compatible object storage) + metadata in Neo4j
|
||||
2. Paperless-ngx (document management with OCR)
|
||||
3. SeaweedFS (distributed file system)
|
||||
4. Custom: filesystem + Neo4j metadata
|
||||
|
||||
### Category Descriptors
|
||||
|
||||
**Wiki page structure for document collections**:
|
||||
```markdown
|
||||
# FastAPI Documentation
|
||||
|
||||
## Overview
|
||||
[LLM-generated summary from web search about FastAPI]
|
||||
|
||||
## Collection Statistics
|
||||
- **Documents**: 342 files
|
||||
- **Last Sync**: 2025-12-24 03:30 UTC
|
||||
- **Source**: github.com/tiangolo/fastapi
|
||||
- **Coverage**: API reference, tutorials, deployment guides
|
||||
|
||||
## What's Included
|
||||
[LLM summary of collection contents based on document analysis]
|
||||
|
||||
## Related Topics
|
||||
- [[Python Web Frameworks]]
|
||||
- [[REST API Design]]
|
||||
```
|
||||
|
||||
### Tasks
|
||||
|
||||
#### 3.1 Storage Research
|
||||
**Deliverable**: Evaluation document comparing options
|
||||
|
||||
#### 3.2 Storage Service Implementation
|
||||
**New file**: `src/services/document_store_service.py`
|
||||
(Details pending research results)
|
||||
|
||||
#### 3.3 Category Descriptor Generation
|
||||
**File**: `src/services/consolidation_service.py`
|
||||
|
||||
Add LLM-powered category descriptor generation:
|
||||
1. Web search for topic overview
|
||||
2. Analyze collection contents
|
||||
3. Generate/update wiki page with template
|
||||
|
||||
---
|
||||
|
||||
## Phase 4: LLM Tester Data Cleanup ✅
|
||||
|
||||
### Problem
|
||||
|
||||
LLM testing creates accumulated cruft across the system:
|
||||
- Wiki.js pages under `llm-tester/` and `llm_tester/` paths
|
||||
- Graph nodes (Document, Entity) linked to test pages
|
||||
- Vector chunks in Qdrant for test content
|
||||
|
||||
This data accumulates over time and clutters Wiki.js visually (no separate tenant scope for tests).
|
||||
|
||||
### Solution
|
||||
|
||||
Add a maintenance endpoint to purge all LLM tester artifacts across wiki, graph, and vectors.
|
||||
|
||||
### Tasks
|
||||
|
||||
#### 4.1 Identify Test Data Patterns ✅
|
||||
**Patterns matched** (security-restricted to test user namespace):
|
||||
- `users/llm-tester/*`
|
||||
- `users/llm_tester/*`
|
||||
|
||||
#### 4.2 Add Cleanup Endpoint ✅
|
||||
**File**: `src/routers/maintenance.py`
|
||||
|
||||
```python
|
||||
@router.post("/cleanup/test-data")
|
||||
async def cleanup_test_data(
|
||||
dry_run: bool = Query(default=True),
|
||||
wiki: WikiJSDep = None,
|
||||
vector_service: VectorServiceDep = None,
|
||||
graph_service: GraphServiceDep = None,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Purge LLM tester data from wiki, graph, and vectors.
|
||||
|
||||
**Security**: Only deletes pages in the test user namespace:
|
||||
- users/llm-tester/*
|
||||
- users/llm_tester/*
|
||||
|
||||
Use dry_run=true to preview what would be deleted.
|
||||
"""
|
||||
```
|
||||
|
||||
#### 4.3 Implementation Steps ✅
|
||||
|
||||
1. **Wiki cleanup**: Delete pages via GraphQL mutation
|
||||
2. **Graph cleanup**: Delete Document nodes using `delete_page()` method
|
||||
3. **Vector cleanup**: Delete chunks using `delete_page_chunks()` method
|
||||
|
||||
#### 4.4 Scheduler Integration ✅
|
||||
**Recommended schedule**: Weekly (Sunday 3:00 AM)
|
||||
|
||||
```json
|
||||
{
|
||||
"task_name": "test_data_cleanup",
|
||||
"schedule": "0 3 * * 0",
|
||||
"endpoint": "POST /maintenance/cleanup/test-data?dry_run=false",
|
||||
"description": "Weekly cleanup of LLM test data"
|
||||
}
|
||||
```
|
||||
|
||||
### Files to Modify
|
||||
|
||||
- `src/routers/maintenance.py` - Add cleanup endpoint
|
||||
- `src/services/wiki_service.py` - Add bulk delete by path pattern (if needed)
|
||||
- `src/services/graph_service.py` - May need pattern-based node deletion
|
||||
- `src/services/vector_service.py` - Add pattern-based chunk deletion
|
||||
|
||||
---
|
||||
|
||||
## Files Modified/Created
|
||||
|
||||
### Phase 1 (Cleanup) ✅
|
||||
- `src/routers/maintenance.py` - Timestamp tracking, cleanup endpoints
|
||||
- `src/services/graph_service.py` - Bidirectional validation
|
||||
- `src/services/vector_service.py` - Cross-reference checks
|
||||
- `LIBRARIAN_INTEGRATION.md` - Scheduler config docs
|
||||
|
||||
### Phase 2 (Volatile) ✅
|
||||
- `src/services/volatile_service.py` - Qdrant-based volatile cache
|
||||
- `src/routers/volatile.py` - Simplified endpoints
|
||||
- `src/models/volatile.py` - Namespaces and models
|
||||
- `src/models/hybrid_rag.py` - Volatile config options
|
||||
- `src/services/hybrid_rag_service.py` - Volatile integration
|
||||
- `src/clients/qdrant_client.py` - Expiry filter methods
|
||||
- `tests/test_volatile.py` - 37 tests
|
||||
|
||||
### Phase 3 (Documents)
|
||||
- `docs/DOCUMENT_STORAGE_RESEARCH.md` - **NEW**
|
||||
- `src/services/document_store_service.py` - **NEW** (post-research)
|
||||
- `src/routers/documents.py` - **NEW** (post-research)
|
||||
|
||||
### Phase 4 (Test Data Cleanup)
|
||||
- `src/routers/maintenance.py` - Add cleanup endpoint
|
||||
- `src/services/wiki_service.py` - Bulk delete by path pattern
|
||||
- `src/services/graph_service.py` - Pattern-based node deletion
|
||||
- `src/services/vector_service.py` - Pattern-based chunk deletion
|
||||
|
||||
---
|
||||
|
||||
## Resolved Design Decisions
|
||||
|
||||
1. **Volatile Storage**: Qdrant vectors (not Redis) for semantic search capability
|
||||
2. **Collection Naming**: `volatile_{user}` for per-user isolation
|
||||
3. **TTL Mechanism**: `ttl_expiry` timestamp in payload, background cleanup job
|
||||
4. **HybridRAG Integration**: Volatile as third source with RRF priority boost
|
||||
5. **Biographer Qdrant**: Same Qdrant instance, different collection
|
||||
6. **Scheduler API**: Has REST API for task registration
|
||||
|
||||
---
|
||||
|
||||
## Future Consideration: Dedicated API Integrations
|
||||
|
||||
For volatile data where quality/consistency matters (weather, financial), consider:
|
||||
- OpenWeatherMap API for weather (daily refresh cycle)
|
||||
- Financial data API (Alpha Vantage, Yahoo Finance)
|
||||
- News APIs (NewsAPI, GDELT)
|
||||
- **NOS.nl** - Explicit source for Dutch news
|
||||
|
||||
This would live in a new `src/clients/` module with:
|
||||
- `weather_client.py` - Daily refresh cycle
|
||||
- `financial_client.py`
|
||||
- `news_client.py` - Include NOS.nl scraper/API for Dutch coverage
|
||||
|
||||
These provide structured, reliable data vs. SearXNG web scraping. Implementation deferred to later phase.
|
||||
|
||||
---
|
||||
|
||||
## Refresh Schedules
|
||||
|
||||
**Note:** TTL should be longer than refresh interval to prevent data gaps.
|
||||
|
||||
| Volatile Type | TTL | Refresh Cycle | Refresh Interval | Sources |
|
||||
|---------------|-----|---------------|------------------|---------|
|
||||
| Weather | 86400s (24hr) | Daily | Every 24hr | OpenWeatherMap |
|
||||
| Dutch News | 28800s (8hr) | 4x daily | Every 6hr | NOS.nl |
|
||||
| Global News | 28800s (8hr) | 4x daily | Every 6hr | NewsAPI, GDELT |
|
||||
| Financial | 600s (10min) | On-demand | N/A | Alpha Vantage |
|
||||
|
||||
**TTL Logic:**
|
||||
- TTL = Refresh Interval × 1.5 (buffer for failed refreshes)
|
||||
- On-demand data gets shorter TTL since it's fetched when needed
|
||||
@@ -0,0 +1,31 @@
|
||||
[project]
|
||||
name = "library-desk"
|
||||
version = "1.4.8"
|
||||
description = "Coordination service for The Library system - HybridRAG queries, document ingestion, entity extraction, and knowledge consolidation"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.12"
|
||||
license = {text = "MIT"}
|
||||
authors = [
|
||||
{name = "JP Schweitzer"}
|
||||
]
|
||||
keywords = ["rag", "knowledge-graph", "wiki", "semantic-search", "neo4j", "qdrant"]
|
||||
classifiers = [
|
||||
"Development Status :: 4 - Beta",
|
||||
"Framework :: FastAPI",
|
||||
"Intended Audience :: Developers",
|
||||
"License :: OSI Approved :: MIT License",
|
||||
"Programming Language :: Python :: 3",
|
||||
"Programming Language :: Python :: 3.12",
|
||||
]
|
||||
|
||||
[project.urls]
|
||||
Homepage = "https://github.com/jpmschweitzer/library-desk"
|
||||
Documentation = "https://github.com/jpmschweitzer/library-desk#readme"
|
||||
|
||||
[build-system]
|
||||
requires = ["setuptools>=61.0"]
|
||||
build-backend = "setuptools.build_meta"
|
||||
|
||||
[tool.setuptools.packages.find]
|
||||
where = ["."]
|
||||
include = ["src*"]
|
||||
@@ -25,6 +25,9 @@ python-multipart~=0.0.20
|
||||
# Utilities
|
||||
python-dateutil~=2.9.0
|
||||
|
||||
# Content Extraction
|
||||
trafilatura~=1.12.0
|
||||
|
||||
# Testing
|
||||
pytest~=8.3.0
|
||||
pytest-asyncio~=0.24.0
|
||||
|
||||
@@ -0,0 +1,310 @@
|
||||
"""
|
||||
Content extraction client for Library Desk.
|
||||
|
||||
A reusable Trafilatura wrapper that can be used throughout library-desk:
|
||||
- RAG search service (extract content from search results)
|
||||
- Ingestion service (extract content from URLs)
|
||||
- Standalone endpoint (ad-hoc content extraction)
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from typing import List, Optional
|
||||
|
||||
import trafilatura
|
||||
|
||||
from src.models.content import ContentExtractionResult
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class ContentExtractor:
|
||||
"""
|
||||
Generic content extraction client using Trafilatura.
|
||||
|
||||
Provides async wrappers around Trafilatura's synchronous extraction,
|
||||
with support for parallel batch processing and configurable timeouts.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
timeout: int = 5,
|
||||
max_length: int = 2000,
|
||||
max_workers: int = 10
|
||||
):
|
||||
"""
|
||||
Initialize ContentExtractor.
|
||||
|
||||
Args:
|
||||
timeout: Per-URL timeout in seconds
|
||||
max_length: Maximum content length to return (truncated if longer)
|
||||
max_workers: Max concurrent extractions for batch operations
|
||||
"""
|
||||
self.timeout = timeout
|
||||
self.max_length = max_length
|
||||
self._executor = ThreadPoolExecutor(max_workers=max_workers)
|
||||
logger.info(
|
||||
f"Initialized ContentExtractor: timeout={timeout}s, "
|
||||
f"max_length={max_length}, max_workers={max_workers}"
|
||||
)
|
||||
|
||||
def _extract_sync(
|
||||
self,
|
||||
url: str,
|
||||
include_metadata: bool = True,
|
||||
max_length: Optional[int] = None
|
||||
) -> ContentExtractionResult:
|
||||
"""
|
||||
Synchronous extraction (runs in thread pool).
|
||||
|
||||
Args:
|
||||
url: URL to extract content from
|
||||
include_metadata: Whether to extract title, author, date
|
||||
max_length: Override default max length
|
||||
|
||||
Returns:
|
||||
ContentExtractionResult with extracted content or error
|
||||
"""
|
||||
effective_max_length = max_length or self.max_length
|
||||
|
||||
try:
|
||||
# Fetch the URL
|
||||
downloaded = trafilatura.fetch_url(url)
|
||||
if not downloaded:
|
||||
return ContentExtractionResult(
|
||||
url=url,
|
||||
content="",
|
||||
success=False,
|
||||
error="Failed to fetch URL"
|
||||
)
|
||||
|
||||
# Extract content
|
||||
content = trafilatura.extract(
|
||||
downloaded,
|
||||
include_comments=False,
|
||||
include_tables=True,
|
||||
output_format='txt'
|
||||
)
|
||||
|
||||
if not content:
|
||||
return ContentExtractionResult(
|
||||
url=url,
|
||||
content="",
|
||||
success=False,
|
||||
error="No content extracted"
|
||||
)
|
||||
|
||||
# Truncate if needed
|
||||
if len(content) > effective_max_length:
|
||||
content = content[:effective_max_length] + "..."
|
||||
|
||||
# Extract metadata if requested
|
||||
title = None
|
||||
author = None
|
||||
date = None
|
||||
language = None
|
||||
|
||||
if include_metadata:
|
||||
metadata = trafilatura.extract(
|
||||
downloaded,
|
||||
output_format='xml',
|
||||
include_comments=False
|
||||
)
|
||||
# Parse metadata from XML if available
|
||||
# trafilatura.extract with output_format='xml' returns XML with metadata
|
||||
# For simplicity, we'll use bare_extraction which returns a dict
|
||||
try:
|
||||
meta_dict = trafilatura.bare_extraction(
|
||||
downloaded,
|
||||
include_comments=False
|
||||
)
|
||||
if meta_dict:
|
||||
title = meta_dict.get('title')
|
||||
author = meta_dict.get('author')
|
||||
date = meta_dict.get('date')
|
||||
language = meta_dict.get('language')
|
||||
except Exception as e:
|
||||
logger.debug(f"Metadata extraction failed for {url}: {e}")
|
||||
|
||||
return ContentExtractionResult(
|
||||
url=url,
|
||||
title=title,
|
||||
content=content,
|
||||
author=author,
|
||||
date=date,
|
||||
language=language,
|
||||
success=True,
|
||||
error=None
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Content extraction failed for {url}: {e}")
|
||||
return ContentExtractionResult(
|
||||
url=url,
|
||||
content="",
|
||||
success=False,
|
||||
error=str(e)
|
||||
)
|
||||
|
||||
async def extract(
|
||||
self,
|
||||
url: str,
|
||||
include_metadata: bool = True,
|
||||
max_length: Optional[int] = None
|
||||
) -> ContentExtractionResult:
|
||||
"""
|
||||
Extract content from a single URL asynchronously.
|
||||
|
||||
Args:
|
||||
url: URL to extract content from
|
||||
include_metadata: Whether to extract title, author, date
|
||||
max_length: Override default max length
|
||||
|
||||
Returns:
|
||||
ContentExtractionResult with extracted content or error
|
||||
"""
|
||||
loop = asyncio.get_event_loop()
|
||||
|
||||
try:
|
||||
result = await asyncio.wait_for(
|
||||
loop.run_in_executor(
|
||||
self._executor,
|
||||
self._extract_sync,
|
||||
url,
|
||||
include_metadata,
|
||||
max_length
|
||||
),
|
||||
timeout=self.timeout
|
||||
)
|
||||
return result
|
||||
except asyncio.TimeoutError:
|
||||
logger.warning(f"Content extraction timed out for {url}")
|
||||
return ContentExtractionResult(
|
||||
url=url,
|
||||
content="",
|
||||
success=False,
|
||||
error=f"Extraction timed out after {self.timeout}s"
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"Unexpected error extracting {url}: {e}")
|
||||
return ContentExtractionResult(
|
||||
url=url,
|
||||
content="",
|
||||
success=False,
|
||||
error=str(e)
|
||||
)
|
||||
|
||||
async def extract_batch(
|
||||
self,
|
||||
urls: List[str],
|
||||
include_metadata: bool = True,
|
||||
max_length: Optional[int] = None
|
||||
) -> List[ContentExtractionResult]:
|
||||
"""
|
||||
Extract content from multiple URLs in parallel.
|
||||
|
||||
Args:
|
||||
urls: List of URLs to extract content from
|
||||
include_metadata: Whether to extract title, author, date
|
||||
max_length: Override default max length
|
||||
|
||||
Returns:
|
||||
List of ContentExtractionResult in same order as input URLs
|
||||
"""
|
||||
tasks = [
|
||||
self.extract(url, include_metadata, max_length)
|
||||
for url in urls
|
||||
]
|
||||
results = await asyncio.gather(*tasks)
|
||||
return list(results)
|
||||
|
||||
async def extract_from_html(
|
||||
self,
|
||||
html: str,
|
||||
url: str = "",
|
||||
include_metadata: bool = True,
|
||||
max_length: Optional[int] = None
|
||||
) -> ContentExtractionResult:
|
||||
"""
|
||||
Extract content from raw HTML string.
|
||||
|
||||
Args:
|
||||
html: Raw HTML content
|
||||
url: Optional URL for reference (not fetched)
|
||||
include_metadata: Whether to extract title, author, date
|
||||
max_length: Override default max length
|
||||
|
||||
Returns:
|
||||
ContentExtractionResult with extracted content or error
|
||||
"""
|
||||
effective_max_length = max_length or self.max_length
|
||||
|
||||
def _extract():
|
||||
try:
|
||||
content = trafilatura.extract(
|
||||
html,
|
||||
include_comments=False,
|
||||
include_tables=True,
|
||||
output_format='txt'
|
||||
)
|
||||
|
||||
if not content:
|
||||
return ContentExtractionResult(
|
||||
url=url,
|
||||
content="",
|
||||
success=False,
|
||||
error="No content extracted from HTML"
|
||||
)
|
||||
|
||||
# Truncate if needed
|
||||
if len(content) > effective_max_length:
|
||||
content = content[:effective_max_length] + "..."
|
||||
|
||||
# Extract metadata
|
||||
title = None
|
||||
author = None
|
||||
date = None
|
||||
language = None
|
||||
|
||||
if include_metadata:
|
||||
try:
|
||||
meta_dict = trafilatura.bare_extraction(
|
||||
html,
|
||||
include_comments=False
|
||||
)
|
||||
if meta_dict:
|
||||
title = meta_dict.get('title')
|
||||
author = meta_dict.get('author')
|
||||
date = meta_dict.get('date')
|
||||
language = meta_dict.get('language')
|
||||
except Exception as e:
|
||||
logger.debug(f"Metadata extraction failed: {e}")
|
||||
|
||||
return ContentExtractionResult(
|
||||
url=url,
|
||||
title=title,
|
||||
content=content,
|
||||
author=author,
|
||||
date=date,
|
||||
language=language,
|
||||
success=True,
|
||||
error=None
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"HTML content extraction failed: {e}")
|
||||
return ContentExtractionResult(
|
||||
url=url,
|
||||
content="",
|
||||
success=False,
|
||||
error=str(e)
|
||||
)
|
||||
|
||||
loop = asyncio.get_event_loop()
|
||||
return await loop.run_in_executor(self._executor, _extract)
|
||||
|
||||
async def close(self):
|
||||
"""Shutdown the thread pool executor."""
|
||||
self._executor.shutdown(wait=False)
|
||||
logger.info("ContentExtractor closed")
|
||||
@@ -249,7 +249,8 @@ class OllamaClient:
|
||||
self,
|
||||
prompt: str,
|
||||
model: Optional[str] = None,
|
||||
stream: bool = False
|
||||
stream: bool = False,
|
||||
temperature: Optional[float] = None
|
||||
) -> Optional[str]:
|
||||
"""
|
||||
Generate text completion (for non-embedding use cases).
|
||||
@@ -258,12 +259,15 @@ class OllamaClient:
|
||||
prompt: Input prompt
|
||||
model: Model name (defaults to self.model)
|
||||
stream: Enable streaming response
|
||||
temperature: Sampling temperature (0.0 = deterministic, higher = more creative)
|
||||
None uses model default (~0.7 for mistral-nemo)
|
||||
|
||||
Returns:
|
||||
Generated text or None on failure
|
||||
|
||||
Note: This is primarily for debugging/testing. Use specialized
|
||||
LLM services for production text generation.
|
||||
Note: Use temperature=0.0 for deterministic outputs like JSON parsing,
|
||||
ranking, and factual extraction. Use higher values (0.3-0.7) for
|
||||
creative content generation.
|
||||
"""
|
||||
try:
|
||||
payload = {
|
||||
@@ -272,6 +276,10 @@ class OllamaClient:
|
||||
"stream": stream
|
||||
}
|
||||
|
||||
# Add temperature to options if specified
|
||||
if temperature is not None:
|
||||
payload["options"] = {"temperature": temperature}
|
||||
|
||||
response = await self.client.post(
|
||||
self.generate_url,
|
||||
json=payload
|
||||
|
||||
@@ -0,0 +1,488 @@
|
||||
"""
|
||||
Paperless-ngx API client for Library Desk.
|
||||
|
||||
Provides async document management via Paperless-ngx:
|
||||
- Document upload and retrieval
|
||||
- Search and filtering
|
||||
- Custom field management
|
||||
- Task status tracking
|
||||
"""
|
||||
|
||||
import httpx
|
||||
from typing import Optional, List, Dict, Any
|
||||
from dataclasses import dataclass
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class PaperlessDocument:
|
||||
"""Represents a document from Paperless-ngx."""
|
||||
id: int
|
||||
title: str
|
||||
content: str
|
||||
created: Optional[str] = None
|
||||
modified: Optional[str] = None
|
||||
added: Optional[str] = None
|
||||
correspondent: Optional[int] = None
|
||||
document_type: Optional[int] = None
|
||||
storage_path: Optional[int] = None
|
||||
tags: List[int] = None
|
||||
archive_serial_number: Optional[int] = None
|
||||
original_file_name: Optional[str] = None
|
||||
archived_file_name: Optional[str] = None
|
||||
custom_fields: List[Dict[str, Any]] = None
|
||||
|
||||
def __post_init__(self):
|
||||
if self.tags is None:
|
||||
self.tags = []
|
||||
if self.custom_fields is None:
|
||||
self.custom_fields = []
|
||||
|
||||
|
||||
@dataclass
|
||||
class SearchHit:
|
||||
"""Search result with relevance info."""
|
||||
document: PaperlessDocument
|
||||
score: float
|
||||
rank: int
|
||||
highlights: Optional[str] = None
|
||||
|
||||
|
||||
class PaperlessClient:
|
||||
"""
|
||||
Paperless-ngx REST API client.
|
||||
|
||||
Documentation: https://docs.paperless-ngx.com/api/
|
||||
"""
|
||||
|
||||
def __init__(self, base_url: str, token: str, timeout: int = 30):
|
||||
"""
|
||||
Initialize Paperless-ngx client.
|
||||
|
||||
Args:
|
||||
base_url: Paperless-ngx base URL (e.g., "http://paperless:8000")
|
||||
token: API token for authentication
|
||||
timeout: Request timeout in seconds
|
||||
"""
|
||||
self.base_url = base_url.rstrip("/")
|
||||
self.api_url = f"{self.base_url}/api"
|
||||
self.headers = {
|
||||
"Authorization": f"Token {token}",
|
||||
"Accept": "application/json",
|
||||
}
|
||||
self.client = httpx.AsyncClient(timeout=float(timeout), headers=self.headers)
|
||||
logger.info(f"Initialized Paperless client: {base_url}")
|
||||
|
||||
async def close(self):
|
||||
"""Close HTTP client."""
|
||||
await self.client.aclose()
|
||||
|
||||
# =========================================================================
|
||||
# Document Operations
|
||||
# =========================================================================
|
||||
|
||||
async def get_document(self, document_id: int) -> Optional[PaperlessDocument]:
|
||||
"""
|
||||
Get a document by ID.
|
||||
|
||||
Args:
|
||||
document_id: Paperless document ID
|
||||
|
||||
Returns:
|
||||
PaperlessDocument or None if not found
|
||||
"""
|
||||
try:
|
||||
response = await self.client.get(f"{self.api_url}/documents/{document_id}/")
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
return self._parse_document(data)
|
||||
except httpx.HTTPStatusError as e:
|
||||
if e.response.status_code == 404:
|
||||
return None
|
||||
logger.error(f"Failed to get document {document_id}: {e}")
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to get document {document_id}: {e}")
|
||||
raise
|
||||
|
||||
async def get_document_content(self, document_id: int) -> Optional[str]:
|
||||
"""
|
||||
Get extracted text content of a document.
|
||||
|
||||
Args:
|
||||
document_id: Paperless document ID
|
||||
|
||||
Returns:
|
||||
Text content or None if not found
|
||||
"""
|
||||
doc = await self.get_document(document_id)
|
||||
return doc.content if doc else None
|
||||
|
||||
async def list_documents(
|
||||
self,
|
||||
page: int = 1,
|
||||
page_size: int = 25,
|
||||
ordering: str = "-added",
|
||||
correspondent: Optional[int] = None,
|
||||
document_type: Optional[int] = None,
|
||||
tags: Optional[List[int]] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""
|
||||
List documents with pagination and filtering.
|
||||
|
||||
Args:
|
||||
page: Page number (starts at 1)
|
||||
page_size: Results per page
|
||||
ordering: Sort order (prefix with - for descending)
|
||||
correspondent: Filter by correspondent ID
|
||||
document_type: Filter by document type ID
|
||||
tags: Filter by tag IDs
|
||||
|
||||
Returns:
|
||||
Paginated response with count, next, previous, results
|
||||
"""
|
||||
params = {
|
||||
"page": page,
|
||||
"page_size": page_size,
|
||||
"ordering": ordering,
|
||||
}
|
||||
if correspondent:
|
||||
params["correspondent__id"] = correspondent
|
||||
if document_type:
|
||||
params["document_type__id"] = document_type
|
||||
if tags:
|
||||
params["tags__id__in"] = ",".join(str(t) for t in tags)
|
||||
|
||||
try:
|
||||
response = await self.client.get(f"{self.api_url}/documents/", params=params)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
return {
|
||||
"count": data.get("count", 0),
|
||||
"next": data.get("next"),
|
||||
"previous": data.get("previous"),
|
||||
"results": [self._parse_document(d) for d in data.get("results", [])],
|
||||
}
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to list documents: {e}")
|
||||
raise
|
||||
|
||||
async def search_documents(
|
||||
self,
|
||||
query: str,
|
||||
page: int = 1,
|
||||
page_size: int = 25,
|
||||
) -> List[SearchHit]:
|
||||
"""
|
||||
Full-text search documents.
|
||||
|
||||
Args:
|
||||
query: Search query string
|
||||
page: Page number
|
||||
page_size: Results per page
|
||||
|
||||
Returns:
|
||||
List of SearchHit with document and relevance info
|
||||
"""
|
||||
params = {
|
||||
"query": query,
|
||||
"page": page,
|
||||
"page_size": page_size,
|
||||
}
|
||||
|
||||
try:
|
||||
response = await self.client.get(f"{self.api_url}/documents/", params=params)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
|
||||
results = []
|
||||
for item in data.get("results", []):
|
||||
doc = self._parse_document(item)
|
||||
hit_info = item.get("__search_hit__", {})
|
||||
results.append(SearchHit(
|
||||
document=doc,
|
||||
score=hit_info.get("score", 0.0),
|
||||
rank=hit_info.get("rank", 0),
|
||||
highlights=hit_info.get("highlights"),
|
||||
))
|
||||
return results
|
||||
except Exception as e:
|
||||
logger.error(f"Search failed for '{query}': {e}")
|
||||
raise
|
||||
|
||||
async def upload_document(
|
||||
self,
|
||||
file_content: bytes,
|
||||
filename: str,
|
||||
title: Optional[str] = None,
|
||||
correspondent: Optional[int] = None,
|
||||
document_type: Optional[int] = None,
|
||||
tags: Optional[List[int]] = None,
|
||||
custom_fields: Optional[List[Dict[str, Any]]] = None,
|
||||
) -> str:
|
||||
"""
|
||||
Upload a document to Paperless-ngx.
|
||||
|
||||
Args:
|
||||
file_content: File bytes
|
||||
filename: Original filename
|
||||
title: Document title (optional, derived from filename if not set)
|
||||
correspondent: Correspondent ID
|
||||
document_type: Document type ID
|
||||
tags: List of tag IDs
|
||||
custom_fields: List of custom field values
|
||||
|
||||
Returns:
|
||||
Task UUID for tracking consumption status
|
||||
"""
|
||||
files = {"document": (filename, file_content)}
|
||||
data = {}
|
||||
|
||||
if title:
|
||||
data["title"] = title
|
||||
if correspondent:
|
||||
data["correspondent"] = correspondent
|
||||
if document_type:
|
||||
data["document_type"] = document_type
|
||||
if tags:
|
||||
# Tags need to be sent multiple times for multiple values
|
||||
data["tags"] = tags
|
||||
if custom_fields:
|
||||
data["custom_fields"] = custom_fields
|
||||
|
||||
try:
|
||||
response = await self.client.post(
|
||||
f"{self.api_url}/documents/post_document/",
|
||||
files=files,
|
||||
data=data,
|
||||
)
|
||||
response.raise_for_status()
|
||||
result = response.json()
|
||||
task_id = result.get("task_id", "")
|
||||
logger.info(f"Uploaded document '{filename}', task_id: {task_id}")
|
||||
return task_id
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to upload document '{filename}': {e}")
|
||||
raise
|
||||
|
||||
async def get_task_status(self, task_id: str) -> Dict[str, Any]:
|
||||
"""
|
||||
Get status of a consumption task.
|
||||
|
||||
Args:
|
||||
task_id: Task UUID from upload
|
||||
|
||||
Returns:
|
||||
Task status with state, result, etc.
|
||||
"""
|
||||
try:
|
||||
response = await self.client.get(
|
||||
f"{self.api_url}/tasks/",
|
||||
params={"task_id": task_id},
|
||||
)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
results = data.get("results", [])
|
||||
if results:
|
||||
return results[0]
|
||||
return {"status": "NOT_FOUND"}
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to get task status {task_id}: {e}")
|
||||
raise
|
||||
|
||||
async def update_document(
|
||||
self,
|
||||
document_id: int,
|
||||
title: Optional[str] = None,
|
||||
correspondent: Optional[int] = None,
|
||||
document_type: Optional[int] = None,
|
||||
tags: Optional[List[int]] = None,
|
||||
custom_fields: Optional[List[Dict[str, Any]]] = None,
|
||||
) -> PaperlessDocument:
|
||||
"""
|
||||
Update a document's metadata.
|
||||
|
||||
Args:
|
||||
document_id: Document ID to update
|
||||
title: New title
|
||||
correspondent: New correspondent ID
|
||||
document_type: New document type ID
|
||||
tags: New tag IDs (replaces existing)
|
||||
custom_fields: New custom field values
|
||||
|
||||
Returns:
|
||||
Updated document
|
||||
"""
|
||||
data = {}
|
||||
if title is not None:
|
||||
data["title"] = title
|
||||
if correspondent is not None:
|
||||
data["correspondent"] = correspondent
|
||||
if document_type is not None:
|
||||
data["document_type"] = document_type
|
||||
if tags is not None:
|
||||
data["tags"] = tags
|
||||
if custom_fields is not None:
|
||||
data["custom_fields"] = custom_fields
|
||||
|
||||
try:
|
||||
response = await self.client.patch(
|
||||
f"{self.api_url}/documents/{document_id}/",
|
||||
json=data,
|
||||
)
|
||||
response.raise_for_status()
|
||||
return self._parse_document(response.json())
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to update document {document_id}: {e}")
|
||||
raise
|
||||
|
||||
# =========================================================================
|
||||
# Custom Fields
|
||||
# =========================================================================
|
||||
|
||||
async def list_custom_fields(self) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
List all custom fields.
|
||||
|
||||
Returns:
|
||||
List of custom field definitions
|
||||
"""
|
||||
try:
|
||||
response = await self.client.get(f"{self.api_url}/custom_fields/")
|
||||
response.raise_for_status()
|
||||
return response.json().get("results", [])
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to list custom fields: {e}")
|
||||
raise
|
||||
|
||||
async def get_custom_field_by_name(self, name: str) -> Optional[Dict[str, Any]]:
|
||||
"""
|
||||
Get a custom field by name.
|
||||
|
||||
Args:
|
||||
name: Custom field name
|
||||
|
||||
Returns:
|
||||
Custom field definition or None
|
||||
"""
|
||||
fields = await self.list_custom_fields()
|
||||
for field in fields:
|
||||
if field.get("name") == name:
|
||||
return field
|
||||
return None
|
||||
|
||||
# =========================================================================
|
||||
# Tags, Correspondents, Document Types
|
||||
# =========================================================================
|
||||
|
||||
async def list_tags(self) -> List[Dict[str, Any]]:
|
||||
"""List all tags."""
|
||||
try:
|
||||
response = await self.client.get(f"{self.api_url}/tags/")
|
||||
response.raise_for_status()
|
||||
return response.json().get("results", [])
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to list tags: {e}")
|
||||
raise
|
||||
|
||||
async def list_correspondents(self) -> List[Dict[str, Any]]:
|
||||
"""List all correspondents."""
|
||||
try:
|
||||
response = await self.client.get(f"{self.api_url}/correspondents/")
|
||||
response.raise_for_status()
|
||||
return response.json().get("results", [])
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to list correspondents: {e}")
|
||||
raise
|
||||
|
||||
async def list_document_types(self) -> List[Dict[str, Any]]:
|
||||
"""List all document types."""
|
||||
try:
|
||||
response = await self.client.get(f"{self.api_url}/document_types/")
|
||||
response.raise_for_status()
|
||||
return response.json().get("results", [])
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to list document types: {e}")
|
||||
raise
|
||||
|
||||
# =========================================================================
|
||||
# Bulk Operations
|
||||
# =========================================================================
|
||||
|
||||
async def bulk_edit(
|
||||
self,
|
||||
document_ids: List[int],
|
||||
method: str,
|
||||
parameters: Optional[Dict[str, Any]] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""
|
||||
Bulk edit documents.
|
||||
|
||||
Args:
|
||||
document_ids: List of document IDs
|
||||
method: Operation (add_tag, remove_tag, set_correspondent, etc.)
|
||||
parameters: Operation parameters
|
||||
|
||||
Returns:
|
||||
Operation result
|
||||
"""
|
||||
data = {
|
||||
"documents": document_ids,
|
||||
"method": method,
|
||||
}
|
||||
if parameters:
|
||||
data["parameters"] = parameters
|
||||
|
||||
try:
|
||||
response = await self.client.post(
|
||||
f"{self.api_url}/documents/bulk_edit/",
|
||||
json=data,
|
||||
)
|
||||
response.raise_for_status()
|
||||
return response.json()
|
||||
except Exception as e:
|
||||
logger.error(f"Bulk edit failed: {e}")
|
||||
raise
|
||||
|
||||
# =========================================================================
|
||||
# Health Check
|
||||
# =========================================================================
|
||||
|
||||
async def health_check(self) -> bool:
|
||||
"""
|
||||
Check if Paperless-ngx is responding.
|
||||
|
||||
Returns:
|
||||
True if service is healthy
|
||||
"""
|
||||
try:
|
||||
response = await self.client.get(f"{self.api_url}/", timeout=5.0)
|
||||
return response.status_code < 400
|
||||
except Exception as e:
|
||||
logger.error(f"Paperless health check failed: {e}")
|
||||
return False
|
||||
|
||||
# =========================================================================
|
||||
# Helpers
|
||||
# =========================================================================
|
||||
|
||||
def _parse_document(self, data: Dict[str, Any]) -> PaperlessDocument:
|
||||
"""Parse API response into PaperlessDocument."""
|
||||
return PaperlessDocument(
|
||||
id=data.get("id", 0),
|
||||
title=data.get("title", ""),
|
||||
content=data.get("content", ""),
|
||||
created=data.get("created"),
|
||||
modified=data.get("modified"),
|
||||
added=data.get("added"),
|
||||
correspondent=data.get("correspondent"),
|
||||
document_type=data.get("document_type"),
|
||||
storage_path=data.get("storage_path"),
|
||||
tags=data.get("tags", []),
|
||||
archive_serial_number=data.get("archive_serial_number"),
|
||||
original_file_name=data.get("original_file_name"),
|
||||
archived_file_name=data.get("archived_file_name"),
|
||||
custom_fields=data.get("custom_fields", []),
|
||||
)
|
||||
@@ -11,7 +11,7 @@ Provides async vector operations with:
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.models import (
|
||||
Distance, VectorParams, PointStruct,
|
||||
Filter, FieldCondition, MatchValue
|
||||
Filter, FieldCondition, MatchValue, Range
|
||||
)
|
||||
from typing import List, Dict, Any, Optional
|
||||
import uuid
|
||||
@@ -551,6 +551,97 @@ class QdrantClientWrapper:
|
||||
logger.error(f"Search failed: {e}", exc_info=True)
|
||||
return []
|
||||
|
||||
async def scroll_all_points(
|
||||
self,
|
||||
collection_name: str,
|
||||
batch_size: int = 100,
|
||||
with_payload: bool = True,
|
||||
with_vectors: bool = False,
|
||||
filter_conditions: Optional[Dict[str, Any]] = None
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Scroll through all points in a collection.
|
||||
|
||||
Args:
|
||||
collection_name: Collection name
|
||||
batch_size: Number of points per batch
|
||||
with_payload: Include payload in results
|
||||
with_vectors: Include vectors in results
|
||||
filter_conditions: Optional filter conditions
|
||||
|
||||
Returns:
|
||||
List of all points with id and payload
|
||||
"""
|
||||
all_points = []
|
||||
offset = None
|
||||
|
||||
# Build filter if provided
|
||||
scroll_filter = None
|
||||
if filter_conditions:
|
||||
conditions = []
|
||||
for key, value in filter_conditions.items():
|
||||
conditions.append(
|
||||
FieldCondition(key=key, match=MatchValue(value=value))
|
||||
)
|
||||
scroll_filter = Filter(must=conditions)
|
||||
|
||||
try:
|
||||
while True:
|
||||
points, next_offset = self.client.scroll(
|
||||
collection_name=collection_name,
|
||||
scroll_filter=scroll_filter,
|
||||
limit=batch_size,
|
||||
offset=offset,
|
||||
with_payload=with_payload,
|
||||
with_vectors=with_vectors
|
||||
)
|
||||
|
||||
for point in points:
|
||||
all_points.append({
|
||||
"id": str(point.id),
|
||||
"payload": dict(point.payload) if point.payload else {}
|
||||
})
|
||||
|
||||
if next_offset is None:
|
||||
break
|
||||
offset = next_offset
|
||||
|
||||
return all_points
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to scroll collection {collection_name}: {e}", exc_info=True)
|
||||
return []
|
||||
|
||||
async def delete_by_ids(
|
||||
self,
|
||||
collection_name: str,
|
||||
point_ids: List[str]
|
||||
) -> int:
|
||||
"""
|
||||
Delete points by their IDs.
|
||||
|
||||
Args:
|
||||
collection_name: Collection name
|
||||
point_ids: List of point IDs to delete
|
||||
|
||||
Returns:
|
||||
Number of points deleted
|
||||
"""
|
||||
if not point_ids:
|
||||
return 0
|
||||
|
||||
try:
|
||||
self.client.delete(
|
||||
collection_name=collection_name,
|
||||
points_selector=point_ids
|
||||
)
|
||||
logger.info(f"Deleted {len(point_ids)} points from {collection_name}")
|
||||
return len(point_ids)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to delete points by IDs: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def list_collections(self) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
List all collections with stats.
|
||||
@@ -585,4 +676,134 @@ class QdrantClientWrapper:
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to list collections: {e}", exc_info=True)
|
||||
return []
|
||||
|
||||
# ========== Volatile Data Methods ==========
|
||||
|
||||
async def search_with_expiry_filter(
|
||||
self,
|
||||
collection_name: str,
|
||||
query_vector: List[float],
|
||||
current_timestamp: int,
|
||||
limit: int = 10,
|
||||
score_threshold: float = 0.7
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Search vectors filtering out expired records.
|
||||
|
||||
Args:
|
||||
collection_name: Collection name
|
||||
query_vector: Query embedding vector
|
||||
current_timestamp: Current time in milliseconds
|
||||
limit: Maximum results
|
||||
score_threshold: Minimum similarity score
|
||||
|
||||
Returns:
|
||||
List of non-expired search results
|
||||
"""
|
||||
# Filter: ttl_expiry > current_timestamp (not expired)
|
||||
expiry_filter = Filter(
|
||||
must=[
|
||||
FieldCondition(
|
||||
key="ttl_expiry",
|
||||
range=Range(gt=current_timestamp)
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
try:
|
||||
response = self.client.query_points(
|
||||
collection_name=collection_name,
|
||||
query=query_vector,
|
||||
limit=limit,
|
||||
score_threshold=score_threshold,
|
||||
query_filter=expiry_filter,
|
||||
with_payload=True
|
||||
)
|
||||
|
||||
return [
|
||||
{
|
||||
"id": str(point.id),
|
||||
"score": point.score,
|
||||
"payload": dict(point.payload)
|
||||
}
|
||||
for point in response.points
|
||||
]
|
||||
except Exception as e:
|
||||
logger.error(f"Volatile search failed: {e}", exc_info=True)
|
||||
return []
|
||||
|
||||
async def delete_expired_vectors(
|
||||
self,
|
||||
collection_name: str,
|
||||
current_timestamp: int
|
||||
) -> int:
|
||||
"""
|
||||
Delete all vectors where ttl_expiry < current_timestamp.
|
||||
|
||||
Args:
|
||||
collection_name: Collection name
|
||||
current_timestamp: Current time in milliseconds
|
||||
|
||||
Returns:
|
||||
Number of points deleted (approximate)
|
||||
"""
|
||||
# Filter: ttl_expiry < current_timestamp (expired)
|
||||
expiry_filter = Filter(
|
||||
must=[
|
||||
FieldCondition(
|
||||
key="ttl_expiry",
|
||||
range=Range(lt=current_timestamp)
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
try:
|
||||
# First count how many will be deleted (scroll to count)
|
||||
count = 0
|
||||
offset = None
|
||||
while True:
|
||||
points, next_offset = self.client.scroll(
|
||||
collection_name=collection_name,
|
||||
scroll_filter=expiry_filter,
|
||||
limit=100,
|
||||
offset=offset,
|
||||
with_payload=False
|
||||
)
|
||||
count += len(points)
|
||||
if next_offset is None:
|
||||
break
|
||||
offset = next_offset
|
||||
|
||||
if count == 0:
|
||||
return 0
|
||||
|
||||
# Delete expired points
|
||||
self.client.delete(
|
||||
collection_name=collection_name,
|
||||
points_selector=expiry_filter
|
||||
)
|
||||
|
||||
logger.info(f"Deleted {count} expired vectors from {collection_name}")
|
||||
return count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to delete expired vectors: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def get_volatile_collections(self) -> List[str]:
|
||||
"""
|
||||
Get all volatile collections (prefixed with 'volatile_').
|
||||
|
||||
Returns:
|
||||
List of volatile collection names
|
||||
"""
|
||||
try:
|
||||
collections = self.client.get_collections()
|
||||
return [
|
||||
c.name for c in collections.collections
|
||||
if c.name.startswith("volatile_")
|
||||
]
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to list volatile collections: {e}", exc_info=True)
|
||||
return []
|
||||
@@ -20,96 +20,34 @@ class WikiJSClient:
|
||||
Wiki.js GraphQL API client.
|
||||
|
||||
Documentation: https://docs.requarks.io/dev/api
|
||||
Authentication: Username/password login to get user-specific JWT token
|
||||
Authentication: API token (JWT) generated from Wiki.js admin panel
|
||||
"""
|
||||
|
||||
def __init__(self, base_url: str, username: str, password: str):
|
||||
def __init__(self, base_url: str, api_token: str):
|
||||
"""
|
||||
Initialize Wiki.js client.
|
||||
|
||||
Args:
|
||||
base_url: Wiki.js base URL (e.g., "http://wiki:3000")
|
||||
username: Wiki.js username (e.g., "librarian@schweitz.net")
|
||||
password: Wiki.js password
|
||||
api_token: Wiki.js API token (JWT from admin panel)
|
||||
"""
|
||||
self.base_url = base_url.rstrip("/")
|
||||
self.graphql_url = f"{self.base_url}/graphql"
|
||||
self.username = username
|
||||
self.password = password
|
||||
self.jwt_token: Optional[str] = None
|
||||
self.api_token = api_token
|
||||
self.client = httpx.AsyncClient(timeout=30.0)
|
||||
logger.info(f"Initialized Wiki.js client: {base_url} (user: {username})")
|
||||
auth_mode = "with API token" if api_token else "without auth (open API)"
|
||||
logger.info(f"Initialized Wiki.js client: {base_url} ({auth_mode})")
|
||||
|
||||
async def close(self):
|
||||
"""Close HTTP client"""
|
||||
await self.client.aclose()
|
||||
|
||||
async def login(self) -> bool:
|
||||
"""
|
||||
Authenticate with Wiki.js using username/password.
|
||||
|
||||
Returns:
|
||||
True if login successful, False otherwise
|
||||
"""
|
||||
login_mutation = """
|
||||
mutation Login($username: String!, $password: String!, $strategy: String!) {
|
||||
authentication {
|
||||
login(username: $username, password: $password, strategy: $strategy) {
|
||||
responseResult {
|
||||
succeeded
|
||||
errorCode
|
||||
message
|
||||
}
|
||||
jwt
|
||||
}
|
||||
}
|
||||
}
|
||||
"""
|
||||
|
||||
variables = {
|
||||
"username": self.username,
|
||||
"password": self.password,
|
||||
"strategy": "local"
|
||||
}
|
||||
|
||||
try:
|
||||
response = await self.client.post(
|
||||
self.graphql_url,
|
||||
headers={"Content-Type": "application/json"},
|
||||
json={"query": login_mutation, "variables": variables}
|
||||
)
|
||||
response.raise_for_status()
|
||||
result = response.json()
|
||||
|
||||
if "errors" in result:
|
||||
logger.error(f"Login failed: {result['errors']}")
|
||||
return False
|
||||
|
||||
login_result = result.get("data", {}).get("authentication", {}).get("login", {})
|
||||
response_result = login_result.get("responseResult", {})
|
||||
|
||||
if not response_result.get("succeeded"):
|
||||
logger.error(f"Login failed: {response_result.get('message')}")
|
||||
return False
|
||||
|
||||
self.jwt_token = login_result.get("jwt")
|
||||
if not self.jwt_token:
|
||||
logger.error("Login succeeded but no JWT token received")
|
||||
return False
|
||||
|
||||
logger.info(f"Successfully authenticated as {self.username}")
|
||||
return True
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Login failed: {e}", exc_info=True)
|
||||
return False
|
||||
|
||||
async def _ensure_authenticated(self):
|
||||
"""Ensure we have a valid JWT token, login if needed."""
|
||||
if not self.jwt_token:
|
||||
success = await self.login()
|
||||
if not success:
|
||||
raise Exception("Failed to authenticate with Wiki.js")
|
||||
def _get_headers(self) -> Dict[str, str]:
|
||||
"""Get request headers, optionally including auth token."""
|
||||
headers = {"Content-Type": "application/json"}
|
||||
if self.api_token:
|
||||
headers["Authorization"] = f"Bearer {self.api_token}"
|
||||
return headers
|
||||
|
||||
async def _execute_query(
|
||||
self,
|
||||
@@ -129,18 +67,12 @@ class WikiJSClient:
|
||||
Raises:
|
||||
Exception: If query fails or returns errors
|
||||
"""
|
||||
# Ensure we're authenticated before making requests
|
||||
await self._ensure_authenticated()
|
||||
|
||||
payload = {
|
||||
"query": query,
|
||||
"variables": variables or {}
|
||||
}
|
||||
|
||||
headers = {
|
||||
"Authorization": f"Bearer {self.jwt_token}",
|
||||
"Content-Type": "application/json"
|
||||
}
|
||||
headers = self._get_headers()
|
||||
|
||||
try:
|
||||
response = await self.client.post(
|
||||
|
||||
+49
-6
@@ -4,9 +4,20 @@ Following best practices: modular settings, environment-based config.
|
||||
"""
|
||||
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
from pydantic import Field
|
||||
from pydantic_settings import BaseSettings, SettingsConfigDict
|
||||
|
||||
# Read version from pyproject.toml
|
||||
try:
|
||||
import tomllib
|
||||
_pyproject_path = Path(__file__).parent.parent / "pyproject.toml"
|
||||
with open(_pyproject_path, "rb") as f:
|
||||
_pyproject = tomllib.load(f)
|
||||
__version__ = _pyproject["project"]["version"]
|
||||
except Exception:
|
||||
__version__ = "0.0.0" # Fallback if pyproject.toml not found
|
||||
|
||||
|
||||
class Settings(BaseSettings):
|
||||
"""Application settings loaded from environment variables."""
|
||||
@@ -32,8 +43,10 @@ class Settings(BaseSettings):
|
||||
|
||||
# Wiki.js Configuration
|
||||
wikijs_url: str = Field(default="http://wiki:3000", description="Wiki.js URL")
|
||||
wikijs_username: str = Field(..., description="Wiki.js username")
|
||||
wikijs_password: str = Field(..., description="Wiki.js password")
|
||||
wiki_graphql_api: str = Field(default="", description="Wiki.js GraphQL API token (optional - API may be open)")
|
||||
# Legacy auth fields - kept for backwards compatibility but deprecated
|
||||
wikijs_username: str = Field(default="", description="Wiki.js username (deprecated, use wiki_graphql_api)")
|
||||
wikijs_password: str = Field(default="", description="Wiki.js password (deprecated, use wiki_graphql_api)")
|
||||
|
||||
# Wiki.js Database Configuration (for change listener)
|
||||
wikijs_db_host: str = Field(default="postgres-shared", description="Wiki.js PostgreSQL host")
|
||||
@@ -51,16 +64,17 @@ class Settings(BaseSettings):
|
||||
# SearXNG Configuration
|
||||
searxng_url: str = Field(default="http://searxng:8080", description="SearXNG URL")
|
||||
|
||||
# Ollama Configuration (for embeddings)
|
||||
# Ollama Configuration
|
||||
ollama_url: str = Field(default="http://ollama:11434", description="Ollama URL")
|
||||
ollama_model: str = Field(default="nomic-embed-text", description="Ollama embedding model")
|
||||
ollama_model: str = Field(default="mistral-nemo-large:latest", description="Ollama LLM model")
|
||||
ollama_embedding_model: str = Field(default="nomic-embed-text", description="Ollama embedding model")
|
||||
|
||||
# HybridRAG Configuration
|
||||
reranker_model: str = Field(default="mistral-nemo", description="Model for LLM re-ranking")
|
||||
reranker_enabled: bool = Field(default=True, description="Enable LLM re-ranking")
|
||||
hybrid_rag_vector_limit: int = Field(default=10, ge=1, le=50, description="Vector search limit")
|
||||
hybrid_rag_graph_limit: int = Field(default=10, ge=1, le=50, description="Graph search limit")
|
||||
hybrid_rag_web_limit: int = Field(default=5, ge=1, le=20, description="Web search limit")
|
||||
vector_similarity_threshold: float = Field(default=0.7, ge=0.0, le=1.0, description="Minimum similarity score for vector results")
|
||||
|
||||
# Entity Linking Fuzzy Matching Configuration
|
||||
entity_linking_min_confidence: float = Field(default=0.70, ge=0.0, le=1.0, description="Minimum confidence for entity-document matching")
|
||||
@@ -75,9 +89,38 @@ class Settings(BaseSettings):
|
||||
|
||||
# Application
|
||||
app_name: str = Field(default="Library Desk", description="Application name")
|
||||
app_version: str = Field(default="1.0.0", description="Application version")
|
||||
app_version: str = Field(default=__version__, description="Application version")
|
||||
debug: bool = Field(default=False, description="Debug mode")
|
||||
|
||||
# RAG Search Configuration
|
||||
search_cache_ttl: int = Field(default=300, ge=0, le=3600, description="Search cache TTL in seconds")
|
||||
search_timeout: int = Field(default=10, ge=1, le=60, description="SearXNG timeout in seconds")
|
||||
search_default_limit: int = Field(default=10, ge=1, le=20, description="Default number of search results")
|
||||
|
||||
# Content Extraction Configuration
|
||||
content_extraction_timeout: int = Field(default=5, ge=1, le=30, description="Trafilatura per-URL timeout in seconds")
|
||||
content_max_length: int = Field(default=2000, ge=500, le=10000, description="Max extracted content length per result")
|
||||
|
||||
# Paperless-ngx Configuration
|
||||
paperless_url: str = Field(default="http://paperless:8000", description="Paperless-ngx URL")
|
||||
paperless_token: str = Field(default="", description="Paperless-ngx API token")
|
||||
paperless_timeout: int = Field(default=30, ge=5, le=120, description="Paperless API timeout in seconds")
|
||||
|
||||
# Document Store Configuration
|
||||
document_store_enabled: bool = Field(default=True, description="Enable document store feature")
|
||||
document_catalog_path_prefix: str = Field(default="docs", description="Wiki path prefix for catalog pages")
|
||||
|
||||
# Volatile Cache Configuration
|
||||
volatile_cache_enabled: bool = Field(default=True, description="Enable volatile cache feature")
|
||||
volatile_default_ttl: int = Field(default=3600, ge=60, le=86400, description="Default TTL in seconds")
|
||||
volatile_weather_ttl: int = Field(default=1800, ge=60, le=7200, description="Weather data TTL in seconds")
|
||||
volatile_news_ttl: int = Field(default=7200, ge=300, le=86400, description="News data TTL in seconds")
|
||||
volatile_financial_ttl: int = Field(default=300, ge=60, le=3600, description="Financial data TTL in seconds")
|
||||
|
||||
# Maintenance Configuration
|
||||
maintenance_orphan_cleanup_enabled: bool = Field(default=True, description="Enable automatic orphan cleanup")
|
||||
maintenance_cleanup_batch_size: int = Field(default=100, ge=10, le=1000, description="Cleanup batch size")
|
||||
|
||||
@property
|
||||
def qdrant_url(self) -> str:
|
||||
"""Computed Qdrant URL."""
|
||||
|
||||
+118
-15
@@ -13,12 +13,16 @@ from typing import Annotated
|
||||
from fastapi import Depends
|
||||
import logging
|
||||
|
||||
import redis.asyncio as aioredis
|
||||
|
||||
from src.config import Settings, get_settings
|
||||
from src.clients.neo4j_client import Neo4jClient
|
||||
from src.clients.qdrant_client import QdrantClientWrapper
|
||||
from src.clients.wikijs_client import WikiJSClient
|
||||
from src.clients.searxng_client import SearXNGClient
|
||||
from src.clients.ollama_client import OllamaClient
|
||||
from src.clients.content_extractor import ContentExtractor
|
||||
from src.clients.paperless_client import PaperlessClient
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -73,13 +77,12 @@ def get_wikijs_client() -> WikiJSClient:
|
||||
Get Wiki.js client singleton.
|
||||
|
||||
Returns:
|
||||
Initialized Wiki.js GraphQL client with username/password auth
|
||||
Initialized Wiki.js GraphQL client with API token auth
|
||||
"""
|
||||
settings = get_settings()
|
||||
client = WikiJSClient(
|
||||
base_url=settings.wikijs_url,
|
||||
username=settings.wikijs_username,
|
||||
password=settings.wikijs_password
|
||||
api_token=settings.wiki_graphql_api
|
||||
)
|
||||
logger.debug("Created Wiki.js client instance")
|
||||
return client
|
||||
@@ -110,12 +113,71 @@ def get_ollama_client() -> OllamaClient:
|
||||
settings = get_settings()
|
||||
client = OllamaClient(
|
||||
base_url=settings.ollama_url,
|
||||
model=settings.ollama_model
|
||||
model=settings.ollama_embedding_model
|
||||
)
|
||||
logger.debug("Created Ollama client instance")
|
||||
return client
|
||||
|
||||
|
||||
@lru_cache
|
||||
def get_redis_client() -> aioredis.Redis:
|
||||
"""
|
||||
Get Redis client singleton for caching.
|
||||
|
||||
Returns:
|
||||
Async Redis client connected to the configured database
|
||||
|
||||
Note: Uses Redis DB 4 (configured for library-desk)
|
||||
"""
|
||||
settings = get_settings()
|
||||
client = aioredis.from_url(
|
||||
settings.redis_url,
|
||||
encoding="utf-8",
|
||||
decode_responses=True
|
||||
)
|
||||
logger.debug(f"Created Redis client: {settings.redis_url}")
|
||||
return client
|
||||
|
||||
|
||||
@lru_cache
|
||||
def get_content_extractor() -> ContentExtractor:
|
||||
"""
|
||||
Get ContentExtractor singleton.
|
||||
|
||||
Returns:
|
||||
Initialized content extraction client using Trafilatura
|
||||
"""
|
||||
settings = get_settings()
|
||||
extractor = ContentExtractor(
|
||||
timeout=settings.content_extraction_timeout,
|
||||
max_length=settings.content_max_length
|
||||
)
|
||||
logger.debug("Created ContentExtractor instance")
|
||||
return extractor
|
||||
|
||||
|
||||
@lru_cache
|
||||
def get_paperless_client() -> PaperlessClient:
|
||||
"""
|
||||
Get Paperless-ngx client singleton.
|
||||
|
||||
Returns:
|
||||
Initialized Paperless-ngx REST API client
|
||||
|
||||
Note: Returns None-like client if paperless_token is not configured
|
||||
"""
|
||||
settings = get_settings()
|
||||
if not settings.paperless_token:
|
||||
logger.warning("Paperless token not configured - document storage disabled")
|
||||
client = PaperlessClient(
|
||||
base_url=settings.paperless_url,
|
||||
token=settings.paperless_token,
|
||||
timeout=settings.paperless_timeout
|
||||
)
|
||||
logger.debug(f"Created Paperless client: {settings.paperless_url}")
|
||||
return client
|
||||
|
||||
|
||||
# Type aliases for FastAPI endpoint dependencies
|
||||
# Usage: def my_endpoint(neo4j: Neo4jDep):
|
||||
Neo4jDep = Annotated[Neo4jClient, Depends(get_neo4j_client)]
|
||||
@@ -123,6 +185,9 @@ QdrantDep = Annotated[QdrantClientWrapper, Depends(get_qdrant_client)]
|
||||
WikiJSDep = Annotated[WikiJSClient, Depends(get_wikijs_client)]
|
||||
SearXNGDep = Annotated[SearXNGClient, Depends(get_searxng_client)]
|
||||
OllamaDep = Annotated[OllamaClient, Depends(get_ollama_client)]
|
||||
RedisDep = Annotated[aioredis.Redis, Depends(get_redis_client)]
|
||||
ContentExtractorDep = Annotated[ContentExtractor, Depends(get_content_extractor)]
|
||||
PaperlessDep = Annotated[PaperlessClient, Depends(get_paperless_client)]
|
||||
|
||||
|
||||
# Lifecycle management functions
|
||||
@@ -161,6 +226,21 @@ async def startup_clients():
|
||||
logger.error(f"✗ Ollama health check failed: {e}")
|
||||
pass
|
||||
|
||||
# Check Paperless availability
|
||||
settings = get_settings()
|
||||
if settings.paperless_token:
|
||||
try:
|
||||
paperless = get_paperless_client()
|
||||
is_healthy = await paperless.health_check()
|
||||
if is_healthy:
|
||||
logger.info(f"✓ Paperless-ngx ready: {settings.paperless_url}")
|
||||
else:
|
||||
logger.warning("✗ Paperless-ngx not responding")
|
||||
except Exception as e:
|
||||
logger.error(f"✗ Paperless health check failed: {e}")
|
||||
else:
|
||||
logger.info("○ Paperless-ngx not configured (document storage disabled)")
|
||||
|
||||
# Qdrant, Wiki.js, SearXNG are lazy-initialized
|
||||
logger.info("Service clients startup complete")
|
||||
|
||||
@@ -189,7 +269,8 @@ async def shutdown_clients():
|
||||
clients_to_close = [
|
||||
("Wiki.js", get_wikijs_client()),
|
||||
("SearXNG", get_searxng_client()),
|
||||
("Ollama", get_ollama_client())
|
||||
("Ollama", get_ollama_client()),
|
||||
("Paperless", get_paperless_client()),
|
||||
]
|
||||
|
||||
for name, client in clients_to_close:
|
||||
@@ -271,6 +352,18 @@ async def check_service_health() -> dict:
|
||||
logger.error(f"Ollama health check failed: {e}")
|
||||
health["ollama"] = False
|
||||
|
||||
# Paperless-ngx
|
||||
settings = get_settings()
|
||||
if settings.paperless_token:
|
||||
try:
|
||||
paperless = get_paperless_client()
|
||||
health["paperless"] = await paperless.health_check()
|
||||
except Exception as e:
|
||||
logger.error(f"Paperless health check failed: {e}")
|
||||
health["paperless"] = False
|
||||
else:
|
||||
health["paperless"] = None # Not configured
|
||||
|
||||
return health
|
||||
|
||||
|
||||
@@ -336,20 +429,21 @@ def get_hybrid_rag_service() -> "HybridRAGService":
|
||||
graph_service=get_graph_service(),
|
||||
searxng_client=get_searxng_client(),
|
||||
ollama_client=get_ollama_client(),
|
||||
content_extractor=get_content_extractor(),
|
||||
settings=get_settings()
|
||||
)
|
||||
|
||||
|
||||
# Utility: Get default user from settings or multi_tenancy
|
||||
def get_default_user() -> str:
|
||||
"""
|
||||
Get default user for operations.
|
||||
|
||||
Returns:
|
||||
Default user identifier
|
||||
"""
|
||||
from src.core.multi_tenancy import DEFAULT_USER
|
||||
return DEFAULT_USER
|
||||
@lru_cache
|
||||
def get_rag_search_service() -> "RAGSearchService":
|
||||
"""Get RAGSearchService singleton."""
|
||||
from src.services.rag_search_service import RAGSearchService
|
||||
return RAGSearchService(
|
||||
searxng_client=get_searxng_client(),
|
||||
content_extractor=get_content_extractor(),
|
||||
redis_client=get_redis_client(),
|
||||
settings=get_settings()
|
||||
)
|
||||
|
||||
|
||||
# Authentication
|
||||
@@ -382,3 +476,12 @@ async def verify_api_key(
|
||||
detail="Invalid API key"
|
||||
)
|
||||
return credentials.credentials
|
||||
|
||||
|
||||
# Service type aliases for FastAPI endpoint dependencies
|
||||
# These are defined after the factory functions
|
||||
from src.services.vector_service import VectorService
|
||||
from src.services.graph_service import GraphService
|
||||
|
||||
VectorServiceDep = Annotated[VectorService, Depends(get_vector_service)]
|
||||
GraphServiceDep = Annotated[GraphService, Depends(get_graph_service)]
|
||||
|
||||
+81
-103
@@ -8,7 +8,7 @@ Following best practices:
|
||||
- OpenAPI documentation
|
||||
"""
|
||||
|
||||
from fastapi import FastAPI, HTTPException, Depends
|
||||
from fastapi import FastAPI, HTTPException, Depends, Query
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
from fastapi.staticfiles import StaticFiles
|
||||
from pydantic import BaseModel
|
||||
@@ -16,8 +16,11 @@ from typing import Dict, Any
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
from src.config import Settings, get_settings
|
||||
from src.core.dependencies import verify_api_key
|
||||
from src.config import Settings, get_settings, __version__
|
||||
from src.core.dependencies import (
|
||||
verify_api_key, QdrantDep, WikiJSDep, OllamaDep, Neo4jDep
|
||||
)
|
||||
from src.core.multi_tenancy import DEFAULT_USER
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(
|
||||
@@ -30,7 +33,7 @@ logger = logging.getLogger(__name__)
|
||||
app = FastAPI(
|
||||
title="Library Desk API",
|
||||
description="Coordination service for The Library system - HybridRAG queries, document ingestion, entity extraction, and mind map generation",
|
||||
version="1.0.0",
|
||||
version=__version__,
|
||||
docs_url="/docs",
|
||||
redoc_url="/redoc",
|
||||
)
|
||||
@@ -45,7 +48,11 @@ app.add_middleware(
|
||||
)
|
||||
|
||||
# Register routers
|
||||
from src.routers import wiki, tools, graph, vector, hybrid_rag, consolidation, ingestion, entity_linking, webhooks
|
||||
from src.routers import (
|
||||
wiki, tools, graph, vector, hybrid_rag, consolidation,
|
||||
ingestion, entity_linking, webhooks, rag_search, content,
|
||||
maintenance, volatile, documents
|
||||
)
|
||||
|
||||
app.include_router(wiki.router)
|
||||
app.include_router(tools.router)
|
||||
@@ -56,6 +63,11 @@ app.include_router(consolidation.router)
|
||||
app.include_router(ingestion.router)
|
||||
app.include_router(entity_linking.router)
|
||||
app.include_router(webhooks.router)
|
||||
app.include_router(rag_search.router)
|
||||
app.include_router(content.router)
|
||||
app.include_router(maintenance.router)
|
||||
app.include_router(volatile.router)
|
||||
app.include_router(documents.router)
|
||||
|
||||
# Mount static files directory for Wiki.js integration scripts
|
||||
static_dir = Path(__file__).parent.parent / "static"
|
||||
@@ -73,13 +85,6 @@ class HealthResponse(BaseModel):
|
||||
services: Dict[str, Any]
|
||||
|
||||
|
||||
class StatsResponse(BaseModel):
|
||||
"""Statistics response model."""
|
||||
wiki_pages: int
|
||||
neo4j_nodes: int
|
||||
qdrant_vectors: int
|
||||
|
||||
|
||||
# Routes
|
||||
@app.get("/", tags=["Root"])
|
||||
async def root() -> Dict[str, str]:
|
||||
@@ -136,77 +141,6 @@ async def health(settings: Settings = Depends(get_settings)) -> HealthResponse:
|
||||
)
|
||||
|
||||
|
||||
@app.get("/stats", response_model=StatsResponse, tags=["System"])
|
||||
async def stats(
|
||||
api_key: str = Depends(verify_api_key)
|
||||
) -> StatsResponse:
|
||||
"""
|
||||
Get system statistics.
|
||||
Protected endpoint - requires API key.
|
||||
|
||||
TODO: Implement actual stats gathering from:
|
||||
- Neo4j (node count)
|
||||
- Qdrant (vector count)
|
||||
- Wiki.js (page count)
|
||||
"""
|
||||
return StatsResponse(
|
||||
wiki_pages=0,
|
||||
neo4j_nodes=0,
|
||||
qdrant_vectors=0
|
||||
)
|
||||
|
||||
|
||||
# Ingestion endpoints (for Scheduler integration)
|
||||
@app.post("/ingest/document", tags=["Ingestion"])
|
||||
async def ingest_document(
|
||||
document: Dict[str, Any],
|
||||
api_key: str = Depends(verify_api_key)
|
||||
) -> Dict[str, Any]:
|
||||
"""
|
||||
Ingest a single document for indexing.
|
||||
Used by The Scheduler to add mirrored documentation to the knowledge base.
|
||||
|
||||
Expected fields:
|
||||
- source: str (e.g., "github", "gitea")
|
||||
- repository: str (e.g., "anthropic-cookbook")
|
||||
- path: str (file path)
|
||||
- content: str (document content)
|
||||
- metadata: dict (commit, author, tags, etc.)
|
||||
|
||||
TODO: Implement document ingestion pipeline:
|
||||
1. Chunk content
|
||||
2. Generate embeddings (Ollama)
|
||||
3. Extract entities (NLP)
|
||||
4. Index in Qdrant
|
||||
5. Create graph nodes/relationships in Neo4j
|
||||
"""
|
||||
return {
|
||||
"message": "Document ingestion not yet implemented",
|
||||
"document_id": f"doc_{document.get('path', 'unknown')}",
|
||||
"status": "stub"
|
||||
}
|
||||
|
||||
|
||||
@app.post("/ingest/batch", tags=["Ingestion"])
|
||||
async def batch_ingest(
|
||||
batch: Dict[str, Any],
|
||||
api_key: str = Depends(verify_api_key)
|
||||
) -> Dict[str, Any]:
|
||||
"""
|
||||
Ingest multiple documents in a batch.
|
||||
More efficient than individual ingestion for large syncs.
|
||||
|
||||
TODO: Implement batch processing with task queue
|
||||
"""
|
||||
document_count = len(batch.get("documents", []))
|
||||
return {
|
||||
"message": "Batch ingestion not yet implemented",
|
||||
"batch_id": "batch_stub",
|
||||
"total_documents": document_count,
|
||||
"status": "stub"
|
||||
}
|
||||
|
||||
|
||||
@app.post("/ingest/check-updates", tags=["Ingestion"])
|
||||
async def check_updates(
|
||||
documents: Dict[str, Any],
|
||||
@@ -264,41 +198,85 @@ async def get_repo_status(
|
||||
}
|
||||
|
||||
|
||||
# Query endpoints (stubs for future implementation)
|
||||
# NOTE: /query/hybrid is now implemented in routers/hybrid_rag.py
|
||||
# Query endpoints
|
||||
# NOTE: /query/hybrid is implemented in routers/hybrid_rag.py
|
||||
|
||||
@app.post("/query/semantic", tags=["Query"])
|
||||
async def semantic_query(
|
||||
query: Dict[str, Any],
|
||||
query: str = Query(..., min_length=1, description="Search query text"),
|
||||
user: str = Query(default=DEFAULT_USER, description="User identifier"),
|
||||
limit: int = Query(default=10, ge=1, le=100, description="Maximum results"),
|
||||
score_threshold: float = Query(default=0.5, ge=0.0, le=1.0, description="Minimum similarity score"),
|
||||
qdrant_client: QdrantDep = None,
|
||||
wiki_client: WikiJSDep = None,
|
||||
ollama_client: OllamaDep = None,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
) -> Dict[str, Any]:
|
||||
):
|
||||
"""
|
||||
Semantic search via Qdrant.
|
||||
Pure vector similarity search.
|
||||
Semantic search via Qdrant vector similarity.
|
||||
|
||||
TODO: Implement semantic search
|
||||
Searches document chunks using embedding similarity. Returns matching
|
||||
chunks with relevance scores, page titles, and paths.
|
||||
|
||||
**Example:**
|
||||
```
|
||||
POST /query/semantic?query=docker%20configuration&user=jpmschweitzer&limit=10
|
||||
```
|
||||
|
||||
**Returns:** List of matching chunks with similarity scores (0-1)
|
||||
"""
|
||||
return {
|
||||
"message": "Semantic search not yet implemented",
|
||||
"query": query
|
||||
}
|
||||
from src.services.vector_service import VectorService
|
||||
|
||||
vector_service = VectorService(qdrant_client, wiki_client, ollama_client)
|
||||
try:
|
||||
return await vector_service.search(
|
||||
query=query,
|
||||
user=user,
|
||||
limit=limit,
|
||||
score_threshold=score_threshold
|
||||
)
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
except Exception as e:
|
||||
logger.error(f"Semantic search failed: {e}", exc_info=True)
|
||||
raise HTTPException(status_code=500, detail="Search failed")
|
||||
|
||||
|
||||
@app.post("/query/graph", tags=["Query"])
|
||||
async def graph_query(
|
||||
query: Dict[str, Any],
|
||||
query: str = Query(..., description="Cypher query to execute"),
|
||||
user: str = Query(default=DEFAULT_USER, description="User for scoping (auto-filters results)"),
|
||||
neo4j_client: Neo4jDep = None,
|
||||
wiki_client: WikiJSDep = None,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
) -> Dict[str, Any]:
|
||||
):
|
||||
"""
|
||||
Graph traversal via Neo4j.
|
||||
Execute Cypher queries.
|
||||
Execute a Cypher query against the Neo4j knowledge graph.
|
||||
|
||||
TODO: Implement graph queries
|
||||
Queries are automatically scoped to the user's data for security.
|
||||
Use this for custom graph traversals beyond what /graph/nodes provides.
|
||||
|
||||
**Example:**
|
||||
```
|
||||
POST /query/graph?query=MATCH%20(d:Document)-[:MENTIONS]->(p:Person)%20RETURN%20d,p&user=jpmschweitzer
|
||||
```
|
||||
|
||||
**Security:** All queries are user-scoped to prevent cross-user data access.
|
||||
"""
|
||||
return {
|
||||
"message": "Graph query not yet implemented",
|
||||
"query": query
|
||||
}
|
||||
from src.services.graph_service import GraphService
|
||||
|
||||
graph_service = GraphService(neo4j_client, wiki_client)
|
||||
try:
|
||||
return await graph_service.execute_query(
|
||||
query=query,
|
||||
parameters={},
|
||||
user=user
|
||||
)
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
except Exception as e:
|
||||
logger.error(f"Graph query failed: {e}", exc_info=True)
|
||||
raise HTTPException(status_code=500, detail="Query execution failed")
|
||||
|
||||
|
||||
# Deduplication endpoints
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
"""
|
||||
Content extraction models for Library Desk.
|
||||
|
||||
Pydantic models for content extraction requests and responses.
|
||||
"""
|
||||
|
||||
from typing import Optional, List
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
|
||||
class ContentExtractionResult(BaseModel):
|
||||
"""Result of extracting content from a single URL."""
|
||||
|
||||
url: str = Field(..., description="The URL that was processed")
|
||||
title: Optional[str] = Field(None, description="Page title if extracted")
|
||||
content: str = Field("", description="Extracted main text content")
|
||||
author: Optional[str] = Field(None, description="Author if available")
|
||||
date: Optional[str] = Field(None, description="Publication date if available (ISO format)")
|
||||
language: Optional[str] = Field(None, description="Detected language code")
|
||||
success: bool = Field(..., description="Whether extraction succeeded")
|
||||
error: Optional[str] = Field(None, description="Error message if extraction failed")
|
||||
|
||||
|
||||
class ContentExtractionRequest(BaseModel):
|
||||
"""Request to extract content from a single URL."""
|
||||
|
||||
url: str = Field(..., min_length=1, description="URL to extract content from")
|
||||
include_metadata: bool = Field(default=True, description="Include title, author, date metadata")
|
||||
max_length: Optional[int] = Field(
|
||||
None,
|
||||
ge=100,
|
||||
le=50000,
|
||||
description="Override default max content length"
|
||||
)
|
||||
|
||||
|
||||
class ContentExtractionResponse(BaseModel):
|
||||
"""Response for single URL extraction."""
|
||||
|
||||
result: ContentExtractionResult
|
||||
extraction_time_ms: int = Field(..., ge=0, description="Time taken to extract content")
|
||||
|
||||
|
||||
class BatchContentExtractionRequest(BaseModel):
|
||||
"""Request to extract content from multiple URLs."""
|
||||
|
||||
urls: List[str] = Field(
|
||||
...,
|
||||
min_length=1,
|
||||
max_length=20,
|
||||
description="URLs to extract content from (max 20)"
|
||||
)
|
||||
include_metadata: bool = Field(default=True, description="Include title, author, date metadata")
|
||||
max_length: Optional[int] = Field(
|
||||
None,
|
||||
ge=100,
|
||||
le=50000,
|
||||
description="Override default max content length"
|
||||
)
|
||||
|
||||
|
||||
class BatchContentExtractionResponse(BaseModel):
|
||||
"""Response for batch URL extraction."""
|
||||
|
||||
results: List[ContentExtractionResult]
|
||||
total_urls: int = Field(..., ge=0, description="Total number of URLs processed")
|
||||
successful: int = Field(..., ge=0, description="Number of successful extractions")
|
||||
failed: int = Field(..., ge=0, description="Number of failed extractions")
|
||||
extraction_time_ms: int = Field(..., ge=0, description="Total time for batch extraction")
|
||||
@@ -0,0 +1,207 @@
|
||||
"""
|
||||
Document storage models for Library Desk.
|
||||
|
||||
Models for Paperless-ngx document management, virus scanning,
|
||||
and document sync operations.
|
||||
"""
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import Dict, Any, Optional, List
|
||||
from datetime import datetime
|
||||
from enum import Enum
|
||||
|
||||
|
||||
class DocumentType(str, Enum):
|
||||
"""Types of documents supported in the document store."""
|
||||
PDF = "pdf"
|
||||
IMAGE = "image"
|
||||
VIDEO = "video"
|
||||
TEXT = "text"
|
||||
ARCHIVE = "archive"
|
||||
OTHER = "other"
|
||||
|
||||
|
||||
class SyncStatus(str, Enum):
|
||||
"""Status of document sync with Library Desk."""
|
||||
PENDING = "pending"
|
||||
INDEXED = "indexed"
|
||||
FAILED = "failed"
|
||||
SKIPPED = "skipped"
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Document Models
|
||||
# =============================================================================
|
||||
|
||||
|
||||
class DocumentMetadata(BaseModel):
|
||||
"""Metadata for a document in Paperless-ngx."""
|
||||
paperless_id: int = Field(..., description="Paperless-ngx document ID")
|
||||
title: str = Field(..., description="Document title")
|
||||
filename: Optional[str] = Field(None, description="Original filename")
|
||||
content: Optional[str] = Field(None, description="Extracted text content")
|
||||
created: Optional[datetime] = Field(None, description="Document creation date")
|
||||
modified: Optional[datetime] = Field(None, description="Last modification date")
|
||||
added: Optional[datetime] = Field(None, description="Date added to Paperless")
|
||||
correspondent: Optional[str] = Field(None, description="Correspondent name")
|
||||
document_type: Optional[str] = Field(None, description="Document type name")
|
||||
tags: List[str] = Field(default_factory=list, description="Tag names")
|
||||
custom_fields: Dict[str, Any] = Field(default_factory=dict, description="Custom field values")
|
||||
|
||||
|
||||
class DocumentRecord(BaseModel):
|
||||
"""A document record with sync status."""
|
||||
metadata: DocumentMetadata = Field(..., description="Document metadata from Paperless")
|
||||
sync_status: SyncStatus = Field(default=SyncStatus.PENDING, description="Library Desk sync status")
|
||||
indexed_at: Optional[datetime] = Field(None, description="When indexed in Library Desk")
|
||||
collection: Optional[str] = Field(None, description="Collection name (e.g., 'fastapi-docs')")
|
||||
source_url: Optional[str] = Field(None, description="Original source URL if uploaded via HybridRAG")
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Upload Request/Response Models
|
||||
# =============================================================================
|
||||
|
||||
|
||||
class DocumentUploadRequest(BaseModel):
|
||||
"""Request to upload a document to Paperless-ngx."""
|
||||
url: Optional[str] = Field(None, description="URL to download document from")
|
||||
title: Optional[str] = Field(None, description="Document title (derived from filename if not set)")
|
||||
collection: Optional[str] = Field(None, description="Collection to add document to")
|
||||
tags: List[str] = Field(default_factory=list, description="Tags to apply")
|
||||
correspondent: Optional[str] = Field(None, description="Correspondent name")
|
||||
document_type: Optional[str] = Field(None, description="Document type name")
|
||||
|
||||
|
||||
class DocumentUploadResponse(BaseModel):
|
||||
"""Response from document upload."""
|
||||
task_id: str = Field(..., description="Paperless task ID for tracking")
|
||||
filename: str = Field(..., description="Uploaded filename")
|
||||
message: str = Field(..., description="Status message")
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Webhook Models
|
||||
# =============================================================================
|
||||
|
||||
|
||||
class PaperlessWebhookPayload(BaseModel):
|
||||
"""
|
||||
Payload from Paperless-ngx webhook.
|
||||
|
||||
Supports Jinja template format:
|
||||
- doc_url: Contains document ID in URL path (e.g., http://paperless:8000/documents/123/)
|
||||
- title: Document title from {{ doc_title }}
|
||||
"""
|
||||
doc_url: str = Field(..., description="Paperless document URL containing ID")
|
||||
title: Optional[str] = Field(None, description="Document title")
|
||||
|
||||
class Config:
|
||||
extra = "ignore" # Ignore extra fields
|
||||
|
||||
@property
|
||||
def document_id(self) -> int:
|
||||
"""Extract document ID from doc_url."""
|
||||
import re
|
||||
match = re.search(r'/documents/(\d+)/?', self.doc_url)
|
||||
if match:
|
||||
return int(match.group(1))
|
||||
raise ValueError(f"Cannot extract document ID from URL: {self.doc_url}")
|
||||
|
||||
|
||||
class WebhookResponse(BaseModel):
|
||||
"""Response to webhook processing."""
|
||||
document_id: int = Field(..., description="Processed document ID")
|
||||
status: str = Field(..., description="Processing status")
|
||||
indexed: bool = Field(..., description="Whether document was indexed")
|
||||
message: Optional[str] = Field(None, description="Additional details")
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Sync Models
|
||||
# =============================================================================
|
||||
|
||||
|
||||
class SyncRequest(BaseModel):
|
||||
"""Request to sync documents from Paperless-ngx."""
|
||||
since: Optional[datetime] = Field(None, description="Only sync documents modified after this time")
|
||||
collection: Optional[str] = Field(None, description="Only sync documents in this collection")
|
||||
limit: int = Field(default=100, ge=1, le=1000, description="Maximum documents to sync")
|
||||
force_reindex: bool = Field(default=False, description="Re-index already indexed documents")
|
||||
|
||||
|
||||
class SyncResult(BaseModel):
|
||||
"""Result of a sync operation."""
|
||||
documents_found: int = Field(..., description="Total documents matching criteria")
|
||||
documents_indexed: int = Field(..., description="Successfully indexed")
|
||||
documents_skipped: int = Field(..., description="Skipped (already indexed)")
|
||||
documents_failed: int = Field(..., description="Failed to index")
|
||||
errors: List[str] = Field(default_factory=list, description="Error messages")
|
||||
duration_seconds: float = Field(..., description="Sync duration")
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Collection Models
|
||||
# =============================================================================
|
||||
|
||||
|
||||
class Collection(BaseModel):
|
||||
"""A logical grouping of documents."""
|
||||
name: str = Field(..., description="Collection name (e.g., 'fastapi-docs')")
|
||||
description: Optional[str] = Field(None, description="Collection description")
|
||||
document_count: int = Field(default=0, description="Number of documents")
|
||||
source: Optional[str] = Field(None, description="Source (e.g., 'github.com/tiangolo/fastapi')")
|
||||
last_sync: Optional[datetime] = Field(None, description="Last sync timestamp")
|
||||
wiki_page: Optional[str] = Field(None, description="Wiki catalog page path")
|
||||
|
||||
|
||||
class CollectionListResponse(BaseModel):
|
||||
"""Response listing all collections."""
|
||||
collections: List[Collection] = Field(..., description="List of collections")
|
||||
total_documents: int = Field(..., description="Total documents across all collections")
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Search Models
|
||||
# =============================================================================
|
||||
|
||||
|
||||
class DocumentSearchRequest(BaseModel):
|
||||
"""Request to search documents."""
|
||||
query: str = Field(..., min_length=1, description="Search query")
|
||||
collection: Optional[str] = Field(None, description="Limit to collection")
|
||||
document_type: Optional[DocumentType] = Field(None, description="Filter by type")
|
||||
limit: int = Field(default=10, ge=1, le=50, description="Maximum results")
|
||||
include_content: bool = Field(default=False, description="Include full text content")
|
||||
|
||||
|
||||
class DocumentSearchHit(BaseModel):
|
||||
"""A document search result."""
|
||||
paperless_id: int = Field(..., description="Paperless document ID")
|
||||
title: str = Field(..., description="Document title")
|
||||
score: float = Field(..., description="Relevance score")
|
||||
highlights: Optional[str] = Field(None, description="Highlighted matching text")
|
||||
collection: Optional[str] = Field(None, description="Collection name")
|
||||
document_type: Optional[str] = Field(None, description="Document type")
|
||||
content_preview: Optional[str] = Field(None, description="Content preview if requested")
|
||||
|
||||
|
||||
class DocumentSearchResponse(BaseModel):
|
||||
"""Response from document search."""
|
||||
query: str = Field(..., description="Original query")
|
||||
hits: List[DocumentSearchHit] = Field(..., description="Search results")
|
||||
total: int = Field(..., description="Total matching documents")
|
||||
duration_ms: int = Field(..., description="Search duration in milliseconds")
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Health Check Models
|
||||
# =============================================================================
|
||||
|
||||
|
||||
class DocumentStoreHealth(BaseModel):
|
||||
"""Health status of document storage components."""
|
||||
paperless_healthy: bool = Field(..., description="Paperless-ngx responding")
|
||||
paperless_version: Optional[str] = Field(None, description="Paperless version")
|
||||
total_documents: Optional[int] = Field(None, description="Total documents in Paperless")
|
||||
indexed_documents: Optional[int] = Field(None, description="Documents indexed in Library Desk")
|
||||
@@ -14,13 +14,16 @@ class HybridRAGConfig(BaseModel):
|
||||
vector_limit: int = Field(default=10, ge=1, le=50, description="Max vector results")
|
||||
graph_limit: int = Field(default=10, ge=1, le=50, description="Max graph results")
|
||||
web_limit: int = Field(default=5, ge=1, le=20, description="Max web results")
|
||||
volatile_limit: int = Field(default=1, ge=1, le=5, description="Max volatile results (typically 1)")
|
||||
enable_vector: bool = Field(default=True, description="Enable vector search")
|
||||
enable_graph: bool = Field(default=True, description="Enable graph search")
|
||||
enable_web: bool = Field(default=True, description="Enable web search")
|
||||
enable_volatile: bool = Field(default=True, description="Enable volatile cache search")
|
||||
enable_reranking: bool = Field(default=True, description="Enable LLM re-ranking")
|
||||
enable_enrichment: bool = Field(default=True, description="Enable graph enrichment")
|
||||
final_result_count: int = Field(default=10, ge=1, le=50, description="Final results to return")
|
||||
rrf_k: int = Field(default=60, ge=1, le=100, description="RRF constant")
|
||||
volatile_threshold: float = Field(default=0.8, ge=0.5, le=1.0, description="Volatile similarity threshold")
|
||||
|
||||
|
||||
class RelatedDossier(BaseModel):
|
||||
@@ -34,7 +37,7 @@ class RelatedDossier(BaseModel):
|
||||
|
||||
class HybridRAGResult(BaseModel):
|
||||
"""Single result from HybridRAG query."""
|
||||
source_type: str = Field(..., description="Source: 'vector', 'graph', 'web'")
|
||||
source_type: str = Field(..., description="Source: 'wiki', 'web', 'volatile'")
|
||||
title: str
|
||||
content: str
|
||||
url: Optional[str] = Field(None, description="URL for web results")
|
||||
@@ -53,6 +56,7 @@ class TimingBreakdown(BaseModel):
|
||||
vector_ms: float = Field(..., description="Phase 1: Vector search")
|
||||
graph_ms: float = Field(..., description="Phase 1: Graph search")
|
||||
web_ms: float = Field(..., description="Phase 1: Web search")
|
||||
volatile_ms: float = Field(default=0, description="Phase 1: Volatile cache search")
|
||||
fusion_ms: float = Field(..., description="Phase 2: RRF fusion")
|
||||
enrichment_ms: float = Field(..., description="Phase 3: Graph enrichment")
|
||||
reranking_ms: float = Field(..., description="Phase 4: LLM re-ranking")
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
"""
|
||||
RAG search models for Library Desk.
|
||||
|
||||
Pydantic models for web/news/image search requests and responses.
|
||||
"""
|
||||
|
||||
from enum import Enum
|
||||
from typing import Optional, List
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from src.core.multi_tenancy import DEFAULT_USER
|
||||
|
||||
|
||||
class SearchType(str, Enum):
|
||||
"""Supported search types."""
|
||||
WEB = "web"
|
||||
NEWS = "news"
|
||||
IMAGES = "images"
|
||||
|
||||
|
||||
class RAGSearchRequest(BaseModel):
|
||||
"""Request for RAG search endpoint."""
|
||||
|
||||
query: str = Field(
|
||||
...,
|
||||
min_length=1,
|
||||
max_length=500,
|
||||
description="The search query"
|
||||
)
|
||||
search_type: SearchType = Field(
|
||||
default=SearchType.WEB,
|
||||
description="Type of search: web, news, or images"
|
||||
)
|
||||
limit: int = Field(
|
||||
default=10,
|
||||
ge=1,
|
||||
le=20,
|
||||
description="Maximum number of results (1-20)"
|
||||
)
|
||||
user: str = Field(
|
||||
default=DEFAULT_USER,
|
||||
description="User identifier for rate limiting/personalization"
|
||||
)
|
||||
|
||||
|
||||
class RAGSearchResult(BaseModel):
|
||||
"""A single search result with extracted content."""
|
||||
|
||||
title: str = Field(..., description="Title of the result")
|
||||
url: str = Field(..., description="URL of the source")
|
||||
content: str = Field(
|
||||
"",
|
||||
description="Full extracted text via Trafilatura (max ~2000 chars)"
|
||||
)
|
||||
snippet: str = Field(
|
||||
"",
|
||||
description="Original search engine snippet (150-300 chars)"
|
||||
)
|
||||
source: str = Field(..., description="Domain name of the source")
|
||||
published_date: Optional[str] = Field(
|
||||
None,
|
||||
description="Publication date in ISO format if available"
|
||||
)
|
||||
|
||||
|
||||
class RAGSearchResponse(BaseModel):
|
||||
"""Response from RAG search endpoint."""
|
||||
|
||||
query: str = Field(..., description="Echo of the original query")
|
||||
search_type: SearchType = Field(..., description="Type of search performed")
|
||||
results: List[RAGSearchResult] = Field(
|
||||
default_factory=list,
|
||||
description="List of search results with extracted content"
|
||||
)
|
||||
total_results: int = Field(
|
||||
...,
|
||||
ge=0,
|
||||
description="Number of results returned"
|
||||
)
|
||||
search_time_ms: int = Field(
|
||||
...,
|
||||
ge=0,
|
||||
description="Total time for search and content extraction"
|
||||
)
|
||||
sources_summary: str = Field(
|
||||
"",
|
||||
description="Markdown-formatted list of all source URLs"
|
||||
)
|
||||
@@ -0,0 +1,130 @@
|
||||
"""
|
||||
Volatile memory models for Library Desk.
|
||||
|
||||
Provides models for ephemeral cached data with TTL - weather, news, financial data,
|
||||
transit schedules, and other time-sensitive external information.
|
||||
"""
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import Dict, Any, Optional, List
|
||||
from datetime import datetime
|
||||
from enum import Enum
|
||||
|
||||
|
||||
class VolatileNamespace(str, Enum):
|
||||
"""
|
||||
Predefined namespaces for volatile data.
|
||||
|
||||
Each namespace can have different default TTLs and refresh schedules.
|
||||
"""
|
||||
# Real-time external data
|
||||
WEATHER = "weather" # Current conditions, forecasts
|
||||
NEWS = "news" # Headlines, breaking news
|
||||
FINANCIAL = "financial" # Stock prices, exchange rates, crypto
|
||||
TRANSIT = "transit" # Train/bus schedules, delays, disruptions
|
||||
TRAFFIC = "traffic" # Commute times, road conditions
|
||||
AIR_QUALITY = "air_quality" # Pollution levels, pollen counts
|
||||
SPORTS = "sports" # Live scores, upcoming matches
|
||||
|
||||
# System/integration data
|
||||
SOCIAL = "social" # Social media mentions, notifications
|
||||
SYSTEM = "system" # Service health, infrastructure status
|
||||
|
||||
# Ephemeral context
|
||||
CONTEXT = "context" # Conversation context, session state
|
||||
CUSTOM = "custom" # User-defined volatile data
|
||||
|
||||
|
||||
# Default TTLs per namespace (in seconds)
|
||||
NAMESPACE_DEFAULT_TTL: Dict[str, int] = {
|
||||
VolatileNamespace.WEATHER: 1800, # 30 min - weather changes slowly
|
||||
VolatileNamespace.NEWS: 3600, # 1 hour - news cycles
|
||||
VolatileNamespace.FINANCIAL: 300, # 5 min - markets move fast
|
||||
VolatileNamespace.TRANSIT: 300, # 5 min - schedules update frequently
|
||||
VolatileNamespace.TRAFFIC: 600, # 10 min - traffic patterns
|
||||
VolatileNamespace.AIR_QUALITY: 3600, # 1 hour - air quality stable
|
||||
VolatileNamespace.SPORTS: 60, # 1 min - live scores
|
||||
VolatileNamespace.SOCIAL: 600, # 10 min - social notifications
|
||||
VolatileNamespace.SYSTEM: 60, # 1 min - system health
|
||||
VolatileNamespace.CONTEXT: 3600, # 1 hour - session context
|
||||
VolatileNamespace.CUSTOM: 3600, # 1 hour - default for custom
|
||||
}
|
||||
|
||||
|
||||
class VolatileRecord(BaseModel):
|
||||
"""
|
||||
A volatile cache record with TTL.
|
||||
|
||||
Volatile records are ephemeral data stored in Redis with automatic expiration.
|
||||
Used for weather, news, financial data, and other time-sensitive information.
|
||||
"""
|
||||
key: str = Field(..., description="Record key (e.g., 'rotterdam', 'nos-headlines')")
|
||||
namespace: str = Field(..., description="Namespace (e.g., 'weather', 'news', 'financial')")
|
||||
data: Dict[str, Any] = Field(..., description="Actual content/payload")
|
||||
source: Optional[str] = Field(None, description="Origin API/service (e.g., 'openweathermap', 'nos.nl')")
|
||||
created_at: datetime = Field(default_factory=datetime.utcnow, description="When record was created")
|
||||
updated_at: datetime = Field(default_factory=datetime.utcnow, description="When record was last updated")
|
||||
ttl: int = Field(..., ge=60, le=604800, description="Time-to-live in seconds (max 7 days)")
|
||||
refresh_schedule: Optional[str] = Field(None, description="Cron expression for scheduled refresh")
|
||||
user: str = Field(..., description="User identifier for multi-tenancy")
|
||||
|
||||
|
||||
class VolatileRecordCreate(BaseModel):
|
||||
"""Request model for creating/updating a volatile record."""
|
||||
data: Dict[str, Any] = Field(..., description="Content to store")
|
||||
source: Optional[str] = Field(None, description="Origin API/service")
|
||||
ttl: Optional[int] = Field(None, ge=60, le=604800, description="TTL in seconds (uses namespace default if not set)")
|
||||
refresh_schedule: Optional[str] = Field(None, description="Cron expression for scheduled refresh")
|
||||
|
||||
|
||||
class VolatileRecordResponse(BaseModel):
|
||||
"""Response model for a volatile record."""
|
||||
key: str = Field(..., description="Record key")
|
||||
namespace: str = Field(..., description="Namespace")
|
||||
data: Dict[str, Any] = Field(..., description="Stored content")
|
||||
source: Optional[str] = Field(None, description="Origin API/service")
|
||||
created_at: datetime = Field(..., description="Creation timestamp")
|
||||
updated_at: datetime = Field(..., description="Last update timestamp")
|
||||
ttl: int = Field(..., description="TTL in seconds")
|
||||
ttl_remaining: int = Field(..., description="Seconds until expiration")
|
||||
refresh_schedule: Optional[str] = Field(None, description="Cron expression if scheduled")
|
||||
user: str = Field(..., description="User identifier")
|
||||
|
||||
|
||||
class VolatileListResponse(BaseModel):
|
||||
"""Response model for listing volatile records."""
|
||||
namespace: str = Field(..., description="Namespace queried")
|
||||
keys: List[str] = Field(..., description="List of keys in namespace")
|
||||
count: int = Field(..., description="Number of keys")
|
||||
user: str = Field(..., description="User identifier")
|
||||
|
||||
|
||||
class VolatileScheduledResponse(BaseModel):
|
||||
"""Response model for records needing refresh."""
|
||||
records: List[VolatileRecordResponse] = Field(..., description="Records with refresh schedules")
|
||||
count: int = Field(..., description="Number of scheduled records")
|
||||
user: str = Field(..., description="User identifier")
|
||||
|
||||
|
||||
class VolatileStatsResponse(BaseModel):
|
||||
"""Response model for volatile cache statistics."""
|
||||
total_records: int = Field(..., description="Total volatile records for user")
|
||||
by_namespace: Dict[str, int] = Field(..., description="Record count per namespace")
|
||||
scheduled_count: int = Field(..., description="Records with refresh schedules")
|
||||
total_memory_bytes: Optional[int] = Field(None, description="Approximate memory usage")
|
||||
user: str = Field(..., description="User identifier")
|
||||
|
||||
|
||||
class VolatileDeleteResponse(BaseModel):
|
||||
"""Response model for delete operation."""
|
||||
key: str = Field(..., description="Deleted key")
|
||||
namespace: str = Field(..., description="Namespace")
|
||||
deleted: bool = Field(..., description="Whether record was found and deleted")
|
||||
user: str = Field(..., description="User identifier")
|
||||
|
||||
|
||||
class VolatileBulkDeleteResponse(BaseModel):
|
||||
"""Response model for bulk delete operations."""
|
||||
namespace: Optional[str] = Field(None, description="Namespace if namespace-wide delete")
|
||||
deleted_count: int = Field(..., description="Number of records deleted")
|
||||
user: str = Field(..., description="User identifier")
|
||||
+42
-1
@@ -8,7 +8,7 @@ Models for:
|
||||
"""
|
||||
|
||||
from pydantic import BaseModel, Field, field_validator
|
||||
from typing import Optional, List
|
||||
from typing import Optional, List, Dict, Any
|
||||
from datetime import datetime
|
||||
|
||||
|
||||
@@ -190,3 +190,44 @@ class DossierOperationResponse(BaseModel):
|
||||
dossier_name: str = Field(..., description="Dossier name")
|
||||
index_page_id: Optional[int] = Field(None, description="Index page ID (if created)")
|
||||
index_page_path: Optional[str] = Field(None, description="Index page path (if created)")
|
||||
|
||||
|
||||
# Smart create models (HybridRAG-powered page creation)
|
||||
class WikiSmartCreateRequest(BaseModel):
|
||||
"""Request model for smart page creation with research."""
|
||||
topic: str = Field(..., min_length=1, max_length=500, description="Topic to research and create page about")
|
||||
path: Optional[str] = Field(None, description="Page path (auto-generated from topic if not provided)")
|
||||
tags: List[str] = Field(default_factory=list, description="Tags for the page")
|
||||
user: Optional[str] = Field(None, description="User identifier")
|
||||
include_web_research: bool = Field(default=True, description="Include web search results")
|
||||
include_wiki_search: bool = Field(default=True, description="Include existing wiki knowledge")
|
||||
|
||||
@field_validator("tags")
|
||||
@classmethod
|
||||
def validate_tags(cls, v: List[str]) -> List[str]:
|
||||
"""Validate and clean tags."""
|
||||
cleaned = [tag.strip() for tag in v if tag.strip()]
|
||||
return list(set(cleaned))
|
||||
|
||||
@field_validator("path")
|
||||
@classmethod
|
||||
def validate_path(cls, v: Optional[str]) -> Optional[str]:
|
||||
"""Validate page path if provided."""
|
||||
if v is None:
|
||||
return None
|
||||
# Ensure path starts with /
|
||||
if not v.startswith("/"):
|
||||
v = f"/{v}"
|
||||
# Remove trailing slash
|
||||
if v.endswith("/") and v != "/":
|
||||
v = v.rstrip("/")
|
||||
return v
|
||||
|
||||
|
||||
class WikiSmartCreateResponse(BaseModel):
|
||||
"""Response model for smart page creation."""
|
||||
page: WikiPage = Field(..., description="Created wiki page")
|
||||
research_summary: Dict[str, Any] = Field(..., description="Summary of research used")
|
||||
sources_used: int = Field(..., description="Number of sources incorporated")
|
||||
search_id: Optional[str] = Field(None, description="HybridRAG search ID for reference")
|
||||
entity_linking: Dict[str, int] = Field(default_factory=dict, description="Entity linking statistics")
|
||||
|
||||
@@ -0,0 +1,132 @@
|
||||
"""
|
||||
Content extraction router for Library Desk API.
|
||||
|
||||
Endpoints for extracting main content from web URLs using Trafilatura.
|
||||
"""
|
||||
|
||||
import time
|
||||
from fastapi import APIRouter, HTTPException, Depends
|
||||
import logging
|
||||
|
||||
from src.models.content import (
|
||||
ContentExtractionRequest,
|
||||
ContentExtractionResponse,
|
||||
BatchContentExtractionRequest,
|
||||
BatchContentExtractionResponse,
|
||||
)
|
||||
from src.clients.content_extractor import ContentExtractor
|
||||
from src.core.dependencies import verify_api_key
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/content", tags=["Content Extraction"])
|
||||
|
||||
|
||||
# Lazy import to avoid circular dependency
|
||||
def get_content_extractor() -> ContentExtractor:
|
||||
"""Get content extractor instance."""
|
||||
from src.core.dependencies import get_content_extractor as _get_extractor
|
||||
return _get_extractor()
|
||||
|
||||
|
||||
@router.post("/extract", response_model=ContentExtractionResponse)
|
||||
async def extract_content(
|
||||
request: ContentExtractionRequest,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Extract main content from a single URL.
|
||||
|
||||
Uses Trafilatura to fetch the URL and extract the main text content,
|
||||
removing navigation, ads, and other boilerplate.
|
||||
|
||||
**Example Request:**
|
||||
```json
|
||||
{
|
||||
"url": "https://example.com/article",
|
||||
"include_metadata": true,
|
||||
"max_length": 2000
|
||||
}
|
||||
```
|
||||
|
||||
**Returns:** Extracted content with optional metadata (title, author, date)
|
||||
"""
|
||||
start_time = time.time()
|
||||
|
||||
try:
|
||||
extractor = get_content_extractor()
|
||||
result = await extractor.extract(
|
||||
url=request.url,
|
||||
include_metadata=request.include_metadata,
|
||||
max_length=request.max_length
|
||||
)
|
||||
|
||||
extraction_time_ms = int((time.time() - start_time) * 1000)
|
||||
|
||||
return ContentExtractionResponse(
|
||||
result=result,
|
||||
extraction_time_ms=extraction_time_ms
|
||||
)
|
||||
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
except Exception as e:
|
||||
logger.error(f"Content extraction failed: {e}", exc_info=True)
|
||||
raise HTTPException(status_code=500, detail="Content extraction failed")
|
||||
|
||||
|
||||
@router.post("/extract/batch", response_model=BatchContentExtractionResponse)
|
||||
async def extract_content_batch(
|
||||
request: BatchContentExtractionRequest,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Extract content from multiple URLs in parallel.
|
||||
|
||||
Processes up to 20 URLs concurrently with per-URL timeouts.
|
||||
Failed extractions are included in results with success=false.
|
||||
|
||||
**Example Request:**
|
||||
```json
|
||||
{
|
||||
"urls": [
|
||||
"https://example.com/article1",
|
||||
"https://example.com/article2"
|
||||
],
|
||||
"include_metadata": true,
|
||||
"max_length": 2000
|
||||
}
|
||||
```
|
||||
|
||||
**Returns:** List of extraction results with success/failure counts
|
||||
"""
|
||||
start_time = time.time()
|
||||
|
||||
if not request.urls:
|
||||
raise HTTPException(status_code=400, detail="URLs list cannot be empty")
|
||||
|
||||
try:
|
||||
extractor = get_content_extractor()
|
||||
results = await extractor.extract_batch(
|
||||
urls=request.urls,
|
||||
include_metadata=request.include_metadata,
|
||||
max_length=request.max_length
|
||||
)
|
||||
|
||||
extraction_time_ms = int((time.time() - start_time) * 1000)
|
||||
successful = sum(1 for r in results if r.success)
|
||||
failed = len(results) - successful
|
||||
|
||||
return BatchContentExtractionResponse(
|
||||
results=results,
|
||||
total_urls=len(request.urls),
|
||||
successful=successful,
|
||||
failed=failed,
|
||||
extraction_time_ms=extraction_time_ms
|
||||
)
|
||||
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
except Exception as e:
|
||||
logger.error(f"Batch content extraction failed: {e}", exc_info=True)
|
||||
raise HTTPException(status_code=500, detail="Batch extraction failed")
|
||||
@@ -0,0 +1,424 @@
|
||||
"""
|
||||
Document storage router for Library Desk API.
|
||||
|
||||
Event-driven integration with Paperless-ngx:
|
||||
- Webhook receiver triggers indexing after Paperless virus scan passes
|
||||
- Upload endpoint sends files to Paperless for processing
|
||||
- Search across indexed documents
|
||||
"""
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Depends, Query, UploadFile, File, Request
|
||||
from typing import Optional
|
||||
import logging
|
||||
import time
|
||||
|
||||
from src.models.document import (
|
||||
PaperlessWebhookPayload,
|
||||
WebhookResponse,
|
||||
DocumentUploadRequest,
|
||||
DocumentUploadResponse,
|
||||
DocumentSearchRequest,
|
||||
DocumentSearchResponse,
|
||||
DocumentStoreHealth,
|
||||
)
|
||||
from src.core.dependencies import (
|
||||
verify_api_key,
|
||||
PaperlessDep,
|
||||
QdrantDep,
|
||||
OllamaDep,
|
||||
Neo4jDep,
|
||||
WikiJSDep,
|
||||
)
|
||||
from src.core.multi_tenancy import DEFAULT_USER
|
||||
from src.config import get_settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/documents", tags=["Documents"])
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Webhook Endpoint (primary integration - event-driven)
|
||||
# =============================================================================
|
||||
|
||||
|
||||
@router.post("/webhook", response_model=WebhookResponse)
|
||||
async def receive_webhook(
|
||||
payload: PaperlessWebhookPayload,
|
||||
paperless: PaperlessDep,
|
||||
qdrant: QdrantDep,
|
||||
ollama: OllamaDep,
|
||||
neo4j: Neo4jDep,
|
||||
wiki: WikiJSDep,
|
||||
user: str = Query(default=DEFAULT_USER, description="User identifier"),
|
||||
):
|
||||
"""
|
||||
Receive webhook events from Paperless-ngx.
|
||||
|
||||
This is the primary integration point. Configure Paperless workflow:
|
||||
1. Trigger: Document Added (after consumption completes)
|
||||
2. Condition: Document passed virus scan (ClamAV in Paperless)
|
||||
3. Action: Webhook POST to this endpoint
|
||||
|
||||
Library Desk indexes the document into vectors and graph.
|
||||
"""
|
||||
from src.services.document_sync_service import DocumentSyncService
|
||||
|
||||
doc_id = payload.document_id
|
||||
logger.info(f"Webhook received: document_id={doc_id}, title={payload.title}")
|
||||
|
||||
settings = get_settings()
|
||||
if not settings.document_store_enabled:
|
||||
return WebhookResponse(
|
||||
document_id=doc_id,
|
||||
status="skipped",
|
||||
indexed=False,
|
||||
message="Document store is disabled"
|
||||
)
|
||||
|
||||
try:
|
||||
sync_service = DocumentSyncService(
|
||||
paperless_client=paperless,
|
||||
qdrant_client=qdrant,
|
||||
ollama_client=ollama,
|
||||
neo4j_client=neo4j,
|
||||
wiki_client=wiki,
|
||||
settings=settings
|
||||
)
|
||||
|
||||
# Fetch content from Paperless (template only provides doc_url and title)
|
||||
result = await sync_service.index_document(
|
||||
document_id=doc_id,
|
||||
user=user,
|
||||
)
|
||||
|
||||
return WebhookResponse(
|
||||
document_id=doc_id,
|
||||
status="indexed" if result.success else "failed",
|
||||
indexed=result.success,
|
||||
message=result.error if not result.success else f"Indexed: {result.title}"
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Webhook processing failed for document {doc_id}: {e}", exc_info=True)
|
||||
return WebhookResponse(
|
||||
document_id=doc_id,
|
||||
status="error",
|
||||
indexed=False,
|
||||
message=str(e)
|
||||
)
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Debug Capture Endpoint
|
||||
# =============================================================================
|
||||
|
||||
|
||||
@router.post("/webhook-capture")
|
||||
async def capture_webhook(request: Request):
|
||||
"""Capture raw webhook payload for debugging."""
|
||||
import json
|
||||
from pathlib import Path
|
||||
from datetime import datetime
|
||||
|
||||
# Get raw body
|
||||
body = await request.body()
|
||||
headers = dict(request.headers)
|
||||
query_params = dict(request.query_params)
|
||||
|
||||
# Build capture data
|
||||
capture = {
|
||||
"timestamp": datetime.now().isoformat(),
|
||||
"method": request.method,
|
||||
"url": str(request.url),
|
||||
"query_params": query_params,
|
||||
"headers": headers,
|
||||
"content_type": headers.get("content-type", "unknown"),
|
||||
"body_raw": body.decode("utf-8", errors="replace"),
|
||||
}
|
||||
|
||||
# Try to parse as JSON
|
||||
try:
|
||||
capture["body_json"] = json.loads(body)
|
||||
except:
|
||||
capture["body_json"] = None
|
||||
|
||||
# Write to file
|
||||
capture_file = Path("logs/webhook_capture.json")
|
||||
capture_file.parent.mkdir(exist_ok=True)
|
||||
with open(capture_file, "w") as f:
|
||||
json.dump(capture, f, indent=2, default=str)
|
||||
|
||||
logger.info(f"Captured webhook: {capture['body_raw'][:200]}")
|
||||
|
||||
return {"status": "captured", "file": str(capture_file)}
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Simple Webhook (URL parameters only)
|
||||
# =============================================================================
|
||||
|
||||
|
||||
@router.post("/webhook-simple", response_model=WebhookResponse)
|
||||
async def receive_webhook_simple(
|
||||
doc_url: str = Query(..., description="Paperless document URL containing ID"),
|
||||
title: str = Query(default="", description="Document title"),
|
||||
user: str = Query(default=DEFAULT_USER, description="User identifier"),
|
||||
paperless: PaperlessDep = None,
|
||||
qdrant: QdrantDep = None,
|
||||
ollama: OllamaDep = None,
|
||||
neo4j: Neo4jDep = None,
|
||||
wiki: WikiJSDep = None,
|
||||
):
|
||||
"""
|
||||
Simple webhook endpoint accepting URL parameters.
|
||||
|
||||
Used when Paperless Jinja templates don't work with JSON body.
|
||||
URL format: /webhook-simple?doc_url=http://...&title=...&user=...
|
||||
"""
|
||||
from src.services.document_sync_service import DocumentSyncService
|
||||
import re
|
||||
|
||||
# Extract document ID from URL
|
||||
match = re.search(r'/documents/(\d+)/?', doc_url)
|
||||
if not match:
|
||||
return WebhookResponse(
|
||||
document_id=0,
|
||||
status="error",
|
||||
indexed=False,
|
||||
message=f"Cannot extract document ID from URL: {doc_url}"
|
||||
)
|
||||
doc_id = int(match.group(1))
|
||||
|
||||
logger.info(f"Webhook-simple received: document_id={doc_id}, title={title}")
|
||||
|
||||
settings = get_settings()
|
||||
if not settings.document_store_enabled:
|
||||
return WebhookResponse(
|
||||
document_id=doc_id,
|
||||
status="skipped",
|
||||
indexed=False,
|
||||
message="Document store is disabled"
|
||||
)
|
||||
|
||||
try:
|
||||
sync_service = DocumentSyncService(
|
||||
paperless_client=paperless,
|
||||
qdrant_client=qdrant,
|
||||
ollama_client=ollama,
|
||||
neo4j_client=neo4j,
|
||||
wiki_client=wiki,
|
||||
settings=settings
|
||||
)
|
||||
|
||||
result = await sync_service.index_document(
|
||||
document_id=doc_id,
|
||||
user=user,
|
||||
)
|
||||
|
||||
return WebhookResponse(
|
||||
document_id=doc_id,
|
||||
status="indexed" if result.success else "failed",
|
||||
indexed=result.success,
|
||||
message=result.error if not result.success else f"Indexed: {result.title}"
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Webhook-simple failed for document {doc_id}: {e}", exc_info=True)
|
||||
return WebhookResponse(
|
||||
document_id=doc_id,
|
||||
status="error",
|
||||
indexed=False,
|
||||
message=str(e)
|
||||
)
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Upload Endpoints
|
||||
# =============================================================================
|
||||
|
||||
|
||||
@router.post("/upload", response_model=DocumentUploadResponse)
|
||||
async def upload_document(
|
||||
file: UploadFile = File(...),
|
||||
title: Optional[str] = Query(None, description="Document title"),
|
||||
collection: Optional[str] = Query(None, description="Collection name"),
|
||||
paperless: PaperlessDep = None,
|
||||
api_key: str = Depends(verify_api_key),
|
||||
):
|
||||
"""
|
||||
Upload a document to Paperless-ngx.
|
||||
|
||||
Paperless handles virus scanning. If clean, Paperless webhook
|
||||
triggers indexing back to Library Desk.
|
||||
"""
|
||||
settings = get_settings()
|
||||
if not settings.document_store_enabled:
|
||||
raise HTTPException(status_code=503, detail="Document store is disabled")
|
||||
|
||||
content = await file.read()
|
||||
filename = file.filename or "document"
|
||||
|
||||
custom_fields = []
|
||||
if collection:
|
||||
custom_fields.append({"field": "collection", "value": collection})
|
||||
|
||||
try:
|
||||
task_id = await paperless.upload_document(
|
||||
file_content=content,
|
||||
filename=filename,
|
||||
title=title,
|
||||
custom_fields=custom_fields if custom_fields else None,
|
||||
)
|
||||
|
||||
return DocumentUploadResponse(
|
||||
task_id=task_id,
|
||||
filename=filename,
|
||||
message=f"Uploaded to Paperless, task {task_id}. Indexing via webhook after scan."
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Upload failed for '{filename}': {e}")
|
||||
raise HTTPException(status_code=500, detail=f"Upload failed: {e}")
|
||||
|
||||
|
||||
@router.post("/upload-url", response_model=DocumentUploadResponse)
|
||||
async def upload_from_url(
|
||||
request: DocumentUploadRequest,
|
||||
paperless: PaperlessDep = None,
|
||||
api_key: str = Depends(verify_api_key),
|
||||
):
|
||||
"""
|
||||
Download document from URL and upload to Paperless-ngx.
|
||||
|
||||
Used by HybridRAG to save discovered PDFs. Paperless scans and
|
||||
webhooks back for indexing.
|
||||
"""
|
||||
import httpx
|
||||
|
||||
settings = get_settings()
|
||||
if not settings.document_store_enabled:
|
||||
raise HTTPException(status_code=503, detail="Document store is disabled")
|
||||
|
||||
if not request.url:
|
||||
raise HTTPException(status_code=400, detail="URL is required")
|
||||
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=60.0) as client:
|
||||
response = await client.get(request.url, follow_redirects=True)
|
||||
response.raise_for_status()
|
||||
content = response.content
|
||||
filename = request.url.split("/")[-1].split("?")[0] or "document"
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Download failed from {request.url}: {e}")
|
||||
raise HTTPException(status_code=400, detail=f"Download failed: {e}")
|
||||
|
||||
try:
|
||||
custom_fields = [{"field": "source_url", "value": request.url}]
|
||||
if request.collection:
|
||||
custom_fields.append({"field": "collection", "value": request.collection})
|
||||
|
||||
task_id = await paperless.upload_document(
|
||||
file_content=content,
|
||||
filename=filename,
|
||||
title=request.title,
|
||||
custom_fields=custom_fields,
|
||||
)
|
||||
|
||||
return DocumentUploadResponse(
|
||||
task_id=task_id,
|
||||
filename=filename,
|
||||
message=f"Uploaded from URL, task {task_id}. Indexing via webhook after scan."
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Upload failed for URL '{request.url}': {e}")
|
||||
raise HTTPException(status_code=500, detail=f"Upload failed: {e}")
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Search
|
||||
# =============================================================================
|
||||
|
||||
|
||||
@router.post("/search", response_model=DocumentSearchResponse)
|
||||
async def search_documents(
|
||||
request: DocumentSearchRequest,
|
||||
qdrant: QdrantDep,
|
||||
ollama: OllamaDep,
|
||||
user: str = Query(default=DEFAULT_USER, description="User identifier"),
|
||||
api_key: str = Depends(verify_api_key),
|
||||
):
|
||||
"""
|
||||
Semantic search across indexed documents.
|
||||
"""
|
||||
from src.services.vector_service import VectorService
|
||||
from src.core.dependencies import get_wikijs_client
|
||||
from src.models.document import DocumentSearchHit
|
||||
|
||||
settings = get_settings()
|
||||
if not settings.document_store_enabled:
|
||||
raise HTTPException(status_code=503, detail="Document store is disabled")
|
||||
|
||||
start_time = time.time()
|
||||
|
||||
try:
|
||||
wiki = get_wikijs_client()
|
||||
vector_service = VectorService(qdrant, wiki, ollama)
|
||||
|
||||
results = await vector_service.search(
|
||||
query=request.query,
|
||||
user=user,
|
||||
limit=request.limit,
|
||||
score_threshold=0.5,
|
||||
doc_type="document"
|
||||
)
|
||||
|
||||
hits = []
|
||||
for result in results.get("results", []):
|
||||
hits.append(DocumentSearchHit(
|
||||
paperless_id=result.get("metadata", {}).get("paperless_id", 0),
|
||||
title=result.get("title", ""),
|
||||
score=result.get("score", 0.0),
|
||||
highlights=result.get("chunk_text", "")[:200] if request.include_content else None,
|
||||
collection=result.get("metadata", {}).get("collection"),
|
||||
document_type=result.get("metadata", {}).get("document_type"),
|
||||
content_preview=result.get("chunk_text", "")[:500] if request.include_content else None,
|
||||
))
|
||||
|
||||
return DocumentSearchResponse(
|
||||
query=request.query,
|
||||
hits=hits,
|
||||
total=len(hits),
|
||||
duration_ms=int((time.time() - start_time) * 1000)
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Document search failed: {e}", exc_info=True)
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Health
|
||||
# =============================================================================
|
||||
|
||||
|
||||
@router.get("/health", response_model=DocumentStoreHealth)
|
||||
async def document_store_health(paperless: PaperlessDep):
|
||||
"""Check Paperless-ngx connectivity."""
|
||||
settings = get_settings()
|
||||
|
||||
paperless_healthy = False
|
||||
if settings.paperless_token:
|
||||
try:
|
||||
paperless_healthy = await paperless.health_check()
|
||||
except Exception as e:
|
||||
logger.error(f"Paperless health check failed: {e}")
|
||||
|
||||
return DocumentStoreHealth(
|
||||
paperless_healthy=paperless_healthy,
|
||||
paperless_version="connected" if paperless_healthy else None,
|
||||
total_documents=None,
|
||||
indexed_documents=None
|
||||
)
|
||||
@@ -15,8 +15,6 @@ from src.models.graph import (
|
||||
MindMapResponse
|
||||
)
|
||||
from src.services.graph_service import GraphService
|
||||
from src.clients.neo4j_client import Neo4jClient
|
||||
from src.clients.wikijs_client import WikiJSClient
|
||||
from src.core.dependencies import Neo4jDep, WikiJSDep, verify_api_key
|
||||
from src.core.multi_tenancy import DEFAULT_USER
|
||||
|
||||
|
||||
+16
-13
@@ -1,7 +1,7 @@
|
||||
"""
|
||||
HybridRAG router for multi-source search API.
|
||||
|
||||
Provides endpoint for combining vector, graph, and web search
|
||||
Provides endpoint for combining vector, graph, volatile cache, and web search
|
||||
with RRF fusion and LLM re-ranking.
|
||||
"""
|
||||
|
||||
@@ -10,13 +10,9 @@ import logging
|
||||
|
||||
from src.models.hybrid_rag import HybridRAGRequest, HybridRAGResponse
|
||||
from src.services.hybrid_rag_service import HybridRAGService
|
||||
from src.services.vector_service import VectorService
|
||||
from src.services.graph_service import GraphService
|
||||
from src.clients.searxng_client import SearXNGClient
|
||||
from src.clients.ollama_client import OllamaClient
|
||||
from src.core.dependencies import (
|
||||
Neo4jDep, WikiJSDep, QdrantDep, OllamaDep,
|
||||
SearXNGDep, verify_api_key, get_settings
|
||||
SearXNGDep, ContentExtractorDep, verify_api_key, get_settings
|
||||
)
|
||||
from src.config import Settings
|
||||
|
||||
@@ -32,15 +28,18 @@ def get_hybrid_rag_service(
|
||||
qdrant_client: QdrantDep,
|
||||
ollama_client: OllamaDep,
|
||||
searxng_client: SearXNGDep,
|
||||
content_extractor: ContentExtractorDep,
|
||||
settings: Settings = Depends(get_settings)
|
||||
) -> HybridRAGService:
|
||||
"""Get HybridRAG service instance with all dependencies."""
|
||||
from src.services.vector_service import VectorService
|
||||
from src.services.graph_service import GraphService
|
||||
from src.services.volatile_service import VolatileCacheService
|
||||
|
||||
# Create component services
|
||||
vector_service = VectorService(qdrant_client, wiki_client, ollama_client)
|
||||
graph_service = GraphService(neo4j_client, wiki_client)
|
||||
volatile_service = VolatileCacheService(qdrant_client, ollama_client, settings)
|
||||
|
||||
# Create HybridRAG service
|
||||
return HybridRAGService(
|
||||
@@ -48,7 +47,9 @@ def get_hybrid_rag_service(
|
||||
graph_service=graph_service,
|
||||
searxng_client=searxng_client,
|
||||
ollama_client=ollama_client,
|
||||
settings=settings
|
||||
content_extractor=content_extractor,
|
||||
settings=settings,
|
||||
volatile_service=volatile_service
|
||||
)
|
||||
|
||||
|
||||
@@ -60,26 +61,28 @@ async def hybrid_search(
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Execute HybridRAG query combining vector, graph, and web search.
|
||||
Execute HybridRAG query combining vector, graph, volatile cache, and web search.
|
||||
|
||||
**6-Phase Pipeline:**
|
||||
1. **Query Enhancement**: Extract keywords/synonyms with LLM
|
||||
2. **Parallel Retrieval**: Search vector (Qdrant), graph (Neo4j), web (SearXNG)
|
||||
3. **RRF Fusion**: Merge results with Reciprocal Rank Fusion
|
||||
2. **Parallel Retrieval**: Search vector (Qdrant), graph (Neo4j), volatile cache, web (SearXNG)
|
||||
3. **RRF Fusion**: Merge results with Reciprocal Rank Fusion (volatile gets priority boost)
|
||||
4. **Enrichment**: Add related documents via shared entities
|
||||
5. **LLM Re-ranking**: Re-rank with mistral-nemo for relevance
|
||||
5. **LLM Re-ranking**: Re-rank with configured model for relevance
|
||||
6. **Context Formatting**: Format for LLM consumption
|
||||
7. **Persistence**: Store for Librarian knowledge consolidation
|
||||
|
||||
**Example Request:**
|
||||
```json
|
||||
{
|
||||
"query": "How does Docker orchestration work with Kubernetes?",
|
||||
"query": "What's the weather in Rotterdam?",
|
||||
"user": "jpmschweitzer",
|
||||
"config": {
|
||||
"vector_limit": 10,
|
||||
"graph_limit": 10,
|
||||
"web_limit": 5,
|
||||
"volatile_limit": 5,
|
||||
"enable_volatile": true,
|
||||
"enable_reranking": true,
|
||||
"final_result_count": 10
|
||||
}
|
||||
@@ -87,7 +90,7 @@ async def hybrid_search(
|
||||
```
|
||||
|
||||
**Returns:**
|
||||
- Ranked results from all sources
|
||||
- Ranked results from all sources (wiki, volatile, web)
|
||||
- Extracted keywords/synonyms
|
||||
- Related dossiers (via graph)
|
||||
- Formatted context for LLM
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,87 @@
|
||||
"""
|
||||
RAG search router for Library Desk API.
|
||||
|
||||
Endpoints for web, news, and image search with content extraction.
|
||||
"""
|
||||
|
||||
import httpx
|
||||
from fastapi import APIRouter, HTTPException, Depends
|
||||
import logging
|
||||
|
||||
from src.models.rag_search import RAGSearchRequest, RAGSearchResponse
|
||||
from src.services.rag_search_service import RAGSearchService
|
||||
from src.core.dependencies import verify_api_key
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/rag", tags=["RAG Search"])
|
||||
|
||||
|
||||
# Lazy import to avoid circular dependency
|
||||
def get_rag_search_service() -> RAGSearchService:
|
||||
"""Get RAG search service instance."""
|
||||
from src.core.dependencies import get_rag_search_service as _get_service
|
||||
return _get_service()
|
||||
|
||||
|
||||
@router.post("/search", response_model=RAGSearchResponse)
|
||||
async def search(
|
||||
request: RAGSearchRequest,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Execute RAG-optimized web search with content extraction.
|
||||
|
||||
Searches via SearXNG and extracts full content from results using
|
||||
Trafilatura. Results are cached in Redis for efficiency.
|
||||
|
||||
**Search Types:**
|
||||
- `web`: General web search (default)
|
||||
- `news`: News articles with recency filtering
|
||||
- `images`: Image search results
|
||||
|
||||
**Example Request:**
|
||||
```json
|
||||
{
|
||||
"query": "Python async programming best practices",
|
||||
"search_type": "web",
|
||||
"limit": 10,
|
||||
"user": "default"
|
||||
}
|
||||
```
|
||||
|
||||
**Response includes:**
|
||||
- Full extracted text content per result
|
||||
- Original search snippets
|
||||
- Source domain names
|
||||
- Markdown sources summary for LLM consumption
|
||||
|
||||
**Error Codes:**
|
||||
- 400: Invalid query (empty or too long)
|
||||
- 502: Search provider (SearXNG) error
|
||||
- 504: Search timeout
|
||||
"""
|
||||
try:
|
||||
service = get_rag_search_service()
|
||||
response = await service.search(
|
||||
query=request.query,
|
||||
search_type=request.search_type,
|
||||
limit=request.limit,
|
||||
user=request.user
|
||||
)
|
||||
return response
|
||||
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
|
||||
except httpx.TimeoutException:
|
||||
logger.error(f"Search timed out for query: {request.query}")
|
||||
raise HTTPException(status_code=504, detail="Search timed out")
|
||||
|
||||
except httpx.HTTPError as e:
|
||||
logger.error(f"Search provider error: {e}")
|
||||
raise HTTPException(status_code=502, detail="Search provider error")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"RAG search failed: {e}", exc_info=True)
|
||||
raise HTTPException(status_code=500, detail="Search failed")
|
||||
@@ -0,0 +1,273 @@
|
||||
"""
|
||||
Volatile cache router for Library Desk API.
|
||||
|
||||
Endpoints for ephemeral cached data with TTL - weather, news, financial, etc.
|
||||
Data is stored as vectors in Qdrant for semantic search retrieval.
|
||||
"""
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Depends, Query
|
||||
import logging
|
||||
|
||||
from src.models.volatile import (
|
||||
VolatileRecordCreate,
|
||||
VolatileRecordResponse,
|
||||
VolatileListResponse,
|
||||
VolatileScheduledResponse,
|
||||
VolatileStatsResponse,
|
||||
VolatileDeleteResponse,
|
||||
VolatileNamespace,
|
||||
NAMESPACE_DEFAULT_TTL,
|
||||
)
|
||||
from src.services.volatile_service import VolatileCacheService
|
||||
from src.core.dependencies import verify_api_key, QdrantDep, OllamaDep
|
||||
from src.core.multi_tenancy import DEFAULT_USER
|
||||
from src.config import get_settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/volatile", tags=["Volatile Cache"])
|
||||
|
||||
|
||||
def get_volatile_service(qdrant: QdrantDep, ollama: OllamaDep) -> VolatileCacheService:
|
||||
"""Get volatile cache service instance."""
|
||||
settings = get_settings()
|
||||
return VolatileCacheService(
|
||||
qdrant_client=qdrant,
|
||||
ollama_client=ollama,
|
||||
settings=settings
|
||||
)
|
||||
|
||||
|
||||
@router.get("/stats", response_model=VolatileStatsResponse)
|
||||
async def get_stats(
|
||||
user: str = Query(default=DEFAULT_USER, description="User identifier"),
|
||||
qdrant: QdrantDep = None,
|
||||
ollama: OllamaDep = None,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Get volatile cache statistics.
|
||||
|
||||
Returns counts of records by namespace and scheduled refresh info.
|
||||
"""
|
||||
service = get_volatile_service(qdrant, ollama)
|
||||
stats = await service.get_stats(user)
|
||||
|
||||
return VolatileStatsResponse(
|
||||
total_records=stats["total_records"],
|
||||
by_namespace=stats["by_namespace"],
|
||||
scheduled_count=stats["scheduled_count"],
|
||||
total_memory_bytes=None,
|
||||
user=user,
|
||||
)
|
||||
|
||||
|
||||
@router.get("/scheduled", response_model=VolatileScheduledResponse)
|
||||
async def get_scheduled(
|
||||
user: str = Query(default=DEFAULT_USER, description="User identifier"),
|
||||
qdrant: QdrantDep = None,
|
||||
ollama: OllamaDep = None,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Get records with refresh schedules.
|
||||
|
||||
Used by scheduler to determine what volatile data needs refreshing.
|
||||
Returns all records that have a refresh_schedule cron expression set.
|
||||
"""
|
||||
service = get_volatile_service(qdrant, ollama)
|
||||
records = await service.get_scheduled(user)
|
||||
|
||||
return VolatileScheduledResponse(
|
||||
records=records,
|
||||
count=len(records),
|
||||
user=user,
|
||||
)
|
||||
|
||||
|
||||
@router.get("/namespaces")
|
||||
async def list_namespaces(
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
List available namespaces and their default TTLs.
|
||||
|
||||
Returns predefined namespaces with their default TTL values.
|
||||
"""
|
||||
return {
|
||||
"namespaces": [
|
||||
{
|
||||
"name": ns.value,
|
||||
"default_ttl": NAMESPACE_DEFAULT_TTL.get(ns, 3600),
|
||||
"description": _get_namespace_description(ns),
|
||||
}
|
||||
for ns in VolatileNamespace
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
def _get_namespace_description(ns: VolatileNamespace) -> str:
|
||||
"""Get human-readable description for namespace."""
|
||||
descriptions = {
|
||||
VolatileNamespace.WEATHER: "Weather conditions and forecasts",
|
||||
VolatileNamespace.NEWS: "Headlines and breaking news",
|
||||
VolatileNamespace.FINANCIAL: "Stock prices, exchange rates, crypto",
|
||||
VolatileNamespace.TRANSIT: "Train/bus schedules, delays",
|
||||
VolatileNamespace.TRAFFIC: "Commute times, road conditions",
|
||||
VolatileNamespace.AIR_QUALITY: "Pollution levels, pollen counts",
|
||||
VolatileNamespace.SPORTS: "Live scores, upcoming matches",
|
||||
VolatileNamespace.SOCIAL: "Social media mentions, notifications",
|
||||
VolatileNamespace.SYSTEM: "Service health, infrastructure status",
|
||||
VolatileNamespace.CONTEXT: "Conversation context, session state",
|
||||
VolatileNamespace.CUSTOM: "User-defined volatile data",
|
||||
}
|
||||
return descriptions.get(ns, "Custom namespace")
|
||||
|
||||
|
||||
@router.get("/search")
|
||||
async def search_volatile(
|
||||
q: str = Query(..., min_length=1, description="Search query"),
|
||||
user: str = Query(default=DEFAULT_USER, description="User identifier"),
|
||||
limit: int = Query(default=5, ge=1, le=20, description="Maximum results"),
|
||||
threshold: float = Query(default=0.75, ge=0.5, le=1.0, description="Minimum similarity score"),
|
||||
qdrant: QdrantDep = None,
|
||||
ollama: OllamaDep = None,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Semantic search across volatile data.
|
||||
|
||||
Searches all volatile data for semantically similar content.
|
||||
Higher threshold = stricter matching.
|
||||
|
||||
**Example:**
|
||||
```
|
||||
GET /volatile/search?q=weather%20rotterdam&user=jpmschweitzer
|
||||
```
|
||||
"""
|
||||
service = get_volatile_service(qdrant, ollama)
|
||||
results = await service.search(user, q, limit=limit, score_threshold=threshold)
|
||||
|
||||
return {
|
||||
"query": q,
|
||||
"results": results,
|
||||
"count": len(results),
|
||||
"user": user,
|
||||
}
|
||||
|
||||
|
||||
@router.post("/store", response_model=VolatileRecordResponse)
|
||||
async def store_volatile(
|
||||
namespace: str = Query(..., description="Data namespace (weather, news, etc.)"),
|
||||
key: str = Query(..., description="Record key (e.g., 'rotterdam', 'nos-headlines')"),
|
||||
request: VolatileRecordCreate = None,
|
||||
user: str = Query(default=DEFAULT_USER, description="User identifier"),
|
||||
qdrant: QdrantDep = None,
|
||||
ollama: OllamaDep = None,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Store volatile data.
|
||||
|
||||
Data is converted to natural language and embedded for semantic search.
|
||||
If the same namespace+key already exists, it will be updated.
|
||||
|
||||
**Example Request:**
|
||||
```json
|
||||
POST /volatile/store?namespace=weather&key=rotterdam
|
||||
{
|
||||
"data": {
|
||||
"temperature": 8,
|
||||
"conditions": "Cloudy",
|
||||
"humidity": 85
|
||||
},
|
||||
"source": "openweathermap",
|
||||
"ttl": 1800,
|
||||
"refresh_schedule": "0 * * * *"
|
||||
}
|
||||
```
|
||||
|
||||
**Refresh Schedule:**
|
||||
Optional cron expression for automatic refresh. The scheduler
|
||||
will query `/volatile/scheduled` and trigger refreshes.
|
||||
"""
|
||||
# Validate namespace if not custom
|
||||
if namespace != VolatileNamespace.CUSTOM:
|
||||
try:
|
||||
VolatileNamespace(namespace)
|
||||
except ValueError:
|
||||
valid = [ns.value for ns in VolatileNamespace]
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"Invalid namespace '{namespace}'. Valid: {valid}"
|
||||
)
|
||||
|
||||
service = get_volatile_service(qdrant, ollama)
|
||||
|
||||
try:
|
||||
record = await service.store(
|
||||
user=user,
|
||||
namespace=namespace,
|
||||
key=key,
|
||||
data=request.data,
|
||||
source=request.source,
|
||||
ttl=request.ttl,
|
||||
refresh_schedule=request.refresh_schedule,
|
||||
)
|
||||
return record
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to store volatile record: {e}")
|
||||
raise HTTPException(status_code=500, detail=f"Failed to store record: {str(e)}")
|
||||
|
||||
|
||||
@router.get("/{namespace}/{key}", response_model=VolatileRecordResponse)
|
||||
async def get_record(
|
||||
namespace: str,
|
||||
key: str,
|
||||
user: str = Query(default=DEFAULT_USER, description="User identifier"),
|
||||
qdrant: QdrantDep = None,
|
||||
ollama: OllamaDep = None,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Get a specific volatile record by namespace and key.
|
||||
|
||||
**Example:**
|
||||
```
|
||||
GET /volatile/weather/rotterdam?user=jpmschweitzer
|
||||
```
|
||||
"""
|
||||
service = get_volatile_service(qdrant, ollama)
|
||||
record = await service.get(user, namespace, key)
|
||||
|
||||
if not record:
|
||||
raise HTTPException(
|
||||
status_code=404,
|
||||
detail=f"Record '{key}' not found in namespace '{namespace}'"
|
||||
)
|
||||
|
||||
return record
|
||||
|
||||
|
||||
@router.delete("/{namespace}/{key}", response_model=VolatileDeleteResponse)
|
||||
async def delete_record(
|
||||
namespace: str,
|
||||
key: str,
|
||||
user: str = Query(default=DEFAULT_USER, description="User identifier"),
|
||||
qdrant: QdrantDep = None,
|
||||
ollama: OllamaDep = None,
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Delete a specific volatile record.
|
||||
"""
|
||||
service = get_volatile_service(qdrant, ollama)
|
||||
deleted = await service.delete(user, namespace, key)
|
||||
|
||||
return VolatileDeleteResponse(
|
||||
key=key,
|
||||
namespace=namespace,
|
||||
deleted=deleted,
|
||||
user=user,
|
||||
)
|
||||
+130
-4
@@ -5,15 +5,15 @@ Endpoints for wiki page and dossier management.
|
||||
All operations are scoped to user namespaces for multi-tenancy.
|
||||
"""
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Depends, Query, Security, BackgroundTasks
|
||||
from fastapi.security import HTTPAuthorizationCredentials
|
||||
from fastapi import APIRouter, HTTPException, Depends, Query, BackgroundTasks
|
||||
from typing import Optional
|
||||
import logging
|
||||
|
||||
from src.models.wiki import (
|
||||
WikiPage, WikiPageList, WikiPageCreate, WikiPageUpdate, WikiPageMove,
|
||||
WikiOperationResponse, WikiSearchResponse,
|
||||
DossierList, WikiSearchResult
|
||||
DossierList, WikiSearchResult,
|
||||
WikiSmartCreateRequest, WikiSmartCreateResponse
|
||||
)
|
||||
from src.services.wiki_service import WikiService
|
||||
from src.services.graph_service import GraphService
|
||||
@@ -22,8 +22,15 @@ from src.clients.wikijs_client import WikiJSClient
|
||||
from src.clients.neo4j_client import Neo4jClient
|
||||
from src.clients.qdrant_client import QdrantClientWrapper
|
||||
from src.clients.ollama_client import OllamaClient
|
||||
from src.core.dependencies import WikiJSDep, Neo4jDep, QdrantDep, OllamaDep, verify_api_key
|
||||
from src.core.dependencies import (
|
||||
WikiJSDep, Neo4jDep, QdrantDep, OllamaDep, SearXNGDep, ContentExtractorDep,
|
||||
verify_api_key, get_settings, get_hybrid_rag_service, get_ingestion_service
|
||||
)
|
||||
from src.core.multi_tenancy import DEFAULT_USER
|
||||
from src.services.hybrid_rag_service import HybridRAGService
|
||||
from src.services.wiki_page_writer import WikiPageWriter
|
||||
from src.services.entity_linking_utils import apply_bidirectional_entity_linking
|
||||
from src.config import Settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -163,6 +170,125 @@ async def create_page(
|
||||
raise HTTPException(status_code=500, detail="Internal server error")
|
||||
|
||||
|
||||
@router.post("/pages/smart-create", response_model=WikiSmartCreateResponse, status_code=201)
|
||||
async def smart_create_page(
|
||||
request: WikiSmartCreateRequest,
|
||||
background_tasks: BackgroundTasks,
|
||||
wiki_client: WikiJSDep,
|
||||
neo4j_client: Neo4jDep,
|
||||
qdrant_client: QdrantDep,
|
||||
ollama_client: OllamaDep,
|
||||
searxng_client: SearXNGDep,
|
||||
content_extractor: ContentExtractorDep,
|
||||
settings: Settings = Depends(get_settings),
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Create wiki page with intelligent research.
|
||||
|
||||
Combines HybridRAG search with LLM content generation to create
|
||||
rich, well-researched wiki pages in a single API call.
|
||||
|
||||
**Process:**
|
||||
1. Runs HybridRAG search on the topic (wiki + graph + web)
|
||||
2. Uses LLM to synthesize findings into structured wiki content
|
||||
3. Creates the page with proper attribution/sources
|
||||
4. Indexes into vectors + knowledge graph (background)
|
||||
5. Applies bidirectional entity linking (background)
|
||||
|
||||
**Example Request:**
|
||||
```json
|
||||
{
|
||||
"topic": "Docker orchestration patterns",
|
||||
"path": "/technology/containers/docker-orchestration",
|
||||
"tags": ["technology", "devops", "containers"],
|
||||
"user": "jpmschweitzer",
|
||||
"include_web_research": true,
|
||||
"include_wiki_search": true
|
||||
}
|
||||
```
|
||||
|
||||
**Returns:**
|
||||
- Created page with ID, path, content
|
||||
- Research summary (wiki/web/graph result counts)
|
||||
- Entity linking statistics (forward/backward links)
|
||||
"""
|
||||
try:
|
||||
user = request.user or DEFAULT_USER
|
||||
|
||||
# Build services
|
||||
wiki_service = WikiService(wiki_client)
|
||||
vector_service = VectorService(qdrant_client, wiki_client, ollama_client)
|
||||
graph_service = GraphService(neo4j_client, wiki_client)
|
||||
hybrid_rag_service = HybridRAGService(
|
||||
vector_service=vector_service,
|
||||
graph_service=graph_service,
|
||||
searxng_client=searxng_client,
|
||||
ollama_client=ollama_client,
|
||||
content_extractor=content_extractor,
|
||||
settings=settings
|
||||
)
|
||||
wiki_page_writer = WikiPageWriter(ollama_client=ollama_client, settings=settings)
|
||||
|
||||
# Step 1-5: Research + Generate + Create page
|
||||
page, research_data = await wiki_service.smart_create_page(
|
||||
topic=request.topic,
|
||||
user=user,
|
||||
path=request.path,
|
||||
tags=request.tags,
|
||||
hybrid_rag_service=hybrid_rag_service,
|
||||
wiki_page_writer=wiki_page_writer,
|
||||
include_web=request.include_web_research,
|
||||
include_wiki=request.include_wiki_search
|
||||
)
|
||||
|
||||
# Schedule graph and vector updates in background
|
||||
background_tasks.add_task(
|
||||
graph_service.update_from_page,
|
||||
page_id=page.id,
|
||||
user=user
|
||||
)
|
||||
background_tasks.add_task(
|
||||
vector_service.update_from_page,
|
||||
page_id=page.id,
|
||||
user=user
|
||||
)
|
||||
|
||||
# Schedule bidirectional entity linking in background
|
||||
async def run_entity_linking():
|
||||
ingestion_service = get_ingestion_service()
|
||||
return await apply_bidirectional_entity_linking(
|
||||
page_id=page.id,
|
||||
page_title=page.title,
|
||||
user=user,
|
||||
neo4j_client=neo4j_client,
|
||||
wiki_service=wiki_service,
|
||||
ingestion_service=ingestion_service
|
||||
)
|
||||
|
||||
background_tasks.add_task(run_entity_linking)
|
||||
|
||||
logger.info(
|
||||
f"Smart page created: id={page.id}, path={page.path}, "
|
||||
f"sources={research_data['sources_used']}"
|
||||
)
|
||||
|
||||
return WikiSmartCreateResponse(
|
||||
page=page,
|
||||
research_summary=research_data["research_summary"],
|
||||
sources_used=research_data["sources_used"],
|
||||
search_id=research_data["search_id"],
|
||||
entity_linking={"forward_links": 0, "backward_links": 0, "pages_updated": 0}
|
||||
# Note: entity_linking stats are 0 here as it runs in background
|
||||
)
|
||||
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to smart create page: {e}", exc_info=True)
|
||||
raise HTTPException(status_code=500, detail="Internal server error")
|
||||
|
||||
|
||||
@router.put("/pages/{page_id}", response_model=WikiPage)
|
||||
async def update_page(
|
||||
page_id: int,
|
||||
|
||||
@@ -47,7 +47,7 @@ class ConsolidationService:
|
||||
self.ollama = ollama
|
||||
self.wiki = wiki
|
||||
self.settings = settings
|
||||
self.wiki_page_writer = WikiPageWriter(ollama_client=ollama)
|
||||
self.wiki_page_writer = WikiPageWriter(ollama_client=ollama, settings=settings)
|
||||
self.ingestion_service = ingestion_service # Optional to avoid circular dependency
|
||||
|
||||
async def consolidate_knowledge(
|
||||
@@ -410,13 +410,19 @@ This is a PERSONAL knowledge base using Schema.org-aligned taxonomy that capture
|
||||
- Projects: Work projects, personal projects (Schema.org: Project)
|
||||
- Reference: General knowledge, how-tos (Custom extension)
|
||||
|
||||
Identify information worth documenting:
|
||||
1. New topics/people/things that deserve their own wiki page
|
||||
2. Facts that could enhance existing pages
|
||||
3. Entities (people, places, things, concepts) for the knowledge graph
|
||||
ANALYSIS STEPS:
|
||||
1. Read each web result carefully for substantive, factual content
|
||||
2. Identify genuinely novel information not likely already known
|
||||
3. Match topics to appropriate taxonomy categories
|
||||
4. Generate valid paths following the exact format below
|
||||
|
||||
Be INCLUSIVE - if someone searched for it, it's likely worth documenting.
|
||||
Personal information is just as valuable as technical information.
|
||||
RULES:
|
||||
- Do NOT suggest pages for topics with insufficient information in results
|
||||
- Do NOT invent entities not explicitly mentioned in results
|
||||
- Do NOT suggest paths that don't match the taxonomy exactly
|
||||
- Do NOT suggest generic or vague page topics
|
||||
- Be CONSERVATIVE - fewer high-quality suggestions is better than many low-quality ones
|
||||
- ONLY suggest documentation for substantive, specific information
|
||||
|
||||
**CRITICAL: Use ONLY these Schema.org-aligned path prefixes (case-sensitive):**
|
||||
|
||||
@@ -462,11 +468,12 @@ Return ONLY valid JSON:
|
||||
JSON:"""
|
||||
|
||||
try:
|
||||
# Call Ollama for analysis
|
||||
# Call Ollama for analysis (temperature=0.0 for consistent classification)
|
||||
response = await self.ollama.generate_text(
|
||||
prompt=prompt,
|
||||
model=self.settings.reranker_model, # Use mistral-nemo
|
||||
stream=False
|
||||
model=self.settings.ollama_model,
|
||||
stream=False,
|
||||
temperature=0.0
|
||||
)
|
||||
|
||||
if not response:
|
||||
|
||||
@@ -0,0 +1,293 @@
|
||||
"""
|
||||
Document sync service for Library Desk.
|
||||
|
||||
Handles indexing of Paperless-ngx documents into vectors and graph.
|
||||
Called by webhook when Paperless completes document processing.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import re
|
||||
import hashlib
|
||||
import uuid
|
||||
from typing import Optional, List
|
||||
from dataclasses import dataclass
|
||||
|
||||
from src.clients.paperless_client import PaperlessClient
|
||||
from src.clients.qdrant_client import QdrantClientWrapper
|
||||
from src.clients.ollama_client import OllamaClient
|
||||
from src.clients.neo4j_client import Neo4jClient
|
||||
from src.clients.wikijs_client import WikiJSClient
|
||||
from src.core.multi_tenancy import get_qdrant_collection_name
|
||||
from src.config import Settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class IndexResult:
|
||||
"""Result of indexing a single document."""
|
||||
success: bool
|
||||
document_id: int
|
||||
title: str = ""
|
||||
chunks_created: int = 0
|
||||
error: Optional[str] = None
|
||||
|
||||
|
||||
class DocumentSyncService:
|
||||
"""
|
||||
Service for syncing Paperless documents to Library Desk indexes.
|
||||
|
||||
Handles:
|
||||
- Fetching document content from Paperless API
|
||||
- Chunking and embedding into Qdrant
|
||||
- Creating graph nodes in Neo4j
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
paperless_client: PaperlessClient,
|
||||
qdrant_client: QdrantClientWrapper,
|
||||
ollama_client: OllamaClient,
|
||||
neo4j_client: Neo4jClient,
|
||||
wiki_client: WikiJSClient,
|
||||
settings: Settings,
|
||||
chunk_size: int = 500,
|
||||
chunk_overlap: int = 50
|
||||
):
|
||||
self.paperless = paperless_client
|
||||
self.qdrant = qdrant_client
|
||||
self.ollama = ollama_client
|
||||
self.neo4j = neo4j_client
|
||||
self.wiki = wiki_client
|
||||
self.settings = settings
|
||||
self.chunk_size = chunk_size
|
||||
self.chunk_overlap = chunk_overlap
|
||||
|
||||
def _chunk_text(self, text: str) -> List[str]:
|
||||
"""Chunk text into overlapping segments."""
|
||||
text = re.sub(r'\s+', ' ', text).strip()
|
||||
words = text.split()
|
||||
|
||||
if len(words) <= self.chunk_size:
|
||||
return [text] if text else []
|
||||
|
||||
chunks = []
|
||||
start = 0
|
||||
|
||||
while start < len(words):
|
||||
end = start + self.chunk_size
|
||||
chunk_words = words[start:end]
|
||||
chunks.append(' '.join(chunk_words))
|
||||
start = end - self.chunk_overlap
|
||||
|
||||
return chunks
|
||||
|
||||
async def index_document(
|
||||
self,
|
||||
document_id: int,
|
||||
user: str,
|
||||
content: Optional[str] = None,
|
||||
title: Optional[str] = None,
|
||||
) -> IndexResult:
|
||||
"""
|
||||
Index a single document from Paperless into vectors and graph.
|
||||
|
||||
Args:
|
||||
document_id: Paperless document ID
|
||||
user: User identifier for multi-tenancy
|
||||
content: Optional document content (if provided, skip Paperless API call)
|
||||
title: Optional document title (if provided, skip Paperless API call)
|
||||
|
||||
Returns:
|
||||
IndexResult with success status and details
|
||||
"""
|
||||
logger.info(f"Indexing document {document_id} for user {user}")
|
||||
|
||||
try:
|
||||
# If content and title provided (from webhook), skip API call
|
||||
if content is not None and title is not None:
|
||||
doc_title = title
|
||||
doc_content = content
|
||||
original_filename = None
|
||||
correspondent = None
|
||||
document_type = None
|
||||
tags = []
|
||||
else:
|
||||
# Fetch document from Paperless
|
||||
doc = await self.paperless.get_document(document_id)
|
||||
if not doc:
|
||||
return IndexResult(
|
||||
success=False,
|
||||
document_id=document_id,
|
||||
error="Document not found in Paperless"
|
||||
)
|
||||
doc_title = doc.title
|
||||
doc_content = doc.content or ""
|
||||
original_filename = doc.original_file_name
|
||||
correspondent = doc.correspondent
|
||||
document_type = doc.document_type
|
||||
tags = doc.tags
|
||||
|
||||
if not doc_content.strip():
|
||||
logger.warning(f"Document {document_id} has no text content")
|
||||
return IndexResult(
|
||||
success=True,
|
||||
document_id=document_id,
|
||||
title=doc_title,
|
||||
chunks_created=0,
|
||||
error="No text content (possibly image/video only)"
|
||||
)
|
||||
|
||||
# Index vectors
|
||||
chunks_created = await self._index_vectors(
|
||||
document_id=document_id,
|
||||
title=doc_title,
|
||||
content=doc_content,
|
||||
user=user,
|
||||
metadata={
|
||||
"paperless_id": document_id,
|
||||
"original_filename": original_filename,
|
||||
"correspondent": correspondent,
|
||||
"document_type": document_type,
|
||||
"tags": tags,
|
||||
}
|
||||
)
|
||||
|
||||
# Index graph node
|
||||
await self._index_graph(
|
||||
document_id=document_id,
|
||||
title=doc_title,
|
||||
content=doc_content,
|
||||
user=user,
|
||||
)
|
||||
|
||||
# Mark as indexed in Paperless (optional - if custom field exists)
|
||||
try:
|
||||
await self._mark_indexed(document_id)
|
||||
except Exception as e:
|
||||
logger.debug(f"Could not mark document as indexed: {e}")
|
||||
|
||||
logger.info(f"Successfully indexed document {document_id}: {chunks_created} chunks")
|
||||
|
||||
return IndexResult(
|
||||
success=True,
|
||||
document_id=document_id,
|
||||
title=doc_title,
|
||||
chunks_created=chunks_created
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to index document {document_id}: {e}", exc_info=True)
|
||||
return IndexResult(
|
||||
success=False,
|
||||
document_id=document_id,
|
||||
error=str(e)
|
||||
)
|
||||
|
||||
async def _index_vectors(
|
||||
self,
|
||||
document_id: int,
|
||||
title: str,
|
||||
content: str,
|
||||
user: str,
|
||||
metadata: dict,
|
||||
) -> int:
|
||||
"""Create vector embeddings for document content."""
|
||||
collection = get_qdrant_collection_name(user)
|
||||
self.qdrant.ensure_collection(collection)
|
||||
|
||||
# Delete existing chunks for this document
|
||||
try:
|
||||
self.qdrant.client.delete(
|
||||
collection_name=collection,
|
||||
points_selector={
|
||||
"filter": {
|
||||
"must": [
|
||||
{"key": "doc_type", "match": {"value": "document"}},
|
||||
{"key": "paperless_id", "match": {"value": document_id}},
|
||||
]
|
||||
}
|
||||
}
|
||||
)
|
||||
except Exception as e:
|
||||
logger.debug(f"No existing chunks to delete: {e}")
|
||||
|
||||
# Chunk content
|
||||
chunks = self._chunk_text(content)
|
||||
if not chunks:
|
||||
return 0
|
||||
|
||||
# Generate embeddings
|
||||
embeddings = await self.ollama.embed_batch(chunks)
|
||||
|
||||
# Build points
|
||||
points = []
|
||||
for i, (chunk, embedding) in enumerate(zip(chunks, embeddings)):
|
||||
point_id = str(uuid.uuid4())
|
||||
content_hash = hashlib.md5(chunk.encode()).hexdigest()
|
||||
|
||||
points.append({
|
||||
"id": point_id,
|
||||
"vector": embedding,
|
||||
"payload": {
|
||||
"doc_type": "document",
|
||||
"paperless_id": document_id,
|
||||
"title": title,
|
||||
"chunk_text": chunk,
|
||||
"chunk_index": i,
|
||||
"content_hash": content_hash,
|
||||
**metadata
|
||||
}
|
||||
})
|
||||
|
||||
# Upsert to Qdrant
|
||||
if points:
|
||||
self.qdrant.client.upsert(
|
||||
collection_name=collection,
|
||||
points=points
|
||||
)
|
||||
|
||||
return len(points)
|
||||
|
||||
async def _index_graph(
|
||||
self,
|
||||
document_id: int,
|
||||
title: str,
|
||||
content: str,
|
||||
user: str,
|
||||
):
|
||||
"""Create graph node for document."""
|
||||
# Create Document node in Neo4j
|
||||
query = """
|
||||
MERGE (d:Document {paperless_id: $paperless_id, user: $user})
|
||||
SET d.title = $title,
|
||||
d.doc_type = 'document',
|
||||
d.updated_at = datetime()
|
||||
RETURN d
|
||||
"""
|
||||
await self.neo4j.execute_query(
|
||||
query,
|
||||
{
|
||||
"paperless_id": document_id,
|
||||
"user": user,
|
||||
"title": title,
|
||||
}
|
||||
)
|
||||
|
||||
# TODO: Extract entities from content and create relationships
|
||||
# This could use the same entity extraction as wiki pages
|
||||
|
||||
async def _mark_indexed(self, document_id: int):
|
||||
"""Mark document as indexed in Paperless custom field."""
|
||||
# Try to update library_indexed custom field if it exists
|
||||
try:
|
||||
# Look up field ID by name (Paperless requires ID, not name)
|
||||
field = await self.paperless.get_custom_field_by_name("library_indexed")
|
||||
if field:
|
||||
await self.paperless.update_document(
|
||||
document_id=document_id,
|
||||
custom_fields=[{"field": field["id"], "value": True}]
|
||||
)
|
||||
except Exception:
|
||||
# Field might not exist, that's OK
|
||||
pass
|
||||
@@ -0,0 +1,160 @@
|
||||
"""
|
||||
Shared entity linking utilities for Library Desk.
|
||||
|
||||
Provides bidirectional entity linking functionality that can be used by:
|
||||
- Consolidation service (knowledge consolidation)
|
||||
- Wiki router (smart page creation)
|
||||
- Any other service that creates wiki pages
|
||||
"""
|
||||
import logging
|
||||
from typing import Dict, Any, Optional
|
||||
|
||||
from src.core.multi_tenancy import get_neo4j_user_base_label
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
async def apply_bidirectional_entity_linking(
|
||||
page_id: int,
|
||||
page_title: str,
|
||||
user: str,
|
||||
neo4j_client: "Neo4jClient",
|
||||
wiki_service: "WikiService",
|
||||
ingestion_service: Optional["IngestionService"] = None
|
||||
) -> Dict[str, int]:
|
||||
"""
|
||||
Apply bidirectional entity linking after page creation/update.
|
||||
|
||||
This runs AFTER ingestion so entities are extracted and in the graph.
|
||||
|
||||
Steps:
|
||||
1. Link entities in the new page (forward links to existing entities)
|
||||
2. Find pages that mention the new entity (reverse references)
|
||||
3. Link entities in those pages (backward links to the new entity)
|
||||
|
||||
Args:
|
||||
page_id: Wiki page ID
|
||||
page_title: Page title (used to find reverse references)
|
||||
user: User identifier
|
||||
neo4j_client: Neo4j client for graph queries
|
||||
wiki_service: Wiki service for page operations
|
||||
ingestion_service: Optional ingestion service for re-indexing
|
||||
|
||||
Returns:
|
||||
Dict with link counts: {
|
||||
"forward_links": int, # Links added to the new page
|
||||
"backward_links": int, # Links added to other pages pointing to new page
|
||||
"pages_updated": int # Number of other pages updated
|
||||
}
|
||||
"""
|
||||
from src.routers.entity_linking import (
|
||||
link_entities_in_page,
|
||||
EntityLinkingRequest,
|
||||
get_entities_with_paths,
|
||||
add_entity_links_to_content
|
||||
)
|
||||
from src.core.dependencies import get_graph_service, get_wiki_service, get_ingestion_service
|
||||
from src.models.wiki import WikiPageUpdate
|
||||
|
||||
forward_links = 0
|
||||
backward_links = 0
|
||||
pages_updated = 0
|
||||
|
||||
try:
|
||||
graph_service = get_graph_service()
|
||||
|
||||
# Use provided services or get defaults
|
||||
wiki_svc = wiki_service
|
||||
ingestion_svc = ingestion_service or get_ingestion_service()
|
||||
|
||||
# STEP 1: Forward linking - link entities in the new page
|
||||
logger.info(f"Step 1/3: Linking entities in page {page_id} ('{page_title}')")
|
||||
try:
|
||||
forward_result = await link_entities_in_page(
|
||||
request=EntityLinkingRequest(
|
||||
user=user,
|
||||
page_id=page_id,
|
||||
create_relationships=True,
|
||||
re_index_if_changed=False # Already indexed, no need to re-index
|
||||
),
|
||||
wiki_service=wiki_svc,
|
||||
graph_service=graph_service,
|
||||
ingestion_service=ingestion_svc,
|
||||
api_key="" # Internal call, no auth needed
|
||||
)
|
||||
forward_links = forward_result.content_links_added
|
||||
logger.info(f"Added {forward_links} forward links in page {page_id}")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to add forward links: {e}")
|
||||
|
||||
# STEP 2: Find reverse references - which pages mention this new entity?
|
||||
logger.info(f"Step 2/3: Finding pages that mention '{page_title}'")
|
||||
user_base_label = get_neo4j_user_base_label(user)
|
||||
|
||||
# Query to find documents that mention entities with this page's title
|
||||
reverse_query = f"""
|
||||
// Find entities with the same name as the page title
|
||||
MATCH (e:{user_base_label})
|
||||
WHERE toLower(e.name) = toLower($title)
|
||||
AND NOT e:Document
|
||||
|
||||
// Find documents that mention those entities
|
||||
MATCH (d:Document)-[r:MENTIONS]->(e)
|
||||
WHERE d.page_id <> $page_id // Exclude the page itself
|
||||
|
||||
RETURN DISTINCT d.page_id as page_id, d.title as title
|
||||
LIMIT 50
|
||||
"""
|
||||
|
||||
try:
|
||||
reverse_refs = await neo4j_client.execute_query(
|
||||
reverse_query,
|
||||
{"title": page_title, "page_id": page_id}
|
||||
)
|
||||
logger.info(f"Found {len(reverse_refs)} pages that mention '{page_title}'")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to find reverse references: {e}")
|
||||
reverse_refs = []
|
||||
|
||||
# STEP 3: Backward linking - add links in those pages to the new entity
|
||||
if reverse_refs:
|
||||
logger.info(f"Step 3/3: Adding backward links in {len(reverse_refs)} pages")
|
||||
for ref in reverse_refs:
|
||||
try:
|
||||
backward_result = await link_entities_in_page(
|
||||
request=EntityLinkingRequest(
|
||||
user=user,
|
||||
page_id=ref['page_id'],
|
||||
create_relationships=False, # Relationships already exist
|
||||
re_index_if_changed=False # Don't re-index for link updates
|
||||
),
|
||||
wiki_service=wiki_svc,
|
||||
graph_service=graph_service,
|
||||
ingestion_service=ingestion_svc,
|
||||
api_key=""
|
||||
)
|
||||
if backward_result.content_links_added > 0:
|
||||
backward_links += backward_result.content_links_added
|
||||
pages_updated += 1
|
||||
logger.info(
|
||||
f"Added {backward_result.content_links_added} links "
|
||||
f"in page {ref['page_id']} ('{ref['title']}')"
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to add backward links in page {ref['page_id']}: {e}")
|
||||
else:
|
||||
logger.info("Step 3/3: No reverse references found, skipping backward linking")
|
||||
|
||||
return {
|
||||
"forward_links": forward_links,
|
||||
"backward_links": backward_links,
|
||||
"pages_updated": pages_updated
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Bidirectional entity linking failed: {e}", exc_info=True)
|
||||
return {
|
||||
"forward_links": 0,
|
||||
"backward_links": 0,
|
||||
"pages_updated": 0
|
||||
}
|
||||
@@ -1077,8 +1077,18 @@ Feel free to expand it with more details!
|
||||
search_query,
|
||||
{"terms": all_terms, "limit": limit}
|
||||
)
|
||||
logger.info(f"Graph search found {len(results)} documents")
|
||||
return results
|
||||
|
||||
# Deduplicate by page_id (safety net for any edge cases)
|
||||
seen_page_ids = set()
|
||||
unique_results = []
|
||||
for r in results:
|
||||
page_id = r.get("page_id")
|
||||
if page_id and page_id not in seen_page_ids:
|
||||
seen_page_ids.add(page_id)
|
||||
unique_results.append(r)
|
||||
|
||||
logger.info(f"Graph search found {len(unique_results)} unique documents (raw: {len(results)})")
|
||||
return unique_results
|
||||
except Exception as e:
|
||||
logger.error(f"Graph document search failed: {e}", exc_info=True)
|
||||
return []
|
||||
@@ -1251,3 +1261,395 @@ Feel free to expand it with more details!
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to create entity mentions: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
# ========== Cleanup Methods ==========
|
||||
|
||||
async def delete_document_node(
|
||||
self,
|
||||
document_id: str,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Delete a Document Store document node and all its relationships.
|
||||
|
||||
Args:
|
||||
document_id: Document UUID (Document Store)
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of nodes deleted (1 if successful, 0 if not found)
|
||||
"""
|
||||
user_doc_label = get_neo4j_user_label(user)
|
||||
|
||||
delete_query = f"""
|
||||
MATCH (d:{user_doc_label}:Document {{document_id: $document_id}})
|
||||
DETACH DELETE d
|
||||
RETURN count(d) as deleted_count
|
||||
"""
|
||||
|
||||
try:
|
||||
result = await self.neo4j.execute_query(
|
||||
delete_query,
|
||||
{"document_id": document_id}
|
||||
)
|
||||
|
||||
deleted_count = result[0]["deleted_count"] if result else 0
|
||||
|
||||
if deleted_count > 0:
|
||||
logger.info(f"Deleted Document node for document {document_id}")
|
||||
else:
|
||||
logger.warning(f"No Document node found for document {document_id}")
|
||||
|
||||
return deleted_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to delete document {document_id} from graph: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def delete_paperless_document(
|
||||
self,
|
||||
paperless_id: int,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Delete a Paperless document node and all its relationships.
|
||||
|
||||
Args:
|
||||
paperless_id: Paperless-ngx document ID
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of nodes deleted (1 if successful, 0 if not found)
|
||||
"""
|
||||
user_doc_label = get_neo4j_user_label(user)
|
||||
|
||||
delete_query = f"""
|
||||
MATCH (d:{user_doc_label}:Document {{paperless_id: $paperless_id}})
|
||||
DETACH DELETE d
|
||||
RETURN count(d) as deleted_count
|
||||
"""
|
||||
|
||||
try:
|
||||
result = await self.neo4j.execute_query(
|
||||
delete_query,
|
||||
{"paperless_id": paperless_id}
|
||||
)
|
||||
|
||||
deleted_count = result[0]["deleted_count"] if result else 0
|
||||
|
||||
if deleted_count > 0:
|
||||
logger.info(f"Deleted Document node for Paperless document {paperless_id}")
|
||||
else:
|
||||
logger.debug(f"No Document node found for Paperless document {paperless_id}")
|
||||
|
||||
return deleted_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to delete Paperless document {paperless_id} from graph: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def delete_collection_node(
|
||||
self,
|
||||
collection_id: str,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Delete a DocumentCollection node and all contained documents.
|
||||
|
||||
Args:
|
||||
collection_id: Collection UUID
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of nodes deleted (collection + documents)
|
||||
"""
|
||||
user_doc_label = get_neo4j_user_label(user)
|
||||
|
||||
# Delete collection and all documents it contains
|
||||
delete_query = f"""
|
||||
MATCH (c:{user_doc_label}:DocumentCollection {{id: $collection_id}})
|
||||
OPTIONAL MATCH (c)-[:CONTAINS]->(d:Document)
|
||||
DETACH DELETE c, d
|
||||
RETURN count(c) + count(d) as deleted_count
|
||||
"""
|
||||
|
||||
try:
|
||||
result = await self.neo4j.execute_query(
|
||||
delete_query,
|
||||
{"collection_id": collection_id}
|
||||
)
|
||||
|
||||
deleted_count = result[0]["deleted_count"] if result else 0
|
||||
logger.info(f"Deleted collection {collection_id} with {deleted_count} total nodes")
|
||||
return deleted_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to delete collection {collection_id}: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def find_orphan_entities(
|
||||
self,
|
||||
user: str
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Find entities with no MENTIONS relationships (orphaned).
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
List of orphaned entities {id, name, type}
|
||||
"""
|
||||
from src.core.multi_tenancy import get_neo4j_user_base_label
|
||||
|
||||
user_base_label = get_neo4j_user_base_label(user)
|
||||
|
||||
query = f"""
|
||||
MATCH (e:{user_base_label})
|
||||
WHERE NOT e:Document
|
||||
AND NOT e:DocumentCollection
|
||||
AND NOT EXISTS {{ (d:Document)-[:MENTIONS]->(e) }}
|
||||
RETURN elementId(e) as id, e.name as name, labels(e) as labels
|
||||
"""
|
||||
|
||||
try:
|
||||
results = await self.neo4j.execute_query(query, {})
|
||||
|
||||
orphans = []
|
||||
for r in results:
|
||||
labels = r.get("labels", [])
|
||||
entity_type = next(
|
||||
(l for l in labels if l != user_base_label),
|
||||
"Unknown"
|
||||
)
|
||||
orphans.append({
|
||||
"id": r["id"],
|
||||
"name": r["name"],
|
||||
"type": entity_type
|
||||
})
|
||||
|
||||
logger.info(f"Found {len(orphans)} orphan entities for user {user}")
|
||||
return orphans
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to find orphan entities: {e}", exc_info=True)
|
||||
return []
|
||||
|
||||
async def purge_orphan_entities(
|
||||
self,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Delete all orphaned entities (entities with no MENTIONS relationships).
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of entities purged
|
||||
"""
|
||||
from src.core.multi_tenancy import get_neo4j_user_base_label
|
||||
|
||||
user_base_label = get_neo4j_user_base_label(user)
|
||||
|
||||
query = f"""
|
||||
MATCH (e:{user_base_label})
|
||||
WHERE NOT e:Document
|
||||
AND NOT e:DocumentCollection
|
||||
AND NOT EXISTS {{ (d:Document)-[:MENTIONS]->(e) }}
|
||||
DETACH DELETE e
|
||||
RETURN count(e) as purged_count
|
||||
"""
|
||||
|
||||
try:
|
||||
results = await self.neo4j.execute_query(query, {})
|
||||
purged_count = results[0]["purged_count"] if results else 0
|
||||
|
||||
logger.info(f"Purged {purged_count} orphan entities for user {user}")
|
||||
return purged_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to purge orphan entities: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def get_all_document_references(
|
||||
self,
|
||||
user: str
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Get all Document node references for orphan detection.
|
||||
|
||||
Returns page_id for wiki docs and document_id for Document Store docs.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
List of document references {page_id, document_id, doc_type, title}
|
||||
"""
|
||||
user_doc_label = get_neo4j_user_label(user)
|
||||
|
||||
query = f"""
|
||||
MATCH (d:{user_doc_label}:Document)
|
||||
RETURN d.page_id as page_id,
|
||||
d.document_id as document_id,
|
||||
COALESCE(d.doc_type, 'wiki') as doc_type,
|
||||
d.title as title
|
||||
"""
|
||||
|
||||
try:
|
||||
results = await self.neo4j.execute_query(query, {})
|
||||
|
||||
references = []
|
||||
for r in results:
|
||||
references.append({
|
||||
"page_id": r.get("page_id"),
|
||||
"document_id": r.get("document_id"),
|
||||
"doc_type": r.get("doc_type", "wiki"),
|
||||
"title": r.get("title")
|
||||
})
|
||||
|
||||
logger.info(f"Found {len(references)} document references for user {user}")
|
||||
return references
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to get document references: {e}", exc_info=True)
|
||||
return []
|
||||
|
||||
async def purge_stale_documents_by_ids(
|
||||
self,
|
||||
user: str,
|
||||
page_ids: List[int] = None,
|
||||
document_ids: List[str] = None
|
||||
) -> int:
|
||||
"""
|
||||
Delete specific stale Document nodes by their IDs.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
page_ids: List of wiki page IDs to delete
|
||||
document_ids: List of Document Store document IDs to delete
|
||||
|
||||
Returns:
|
||||
Number of documents purged
|
||||
"""
|
||||
user_doc_label = get_neo4j_user_label(user)
|
||||
total_purged = 0
|
||||
|
||||
try:
|
||||
# Purge by page_id (wiki docs)
|
||||
if page_ids:
|
||||
query = f"""
|
||||
MATCH (d:{user_doc_label}:Document)
|
||||
WHERE d.page_id IN $page_ids
|
||||
DETACH DELETE d
|
||||
RETURN count(d) as purged_count
|
||||
"""
|
||||
results = await self.neo4j.execute_query(query, {"page_ids": page_ids})
|
||||
count = results[0]["purged_count"] if results else 0
|
||||
total_purged += count
|
||||
logger.info(f"Purged {count} wiki Document nodes")
|
||||
|
||||
# Purge by document_id (Document Store docs)
|
||||
if document_ids:
|
||||
query = f"""
|
||||
MATCH (d:{user_doc_label}:Document)
|
||||
WHERE d.document_id IN $document_ids
|
||||
DETACH DELETE d
|
||||
RETURN count(d) as purged_count
|
||||
"""
|
||||
results = await self.neo4j.execute_query(query, {"document_ids": document_ids})
|
||||
count = results[0]["purged_count"] if results else 0
|
||||
total_purged += count
|
||||
logger.info(f"Purged {count} Document Store Document nodes")
|
||||
|
||||
return total_purged
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to purge stale documents: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def cleanup_broken_relationships(
|
||||
self,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Clean up broken FOUND relationships from SearchQuery nodes.
|
||||
|
||||
Removes relationships pointing to deleted documents.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of relationships cleaned
|
||||
"""
|
||||
query = """
|
||||
MATCH (sq:SearchQuery)-[r:FOUND]->(d)
|
||||
WHERE NOT EXISTS { (d) }
|
||||
DELETE r
|
||||
RETURN count(r) as cleaned_count
|
||||
"""
|
||||
|
||||
try:
|
||||
results = await self.neo4j.execute_query(query, {})
|
||||
cleaned_count = results[0]["cleaned_count"] if results else 0
|
||||
|
||||
if cleaned_count > 0:
|
||||
logger.info(f"Cleaned {cleaned_count} broken FOUND relationships")
|
||||
|
||||
return cleaned_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to cleanup broken relationships: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def find_documents_without_vectors(
|
||||
self,
|
||||
user: str,
|
||||
vector_references: List[Dict[str, Any]]
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Find Document nodes that have no corresponding vectors.
|
||||
|
||||
Used for bidirectional orphan detection - graph nodes without vector data.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
vector_references: List of vector refs from VectorService.get_all_chunk_references()
|
||||
|
||||
Returns:
|
||||
List of orphan documents {page_id, document_id, doc_type, title}
|
||||
"""
|
||||
# Get all graph document references
|
||||
graph_docs = await self.get_all_document_references(user)
|
||||
|
||||
if not graph_docs:
|
||||
return []
|
||||
|
||||
# Build sets of IDs that have vectors
|
||||
vector_page_ids = {
|
||||
ref.get("page_id") for ref in vector_references
|
||||
if ref.get("doc_type") == "wiki" and ref.get("page_id")
|
||||
}
|
||||
vector_doc_ids = {
|
||||
ref.get("document_id") for ref in vector_references
|
||||
if ref.get("doc_type") != "wiki" and ref.get("document_id")
|
||||
}
|
||||
|
||||
# Find graph docs with no vectors
|
||||
orphans = []
|
||||
for doc in graph_docs:
|
||||
doc_type = doc.get("doc_type", "wiki")
|
||||
|
||||
if doc_type == "wiki":
|
||||
page_id = doc.get("page_id")
|
||||
if page_id and page_id not in vector_page_ids:
|
||||
orphans.append(doc)
|
||||
else:
|
||||
document_id = doc.get("document_id")
|
||||
if document_id and document_id not in vector_doc_ids:
|
||||
orphans.append(doc)
|
||||
|
||||
logger.info(f"Found {len(orphans)} graph documents without vectors for user {user}")
|
||||
return orphans
|
||||
|
||||
@@ -6,7 +6,7 @@ HybridRAG service combining vector, graph, and web search.
|
||||
1. Parallel Retrieval - Vector + Graph + Web search
|
||||
2. RRF Fusion - Merge results with Reciprocal Rank Fusion
|
||||
3. Enrichment - Add related dossiers via graph
|
||||
4. LLM Re-ranking - Re-rank with mistral-nemo
|
||||
4. LLM Re-ranking - Re-rank with configured Ollama model
|
||||
5. Context Formatting - Format for LLM consumption
|
||||
6. Persistence - Store for Librarian processing
|
||||
"""
|
||||
@@ -20,8 +20,10 @@ import logging
|
||||
|
||||
from src.services.vector_service import VectorService
|
||||
from src.services.graph_service import GraphService
|
||||
from src.services.volatile_service import VolatileCacheService
|
||||
from src.clients.searxng_client import SearXNGClient
|
||||
from src.clients.ollama_client import OllamaClient
|
||||
from src.clients.content_extractor import ContentExtractor
|
||||
from src.config import Settings
|
||||
from src.models.hybrid_rag import (
|
||||
HybridRAGConfig, HybridRAGRequest, HybridRAGResponse,
|
||||
@@ -44,7 +46,9 @@ class HybridRAGService:
|
||||
graph_service: GraphService,
|
||||
searxng_client: SearXNGClient,
|
||||
ollama_client: OllamaClient,
|
||||
settings: Settings
|
||||
content_extractor: ContentExtractor,
|
||||
settings: Settings,
|
||||
volatile_service: Optional[VolatileCacheService] = None
|
||||
):
|
||||
"""
|
||||
Initialize HybridRAG service.
|
||||
@@ -54,14 +58,18 @@ class HybridRAGService:
|
||||
graph_service: Service for Neo4j graph search
|
||||
searxng_client: Client for web search
|
||||
ollama_client: Client for LLM (keyword extraction, re-ranking)
|
||||
content_extractor: Client for extracting full content from URLs
|
||||
settings: Application settings
|
||||
volatile_service: Service for volatile cache search (optional)
|
||||
"""
|
||||
self.vector = vector_service
|
||||
self.graph = graph_service
|
||||
self.searxng = searxng_client
|
||||
self.ollama = ollama_client
|
||||
self.content_extractor = content_extractor
|
||||
self.settings = settings
|
||||
self.reranker_model = settings.reranker_model
|
||||
self.volatile = volatile_service
|
||||
self.reranker_model = settings.ollama_model
|
||||
|
||||
async def search(
|
||||
self,
|
||||
@@ -100,15 +108,24 @@ class HybridRAGService:
|
||||
timing["vector_ms"] = raw_results.get("timing", {}).get("vector_ms", 0)
|
||||
timing["graph_ms"] = raw_results.get("timing", {}).get("graph_ms", 0)
|
||||
timing["web_ms"] = raw_results.get("timing", {}).get("web_ms", 0)
|
||||
timing["volatile_ms"] = raw_results.get("timing", {}).get("volatile_ms", 0)
|
||||
|
||||
# Phase 2: RRF Fusion
|
||||
# Phase 2: Three-Source RRF Fusion
|
||||
phase2_start = time.time()
|
||||
|
||||
# Stage 1: Merge wiki sources (vector + graph) into single ranking
|
||||
wiki_merged = self._merge_wiki_sources(
|
||||
vector_results=raw_results.get("vector", []),
|
||||
graph_results=raw_results.get("graph", []),
|
||||
k=config.rrf_k
|
||||
)
|
||||
|
||||
# Stage 2: Final RRF between wiki, volatile, and web
|
||||
# Volatile gets priority boost (smaller k = higher contribution per rank)
|
||||
fused_results = self._reciprocal_rank_fusion(
|
||||
results_by_source={
|
||||
"vector": raw_results.get("vector", []),
|
||||
"graph": raw_results.get("graph", []),
|
||||
"web": raw_results.get("web", [])
|
||||
},
|
||||
wiki_results=wiki_merged,
|
||||
web_results=raw_results.get("web", []),
|
||||
volatile_results=raw_results.get("volatile", []),
|
||||
k=config.rrf_k
|
||||
)
|
||||
timing["fusion_ms"] = (time.time() - phase2_start) * 1000
|
||||
@@ -185,25 +202,21 @@ class HybridRAGService:
|
||||
Returns:
|
||||
Dictionary with keywords, entities, synonyms, expansions
|
||||
"""
|
||||
prompt = f"""Extract search terms from this query. For each important word, provide synonyms and expansions.
|
||||
prompt = f"""Extract search terms from this query.
|
||||
|
||||
Query: "{query}"
|
||||
|
||||
Return ONLY valid JSON:
|
||||
{{
|
||||
"core_keywords": ["key", "words", "from", "query"],
|
||||
"synonyms": {{
|
||||
"word": ["alternative", "terms"]
|
||||
}}
|
||||
}}
|
||||
RULES:
|
||||
- Extract ONLY keywords explicitly present or directly implied in the query
|
||||
- Do NOT invent terms, concepts, or synonyms not clearly related
|
||||
- Do NOT add general knowledge or associations
|
||||
- Provide synonyms ONLY for technical terms with well-known alternatives
|
||||
- Return valid JSON only, no commentary
|
||||
|
||||
Example for "Docker container hosting":
|
||||
Return format:
|
||||
{{
|
||||
"core_keywords": ["docker", "container", "hosting"],
|
||||
"synonyms": {{
|
||||
"docker": ["containerization", "container runtime"],
|
||||
"hosting": ["server", "infrastructure"]
|
||||
}}
|
||||
"core_keywords": ["words", "from", "query"],
|
||||
"synonyms": {{"term": ["direct", "alternatives"]}}
|
||||
}}
|
||||
|
||||
JSON:"""
|
||||
@@ -211,7 +224,8 @@ JSON:"""
|
||||
try:
|
||||
response = await self.ollama.generate_text(
|
||||
prompt=prompt,
|
||||
model=self.reranker_model
|
||||
model=self.reranker_model,
|
||||
temperature=0.0 # Deterministic for consistent extraction
|
||||
)
|
||||
|
||||
# Parse JSON response (handle potential extra text)
|
||||
@@ -284,7 +298,8 @@ JSON:"""
|
||||
response = await self.vector.search(
|
||||
query=query,
|
||||
user=user,
|
||||
limit=config.vector_limit
|
||||
limit=config.vector_limit,
|
||||
score_threshold=self.settings.vector_similarity_threshold
|
||||
)
|
||||
results = [
|
||||
{
|
||||
@@ -309,11 +324,19 @@ JSON:"""
|
||||
async def graph_search():
|
||||
start = time.time()
|
||||
try:
|
||||
# Skip synonyms for graph search - only use core keywords
|
||||
# Synonyms like "author" can match unrelated entities like "author2000"
|
||||
graph_keywords = {
|
||||
"core_keywords": keywords_data.get("core_keywords", []),
|
||||
"entities": keywords_data.get("entities", []),
|
||||
"synonyms": {}, # No synonyms for exact entity matching
|
||||
"expansions": {}
|
||||
}
|
||||
results = await self.graph.search_documents(
|
||||
query=query,
|
||||
user=user,
|
||||
limit=config.graph_limit,
|
||||
keywords_data=keywords_data
|
||||
keywords_data=graph_keywords
|
||||
)
|
||||
formatted = [
|
||||
{
|
||||
@@ -334,7 +357,7 @@ JSON:"""
|
||||
|
||||
tasks["graph"] = graph_search()
|
||||
|
||||
# Web search
|
||||
# Web search with content extraction
|
||||
if config.enable_web:
|
||||
async def web_search():
|
||||
start = time.time()
|
||||
@@ -343,11 +366,24 @@ JSON:"""
|
||||
query=query,
|
||||
limit=config.web_limit
|
||||
)
|
||||
|
||||
# Extract full content from URLs using Trafilatura
|
||||
urls = [r.get("url") for r in results if r.get("url")]
|
||||
extraction_results = await self.content_extractor.extract_batch(urls)
|
||||
|
||||
# Map extracted content back to results by URL
|
||||
url_to_content = {
|
||||
ext.url: ext.content
|
||||
for ext in extraction_results
|
||||
if ext.success and ext.content
|
||||
}
|
||||
|
||||
formatted = [
|
||||
{
|
||||
"url": r.get("url"),
|
||||
"title": r.get("title", ""),
|
||||
"content": r.get("content", ""),
|
||||
"content": url_to_content.get(r.get("url"), r.get("content", "")),
|
||||
"snippet": r.get("content", ""), # Keep original snippet
|
||||
"engine": r.get("engine", ""),
|
||||
"source": "web"
|
||||
}
|
||||
@@ -360,6 +396,37 @@ JSON:"""
|
||||
|
||||
tasks["web"] = web_search()
|
||||
|
||||
# Volatile cache search
|
||||
if config.enable_volatile and self.volatile:
|
||||
async def volatile_search():
|
||||
start = time.time()
|
||||
try:
|
||||
results = await self.volatile.search(
|
||||
user=user,
|
||||
query=query,
|
||||
limit=config.volatile_limit,
|
||||
score_threshold=config.volatile_threshold
|
||||
)
|
||||
formatted = [
|
||||
{
|
||||
"key": r.key,
|
||||
"namespace": r.namespace,
|
||||
"title": f"{r.namespace}: {r.key}",
|
||||
"content": r.data.get("text", "") if isinstance(r.data, dict) else str(r.data),
|
||||
"raw_data": r.data,
|
||||
"source_api": r.source,
|
||||
"ttl_remaining": r.ttl_remaining,
|
||||
"source": "volatile"
|
||||
}
|
||||
for r in results
|
||||
]
|
||||
return formatted, (time.time() - start) * 1000
|
||||
except Exception as e:
|
||||
logger.error(f"Volatile search failed: {e}", exc_info=True)
|
||||
return [], (time.time() - start) * 1000
|
||||
|
||||
tasks["volatile"] = volatile_search()
|
||||
|
||||
# Execute all searches in parallel
|
||||
results_dict = await asyncio.gather(*tasks.values())
|
||||
|
||||
@@ -372,57 +439,159 @@ JSON:"""
|
||||
|
||||
logger.info(
|
||||
f"Parallel retrieval: vector={len(output.get('vector', []))}, "
|
||||
f"graph={len(output.get('graph', []))}, web={len(output.get('web', []))}"
|
||||
f"graph={len(output.get('graph', []))}, web={len(output.get('web', []))}, "
|
||||
f"volatile={len(output.get('volatile', []))}"
|
||||
)
|
||||
|
||||
return output
|
||||
|
||||
def _reciprocal_rank_fusion(
|
||||
def _merge_wiki_sources(
|
||||
self,
|
||||
results_by_source: Dict[str, List],
|
||||
vector_results: List[Dict],
|
||||
graph_results: List[Dict],
|
||||
k: int = 60
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Phase 2: Merge results using Reciprocal Rank Fusion.
|
||||
Stage 1: Merge vector and graph into single wiki ranking using RRF.
|
||||
|
||||
RRF formula: score = sum(1 / (k + rank)) for each source
|
||||
Both sources search the same wiki pool, so we combine them before
|
||||
final RRF with web to avoid double-counting wiki pages.
|
||||
|
||||
Args:
|
||||
results_by_source: Results from each source
|
||||
vector_results: Results from vector search
|
||||
graph_results: Results from graph search
|
||||
k: RRF constant (default 60)
|
||||
|
||||
Returns:
|
||||
Merged and sorted results
|
||||
Merged wiki results sorted by wiki RRF score
|
||||
"""
|
||||
wiki_scores = {}
|
||||
|
||||
# Process vector results
|
||||
for rank, result in enumerate(vector_results, start=1):
|
||||
page_id = result.get("page_id")
|
||||
if not page_id:
|
||||
continue
|
||||
result_id = f"page_{page_id}"
|
||||
|
||||
if result_id not in wiki_scores:
|
||||
wiki_scores[result_id] = {
|
||||
"result": dict(result), # Copy to avoid mutation
|
||||
"wiki_rrf_score": 0.0,
|
||||
"found_by": []
|
||||
}
|
||||
|
||||
wiki_scores[result_id]["wiki_rrf_score"] += 1 / (k + rank)
|
||||
wiki_scores[result_id]["found_by"].append("vector")
|
||||
|
||||
# Process graph results
|
||||
for rank, result in enumerate(graph_results, start=1):
|
||||
page_id = result.get("page_id")
|
||||
if not page_id:
|
||||
continue
|
||||
result_id = f"page_{page_id}"
|
||||
|
||||
if result_id not in wiki_scores:
|
||||
wiki_scores[result_id] = {
|
||||
"result": dict(result),
|
||||
"wiki_rrf_score": 0.0,
|
||||
"found_by": []
|
||||
}
|
||||
|
||||
wiki_scores[result_id]["wiki_rrf_score"] += 1 / (k + rank)
|
||||
wiki_scores[result_id]["found_by"].append("graph")
|
||||
|
||||
# Add graph metadata to existing result
|
||||
wiki_scores[result_id]["result"]["entity_matches"] = result.get("entity_matches")
|
||||
wiki_scores[result_id]["result"]["matched_entities"] = result.get("matched_entities")
|
||||
|
||||
# Sort by wiki RRF score
|
||||
sorted_wiki = sorted(
|
||||
wiki_scores.values(),
|
||||
key=lambda x: x["wiki_rrf_score"],
|
||||
reverse=True
|
||||
)
|
||||
|
||||
# Return merged results with wiki ranking
|
||||
merged = []
|
||||
for wiki_rank, item in enumerate(sorted_wiki, start=1):
|
||||
merged.append({
|
||||
**item["result"],
|
||||
"wiki_rank": wiki_rank,
|
||||
"wiki_rrf_score": item["wiki_rrf_score"],
|
||||
"found_by": item["found_by"],
|
||||
"source": "wiki"
|
||||
})
|
||||
|
||||
logger.info(f"Wiki merge: {len(merged)} unique pages from vector+graph")
|
||||
return merged
|
||||
|
||||
def _reciprocal_rank_fusion(
|
||||
self,
|
||||
wiki_results: List[Dict],
|
||||
web_results: List[Dict],
|
||||
volatile_results: Optional[List[Dict]] = None,
|
||||
k: int = 60
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Stage 2: Final RRF between wiki, volatile, and web.
|
||||
|
||||
Wiki results are pre-merged from vector+graph. Volatile results
|
||||
get a priority boost (smaller effective k) since they represent
|
||||
current, time-sensitive information.
|
||||
|
||||
Args:
|
||||
wiki_results: Pre-merged wiki results from _merge_wiki_sources()
|
||||
web_results: Results from web search
|
||||
volatile_results: Results from volatile cache (fresh data)
|
||||
k: RRF constant (default 60)
|
||||
|
||||
Returns:
|
||||
Final merged and sorted results
|
||||
"""
|
||||
rrf_scores = {}
|
||||
volatile_results = volatile_results or []
|
||||
|
||||
for source, results in results_by_source.items():
|
||||
for rank, result in enumerate(results, start=1):
|
||||
# Use page_id for wiki results, url hash for web results
|
||||
if result.get("page_id"):
|
||||
result_id = f"page_{result['page_id']}"
|
||||
elif result.get("url"):
|
||||
result_id = f"url_{hash(result['url'])}"
|
||||
else:
|
||||
continue # Skip results without ID
|
||||
# Volatile results get priority boost (k/2 = stronger score per rank)
|
||||
volatile_k = k // 2
|
||||
for rank, result in enumerate(volatile_results, start=1):
|
||||
key = result.get("key")
|
||||
namespace = result.get("namespace", "unknown")
|
||||
if not key:
|
||||
continue
|
||||
result_id = f"volatile_{namespace}_{key}"
|
||||
rrf_scores[result_id] = {
|
||||
"result": result,
|
||||
"rrf_score": 1 / (volatile_k + rank), # Priority boost
|
||||
"sources": ["volatile"],
|
||||
"source_type": "volatile"
|
||||
}
|
||||
|
||||
if result_id not in rrf_scores:
|
||||
rrf_scores[result_id] = {
|
||||
"result": result,
|
||||
"rrf_score": 0.0,
|
||||
"sources": [],
|
||||
"source_type": source
|
||||
}
|
||||
# Wiki results (single source, already merged)
|
||||
for rank, result in enumerate(wiki_results, start=1):
|
||||
page_id = result.get("page_id")
|
||||
if not page_id:
|
||||
continue
|
||||
result_id = f"page_{page_id}"
|
||||
rrf_scores[result_id] = {
|
||||
"result": result,
|
||||
"rrf_score": 1 / (k + rank),
|
||||
"sources": result.get("found_by", ["wiki"]),
|
||||
"source_type": "wiki"
|
||||
}
|
||||
|
||||
# RRF formula: sum of 1/(k + rank) across sources
|
||||
rrf_scores[result_id]["rrf_score"] += 1 / (k + rank)
|
||||
rrf_scores[result_id]["sources"].append(source)
|
||||
|
||||
# If result appears in multiple sources, update source_type
|
||||
if len(rrf_scores[result_id]["sources"]) > 1:
|
||||
rrf_scores[result_id]["source_type"] = "+".join(
|
||||
sorted(set(rrf_scores[result_id]["sources"]))
|
||||
)
|
||||
# Web results (single source)
|
||||
for rank, result in enumerate(web_results, start=1):
|
||||
url = result.get("url")
|
||||
if not url:
|
||||
continue
|
||||
result_id = f"url_{hash(url)}"
|
||||
rrf_scores[result_id] = {
|
||||
"result": result,
|
||||
"rrf_score": 1 / (k + rank),
|
||||
"sources": ["web"],
|
||||
"source_type": "web"
|
||||
}
|
||||
|
||||
# Sort by RRF score descending
|
||||
sorted_results = sorted(
|
||||
@@ -431,7 +600,8 @@ JSON:"""
|
||||
reverse=True
|
||||
)
|
||||
|
||||
logger.info(f"RRF fusion: {len(sorted_results)} unique results from {len(results_by_source)} sources")
|
||||
volatile_count = len([r for r in sorted_results if r["source_type"] == "volatile"])
|
||||
logger.info(f"Final RRF: {len(sorted_results)} results (wiki + volatile[{volatile_count}] + web)")
|
||||
|
||||
return sorted_results
|
||||
|
||||
@@ -509,21 +679,27 @@ JSON:"""
|
||||
for i, r in enumerate(results)
|
||||
])
|
||||
|
||||
prompt = f"""Given this search query and documents, rank them by relevance.
|
||||
prompt = f"""Rank these documents by relevance to the query.
|
||||
|
||||
Query: {query}
|
||||
|
||||
Documents:
|
||||
{docs_text}
|
||||
|
||||
Return only the numbers in order of relevance (most relevant first).
|
||||
Example: 3,1,5,2,4
|
||||
RULES:
|
||||
- Rank ONLY by how well content answers the query
|
||||
- Do NOT consider document length, formatting, or style
|
||||
- Do NOT add explanation or commentary
|
||||
- Return ONLY comma-separated numbers, most relevant first
|
||||
|
||||
Example output: 3,1,5,2,4
|
||||
|
||||
Ranking:"""
|
||||
|
||||
response = await self.ollama.generate_text(
|
||||
prompt=prompt,
|
||||
model=self.reranker_model
|
||||
model=self.reranker_model,
|
||||
temperature=0.0 # Deterministic for consistent rankings
|
||||
)
|
||||
|
||||
# Parse response: "3,1,5,2,4" → [2, 0, 4, 1, 3] (0-indexed)
|
||||
|
||||
@@ -0,0 +1,265 @@
|
||||
"""
|
||||
RAG Search service for Library Desk.
|
||||
|
||||
Provides web, news, and image search with content extraction:
|
||||
- Uses SearXNG for search queries
|
||||
- Uses Trafilatura for content extraction
|
||||
- Caches results in Redis
|
||||
"""
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
import time
|
||||
from typing import List, Optional
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import redis.asyncio as aioredis
|
||||
|
||||
from src.clients.searxng_client import SearXNGClient
|
||||
from src.clients.content_extractor import ContentExtractor
|
||||
from src.config import Settings
|
||||
from src.models.rag_search import (
|
||||
SearchType,
|
||||
RAGSearchRequest,
|
||||
RAGSearchResult,
|
||||
RAGSearchResponse,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def extract_domain(url: str) -> str:
|
||||
"""Extract domain name from URL, removing 'www.' prefix."""
|
||||
try:
|
||||
parsed = urlparse(url)
|
||||
domain = parsed.netloc
|
||||
return domain.removeprefix("www.")
|
||||
except Exception:
|
||||
return url
|
||||
|
||||
|
||||
class RAGSearchService:
|
||||
"""
|
||||
Service for RAG-optimized web search with content extraction.
|
||||
|
||||
Combines SearXNG search with Trafilatura content extraction
|
||||
and Redis caching for efficient RAG pipeline integration.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
searxng_client: SearXNGClient,
|
||||
content_extractor: ContentExtractor,
|
||||
redis_client: aioredis.Redis,
|
||||
settings: Settings
|
||||
):
|
||||
"""
|
||||
Initialize RAG search service.
|
||||
|
||||
Args:
|
||||
searxng_client: SearXNG search client
|
||||
content_extractor: Trafilatura content extractor
|
||||
redis_client: Async Redis client for caching
|
||||
settings: Application settings
|
||||
"""
|
||||
self.searxng = searxng_client
|
||||
self.extractor = content_extractor
|
||||
self.redis = redis_client
|
||||
self.settings = settings
|
||||
|
||||
self.cache_ttl = settings.search_cache_ttl
|
||||
self.default_limit = settings.search_default_limit
|
||||
|
||||
logger.info(
|
||||
f"Initialized RAGSearchService: cache_ttl={self.cache_ttl}s, "
|
||||
f"default_limit={self.default_limit}"
|
||||
)
|
||||
|
||||
def _cache_key(self, query: str, search_type: str, limit: int) -> str:
|
||||
"""Generate cache key from search parameters."""
|
||||
key_data = f"{query}:{search_type}:{limit}"
|
||||
key_hash = hashlib.md5(key_data.encode()).hexdigest()
|
||||
return f"rag_search:{key_hash}"
|
||||
|
||||
async def _get_cached_result(self, cache_key: str) -> Optional[RAGSearchResponse]:
|
||||
"""Try to get cached search result."""
|
||||
try:
|
||||
cached = await self.redis.get(cache_key)
|
||||
if cached:
|
||||
data = json.loads(cached)
|
||||
logger.debug(f"Cache hit: {cache_key}")
|
||||
return RAGSearchResponse(**data)
|
||||
except Exception as e:
|
||||
logger.warning(f"Cache read failed: {e}")
|
||||
return None
|
||||
|
||||
async def _set_cached_result(self, cache_key: str, result: RAGSearchResponse):
|
||||
"""Cache search result."""
|
||||
try:
|
||||
await self.redis.setex(
|
||||
cache_key,
|
||||
self.cache_ttl,
|
||||
result.model_dump_json()
|
||||
)
|
||||
logger.debug(f"Cached result: {cache_key} (TTL={self.cache_ttl}s)")
|
||||
except Exception as e:
|
||||
logger.warning(f"Cache write failed: {e}")
|
||||
|
||||
async def _search_searxng(
|
||||
self,
|
||||
query: str,
|
||||
search_type: SearchType,
|
||||
limit: int
|
||||
) -> List[dict]:
|
||||
"""Execute search via SearXNG based on search type."""
|
||||
try:
|
||||
if search_type == SearchType.WEB:
|
||||
results = await self.searxng.search_general(
|
||||
query=query,
|
||||
limit=limit
|
||||
)
|
||||
elif search_type == SearchType.NEWS:
|
||||
results = await self.searxng.search_news(
|
||||
query=query,
|
||||
limit=limit
|
||||
)
|
||||
elif search_type == SearchType.IMAGES:
|
||||
results = await self.searxng.search_images(
|
||||
query=query,
|
||||
limit=limit
|
||||
)
|
||||
else:
|
||||
results = await self.searxng.search_general(
|
||||
query=query,
|
||||
limit=limit
|
||||
)
|
||||
|
||||
return results
|
||||
except Exception as e:
|
||||
logger.error(f"SearXNG search failed: {e}")
|
||||
raise
|
||||
|
||||
async def _extract_content_for_results(
|
||||
self,
|
||||
results: List[dict]
|
||||
) -> List[RAGSearchResult]:
|
||||
"""Extract full content from search result URLs."""
|
||||
# Get URLs for extraction
|
||||
urls = [r.get("url", "") for r in results if r.get("url")]
|
||||
|
||||
# Extract content in parallel
|
||||
extraction_results = await self.extractor.extract_batch(urls)
|
||||
|
||||
# Build result objects
|
||||
search_results = []
|
||||
for i, raw_result in enumerate(results):
|
||||
url = raw_result.get("url", "")
|
||||
|
||||
# Find matching extraction result
|
||||
extracted_content = ""
|
||||
for ext_result in extraction_results:
|
||||
if ext_result.url == url and ext_result.success:
|
||||
extracted_content = ext_result.content
|
||||
break
|
||||
|
||||
# Get original snippet
|
||||
snippet = raw_result.get("content", "")
|
||||
if len(snippet) > 300:
|
||||
snippet = snippet[:300] + "..."
|
||||
|
||||
# Build result
|
||||
search_results.append(RAGSearchResult(
|
||||
title=raw_result.get("title", ""),
|
||||
url=url,
|
||||
content=extracted_content,
|
||||
snippet=snippet,
|
||||
source=extract_domain(url),
|
||||
published_date=raw_result.get("publishedDate")
|
||||
))
|
||||
|
||||
return search_results
|
||||
|
||||
def _generate_sources_summary(self, results: List[RAGSearchResult]) -> str:
|
||||
"""Generate markdown list of source URLs."""
|
||||
if not results:
|
||||
return ""
|
||||
|
||||
lines = ["## Sources"]
|
||||
for i, r in enumerate(results, 1):
|
||||
lines.append(f"{i}. [{r.title}]({r.url})")
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
async def search(
|
||||
self,
|
||||
query: str,
|
||||
search_type: SearchType = SearchType.WEB,
|
||||
limit: Optional[int] = None,
|
||||
user: str = "default"
|
||||
) -> RAGSearchResponse:
|
||||
"""
|
||||
Execute RAG-optimized search.
|
||||
|
||||
Args:
|
||||
query: Search query string
|
||||
search_type: Type of search (web, news, images)
|
||||
limit: Maximum results to return (default from settings)
|
||||
user: User identifier for logging/rate limiting
|
||||
|
||||
Returns:
|
||||
RAGSearchResponse with extracted content and sources
|
||||
|
||||
Raises:
|
||||
ValueError: If query is empty
|
||||
Exception: If search fails
|
||||
"""
|
||||
start_time = time.time()
|
||||
|
||||
if not query or not query.strip():
|
||||
raise ValueError("Query cannot be empty")
|
||||
|
||||
effective_limit = limit or self.default_limit
|
||||
|
||||
# Check cache
|
||||
cache_key = self._cache_key(query, search_type.value, effective_limit)
|
||||
cached = await self._get_cached_result(cache_key)
|
||||
if cached:
|
||||
return cached
|
||||
|
||||
logger.info(
|
||||
f"RAG search: '{query}' type={search_type.value} "
|
||||
f"limit={effective_limit} user={user}"
|
||||
)
|
||||
|
||||
# Execute search
|
||||
raw_results = await self._search_searxng(query, search_type, effective_limit)
|
||||
|
||||
# Extract content from results
|
||||
search_results = await self._extract_content_for_results(raw_results)
|
||||
|
||||
# Generate sources summary
|
||||
sources_summary = self._generate_sources_summary(search_results)
|
||||
|
||||
# Calculate timing
|
||||
search_time_ms = int((time.time() - start_time) * 1000)
|
||||
|
||||
# Build response
|
||||
response = RAGSearchResponse(
|
||||
query=query,
|
||||
search_type=search_type,
|
||||
results=search_results,
|
||||
total_results=len(search_results),
|
||||
search_time_ms=search_time_ms,
|
||||
sources_summary=sources_summary
|
||||
)
|
||||
|
||||
# Cache result
|
||||
await self._set_cached_result(cache_key, response)
|
||||
|
||||
logger.info(
|
||||
f"RAG search completed: {len(search_results)} results "
|
||||
f"in {search_time_ms}ms"
|
||||
)
|
||||
|
||||
return response
|
||||
@@ -356,3 +356,222 @@ class VectorService:
|
||||
collections=[],
|
||||
total=0
|
||||
)
|
||||
|
||||
# ========== Cleanup Methods ==========
|
||||
|
||||
async def delete_document_chunks(
|
||||
self,
|
||||
document_id: str,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Delete all chunks for a document (Document Store).
|
||||
|
||||
Args:
|
||||
document_id: Document UUID
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of chunks deleted
|
||||
"""
|
||||
collection_name = get_qdrant_collection_name(user)
|
||||
|
||||
try:
|
||||
deleted_count = await self.qdrant.delete_by_filter(
|
||||
collection_name=collection_name,
|
||||
filter_conditions={"document_id": document_id}
|
||||
)
|
||||
|
||||
logger.info(f"Deleted chunks for document {document_id}")
|
||||
return deleted_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to delete chunks for document {document_id}: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def delete_paperless_document_chunks(
|
||||
self,
|
||||
paperless_id: int,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Delete all chunks for a Paperless document.
|
||||
|
||||
Args:
|
||||
paperless_id: Paperless-ngx document ID
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of chunks deleted
|
||||
"""
|
||||
collection_name = get_qdrant_collection_name(user)
|
||||
|
||||
try:
|
||||
deleted_count = await self.qdrant.delete_by_filter(
|
||||
collection_name=collection_name,
|
||||
filter_conditions={
|
||||
"doc_type": "document",
|
||||
"paperless_id": paperless_id
|
||||
}
|
||||
)
|
||||
|
||||
logger.info(f"Deleted chunks for Paperless document {paperless_id}")
|
||||
return deleted_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to delete chunks for Paperless document {paperless_id}: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def delete_collection_chunks(
|
||||
self,
|
||||
collection_id: str,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Delete all chunks for a document collection.
|
||||
|
||||
Args:
|
||||
collection_id: Collection UUID
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of chunks deleted
|
||||
"""
|
||||
collection_name = get_qdrant_collection_name(user)
|
||||
|
||||
try:
|
||||
deleted_count = await self.qdrant.delete_by_filter(
|
||||
collection_name=collection_name,
|
||||
filter_conditions={"collection_id": collection_id}
|
||||
)
|
||||
|
||||
logger.info(f"Deleted chunks for collection {collection_id}")
|
||||
return deleted_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to delete chunks for collection {collection_id}: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
async def get_all_chunk_references(
|
||||
self,
|
||||
user: str
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Get all chunk references for orphan detection.
|
||||
|
||||
Returns list of {id, page_id, document_id} for all chunks.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
List of chunk references
|
||||
"""
|
||||
collection_name = get_qdrant_collection_name(user)
|
||||
|
||||
try:
|
||||
# Check if collection exists
|
||||
exists = await self.qdrant.collection_exists(collection_name)
|
||||
if not exists:
|
||||
return []
|
||||
|
||||
all_points = await self.qdrant.scroll_all_points(
|
||||
collection_name=collection_name,
|
||||
batch_size=100,
|
||||
with_payload=True
|
||||
)
|
||||
|
||||
references = []
|
||||
for point in all_points:
|
||||
payload = point.get("payload", {})
|
||||
references.append({
|
||||
"chunk_id": point["id"],
|
||||
"page_id": payload.get("page_id"),
|
||||
"document_id": payload.get("document_id"),
|
||||
"collection_id": payload.get("collection_id"),
|
||||
"doc_type": payload.get("doc_type", "wiki")
|
||||
})
|
||||
|
||||
logger.info(f"Found {len(references)} chunks for user {user}")
|
||||
return references
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to get chunk references: {e}", exc_info=True)
|
||||
return []
|
||||
|
||||
async def purge_chunks_by_ids(
|
||||
self,
|
||||
user: str,
|
||||
chunk_ids: List[str]
|
||||
) -> int:
|
||||
"""
|
||||
Delete specific chunks by their IDs.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
chunk_ids: List of chunk IDs to delete
|
||||
|
||||
Returns:
|
||||
Number of chunks deleted
|
||||
"""
|
||||
if not chunk_ids:
|
||||
return 0
|
||||
|
||||
collection_name = get_qdrant_collection_name(user)
|
||||
|
||||
try:
|
||||
deleted_count = await self.qdrant.delete_by_ids(
|
||||
collection_name=collection_name,
|
||||
point_ids=chunk_ids
|
||||
)
|
||||
|
||||
logger.info(f"Purged {deleted_count} orphan chunks for user {user}")
|
||||
return deleted_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to purge chunks: {e}", exc_info=True)
|
||||
return 0
|
||||
|
||||
def find_chunks_without_graph_nodes(
|
||||
self,
|
||||
chunk_references: List[Dict[str, Any]],
|
||||
graph_references: List[Dict[str, Any]]
|
||||
) -> List[str]:
|
||||
"""
|
||||
Find vector chunks that have no corresponding graph Document node.
|
||||
|
||||
Used for bidirectional orphan detection - vectors without graph representation.
|
||||
|
||||
Args:
|
||||
chunk_references: List from get_all_chunk_references()
|
||||
graph_references: List from GraphService.get_all_document_references()
|
||||
|
||||
Returns:
|
||||
List of orphan chunk IDs
|
||||
"""
|
||||
# Build sets of IDs that have graph nodes
|
||||
graph_page_ids = {
|
||||
ref.get("page_id") for ref in graph_references
|
||||
if ref.get("doc_type") == "wiki" and ref.get("page_id")
|
||||
}
|
||||
graph_doc_ids = {
|
||||
ref.get("document_id") for ref in graph_references
|
||||
if ref.get("doc_type") != "wiki" and ref.get("document_id")
|
||||
}
|
||||
|
||||
# Find chunks with no graph node
|
||||
orphan_ids = []
|
||||
for chunk in chunk_references:
|
||||
doc_type = chunk.get("doc_type", "wiki")
|
||||
|
||||
if doc_type == "wiki":
|
||||
page_id = chunk.get("page_id")
|
||||
if page_id and page_id not in graph_page_ids:
|
||||
orphan_ids.append(chunk["chunk_id"])
|
||||
else:
|
||||
document_id = chunk.get("document_id")
|
||||
if document_id and document_id not in graph_doc_ids:
|
||||
orphan_ids.append(chunk["chunk_id"])
|
||||
|
||||
logger.info(f"Found {len(orphan_ids)} vector chunks without graph nodes")
|
||||
return orphan_ids
|
||||
|
||||
@@ -0,0 +1,564 @@
|
||||
"""
|
||||
Volatile Cache service for Library Desk.
|
||||
|
||||
Provides ephemeral data storage with TTL using Qdrant vectors:
|
||||
- Weather, news, financial data
|
||||
- Transit schedules, traffic conditions
|
||||
- System status, social notifications
|
||||
|
||||
Data is stored as embedded vectors for semantic search retrieval.
|
||||
"""
|
||||
|
||||
import hashlib
|
||||
import logging
|
||||
import time
|
||||
from datetime import datetime
|
||||
from typing import List, Optional, Dict, Any
|
||||
|
||||
from src.clients.qdrant_client import QdrantClientWrapper
|
||||
from src.clients.ollama_client import OllamaClient
|
||||
from src.config import Settings
|
||||
from src.models.volatile import (
|
||||
VolatileRecordResponse,
|
||||
VolatileNamespace,
|
||||
NAMESPACE_DEFAULT_TTL,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class VolatileCacheService:
|
||||
"""
|
||||
Service for volatile data with TTL stored in Qdrant.
|
||||
|
||||
Stores ephemeral data as vectors for semantic search retrieval.
|
||||
Each user has an isolated volatile collection.
|
||||
"""
|
||||
|
||||
COLLECTION_PREFIX = "volatile_"
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
qdrant_client: QdrantClientWrapper,
|
||||
ollama_client: OllamaClient,
|
||||
settings: Settings
|
||||
):
|
||||
"""
|
||||
Initialize volatile cache service.
|
||||
|
||||
Args:
|
||||
qdrant_client: Qdrant client for vector storage
|
||||
ollama_client: Ollama client for embeddings
|
||||
settings: Application settings
|
||||
"""
|
||||
self.qdrant = qdrant_client
|
||||
self.ollama = ollama_client
|
||||
self.settings = settings
|
||||
|
||||
logger.info("Initialized VolatileCacheService (Qdrant backend)")
|
||||
|
||||
def _collection_name(self, user: str) -> str:
|
||||
"""Get volatile collection name for user."""
|
||||
return f"{self.COLLECTION_PREFIX}{user}"
|
||||
|
||||
def _make_vector_id(self, namespace: str, key: str) -> str:
|
||||
"""
|
||||
Generate deterministic vector ID for namespace/key.
|
||||
|
||||
Same namespace+key always produces same ID for upsert behavior.
|
||||
"""
|
||||
combined = f"{namespace}:{key}"
|
||||
return hashlib.md5(combined.encode()).hexdigest()
|
||||
|
||||
def _get_default_ttl(self, namespace: str) -> int:
|
||||
"""Get default TTL for a namespace."""
|
||||
try:
|
||||
ns = VolatileNamespace(namespace)
|
||||
return NAMESPACE_DEFAULT_TTL.get(ns, self.settings.volatile_default_ttl)
|
||||
except ValueError:
|
||||
return self.settings.volatile_default_ttl
|
||||
|
||||
def _current_timestamp_ms(self) -> int:
|
||||
"""Get current timestamp in milliseconds."""
|
||||
return int(time.time() * 1000)
|
||||
|
||||
def _to_natural_language(
|
||||
self,
|
||||
namespace: str,
|
||||
key: str,
|
||||
data: Dict[str, Any]
|
||||
) -> str:
|
||||
"""
|
||||
Convert structured data to natural language for embedding.
|
||||
|
||||
This creates a text representation that embeds well semantically.
|
||||
"""
|
||||
# Template-based conversion for known namespaces
|
||||
if namespace == VolatileNamespace.WEATHER:
|
||||
temp = data.get("temperature", data.get("temp", "unknown"))
|
||||
conditions = data.get("conditions", data.get("weather", ""))
|
||||
humidity = data.get("humidity", "")
|
||||
text = f"Current weather in {key}: {temp}°C"
|
||||
if conditions:
|
||||
text += f", {conditions}"
|
||||
if humidity:
|
||||
text += f", humidity {humidity}%"
|
||||
return text
|
||||
|
||||
elif namespace == VolatileNamespace.NEWS:
|
||||
title = data.get("title", data.get("headline", ""))
|
||||
summary = data.get("summary", data.get("description", ""))
|
||||
source = data.get("source", "")
|
||||
text = f"News: {title}"
|
||||
if summary:
|
||||
text += f". {summary}"
|
||||
if source:
|
||||
text += f" (Source: {source})"
|
||||
return text
|
||||
|
||||
elif namespace == VolatileNamespace.FINANCIAL:
|
||||
symbol = data.get("symbol", key)
|
||||
price = data.get("price", "")
|
||||
change = data.get("change", data.get("change_percent", ""))
|
||||
text = f"Financial data for {symbol}"
|
||||
if price:
|
||||
text += f": price {price}"
|
||||
if change:
|
||||
text += f", change {change}%"
|
||||
return text
|
||||
|
||||
elif namespace == VolatileNamespace.TRANSIT:
|
||||
route = data.get("route", data.get("line", key))
|
||||
status = data.get("status", "")
|
||||
delay = data.get("delay", data.get("delay_minutes", ""))
|
||||
text = f"Transit {route}"
|
||||
if status:
|
||||
text += f": {status}"
|
||||
if delay:
|
||||
text += f", delay {delay} minutes"
|
||||
return text
|
||||
|
||||
elif namespace == VolatileNamespace.TRAFFIC:
|
||||
location = data.get("location", key)
|
||||
duration = data.get("duration", data.get("travel_time", ""))
|
||||
congestion = data.get("congestion", "")
|
||||
text = f"Traffic for {location}"
|
||||
if duration:
|
||||
text += f": {duration} minutes"
|
||||
if congestion:
|
||||
text += f", congestion level {congestion}"
|
||||
return text
|
||||
|
||||
elif namespace == VolatileNamespace.AIR_QUALITY:
|
||||
location = data.get("location", key)
|
||||
aqi = data.get("aqi", data.get("index", ""))
|
||||
quality = data.get("quality", "")
|
||||
text = f"Air quality in {location}"
|
||||
if aqi:
|
||||
text += f": AQI {aqi}"
|
||||
if quality:
|
||||
text += f" ({quality})"
|
||||
return text
|
||||
|
||||
elif namespace == VolatileNamespace.SPORTS:
|
||||
event = data.get("event", data.get("match", key))
|
||||
score = data.get("score", "")
|
||||
status = data.get("status", "")
|
||||
text = f"Sports: {event}"
|
||||
if score:
|
||||
text += f" - Score: {score}"
|
||||
if status:
|
||||
text += f" ({status})"
|
||||
return text
|
||||
|
||||
elif namespace == VolatileNamespace.SYSTEM:
|
||||
service = data.get("service", key)
|
||||
status = data.get("status", "unknown")
|
||||
message = data.get("message", "")
|
||||
text = f"System status for {service}: {status}"
|
||||
if message:
|
||||
text += f". {message}"
|
||||
return text
|
||||
|
||||
# Fallback: serialize key fields
|
||||
text_parts = [f"{namespace} data for {key}:"]
|
||||
for k, v in data.items():
|
||||
if isinstance(v, (str, int, float, bool)):
|
||||
text_parts.append(f"{k}: {v}")
|
||||
return " ".join(text_parts)
|
||||
|
||||
async def store(
|
||||
self,
|
||||
user: str,
|
||||
namespace: str,
|
||||
key: str,
|
||||
data: Dict[str, Any],
|
||||
source: Optional[str] = None,
|
||||
ttl: Optional[int] = None,
|
||||
refresh_schedule: Optional[str] = None
|
||||
) -> VolatileRecordResponse:
|
||||
"""
|
||||
Store volatile data as an embedded vector.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
namespace: Data namespace (from controlled list)
|
||||
key: Record key (normalized slug)
|
||||
data: Structured data to store
|
||||
source: Origin API/service
|
||||
ttl: TTL in seconds (uses namespace default if not set)
|
||||
refresh_schedule: Optional cron expression for refresh
|
||||
|
||||
Returns:
|
||||
The stored record
|
||||
"""
|
||||
collection = self._collection_name(user)
|
||||
|
||||
# Ensure collection exists
|
||||
await self.qdrant.ensure_collection(collection)
|
||||
|
||||
# Calculate TTL and expiry
|
||||
effective_ttl = ttl if ttl is not None else self._get_default_ttl(namespace)
|
||||
now_ms = self._current_timestamp_ms()
|
||||
expiry_ms = now_ms + (effective_ttl * 1000)
|
||||
|
||||
# Convert to natural language for embedding
|
||||
text = self._to_natural_language(namespace, key, data)
|
||||
|
||||
# Generate embedding
|
||||
embedding = await self.ollama.embed(text)
|
||||
if not embedding:
|
||||
raise ValueError("Failed to generate embedding for volatile data")
|
||||
|
||||
# Build payload
|
||||
now = datetime.utcnow()
|
||||
payload = {
|
||||
"doc_type": "volatile",
|
||||
"namespace": namespace,
|
||||
"key": key,
|
||||
"text": text,
|
||||
"raw_data": data,
|
||||
"source": source,
|
||||
"created_at": now.isoformat(),
|
||||
"updated_at": now.isoformat(),
|
||||
"ttl": effective_ttl,
|
||||
"ttl_expiry": expiry_ms,
|
||||
"refresh_schedule": refresh_schedule,
|
||||
"user": user,
|
||||
}
|
||||
|
||||
# Upsert vector (same namespace+key = same ID = update)
|
||||
vector_id = self._make_vector_id(namespace, key)
|
||||
success = await self.qdrant.upsert_vector(
|
||||
collection_name=collection,
|
||||
vector_id=vector_id,
|
||||
vector=embedding,
|
||||
payload=payload
|
||||
)
|
||||
|
||||
if not success:
|
||||
raise ValueError("Failed to store volatile vector")
|
||||
|
||||
logger.debug(f"Stored volatile {namespace}:{key} with TTL {effective_ttl}s")
|
||||
|
||||
return VolatileRecordResponse(
|
||||
key=key,
|
||||
namespace=namespace,
|
||||
data=data,
|
||||
source=source,
|
||||
created_at=now,
|
||||
updated_at=now,
|
||||
ttl=effective_ttl,
|
||||
ttl_remaining=effective_ttl,
|
||||
refresh_schedule=refresh_schedule,
|
||||
user=user,
|
||||
)
|
||||
|
||||
async def search(
|
||||
self,
|
||||
user: str,
|
||||
query: str,
|
||||
limit: int = 5,
|
||||
score_threshold: float = 0.75
|
||||
) -> List[VolatileRecordResponse]:
|
||||
"""
|
||||
Semantic search across volatile data.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
query: Search query
|
||||
limit: Maximum results
|
||||
score_threshold: Minimum similarity score (higher = stricter)
|
||||
|
||||
Returns:
|
||||
List of matching volatile records
|
||||
"""
|
||||
collection = self._collection_name(user)
|
||||
|
||||
# Check if collection exists
|
||||
if not await self.qdrant.collection_exists(collection):
|
||||
return []
|
||||
|
||||
# Generate query embedding
|
||||
query_embedding = await self.ollama.embed(query)
|
||||
if not query_embedding:
|
||||
logger.error("Failed to embed query for volatile search")
|
||||
return []
|
||||
|
||||
# Search with expiry filter
|
||||
now_ms = self._current_timestamp_ms()
|
||||
results = await self.qdrant.search_with_expiry_filter(
|
||||
collection_name=collection,
|
||||
query_vector=query_embedding,
|
||||
current_timestamp=now_ms,
|
||||
limit=limit,
|
||||
score_threshold=score_threshold
|
||||
)
|
||||
|
||||
# Convert to response models
|
||||
responses = []
|
||||
for result in results:
|
||||
payload = result["payload"]
|
||||
ttl_expiry = payload.get("ttl_expiry", 0)
|
||||
ttl_remaining = max(0, (ttl_expiry - now_ms) // 1000)
|
||||
|
||||
responses.append(VolatileRecordResponse(
|
||||
key=payload["key"],
|
||||
namespace=payload["namespace"],
|
||||
data=payload.get("raw_data", {}),
|
||||
source=payload.get("source"),
|
||||
created_at=datetime.fromisoformat(payload["created_at"]),
|
||||
updated_at=datetime.fromisoformat(payload["updated_at"]),
|
||||
ttl=payload.get("ttl", 0),
|
||||
ttl_remaining=ttl_remaining,
|
||||
refresh_schedule=payload.get("refresh_schedule"),
|
||||
user=payload["user"],
|
||||
))
|
||||
|
||||
return responses
|
||||
|
||||
async def get(
|
||||
self,
|
||||
user: str,
|
||||
namespace: str,
|
||||
key: str
|
||||
) -> Optional[VolatileRecordResponse]:
|
||||
"""
|
||||
Get a specific volatile record by namespace and key.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
namespace: Data namespace
|
||||
key: Record key
|
||||
|
||||
Returns:
|
||||
Record if found and not expired, None otherwise
|
||||
"""
|
||||
# Use search with high threshold to find exact match
|
||||
query = self._to_natural_language(namespace, key, {"key": key})
|
||||
results = await self.search(user, query, limit=10, score_threshold=0.5)
|
||||
|
||||
# Find exact namespace+key match
|
||||
for result in results:
|
||||
if result.namespace == namespace and result.key == key:
|
||||
return result
|
||||
|
||||
return None
|
||||
|
||||
async def delete(
|
||||
self,
|
||||
user: str,
|
||||
namespace: str,
|
||||
key: str
|
||||
) -> bool:
|
||||
"""
|
||||
Delete a specific volatile record.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
namespace: Data namespace
|
||||
key: Record key
|
||||
|
||||
Returns:
|
||||
True if deleted, False if not found
|
||||
"""
|
||||
collection = self._collection_name(user)
|
||||
|
||||
if not await self.qdrant.collection_exists(collection):
|
||||
return False
|
||||
|
||||
vector_id = self._make_vector_id(namespace, key)
|
||||
|
||||
try:
|
||||
deleted = await self.qdrant.delete_by_ids(
|
||||
collection_name=collection,
|
||||
point_ids=[vector_id]
|
||||
)
|
||||
return deleted > 0
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to delete volatile {namespace}:{key}: {e}")
|
||||
return False
|
||||
|
||||
async def get_scheduled(
|
||||
self,
|
||||
user: str
|
||||
) -> List[VolatileRecordResponse]:
|
||||
"""
|
||||
Get all records with refresh schedules.
|
||||
|
||||
Used by scheduler to determine what needs refreshing.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
List of records with refresh_schedule set
|
||||
"""
|
||||
collection = self._collection_name(user)
|
||||
|
||||
if not await self.qdrant.collection_exists(collection):
|
||||
return []
|
||||
|
||||
now_ms = self._current_timestamp_ms()
|
||||
scheduled = []
|
||||
|
||||
# Scroll through all non-expired records
|
||||
try:
|
||||
all_points = await self.qdrant.scroll_all_points(
|
||||
collection_name=collection,
|
||||
with_payload=True
|
||||
)
|
||||
|
||||
for point in all_points:
|
||||
payload = point.get("payload", {})
|
||||
ttl_expiry = payload.get("ttl_expiry", 0)
|
||||
|
||||
# Skip expired
|
||||
if ttl_expiry <= now_ms:
|
||||
continue
|
||||
|
||||
# Only include if has refresh schedule
|
||||
if payload.get("refresh_schedule"):
|
||||
ttl_remaining = max(0, (ttl_expiry - now_ms) // 1000)
|
||||
scheduled.append(VolatileRecordResponse(
|
||||
key=payload["key"],
|
||||
namespace=payload["namespace"],
|
||||
data=payload.get("raw_data", {}),
|
||||
source=payload.get("source"),
|
||||
created_at=datetime.fromisoformat(payload["created_at"]),
|
||||
updated_at=datetime.fromisoformat(payload["updated_at"]),
|
||||
ttl=payload.get("ttl", 0),
|
||||
ttl_remaining=ttl_remaining,
|
||||
refresh_schedule=payload["refresh_schedule"],
|
||||
user=payload["user"],
|
||||
))
|
||||
|
||||
return scheduled
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to get scheduled volatile records: {e}")
|
||||
return []
|
||||
|
||||
async def get_stats(
|
||||
self,
|
||||
user: str
|
||||
) -> Dict[str, Any]:
|
||||
"""
|
||||
Get cache statistics for user.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Statistics dict
|
||||
"""
|
||||
collection = self._collection_name(user)
|
||||
|
||||
if not await self.qdrant.collection_exists(collection):
|
||||
return {
|
||||
"total_records": 0,
|
||||
"by_namespace": {},
|
||||
"scheduled_count": 0,
|
||||
"expired_count": 0,
|
||||
}
|
||||
|
||||
now_ms = self._current_timestamp_ms()
|
||||
by_namespace: Dict[str, int] = {}
|
||||
total = 0
|
||||
scheduled = 0
|
||||
expired = 0
|
||||
|
||||
try:
|
||||
all_points = await self.qdrant.scroll_all_points(
|
||||
collection_name=collection,
|
||||
with_payload=True
|
||||
)
|
||||
|
||||
for point in all_points:
|
||||
payload = point.get("payload", {})
|
||||
namespace = payload.get("namespace", "unknown")
|
||||
ttl_expiry = payload.get("ttl_expiry", 0)
|
||||
|
||||
if ttl_expiry <= now_ms:
|
||||
expired += 1
|
||||
else:
|
||||
total += 1
|
||||
by_namespace[namespace] = by_namespace.get(namespace, 0) + 1
|
||||
if payload.get("refresh_schedule"):
|
||||
scheduled += 1
|
||||
|
||||
return {
|
||||
"total_records": total,
|
||||
"by_namespace": by_namespace,
|
||||
"scheduled_count": scheduled,
|
||||
"expired_count": expired,
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to get volatile stats: {e}")
|
||||
return {
|
||||
"total_records": 0,
|
||||
"by_namespace": {},
|
||||
"scheduled_count": 0,
|
||||
"expired_count": 0,
|
||||
}
|
||||
|
||||
async def purge_expired(
|
||||
self,
|
||||
user: str
|
||||
) -> int:
|
||||
"""
|
||||
Purge all expired volatile records for user.
|
||||
|
||||
Args:
|
||||
user: User identifier
|
||||
|
||||
Returns:
|
||||
Number of records purged
|
||||
"""
|
||||
collection = self._collection_name(user)
|
||||
|
||||
if not await self.qdrant.collection_exists(collection):
|
||||
return 0
|
||||
|
||||
now_ms = self._current_timestamp_ms()
|
||||
return await self.qdrant.delete_expired_vectors(collection, now_ms)
|
||||
|
||||
async def purge_all_expired(self) -> Dict[str, int]:
|
||||
"""
|
||||
Purge expired records from all volatile collections.
|
||||
|
||||
Returns:
|
||||
Dict of collection -> purged count
|
||||
"""
|
||||
collections = await self.qdrant.get_volatile_collections()
|
||||
results = {}
|
||||
now_ms = self._current_timestamp_ms()
|
||||
|
||||
for collection in collections:
|
||||
purged = await self.qdrant.delete_expired_vectors(collection, now_ms)
|
||||
if purged > 0:
|
||||
results[collection] = purged
|
||||
logger.info(f"Purged {purged} expired from {collection}")
|
||||
|
||||
return results
|
||||
@@ -1,7 +1,7 @@
|
||||
"""
|
||||
Intelligent Wiki Page Writer Service
|
||||
|
||||
Uses LLM (mistral-nemo) to create and reconstruct wiki pages with:
|
||||
Uses LLM to create and reconstruct wiki pages with:
|
||||
- Holistic content restructuring
|
||||
- Zero fact loss (unless superseded)
|
||||
- Conflict detection and flagging
|
||||
@@ -25,15 +25,16 @@ class WikiPageWriter:
|
||||
Intelligent wiki page writer using LLM for content generation and restructuring.
|
||||
"""
|
||||
|
||||
def __init__(self, ollama_client):
|
||||
def __init__(self, ollama_client, settings):
|
||||
"""
|
||||
Initialize wiki page writer.
|
||||
|
||||
Args:
|
||||
ollama_client: OllamaClient for LLM operations
|
||||
settings: Application settings
|
||||
"""
|
||||
self.ollama = ollama_client
|
||||
self.model = "mistral-nemo" # Default model for writing
|
||||
self.model = settings.ollama_model
|
||||
|
||||
async def create_page(
|
||||
self,
|
||||
@@ -130,8 +131,8 @@ class WikiPageWriter:
|
||||
conflicts=conflicts
|
||||
)
|
||||
|
||||
# Reconstruct with LLM
|
||||
reconstructed = await self._call_llm(prompt)
|
||||
# Reconstruct with LLM (lower temperature for precise merging)
|
||||
reconstructed = await self._call_llm(prompt, temperature=0.2)
|
||||
|
||||
# Ensure standard sections are present
|
||||
reconstructed = self._ensure_standard_sections(
|
||||
@@ -154,7 +155,7 @@ class WikiPageWriter:
|
||||
Returns:
|
||||
List of conflicts with: {fact_a, fact_b, confidence, context}
|
||||
"""
|
||||
prompt = f"""Analyze these two pieces of content for factual conflicts.
|
||||
prompt = f"""Analyze these contents for direct factual conflicts.
|
||||
|
||||
EXISTING CONTENT:
|
||||
{existing_content[:2000]}
|
||||
@@ -162,25 +163,26 @@ EXISTING CONTENT:
|
||||
NEW INFORMATION:
|
||||
{new_information[:2000]}
|
||||
|
||||
Identify any facts that contradict each other. For each conflict, provide:
|
||||
1. The fact from existing content
|
||||
2. The contradicting fact from new information
|
||||
3. Confidence level (low/medium/high)
|
||||
4. Context/explanation
|
||||
ANALYSIS STEPS:
|
||||
1. Identify specific factual claims in existing content (dates, numbers, names, states)
|
||||
2. Identify specific factual claims in new content
|
||||
3. Compare ONLY for direct contradictions (X says A, Y says not-A)
|
||||
|
||||
Return ONLY valid JSON:
|
||||
RULES:
|
||||
- Do NOT flag differences in wording or phrasing as conflicts
|
||||
- Do NOT flag new/additional information as conflicts
|
||||
- Do NOT flag opinion differences as conflicts
|
||||
- ONLY flag direct factual contradictions
|
||||
- Return valid JSON only, no commentary
|
||||
|
||||
Return format:
|
||||
{{
|
||||
"conflicts": [
|
||||
{{
|
||||
"existing_fact": "fact from old content",
|
||||
"new_fact": "contradicting fact",
|
||||
"confidence": "medium",
|
||||
"context": "explanation of why these conflict"
|
||||
}}
|
||||
{{"existing_fact": "...", "new_fact": "...", "confidence": "low/medium/high", "context": "..."}}
|
||||
]
|
||||
}}
|
||||
|
||||
If no conflicts, return: {{"conflicts": []}}
|
||||
If no conflicts: {{"conflicts": []}}
|
||||
|
||||
JSON:"""
|
||||
|
||||
@@ -188,7 +190,8 @@ JSON:"""
|
||||
response = await self.ollama.generate_text(
|
||||
prompt=prompt,
|
||||
model=self.model,
|
||||
stream=False
|
||||
stream=False,
|
||||
temperature=0.0 # Deterministic for consistent conflict detection
|
||||
)
|
||||
|
||||
# Extract JSON
|
||||
@@ -347,6 +350,13 @@ FORMATTING RULES:
|
||||
- Keep sections focused and scannable
|
||||
- Adapt structure to content - not all sections apply to all topics
|
||||
|
||||
CRITICAL CONSTRAINTS:
|
||||
- Do NOT invent facts not present in the source information above
|
||||
- Do NOT add speculative information or assumptions
|
||||
- Do NOT fill sections with placeholder text or generic statements
|
||||
- If information for a section is not available, OMIT the section entirely
|
||||
- Base ALL content strictly on provided source information
|
||||
|
||||
Generate ONLY the markdown content (do not include Sources, Knowledge Graph, or Mind Map sections - those are added automatically).
|
||||
|
||||
MARKDOWN:"""
|
||||
@@ -402,6 +412,13 @@ FORMATTING RULES:
|
||||
- Bold important terms
|
||||
- Add subsections (###) where it improves clarity
|
||||
|
||||
CRITICAL CONSTRAINTS:
|
||||
- Do NOT rephrase facts in ways that change their meaning
|
||||
- Do NOT remove ANY information unless explicitly superseded by newer facts
|
||||
- Do NOT add information not present in existing content or new information
|
||||
- Preserve exact quotes, dates, numbers, and names verbatim
|
||||
- Do NOT fill gaps with assumptions or general knowledge
|
||||
|
||||
OUTPUT INSTRUCTIONS:
|
||||
- Return complete page content (do not include Sources, Knowledge Graph, Mind Map - those are added automatically)
|
||||
- Include updated "Changes & Updates" section noting what was changed today
|
||||
@@ -409,13 +426,21 @@ OUTPUT INSTRUCTIONS:
|
||||
|
||||
RECONSTRUCTED MARKDOWN:"""
|
||||
|
||||
async def _call_llm(self, prompt: str) -> str:
|
||||
"""Call LLM with prompt and return response."""
|
||||
async def _call_llm(self, prompt: str, temperature: float = 0.3) -> str:
|
||||
"""
|
||||
Call LLM with prompt and return response.
|
||||
|
||||
Args:
|
||||
prompt: The prompt text
|
||||
temperature: Sampling temperature (0.0=deterministic, higher=creative)
|
||||
Default 0.3 for controlled but natural content generation
|
||||
"""
|
||||
try:
|
||||
response = await self.ollama.generate_text(
|
||||
prompt=prompt,
|
||||
model=self.model,
|
||||
stream=False
|
||||
stream=False,
|
||||
temperature=temperature
|
||||
)
|
||||
|
||||
if not response:
|
||||
|
||||
@@ -431,3 +431,166 @@ class WikiService:
|
||||
WikiPageList filtered by dossier tag
|
||||
"""
|
||||
return await self.list_pages(user, tag=dossier_name, limit=limit)
|
||||
|
||||
async def smart_create_page(
|
||||
self,
|
||||
topic: str,
|
||||
user: str,
|
||||
path: Optional[str],
|
||||
tags: List[str],
|
||||
hybrid_rag_service: "HybridRAGService",
|
||||
wiki_page_writer: "WikiPageWriter",
|
||||
include_web: bool = True,
|
||||
include_wiki: bool = True
|
||||
) -> tuple["WikiPage", Dict[str, Any]]:
|
||||
"""
|
||||
Create wiki page with research from HybridRAG.
|
||||
|
||||
This method combines research + content generation + page creation:
|
||||
1. Run HybridRAG search on topic
|
||||
2. Format results for WikiPageWriter
|
||||
3. Generate page content with LLM
|
||||
4. Create page in Wiki.js
|
||||
5. Return page + research summary
|
||||
|
||||
Args:
|
||||
topic: Topic to research and create page about
|
||||
user: User identifier
|
||||
path: Optional page path (auto-generated from topic if not provided)
|
||||
tags: Tags for the page
|
||||
hybrid_rag_service: HybridRAG service for multi-source search
|
||||
wiki_page_writer: WikiPageWriter for LLM content generation
|
||||
include_web: Include web search results
|
||||
include_wiki: Include existing wiki knowledge
|
||||
|
||||
Returns:
|
||||
Tuple of (created WikiPage, research summary dict)
|
||||
"""
|
||||
from src.models.hybrid_rag import HybridRAGConfig
|
||||
|
||||
logger.info(f"Smart create page: topic='{topic}', user='{user}'")
|
||||
|
||||
# Step 1: Run HybridRAG search on the topic
|
||||
config = HybridRAGConfig(
|
||||
enable_vector=include_wiki,
|
||||
enable_graph=include_wiki,
|
||||
enable_web=include_web,
|
||||
enable_reranking=True,
|
||||
enable_enrichment=True,
|
||||
final_result_count=15 # Get more results for rich content
|
||||
)
|
||||
|
||||
search_response = await hybrid_rag_service.search(
|
||||
query=topic,
|
||||
user=user,
|
||||
config=config
|
||||
)
|
||||
|
||||
logger.info(
|
||||
f"HybridRAG search completed: {search_response.total_results} results, "
|
||||
f"search_id={search_response.search_id}"
|
||||
)
|
||||
|
||||
# Step 2: Format results for WikiPageWriter
|
||||
source_information = []
|
||||
wiki_results_count = 0
|
||||
web_results_count = 0
|
||||
graph_entities_count = 0
|
||||
|
||||
for result in search_response.results:
|
||||
source_type = result.source_type
|
||||
|
||||
if "web" in source_type:
|
||||
web_results_count += 1
|
||||
source_information.append({
|
||||
"title": result.title,
|
||||
"url": result.url or "",
|
||||
"content": result.content[:500] if result.content else ""
|
||||
})
|
||||
elif "vector" in source_type or "graph" in source_type:
|
||||
wiki_results_count += 1
|
||||
# For wiki results, use page path as URL
|
||||
source_information.append({
|
||||
"title": result.title,
|
||||
"url": f"/{result.page_path}" if result.page_path else "",
|
||||
"content": result.content[:500] if result.content else ""
|
||||
})
|
||||
|
||||
# Count entities from related dossiers
|
||||
if result.related_dossiers:
|
||||
graph_entities_count += len(result.related_dossiers)
|
||||
|
||||
# Step 3: Generate page content with LLM
|
||||
# Use topic as summary and let WikiPageWriter create structured content
|
||||
topic_summary = f"Research findings about: {topic}"
|
||||
if search_response.keywords:
|
||||
topic_summary += f"\n\nKey concepts: {', '.join(search_response.keywords.core_keywords)}"
|
||||
|
||||
# Extract entities from search results for knowledge graph linking
|
||||
entities = []
|
||||
if search_response.keywords and search_response.keywords.core_keywords:
|
||||
entities = search_response.keywords.core_keywords[:10]
|
||||
|
||||
# Get related documents for cross-linking
|
||||
related_docs = []
|
||||
for result in search_response.results[:5]:
|
||||
if result.page_path:
|
||||
related_docs.append(f"[{result.title}](/{result.page_path})")
|
||||
|
||||
content = await wiki_page_writer.create_page(
|
||||
title=topic,
|
||||
topic_summary=topic_summary,
|
||||
source_information=source_information[:10], # Limit sources
|
||||
entities=entities,
|
||||
related_docs=related_docs
|
||||
)
|
||||
|
||||
logger.info(f"Generated page content: {len(content)} characters")
|
||||
|
||||
# Step 4: Auto-generate path from topic if not provided
|
||||
if not path:
|
||||
# Convert topic to kebab-case path
|
||||
import re
|
||||
path_slug = topic.lower()
|
||||
path_slug = re.sub(r'[^\w\s-]', '', path_slug) # Remove special chars
|
||||
path_slug = re.sub(r'\s+', '-', path_slug) # Spaces to hyphens
|
||||
path_slug = re.sub(r'-+', '-', path_slug) # Multiple hyphens to single
|
||||
path_slug = path_slug.strip('-')
|
||||
|
||||
# Infer category from tags or use reference
|
||||
category = "reference"
|
||||
if tags:
|
||||
category = tags[0].lower()
|
||||
|
||||
path = f"/{category}/{path_slug}"
|
||||
|
||||
# Step 5: Create page using existing create_page method
|
||||
from src.models.wiki import WikiPageCreate
|
||||
|
||||
page_data = WikiPageCreate(
|
||||
title=topic,
|
||||
path=path,
|
||||
content=content,
|
||||
description=f"Research summary about {topic}",
|
||||
tags=tags,
|
||||
user=user
|
||||
)
|
||||
|
||||
page = await self.create_page(page_data)
|
||||
|
||||
logger.info(f"Created page: id={page.id}, path={page.path}")
|
||||
|
||||
# Build research summary
|
||||
research_summary = {
|
||||
"wiki_results": wiki_results_count,
|
||||
"web_results": web_results_count,
|
||||
"graph_entities": graph_entities_count,
|
||||
"keywords_extracted": len(search_response.keywords.core_keywords) if search_response.keywords else 0,
|
||||
"timing_ms": search_response.timing.total_ms if search_response.timing else 0
|
||||
}
|
||||
|
||||
return page, {
|
||||
"research_summary": research_summary,
|
||||
"sources_used": len(source_information),
|
||||
"search_id": search_response.search_id
|
||||
}
|
||||
|
||||
+20
-9
@@ -1,5 +1,6 @@
|
||||
"""Pytest configuration and shared fixtures for Library Desk tests."""
|
||||
|
||||
import os
|
||||
import pytest
|
||||
import pytest_asyncio
|
||||
from typing import AsyncGenerator
|
||||
@@ -7,6 +8,9 @@ from typing import AsyncGenerator
|
||||
# Test configuration
|
||||
pytest_plugins = ("pytest_asyncio",)
|
||||
|
||||
# Use real host for tests (services available at this IP)
|
||||
TEST_HOST = os.environ.get("TEST_HOST", "192.168.86.149")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def test_user() -> str:
|
||||
@@ -17,49 +21,56 @@ def test_user() -> str:
|
||||
@pytest.fixture
|
||||
def neo4j_test_uri() -> str:
|
||||
"""Test Neo4j URI."""
|
||||
return "bolt://neo4j:7687"
|
||||
return f"bolt://{TEST_HOST}:7687"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def neo4j_test_auth() -> tuple:
|
||||
"""Test Neo4j authentication."""
|
||||
return ("neo4j", "test_password")
|
||||
from src.config import get_settings
|
||||
settings = get_settings()
|
||||
return ("neo4j", settings.neo4j_password)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def qdrant_test_url() -> str:
|
||||
"""Test Qdrant URL."""
|
||||
return "http://qdrant:6333"
|
||||
return f"http://{TEST_HOST}:6333"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def wikijs_test_config() -> dict:
|
||||
"""Test Wiki.js configuration."""
|
||||
from src.config import get_settings
|
||||
settings = get_settings()
|
||||
return {
|
||||
"base_url": "http://wiki:3000",
|
||||
"api_key": "test_api_key"
|
||||
"base_url": f"http://{TEST_HOST}:3000",
|
||||
"username": settings.wikijs_username,
|
||||
"password": settings.wikijs_password
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def searxng_test_url() -> str:
|
||||
"""Test SearXNG URL."""
|
||||
return "http://searxng:8080"
|
||||
return f"http://{TEST_HOST}:8080"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ollama_test_config() -> dict:
|
||||
"""Test Ollama configuration."""
|
||||
from src.config import get_settings
|
||||
settings = get_settings()
|
||||
return {
|
||||
"base_url": "http://ollama:11434",
|
||||
"model": "nomic-embed-text"
|
||||
"base_url": f"http://{TEST_HOST}:11434",
|
||||
"model": settings.ollama_embedding_model
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def redis_test_url() -> str:
|
||||
"""Test Redis URL."""
|
||||
return "redis://redis-shared:6379/4"
|
||||
return f"redis://{TEST_HOST}:6379/4"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
|
||||
@@ -0,0 +1,183 @@
|
||||
"""Tests for ContentExtractor client."""
|
||||
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
from src.clients.content_extractor import ContentExtractor
|
||||
from src.models.content import ContentExtractionResult
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def content_extractor():
|
||||
"""Create ContentExtractor with test configuration."""
|
||||
return ContentExtractor(timeout=5, max_length=2000)
|
||||
|
||||
|
||||
class TestContentExtractor:
|
||||
"""Tests for ContentExtractor client."""
|
||||
|
||||
def test_init(self, content_extractor):
|
||||
"""Test ContentExtractor initialization."""
|
||||
assert content_extractor.timeout == 5
|
||||
assert content_extractor.max_length == 2000
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_extract_success(self, content_extractor):
|
||||
"""Test successful content extraction."""
|
||||
test_url = "https://example.com/article"
|
||||
test_content = "This is the extracted article content."
|
||||
|
||||
with patch('src.clients.content_extractor.trafilatura') as mock_traf:
|
||||
mock_traf.fetch_url.return_value = "<html><body>Test</body></html>"
|
||||
mock_traf.extract.return_value = test_content
|
||||
mock_traf.bare_extraction.return_value = {
|
||||
"title": "Test Article",
|
||||
"author": "John Doe",
|
||||
"date": "2024-01-15",
|
||||
"language": "en"
|
||||
}
|
||||
|
||||
result = await content_extractor.extract(test_url)
|
||||
|
||||
assert result.success is True
|
||||
assert result.url == test_url
|
||||
assert result.content == test_content
|
||||
assert result.error is None
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_extract_fetch_failure(self, content_extractor):
|
||||
"""Test extraction when URL fetch fails."""
|
||||
test_url = "https://example.com/nonexistent"
|
||||
|
||||
with patch('src.clients.content_extractor.trafilatura') as mock_traf:
|
||||
mock_traf.fetch_url.return_value = None
|
||||
|
||||
result = await content_extractor.extract(test_url)
|
||||
|
||||
assert result.success is False
|
||||
assert result.url == test_url
|
||||
assert result.content == ""
|
||||
assert "Failed to fetch URL" in result.error
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_extract_no_content(self, content_extractor):
|
||||
"""Test extraction when page has no extractable content."""
|
||||
test_url = "https://example.com/empty"
|
||||
|
||||
with patch('src.clients.content_extractor.trafilatura') as mock_traf:
|
||||
mock_traf.fetch_url.return_value = "<html><body></body></html>"
|
||||
mock_traf.extract.return_value = None
|
||||
|
||||
result = await content_extractor.extract(test_url)
|
||||
|
||||
assert result.success is False
|
||||
assert "No content extracted" in result.error
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_extract_max_length_truncation(self, content_extractor):
|
||||
"""Test that content is truncated to max length."""
|
||||
test_url = "https://example.com/long-article"
|
||||
# Content longer than max_length (2000)
|
||||
long_content = "x" * 3000
|
||||
|
||||
with patch('src.clients.content_extractor.trafilatura') as mock_traf:
|
||||
mock_traf.fetch_url.return_value = "<html><body>Test</body></html>"
|
||||
mock_traf.extract.return_value = long_content
|
||||
mock_traf.bare_extraction.return_value = {}
|
||||
|
||||
result = await content_extractor.extract(test_url)
|
||||
|
||||
assert result.success is True
|
||||
assert len(result.content) <= content_extractor.max_length + 3 # +3 for "..."
|
||||
assert result.content.endswith("...")
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_extract_batch(self, content_extractor):
|
||||
"""Test batch extraction of multiple URLs."""
|
||||
test_urls = [
|
||||
"https://example.com/article1",
|
||||
"https://example.com/article2",
|
||||
"https://example.com/article3"
|
||||
]
|
||||
|
||||
with patch('src.clients.content_extractor.trafilatura') as mock_traf:
|
||||
mock_traf.fetch_url.return_value = "<html><body>Test</body></html>"
|
||||
mock_traf.extract.return_value = "Extracted content"
|
||||
mock_traf.bare_extraction.return_value = {}
|
||||
|
||||
results = await content_extractor.extract_batch(test_urls)
|
||||
|
||||
assert len(results) == 3
|
||||
for i, result in enumerate(results):
|
||||
assert result.url == test_urls[i]
|
||||
assert result.success is True
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_extract_timeout(self):
|
||||
"""Test extraction timeout handling."""
|
||||
import time
|
||||
|
||||
test_url = "https://example.com/slow"
|
||||
|
||||
# Create an extractor with very short timeout
|
||||
fast_extractor = ContentExtractor(timeout=0.001, max_length=2000)
|
||||
|
||||
def slow_fetch(url):
|
||||
time.sleep(1) # Sleep synchronously (this runs in thread pool)
|
||||
return "<html></html>"
|
||||
|
||||
with patch('src.clients.content_extractor.trafilatura') as mock_traf:
|
||||
mock_traf.fetch_url = slow_fetch
|
||||
|
||||
result = await fast_extractor.extract(test_url)
|
||||
|
||||
assert result.success is False
|
||||
assert "timed out" in result.error.lower()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_extract_from_html(self, content_extractor):
|
||||
"""Test extraction from raw HTML."""
|
||||
test_html = "<html><body><article>Article content here.</article></body></html>"
|
||||
|
||||
with patch('src.clients.content_extractor.trafilatura') as mock_traf:
|
||||
mock_traf.extract.return_value = "Article content here."
|
||||
mock_traf.bare_extraction.return_value = {"title": "Test"}
|
||||
|
||||
result = await content_extractor.extract_from_html(test_html, url="https://example.com")
|
||||
|
||||
assert result.success is True
|
||||
assert result.content == "Article content here."
|
||||
|
||||
|
||||
class TestContentExtractionResult:
|
||||
"""Tests for ContentExtractionResult model."""
|
||||
|
||||
def test_success_result(self):
|
||||
"""Test creating a successful result."""
|
||||
result = ContentExtractionResult(
|
||||
url="https://example.com",
|
||||
title="Test Article",
|
||||
content="Article content",
|
||||
author="John Doe",
|
||||
date="2024-01-15",
|
||||
language="en",
|
||||
success=True,
|
||||
error=None
|
||||
)
|
||||
|
||||
assert result.url == "https://example.com"
|
||||
assert result.success is True
|
||||
assert result.error is None
|
||||
|
||||
def test_failure_result(self):
|
||||
"""Test creating a failure result."""
|
||||
result = ContentExtractionResult(
|
||||
url="https://example.com/error",
|
||||
content="",
|
||||
success=False,
|
||||
error="Failed to fetch URL"
|
||||
)
|
||||
|
||||
assert result.url == "https://example.com/error"
|
||||
assert result.success is False
|
||||
assert result.error == "Failed to fetch URL"
|
||||
@@ -38,10 +38,10 @@ def settings():
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def neo4j_client(settings) -> AsyncGenerator[Neo4jClient, None]:
|
||||
async def neo4j_client(settings, neo4j_test_uri) -> AsyncGenerator[Neo4jClient, None]:
|
||||
"""Get connected Neo4j client."""
|
||||
client = Neo4jClient(
|
||||
uri=settings.neo4j_uri,
|
||||
uri=neo4j_test_uri,
|
||||
user=settings.neo4j_user,
|
||||
password=settings.neo4j_password
|
||||
)
|
||||
@@ -51,12 +51,12 @@ async def neo4j_client(settings) -> AsyncGenerator[Neo4jClient, None]:
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def wiki_client(settings) -> AsyncGenerator[WikiJSClient, None]:
|
||||
async def wiki_client(wikijs_test_config) -> AsyncGenerator[WikiJSClient, None]:
|
||||
"""Get Wiki.js client."""
|
||||
client = WikiJSClient(
|
||||
base_url=settings.wikijs_url,
|
||||
username=settings.wikijs_username,
|
||||
password=settings.wikijs_password
|
||||
base_url=wikijs_test_config["base_url"],
|
||||
username=wikijs_test_config["username"],
|
||||
password=wikijs_test_config["password"]
|
||||
)
|
||||
yield client
|
||||
|
||||
|
||||
+65
-49
@@ -25,6 +25,7 @@ from src.clients.qdrant_client import QdrantClientWrapper
|
||||
from src.clients.wikijs_client import WikiJSClient
|
||||
from src.clients.searxng_client import SearXNGClient
|
||||
from src.clients.ollama_client import OllamaClient
|
||||
from src.clients.content_extractor import ContentExtractor
|
||||
from src.services.hybrid_rag_service import HybridRAGService
|
||||
from src.services.vector_service import VectorService
|
||||
from src.services.graph_service import GraphService
|
||||
@@ -42,10 +43,10 @@ def settings():
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def neo4j_client(settings) -> AsyncGenerator[Neo4jClient, None]:
|
||||
async def neo4j_client(settings, neo4j_test_uri) -> AsyncGenerator[Neo4jClient, None]:
|
||||
"""Get connected Neo4j client."""
|
||||
client = Neo4jClient(
|
||||
uri=settings.neo4j_uri,
|
||||
uri=neo4j_test_uri,
|
||||
user=settings.neo4j_user,
|
||||
password=settings.neo4j_password
|
||||
)
|
||||
@@ -55,32 +56,44 @@ async def neo4j_client(settings) -> AsyncGenerator[Neo4jClient, None]:
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def qdrant_client(settings) -> QdrantClientWrapper:
|
||||
def qdrant_client(qdrant_test_url) -> QdrantClientWrapper:
|
||||
"""Get Qdrant client."""
|
||||
return QdrantClientWrapper(url=settings.qdrant_url)
|
||||
return QdrantClientWrapper(url=qdrant_test_url)
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def wiki_client(settings) -> AsyncGenerator[WikiJSClient, None]:
|
||||
async def wiki_client(wikijs_test_config) -> AsyncGenerator[WikiJSClient, None]:
|
||||
"""Get Wiki.js client."""
|
||||
client = WikiJSClient(
|
||||
base_url=settings.wikijs_url,
|
||||
username=settings.wikijs_username,
|
||||
password=settings.wikijs_password
|
||||
base_url=wikijs_test_config["base_url"],
|
||||
username=wikijs_test_config["username"],
|
||||
password=wikijs_test_config["password"]
|
||||
)
|
||||
yield client
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def searxng_client(settings) -> SearXNGClient:
|
||||
def searxng_client(searxng_test_url) -> SearXNGClient:
|
||||
"""Get SearXNG client."""
|
||||
return SearXNGClient(base_url=settings.searxng_url)
|
||||
return SearXNGClient(base_url=searxng_test_url)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ollama_client(settings) -> OllamaClient:
|
||||
def ollama_client(ollama_test_config) -> OllamaClient:
|
||||
"""Get Ollama client."""
|
||||
return OllamaClient(base_url=settings.ollama_url)
|
||||
return OllamaClient(
|
||||
base_url=ollama_test_config["base_url"],
|
||||
model=ollama_test_config["model"]
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def content_extractor(settings) -> ContentExtractor:
|
||||
"""Get ContentExtractor client."""
|
||||
return ContentExtractor(
|
||||
timeout=settings.content_extraction_timeout,
|
||||
max_length=settings.content_max_length
|
||||
)
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
@@ -101,6 +114,7 @@ async def hybrid_rag_service(
|
||||
graph_service,
|
||||
searxng_client,
|
||||
ollama_client,
|
||||
content_extractor,
|
||||
settings
|
||||
):
|
||||
"""Get HybridRAGService instance."""
|
||||
@@ -109,6 +123,7 @@ async def hybrid_rag_service(
|
||||
graph_service=graph_service,
|
||||
searxng_client=searxng_client,
|
||||
ollama_client=ollama_client,
|
||||
content_extractor=content_extractor,
|
||||
settings=settings
|
||||
)
|
||||
|
||||
@@ -207,53 +222,54 @@ async def test_vector_data(vector_service, test_wiki_page):
|
||||
# ============================================================================
|
||||
|
||||
class TestRRFFusion:
|
||||
"""Test Reciprocal Rank Fusion algorithm."""
|
||||
"""Test two-stage Reciprocal Rank Fusion algorithm."""
|
||||
|
||||
def test_rrf_single_source(self, hybrid_rag_service):
|
||||
"""Test RRF with single source."""
|
||||
results_by_source = {
|
||||
"vector": [
|
||||
{"page_id": 1, "title": "Doc 1", "content": "test"},
|
||||
{"page_id": 2, "title": "Doc 2", "content": "test"}
|
||||
]
|
||||
}
|
||||
def test_wiki_merge_single_source(self, hybrid_rag_service):
|
||||
"""Test wiki merge with single source (vector only)."""
|
||||
vector_results = [
|
||||
{"page_id": 1, "title": "Doc 1", "content": "test"},
|
||||
{"page_id": 2, "title": "Doc 2", "content": "test"}
|
||||
]
|
||||
|
||||
fused = hybrid_rag_service._reciprocal_rank_fusion(results_by_source, k=60)
|
||||
merged = hybrid_rag_service._merge_wiki_sources(vector_results, [], k=60)
|
||||
|
||||
assert len(fused) == 2
|
||||
assert fused[0]["rrf_score"] > fused[1]["rrf_score"] # Rank 1 > Rank 2
|
||||
assert fused[0]["sources"] == ["vector"]
|
||||
assert len(merged) == 2
|
||||
assert merged[0]["wiki_rrf_score"] > merged[1]["wiki_rrf_score"] # Rank 1 > Rank 2
|
||||
assert merged[0]["found_by"] == ["vector"]
|
||||
|
||||
def test_rrf_multiple_sources_same_doc(self, hybrid_rag_service):
|
||||
"""Test RRF with same document from multiple sources."""
|
||||
results_by_source = {
|
||||
"vector": [{"page_id": 1, "title": "Doc 1", "content": "test"}],
|
||||
"graph": [{"page_id": 1, "title": "Doc 1", "content": ""}],
|
||||
}
|
||||
def test_wiki_merge_multiple_sources_same_doc(self, hybrid_rag_service):
|
||||
"""Test wiki merge with same document from vector and graph."""
|
||||
vector_results = [{"page_id": 1, "title": "Doc 1", "content": "test"}]
|
||||
graph_results = [{"page_id": 1, "title": "Doc 1", "content": ""}]
|
||||
|
||||
fused = hybrid_rag_service._reciprocal_rank_fusion(results_by_source, k=60)
|
||||
merged = hybrid_rag_service._merge_wiki_sources(vector_results, graph_results, k=60)
|
||||
|
||||
assert len(fused) == 1 # Deduplicated
|
||||
assert len(fused[0]["sources"]) == 2 # Both sources
|
||||
assert "vector" in fused[0]["sources"]
|
||||
assert "graph" in fused[0]["sources"]
|
||||
# RRF score should be sum: 1/(60+1) + 1/(60+1)
|
||||
assert len(merged) == 1 # Deduplicated
|
||||
assert len(merged[0]["found_by"]) == 2 # Both sources
|
||||
assert "vector" in merged[0]["found_by"]
|
||||
assert "graph" in merged[0]["found_by"]
|
||||
# Wiki RRF score should be sum: 1/(60+1) + 1/(60+1)
|
||||
expected_score = 1/61 + 1/61
|
||||
assert abs(fused[0]["rrf_score"] - expected_score) < 0.001
|
||||
assert abs(merged[0]["wiki_rrf_score"] - expected_score) < 0.001
|
||||
|
||||
def test_rrf_web_results(self, hybrid_rag_service):
|
||||
"""Test RRF with web results (URL-based)."""
|
||||
results_by_source = {
|
||||
"web": [
|
||||
{"url": "https://example.com/1", "title": "Web 1", "content": "test"},
|
||||
{"url": "https://example.com/2", "title": "Web 2", "content": "test"}
|
||||
]
|
||||
}
|
||||
def test_final_rrf_wiki_and_web(self, hybrid_rag_service):
|
||||
"""Test final RRF between wiki and web results."""
|
||||
# Pre-merged wiki results
|
||||
wiki_results = [
|
||||
{"page_id": 1, "title": "Wiki 1", "content": "test", "found_by": ["vector"]}
|
||||
]
|
||||
web_results = [
|
||||
{"url": "https://example.com/1", "title": "Web 1", "content": "test"},
|
||||
{"url": "https://example.com/2", "title": "Web 2", "content": "test"}
|
||||
]
|
||||
|
||||
fused = hybrid_rag_service._reciprocal_rank_fusion(results_by_source, k=60)
|
||||
fused = hybrid_rag_service._reciprocal_rank_fusion(wiki_results, web_results, k=60)
|
||||
|
||||
assert len(fused) == 2
|
||||
assert fused[0]["result"]["url"] == "https://example.com/1"
|
||||
assert len(fused) == 3
|
||||
# Wiki rank 1 and web rank 1 should have same RRF score
|
||||
wiki_score = next(r["rrf_score"] for r in fused if r["source_type"] == "wiki")
|
||||
web_score = next(r["rrf_score"] for r in fused if r["source_type"] == "web")
|
||||
assert abs(wiki_score - web_score) < 0.001 # Equal footing
|
||||
|
||||
|
||||
class TestContextFormatting:
|
||||
|
||||
+16
-15
@@ -30,10 +30,10 @@ def settings():
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def neo4j_client(settings) -> AsyncGenerator[Neo4jClient, None]:
|
||||
async def neo4j_client(settings, neo4j_test_uri) -> AsyncGenerator[Neo4jClient, None]:
|
||||
"""Get connected Neo4j client."""
|
||||
client = Neo4jClient(
|
||||
uri=settings.neo4j_uri,
|
||||
uri=neo4j_test_uri,
|
||||
user=settings.neo4j_user,
|
||||
password=settings.neo4j_password
|
||||
)
|
||||
@@ -43,45 +43,46 @@ async def neo4j_client(settings) -> AsyncGenerator[Neo4jClient, None]:
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def qdrant_client(settings) -> QdrantClientWrapper:
|
||||
def qdrant_client(settings, qdrant_test_url) -> QdrantClientWrapper:
|
||||
"""Get Qdrant client."""
|
||||
return QdrantClientWrapper(url=settings.qdrant_url)
|
||||
return QdrantClientWrapper(url=qdrant_test_url)
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def wikijs_client(settings) -> AsyncGenerator[WikiJSClient, None]:
|
||||
async def wikijs_client(wikijs_test_config) -> AsyncGenerator[WikiJSClient, None]:
|
||||
"""Get Wiki.js client."""
|
||||
client = WikiJSClient(
|
||||
base_url=settings.wikijs_url,
|
||||
api_key=settings.wikijs_api_key
|
||||
base_url=wikijs_test_config["base_url"],
|
||||
username=wikijs_test_config["username"],
|
||||
password=wikijs_test_config["password"]
|
||||
)
|
||||
yield client
|
||||
await client.close()
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def searxng_client(settings) -> AsyncGenerator[SearXNGClient, None]:
|
||||
async def searxng_client(searxng_test_url) -> AsyncGenerator[SearXNGClient, None]:
|
||||
"""Get SearXNG client."""
|
||||
client = SearXNGClient(base_url=settings.searxng_url)
|
||||
client = SearXNGClient(base_url=searxng_test_url)
|
||||
yield client
|
||||
await client.close()
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def ollama_client(settings) -> AsyncGenerator[OllamaClient, None]:
|
||||
"""Get Ollama client."""
|
||||
async def ollama_client(ollama_test_config) -> AsyncGenerator[OllamaClient, None]:
|
||||
"""Get Ollama client for embeddings."""
|
||||
client = OllamaClient(
|
||||
base_url=settings.ollama_url,
|
||||
model=settings.ollama_model
|
||||
base_url=ollama_test_config["base_url"],
|
||||
model=ollama_test_config["model"]
|
||||
)
|
||||
yield client
|
||||
await client.close()
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def job_manager(settings) -> AsyncGenerator[JobManager, None]:
|
||||
async def job_manager(redis_test_url) -> AsyncGenerator[JobManager, None]:
|
||||
"""Get job manager."""
|
||||
manager = JobManager(redis_url=settings.redis_url)
|
||||
manager = JobManager(redis_url=redis_test_url)
|
||||
await manager.connect()
|
||||
yield manager
|
||||
await manager.close()
|
||||
|
||||
@@ -0,0 +1,488 @@
|
||||
"""
|
||||
Tests for maintenance router and cleanup functionality.
|
||||
|
||||
Tests cleanup of:
|
||||
- Orphan vector chunks
|
||||
- Orphan entities in graph
|
||||
- Stale document nodes
|
||||
"""
|
||||
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
from src.routers.maintenance import (
|
||||
cleanup_vectors,
|
||||
cleanup_graph,
|
||||
cleanup_all,
|
||||
maintenance_health,
|
||||
reindex_page,
|
||||
CleanupResult,
|
||||
VectorCleanupResponse,
|
||||
GraphCleanupResponse,
|
||||
FullCleanupResponse,
|
||||
HealthCheckResponse,
|
||||
ReindexResponse
|
||||
)
|
||||
|
||||
|
||||
class TestCleanupResult:
|
||||
"""Test CleanupResult model."""
|
||||
|
||||
def test_cleanup_result_defaults(self):
|
||||
"""Test CleanupResult with default values."""
|
||||
result = CleanupResult(duration_ms=100.0)
|
||||
assert result.orphans_found == 0
|
||||
assert result.orphans_purged == 0
|
||||
assert result.duration_ms == 100.0
|
||||
|
||||
def test_cleanup_result_with_values(self):
|
||||
"""Test CleanupResult with actual values."""
|
||||
result = CleanupResult(
|
||||
orphans_found=10,
|
||||
orphans_purged=8,
|
||||
duration_ms=250.5
|
||||
)
|
||||
assert result.orphans_found == 10
|
||||
assert result.orphans_purged == 8
|
||||
assert result.duration_ms == 250.5
|
||||
|
||||
|
||||
class TestVectorCleanupResponse:
|
||||
"""Test VectorCleanupResponse model."""
|
||||
|
||||
def test_vector_cleanup_response(self):
|
||||
"""Test VectorCleanupResponse structure."""
|
||||
response = VectorCleanupResponse(
|
||||
success=True,
|
||||
wiki_chunks=CleanupResult(orphans_found=5, orphans_purged=5, duration_ms=50),
|
||||
document_chunks=CleanupResult(orphans_found=3, orphans_purged=3, duration_ms=50),
|
||||
chunks_without_graph=CleanupResult(orphans_found=2, orphans_purged=2, duration_ms=50),
|
||||
total_chunks_scanned=100,
|
||||
total_orphans_purged=10,
|
||||
duration_ms=100
|
||||
)
|
||||
assert response.success is True
|
||||
assert response.wiki_chunks.orphans_found == 5
|
||||
assert response.document_chunks.orphans_found == 3
|
||||
assert response.chunks_without_graph.orphans_found == 2
|
||||
assert response.total_orphans_purged == 10
|
||||
|
||||
|
||||
class TestGraphCleanupResponse:
|
||||
"""Test GraphCleanupResponse model."""
|
||||
|
||||
def test_graph_cleanup_response(self):
|
||||
"""Test GraphCleanupResponse structure."""
|
||||
response = GraphCleanupResponse(
|
||||
success=True,
|
||||
orphan_entities=CleanupResult(orphans_found=10, orphans_purged=10, duration_ms=25),
|
||||
stale_wiki_documents=CleanupResult(orphans_found=2, orphans_purged=2, duration_ms=25),
|
||||
stale_store_documents=CleanupResult(orphans_found=0, orphans_purged=0, duration_ms=25),
|
||||
docs_without_vectors=CleanupResult(orphans_found=1, orphans_purged=1, duration_ms=25),
|
||||
broken_relationships_cleaned=5,
|
||||
duration_ms=100
|
||||
)
|
||||
assert response.success is True
|
||||
assert response.orphan_entities.orphans_found == 10
|
||||
assert response.docs_without_vectors.orphans_found == 1
|
||||
assert response.broken_relationships_cleaned == 5
|
||||
|
||||
|
||||
class TestHealthCheckResponse:
|
||||
"""Test HealthCheckResponse model."""
|
||||
|
||||
def test_health_check_healthy(self):
|
||||
"""Test healthy status."""
|
||||
response = HealthCheckResponse(
|
||||
status="healthy",
|
||||
orphan_vector_count=0,
|
||||
orphan_entity_count=0,
|
||||
stale_document_count=0
|
||||
)
|
||||
assert response.status == "healthy"
|
||||
assert response.recommendations == []
|
||||
|
||||
def test_health_check_degraded(self):
|
||||
"""Test degraded status with recommendations."""
|
||||
response = HealthCheckResponse(
|
||||
status="degraded",
|
||||
orphan_vector_count=15,
|
||||
orphan_entity_count=3,
|
||||
stale_document_count=0,
|
||||
recommendations=[
|
||||
"Found 15 orphan vector chunks. Consider running POST /maintenance/cleanup/vectors"
|
||||
]
|
||||
)
|
||||
assert response.status == "degraded"
|
||||
assert len(response.recommendations) == 1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestVectorCleanup:
|
||||
"""Test vector cleanup endpoint."""
|
||||
|
||||
async def test_cleanup_vectors_no_orphans(self):
|
||||
"""Test cleanup when no orphans exist."""
|
||||
# Mock services
|
||||
vector_service = AsyncMock()
|
||||
vector_service.get_all_chunk_references.return_value = [
|
||||
{"chunk_id": "c1", "page_id": 1, "doc_type": "wiki"}
|
||||
]
|
||||
# find_chunks_without_graph_nodes is not async
|
||||
vector_service.find_chunks_without_graph_nodes = MagicMock(return_value=[])
|
||||
|
||||
graph_service = AsyncMock()
|
||||
graph_service.get_all_document_references.return_value = [
|
||||
{"page_id": 1, "doc_type": "wiki", "title": "Test"}
|
||||
]
|
||||
|
||||
wiki_client = AsyncMock()
|
||||
wiki_client.list_all_pages.return_value = [{"id": 1, "path": "test"}]
|
||||
|
||||
# Call cleanup
|
||||
result = await cleanup_vectors(
|
||||
user="testuser",
|
||||
dry_run=False,
|
||||
vector_service=vector_service,
|
||||
graph_service=graph_service,
|
||||
wiki_client=wiki_client,
|
||||
api_key="test"
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
assert result.wiki_chunks.orphans_found == 0
|
||||
assert result.chunks_without_graph.orphans_found == 0
|
||||
assert result.total_orphans_purged == 0
|
||||
|
||||
async def test_cleanup_vectors_with_orphans(self):
|
||||
"""Test cleanup when orphans exist."""
|
||||
# Mock services
|
||||
vector_service = AsyncMock()
|
||||
vector_service.get_all_chunk_references.return_value = [
|
||||
{"chunk_id": "c1", "page_id": 1, "doc_type": "wiki"},
|
||||
{"chunk_id": "c2", "page_id": 999, "doc_type": "wiki"}, # Orphan
|
||||
{"chunk_id": "c3", "page_id": 999, "doc_type": "wiki"}, # Orphan
|
||||
]
|
||||
vector_service.purge_chunks_by_ids.return_value = 2
|
||||
# find_chunks_without_graph_nodes is not async
|
||||
vector_service.find_chunks_without_graph_nodes = MagicMock(return_value=[])
|
||||
|
||||
graph_service = AsyncMock()
|
||||
graph_service.get_all_document_references.return_value = [
|
||||
{"page_id": 1, "doc_type": "wiki", "title": "Test"}
|
||||
]
|
||||
|
||||
wiki_client = AsyncMock()
|
||||
wiki_client.list_all_pages.return_value = [{"id": 1, "path": "test"}]
|
||||
|
||||
# Call cleanup
|
||||
result = await cleanup_vectors(
|
||||
user="testuser",
|
||||
dry_run=False,
|
||||
vector_service=vector_service,
|
||||
graph_service=graph_service,
|
||||
wiki_client=wiki_client,
|
||||
api_key="test"
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
assert result.wiki_chunks.orphans_found == 2
|
||||
assert result.wiki_chunks.orphans_purged == 2
|
||||
assert result.total_orphans_purged == 2
|
||||
|
||||
async def test_cleanup_vectors_dry_run(self):
|
||||
"""Test cleanup dry run doesn't purge."""
|
||||
# Mock services
|
||||
vector_service = AsyncMock()
|
||||
vector_service.get_all_chunk_references.return_value = [
|
||||
{"chunk_id": "c1", "page_id": 999, "doc_type": "wiki"}, # Orphan
|
||||
]
|
||||
# find_chunks_without_graph_nodes is not async
|
||||
vector_service.find_chunks_without_graph_nodes = MagicMock(return_value=[])
|
||||
|
||||
graph_service = AsyncMock()
|
||||
graph_service.get_all_document_references.return_value = []
|
||||
|
||||
wiki_client = AsyncMock()
|
||||
wiki_client.list_all_pages.return_value = []
|
||||
|
||||
# Call cleanup in dry run mode
|
||||
result = await cleanup_vectors(
|
||||
user="testuser",
|
||||
dry_run=True,
|
||||
vector_service=vector_service,
|
||||
graph_service=graph_service,
|
||||
wiki_client=wiki_client,
|
||||
api_key="test"
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
assert result.wiki_chunks.orphans_found == 1
|
||||
assert result.wiki_chunks.orphans_purged == 0 # Not purged due to dry run
|
||||
vector_service.purge_chunks_by_ids.assert_not_called()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestGraphCleanup:
|
||||
"""Test graph cleanup endpoint."""
|
||||
|
||||
async def test_cleanup_graph_no_orphans(self):
|
||||
"""Test cleanup when no orphans exist."""
|
||||
vector_service = AsyncMock()
|
||||
vector_service.get_all_chunk_references.return_value = [
|
||||
{"chunk_id": "c1", "page_id": 1, "doc_type": "wiki"}
|
||||
]
|
||||
|
||||
graph_service = AsyncMock()
|
||||
graph_service.find_orphan_entities.return_value = []
|
||||
graph_service.get_all_document_references.return_value = [
|
||||
{"page_id": 1, "doc_type": "wiki", "title": "Test"}
|
||||
]
|
||||
graph_service.find_documents_without_vectors.return_value = []
|
||||
graph_service.cleanup_broken_relationships.return_value = 0
|
||||
|
||||
wiki_client = AsyncMock()
|
||||
wiki_client.list_all_pages.return_value = [{"id": 1, "path": "test"}]
|
||||
|
||||
result = await cleanup_graph(
|
||||
user="testuser",
|
||||
dry_run=False,
|
||||
vector_service=vector_service,
|
||||
graph_service=graph_service,
|
||||
wiki_client=wiki_client,
|
||||
api_key="test"
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
assert result.orphan_entities.orphans_found == 0
|
||||
assert result.stale_wiki_documents.orphans_found == 0
|
||||
assert result.docs_without_vectors.orphans_found == 0
|
||||
|
||||
async def test_cleanup_graph_with_orphan_entities(self):
|
||||
"""Test cleanup of orphan entities."""
|
||||
vector_service = AsyncMock()
|
||||
vector_service.get_all_chunk_references.return_value = []
|
||||
|
||||
graph_service = AsyncMock()
|
||||
graph_service.find_orphan_entities.return_value = [
|
||||
{"id": "e1", "name": "Orphan1", "type": "Person"},
|
||||
{"id": "e2", "name": "Orphan2", "type": "Technology"},
|
||||
]
|
||||
graph_service.purge_orphan_entities.return_value = 2
|
||||
graph_service.get_all_document_references.return_value = []
|
||||
graph_service.find_documents_without_vectors.return_value = []
|
||||
graph_service.cleanup_broken_relationships.return_value = 0
|
||||
|
||||
wiki_client = AsyncMock()
|
||||
wiki_client.list_all_pages.return_value = []
|
||||
|
||||
result = await cleanup_graph(
|
||||
user="testuser",
|
||||
dry_run=False,
|
||||
vector_service=vector_service,
|
||||
graph_service=graph_service,
|
||||
wiki_client=wiki_client,
|
||||
api_key="test"
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
assert result.orphan_entities.orphans_found == 2
|
||||
assert result.orphan_entities.orphans_purged == 2
|
||||
|
||||
async def test_cleanup_graph_with_stale_documents(self):
|
||||
"""Test cleanup of stale document nodes."""
|
||||
vector_service = AsyncMock()
|
||||
vector_service.get_all_chunk_references.return_value = [
|
||||
{"chunk_id": "c1", "page_id": 1, "doc_type": "wiki"}
|
||||
]
|
||||
|
||||
graph_service = AsyncMock()
|
||||
graph_service.find_orphan_entities.return_value = []
|
||||
graph_service.get_all_document_references.return_value = [
|
||||
{"page_id": 1, "doc_type": "wiki", "title": "Exists"},
|
||||
{"page_id": 999, "doc_type": "wiki", "title": "Deleted"}, # Stale
|
||||
]
|
||||
graph_service.find_documents_without_vectors.return_value = []
|
||||
graph_service.purge_stale_documents_by_ids.return_value = 1
|
||||
graph_service.cleanup_broken_relationships.return_value = 0
|
||||
|
||||
wiki_client = AsyncMock()
|
||||
wiki_client.list_all_pages.return_value = [{"id": 1, "path": "test"}]
|
||||
|
||||
result = await cleanup_graph(
|
||||
user="testuser",
|
||||
dry_run=False,
|
||||
vector_service=vector_service,
|
||||
graph_service=graph_service,
|
||||
wiki_client=wiki_client,
|
||||
api_key="test"
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
assert result.stale_wiki_documents.orphans_found == 1
|
||||
assert result.stale_wiki_documents.orphans_purged == 1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestFullCleanup:
|
||||
"""Test full cleanup endpoint."""
|
||||
|
||||
async def test_full_cleanup(self):
|
||||
"""Test full cleanup runs both vector and graph cleanup."""
|
||||
vector_service = AsyncMock()
|
||||
vector_service.get_all_chunk_references.return_value = []
|
||||
# find_chunks_without_graph_nodes is not async
|
||||
vector_service.find_chunks_without_graph_nodes = MagicMock(return_value=[])
|
||||
|
||||
graph_service = AsyncMock()
|
||||
graph_service.find_orphan_entities.return_value = []
|
||||
graph_service.get_all_document_references.return_value = []
|
||||
graph_service.find_documents_without_vectors.return_value = []
|
||||
graph_service.cleanup_broken_relationships.return_value = 0
|
||||
|
||||
wiki_client = AsyncMock()
|
||||
wiki_client.list_all_pages.return_value = []
|
||||
|
||||
result = await cleanup_all(
|
||||
user="testuser",
|
||||
dry_run=False,
|
||||
vector_service=vector_service,
|
||||
graph_service=graph_service,
|
||||
wiki_client=wiki_client,
|
||||
api_key="test"
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
assert result.vector_cleanup.success is True
|
||||
assert result.graph_cleanup.success is True
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestMaintenanceHealth:
|
||||
"""Test maintenance health endpoint."""
|
||||
|
||||
async def test_health_healthy(self):
|
||||
"""Test healthy status when no orphans."""
|
||||
vector_service = AsyncMock()
|
||||
vector_service.get_all_chunk_references.return_value = []
|
||||
# find_chunks_without_graph_nodes is not async
|
||||
vector_service.find_chunks_without_graph_nodes = MagicMock(return_value=[])
|
||||
|
||||
graph_service = AsyncMock()
|
||||
graph_service.find_orphan_entities.return_value = []
|
||||
graph_service.get_all_document_references.return_value = []
|
||||
graph_service.find_documents_without_vectors.return_value = []
|
||||
|
||||
wiki_client = AsyncMock()
|
||||
wiki_client.list_all_pages.return_value = []
|
||||
|
||||
result = await maintenance_health(
|
||||
user="testuser",
|
||||
vector_service=vector_service,
|
||||
graph_service=graph_service,
|
||||
wiki_client=wiki_client,
|
||||
api_key="test"
|
||||
)
|
||||
|
||||
assert result.status == "healthy"
|
||||
assert result.orphan_vector_count == 0
|
||||
assert result.orphan_entity_count == 0
|
||||
assert result.vectors_without_graph == 0
|
||||
assert result.docs_without_vectors == 0
|
||||
|
||||
async def test_health_degraded(self):
|
||||
"""Test degraded status with orphans."""
|
||||
vector_service = AsyncMock()
|
||||
vector_service.get_all_chunk_references.return_value = [
|
||||
{"chunk_id": f"c{i}", "page_id": 999, "doc_type": "wiki"}
|
||||
for i in range(15)
|
||||
]
|
||||
# find_chunks_without_graph_nodes is not async
|
||||
vector_service.find_chunks_without_graph_nodes = MagicMock(return_value=[])
|
||||
|
||||
graph_service = AsyncMock()
|
||||
graph_service.find_orphan_entities.return_value = [
|
||||
{"id": f"e{i}", "name": f"Entity{i}", "type": "Entity"}
|
||||
for i in range(3)
|
||||
]
|
||||
graph_service.get_all_document_references.return_value = []
|
||||
graph_service.find_documents_without_vectors.return_value = []
|
||||
|
||||
wiki_client = AsyncMock()
|
||||
wiki_client.list_all_pages.return_value = []
|
||||
|
||||
result = await maintenance_health(
|
||||
user="testuser",
|
||||
vector_service=vector_service,
|
||||
graph_service=graph_service,
|
||||
wiki_client=wiki_client,
|
||||
api_key="test"
|
||||
)
|
||||
|
||||
assert result.status == "degraded"
|
||||
assert result.orphan_vector_count == 15
|
||||
assert result.orphan_entity_count == 3
|
||||
assert len(result.recommendations) >= 1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestReindexPage:
|
||||
"""Test reindex page endpoint."""
|
||||
|
||||
async def test_reindex_success(self):
|
||||
"""Test successful page reindex."""
|
||||
vector_service = AsyncMock()
|
||||
vector_service.delete_page_chunks.return_value = 5
|
||||
vector_service.update_from_page.return_value = MagicMock(
|
||||
success=True,
|
||||
chunks_created=6,
|
||||
error_message=None
|
||||
)
|
||||
|
||||
graph_service = AsyncMock()
|
||||
graph_service.delete_page.return_value = 1
|
||||
graph_service.update_from_page.return_value = MagicMock(
|
||||
success=True,
|
||||
error_message=None
|
||||
)
|
||||
|
||||
result = await reindex_page(
|
||||
page_id=123,
|
||||
user="testuser",
|
||||
vector_service=vector_service,
|
||||
graph_service=graph_service,
|
||||
api_key="test"
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
assert result.page_id == 123
|
||||
assert result.vectors_deleted == 5
|
||||
assert result.vectors_created == 6
|
||||
assert result.graph_updated is True
|
||||
|
||||
async def test_reindex_failure(self):
|
||||
"""Test reindex with failure."""
|
||||
vector_service = AsyncMock()
|
||||
vector_service.delete_page_chunks.return_value = 0
|
||||
vector_service.update_from_page.return_value = MagicMock(
|
||||
success=False,
|
||||
chunks_created=0,
|
||||
error_message="Page not found"
|
||||
)
|
||||
|
||||
graph_service = AsyncMock()
|
||||
graph_service.delete_page.return_value = 0
|
||||
graph_service.update_from_page.return_value = MagicMock(
|
||||
success=False,
|
||||
error_message="Page not found"
|
||||
)
|
||||
|
||||
result = await reindex_page(
|
||||
page_id=999,
|
||||
user="testuser",
|
||||
vector_service=vector_service,
|
||||
graph_service=graph_service,
|
||||
api_key="test"
|
||||
)
|
||||
|
||||
assert result.success is False
|
||||
assert result.error == "Page not found"
|
||||
@@ -0,0 +1,369 @@
|
||||
"""Tests for RAG search service and endpoints."""
|
||||
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
from src.models.rag_search import (
|
||||
SearchType,
|
||||
RAGSearchRequest,
|
||||
RAGSearchResult,
|
||||
RAGSearchResponse,
|
||||
)
|
||||
from src.services.rag_search_service import RAGSearchService, extract_domain
|
||||
|
||||
|
||||
class TestExtractDomain:
|
||||
"""Tests for domain extraction utility."""
|
||||
|
||||
def test_extract_simple_domain(self):
|
||||
"""Test extracting domain from simple URL."""
|
||||
assert extract_domain("https://example.com/page") == "example.com"
|
||||
|
||||
def test_extract_domain_with_www(self):
|
||||
"""Test extracting domain removes www prefix."""
|
||||
assert extract_domain("https://www.example.com/page") == "example.com"
|
||||
|
||||
def test_extract_domain_with_subdomain(self):
|
||||
"""Test extracting domain preserves subdomains."""
|
||||
assert extract_domain("https://blog.example.com/post") == "blog.example.com"
|
||||
|
||||
def test_extract_domain_invalid_url(self):
|
||||
"""Test extracting domain from invalid URL returns empty string."""
|
||||
# urlparse returns empty netloc for invalid URLs
|
||||
assert extract_domain("not-a-url") == ""
|
||||
|
||||
|
||||
class TestRAGSearchModels:
|
||||
"""Tests for RAG search Pydantic models."""
|
||||
|
||||
def test_search_request_defaults(self):
|
||||
"""Test RAGSearchRequest with default values."""
|
||||
request = RAGSearchRequest(query="test query")
|
||||
|
||||
assert request.query == "test query"
|
||||
assert request.search_type == SearchType.WEB
|
||||
assert request.limit == 10
|
||||
|
||||
def test_search_request_custom_values(self):
|
||||
"""Test RAGSearchRequest with custom values."""
|
||||
request = RAGSearchRequest(
|
||||
query="news about AI",
|
||||
search_type=SearchType.NEWS,
|
||||
limit=5,
|
||||
user="custom_user"
|
||||
)
|
||||
|
||||
assert request.query == "news about AI"
|
||||
assert request.search_type == SearchType.NEWS
|
||||
assert request.limit == 5
|
||||
assert request.user == "custom_user"
|
||||
|
||||
def test_search_result(self):
|
||||
"""Test RAGSearchResult model."""
|
||||
result = RAGSearchResult(
|
||||
title="Test Article",
|
||||
url="https://example.com/article",
|
||||
content="Full article content",
|
||||
snippet="Article snippet...",
|
||||
source="example.com",
|
||||
published_date="2024-01-15"
|
||||
)
|
||||
|
||||
assert result.title == "Test Article"
|
||||
assert result.source == "example.com"
|
||||
assert result.published_date == "2024-01-15"
|
||||
|
||||
def test_search_response(self):
|
||||
"""Test RAGSearchResponse model."""
|
||||
response = RAGSearchResponse(
|
||||
query="test",
|
||||
search_type=SearchType.WEB,
|
||||
results=[],
|
||||
total_results=0,
|
||||
search_time_ms=100,
|
||||
sources_summary=""
|
||||
)
|
||||
|
||||
assert response.query == "test"
|
||||
assert response.total_results == 0
|
||||
assert response.search_time_ms == 100
|
||||
|
||||
|
||||
class TestRAGSearchService:
|
||||
"""Tests for RAGSearchService."""
|
||||
|
||||
@pytest.fixture
|
||||
def mock_searxng_client(self):
|
||||
"""Create mock SearXNG client."""
|
||||
client = MagicMock()
|
||||
client.search_general = AsyncMock(return_value=[
|
||||
{
|
||||
"title": "Test Result 1",
|
||||
"url": "https://example.com/1",
|
||||
"content": "Snippet 1",
|
||||
"publishedDate": "2024-01-15"
|
||||
},
|
||||
{
|
||||
"title": "Test Result 2",
|
||||
"url": "https://example.com/2",
|
||||
"content": "Snippet 2",
|
||||
"publishedDate": None
|
||||
}
|
||||
])
|
||||
client.search_news = AsyncMock(return_value=[])
|
||||
client.search_images = AsyncMock(return_value=[])
|
||||
return client
|
||||
|
||||
@pytest.fixture
|
||||
def mock_content_extractor(self):
|
||||
"""Create mock ContentExtractor."""
|
||||
from src.models.content import ContentExtractionResult
|
||||
|
||||
extractor = MagicMock()
|
||||
extractor.extract_batch = AsyncMock(return_value=[
|
||||
ContentExtractionResult(
|
||||
url="https://example.com/1",
|
||||
content="Full extracted content 1",
|
||||
success=True
|
||||
),
|
||||
ContentExtractionResult(
|
||||
url="https://example.com/2",
|
||||
content="Full extracted content 2",
|
||||
success=True
|
||||
)
|
||||
])
|
||||
return extractor
|
||||
|
||||
@pytest.fixture
|
||||
def mock_redis_client(self):
|
||||
"""Create mock Redis client."""
|
||||
redis = MagicMock()
|
||||
redis.get = AsyncMock(return_value=None) # No cache hit
|
||||
redis.setex = AsyncMock()
|
||||
return redis
|
||||
|
||||
@pytest.fixture
|
||||
def mock_settings(self):
|
||||
"""Create mock settings."""
|
||||
settings = MagicMock()
|
||||
settings.search_cache_ttl = 300
|
||||
settings.search_default_limit = 10
|
||||
return settings
|
||||
|
||||
@pytest.fixture
|
||||
def rag_search_service(
|
||||
self,
|
||||
mock_searxng_client,
|
||||
mock_content_extractor,
|
||||
mock_redis_client,
|
||||
mock_settings
|
||||
):
|
||||
"""Create RAGSearchService with mocked dependencies."""
|
||||
return RAGSearchService(
|
||||
searxng_client=mock_searxng_client,
|
||||
content_extractor=mock_content_extractor,
|
||||
redis_client=mock_redis_client,
|
||||
settings=mock_settings
|
||||
)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_search_basic(self, rag_search_service, mock_searxng_client):
|
||||
"""Test basic web search."""
|
||||
response = await rag_search_service.search(
|
||||
query="test query",
|
||||
search_type=SearchType.WEB,
|
||||
limit=10
|
||||
)
|
||||
|
||||
assert response.query == "test query"
|
||||
assert response.search_type == SearchType.WEB
|
||||
assert len(response.results) == 2
|
||||
assert response.total_results == 2
|
||||
assert response.search_time_ms >= 0
|
||||
|
||||
mock_searxng_client.search_general.assert_called_once()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_search_news(self, rag_search_service, mock_searxng_client):
|
||||
"""Test news search type."""
|
||||
mock_searxng_client.search_news.return_value = [
|
||||
{"title": "News", "url": "https://news.com/1", "content": "News content"}
|
||||
]
|
||||
|
||||
response = await rag_search_service.search(
|
||||
query="latest news",
|
||||
search_type=SearchType.NEWS
|
||||
)
|
||||
|
||||
assert response.search_type == SearchType.NEWS
|
||||
mock_searxng_client.search_news.assert_called_once()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_search_images(self, rag_search_service, mock_searxng_client):
|
||||
"""Test image search type."""
|
||||
mock_searxng_client.search_images.return_value = [
|
||||
{"title": "Image", "url": "https://images.com/1.jpg", "content": ""}
|
||||
]
|
||||
|
||||
response = await rag_search_service.search(
|
||||
query="cat photos",
|
||||
search_type=SearchType.IMAGES
|
||||
)
|
||||
|
||||
assert response.search_type == SearchType.IMAGES
|
||||
mock_searxng_client.search_images.assert_called_once()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_search_empty_query(self, rag_search_service):
|
||||
"""Test search with empty query raises ValueError."""
|
||||
with pytest.raises(ValueError, match="Query cannot be empty"):
|
||||
await rag_search_service.search(query="", search_type=SearchType.WEB)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_search_caching_miss(
|
||||
self,
|
||||
rag_search_service,
|
||||
mock_redis_client,
|
||||
mock_searxng_client
|
||||
):
|
||||
"""Test search caches results on cache miss."""
|
||||
mock_redis_client.get.return_value = None # Cache miss
|
||||
|
||||
await rag_search_service.search(query="test", search_type=SearchType.WEB)
|
||||
|
||||
# Should call SearXNG (cache miss)
|
||||
mock_searxng_client.search_general.assert_called_once()
|
||||
# Should cache result
|
||||
mock_redis_client.setex.assert_called_once()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_search_caching_hit(
|
||||
self,
|
||||
rag_search_service,
|
||||
mock_redis_client,
|
||||
mock_searxng_client
|
||||
):
|
||||
"""Test search returns cached results on cache hit."""
|
||||
# Simulate cache hit
|
||||
cached_response = RAGSearchResponse(
|
||||
query="test",
|
||||
search_type=SearchType.WEB,
|
||||
results=[],
|
||||
total_results=0,
|
||||
search_time_ms=50,
|
||||
sources_summary=""
|
||||
)
|
||||
mock_redis_client.get.return_value = cached_response.model_dump_json()
|
||||
|
||||
response = await rag_search_service.search(query="test", search_type=SearchType.WEB)
|
||||
|
||||
# Should NOT call SearXNG (cache hit)
|
||||
mock_searxng_client.search_general.assert_not_called()
|
||||
assert response.query == "test"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_search_content_extraction(
|
||||
self,
|
||||
rag_search_service,
|
||||
mock_content_extractor
|
||||
):
|
||||
"""Test search extracts content from result URLs."""
|
||||
response = await rag_search_service.search(
|
||||
query="test",
|
||||
search_type=SearchType.WEB
|
||||
)
|
||||
|
||||
# Should have called content extractor
|
||||
mock_content_extractor.extract_batch.assert_called_once()
|
||||
|
||||
# Results should have extracted content
|
||||
for result in response.results:
|
||||
assert result.content # Content should be populated
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_search_sources_summary(self, rag_search_service):
|
||||
"""Test search generates sources summary."""
|
||||
response = await rag_search_service.search(
|
||||
query="test",
|
||||
search_type=SearchType.WEB
|
||||
)
|
||||
|
||||
assert response.sources_summary
|
||||
assert "## Sources" in response.sources_summary
|
||||
assert "[Test Result 1]" in response.sources_summary
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_search_limit(self, rag_search_service, mock_searxng_client):
|
||||
"""Test search respects limit parameter."""
|
||||
await rag_search_service.search(
|
||||
query="test",
|
||||
search_type=SearchType.WEB,
|
||||
limit=5
|
||||
)
|
||||
|
||||
# Check limit was passed to SearXNG
|
||||
mock_searxng_client.search_general.assert_called_once_with(
|
||||
query="test",
|
||||
limit=5
|
||||
)
|
||||
|
||||
|
||||
class TestRAGSearchServiceIntegration:
|
||||
"""Integration-style tests (still mocked but test more of the flow)."""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_full_search_flow(self):
|
||||
"""Test full search flow with all components mocked."""
|
||||
from src.models.content import ContentExtractionResult
|
||||
|
||||
# Setup mocks
|
||||
mock_searxng = MagicMock()
|
||||
mock_searxng.search_general = AsyncMock(return_value=[
|
||||
{
|
||||
"title": "Python Tutorial",
|
||||
"url": "https://python.org/tutorial",
|
||||
"content": "Learn Python programming",
|
||||
"publishedDate": "2024-01-10"
|
||||
}
|
||||
])
|
||||
|
||||
mock_extractor = MagicMock()
|
||||
mock_extractor.extract_batch = AsyncMock(return_value=[
|
||||
ContentExtractionResult(
|
||||
url="https://python.org/tutorial",
|
||||
title="Python Tutorial",
|
||||
content="This is a comprehensive Python tutorial covering basics to advanced topics.",
|
||||
success=True
|
||||
)
|
||||
])
|
||||
|
||||
mock_redis = MagicMock()
|
||||
mock_redis.get = AsyncMock(return_value=None)
|
||||
mock_redis.setex = AsyncMock()
|
||||
|
||||
mock_settings = MagicMock()
|
||||
mock_settings.search_cache_ttl = 300
|
||||
mock_settings.search_default_limit = 10
|
||||
|
||||
# Create service and execute search
|
||||
service = RAGSearchService(
|
||||
searxng_client=mock_searxng,
|
||||
content_extractor=mock_extractor,
|
||||
redis_client=mock_redis,
|
||||
settings=mock_settings
|
||||
)
|
||||
|
||||
response = await service.search(
|
||||
query="python tutorial",
|
||||
search_type=SearchType.WEB,
|
||||
limit=10,
|
||||
user="test_user"
|
||||
)
|
||||
|
||||
# Verify response
|
||||
assert response.query == "python tutorial"
|
||||
assert len(response.results) == 1
|
||||
assert response.results[0].title == "Python Tutorial"
|
||||
assert response.results[0].source == "python.org"
|
||||
assert "comprehensive Python tutorial" in response.results[0].content
|
||||
assert response.results[0].snippet == "Learn Python programming"
|
||||
@@ -0,0 +1,553 @@
|
||||
"""
|
||||
Tests for Smart Page Creation functionality.
|
||||
|
||||
Tests the new smart-create feature including:
|
||||
- WikiSmartCreateRequest/Response models
|
||||
- smart_create_page() method in WikiService
|
||||
- Bidirectional entity linking utilities
|
||||
- POST /wiki/pages/smart-create endpoint
|
||||
"""
|
||||
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
from src.models.wiki import (
|
||||
WikiSmartCreateRequest,
|
||||
WikiSmartCreateResponse,
|
||||
WikiPage
|
||||
)
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Model Tests
|
||||
# =============================================================================
|
||||
|
||||
class TestWikiSmartCreateRequest:
|
||||
"""Tests for WikiSmartCreateRequest model validation."""
|
||||
|
||||
def test_minimal_request(self):
|
||||
"""Test request with only required field."""
|
||||
request = WikiSmartCreateRequest(topic="Docker containers")
|
||||
assert request.topic == "Docker containers"
|
||||
assert request.path is None
|
||||
assert request.tags == []
|
||||
assert request.user is None
|
||||
assert request.include_web_research is True
|
||||
assert request.include_wiki_search is True
|
||||
|
||||
def test_full_request(self):
|
||||
"""Test request with all fields."""
|
||||
request = WikiSmartCreateRequest(
|
||||
topic="Kubernetes orchestration",
|
||||
path="/technology/kubernetes",
|
||||
tags=["devops", "containers"],
|
||||
user="testuser",
|
||||
include_web_research=False,
|
||||
include_wiki_search=True
|
||||
)
|
||||
assert request.topic == "Kubernetes orchestration"
|
||||
assert request.path == "/technology/kubernetes"
|
||||
# Tags are deduplicated via set, so order is not guaranteed
|
||||
assert set(request.tags) == {"devops", "containers"}
|
||||
assert request.user == "testuser"
|
||||
assert request.include_web_research is False
|
||||
assert request.include_wiki_search is True
|
||||
|
||||
def test_topic_min_length(self):
|
||||
"""Test that topic requires at least 1 character."""
|
||||
with pytest.raises(ValueError):
|
||||
WikiSmartCreateRequest(topic="")
|
||||
|
||||
def test_topic_max_length(self):
|
||||
"""Test that topic is limited to 500 characters."""
|
||||
long_topic = "x" * 501
|
||||
with pytest.raises(ValueError):
|
||||
WikiSmartCreateRequest(topic=long_topic)
|
||||
|
||||
def test_path_validation_adds_leading_slash(self):
|
||||
"""Test that path without leading slash gets one added."""
|
||||
request = WikiSmartCreateRequest(
|
||||
topic="Test",
|
||||
path="technology/test"
|
||||
)
|
||||
assert request.path == "/technology/test"
|
||||
|
||||
def test_path_validation_removes_trailing_slash(self):
|
||||
"""Test that trailing slash is removed."""
|
||||
request = WikiSmartCreateRequest(
|
||||
topic="Test",
|
||||
path="/technology/test/"
|
||||
)
|
||||
assert request.path == "/technology/test"
|
||||
|
||||
def test_tags_deduplication(self):
|
||||
"""Test that duplicate tags are removed."""
|
||||
request = WikiSmartCreateRequest(
|
||||
topic="Test",
|
||||
tags=["devops", "devops", "containers", "devops"]
|
||||
)
|
||||
assert len(request.tags) == 2
|
||||
assert "devops" in request.tags
|
||||
assert "containers" in request.tags
|
||||
|
||||
def test_tags_whitespace_cleanup(self):
|
||||
"""Test that tag whitespace is cleaned."""
|
||||
request = WikiSmartCreateRequest(
|
||||
topic="Test",
|
||||
tags=[" devops ", "containers", " ", ""]
|
||||
)
|
||||
assert "devops" in request.tags
|
||||
assert "containers" in request.tags
|
||||
assert "" not in request.tags
|
||||
assert " " not in request.tags
|
||||
|
||||
|
||||
class TestWikiSmartCreateResponse:
|
||||
"""Tests for WikiSmartCreateResponse model."""
|
||||
|
||||
def test_response_structure(self):
|
||||
"""Test response model with all fields."""
|
||||
page = WikiPage(
|
||||
id=123,
|
||||
path="/users/test/technology/docker",
|
||||
title="Docker",
|
||||
content="# Docker\n\nContent here",
|
||||
tags=["technology"],
|
||||
is_published=True,
|
||||
created_at="2024-01-15T10:00:00Z",
|
||||
updated_at="2024-01-15T10:00:00Z"
|
||||
)
|
||||
|
||||
response = WikiSmartCreateResponse(
|
||||
page=page,
|
||||
research_summary={
|
||||
"wiki_results": 3,
|
||||
"web_results": 5,
|
||||
"graph_entities": 2
|
||||
},
|
||||
sources_used=8,
|
||||
search_id="test-uuid-123",
|
||||
entity_linking={
|
||||
"forward_links": 4,
|
||||
"backward_links": 2,
|
||||
"pages_updated": 1
|
||||
}
|
||||
)
|
||||
|
||||
assert response.page.id == 123
|
||||
assert response.sources_used == 8
|
||||
assert response.research_summary["wiki_results"] == 3
|
||||
assert response.entity_linking["forward_links"] == 4
|
||||
|
||||
def test_response_default_entity_linking(self):
|
||||
"""Test that entity_linking defaults to empty dict."""
|
||||
page = WikiPage(
|
||||
id=1,
|
||||
path="/test",
|
||||
title="Test",
|
||||
content="Content",
|
||||
tags=[],
|
||||
is_published=True,
|
||||
created_at="2024-01-15T10:00:00Z",
|
||||
updated_at="2024-01-15T10:00:00Z"
|
||||
)
|
||||
|
||||
response = WikiSmartCreateResponse(
|
||||
page=page,
|
||||
research_summary={},
|
||||
sources_used=0
|
||||
)
|
||||
|
||||
assert response.entity_linking == {}
|
||||
assert response.search_id is None
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# WikiService.smart_create_page Tests
|
||||
# =============================================================================
|
||||
|
||||
class TestSmartCreatePage:
|
||||
"""Tests for WikiService.smart_create_page method."""
|
||||
|
||||
@pytest.fixture
|
||||
def mock_hybrid_rag_service(self):
|
||||
"""Mock HybridRAG service."""
|
||||
service = AsyncMock()
|
||||
|
||||
# Create mock response
|
||||
mock_response = MagicMock()
|
||||
mock_response.total_results = 5
|
||||
mock_response.search_id = "search-123"
|
||||
mock_response.results = [
|
||||
MagicMock(
|
||||
source_type="vector",
|
||||
title="Existing Docker Page",
|
||||
url=None,
|
||||
page_path="users/testuser/docker-basics",
|
||||
content="Docker is a containerization platform...",
|
||||
related_dossiers=["containers"]
|
||||
),
|
||||
MagicMock(
|
||||
source_type="web",
|
||||
title="Docker Documentation",
|
||||
url="https://docs.docker.com",
|
||||
page_path=None,
|
||||
content="Official Docker documentation...",
|
||||
related_dossiers=None
|
||||
)
|
||||
]
|
||||
mock_response.keywords = MagicMock()
|
||||
mock_response.keywords.core_keywords = ["docker", "containers", "virtualization"]
|
||||
mock_response.timing = MagicMock()
|
||||
mock_response.timing.total_ms = 1500
|
||||
|
||||
service.search = AsyncMock(return_value=mock_response)
|
||||
return service
|
||||
|
||||
@pytest.fixture
|
||||
def mock_wiki_page_writer(self):
|
||||
"""Mock WikiPageWriter."""
|
||||
writer = AsyncMock()
|
||||
writer.create_page = AsyncMock(return_value="# Docker Containers\n\n## Overview\n\nGenerated content about Docker...")
|
||||
return writer
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_smart_create_basic(
|
||||
self,
|
||||
mock_hybrid_rag_service,
|
||||
mock_wiki_page_writer
|
||||
):
|
||||
"""Test basic smart page creation flow."""
|
||||
from src.services.wiki_service import WikiService
|
||||
|
||||
mock_wiki_client = AsyncMock()
|
||||
wiki_service = WikiService(mock_wiki_client)
|
||||
|
||||
# Mock the create_page method on the service itself
|
||||
mock_page = WikiPage(
|
||||
id=42,
|
||||
path="/users/testuser/technology/docker",
|
||||
title="Docker containers",
|
||||
content="# Docker\n\nGenerated content",
|
||||
tags=["technology"],
|
||||
is_published=True,
|
||||
created_at="2024-01-15T10:00:00Z",
|
||||
updated_at="2024-01-15T10:00:00Z"
|
||||
)
|
||||
|
||||
with patch.object(wiki_service, 'create_page', new_callable=AsyncMock) as mock_create:
|
||||
mock_create.return_value = mock_page
|
||||
|
||||
page, research_data = await wiki_service.smart_create_page(
|
||||
topic="Docker containers",
|
||||
user="testuser",
|
||||
path="/technology/docker",
|
||||
tags=["technology"],
|
||||
hybrid_rag_service=mock_hybrid_rag_service,
|
||||
wiki_page_writer=mock_wiki_page_writer,
|
||||
include_web=True,
|
||||
include_wiki=True
|
||||
)
|
||||
|
||||
# Verify HybridRAG was called
|
||||
mock_hybrid_rag_service.search.assert_called_once()
|
||||
|
||||
# Verify WikiPageWriter was called
|
||||
mock_wiki_page_writer.create_page.assert_called_once()
|
||||
|
||||
# Verify page was created
|
||||
mock_create.assert_called_once()
|
||||
|
||||
# Verify research data
|
||||
assert "research_summary" in research_data
|
||||
assert "sources_used" in research_data
|
||||
assert "search_id" in research_data
|
||||
assert research_data["search_id"] == "search-123"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_smart_create_auto_generates_path(
|
||||
self,
|
||||
mock_hybrid_rag_service,
|
||||
mock_wiki_page_writer
|
||||
):
|
||||
"""Test that path is auto-generated from topic when not provided."""
|
||||
from src.services.wiki_service import WikiService
|
||||
|
||||
mock_wiki_client = AsyncMock()
|
||||
wiki_service = WikiService(mock_wiki_client)
|
||||
|
||||
mock_page = WikiPage(
|
||||
id=42,
|
||||
path="/users/testuser/tutorials/docker-compose-tutorial",
|
||||
title="Docker Compose Tutorial",
|
||||
content="# Docker Compose\n\nContent",
|
||||
tags=["tutorials"],
|
||||
is_published=True,
|
||||
created_at="2024-01-15T10:00:00Z",
|
||||
updated_at="2024-01-15T10:00:00Z"
|
||||
)
|
||||
|
||||
with patch.object(wiki_service, 'create_page', new_callable=AsyncMock) as mock_create:
|
||||
mock_create.return_value = mock_page
|
||||
|
||||
await wiki_service.smart_create_page(
|
||||
topic="Docker Compose Tutorial",
|
||||
user="testuser",
|
||||
path=None, # No path provided
|
||||
tags=["tutorials"],
|
||||
hybrid_rag_service=mock_hybrid_rag_service,
|
||||
wiki_page_writer=mock_wiki_page_writer
|
||||
)
|
||||
|
||||
# Check that create_page was called
|
||||
mock_create.assert_called_once()
|
||||
# The WikiPageCreate passed should have auto-generated path
|
||||
call_args = mock_create.call_args[0][0] # First positional arg
|
||||
assert "docker-compose-tutorial" in call_args.path.lower()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_smart_create_respects_web_flag(
|
||||
self,
|
||||
mock_hybrid_rag_service,
|
||||
mock_wiki_page_writer
|
||||
):
|
||||
"""Test that include_web flag is passed to HybridRAG."""
|
||||
from src.services.wiki_service import WikiService
|
||||
|
||||
mock_wiki_client = AsyncMock()
|
||||
wiki_service = WikiService(mock_wiki_client)
|
||||
|
||||
mock_page = WikiPage(
|
||||
id=1,
|
||||
path="/test",
|
||||
title="Test",
|
||||
content="Content",
|
||||
tags=[],
|
||||
is_published=True,
|
||||
created_at="2024-01-15T10:00:00Z",
|
||||
updated_at="2024-01-15T10:00:00Z"
|
||||
)
|
||||
|
||||
with patch.object(wiki_service, 'create_page', new_callable=AsyncMock) as mock_create:
|
||||
mock_create.return_value = mock_page
|
||||
|
||||
await wiki_service.smart_create_page(
|
||||
topic="Test",
|
||||
user="testuser",
|
||||
path="/test",
|
||||
tags=[],
|
||||
hybrid_rag_service=mock_hybrid_rag_service,
|
||||
wiki_page_writer=mock_wiki_page_writer,
|
||||
include_web=False,
|
||||
include_wiki=True
|
||||
)
|
||||
|
||||
# Check HybridRAG config
|
||||
call_args = mock_hybrid_rag_service.search.call_args
|
||||
config = call_args[1]["config"]
|
||||
assert config.enable_web is False
|
||||
assert config.enable_vector is True
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_smart_create_counts_sources(
|
||||
self,
|
||||
mock_hybrid_rag_service,
|
||||
mock_wiki_page_writer
|
||||
):
|
||||
"""Test that sources are counted correctly."""
|
||||
from src.services.wiki_service import WikiService
|
||||
|
||||
mock_wiki_client = AsyncMock()
|
||||
wiki_service = WikiService(mock_wiki_client)
|
||||
|
||||
mock_page = WikiPage(
|
||||
id=1,
|
||||
path="/test",
|
||||
title="Test",
|
||||
content="Content",
|
||||
tags=[],
|
||||
is_published=True,
|
||||
created_at="2024-01-15T10:00:00Z",
|
||||
updated_at="2024-01-15T10:00:00Z"
|
||||
)
|
||||
|
||||
with patch.object(wiki_service, 'create_page', new_callable=AsyncMock) as mock_create:
|
||||
mock_create.return_value = mock_page
|
||||
|
||||
page, research_data = await wiki_service.smart_create_page(
|
||||
topic="Test",
|
||||
user="testuser",
|
||||
path="/test",
|
||||
tags=[],
|
||||
hybrid_rag_service=mock_hybrid_rag_service,
|
||||
wiki_page_writer=mock_wiki_page_writer
|
||||
)
|
||||
|
||||
# Should have 2 sources (1 wiki + 1 web from mock)
|
||||
assert research_data["sources_used"] == 2
|
||||
assert research_data["research_summary"]["wiki_results"] == 1
|
||||
assert research_data["research_summary"]["web_results"] == 1
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Entity Linking Utils Tests
|
||||
# =============================================================================
|
||||
|
||||
class TestBidirectionalEntityLinking:
|
||||
"""Tests for entity_linking_utils.apply_bidirectional_entity_linking."""
|
||||
|
||||
@pytest.fixture
|
||||
def mock_neo4j_client(self):
|
||||
"""Mock Neo4j client."""
|
||||
client = AsyncMock()
|
||||
client.execute_query = AsyncMock(return_value=[])
|
||||
return client
|
||||
|
||||
@pytest.fixture
|
||||
def mock_wiki_service(self):
|
||||
"""Mock WikiService."""
|
||||
service = AsyncMock()
|
||||
return service
|
||||
|
||||
@pytest.fixture
|
||||
def mock_ingestion_service(self):
|
||||
"""Mock IngestionService."""
|
||||
service = AsyncMock()
|
||||
return service
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_returns_link_counts(
|
||||
self,
|
||||
mock_neo4j_client,
|
||||
mock_wiki_service,
|
||||
mock_ingestion_service
|
||||
):
|
||||
"""Test that function returns proper link count structure."""
|
||||
from src.services.entity_linking_utils import apply_bidirectional_entity_linking
|
||||
|
||||
# Patch at the import location within the module
|
||||
with patch('src.routers.entity_linking.link_entities_in_page') as mock_link:
|
||||
mock_result = MagicMock()
|
||||
mock_result.content_links_added = 3
|
||||
mock_link.return_value = mock_result
|
||||
|
||||
with patch('src.core.dependencies.get_graph_service'):
|
||||
with patch('src.core.dependencies.get_ingestion_service', return_value=mock_ingestion_service):
|
||||
result = await apply_bidirectional_entity_linking(
|
||||
page_id=42,
|
||||
page_title="Docker",
|
||||
user="testuser",
|
||||
neo4j_client=mock_neo4j_client,
|
||||
wiki_service=mock_wiki_service,
|
||||
ingestion_service=mock_ingestion_service
|
||||
)
|
||||
|
||||
assert "forward_links" in result
|
||||
assert "backward_links" in result
|
||||
assert "pages_updated" in result
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_handles_no_reverse_references(
|
||||
self,
|
||||
mock_neo4j_client,
|
||||
mock_wiki_service,
|
||||
mock_ingestion_service
|
||||
):
|
||||
"""Test graceful handling when no reverse references found."""
|
||||
from src.services.entity_linking_utils import apply_bidirectional_entity_linking
|
||||
|
||||
# No reverse references
|
||||
mock_neo4j_client.execute_query = AsyncMock(return_value=[])
|
||||
|
||||
with patch('src.routers.entity_linking.link_entities_in_page') as mock_link:
|
||||
mock_result = MagicMock()
|
||||
mock_result.content_links_added = 2
|
||||
mock_link.return_value = mock_result
|
||||
|
||||
with patch('src.core.dependencies.get_graph_service'):
|
||||
with patch('src.core.dependencies.get_ingestion_service', return_value=mock_ingestion_service):
|
||||
result = await apply_bidirectional_entity_linking(
|
||||
page_id=42,
|
||||
page_title="NewEntity",
|
||||
user="testuser",
|
||||
neo4j_client=mock_neo4j_client,
|
||||
wiki_service=mock_wiki_service,
|
||||
ingestion_service=mock_ingestion_service
|
||||
)
|
||||
|
||||
assert result["backward_links"] == 0
|
||||
assert result["pages_updated"] == 0
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_handles_errors_gracefully(
|
||||
self,
|
||||
mock_neo4j_client,
|
||||
mock_wiki_service,
|
||||
mock_ingestion_service
|
||||
):
|
||||
"""Test that errors don't crash the function."""
|
||||
from src.services.entity_linking_utils import apply_bidirectional_entity_linking
|
||||
|
||||
with patch('src.routers.entity_linking.link_entities_in_page') as mock_link:
|
||||
mock_link.side_effect = Exception("Test error")
|
||||
|
||||
with patch('src.core.dependencies.get_graph_service'):
|
||||
with patch('src.core.dependencies.get_ingestion_service', return_value=mock_ingestion_service):
|
||||
result = await apply_bidirectional_entity_linking(
|
||||
page_id=42,
|
||||
page_title="Test",
|
||||
user="testuser",
|
||||
neo4j_client=mock_neo4j_client,
|
||||
wiki_service=mock_wiki_service,
|
||||
ingestion_service=mock_ingestion_service
|
||||
)
|
||||
|
||||
# Should return zeros, not raise
|
||||
assert result["forward_links"] == 0
|
||||
assert result["backward_links"] == 0
|
||||
assert result["pages_updated"] == 0
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Endpoint Tests
|
||||
# =============================================================================
|
||||
|
||||
class TestSmartCreateEndpoint:
|
||||
"""Tests for POST /wiki/pages/smart-create endpoint."""
|
||||
|
||||
@pytest.fixture
|
||||
def mock_clients(self):
|
||||
"""Create all mock clients needed for the endpoint."""
|
||||
return {
|
||||
"wiki_client": AsyncMock(),
|
||||
"neo4j_client": AsyncMock(),
|
||||
"qdrant_client": MagicMock(),
|
||||
"ollama_client": AsyncMock(),
|
||||
"searxng_client": AsyncMock()
|
||||
}
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_endpoint_returns_201(self, mock_clients):
|
||||
"""Test that successful creation returns 201 status."""
|
||||
from fastapi.testclient import TestClient
|
||||
from unittest.mock import patch
|
||||
|
||||
# This test would require more setup with FastAPI TestClient
|
||||
# For now, we test the model validation
|
||||
request = WikiSmartCreateRequest(
|
||||
topic="Test Topic",
|
||||
tags=["test"]
|
||||
)
|
||||
assert request.topic == "Test Topic"
|
||||
|
||||
def test_request_validation_rejects_empty_topic(self):
|
||||
"""Test that empty topic is rejected."""
|
||||
with pytest.raises(ValueError):
|
||||
WikiSmartCreateRequest(topic="")
|
||||
|
||||
def test_request_accepts_minimal_input(self):
|
||||
"""Test that only topic is required."""
|
||||
request = WikiSmartCreateRequest(topic="Minimal test")
|
||||
assert request.topic == "Minimal test"
|
||||
assert request.include_web_research is True # default
|
||||
assert request.include_wiki_search is True # default
|
||||
@@ -0,0 +1,567 @@
|
||||
"""
|
||||
Tests for volatile cache router and service (Qdrant backend).
|
||||
|
||||
Tests:
|
||||
- Volatile record CRUD operations
|
||||
- Namespace listing and management
|
||||
- Scheduled record retrieval
|
||||
- TTL behavior and expiry filtering
|
||||
- Semantic search
|
||||
- Natural language conversion
|
||||
"""
|
||||
|
||||
import pytest
|
||||
from datetime import datetime
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
from src.models.volatile import (
|
||||
VolatileRecord,
|
||||
VolatileRecordCreate,
|
||||
VolatileRecordResponse,
|
||||
VolatileListResponse,
|
||||
VolatileScheduledResponse,
|
||||
VolatileStatsResponse,
|
||||
VolatileDeleteResponse,
|
||||
VolatileBulkDeleteResponse,
|
||||
VolatileNamespace,
|
||||
NAMESPACE_DEFAULT_TTL,
|
||||
)
|
||||
|
||||
|
||||
class TestVolatileModels:
|
||||
"""Test volatile data models."""
|
||||
|
||||
def test_volatile_record_creation(self):
|
||||
"""Test VolatileRecord model creation."""
|
||||
record = VolatileRecord(
|
||||
key="rotterdam",
|
||||
namespace="weather",
|
||||
data={"temperature": 18, "conditions": "Cloudy"},
|
||||
source="openweathermap",
|
||||
ttl=1800,
|
||||
user="jpmschweitzer",
|
||||
)
|
||||
assert record.key == "rotterdam"
|
||||
assert record.namespace == "weather"
|
||||
assert record.data["temperature"] == 18
|
||||
assert record.ttl == 1800
|
||||
assert record.refresh_schedule is None
|
||||
|
||||
def test_volatile_record_with_schedule(self):
|
||||
"""Test VolatileRecord with refresh schedule."""
|
||||
record = VolatileRecord(
|
||||
key="nos-headlines",
|
||||
namespace="news",
|
||||
data={"headlines": ["Test headline"]},
|
||||
source="nos.nl",
|
||||
ttl=3600,
|
||||
refresh_schedule="0 * * * *",
|
||||
user="jpmschweitzer",
|
||||
)
|
||||
assert record.refresh_schedule == "0 * * * *"
|
||||
|
||||
def test_volatile_record_create(self):
|
||||
"""Test VolatileRecordCreate model."""
|
||||
create = VolatileRecordCreate(
|
||||
data={"price": 150.50, "change": 2.3},
|
||||
source="alpha_vantage",
|
||||
ttl=300,
|
||||
)
|
||||
assert create.data["price"] == 150.50
|
||||
assert create.ttl == 300
|
||||
|
||||
def test_volatile_record_response(self):
|
||||
"""Test VolatileRecordResponse model."""
|
||||
response = VolatileRecordResponse(
|
||||
key="rotterdam",
|
||||
namespace="weather",
|
||||
data={"temperature": 18},
|
||||
source="openweathermap",
|
||||
created_at=datetime.utcnow(),
|
||||
updated_at=datetime.utcnow(),
|
||||
ttl=1800,
|
||||
ttl_remaining=1500,
|
||||
user="jpmschweitzer",
|
||||
)
|
||||
assert response.ttl_remaining == 1500
|
||||
assert response.ttl == 1800
|
||||
|
||||
|
||||
class TestVolatileNamespaces:
|
||||
"""Test volatile namespaces and defaults."""
|
||||
|
||||
def test_all_namespaces_have_default_ttl(self):
|
||||
"""Verify all namespaces have default TTLs defined."""
|
||||
for ns in VolatileNamespace:
|
||||
assert ns in NAMESPACE_DEFAULT_TTL, f"Missing TTL for {ns}"
|
||||
assert NAMESPACE_DEFAULT_TTL[ns] > 0
|
||||
|
||||
def test_weather_default_ttl(self):
|
||||
"""Test weather namespace default TTL."""
|
||||
assert NAMESPACE_DEFAULT_TTL[VolatileNamespace.WEATHER] == 1800 # 30 min
|
||||
|
||||
def test_financial_default_ttl(self):
|
||||
"""Test financial namespace default TTL."""
|
||||
assert NAMESPACE_DEFAULT_TTL[VolatileNamespace.FINANCIAL] == 300 # 5 min
|
||||
|
||||
def test_sports_default_ttl(self):
|
||||
"""Test sports namespace default TTL (fast updates)."""
|
||||
assert NAMESPACE_DEFAULT_TTL[VolatileNamespace.SPORTS] == 60 # 1 min
|
||||
|
||||
def test_namespace_count(self):
|
||||
"""Test we have the expected number of namespaces."""
|
||||
assert len(VolatileNamespace) == 11
|
||||
|
||||
|
||||
class TestVolatileListResponse:
|
||||
"""Test list response models."""
|
||||
|
||||
def test_list_response(self):
|
||||
"""Test VolatileListResponse model."""
|
||||
response = VolatileListResponse(
|
||||
namespace="weather",
|
||||
keys=["rotterdam", "amsterdam", "utrecht"],
|
||||
count=3,
|
||||
user="jpmschweitzer",
|
||||
)
|
||||
assert response.count == 3
|
||||
assert "rotterdam" in response.keys
|
||||
|
||||
|
||||
class TestVolatileScheduledResponse:
|
||||
"""Test scheduled records response."""
|
||||
|
||||
def test_scheduled_response_empty(self):
|
||||
"""Test empty scheduled response."""
|
||||
response = VolatileScheduledResponse(
|
||||
records=[],
|
||||
count=0,
|
||||
user="jpmschweitzer",
|
||||
)
|
||||
assert response.count == 0
|
||||
assert response.records == []
|
||||
|
||||
def test_scheduled_response_with_records(self):
|
||||
"""Test scheduled response with records."""
|
||||
record = VolatileRecordResponse(
|
||||
key="nos-headlines",
|
||||
namespace="news",
|
||||
data={"headlines": []},
|
||||
source="nos.nl",
|
||||
created_at=datetime.utcnow(),
|
||||
updated_at=datetime.utcnow(),
|
||||
ttl=3600,
|
||||
ttl_remaining=3000,
|
||||
refresh_schedule="0 */6 * * *",
|
||||
user="jpmschweitzer",
|
||||
)
|
||||
response = VolatileScheduledResponse(
|
||||
records=[record],
|
||||
count=1,
|
||||
user="jpmschweitzer",
|
||||
)
|
||||
assert response.count == 1
|
||||
assert response.records[0].refresh_schedule == "0 */6 * * *"
|
||||
|
||||
|
||||
class TestVolatileStatsResponse:
|
||||
"""Test stats response model."""
|
||||
|
||||
def test_stats_response(self):
|
||||
"""Test VolatileStatsResponse model."""
|
||||
response = VolatileStatsResponse(
|
||||
total_records=15,
|
||||
by_namespace={"weather": 3, "news": 5, "financial": 7},
|
||||
scheduled_count=2,
|
||||
total_memory_bytes=None,
|
||||
user="jpmschweitzer",
|
||||
)
|
||||
assert response.total_records == 15
|
||||
assert response.by_namespace["weather"] == 3
|
||||
assert response.scheduled_count == 2
|
||||
|
||||
|
||||
class TestVolatileDeleteResponses:
|
||||
"""Test delete response models."""
|
||||
|
||||
def test_delete_response(self):
|
||||
"""Test VolatileDeleteResponse model."""
|
||||
response = VolatileDeleteResponse(
|
||||
key="rotterdam",
|
||||
namespace="weather",
|
||||
deleted=True,
|
||||
user="jpmschweitzer",
|
||||
)
|
||||
assert response.deleted is True
|
||||
|
||||
def test_delete_not_found(self):
|
||||
"""Test delete response when record not found."""
|
||||
response = VolatileDeleteResponse(
|
||||
key="nonexistent",
|
||||
namespace="weather",
|
||||
deleted=False,
|
||||
user="jpmschweitzer",
|
||||
)
|
||||
assert response.deleted is False
|
||||
|
||||
def test_bulk_delete_response(self):
|
||||
"""Test VolatileBulkDeleteResponse model."""
|
||||
response = VolatileBulkDeleteResponse(
|
||||
namespace="weather",
|
||||
deleted_count=5,
|
||||
user="jpmschweitzer",
|
||||
)
|
||||
assert response.deleted_count == 5
|
||||
assert response.namespace == "weather"
|
||||
|
||||
|
||||
class TestVolatileService:
|
||||
"""Test VolatileCacheService functionality (Qdrant backend)."""
|
||||
|
||||
@pytest.fixture
|
||||
def mock_qdrant(self):
|
||||
"""Create mock Qdrant client."""
|
||||
qdrant = AsyncMock()
|
||||
qdrant.ensure_collection = AsyncMock()
|
||||
qdrant.collection_exists = AsyncMock(return_value=True)
|
||||
qdrant.upsert_vector = AsyncMock(return_value=True)
|
||||
qdrant.delete_by_ids = AsyncMock(return_value=1)
|
||||
qdrant.search_with_expiry_filter = AsyncMock(return_value=[])
|
||||
qdrant.scroll_all_points = AsyncMock(return_value=[])
|
||||
qdrant.delete_expired_vectors = AsyncMock(return_value=0)
|
||||
qdrant.get_volatile_collections = AsyncMock(return_value=[])
|
||||
return qdrant
|
||||
|
||||
@pytest.fixture
|
||||
def mock_ollama(self):
|
||||
"""Create mock Ollama client."""
|
||||
ollama = AsyncMock()
|
||||
ollama.embed = AsyncMock(return_value=[0.1] * 768) # Return 768-dim embedding
|
||||
return ollama
|
||||
|
||||
@pytest.fixture
|
||||
def mock_settings(self):
|
||||
"""Create mock settings."""
|
||||
settings = MagicMock()
|
||||
settings.volatile_default_ttl = 3600
|
||||
return settings
|
||||
|
||||
@pytest.fixture
|
||||
def volatile_service(self, mock_qdrant, mock_ollama, mock_settings):
|
||||
"""Create VolatileCacheService with mocks."""
|
||||
from src.services.volatile_service import VolatileCacheService
|
||||
return VolatileCacheService(
|
||||
qdrant_client=mock_qdrant,
|
||||
ollama_client=mock_ollama,
|
||||
settings=mock_settings
|
||||
)
|
||||
|
||||
def test_collection_name(self, volatile_service):
|
||||
"""Test collection naming pattern."""
|
||||
name = volatile_service._collection_name("jpmschweitzer")
|
||||
assert name == "volatile_jpmschweitzer"
|
||||
|
||||
def test_make_vector_id(self, volatile_service):
|
||||
"""Test deterministic vector ID generation."""
|
||||
id1 = volatile_service._make_vector_id("weather", "rotterdam")
|
||||
id2 = volatile_service._make_vector_id("weather", "rotterdam")
|
||||
id3 = volatile_service._make_vector_id("weather", "amsterdam")
|
||||
|
||||
assert id1 == id2 # Same namespace+key = same ID
|
||||
assert id1 != id3 # Different key = different ID
|
||||
assert len(id1) == 32 # MD5 hex length
|
||||
|
||||
def test_get_default_ttl_known_namespace(self, volatile_service):
|
||||
"""Test default TTL for known namespace."""
|
||||
ttl = volatile_service._get_default_ttl("weather")
|
||||
assert ttl == 1800 # Weather namespace default
|
||||
|
||||
def test_get_default_ttl_unknown_namespace(self, volatile_service):
|
||||
"""Test default TTL for unknown namespace."""
|
||||
ttl = volatile_service._get_default_ttl("unknown_namespace")
|
||||
assert ttl == 3600 # Falls back to settings default
|
||||
|
||||
def test_to_natural_language_weather(self, volatile_service):
|
||||
"""Test natural language conversion for weather data."""
|
||||
text = volatile_service._to_natural_language(
|
||||
namespace="weather",
|
||||
key="rotterdam",
|
||||
data={"temperature": 18, "conditions": "Cloudy", "humidity": 75}
|
||||
)
|
||||
assert "rotterdam" in text.lower()
|
||||
assert "18" in text
|
||||
assert "Cloudy" in text
|
||||
assert "75" in text
|
||||
|
||||
def test_to_natural_language_news(self, volatile_service):
|
||||
"""Test natural language conversion for news data."""
|
||||
text = volatile_service._to_natural_language(
|
||||
namespace="news",
|
||||
key="nos-headlines",
|
||||
data={"title": "Breaking News", "summary": "Something happened", "source": "NOS"}
|
||||
)
|
||||
assert "Breaking News" in text
|
||||
assert "Something happened" in text
|
||||
assert "NOS" in text
|
||||
|
||||
def test_to_natural_language_financial(self, volatile_service):
|
||||
"""Test natural language conversion for financial data."""
|
||||
text = volatile_service._to_natural_language(
|
||||
namespace="financial",
|
||||
key="AAPL",
|
||||
data={"symbol": "AAPL", "price": 150.50, "change": 2.3}
|
||||
)
|
||||
assert "AAPL" in text
|
||||
assert "price" in text.lower()
|
||||
assert "change" in text.lower()
|
||||
|
||||
def test_to_natural_language_transit(self, volatile_service):
|
||||
"""Test natural language conversion for transit data."""
|
||||
text = volatile_service._to_natural_language(
|
||||
namespace="transit",
|
||||
key="ns-intercity",
|
||||
data={"route": "Amsterdam-Rotterdam", "status": "On time", "delay": 0}
|
||||
)
|
||||
assert "Amsterdam-Rotterdam" in text or "ns-intercity" in text.lower()
|
||||
assert "On time" in text
|
||||
|
||||
def test_to_natural_language_fallback(self, volatile_service):
|
||||
"""Test natural language fallback for unknown namespace."""
|
||||
text = volatile_service._to_natural_language(
|
||||
namespace="custom",
|
||||
key="test-key",
|
||||
data={"foo": "bar", "count": 42}
|
||||
)
|
||||
assert "custom" in text.lower()
|
||||
assert "foo" in text or "bar" in text
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_store_success(self, volatile_service, mock_qdrant, mock_ollama):
|
||||
"""Test successful store operation."""
|
||||
result = await volatile_service.store(
|
||||
user="jpmschweitzer",
|
||||
namespace="weather",
|
||||
key="rotterdam",
|
||||
data={"temperature": 18, "conditions": "Sunny"},
|
||||
source="openweathermap",
|
||||
ttl=1800
|
||||
)
|
||||
|
||||
assert result.key == "rotterdam"
|
||||
assert result.namespace == "weather"
|
||||
assert result.ttl == 1800
|
||||
mock_qdrant.ensure_collection.assert_called_once()
|
||||
mock_ollama.embed.assert_called_once()
|
||||
mock_qdrant.upsert_vector.assert_called_once()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_store_uses_namespace_default_ttl(self, volatile_service, mock_qdrant, mock_ollama):
|
||||
"""Test store uses namespace default TTL when not specified."""
|
||||
result = await volatile_service.store(
|
||||
user="jpmschweitzer",
|
||||
namespace="weather",
|
||||
key="amsterdam",
|
||||
data={"temperature": 16},
|
||||
source="openweathermap",
|
||||
ttl=None # Not specified
|
||||
)
|
||||
|
||||
assert result.ttl == 1800 # Weather default
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_search_empty_collection(self, volatile_service, mock_qdrant, mock_ollama):
|
||||
"""Test search when collection doesn't exist."""
|
||||
mock_qdrant.collection_exists.return_value = False
|
||||
|
||||
results = await volatile_service.search(
|
||||
user="jpmschweitzer",
|
||||
query="weather rotterdam"
|
||||
)
|
||||
|
||||
assert results == []
|
||||
mock_ollama.embed.assert_not_called()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_search_with_results(self, volatile_service, mock_qdrant, mock_ollama):
|
||||
"""Test search returns results."""
|
||||
import time
|
||||
now_ms = int(time.time() * 1000)
|
||||
|
||||
mock_qdrant.search_with_expiry_filter.return_value = [
|
||||
{
|
||||
"score": 0.95,
|
||||
"payload": {
|
||||
"key": "rotterdam",
|
||||
"namespace": "weather",
|
||||
"raw_data": {"temperature": 18},
|
||||
"source": "openweathermap",
|
||||
"created_at": datetime.utcnow().isoformat(),
|
||||
"updated_at": datetime.utcnow().isoformat(),
|
||||
"ttl": 1800,
|
||||
"ttl_expiry": now_ms + 900000, # 15 min remaining
|
||||
"refresh_schedule": None,
|
||||
"user": "jpmschweitzer"
|
||||
}
|
||||
}
|
||||
]
|
||||
|
||||
results = await volatile_service.search(
|
||||
user="jpmschweitzer",
|
||||
query="weather rotterdam"
|
||||
)
|
||||
|
||||
assert len(results) == 1
|
||||
assert results[0].key == "rotterdam"
|
||||
assert results[0].namespace == "weather"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_success(self, volatile_service, mock_qdrant):
|
||||
"""Test successful delete."""
|
||||
mock_qdrant.delete_by_ids.return_value = 1
|
||||
|
||||
result = await volatile_service.delete("jpmschweitzer", "weather", "rotterdam")
|
||||
|
||||
assert result is True
|
||||
mock_qdrant.delete_by_ids.assert_called_once()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_not_found(self, volatile_service, mock_qdrant):
|
||||
"""Test delete when record not found."""
|
||||
mock_qdrant.delete_by_ids.return_value = 0
|
||||
|
||||
result = await volatile_service.delete("jpmschweitzer", "weather", "nonexistent")
|
||||
|
||||
assert result is False
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_stats_empty(self, volatile_service, mock_qdrant):
|
||||
"""Test stats with no records."""
|
||||
mock_qdrant.collection_exists.return_value = False
|
||||
|
||||
stats = await volatile_service.get_stats("jpmschweitzer")
|
||||
|
||||
assert stats["total_records"] == 0
|
||||
assert stats["by_namespace"] == {}
|
||||
assert stats["scheduled_count"] == 0
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_stats_with_records(self, volatile_service, mock_qdrant):
|
||||
"""Test stats with records."""
|
||||
import time
|
||||
now_ms = int(time.time() * 1000)
|
||||
|
||||
mock_qdrant.scroll_all_points.return_value = [
|
||||
{"payload": {"namespace": "weather", "ttl_expiry": now_ms + 100000}},
|
||||
{"payload": {"namespace": "weather", "ttl_expiry": now_ms + 100000, "refresh_schedule": "0 * * * *"}},
|
||||
{"payload": {"namespace": "news", "ttl_expiry": now_ms + 100000}},
|
||||
{"payload": {"namespace": "weather", "ttl_expiry": now_ms - 100000}}, # Expired
|
||||
]
|
||||
|
||||
stats = await volatile_service.get_stats("jpmschweitzer")
|
||||
|
||||
assert stats["total_records"] == 3 # Excludes expired
|
||||
assert stats["by_namespace"]["weather"] == 2
|
||||
assert stats["by_namespace"]["news"] == 1
|
||||
assert stats["scheduled_count"] == 1
|
||||
assert stats["expired_count"] == 1
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_purge_expired(self, volatile_service, mock_qdrant):
|
||||
"""Test purging expired records."""
|
||||
mock_qdrant.delete_expired_vectors.return_value = 5
|
||||
|
||||
result = await volatile_service.purge_expired("jpmschweitzer")
|
||||
|
||||
assert result == 5
|
||||
mock_qdrant.delete_expired_vectors.assert_called_once()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_purge_all_expired(self, volatile_service, mock_qdrant):
|
||||
"""Test purging expired from all collections."""
|
||||
mock_qdrant.get_volatile_collections.return_value = [
|
||||
"volatile_user1",
|
||||
"volatile_user2"
|
||||
]
|
||||
mock_qdrant.delete_expired_vectors.side_effect = [3, 2]
|
||||
|
||||
results = await volatile_service.purge_all_expired()
|
||||
|
||||
assert results["volatile_user1"] == 3
|
||||
assert results["volatile_user2"] == 2
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_scheduled(self, volatile_service, mock_qdrant):
|
||||
"""Test getting scheduled records."""
|
||||
import time
|
||||
now_ms = int(time.time() * 1000)
|
||||
|
||||
mock_qdrant.scroll_all_points.return_value = [
|
||||
{
|
||||
"payload": {
|
||||
"key": "nos-headlines",
|
||||
"namespace": "news",
|
||||
"raw_data": {"headlines": []},
|
||||
"source": "nos.nl",
|
||||
"created_at": datetime.utcnow().isoformat(),
|
||||
"updated_at": datetime.utcnow().isoformat(),
|
||||
"ttl": 3600,
|
||||
"ttl_expiry": now_ms + 1800000,
|
||||
"refresh_schedule": "0 */6 * * *",
|
||||
"user": "jpmschweitzer"
|
||||
}
|
||||
},
|
||||
{
|
||||
"payload": {
|
||||
"key": "rotterdam",
|
||||
"namespace": "weather",
|
||||
"raw_data": {"temperature": 18},
|
||||
"source": "openweathermap",
|
||||
"created_at": datetime.utcnow().isoformat(),
|
||||
"updated_at": datetime.utcnow().isoformat(),
|
||||
"ttl": 1800,
|
||||
"ttl_expiry": now_ms + 900000,
|
||||
"refresh_schedule": None, # Not scheduled
|
||||
"user": "jpmschweitzer"
|
||||
}
|
||||
}
|
||||
]
|
||||
|
||||
scheduled = await volatile_service.get_scheduled("jpmschweitzer")
|
||||
|
||||
assert len(scheduled) == 1
|
||||
assert scheduled[0].key == "nos-headlines"
|
||||
assert scheduled[0].refresh_schedule == "0 */6 * * *"
|
||||
|
||||
|
||||
class TestVolatileCleanupEndpoint:
|
||||
"""Test volatile cleanup in maintenance router."""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_cleanup_volatile(self):
|
||||
"""Test volatile cleanup endpoint."""
|
||||
from src.routers.maintenance import cleanup_volatile, VolatileCleanupResponse
|
||||
|
||||
mock_qdrant = AsyncMock()
|
||||
mock_qdrant.get_volatile_collections = AsyncMock(return_value=[
|
||||
"volatile_user1",
|
||||
"volatile_user2"
|
||||
])
|
||||
mock_qdrant.delete_expired_vectors = AsyncMock(side_effect=[3, 2])
|
||||
|
||||
mock_ollama = AsyncMock()
|
||||
|
||||
mock_settings = MagicMock()
|
||||
mock_settings.volatile_default_ttl = 3600
|
||||
|
||||
with patch('src.routers.maintenance.get_settings', return_value=mock_settings):
|
||||
result = await cleanup_volatile(
|
||||
qdrant=mock_qdrant,
|
||||
ollama=mock_ollama,
|
||||
api_key="test"
|
||||
)
|
||||
|
||||
assert result.success is True
|
||||
assert result.collections_processed == 2
|
||||
assert result.total_expired_purged == 5
|
||||
assert result.by_collection["volatile_user1"] == 3
|
||||
assert result.by_collection["volatile_user2"] == 2
|
||||
@@ -0,0 +1,50 @@
|
||||
|
||||
|
||||
#!/bin/bash
|
||||
# Library-Desk Server Startup Script
|
||||
|
||||
set -e
|
||||
|
||||
# Colors for output
|
||||
GREEN='\033[0;32m'
|
||||
YELLOW='\033[1;33m'
|
||||
RED='\033[0;31m'
|
||||
NC='\033[0m' # No Color
|
||||
|
||||
echo -e "${GREEN}Starting Library-Desk server...${NC}"
|
||||
|
||||
# Check if port 8778 is already in use
|
||||
if lsof -Pi :8778 -sTCP:LISTEN -t >/dev/null 2>&1 ; then
|
||||
echo -e "${RED}Error: Port 8778 is already in use${NC}"
|
||||
echo "Run: lsof -i :8778 to see what's using it"
|
||||
echo "Or run: kill \$(lsof -t -i:8778) to stop it"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Activate virtual environment if not already activated
|
||||
if [ -z "$VIRTUAL_ENV" ]; then
|
||||
if [ -d ".venv" ]; then
|
||||
echo -e "${YELLOW}Activating virtual environment...${NC}"
|
||||
source .venv/bin/activate
|
||||
else
|
||||
echo -e "${RED}Error: Virtual environment not found${NC}"
|
||||
echo "Run: python -m venv .venv && source .venv/bin/activate && pip install -r requirements.txt"
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
# Create logs directory if it doesn't exist
|
||||
LOGS_DIR="logs"
|
||||
mkdir -p "$LOGS_DIR"
|
||||
|
||||
# Clear/create log file
|
||||
LOG_FILE="$LOGS_DIR/server.log"
|
||||
> "$LOG_FILE"
|
||||
echo -e "${YELLOW}Logs will be written to: ${LOG_FILE}${NC}"
|
||||
|
||||
# Start the server
|
||||
echo -e "${GREEN}Starting uvicorn server on http://tower-of-joy:8778${NC}"
|
||||
echo -e "${YELLOW}Press Ctrl+C to stop the server${NC}"
|
||||
echo ""
|
||||
|
||||
uvicorn src.main:app --reload --host 0.0.0.0 --port 8778 2>&1 | tee "$LOG_FILE"
|
||||
Reference in New Issue
Block a user