feat(library-desk): add knowledge consolidation system
Consolidation Service: - Automated wiki page creation from research results - Entity extraction and disambiguation - Multi-source fact integration - Confidence scoring and source citation - Template-based page generation - Schema.org taxonomy integration Wiki Page Writer: - Structured markdown generation - Standard templates (person, organization, technology) - Metadata formatting (tags, categories, timestamps) - Citation and source linking Router: - POST /consolidate/research endpoint - Batch consolidation support - Manual and automated triggers Models: - ConsolidationRequest with source data - ConsolidationResult with page details - Entity resolution metadata Tests: - Page generation validation - Entity extraction accuracy - Multi-source merging logic
This commit is contained in:
@@ -0,0 +1,49 @@
|
||||
"""
|
||||
Knowledge Consolidation models for Librarian processing.
|
||||
|
||||
Used by the consolidation endpoint to process SearchQuery nodes
|
||||
and consolidate knowledge into wiki pages.
|
||||
"""
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import List, Optional, Dict, Any
|
||||
|
||||
|
||||
class ConsolidationRequest(BaseModel):
|
||||
"""Request for knowledge consolidation from search results."""
|
||||
process_limit: int = Field(default=10, ge=1, le=100, description="Max searches to process")
|
||||
lookback_days: int = Field(default=7, ge=1, le=90, description="Process searches from last N days")
|
||||
min_web_results: int = Field(default=2, ge=1, le=20, description="Minimum web results needed")
|
||||
dry_run: bool = Field(default=False, description="If true, analyze but don't create pages")
|
||||
|
||||
|
||||
class SearchQueryInfo(BaseModel):
|
||||
"""Information about a search query to process."""
|
||||
id: str
|
||||
query: str
|
||||
user: str
|
||||
timestamp: str
|
||||
total_results: int
|
||||
web_count: int
|
||||
keywords: List[str] = []
|
||||
|
||||
|
||||
class ConsolidationResult(BaseModel):
|
||||
"""Result of processing a single search query."""
|
||||
search_id: str
|
||||
query: str
|
||||
pages_created: int = 0
|
||||
pages_updated: int = 0
|
||||
entities_added: int = 0
|
||||
error: Optional[str] = None
|
||||
|
||||
|
||||
class ConsolidationResponse(BaseModel):
|
||||
"""Response from knowledge consolidation."""
|
||||
total_found: int = Field(description="Total unprocessed searches found")
|
||||
processed_count: int = Field(description="Successfully processed searches")
|
||||
pages_created: int = Field(description="New wiki pages created")
|
||||
pages_updated: int = Field(description="Existing pages updated")
|
||||
entities_added: int = Field(description="New entities added to graph")
|
||||
errors: List[str] = Field(default=[], description="Error messages")
|
||||
results: List[ConsolidationResult] = Field(description="Per-search results")
|
||||
dry_run: bool = Field(description="Whether this was a dry run")
|
||||
@@ -0,0 +1,159 @@
|
||||
"""
|
||||
Knowledge Consolidation router for Librarian processing.
|
||||
|
||||
Provides endpoints for the Scheduler to trigger knowledge consolidation
|
||||
from HybridRAG search results into wiki pages.
|
||||
"""
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Depends
|
||||
import logging
|
||||
|
||||
from src.models.consolidation import ConsolidationRequest, ConsolidationResponse
|
||||
from src.services.consolidation_service import ConsolidationService
|
||||
from src.core.dependencies import (
|
||||
Neo4jDep, OllamaDep, WikiJSDep,
|
||||
verify_api_key, get_settings
|
||||
)
|
||||
from src.config import Settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/consolidate", tags=["Consolidation"])
|
||||
|
||||
|
||||
# Dependency to get Consolidation service
|
||||
def get_consolidation_service(
|
||||
neo4j_client: Neo4jDep,
|
||||
ollama_client: OllamaDep,
|
||||
wiki_client: WikiJSDep,
|
||||
settings: Settings = Depends(get_settings)
|
||||
) -> ConsolidationService:
|
||||
"""Get ConsolidationService instance with all dependencies."""
|
||||
return ConsolidationService(
|
||||
neo4j=neo4j_client,
|
||||
ollama=ollama_client,
|
||||
wiki=wiki_client,
|
||||
settings=settings
|
||||
)
|
||||
|
||||
|
||||
@router.post("/knowledge", response_model=ConsolidationResponse)
|
||||
async def consolidate_knowledge(
|
||||
request: ConsolidationRequest,
|
||||
consolidation_service: ConsolidationService = Depends(get_consolidation_service),
|
||||
api_key: str = Depends(verify_api_key)
|
||||
):
|
||||
"""
|
||||
Consolidate knowledge from HybridRAG search results into wiki pages.
|
||||
|
||||
**Librarian Task** - Processes unprocessed SearchQuery nodes to:
|
||||
1. Find searches with web results from last N days
|
||||
2. Analyze web content with LLM for novel information
|
||||
3. Create new wiki pages for new concepts/technologies
|
||||
4. Update existing pages with new facts and citations
|
||||
5. Add new entities to knowledge graph
|
||||
6. Mark SearchQuery nodes as processed
|
||||
|
||||
**Typically called by The Scheduler** on a periodic basis (e.g., hourly).
|
||||
|
||||
**Parameters:**
|
||||
- `process_limit`: Maximum searches to process per run (default: 10)
|
||||
- `lookback_days`: Only process searches from last N days (default: 7)
|
||||
- `min_web_results`: Minimum web results required to consolidate (default: 2)
|
||||
- `dry_run`: If true, analyze but don't create pages (default: false)
|
||||
|
||||
**Returns:**
|
||||
- `total_found`: Number of unprocessed searches found
|
||||
- `processed_count`: Successfully processed searches
|
||||
- `pages_created`: New wiki pages created
|
||||
- `pages_updated`: Existing pages updated with new facts
|
||||
- `entities_added`: New entities added to knowledge graph
|
||||
- `errors`: List of error messages if any failed
|
||||
- `results`: Per-search processing results
|
||||
|
||||
**Example Request:**
|
||||
```json
|
||||
{
|
||||
"process_limit": 10,
|
||||
"lookback_days": 7,
|
||||
"min_web_results": 2,
|
||||
"dry_run": false
|
||||
}
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
```json
|
||||
{
|
||||
"total_found": 5,
|
||||
"processed_count": 4,
|
||||
"pages_created": 2,
|
||||
"pages_updated": 3,
|
||||
"entities_added": 7,
|
||||
"errors": ["Search abc123: Failed to parse response"],
|
||||
"results": [
|
||||
{
|
||||
"search_id": "uuid-1",
|
||||
"query": "docker orchestration kubernetes",
|
||||
"pages_created": 1,
|
||||
"pages_updated": 1,
|
||||
"entities_added": 3
|
||||
}
|
||||
],
|
||||
"dry_run": false
|
||||
}
|
||||
```
|
||||
|
||||
**Scheduler Task Configuration:**
|
||||
```json
|
||||
{
|
||||
"task_name": "knowledge_consolidation",
|
||||
"service": "library-desk",
|
||||
"executor": "rest_api_executor",
|
||||
"priority": 50,
|
||||
"minute": 0,
|
||||
"hour": -1,
|
||||
"description": "Hourly knowledge consolidation from search results",
|
||||
"config": {
|
||||
"url": "http://library-desk:8089/consolidate/knowledge",
|
||||
"method": "POST",
|
||||
"payload": {
|
||||
"process_limit": 10,
|
||||
"lookback_days": 7,
|
||||
"min_web_results": 2,
|
||||
"dry_run": false
|
||||
},
|
||||
"auth": {
|
||||
"type": "bearer",
|
||||
"token": "${LIBRARY_DESK_API_KEY}"
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
"""
|
||||
try:
|
||||
logger.info(
|
||||
f"Knowledge consolidation requested: "
|
||||
f"limit={request.process_limit}, lookback={request.lookback_days}d, "
|
||||
f"dry_run={request.dry_run}"
|
||||
)
|
||||
|
||||
response = await consolidation_service.consolidate_knowledge(
|
||||
process_limit=request.process_limit,
|
||||
lookback_days=request.lookback_days,
|
||||
min_web_results=request.min_web_results,
|
||||
dry_run=request.dry_run
|
||||
)
|
||||
|
||||
logger.info(
|
||||
f"Consolidation completed: {response.processed_count}/{response.total_found} searches, "
|
||||
f"{response.pages_created} pages created, {response.pages_updated} updated"
|
||||
)
|
||||
|
||||
return response
|
||||
|
||||
except ValueError as e:
|
||||
logger.error(f"Invalid request: {e}")
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
except Exception as e:
|
||||
logger.error(f"Knowledge consolidation failed: {e}", exc_info=True)
|
||||
raise HTTPException(status_code=500, detail="Consolidation failed")
|
||||
@@ -0,0 +1,702 @@
|
||||
"""
|
||||
Knowledge Consolidation Service (Librarian Logic)
|
||||
|
||||
Processes unprocessed SearchQuery nodes from HybridRAG searches
|
||||
to consolidate new knowledge into wiki pages.
|
||||
|
||||
This service:
|
||||
1. Queries Neo4j for unprocessed SearchQuery nodes
|
||||
2. Analyzes web results with Ollama for novel information
|
||||
3. Creates/updates wiki pages with new facts
|
||||
4. Updates knowledge graph with new entities
|
||||
5. Marks SearchQuery nodes as processed
|
||||
"""
|
||||
import logging
|
||||
import json
|
||||
from datetime import datetime, timedelta
|
||||
from typing import List, Dict, Any, Optional
|
||||
|
||||
from src.clients.neo4j_client import Neo4jClient
|
||||
from src.clients.ollama_client import OllamaClient
|
||||
from src.clients.wikijs_client import WikiJSClient
|
||||
from src.services.wiki_page_writer import WikiPageWriter
|
||||
from src.models.consolidation import (
|
||||
SearchQueryInfo,
|
||||
ConsolidationResult,
|
||||
ConsolidationResponse
|
||||
)
|
||||
from src.config import Settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class ConsolidationService:
|
||||
"""
|
||||
Service for consolidating knowledge from search results.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
neo4j: Neo4jClient,
|
||||
ollama: OllamaClient,
|
||||
wiki: WikiJSClient,
|
||||
settings: Settings,
|
||||
ingestion_service: Optional["IngestionService"] = None
|
||||
):
|
||||
self.neo4j = neo4j
|
||||
self.ollama = ollama
|
||||
self.wiki = wiki
|
||||
self.settings = settings
|
||||
self.wiki_page_writer = WikiPageWriter(ollama_client=ollama)
|
||||
self.ingestion_service = ingestion_service # Optional to avoid circular dependency
|
||||
|
||||
async def consolidate_knowledge(
|
||||
self,
|
||||
process_limit: int = 10,
|
||||
lookback_days: int = 7,
|
||||
min_web_results: int = 2,
|
||||
dry_run: bool = False
|
||||
) -> ConsolidationResponse:
|
||||
"""
|
||||
Process unprocessed search queries and consolidate knowledge.
|
||||
|
||||
Args:
|
||||
process_limit: Maximum searches to process
|
||||
lookback_days: Only process searches from last N days
|
||||
min_web_results: Minimum web results required to consolidate
|
||||
dry_run: If True, analyze but don't create pages
|
||||
|
||||
Returns:
|
||||
ConsolidationResponse with processing results
|
||||
"""
|
||||
logger.info(f"Starting knowledge consolidation")
|
||||
logger.info(f"Limits: process={process_limit}, lookback={lookback_days}d, min_web={min_web_results}")
|
||||
if dry_run:
|
||||
logger.warning("DRY RUN MODE - will not create wiki pages")
|
||||
|
||||
# Find unprocessed searches
|
||||
unprocessed = await self._find_unprocessed_searches(lookback_days, process_limit)
|
||||
|
||||
if not unprocessed:
|
||||
logger.info("No unprocessed searches found")
|
||||
return ConsolidationResponse(
|
||||
total_found=0,
|
||||
processed_count=0,
|
||||
pages_created=0,
|
||||
pages_updated=0,
|
||||
entities_added=0,
|
||||
errors=[],
|
||||
results=[],
|
||||
dry_run=dry_run
|
||||
)
|
||||
|
||||
logger.info(f"Found {len(unprocessed)} unprocessed searches")
|
||||
|
||||
# Process each search
|
||||
results: List[ConsolidationResult] = []
|
||||
total_pages_created = 0
|
||||
total_pages_updated = 0
|
||||
total_entities_added = 0
|
||||
errors: List[str] = []
|
||||
|
||||
for search in unprocessed:
|
||||
try:
|
||||
result = await self._process_search(
|
||||
search=search,
|
||||
min_web_results=min_web_results,
|
||||
dry_run=dry_run
|
||||
)
|
||||
|
||||
if result:
|
||||
results.append(result)
|
||||
total_pages_created += result.pages_created
|
||||
total_pages_updated += result.pages_updated
|
||||
total_entities_added += result.entities_added
|
||||
|
||||
# Mark as processed if not dry run
|
||||
if not dry_run and not result.error:
|
||||
await self._mark_search_processed(search['id'])
|
||||
|
||||
except Exception as e:
|
||||
error_msg = f"Search {search['id'][:8]}: {str(e)}"
|
||||
logger.error(f"Failed to process search: {error_msg}", exc_info=True)
|
||||
errors.append(error_msg)
|
||||
results.append(ConsolidationResult(
|
||||
search_id=search['id'],
|
||||
query=search['query'],
|
||||
error=str(e)
|
||||
))
|
||||
|
||||
# Build response
|
||||
processed_count = len([r for r in results if not r.error])
|
||||
|
||||
response = ConsolidationResponse(
|
||||
total_found=len(unprocessed),
|
||||
processed_count=processed_count,
|
||||
pages_created=total_pages_created,
|
||||
pages_updated=total_pages_updated,
|
||||
entities_added=total_entities_added,
|
||||
errors=errors,
|
||||
results=results,
|
||||
dry_run=dry_run
|
||||
)
|
||||
|
||||
logger.info(
|
||||
f"Consolidation complete: {processed_count}/{len(unprocessed)} searches, "
|
||||
f"{total_pages_created} pages created, {total_pages_updated} updated, "
|
||||
f"{total_entities_added} entities added"
|
||||
)
|
||||
|
||||
return response
|
||||
|
||||
async def _find_unprocessed_searches(
|
||||
self,
|
||||
lookback_days: int,
|
||||
limit: int
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Find unprocessed SearchQuery nodes from Neo4j.
|
||||
"""
|
||||
lookback_date = datetime.now() - timedelta(days=lookback_days)
|
||||
|
||||
query = """
|
||||
MATCH (sq:SearchQuery {processed: false})
|
||||
WHERE sq.timestamp > datetime($lookback_date)
|
||||
RETURN sq.id as id,
|
||||
sq.query as query,
|
||||
sq.user as user,
|
||||
sq.timestamp as timestamp,
|
||||
sq.total_results as total_results,
|
||||
sq.web_count as web_count,
|
||||
sq.keywords as keywords
|
||||
ORDER BY sq.timestamp DESC
|
||||
LIMIT $limit
|
||||
"""
|
||||
|
||||
try:
|
||||
results = await self.neo4j.execute_query(
|
||||
query,
|
||||
{
|
||||
"lookback_date": lookback_date.isoformat(),
|
||||
"limit": limit
|
||||
}
|
||||
)
|
||||
|
||||
searches = []
|
||||
for record in results:
|
||||
searches.append({
|
||||
'id': record['id'],
|
||||
'query': record['query'],
|
||||
'user': record['user'],
|
||||
'timestamp': record['timestamp'],
|
||||
'total_results': record.get('total_results', 0),
|
||||
'web_count': record.get('web_count', 0),
|
||||
'keywords': record.get('keywords', [])
|
||||
})
|
||||
|
||||
return searches
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to find unprocessed searches: {e}")
|
||||
return []
|
||||
|
||||
async def _process_search(
|
||||
self,
|
||||
search: Dict[str, Any],
|
||||
min_web_results: int,
|
||||
dry_run: bool
|
||||
) -> Optional[ConsolidationResult]:
|
||||
"""
|
||||
Process a single search query for knowledge consolidation.
|
||||
"""
|
||||
search_id = search['id']
|
||||
query = search['query']
|
||||
user = search['user']
|
||||
web_count = search.get('web_count', 0)
|
||||
|
||||
logger.info(f"Processing: '{query}' (user: {user}, web: {web_count})")
|
||||
|
||||
# Skip if insufficient web results
|
||||
if web_count < min_web_results:
|
||||
logger.info(f"Skipping - insufficient web results ({web_count} < {min_web_results})")
|
||||
return None
|
||||
|
||||
# Get web results from SearchQuery
|
||||
web_results = await self._get_web_results(search_id)
|
||||
if not web_results:
|
||||
logger.info("No web results found in database")
|
||||
return None
|
||||
|
||||
logger.info(f"Retrieved {len(web_results)} web results")
|
||||
|
||||
# Analyze web results with Ollama for novel information
|
||||
analysis = await self._analyze_web_results(
|
||||
query=query,
|
||||
web_results=web_results,
|
||||
keywords=search.get('keywords', [])
|
||||
)
|
||||
|
||||
if not analysis or not analysis.get('has_novel_info'):
|
||||
logger.info("No novel information found")
|
||||
return ConsolidationResult(
|
||||
search_id=search_id,
|
||||
query=query
|
||||
)
|
||||
|
||||
# Extract consolidation actions
|
||||
pages_to_create = analysis.get('new_pages', [])
|
||||
pages_to_update = analysis.get('update_pages', [])
|
||||
new_entities = analysis.get('new_entities', [])
|
||||
|
||||
logger.info(
|
||||
f"Analysis: {len(pages_to_create)} new pages, "
|
||||
f"{len(pages_to_update)} updates, {len(new_entities)} entities"
|
||||
)
|
||||
|
||||
if dry_run:
|
||||
logger.info("[DRY RUN] Would create/update pages and entities")
|
||||
return ConsolidationResult(
|
||||
search_id=search_id,
|
||||
query=query,
|
||||
pages_created=len(pages_to_create),
|
||||
pages_updated=len(pages_to_update),
|
||||
entities_added=len(new_entities)
|
||||
)
|
||||
|
||||
# Create/update wiki pages
|
||||
pages_created = 0
|
||||
pages_updated = 0
|
||||
entities_added = 0
|
||||
|
||||
# Create new pages
|
||||
for page_data in pages_to_create:
|
||||
try:
|
||||
await self._create_or_consolidate_page(
|
||||
user=user,
|
||||
title=page_data.get('title'),
|
||||
path=page_data.get('path'),
|
||||
summary=page_data.get('summary'),
|
||||
source_query=query,
|
||||
web_results=web_results
|
||||
)
|
||||
pages_created += 1
|
||||
logger.info(f"Created page: {page_data.get('title')}")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to create page {page_data.get('title')}: {e}")
|
||||
|
||||
# Update existing pages
|
||||
for page_data in pages_to_update:
|
||||
try:
|
||||
await self._update_page_with_facts(
|
||||
title=page_data.get('title'),
|
||||
new_facts=page_data.get('new_facts', []),
|
||||
source_url=page_data.get('source_url'),
|
||||
user=user
|
||||
)
|
||||
pages_updated += 1
|
||||
logger.info(f"Updated page: {page_data.get('title')}")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to update page {page_data.get('title')}: {e}")
|
||||
|
||||
# Add new entities to graph
|
||||
for entity_data in new_entities:
|
||||
try:
|
||||
await self._add_entity_to_graph(
|
||||
user=user,
|
||||
entity_name=entity_data.get('name'),
|
||||
entity_type=entity_data.get('type'),
|
||||
description=entity_data.get('description'),
|
||||
source_search_id=search_id
|
||||
)
|
||||
entities_added += 1
|
||||
logger.info(f"Added entity: {entity_data.get('name')}")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to add entity {entity_data.get('name')}: {e}")
|
||||
|
||||
return ConsolidationResult(
|
||||
search_id=search_id,
|
||||
query=query,
|
||||
pages_created=pages_created,
|
||||
pages_updated=pages_updated,
|
||||
entities_added=entities_added
|
||||
)
|
||||
|
||||
async def _get_web_results(self, search_id: str) -> List[Dict[str, Any]]:
|
||||
"""Get web results for a search from Neo4j."""
|
||||
query = """
|
||||
MATCH (sq:SearchQuery {id: $search_id})-[f:FOUND]->(wr:WebResult)
|
||||
RETURN wr.url as url,
|
||||
wr.title as title,
|
||||
wr.content as content,
|
||||
f.rank as rank,
|
||||
f.rrf_score as rrf_score
|
||||
ORDER BY f.rank
|
||||
LIMIT 20
|
||||
"""
|
||||
|
||||
try:
|
||||
results = await self.neo4j.execute_query(query, {"search_id": search_id})
|
||||
|
||||
web_results = []
|
||||
for record in results:
|
||||
web_results.append({
|
||||
'url': record['url'],
|
||||
'title': record['title'],
|
||||
'content': record['content'],
|
||||
'rank': record['rank'],
|
||||
'rrf_score': record['rrf_score']
|
||||
})
|
||||
|
||||
return web_results
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to get web results: {e}")
|
||||
return []
|
||||
|
||||
async def _analyze_web_results(
|
||||
self,
|
||||
query: str,
|
||||
web_results: List[Dict[str, Any]],
|
||||
keywords: List[str]
|
||||
) -> Optional[Dict[str, Any]]:
|
||||
"""
|
||||
Analyze web results with Ollama for novel information.
|
||||
|
||||
Returns analysis with has_novel_info, new_pages, update_pages, new_entities.
|
||||
"""
|
||||
# Build analysis prompt
|
||||
web_summary = "\n\n".join([
|
||||
f"[{i+1}] {r['title']}\n{r['url']}\n{r['content'][:300]}..."
|
||||
for i, r in enumerate(web_results[:5])
|
||||
])
|
||||
|
||||
prompt = f"""You are a Librarian helping build a personal knowledge base and extended memory system.
|
||||
|
||||
Analyze these web search results for information worth documenting in our personal wiki.
|
||||
|
||||
Query: "{query}"
|
||||
Keywords: {', '.join(keywords) if keywords else 'none'}
|
||||
|
||||
Web Results:
|
||||
{web_summary}
|
||||
|
||||
This is a PERSONAL knowledge base using Schema.org-aligned taxonomy that captures:
|
||||
- People: Family members, friends, colleagues, public figures (Schema.org: Person)
|
||||
- Companies: Businesses, organizations, institutions (Schema.org: Organization)
|
||||
- Places: Locations, restaurants, travel destinations (Schema.org: Place)
|
||||
- Entertainment: Books, movies, TV, music, games (Schema.org: CreativeWork)
|
||||
- Recipes: Food, cooking techniques, ingredients (Schema.org: CreativeWork/Recipe)
|
||||
- Products: Purchased items, gear, tools, equipment (Schema.org: Product)
|
||||
- Technology: Software, applications, infrastructure (Schema.org: SoftwareApplication)
|
||||
- Health: Medical info, fitness, wellness (Schema.org: MedicalEntity)
|
||||
- Events: Concerts, travel, appointments, important dates (Schema.org: Event)
|
||||
- Hobbies: Personal interests, activities, pastimes (Custom extension)
|
||||
- Projects: Work projects, personal projects (Schema.org: Project)
|
||||
- Reference: General knowledge, how-tos (Custom extension)
|
||||
|
||||
Identify information worth documenting:
|
||||
1. New topics/people/things that deserve their own wiki page
|
||||
2. Facts that could enhance existing pages
|
||||
3. Entities (people, places, things, concepts) for the knowledge graph
|
||||
|
||||
Be INCLUSIVE - if someone searched for it, it's likely worth documenting.
|
||||
Personal information is just as valuable as technical information.
|
||||
|
||||
**CRITICAL: Use ONLY these Schema.org-aligned path prefixes (case-sensitive):**
|
||||
|
||||
- People: `people/<name>` (Schema.org: Person)
|
||||
- Companies: `companies/<company-name>` (Schema.org: Organization)
|
||||
- Places: `places/<location>` (Schema.org: Place)
|
||||
- Entertainment (Schema.org: CreativeWork):
|
||||
- Books: `entertainment/books/<title>`
|
||||
- Movies: `entertainment/movies/<title>`
|
||||
- TV: `entertainment/tv/<title>`
|
||||
- Music: `entertainment/music/<artist-or-album>`
|
||||
- Games: `entertainment/games/<title>`
|
||||
- Recipes: `recipes/<cuisine-or-category>/<dish>` (Schema.org: Recipe)
|
||||
- Products: `products/<category>/<product-name>` (Schema.org: Product)
|
||||
- Technology: `technology/<category>/<topic>` (Schema.org: SoftwareApplication)
|
||||
- Health: `health/<category>/<topic>` (Schema.org: MedicalEntity)
|
||||
- Events: `events/<event-type>/<event-name>` (Schema.org: Event)
|
||||
- Hobbies: `hobbies/<hobby-name>` (Custom extension)
|
||||
- Projects: `projects/<project-name>` (Schema.org: Project)
|
||||
- Reference: `reference/<category>/<topic>` (Custom extension)
|
||||
|
||||
**Path Rules:**
|
||||
- Use lowercase with hyphens (kebab-case): "machine-learning" not "Machine_Learning"
|
||||
- Keep paths 2-3 levels deep maximum
|
||||
- Be consistent with existing paths when possible
|
||||
|
||||
Return ONLY valid JSON:
|
||||
{{
|
||||
"has_novel_info": true,
|
||||
"new_pages": [
|
||||
{{"title": "Page Title", "path": "companies/example-company", "summary": "What information to include"}}
|
||||
],
|
||||
"update_pages": [
|
||||
{{"title": "Existing Page", "new_facts": ["fact 1"], "source_url": "url"}}
|
||||
],
|
||||
"new_entities": [
|
||||
{{"name": "Entity Name", "type": "person/place/thing/concept/recipe/media", "description": "Brief description"}}
|
||||
]
|
||||
}}
|
||||
|
||||
JSON:"""
|
||||
|
||||
try:
|
||||
# Call Ollama for analysis
|
||||
response = await self.ollama.generate_text(
|
||||
prompt=prompt,
|
||||
model=self.settings.reranker_model, # Use mistral-nemo
|
||||
stream=False
|
||||
)
|
||||
|
||||
if not response:
|
||||
logger.warning("Empty response from Ollama")
|
||||
return None
|
||||
|
||||
# Extract JSON from response
|
||||
response_clean = response.strip()
|
||||
if '{' in response_clean:
|
||||
json_start = response_clean.find('{')
|
||||
json_end = response_clean.rfind('}') + 1
|
||||
response_clean = response_clean[json_start:json_end]
|
||||
|
||||
analysis = json.loads(response_clean)
|
||||
return analysis
|
||||
|
||||
except json.JSONDecodeError as e:
|
||||
logger.error(f"Failed to parse Ollama response as JSON: {e}")
|
||||
logger.debug(f"Response was: {response[:500]}")
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(f"Analysis failed: {e}", exc_info=True)
|
||||
return None
|
||||
|
||||
async def _mark_search_processed(self, search_id: str):
|
||||
"""Mark SearchQuery node as processed."""
|
||||
query = """
|
||||
MATCH (sq:SearchQuery {id: $search_id})
|
||||
SET sq.processed = true,
|
||||
sq.processed_at = datetime()
|
||||
RETURN sq.id
|
||||
"""
|
||||
|
||||
try:
|
||||
await self.neo4j.execute_query(query, {"search_id": search_id})
|
||||
logger.debug(f"Marked search {search_id} as processed")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to mark search as processed: {e}")
|
||||
|
||||
async def _create_or_consolidate_page(
|
||||
self,
|
||||
user: str,
|
||||
title: str,
|
||||
path: str,
|
||||
summary: str,
|
||||
source_query: str,
|
||||
web_results: List[Dict[str, Any]]
|
||||
):
|
||||
"""
|
||||
Create wiki page or consolidate with existing synonym page.
|
||||
|
||||
Uses WikiPageWriter for intelligent LLM-based content generation:
|
||||
- For new pages: Holistic structured content creation
|
||||
- For existing pages: Zero-loss reconstruction with conflict detection
|
||||
"""
|
||||
# Normalize path to user namespace
|
||||
if not path.startswith(f"users/{user}"):
|
||||
path = f"users/{user}/{path.lstrip('/')}"
|
||||
|
||||
# Format web results as source information
|
||||
source_information = [
|
||||
{
|
||||
'title': r['title'],
|
||||
'url': r['url'],
|
||||
'content': r['content']
|
||||
}
|
||||
for r in web_results[:5] # Top 5 web results
|
||||
]
|
||||
|
||||
# Search for existing pages with similar titles (synonym consolidation)
|
||||
existing_pages = await self.wiki.search_pages(title, path_prefix=f"users/{user}")
|
||||
|
||||
if existing_pages:
|
||||
# Page exists - reconstruct with new information using LLM
|
||||
logger.info(f"Found existing page for '{title}', will reconstruct with new info")
|
||||
page_id = existing_pages[0]['id']
|
||||
|
||||
# Get current content
|
||||
existing_page = await self.wiki.get_page(page_id)
|
||||
if existing_page:
|
||||
# Build new information text from summary and web results
|
||||
new_information = f"{summary}\n\n"
|
||||
for r in web_results[:3]:
|
||||
new_information += f"- {r['title']}: {r['content'][:200]}...\n"
|
||||
|
||||
# Use WikiPageWriter to reconstruct with LLM
|
||||
reconstructed_content, conflicts = await self.wiki_page_writer.reconstruct_page(
|
||||
title=title,
|
||||
existing_content=existing_page['content'],
|
||||
new_information=new_information,
|
||||
new_sources=source_information,
|
||||
detect_conflicts=True
|
||||
)
|
||||
|
||||
if conflicts:
|
||||
logger.warning(
|
||||
f"Detected {len(conflicts)} conflicts when updating '{title}' - "
|
||||
"LLM chose most authoritative sources"
|
||||
)
|
||||
|
||||
await self.wiki.update_page(
|
||||
page_id=page_id,
|
||||
content=reconstructed_content
|
||||
)
|
||||
logger.info(f"Reconstructed existing page: {title}")
|
||||
|
||||
# Trigger ingestion to update vectors and graph
|
||||
if self.ingestion_service:
|
||||
try:
|
||||
await self.ingestion_service.ingest_page(
|
||||
page_id=page_id,
|
||||
user=user,
|
||||
force_refresh=True
|
||||
)
|
||||
logger.info(f"Ingested updated page {page_id} into knowledge base")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to ingest updated page {page_id}: {e}")
|
||||
|
||||
return
|
||||
|
||||
# Create new page with LLM-generated structured content
|
||||
logger.info(f"Creating new page: {title}")
|
||||
|
||||
# Use WikiPageWriter to create structured content
|
||||
content = await self.wiki_page_writer.create_page(
|
||||
title=title,
|
||||
topic_summary=summary,
|
||||
source_information=source_information,
|
||||
entities=None, # Could extract from keywords if available
|
||||
related_docs=None
|
||||
)
|
||||
|
||||
# Extract tags from path for dossier organization
|
||||
path_parts = path.split('/')
|
||||
tags = [part for part in path_parts if part and part not in ['users', user]]
|
||||
|
||||
created_page = await self.wiki.create_page(
|
||||
path=path,
|
||||
title=title,
|
||||
content=content,
|
||||
description=f"Consolidated from search: {source_query}",
|
||||
tags=tags[:3], # Limit to 3 tags
|
||||
is_published=True
|
||||
)
|
||||
|
||||
page_id = created_page.get("id") if created_page else None
|
||||
logger.info(f"Created new page: {path} (page_id: {page_id})")
|
||||
|
||||
# Trigger ingestion to update vectors and graph
|
||||
if self.ingestion_service and page_id:
|
||||
try:
|
||||
await self.ingestion_service.ingest_page(
|
||||
page_id=page_id,
|
||||
user=user,
|
||||
force_refresh=False # New page, no need to force
|
||||
)
|
||||
logger.info(f"Ingested new page {page_id} into knowledge base")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to ingest new page {page_id}: {e}")
|
||||
|
||||
async def _update_page_with_facts(
|
||||
self,
|
||||
title: str,
|
||||
new_facts: List[str],
|
||||
source_url: str,
|
||||
user: str
|
||||
):
|
||||
"""
|
||||
Update existing page with new facts using LLM reconstruction.
|
||||
|
||||
Uses WikiPageWriter to intelligently merge facts with zero loss.
|
||||
"""
|
||||
# Search for page
|
||||
pages = await self.wiki.search_pages(title, path_prefix=f"users/{user}")
|
||||
|
||||
if not pages:
|
||||
logger.warning(f"Page '{title}' not found for update")
|
||||
return
|
||||
|
||||
page_id = pages[0]['id']
|
||||
existing_page = await self.wiki.get_page(page_id)
|
||||
|
||||
if not existing_page:
|
||||
return
|
||||
|
||||
# Build new information from facts
|
||||
new_information = "\n".join([f"- {fact}" for fact in new_facts])
|
||||
|
||||
# Format source
|
||||
source_information = [{
|
||||
'title': source_url,
|
||||
'url': source_url,
|
||||
'content': new_information
|
||||
}]
|
||||
|
||||
# Use WikiPageWriter to reconstruct with LLM
|
||||
reconstructed_content, conflicts = await self.wiki_page_writer.reconstruct_page(
|
||||
title=title,
|
||||
existing_content=existing_page['content'],
|
||||
new_information=new_information,
|
||||
new_sources=source_information,
|
||||
detect_conflicts=True
|
||||
)
|
||||
|
||||
if conflicts:
|
||||
logger.warning(
|
||||
f"Detected {len(conflicts)} conflicts when updating '{title}' with new facts"
|
||||
)
|
||||
|
||||
await self.wiki.update_page(
|
||||
page_id=page_id,
|
||||
content=reconstructed_content
|
||||
)
|
||||
|
||||
async def _add_entity_to_graph(
|
||||
self,
|
||||
user: str,
|
||||
entity_name: str,
|
||||
entity_type: str,
|
||||
description: str,
|
||||
source_search_id: str
|
||||
):
|
||||
"""Add new entity to knowledge graph."""
|
||||
from src.core.multi_tenancy import get_neo4j_user_base_label
|
||||
|
||||
user_base_label = get_neo4j_user_base_label(user)
|
||||
|
||||
# Create entity node with appropriate type label
|
||||
type_label = entity_type.capitalize() if entity_type else "Entity"
|
||||
|
||||
query = f"""
|
||||
MERGE (e:{user_base_label}:{type_label} {{name: $name}})
|
||||
ON CREATE SET
|
||||
e.description = $description,
|
||||
e.created_at = datetime(),
|
||||
e.source = 'librarian_consolidation',
|
||||
e.source_search_id = $search_id
|
||||
ON MATCH SET
|
||||
e.updated_at = datetime()
|
||||
RETURN e
|
||||
"""
|
||||
|
||||
try:
|
||||
await self.neo4j.execute_query(query, {
|
||||
"name": entity_name,
|
||||
"description": description,
|
||||
"search_id": source_search_id
|
||||
})
|
||||
logger.debug(f"Added entity to graph: {entity_name} ({entity_type})")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to add entity to graph: {e}")
|
||||
@@ -0,0 +1,493 @@
|
||||
"""
|
||||
Intelligent Wiki Page Writer Service
|
||||
|
||||
Uses LLM (mistral-nemo) to create and reconstruct wiki pages with:
|
||||
- Holistic content restructuring
|
||||
- Zero fact loss (unless superseded)
|
||||
- Conflict detection and flagging
|
||||
- Standard formatting with template adherence
|
||||
- Professional organization (summary, tables, chapters)
|
||||
|
||||
This service is used by:
|
||||
- Consolidation service (Librarian knowledge consolidation)
|
||||
- Any other service that needs to create/update wiki pages
|
||||
"""
|
||||
import logging
|
||||
import json
|
||||
from typing import Dict, Any, List, Optional, Tuple
|
||||
from datetime import datetime
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class WikiPageWriter:
|
||||
"""
|
||||
Intelligent wiki page writer using LLM for content generation and restructuring.
|
||||
"""
|
||||
|
||||
def __init__(self, ollama_client):
|
||||
"""
|
||||
Initialize wiki page writer.
|
||||
|
||||
Args:
|
||||
ollama_client: OllamaClient for LLM operations
|
||||
"""
|
||||
self.ollama = ollama_client
|
||||
self.model = "mistral-nemo" # Default model for writing
|
||||
|
||||
async def create_page(
|
||||
self,
|
||||
title: str,
|
||||
topic_summary: str,
|
||||
source_information: List[Dict[str, str]],
|
||||
entities: Optional[List[str]] = None,
|
||||
related_docs: Optional[List[str]] = None
|
||||
) -> str:
|
||||
"""
|
||||
Create new wiki page with structured content.
|
||||
|
||||
Args:
|
||||
title: Page title
|
||||
topic_summary: Brief summary of the topic
|
||||
source_information: List of {title, url, content} dicts
|
||||
entities: Related entities from knowledge graph
|
||||
related_docs: Related documents/pages
|
||||
|
||||
Returns:
|
||||
Formatted markdown content
|
||||
"""
|
||||
logger.info(f"Creating wiki page: {title}")
|
||||
|
||||
# Build source context
|
||||
sources_text = self._format_sources_for_llm(source_information)
|
||||
|
||||
# Create page using LLM
|
||||
prompt = self._build_create_prompt(
|
||||
title=title,
|
||||
summary=topic_summary,
|
||||
sources=sources_text,
|
||||
entities=entities or [],
|
||||
related_docs=related_docs or []
|
||||
)
|
||||
|
||||
content = await self._call_llm(prompt)
|
||||
|
||||
# Post-process to ensure template compliance
|
||||
content = self._ensure_standard_sections(
|
||||
content=content,
|
||||
title=title,
|
||||
sources=source_information,
|
||||
entities=entities or [],
|
||||
related_docs=related_docs or []
|
||||
)
|
||||
|
||||
return content
|
||||
|
||||
async def reconstruct_page(
|
||||
self,
|
||||
title: str,
|
||||
existing_content: str,
|
||||
new_information: str,
|
||||
new_sources: List[Dict[str, str]],
|
||||
detect_conflicts: bool = True
|
||||
) -> Tuple[str, Optional[List[Dict[str, Any]]]]:
|
||||
"""
|
||||
Reconstruct existing page with new information.
|
||||
|
||||
Intelligently merges new content with existing, restructures for clarity,
|
||||
and detects factual conflicts.
|
||||
|
||||
Args:
|
||||
title: Page title
|
||||
existing_content: Current page content
|
||||
new_information: New information to integrate
|
||||
new_sources: Sources for new information
|
||||
detect_conflicts: Whether to detect and flag conflicts
|
||||
|
||||
Returns:
|
||||
Tuple of (reconstructed_content, conflicts)
|
||||
conflicts: List of detected conflicts or None
|
||||
"""
|
||||
logger.info(f"Reconstructing wiki page: {title}")
|
||||
|
||||
# Detect conflicts first
|
||||
conflicts = None
|
||||
if detect_conflicts:
|
||||
conflicts = await self._detect_conflicts(
|
||||
existing_content=existing_content,
|
||||
new_information=new_information
|
||||
)
|
||||
|
||||
if conflicts:
|
||||
logger.warning(f"Detected {len(conflicts)} potential conflicts in {title}")
|
||||
|
||||
# Build reconstruction prompt
|
||||
prompt = self._build_reconstruct_prompt(
|
||||
title=title,
|
||||
existing_content=existing_content,
|
||||
new_information=new_information,
|
||||
new_sources=self._format_sources_for_llm(new_sources),
|
||||
conflicts=conflicts
|
||||
)
|
||||
|
||||
# Reconstruct with LLM
|
||||
reconstructed = await self._call_llm(prompt)
|
||||
|
||||
# Ensure standard sections are present
|
||||
reconstructed = self._ensure_standard_sections(
|
||||
content=reconstructed,
|
||||
title=title,
|
||||
sources=new_sources,
|
||||
is_update=True
|
||||
)
|
||||
|
||||
return reconstructed, conflicts
|
||||
|
||||
async def _detect_conflicts(
|
||||
self,
|
||||
existing_content: str,
|
||||
new_information: str
|
||||
) -> Optional[List[Dict[str, Any]]]:
|
||||
"""
|
||||
Detect factual conflicts between existing and new content.
|
||||
|
||||
Returns:
|
||||
List of conflicts with: {fact_a, fact_b, confidence, context}
|
||||
"""
|
||||
prompt = f"""Analyze these two pieces of content for factual conflicts.
|
||||
|
||||
EXISTING CONTENT:
|
||||
{existing_content[:2000]}
|
||||
|
||||
NEW INFORMATION:
|
||||
{new_information[:2000]}
|
||||
|
||||
Identify any facts that contradict each other. For each conflict, provide:
|
||||
1. The fact from existing content
|
||||
2. The contradicting fact from new information
|
||||
3. Confidence level (low/medium/high)
|
||||
4. Context/explanation
|
||||
|
||||
Return ONLY valid JSON:
|
||||
{{
|
||||
"conflicts": [
|
||||
{{
|
||||
"existing_fact": "fact from old content",
|
||||
"new_fact": "contradicting fact",
|
||||
"confidence": "medium",
|
||||
"context": "explanation of why these conflict"
|
||||
}}
|
||||
]
|
||||
}}
|
||||
|
||||
If no conflicts, return: {{"conflicts": []}}
|
||||
|
||||
JSON:"""
|
||||
|
||||
try:
|
||||
response = await self.ollama.generate_text(
|
||||
prompt=prompt,
|
||||
model=self.model,
|
||||
stream=False
|
||||
)
|
||||
|
||||
# Extract JSON
|
||||
response_clean = response.strip()
|
||||
if '{' in response_clean:
|
||||
json_start = response_clean.find('{')
|
||||
json_end = response_clean.rfind('}') + 1
|
||||
response_clean = response_clean[json_start:json_end]
|
||||
|
||||
result = json.loads(response_clean)
|
||||
conflicts = result.get('conflicts', [])
|
||||
|
||||
return conflicts if conflicts else None
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Conflict detection failed: {e}")
|
||||
return None
|
||||
|
||||
def _build_create_prompt(
|
||||
self,
|
||||
title: str,
|
||||
summary: str,
|
||||
sources: str,
|
||||
entities: List[str],
|
||||
related_docs: List[str]
|
||||
) -> str:
|
||||
"""Build LLM prompt for creating new page."""
|
||||
return f"""You are a Librarian creating a dossier for a personal knowledge base and extended memory system.
|
||||
|
||||
Create a comprehensive, well-structured wiki page with appropriate sections for the content type.
|
||||
|
||||
TOPIC: {title}
|
||||
|
||||
SUMMARY: {summary}
|
||||
|
||||
SOURCE INFORMATION:
|
||||
{sources}
|
||||
|
||||
RELATED ENTITIES: {', '.join(entities) if entities else 'None'}
|
||||
|
||||
RELATED DOCUMENTS: {', '.join(related_docs) if related_docs else 'None'}
|
||||
|
||||
CONTENT TYPE GUIDELINES (Schema.org-aligned):
|
||||
|
||||
For PEOPLE (family, friends, colleagues, public figures) (Schema.org: Person):
|
||||
- Executive Summary (who they are, key facts)
|
||||
- Background & Biography
|
||||
- Relationships & Connections
|
||||
- Professional Info / Career
|
||||
- Interests & Preferences
|
||||
- Important Dates & Events
|
||||
- Notes & Observations
|
||||
|
||||
For COMPANIES (businesses, organizations, startups) (Schema.org: Organization):
|
||||
- Executive Summary (what they do, industry, key facts)
|
||||
- Overview & Mission
|
||||
- Products & Services
|
||||
- History & Milestones
|
||||
- Leadership & Team
|
||||
- Personal Connection / Experience
|
||||
- Notable Projects or Achievements
|
||||
|
||||
For PLACES (locations, restaurants, destinations) (Schema.org: Place):
|
||||
- Executive Summary (what/where, key details)
|
||||
- Location & How to Get There
|
||||
- Description & Atmosphere
|
||||
- Features & Amenities
|
||||
- Personal Experiences / Visits
|
||||
- Recommendations & Tips
|
||||
|
||||
For ENTERTAINMENT (books, movies, TV, music, games) (Schema.org: CreativeWork):
|
||||
- Executive Summary (title, creator, key facts)
|
||||
- Synopsis / Overview
|
||||
- Key Characters / Themes
|
||||
- Personal Thoughts & Ratings
|
||||
- Memorable Moments / Quotes
|
||||
- Related Works
|
||||
|
||||
For RECIPES & FOOD (Schema.org: Recipe):
|
||||
- Executive Summary (dish name, cuisine type)
|
||||
- Ingredients (formatted as table or list)
|
||||
- Instructions (step-by-step)
|
||||
- Cooking Tips & Variations
|
||||
- Personal Notes & Modifications
|
||||
- Source / Origin
|
||||
|
||||
For PRODUCTS (gear, tools, purchases) (Schema.org: Product):
|
||||
- Executive Summary (what it is, brand/model, key specs)
|
||||
- Overview & Purpose
|
||||
- Specifications (formatted as table)
|
||||
- Purchase Information (where, when, price)
|
||||
- Personal Experience / Review
|
||||
- Maintenance & Care
|
||||
- Related Products / Alternatives
|
||||
|
||||
For TECHNOLOGY (software, applications, infrastructure) (Schema.org: SoftwareApplication):
|
||||
- Executive Summary (what it is, key facts)
|
||||
- Overview & Purpose
|
||||
- Technical Details (tables for specs)
|
||||
- Setup & Configuration
|
||||
- Use Cases & Applications
|
||||
- Best Practices
|
||||
- Common Issues & Solutions
|
||||
|
||||
For EVENTS (concerts, travel, appointments) (Schema.org: Event):
|
||||
- Executive Summary (what, when, where)
|
||||
- Event Details (date, time, location, venue)
|
||||
- Participants / Attendees
|
||||
- Planning & Preparation
|
||||
- Experience / Highlights
|
||||
- Photos / Media
|
||||
- Notes & Reflections
|
||||
|
||||
For HEALTH (medical, fitness, wellness) (Schema.org: MedicalEntity):
|
||||
- Executive Summary (condition/topic, key facts)
|
||||
- Overview & Background
|
||||
- Symptoms / Signs / Characteristics
|
||||
- Treatments / Approaches / Recommendations
|
||||
- Personal Experience / Progress
|
||||
- Resources & References
|
||||
- Important Dates (appointments, changes)
|
||||
|
||||
For HOBBIES (activities, interests, pastimes) (Custom extension):
|
||||
- Executive Summary (what it is, why interesting)
|
||||
- Getting Started / Basics
|
||||
- Equipment & Materials
|
||||
- Techniques & Skills
|
||||
- Personal Progress / Achievements
|
||||
- Resources & Communities
|
||||
- Goals & Future Plans
|
||||
|
||||
For PROJECTS (work projects, personal projects) (Schema.org: Project):
|
||||
- Executive Summary (what, why, status)
|
||||
- Goals & Objectives
|
||||
- Timeline & Milestones
|
||||
- Team / Collaborators
|
||||
- Technical Details / Architecture
|
||||
- Current Status & Next Steps
|
||||
- Lessons Learned / Reflections
|
||||
|
||||
For REFERENCE (general knowledge, how-tos) (Custom extension):
|
||||
- Executive Summary
|
||||
- Overview & Context
|
||||
- Key Concepts & Definitions
|
||||
- Step-by-Step Guide (if applicable)
|
||||
- Examples & Use Cases
|
||||
- Tips & Best Practices
|
||||
- Related Topics & Further Reading
|
||||
|
||||
FORMATTING RULES:
|
||||
- Use markdown headers (##, ###)
|
||||
- Create tables for structured data (ingredients, specs, comparisons)
|
||||
- Use bullet points for lists
|
||||
- Include code blocks with ``` where applicable
|
||||
- Bold important terms
|
||||
- Keep sections focused and scannable
|
||||
- Adapt structure to content - not all sections apply to all topics
|
||||
|
||||
Generate ONLY the markdown content (do not include Sources, Knowledge Graph, or Mind Map sections - those are added automatically).
|
||||
|
||||
MARKDOWN:"""
|
||||
|
||||
def _build_reconstruct_prompt(
|
||||
self,
|
||||
title: str,
|
||||
existing_content: str,
|
||||
new_information: str,
|
||||
new_sources: str,
|
||||
conflicts: Optional[List[Dict[str, Any]]]
|
||||
) -> str:
|
||||
"""Build LLM prompt for reconstructing page."""
|
||||
conflicts_note = ""
|
||||
if conflicts:
|
||||
conflicts_note = "\n\nDETECTED CONFLICTS:\n"
|
||||
for i, c in enumerate(conflicts, 1):
|
||||
conflicts_note += f"{i}. Existing: '{c['existing_fact']}'\n"
|
||||
conflicts_note += f" New: '{c['new_fact']}'\n"
|
||||
conflicts_note += f" Confidence: {c['confidence']}\n"
|
||||
conflicts_note += f" Note: {c['context']}\n\n"
|
||||
conflicts_note += "IMPORTANT: For conflicts, prefer the most recent/authoritative source. Add a note in 'Changes & Updates' section when facts are superseded.\n"
|
||||
|
||||
return f"""Reconstruct this wiki page by intelligently merging new information with existing content.
|
||||
|
||||
TITLE: {title}
|
||||
|
||||
EXISTING CONTENT:
|
||||
{existing_content}
|
||||
|
||||
NEW INFORMATION TO INTEGRATE:
|
||||
{new_information}
|
||||
|
||||
NEW SOURCES:
|
||||
{new_sources}
|
||||
{conflicts_note}
|
||||
|
||||
RECONSTRUCTION REQUIREMENTS:
|
||||
1. **Zero Fact Loss**: Preserve ALL facts from existing content unless superseded
|
||||
2. **Holistic Restructuring**: Reorganize for better flow and clarity
|
||||
3. **Conflict Resolution**: When facts conflict, choose most authoritative/recent
|
||||
4. **Professional Structure**:
|
||||
- Update Executive Summary with key facts
|
||||
- Organize into clear chapters
|
||||
- Use tables for specifications/comparisons
|
||||
- Maintain consistent formatting
|
||||
5. **Update Tracking**: Add entry to "Changes & Updates" section with today's date
|
||||
|
||||
FORMATTING RULES:
|
||||
- Maintain markdown structure
|
||||
- Use tables for data (| col1 | col2 |)
|
||||
- Keep existing good structure, improve where needed
|
||||
- Bold important terms
|
||||
- Add subsections (###) where it improves clarity
|
||||
|
||||
OUTPUT INSTRUCTIONS:
|
||||
- Return complete page content (do not include Sources, Knowledge Graph, Mind Map - those are added automatically)
|
||||
- Include updated "Changes & Updates" section noting what was changed today
|
||||
- If facts were superseded, note it clearly
|
||||
|
||||
RECONSTRUCTED MARKDOWN:"""
|
||||
|
||||
async def _call_llm(self, prompt: str) -> str:
|
||||
"""Call LLM with prompt and return response."""
|
||||
try:
|
||||
response = await self.ollama.generate_text(
|
||||
prompt=prompt,
|
||||
model=self.model,
|
||||
stream=False
|
||||
)
|
||||
|
||||
if not response:
|
||||
raise Exception("Empty response from LLM")
|
||||
|
||||
return response.strip()
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"LLM call failed: {e}")
|
||||
raise
|
||||
|
||||
def _format_sources_for_llm(self, sources: List[Dict[str, str]]) -> str:
|
||||
"""Format source information for LLM prompt."""
|
||||
formatted = []
|
||||
for i, source in enumerate(sources, 1):
|
||||
formatted.append(f"[{i}] {source.get('title', 'Untitled')}")
|
||||
formatted.append(f" URL: {source.get('url', 'N/A')}")
|
||||
content = source.get('content', '')[:500] # Limit content length
|
||||
formatted.append(f" Content: {content}...\n")
|
||||
|
||||
return "\n".join(formatted)
|
||||
|
||||
def _ensure_standard_sections(
|
||||
self,
|
||||
content: str,
|
||||
title: str,
|
||||
sources: List[Dict[str, str]],
|
||||
entities: Optional[List[str]] = None,
|
||||
related_docs: Optional[List[str]] = None,
|
||||
is_update: bool = False
|
||||
) -> str:
|
||||
"""
|
||||
Ensure page has standard footer sections (Sources, Knowledge Graph, Mind Map).
|
||||
|
||||
These sections are standardized and appended automatically.
|
||||
"""
|
||||
# Remove any existing standard sections
|
||||
for section in ["## Sources", "## Knowledge Graph", "## Mind Map"]:
|
||||
if section in content:
|
||||
content = content.split(section)[0]
|
||||
|
||||
# Add horizontal rule before footer
|
||||
content = content.rstrip() + "\n\n---\n\n"
|
||||
|
||||
# Add Sources section
|
||||
content += "## Sources\n\n"
|
||||
if sources:
|
||||
for i, source in enumerate(sources, 1):
|
||||
content += f"{i}. [{source.get('title', 'Source')}]({source.get('url', '#')})\n"
|
||||
else:
|
||||
content += "*No sources listed*\n"
|
||||
|
||||
# Add Knowledge Graph section
|
||||
content += "\n## Knowledge Graph\n\n"
|
||||
if entities:
|
||||
content += "**Related Entities:**\n"
|
||||
for entity in entities[:10]: # Limit to 10
|
||||
content += f"- {entity}\n"
|
||||
else:
|
||||
content += "*No entities linked yet*\n"
|
||||
|
||||
content += "\n**View in Neo4j:** [Explore Graph](/graph)\n"
|
||||
|
||||
# Add Mind Map section
|
||||
content += "\n## Mind Map\n\n"
|
||||
content += f"**Interactive Mind Map:** [View Topic Map](/mindmap?topic={title.replace(' ', '+')})\n"
|
||||
|
||||
# Add footer metadata
|
||||
content += "\n---\n\n"
|
||||
timestamp = datetime.now().strftime('%Y-%m-%d %H:%M')
|
||||
action = "Updated" if is_update else "Created"
|
||||
content += f"*{action}: {timestamp} | Generated by: Librarian Agent* \n"
|
||||
content += "*Template: Library Desk Wiki Standard v1.0*\n"
|
||||
|
||||
return content
|
||||
@@ -0,0 +1,665 @@
|
||||
"""
|
||||
Comprehensive tests for Knowledge Consolidation system.
|
||||
|
||||
Tests cover:
|
||||
- ConsolidationService (unit tests with mocks)
|
||||
- Consolidation API endpoint (integration tests)
|
||||
- Model validation
|
||||
- Error handling
|
||||
- Dry run mode
|
||||
|
||||
Run with: pytest tests/test_consolidation.py -v -s
|
||||
"""
|
||||
|
||||
import pytest
|
||||
import pytest_asyncio
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
from typing import AsyncGenerator
|
||||
from datetime import datetime
|
||||
import json
|
||||
|
||||
from src.services.consolidation_service import ConsolidationService
|
||||
from src.models.consolidation import (
|
||||
ConsolidationRequest,
|
||||
ConsolidationResponse,
|
||||
ConsolidationResult,
|
||||
SearchQueryInfo
|
||||
)
|
||||
from src.config import get_settings
|
||||
|
||||
# Test constants
|
||||
TEST_USER = "consolidation-tester"
|
||||
TEST_SEARCH_ID = "test-search-123"
|
||||
|
||||
|
||||
# Fixtures
|
||||
|
||||
@pytest.fixture
|
||||
def settings():
|
||||
"""Get application settings."""
|
||||
return get_settings()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_neo4j():
|
||||
"""Mock Neo4j client."""
|
||||
mock = AsyncMock()
|
||||
mock.execute_query = AsyncMock()
|
||||
return mock
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_ollama():
|
||||
"""Mock Ollama client."""
|
||||
mock = AsyncMock()
|
||||
mock.generate_text = AsyncMock()
|
||||
return mock
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_wiki():
|
||||
"""Mock Wiki.js client."""
|
||||
mock = AsyncMock()
|
||||
return mock
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def consolidation_service(mock_neo4j, mock_ollama, mock_wiki, settings):
|
||||
"""Get ConsolidationService with mocked dependencies."""
|
||||
return ConsolidationService(
|
||||
neo4j=mock_neo4j,
|
||||
ollama=mock_ollama,
|
||||
wiki=mock_wiki,
|
||||
settings=settings
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def sample_unprocessed_searches():
|
||||
"""Sample unprocessed search queries."""
|
||||
return [
|
||||
{
|
||||
'id': 'search-1',
|
||||
'query': 'docker orchestration kubernetes',
|
||||
'user': TEST_USER,
|
||||
'timestamp': datetime.now().isoformat(),
|
||||
'total_results': 10,
|
||||
'web_count': 5,
|
||||
'keywords': ['docker', 'orchestration', 'kubernetes']
|
||||
},
|
||||
{
|
||||
'id': 'search-2',
|
||||
'query': 'python async programming',
|
||||
'user': TEST_USER,
|
||||
'timestamp': datetime.now().isoformat(),
|
||||
'total_results': 8,
|
||||
'web_count': 3,
|
||||
'keywords': ['python', 'async', 'programming']
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def sample_web_results():
|
||||
"""Sample web search results."""
|
||||
return [
|
||||
{
|
||||
'url': 'https://kubernetes.io/docs',
|
||||
'title': 'Kubernetes Documentation',
|
||||
'content': 'Kubernetes is an orchestration platform for containers...',
|
||||
'rank': 1,
|
||||
'rrf_score': 0.05
|
||||
},
|
||||
{
|
||||
'url': 'https://docs.docker.com/swarm',
|
||||
'title': 'Docker Swarm Documentation',
|
||||
'content': 'Docker Swarm is a container orchestration tool...',
|
||||
'rank': 2,
|
||||
'rrf_score': 0.04
|
||||
},
|
||||
{
|
||||
'url': 'https://example.com/k8s-tutorial',
|
||||
'title': 'Kubernetes Tutorial',
|
||||
'content': 'Learn how to use Kubernetes for container orchestration...',
|
||||
'rank': 3,
|
||||
'rrf_score': 0.03
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def sample_llm_analysis():
|
||||
"""Sample LLM analysis response."""
|
||||
return {
|
||||
"has_novel_info": True,
|
||||
"new_pages": [
|
||||
{
|
||||
"title": "Kubernetes Container Orchestration",
|
||||
"path": "infrastructure/kubernetes",
|
||||
"summary": "Overview of Kubernetes orchestration capabilities"
|
||||
}
|
||||
],
|
||||
"update_pages": [
|
||||
{
|
||||
"title": "Docker Infrastructure",
|
||||
"new_facts": [
|
||||
"Kubernetes provides automatic bin packing",
|
||||
"Self-healing capabilities with automatic restarts"
|
||||
],
|
||||
"source_url": "https://kubernetes.io/docs"
|
||||
}
|
||||
],
|
||||
"new_entities": [
|
||||
{
|
||||
"name": "Kubernetes",
|
||||
"type": "technology",
|
||||
"description": "Container orchestration platform"
|
||||
},
|
||||
{
|
||||
"name": "Docker Swarm",
|
||||
"type": "technology",
|
||||
"description": "Docker's native orchestration tool"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
# Model Tests
|
||||
|
||||
def test_consolidation_request_validation():
|
||||
"""Test ConsolidationRequest model validation."""
|
||||
# Valid request
|
||||
request = ConsolidationRequest(
|
||||
process_limit=10,
|
||||
lookback_days=7,
|
||||
min_web_results=2,
|
||||
dry_run=False
|
||||
)
|
||||
assert request.process_limit == 10
|
||||
assert request.lookback_days == 7
|
||||
assert request.min_web_results == 2
|
||||
assert request.dry_run is False
|
||||
|
||||
# Default values
|
||||
request = ConsolidationRequest()
|
||||
assert request.process_limit == 10
|
||||
assert request.lookback_days == 7
|
||||
assert request.min_web_results == 2
|
||||
assert request.dry_run is False
|
||||
|
||||
# Validate limits
|
||||
with pytest.raises(Exception):
|
||||
ConsolidationRequest(process_limit=0) # Too low
|
||||
|
||||
with pytest.raises(Exception):
|
||||
ConsolidationRequest(process_limit=101) # Too high
|
||||
|
||||
|
||||
def test_consolidation_response_model():
|
||||
"""Test ConsolidationResponse model."""
|
||||
response = ConsolidationResponse(
|
||||
total_found=5,
|
||||
processed_count=4,
|
||||
pages_created=2,
|
||||
pages_updated=3,
|
||||
entities_added=5,
|
||||
errors=["Error 1"],
|
||||
results=[],
|
||||
dry_run=False
|
||||
)
|
||||
|
||||
assert response.total_found == 5
|
||||
assert response.processed_count == 4
|
||||
assert response.pages_created == 2
|
||||
assert len(response.errors) == 1
|
||||
|
||||
|
||||
def test_consolidation_result_model():
|
||||
"""Test ConsolidationResult model."""
|
||||
result = ConsolidationResult(
|
||||
search_id="test-123",
|
||||
query="test query",
|
||||
pages_created=1,
|
||||
pages_updated=2,
|
||||
entities_added=3,
|
||||
error=None
|
||||
)
|
||||
|
||||
assert result.search_id == "test-123"
|
||||
assert result.query == "test query"
|
||||
assert result.pages_created == 1
|
||||
assert result.error is None
|
||||
|
||||
|
||||
# Service Unit Tests
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_find_unprocessed_searches_empty(consolidation_service, mock_neo4j):
|
||||
"""Test finding unprocessed searches when none exist."""
|
||||
# Mock empty result
|
||||
mock_neo4j.execute_query.return_value = []
|
||||
|
||||
searches = await consolidation_service._find_unprocessed_searches(
|
||||
lookback_days=7,
|
||||
limit=10
|
||||
)
|
||||
|
||||
assert len(searches) == 0
|
||||
mock_neo4j.execute_query.assert_called_once()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_find_unprocessed_searches_with_results(
|
||||
consolidation_service,
|
||||
mock_neo4j,
|
||||
sample_unprocessed_searches
|
||||
):
|
||||
"""Test finding unprocessed searches with results."""
|
||||
# Mock Neo4j response
|
||||
mock_neo4j.execute_query.return_value = sample_unprocessed_searches
|
||||
|
||||
searches = await consolidation_service._find_unprocessed_searches(
|
||||
lookback_days=7,
|
||||
limit=10
|
||||
)
|
||||
|
||||
assert len(searches) == 2
|
||||
assert searches[0]['query'] == 'docker orchestration kubernetes'
|
||||
assert searches[1]['query'] == 'python async programming'
|
||||
mock_neo4j.execute_query.assert_called_once()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_web_results(consolidation_service, mock_neo4j, sample_web_results):
|
||||
"""Test retrieving web results for a search."""
|
||||
# Mock Neo4j response
|
||||
mock_neo4j.execute_query.return_value = sample_web_results
|
||||
|
||||
results = await consolidation_service._get_web_results(TEST_SEARCH_ID)
|
||||
|
||||
assert len(results) == 3
|
||||
assert results[0]['title'] == 'Kubernetes Documentation'
|
||||
assert results[1]['url'] == 'https://docs.docker.com/swarm'
|
||||
mock_neo4j.execute_query.assert_called_once()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_web_results_empty(consolidation_service, mock_neo4j):
|
||||
"""Test retrieving web results when none exist."""
|
||||
mock_neo4j.execute_query.return_value = []
|
||||
|
||||
results = await consolidation_service._get_web_results(TEST_SEARCH_ID)
|
||||
|
||||
assert len(results) == 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_analyze_web_results_with_novel_info(
|
||||
consolidation_service,
|
||||
mock_ollama,
|
||||
sample_web_results,
|
||||
sample_llm_analysis
|
||||
):
|
||||
"""Test analyzing web results with Ollama - novel info found."""
|
||||
# Mock Ollama response
|
||||
mock_ollama.generate_text.return_value = json.dumps(sample_llm_analysis)
|
||||
|
||||
analysis = await consolidation_service._analyze_web_results(
|
||||
query="docker orchestration",
|
||||
web_results=sample_web_results,
|
||||
keywords=["docker", "orchestration"]
|
||||
)
|
||||
|
||||
assert analysis is not None
|
||||
assert analysis['has_novel_info'] is True
|
||||
assert len(analysis['new_pages']) == 1
|
||||
assert len(analysis['update_pages']) == 1
|
||||
assert len(analysis['new_entities']) == 2
|
||||
mock_ollama.generate_text.assert_called_once()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_analyze_web_results_no_novel_info(
|
||||
consolidation_service,
|
||||
mock_ollama,
|
||||
sample_web_results
|
||||
):
|
||||
"""Test analyzing web results - no novel info."""
|
||||
# Mock Ollama response with no novel info
|
||||
analysis_no_novel = {
|
||||
"has_novel_info": False,
|
||||
"new_pages": [],
|
||||
"update_pages": [],
|
||||
"new_entities": []
|
||||
}
|
||||
mock_ollama.generate_text.return_value = json.dumps(analysis_no_novel)
|
||||
|
||||
analysis = await consolidation_service._analyze_web_results(
|
||||
query="common topic",
|
||||
web_results=sample_web_results,
|
||||
keywords=[]
|
||||
)
|
||||
|
||||
assert analysis is not None
|
||||
assert analysis['has_novel_info'] is False
|
||||
assert len(analysis['new_pages']) == 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_analyze_web_results_invalid_json(
|
||||
consolidation_service,
|
||||
mock_ollama,
|
||||
sample_web_results
|
||||
):
|
||||
"""Test analyzing web results with invalid JSON response."""
|
||||
# Mock Ollama response with invalid JSON
|
||||
mock_ollama.generate_text.return_value = "This is not JSON"
|
||||
|
||||
analysis = await consolidation_service._analyze_web_results(
|
||||
query="test query",
|
||||
web_results=sample_web_results,
|
||||
keywords=[]
|
||||
)
|
||||
|
||||
assert analysis is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_analyze_web_results_json_in_markdown(
|
||||
consolidation_service,
|
||||
mock_ollama,
|
||||
sample_web_results,
|
||||
sample_llm_analysis
|
||||
):
|
||||
"""Test extracting JSON from markdown-wrapped response."""
|
||||
# Mock Ollama response with JSON wrapped in markdown
|
||||
wrapped_response = f"""Here's the analysis:
|
||||
|
||||
```json
|
||||
{json.dumps(sample_llm_analysis)}
|
||||
```
|
||||
|
||||
Hope this helps!"""
|
||||
mock_ollama.generate_text.return_value = wrapped_response
|
||||
|
||||
analysis = await consolidation_service._analyze_web_results(
|
||||
query="test",
|
||||
web_results=sample_web_results,
|
||||
keywords=[]
|
||||
)
|
||||
|
||||
assert analysis is not None
|
||||
assert analysis['has_novel_info'] is True
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_mark_search_processed(consolidation_service, mock_neo4j):
|
||||
"""Test marking search as processed."""
|
||||
await consolidation_service._mark_search_processed(TEST_SEARCH_ID)
|
||||
|
||||
mock_neo4j.execute_query.assert_called_once()
|
||||
call_args = mock_neo4j.execute_query.call_args
|
||||
assert TEST_SEARCH_ID in str(call_args)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_process_search_insufficient_web_results(
|
||||
consolidation_service,
|
||||
sample_unprocessed_searches
|
||||
):
|
||||
"""Test processing search with insufficient web results."""
|
||||
search = sample_unprocessed_searches[1].copy()
|
||||
search['web_count'] = 1 # Below minimum
|
||||
|
||||
result = await consolidation_service._process_search(
|
||||
search=search,
|
||||
min_web_results=2,
|
||||
dry_run=False
|
||||
)
|
||||
|
||||
assert result is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_process_search_no_web_results_in_db(
|
||||
consolidation_service,
|
||||
mock_neo4j,
|
||||
sample_unprocessed_searches
|
||||
):
|
||||
"""Test processing search when web results not found in DB."""
|
||||
mock_neo4j.execute_query.return_value = []
|
||||
|
||||
result = await consolidation_service._process_search(
|
||||
search=sample_unprocessed_searches[0],
|
||||
min_web_results=2,
|
||||
dry_run=False
|
||||
)
|
||||
|
||||
assert result is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_process_search_dry_run(
|
||||
consolidation_service,
|
||||
mock_neo4j,
|
||||
mock_ollama,
|
||||
sample_unprocessed_searches,
|
||||
sample_web_results,
|
||||
sample_llm_analysis
|
||||
):
|
||||
"""Test processing search in dry run mode."""
|
||||
# Mock responses
|
||||
mock_neo4j.execute_query.return_value = sample_web_results
|
||||
mock_ollama.generate_text.return_value = json.dumps(sample_llm_analysis)
|
||||
|
||||
result = await consolidation_service._process_search(
|
||||
search=sample_unprocessed_searches[0],
|
||||
min_web_results=2,
|
||||
dry_run=True
|
||||
)
|
||||
|
||||
assert result is not None
|
||||
assert result.search_id == 'search-1'
|
||||
assert result.pages_created == 1
|
||||
assert result.pages_updated == 1
|
||||
assert result.entities_added == 2
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_consolidate_knowledge_no_searches(
|
||||
consolidation_service,
|
||||
mock_neo4j
|
||||
):
|
||||
"""Test consolidation when no unprocessed searches found."""
|
||||
mock_neo4j.execute_query.return_value = []
|
||||
|
||||
response = await consolidation_service.consolidate_knowledge(
|
||||
process_limit=10,
|
||||
lookback_days=7,
|
||||
min_web_results=2,
|
||||
dry_run=False
|
||||
)
|
||||
|
||||
assert response.total_found == 0
|
||||
assert response.processed_count == 0
|
||||
assert response.pages_created == 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_consolidate_knowledge_success(
|
||||
consolidation_service,
|
||||
mock_neo4j,
|
||||
mock_ollama,
|
||||
mock_wiki,
|
||||
sample_unprocessed_searches,
|
||||
sample_web_results,
|
||||
sample_llm_analysis
|
||||
):
|
||||
"""Test successful knowledge consolidation."""
|
||||
# Mock finding searches and entity creation
|
||||
# Each search processes: get web results, add 2 entities, mark processed
|
||||
mock_neo4j.execute_query.side_effect = [
|
||||
sample_unprocessed_searches, # Find searches
|
||||
sample_web_results, # Get web results for search 1
|
||||
None, # Add entity 1 (Kubernetes)
|
||||
None, # Add entity 2 (Docker Swarm)
|
||||
None, # Mark search 1 processed
|
||||
sample_web_results, # Get web results for search 2
|
||||
None, # Add entity 1 (Kubernetes)
|
||||
None, # Add entity 2 (Docker Swarm)
|
||||
None, # Mark search 2 processed
|
||||
]
|
||||
|
||||
# Mock wiki operations
|
||||
mock_wiki.search_pages.return_value = [] # No existing pages
|
||||
mock_wiki.create_page.return_value = None
|
||||
mock_wiki.update_page.return_value = None
|
||||
mock_wiki.get_page.return_value = None
|
||||
|
||||
# Mock LLM analysis and WikiPageWriter LLM calls
|
||||
mock_ollama.generate_text.return_value = json.dumps(sample_llm_analysis)
|
||||
|
||||
response = await consolidation_service.consolidate_knowledge(
|
||||
process_limit=10,
|
||||
lookback_days=7,
|
||||
min_web_results=2,
|
||||
dry_run=False
|
||||
)
|
||||
|
||||
assert response.total_found == 2
|
||||
assert response.processed_count == 2
|
||||
assert response.dry_run is False
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_consolidate_knowledge_with_errors(
|
||||
consolidation_service,
|
||||
mock_neo4j,
|
||||
mock_ollama,
|
||||
sample_unprocessed_searches
|
||||
):
|
||||
"""Test consolidation with some searches failing."""
|
||||
# Mock finding searches - return empty for web results to trigger internal error handling
|
||||
mock_neo4j.execute_query.side_effect = [
|
||||
sample_unprocessed_searches, # Find searches
|
||||
[], # Empty web results for search 1 (causes skip, not error)
|
||||
[], # Empty web results for search 2 (causes skip, not error)
|
||||
]
|
||||
|
||||
response = await consolidation_service.consolidate_knowledge(
|
||||
process_limit=10,
|
||||
lookback_days=7,
|
||||
min_web_results=2,
|
||||
dry_run=False
|
||||
)
|
||||
|
||||
assert response.total_found == 2
|
||||
# Both searches skipped due to no web results (not errors)
|
||||
assert response.processed_count == 0
|
||||
|
||||
|
||||
# Integration Tests (API Endpoint)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_consolidation_endpoint_minimal_request(consolidation_service):
|
||||
"""Test consolidation endpoint with minimal request."""
|
||||
from fastapi.testclient import TestClient
|
||||
from src.main import app
|
||||
|
||||
# This would require proper test client setup
|
||||
# Placeholder for integration test structure
|
||||
request = ConsolidationRequest()
|
||||
assert request.process_limit == 10
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_consolidation_endpoint_custom_config(consolidation_service):
|
||||
"""Test consolidation endpoint with custom configuration."""
|
||||
request = ConsolidationRequest(
|
||||
process_limit=5,
|
||||
lookback_days=14,
|
||||
min_web_results=3,
|
||||
dry_run=True
|
||||
)
|
||||
|
||||
assert request.process_limit == 5
|
||||
assert request.lookback_days == 14
|
||||
assert request.min_web_results == 3
|
||||
assert request.dry_run is True
|
||||
|
||||
|
||||
# Edge Cases
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_consolidate_with_max_limits(consolidation_service, mock_neo4j):
|
||||
"""Test consolidation with maximum limits."""
|
||||
mock_neo4j.execute_query.return_value = []
|
||||
|
||||
response = await consolidation_service.consolidate_knowledge(
|
||||
process_limit=100, # Max
|
||||
lookback_days=90, # Max
|
||||
min_web_results=20, # Max
|
||||
dry_run=True
|
||||
)
|
||||
|
||||
assert response.total_found == 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_analyze_empty_web_results(consolidation_service, mock_ollama):
|
||||
"""Test analyzing with empty web results list."""
|
||||
mock_ollama.generate_text.return_value = json.dumps({
|
||||
"has_novel_info": False,
|
||||
"new_pages": [],
|
||||
"update_pages": [],
|
||||
"new_entities": []
|
||||
})
|
||||
|
||||
analysis = await consolidation_service._analyze_web_results(
|
||||
query="test",
|
||||
web_results=[],
|
||||
keywords=[]
|
||||
)
|
||||
|
||||
# Should still call LLM but return no novel info
|
||||
assert analysis is not None
|
||||
|
||||
|
||||
# Performance/Load Tests (optional)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_process_many_searches_dry_run(
|
||||
consolidation_service,
|
||||
mock_neo4j,
|
||||
mock_ollama
|
||||
):
|
||||
"""Test processing many searches in dry run mode."""
|
||||
# Generate many test searches
|
||||
many_searches = [
|
||||
{
|
||||
'id': f'search-{i}',
|
||||
'query': f'test query {i}',
|
||||
'user': TEST_USER,
|
||||
'timestamp': datetime.now().isoformat(),
|
||||
'total_results': 5,
|
||||
'web_count': 3,
|
||||
'keywords': ['test']
|
||||
}
|
||||
for i in range(50)
|
||||
]
|
||||
|
||||
mock_neo4j.execute_query.return_value = many_searches[:10] # Limit by config
|
||||
|
||||
response = await consolidation_service.consolidate_knowledge(
|
||||
process_limit=10,
|
||||
lookback_days=7,
|
||||
min_web_results=2,
|
||||
dry_run=True
|
||||
)
|
||||
|
||||
# Should only process up to limit
|
||||
assert response.total_found == 10
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
pytest.main([__file__, "-v", "-s"])
|
||||
Reference in New Issue
Block a user