Apply llm-findings.md recommendations: Web results analysis (temp 0.0): - Add ANALYSIS STEPS for chain-of-thought reasoning - Strict RULES section with negative constraints: - "Do NOT suggest pages with insufficient info" - "Do NOT invent entities not mentioned" - "Do NOT suggest paths outside taxonomy" - Conservative approach: quality over quantity - Removed "be INCLUSIVE" guidance (caused over-suggestion) 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
940 lines
35 KiB
Python
940 lines
35 KiB
Python
"""
|
|
Knowledge Consolidation Service (Librarian Logic)
|
|
|
|
Processes unprocessed SearchQuery nodes from HybridRAG searches
|
|
to consolidate new knowledge into wiki pages.
|
|
|
|
This service:
|
|
1. Queries Neo4j for unprocessed SearchQuery nodes
|
|
2. Analyzes web results with Ollama for novel information
|
|
3. Creates/updates wiki pages with new facts
|
|
4. Updates knowledge graph with new entities
|
|
5. Marks SearchQuery nodes as processed
|
|
"""
|
|
import logging
|
|
import json
|
|
from datetime import datetime, timedelta
|
|
from typing import List, Dict, Any, Optional
|
|
|
|
from src.clients.neo4j_client import Neo4jClient
|
|
from src.clients.ollama_client import OllamaClient
|
|
from src.clients.wikijs_client import WikiJSClient
|
|
from src.services.wiki_page_writer import WikiPageWriter
|
|
from src.models.consolidation import (
|
|
SearchQueryInfo,
|
|
ConsolidationResult,
|
|
ConsolidationResponse
|
|
)
|
|
from src.config import Settings
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
class ConsolidationService:
|
|
"""
|
|
Service for consolidating knowledge from search results.
|
|
"""
|
|
|
|
def __init__(
|
|
self,
|
|
neo4j: Neo4jClient,
|
|
ollama: OllamaClient,
|
|
wiki: WikiJSClient,
|
|
settings: Settings,
|
|
ingestion_service: Optional["IngestionService"] = None
|
|
):
|
|
self.neo4j = neo4j
|
|
self.ollama = ollama
|
|
self.wiki = wiki
|
|
self.settings = settings
|
|
self.wiki_page_writer = WikiPageWriter(ollama_client=ollama, settings=settings)
|
|
self.ingestion_service = ingestion_service # Optional to avoid circular dependency
|
|
|
|
async def consolidate_knowledge(
|
|
self,
|
|
process_limit: int = 10,
|
|
lookback_days: int = 7,
|
|
min_web_results: int = 2,
|
|
dry_run: bool = False
|
|
) -> ConsolidationResponse:
|
|
"""
|
|
Process unprocessed search queries and consolidate knowledge.
|
|
|
|
Args:
|
|
process_limit: Maximum searches to process
|
|
lookback_days: Only process searches from last N days
|
|
min_web_results: Minimum web results required to consolidate
|
|
dry_run: If True, analyze but don't create pages
|
|
|
|
Returns:
|
|
ConsolidationResponse with processing results
|
|
"""
|
|
logger.info(f"Starting knowledge consolidation")
|
|
logger.info(f"Limits: process={process_limit}, lookback={lookback_days}d, min_web={min_web_results}")
|
|
if dry_run:
|
|
logger.warning("DRY RUN MODE - will not create wiki pages")
|
|
|
|
# Find unprocessed searches
|
|
unprocessed = await self._find_unprocessed_searches(lookback_days, process_limit)
|
|
|
|
if not unprocessed:
|
|
logger.info("No unprocessed searches found")
|
|
return ConsolidationResponse(
|
|
total_found=0,
|
|
processed_count=0,
|
|
pages_created=0,
|
|
pages_updated=0,
|
|
entities_added=0,
|
|
errors=[],
|
|
results=[],
|
|
dry_run=dry_run
|
|
)
|
|
|
|
logger.info(f"Found {len(unprocessed)} unprocessed searches")
|
|
|
|
# Process each search
|
|
results: List[ConsolidationResult] = []
|
|
total_pages_created = 0
|
|
total_pages_updated = 0
|
|
total_entities_added = 0
|
|
errors: List[str] = []
|
|
|
|
for search in unprocessed:
|
|
try:
|
|
result = await self._process_search(
|
|
search=search,
|
|
min_web_results=min_web_results,
|
|
dry_run=dry_run
|
|
)
|
|
|
|
if result:
|
|
results.append(result)
|
|
total_pages_created += result.pages_created
|
|
total_pages_updated += result.pages_updated
|
|
total_entities_added += result.entities_added
|
|
|
|
# Mark as processed if not dry run (even if skipped)
|
|
# This prevents searches from accumulating when they don't meet criteria
|
|
if not dry_run:
|
|
await self._mark_search_processed(search['id'])
|
|
|
|
except Exception as e:
|
|
error_msg = f"Search {search['id'][:8]}: {str(e)}"
|
|
logger.error(f"Failed to process search: {error_msg}", exc_info=True)
|
|
errors.append(error_msg)
|
|
results.append(ConsolidationResult(
|
|
search_id=search['id'],
|
|
query=search['query'],
|
|
error=str(e)
|
|
))
|
|
|
|
# Mark as processed even on error (to avoid retrying failed searches forever)
|
|
if not dry_run:
|
|
await self._mark_search_processed(search['id'])
|
|
|
|
# Build response
|
|
processed_count = len([r for r in results if not r.error])
|
|
|
|
response = ConsolidationResponse(
|
|
total_found=len(unprocessed),
|
|
processed_count=processed_count,
|
|
pages_created=total_pages_created,
|
|
pages_updated=total_pages_updated,
|
|
entities_added=total_entities_added,
|
|
errors=errors,
|
|
results=results,
|
|
dry_run=dry_run
|
|
)
|
|
|
|
logger.info(
|
|
f"Consolidation complete: {processed_count}/{len(unprocessed)} searches, "
|
|
f"{total_pages_created} pages created, {total_pages_updated} updated, "
|
|
f"{total_entities_added} entities added"
|
|
)
|
|
|
|
return response
|
|
|
|
async def _find_unprocessed_searches(
|
|
self,
|
|
lookback_days: int,
|
|
limit: int
|
|
) -> List[Dict[str, Any]]:
|
|
"""
|
|
Find unprocessed SearchQuery nodes from Neo4j.
|
|
"""
|
|
lookback_date = datetime.now() - timedelta(days=lookback_days)
|
|
|
|
query = """
|
|
MATCH (sq:SearchQuery {processed: false})
|
|
WHERE sq.timestamp > datetime($lookback_date)
|
|
RETURN sq.id as id,
|
|
sq.query as query,
|
|
sq.user as user,
|
|
sq.timestamp as timestamp,
|
|
sq.total_results as total_results,
|
|
sq.web_count as web_count,
|
|
sq.keywords as keywords
|
|
ORDER BY sq.timestamp DESC
|
|
LIMIT $limit
|
|
"""
|
|
|
|
try:
|
|
results = await self.neo4j.execute_query(
|
|
query,
|
|
{
|
|
"lookback_date": lookback_date.isoformat(),
|
|
"limit": limit
|
|
}
|
|
)
|
|
|
|
searches = []
|
|
for record in results:
|
|
searches.append({
|
|
'id': record['id'],
|
|
'query': record['query'],
|
|
'user': record['user'],
|
|
'timestamp': record['timestamp'],
|
|
'total_results': record.get('total_results', 0),
|
|
'web_count': record.get('web_count', 0),
|
|
'keywords': record.get('keywords', [])
|
|
})
|
|
|
|
return searches
|
|
|
|
except Exception as e:
|
|
logger.error(f"Failed to find unprocessed searches: {e}")
|
|
return []
|
|
|
|
async def _process_search(
|
|
self,
|
|
search: Dict[str, Any],
|
|
min_web_results: int,
|
|
dry_run: bool
|
|
) -> Optional[ConsolidationResult]:
|
|
"""
|
|
Process a single search query for knowledge consolidation.
|
|
"""
|
|
search_id = search['id']
|
|
query = search['query']
|
|
user = search['user']
|
|
web_count = search.get('web_count', 0)
|
|
|
|
logger.info(f"Processing: '{query}' (user: {user}, web: {web_count})")
|
|
|
|
# Skip if insufficient web results
|
|
if web_count < min_web_results:
|
|
logger.info(f"Skipping - insufficient web results ({web_count} < {min_web_results})")
|
|
return None
|
|
|
|
# Get web results from SearchQuery
|
|
web_results = await self._get_web_results(search_id)
|
|
if not web_results:
|
|
logger.info("No web results found in database")
|
|
return None
|
|
|
|
logger.info(f"Retrieved {len(web_results)} web results")
|
|
|
|
# Analyze web results with Ollama for novel information
|
|
analysis = await self._analyze_web_results(
|
|
query=query,
|
|
web_results=web_results,
|
|
keywords=search.get('keywords', []),
|
|
user=user
|
|
)
|
|
|
|
if not analysis or not analysis.get('has_novel_info'):
|
|
logger.info("No novel information found")
|
|
return ConsolidationResult(
|
|
search_id=search_id,
|
|
query=query
|
|
)
|
|
|
|
# Extract consolidation actions
|
|
pages_to_create = analysis.get('new_pages', [])
|
|
pages_to_update = analysis.get('update_pages', [])
|
|
new_entities = analysis.get('new_entities', [])
|
|
|
|
logger.info(
|
|
f"Analysis: {len(pages_to_create)} new pages, "
|
|
f"{len(pages_to_update)} updates, {len(new_entities)} entities"
|
|
)
|
|
|
|
if dry_run:
|
|
logger.info("[DRY RUN] Would create/update pages and entities")
|
|
return ConsolidationResult(
|
|
search_id=search_id,
|
|
query=query,
|
|
pages_created=len(pages_to_create),
|
|
pages_updated=len(pages_to_update),
|
|
entities_added=len(new_entities)
|
|
)
|
|
|
|
# Create/update wiki pages
|
|
pages_created = 0
|
|
pages_updated = 0
|
|
entities_added = 0
|
|
|
|
# Create new pages
|
|
for page_data in pages_to_create:
|
|
try:
|
|
await self._create_or_consolidate_page(
|
|
user=user,
|
|
title=page_data.get('title'),
|
|
path=page_data.get('path'),
|
|
summary=page_data.get('summary'),
|
|
source_query=query,
|
|
web_results=web_results
|
|
)
|
|
pages_created += 1
|
|
logger.info(f"Created page: {page_data.get('title')}")
|
|
except Exception as e:
|
|
logger.error(f"Failed to create page {page_data.get('title')}: {e}")
|
|
|
|
# Update existing pages
|
|
for page_data in pages_to_update:
|
|
try:
|
|
await self._update_page_with_facts(
|
|
title=page_data.get('title'),
|
|
new_facts=page_data.get('new_facts', []),
|
|
source_url=page_data.get('source_url'),
|
|
user=user
|
|
)
|
|
pages_updated += 1
|
|
logger.info(f"Updated page: {page_data.get('title')}")
|
|
except Exception as e:
|
|
logger.error(f"Failed to update page {page_data.get('title')}: {e}")
|
|
|
|
# Add new entities to graph
|
|
for entity_data in new_entities:
|
|
try:
|
|
await self._add_entity_to_graph(
|
|
user=user,
|
|
entity_name=entity_data.get('name'),
|
|
entity_type=entity_data.get('type'),
|
|
description=entity_data.get('description'),
|
|
source_search_id=search_id
|
|
)
|
|
entities_added += 1
|
|
logger.info(f"Added entity: {entity_data.get('name')}")
|
|
except Exception as e:
|
|
logger.error(f"Failed to add entity {entity_data.get('name')}: {e}")
|
|
|
|
return ConsolidationResult(
|
|
search_id=search_id,
|
|
query=query,
|
|
pages_created=pages_created,
|
|
pages_updated=pages_updated,
|
|
entities_added=entities_added
|
|
)
|
|
|
|
async def _get_web_results(self, search_id: str) -> List[Dict[str, Any]]:
|
|
"""Get web results for a search from Neo4j."""
|
|
query = """
|
|
MATCH (sq:SearchQuery {id: $search_id})-[f:FOUND]->(wr:WebResult)
|
|
RETURN wr.url as url,
|
|
wr.title as title,
|
|
wr.content as content,
|
|
f.rank as rank,
|
|
f.rrf_score as rrf_score
|
|
ORDER BY f.rank
|
|
LIMIT 20
|
|
"""
|
|
|
|
try:
|
|
results = await self.neo4j.execute_query(query, {"search_id": search_id})
|
|
|
|
web_results = []
|
|
for record in results:
|
|
web_results.append({
|
|
'url': record['url'],
|
|
'title': record['title'],
|
|
'content': record['content'],
|
|
'rank': record['rank'],
|
|
'rrf_score': record['rrf_score']
|
|
})
|
|
|
|
return web_results
|
|
|
|
except Exception as e:
|
|
logger.error(f"Failed to get web results: {e}")
|
|
return []
|
|
|
|
async def _analyze_web_results(
|
|
self,
|
|
query: str,
|
|
web_results: List[Dict[str, Any]],
|
|
keywords: List[str],
|
|
user: str = "jpmschweitzer"
|
|
) -> Optional[Dict[str, Any]]:
|
|
"""
|
|
Analyze web results with Ollama for novel information.
|
|
|
|
Returns analysis with has_novel_info, new_pages, update_pages, new_entities.
|
|
"""
|
|
# Fetch existing taxonomy structure for this user
|
|
try:
|
|
taxonomy_structure = await self.wiki.get_taxonomy_structure(f"users/{user}")
|
|
existing_paths_info = self._format_taxonomy_for_prompt(taxonomy_structure)
|
|
logger.info(f"Fetched taxonomy with {len(taxonomy_structure)} categories for user {user}")
|
|
except Exception as e:
|
|
logger.warning(f"Failed to fetch taxonomy structure: {e}")
|
|
existing_paths_info = ""
|
|
|
|
# Build analysis prompt
|
|
web_summary = "\n\n".join([
|
|
f"[{i+1}] {r['title']}\n{r['url']}\n{r['content'][:300]}..."
|
|
for i, r in enumerate(web_results[:5])
|
|
])
|
|
|
|
prompt = f"""You are a Librarian helping build a personal knowledge base and extended memory system.
|
|
|
|
Analyze these web search results for information worth documenting in our personal wiki.
|
|
|
|
Query: "{query}"
|
|
Keywords: {', '.join(keywords) if keywords else 'none'}
|
|
|
|
Web Results:
|
|
{web_summary}
|
|
|
|
This is a PERSONAL knowledge base using Schema.org-aligned taxonomy that captures:
|
|
- People: Family members, friends, colleagues, public figures (Schema.org: Person)
|
|
- Companies: Businesses, organizations, institutions (Schema.org: Organization)
|
|
- Places: Locations, restaurants, travel destinations (Schema.org: Place)
|
|
- Entertainment: Books, movies, TV, music, games (Schema.org: CreativeWork)
|
|
- Recipes: Food, cooking techniques, ingredients (Schema.org: CreativeWork/Recipe)
|
|
- Products: Purchased items, gear, tools, equipment (Schema.org: Product)
|
|
- Technology: Software, applications, infrastructure (Schema.org: SoftwareApplication)
|
|
- Health: Medical info, fitness, wellness (Schema.org: MedicalEntity)
|
|
- Events: Concerts, travel, appointments, important dates (Schema.org: Event)
|
|
- Hobbies: Personal interests, activities, pastimes (Custom extension)
|
|
- Projects: Work projects, personal projects (Schema.org: Project)
|
|
- Reference: General knowledge, how-tos (Custom extension)
|
|
|
|
ANALYSIS STEPS:
|
|
1. Read each web result carefully for substantive, factual content
|
|
2. Identify genuinely novel information not likely already known
|
|
3. Match topics to appropriate taxonomy categories
|
|
4. Generate valid paths following the exact format below
|
|
|
|
RULES:
|
|
- Do NOT suggest pages for topics with insufficient information in results
|
|
- Do NOT invent entities not explicitly mentioned in results
|
|
- Do NOT suggest paths that don't match the taxonomy exactly
|
|
- Do NOT suggest generic or vague page topics
|
|
- Be CONSERVATIVE - fewer high-quality suggestions is better than many low-quality ones
|
|
- ONLY suggest documentation for substantive, specific information
|
|
|
|
**CRITICAL: Use ONLY these Schema.org-aligned path prefixes (case-sensitive):**
|
|
|
|
- People: `people/<name>` (Schema.org: Person)
|
|
- Companies: `companies/<company-name>` (Schema.org: Organization)
|
|
- Places: `places/<location>` (Schema.org: Place)
|
|
- Entertainment (Schema.org: CreativeWork):
|
|
- Books: `entertainment/books/<title>`
|
|
- Movies: `entertainment/movies/<title>`
|
|
- TV: `entertainment/tv/<title>`
|
|
- Music: `entertainment/music/<artist-or-album>`
|
|
- Games: `entertainment/games/<title>`
|
|
- Recipes: `recipes/<cuisine-or-category>/<dish>` (Schema.org: Recipe)
|
|
- Products: `products/<category>/<product-name>` (Schema.org: Product)
|
|
- Technology: `technology/<category>/<topic>` (Schema.org: SoftwareApplication)
|
|
- Health: `health/<category>/<topic>` (Schema.org: MedicalEntity)
|
|
- Events: `events/<event-type>/<event-name>` (Schema.org: Event)
|
|
- Hobbies: `hobbies/<hobby-name>` (Custom extension)
|
|
- Projects: `projects/<project-name>` (Schema.org: Project)
|
|
- Reference: `reference/<category>/<topic>` (Custom extension)
|
|
|
|
**Path Rules:**
|
|
- Use lowercase with hyphens (kebab-case): "machine-learning" not "Machine_Learning"
|
|
- Keep paths 2-3 levels deep maximum
|
|
- Be consistent with existing paths when possible
|
|
|
|
{existing_paths_info}
|
|
|
|
Return ONLY valid JSON:
|
|
{{
|
|
"has_novel_info": true,
|
|
"new_pages": [
|
|
{{"title": "Page Title", "path": "companies/example-company", "summary": "What information to include"}}
|
|
],
|
|
"update_pages": [
|
|
{{"title": "Existing Page", "new_facts": ["fact 1"], "source_url": "url"}}
|
|
],
|
|
"new_entities": [
|
|
{{"name": "Entity Name", "type": "person/place/thing/concept/recipe/media", "description": "Brief description"}}
|
|
]
|
|
}}
|
|
|
|
JSON:"""
|
|
|
|
try:
|
|
# Call Ollama for analysis (temperature=0.0 for consistent classification)
|
|
response = await self.ollama.generate_text(
|
|
prompt=prompt,
|
|
model=self.settings.ollama_model,
|
|
stream=False,
|
|
temperature=0.0
|
|
)
|
|
|
|
if not response:
|
|
logger.warning("Empty response from Ollama")
|
|
return None
|
|
|
|
# Extract JSON from response
|
|
response_clean = response.strip()
|
|
if '{' in response_clean:
|
|
json_start = response_clean.find('{')
|
|
json_end = response_clean.rfind('}') + 1
|
|
response_clean = response_clean[json_start:json_end]
|
|
|
|
analysis = json.loads(response_clean)
|
|
return analysis
|
|
|
|
except json.JSONDecodeError as e:
|
|
logger.error(f"Failed to parse Ollama response as JSON: {e}")
|
|
logger.debug(f"Response was: {response[:500]}")
|
|
return None
|
|
except Exception as e:
|
|
logger.error(f"Analysis failed: {e}", exc_info=True)
|
|
return None
|
|
|
|
def _format_taxonomy_for_prompt(self, taxonomy: Dict[str, List[str]]) -> str:
|
|
"""
|
|
Format taxonomy structure for inclusion in LLM prompt.
|
|
|
|
Args:
|
|
taxonomy: Dict mapping categories to subcategories
|
|
|
|
Returns:
|
|
Formatted string showing existing paths
|
|
"""
|
|
if not taxonomy:
|
|
return ""
|
|
|
|
lines = ["**Existing paths in your wiki (PREFER these over creating new ones):**"]
|
|
for category, subcategories in taxonomy.items():
|
|
if subcategories:
|
|
lines.append(f"- {category}/")
|
|
for sub in subcategories:
|
|
lines.append(f" - {category}/{sub}/")
|
|
else:
|
|
lines.append(f"- {category}/")
|
|
|
|
lines.append("")
|
|
lines.append("**IMPORTANT:** If a suitable existing path exists, use it instead of creating a new category.")
|
|
lines.append("Example: NATO should go in `reference/political-entities/` not a new `reference/military-alliances/`")
|
|
|
|
return "\n".join(lines)
|
|
|
|
async def _mark_search_processed(self, search_id: str):
|
|
"""Mark SearchQuery node as processed."""
|
|
query = """
|
|
MATCH (sq:SearchQuery {id: $search_id})
|
|
SET sq.processed = true,
|
|
sq.processed_at = datetime()
|
|
RETURN sq.id
|
|
"""
|
|
|
|
try:
|
|
await self.neo4j.execute_query(query, {"search_id": search_id})
|
|
logger.debug(f"Marked search {search_id} as processed")
|
|
except Exception as e:
|
|
logger.error(f"Failed to mark search as processed: {e}")
|
|
|
|
async def _apply_bidirectional_entity_linking(
|
|
self,
|
|
page_id: int,
|
|
page_title: str,
|
|
user: str
|
|
) -> Dict[str, int]:
|
|
"""
|
|
Apply bidirectional entity linking after page creation/update.
|
|
|
|
This runs AFTER ingestion so entities are extracted and in the graph.
|
|
|
|
Steps:
|
|
1. Link entities in the new page (forward links to existing entities)
|
|
2. Find pages that mention the new entity (reverse references)
|
|
3. Link entities in those pages (backward links to the new entity)
|
|
|
|
Args:
|
|
page_id: Wiki page ID
|
|
page_title: Page title (used to find reverse references)
|
|
user: User identifier
|
|
|
|
Returns:
|
|
Dict with link counts: {
|
|
"forward_links": int, # Links added to the new page
|
|
"backward_links": int, # Links added to other pages pointing to new page
|
|
"pages_updated": int # Number of other pages updated
|
|
}
|
|
"""
|
|
from src.core.multi_tenancy import get_neo4j_user_base_label
|
|
|
|
forward_links = 0
|
|
backward_links = 0
|
|
pages_updated = 0
|
|
|
|
try:
|
|
# Import here to avoid circular dependency
|
|
from src.routers.entity_linking import link_entities_in_page, EntityLinkingRequest
|
|
from src.core.dependencies import get_wiki_service, get_graph_service
|
|
|
|
wiki_service = get_wiki_service()
|
|
graph_service = get_graph_service()
|
|
|
|
# STEP 1: Forward linking - link entities in the new page
|
|
logger.info(f"Step 1/3: Linking entities in page {page_id} ('{page_title}')")
|
|
try:
|
|
forward_result = await link_entities_in_page(
|
|
request=EntityLinkingRequest(
|
|
user=user,
|
|
page_id=page_id,
|
|
create_relationships=True,
|
|
re_index_if_changed=False # Already indexed, no need to re-index
|
|
),
|
|
wiki_service=wiki_service,
|
|
graph_service=graph_service,
|
|
ingestion_service=self.ingestion_service,
|
|
api_key="" # Internal call, no auth needed
|
|
)
|
|
forward_links = forward_result.content_links_added
|
|
logger.info(f"Added {forward_links} forward links in page {page_id}")
|
|
except Exception as e:
|
|
logger.error(f"Failed to add forward links: {e}")
|
|
|
|
# STEP 2: Find reverse references - which pages mention this new entity?
|
|
logger.info(f"Step 2/3: Finding pages that mention '{page_title}'")
|
|
user_base_label = get_neo4j_user_base_label(user)
|
|
|
|
# Query to find documents that mention entities with this page's title
|
|
reverse_query = f"""
|
|
// Find entities with the same name as the page title
|
|
MATCH (e:{user_base_label})
|
|
WHERE toLower(e.name) = toLower($title)
|
|
AND NOT e:Document
|
|
|
|
// Find documents that mention those entities
|
|
MATCH (d:Document)-[r:MENTIONS]->(e)
|
|
WHERE d.page_id <> $page_id // Exclude the page itself
|
|
|
|
RETURN DISTINCT d.page_id as page_id, d.title as title
|
|
LIMIT 50
|
|
"""
|
|
|
|
try:
|
|
reverse_refs = await self.neo4j.execute_query(
|
|
reverse_query,
|
|
{"title": page_title, "page_id": page_id}
|
|
)
|
|
logger.info(f"Found {len(reverse_refs)} pages that mention '{page_title}'")
|
|
except Exception as e:
|
|
logger.error(f"Failed to find reverse references: {e}")
|
|
reverse_refs = []
|
|
|
|
# STEP 3: Backward linking - add links in those pages to the new entity
|
|
if reverse_refs:
|
|
logger.info(f"Step 3/3: Adding backward links in {len(reverse_refs)} pages")
|
|
for ref in reverse_refs:
|
|
try:
|
|
backward_result = await link_entities_in_page(
|
|
request=EntityLinkingRequest(
|
|
user=user,
|
|
page_id=ref['page_id'],
|
|
create_relationships=False, # Relationships already exist
|
|
re_index_if_changed=False # Don't re-index for link updates
|
|
),
|
|
wiki_service=wiki_service,
|
|
graph_service=graph_service,
|
|
ingestion_service=self.ingestion_service,
|
|
api_key=""
|
|
)
|
|
if backward_result.content_links_added > 0:
|
|
backward_links += backward_result.content_links_added
|
|
pages_updated += 1
|
|
logger.info(
|
|
f"Added {backward_result.content_links_added} links "
|
|
f"in page {ref['page_id']} ('{ref['title']}')"
|
|
)
|
|
except Exception as e:
|
|
logger.error(f"Failed to add backward links in page {ref['page_id']}: {e}")
|
|
else:
|
|
logger.info("Step 3/3: No reverse references found, skipping backward linking")
|
|
|
|
return {
|
|
"forward_links": forward_links,
|
|
"backward_links": backward_links,
|
|
"pages_updated": pages_updated
|
|
}
|
|
|
|
except Exception as e:
|
|
logger.error(f"Bidirectional entity linking failed: {e}", exc_info=True)
|
|
return {
|
|
"forward_links": 0,
|
|
"backward_links": 0,
|
|
"pages_updated": 0
|
|
}
|
|
|
|
async def _create_or_consolidate_page(
|
|
self,
|
|
user: str,
|
|
title: str,
|
|
path: str,
|
|
summary: str,
|
|
source_query: str,
|
|
web_results: List[Dict[str, Any]]
|
|
):
|
|
"""
|
|
Create wiki page or consolidate with existing synonym page.
|
|
|
|
Uses WikiPageWriter for intelligent LLM-based content generation:
|
|
- For new pages: Holistic structured content creation
|
|
- For existing pages: Zero-loss reconstruction with conflict detection
|
|
"""
|
|
# Normalize path to user namespace
|
|
if not path.startswith(f"users/{user}"):
|
|
path = f"users/{user}/{path.lstrip('/')}"
|
|
|
|
# Format web results as source information
|
|
source_information = [
|
|
{
|
|
'title': r['title'],
|
|
'url': r['url'],
|
|
'content': r['content']
|
|
}
|
|
for r in web_results[:5] # Top 5 web results
|
|
]
|
|
|
|
# Search for existing pages with similar titles (synonym consolidation)
|
|
existing_pages = await self.wiki.search_pages(title, path_prefix=f"users/{user}")
|
|
|
|
if existing_pages:
|
|
# Page exists - reconstruct with new information using LLM
|
|
logger.info(f"Found existing page for '{title}', will reconstruct with new info")
|
|
page_id = existing_pages[0]['id']
|
|
|
|
# Get current content
|
|
existing_page = await self.wiki.get_page(page_id)
|
|
if existing_page:
|
|
# Build new information text from summary and web results
|
|
new_information = f"{summary}\n\n"
|
|
for r in web_results[:3]:
|
|
new_information += f"- {r['title']}: {r['content'][:200]}...\n"
|
|
|
|
# Use WikiPageWriter to reconstruct with LLM
|
|
reconstructed_content, conflicts = await self.wiki_page_writer.reconstruct_page(
|
|
title=title,
|
|
existing_content=existing_page['content'],
|
|
new_information=new_information,
|
|
new_sources=source_information,
|
|
detect_conflicts=True
|
|
)
|
|
|
|
if conflicts:
|
|
logger.warning(
|
|
f"Detected {len(conflicts)} conflicts when updating '{title}' - "
|
|
"LLM chose most authoritative sources"
|
|
)
|
|
|
|
await self.wiki.update_page(
|
|
page_id=page_id,
|
|
content=reconstructed_content
|
|
)
|
|
logger.info(f"Reconstructed existing page: {title}")
|
|
|
|
# Trigger ingestion to update vectors and graph
|
|
if self.ingestion_service:
|
|
try:
|
|
await self.ingestion_service.ingest_page(
|
|
page_id=page_id,
|
|
user=user,
|
|
force_refresh=True
|
|
)
|
|
logger.info(f"Ingested updated page {page_id} into knowledge base")
|
|
|
|
# Apply bidirectional entity linking after ingestion
|
|
link_stats = await self._apply_bidirectional_entity_linking(
|
|
page_id=page_id,
|
|
page_title=title,
|
|
user=user
|
|
)
|
|
logger.info(
|
|
f"Entity linking complete: {link_stats['forward_links']} forward links, "
|
|
f"{link_stats['backward_links']} backward links "
|
|
f"({link_stats['pages_updated']} pages updated)"
|
|
)
|
|
except Exception as e:
|
|
logger.error(f"Failed to ingest updated page {page_id}: {e}")
|
|
|
|
return
|
|
|
|
# Create new page with LLM-generated structured content
|
|
logger.info(f"Creating new page: {title}")
|
|
|
|
# Use WikiPageWriter to create structured content
|
|
content = await self.wiki_page_writer.create_page(
|
|
title=title,
|
|
topic_summary=summary,
|
|
source_information=source_information,
|
|
entities=None, # Could extract from keywords if available
|
|
related_docs=None
|
|
)
|
|
|
|
# Extract tags from path for dossier organization
|
|
path_parts = path.split('/')
|
|
tags = [part for part in path_parts if part and part not in ['users', user]]
|
|
|
|
created_page = await self.wiki.create_page(
|
|
path=path,
|
|
title=title,
|
|
content=content,
|
|
description=f"Consolidated from search: {source_query}",
|
|
tags=tags[:3], # Limit to 3 tags
|
|
is_published=True
|
|
)
|
|
|
|
page_id = created_page.get("id") if created_page else None
|
|
logger.info(f"Created new page: {path} (page_id: {page_id})")
|
|
|
|
# Trigger ingestion to update vectors and graph
|
|
if self.ingestion_service and page_id:
|
|
try:
|
|
await self.ingestion_service.ingest_page(
|
|
page_id=page_id,
|
|
user=user,
|
|
force_refresh=False # New page, no need to force
|
|
)
|
|
logger.info(f"Ingested new page {page_id} into knowledge base")
|
|
|
|
# Apply bidirectional entity linking after ingestion
|
|
link_stats = await self._apply_bidirectional_entity_linking(
|
|
page_id=page_id,
|
|
page_title=title,
|
|
user=user
|
|
)
|
|
logger.info(
|
|
f"Entity linking complete: {link_stats['forward_links']} forward links, "
|
|
f"{link_stats['backward_links']} backward links "
|
|
f"({link_stats['pages_updated']} pages updated)"
|
|
)
|
|
except Exception as e:
|
|
logger.error(f"Failed to ingest new page {page_id}: {e}")
|
|
|
|
async def _update_page_with_facts(
|
|
self,
|
|
title: str,
|
|
new_facts: List[str],
|
|
source_url: str,
|
|
user: str
|
|
):
|
|
"""
|
|
Update existing page with new facts using LLM reconstruction.
|
|
|
|
Uses WikiPageWriter to intelligently merge facts with zero loss.
|
|
"""
|
|
# Search for page
|
|
pages = await self.wiki.search_pages(title, path_prefix=f"users/{user}")
|
|
|
|
if not pages:
|
|
logger.warning(f"Page '{title}' not found for update")
|
|
return
|
|
|
|
page_id = pages[0]['id']
|
|
existing_page = await self.wiki.get_page(page_id)
|
|
|
|
if not existing_page:
|
|
return
|
|
|
|
# Build new information from facts
|
|
new_information = "\n".join([f"- {fact}" for fact in new_facts])
|
|
|
|
# Format source
|
|
source_information = [{
|
|
'title': source_url,
|
|
'url': source_url,
|
|
'content': new_information
|
|
}]
|
|
|
|
# Use WikiPageWriter to reconstruct with LLM
|
|
reconstructed_content, conflicts = await self.wiki_page_writer.reconstruct_page(
|
|
title=title,
|
|
existing_content=existing_page['content'],
|
|
new_information=new_information,
|
|
new_sources=source_information,
|
|
detect_conflicts=True
|
|
)
|
|
|
|
if conflicts:
|
|
logger.warning(
|
|
f"Detected {len(conflicts)} conflicts when updating '{title}' with new facts"
|
|
)
|
|
|
|
await self.wiki.update_page(
|
|
page_id=page_id,
|
|
content=reconstructed_content
|
|
)
|
|
|
|
# Trigger ingestion to update vectors and graph
|
|
logger.debug(f"ingestion_service available: {self.ingestion_service is not None}")
|
|
if self.ingestion_service:
|
|
try:
|
|
logger.info(f"Starting ingestion for updated page {page_id}")
|
|
await self.ingestion_service.ingest_page(
|
|
page_id=page_id,
|
|
user=user,
|
|
force_refresh=True
|
|
)
|
|
logger.info(f"Ingested updated page {page_id} into knowledge base")
|
|
|
|
# Apply bidirectional entity linking after ingestion
|
|
link_stats = await self._apply_bidirectional_entity_linking(
|
|
page_id=page_id,
|
|
page_title=title,
|
|
user=user
|
|
)
|
|
logger.info(
|
|
f"Entity linking complete: {link_stats['forward_links']} forward links, "
|
|
f"{link_stats['backward_links']} backward links "
|
|
f"({link_stats['pages_updated']} pages updated)"
|
|
)
|
|
except Exception as e:
|
|
logger.error(f"Failed to ingest updated page {page_id}: {e}")
|
|
|
|
async def _add_entity_to_graph(
|
|
self,
|
|
user: str,
|
|
entity_name: str,
|
|
entity_type: str,
|
|
description: str,
|
|
source_search_id: str
|
|
):
|
|
"""Add new entity to knowledge graph."""
|
|
from src.core.multi_tenancy import get_neo4j_user_base_label
|
|
|
|
user_base_label = get_neo4j_user_base_label(user)
|
|
|
|
# Create entity node with appropriate type label
|
|
type_label = entity_type.capitalize() if entity_type else "Entity"
|
|
|
|
query = f"""
|
|
MERGE (e:{user_base_label}:{type_label} {{name: $name}})
|
|
ON CREATE SET
|
|
e.description = $description,
|
|
e.created_at = datetime(),
|
|
e.source = 'librarian_consolidation',
|
|
e.source_search_id = $search_id
|
|
ON MATCH SET
|
|
e.updated_at = datetime()
|
|
RETURN e
|
|
"""
|
|
|
|
try:
|
|
await self.neo4j.execute_query(query, {
|
|
"name": entity_name,
|
|
"description": description,
|
|
"search_id": source_search_id
|
|
})
|
|
logger.debug(f"Added entity to graph: {entity_name} ({entity_type})")
|
|
except Exception as e:
|
|
logger.error(f"Failed to add entity to graph: {e}")
|