Files
library-desk/src/services/consolidation_service.py
T
jpmschweitzerandClaude Opus 4.5 b976da0092 refactor: improve web results analysis prompt
Apply llm-findings.md recommendations:

Web results analysis (temp 0.0):
- Add ANALYSIS STEPS for chain-of-thought reasoning
- Strict RULES section with negative constraints:
  - "Do NOT suggest pages with insufficient info"
  - "Do NOT invent entities not mentioned"
  - "Do NOT suggest paths outside taxonomy"
- Conservative approach: quality over quantity
- Removed "be INCLUSIVE" guidance (caused over-suggestion)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-23 17:22:12 +01:00

940 lines
35 KiB
Python

"""
Knowledge Consolidation Service (Librarian Logic)
Processes unprocessed SearchQuery nodes from HybridRAG searches
to consolidate new knowledge into wiki pages.
This service:
1. Queries Neo4j for unprocessed SearchQuery nodes
2. Analyzes web results with Ollama for novel information
3. Creates/updates wiki pages with new facts
4. Updates knowledge graph with new entities
5. Marks SearchQuery nodes as processed
"""
import logging
import json
from datetime import datetime, timedelta
from typing import List, Dict, Any, Optional
from src.clients.neo4j_client import Neo4jClient
from src.clients.ollama_client import OllamaClient
from src.clients.wikijs_client import WikiJSClient
from src.services.wiki_page_writer import WikiPageWriter
from src.models.consolidation import (
SearchQueryInfo,
ConsolidationResult,
ConsolidationResponse
)
from src.config import Settings
logger = logging.getLogger(__name__)
class ConsolidationService:
"""
Service for consolidating knowledge from search results.
"""
def __init__(
self,
neo4j: Neo4jClient,
ollama: OllamaClient,
wiki: WikiJSClient,
settings: Settings,
ingestion_service: Optional["IngestionService"] = None
):
self.neo4j = neo4j
self.ollama = ollama
self.wiki = wiki
self.settings = settings
self.wiki_page_writer = WikiPageWriter(ollama_client=ollama, settings=settings)
self.ingestion_service = ingestion_service # Optional to avoid circular dependency
async def consolidate_knowledge(
self,
process_limit: int = 10,
lookback_days: int = 7,
min_web_results: int = 2,
dry_run: bool = False
) -> ConsolidationResponse:
"""
Process unprocessed search queries and consolidate knowledge.
Args:
process_limit: Maximum searches to process
lookback_days: Only process searches from last N days
min_web_results: Minimum web results required to consolidate
dry_run: If True, analyze but don't create pages
Returns:
ConsolidationResponse with processing results
"""
logger.info(f"Starting knowledge consolidation")
logger.info(f"Limits: process={process_limit}, lookback={lookback_days}d, min_web={min_web_results}")
if dry_run:
logger.warning("DRY RUN MODE - will not create wiki pages")
# Find unprocessed searches
unprocessed = await self._find_unprocessed_searches(lookback_days, process_limit)
if not unprocessed:
logger.info("No unprocessed searches found")
return ConsolidationResponse(
total_found=0,
processed_count=0,
pages_created=0,
pages_updated=0,
entities_added=0,
errors=[],
results=[],
dry_run=dry_run
)
logger.info(f"Found {len(unprocessed)} unprocessed searches")
# Process each search
results: List[ConsolidationResult] = []
total_pages_created = 0
total_pages_updated = 0
total_entities_added = 0
errors: List[str] = []
for search in unprocessed:
try:
result = await self._process_search(
search=search,
min_web_results=min_web_results,
dry_run=dry_run
)
if result:
results.append(result)
total_pages_created += result.pages_created
total_pages_updated += result.pages_updated
total_entities_added += result.entities_added
# Mark as processed if not dry run (even if skipped)
# This prevents searches from accumulating when they don't meet criteria
if not dry_run:
await self._mark_search_processed(search['id'])
except Exception as e:
error_msg = f"Search {search['id'][:8]}: {str(e)}"
logger.error(f"Failed to process search: {error_msg}", exc_info=True)
errors.append(error_msg)
results.append(ConsolidationResult(
search_id=search['id'],
query=search['query'],
error=str(e)
))
# Mark as processed even on error (to avoid retrying failed searches forever)
if not dry_run:
await self._mark_search_processed(search['id'])
# Build response
processed_count = len([r for r in results if not r.error])
response = ConsolidationResponse(
total_found=len(unprocessed),
processed_count=processed_count,
pages_created=total_pages_created,
pages_updated=total_pages_updated,
entities_added=total_entities_added,
errors=errors,
results=results,
dry_run=dry_run
)
logger.info(
f"Consolidation complete: {processed_count}/{len(unprocessed)} searches, "
f"{total_pages_created} pages created, {total_pages_updated} updated, "
f"{total_entities_added} entities added"
)
return response
async def _find_unprocessed_searches(
self,
lookback_days: int,
limit: int
) -> List[Dict[str, Any]]:
"""
Find unprocessed SearchQuery nodes from Neo4j.
"""
lookback_date = datetime.now() - timedelta(days=lookback_days)
query = """
MATCH (sq:SearchQuery {processed: false})
WHERE sq.timestamp > datetime($lookback_date)
RETURN sq.id as id,
sq.query as query,
sq.user as user,
sq.timestamp as timestamp,
sq.total_results as total_results,
sq.web_count as web_count,
sq.keywords as keywords
ORDER BY sq.timestamp DESC
LIMIT $limit
"""
try:
results = await self.neo4j.execute_query(
query,
{
"lookback_date": lookback_date.isoformat(),
"limit": limit
}
)
searches = []
for record in results:
searches.append({
'id': record['id'],
'query': record['query'],
'user': record['user'],
'timestamp': record['timestamp'],
'total_results': record.get('total_results', 0),
'web_count': record.get('web_count', 0),
'keywords': record.get('keywords', [])
})
return searches
except Exception as e:
logger.error(f"Failed to find unprocessed searches: {e}")
return []
async def _process_search(
self,
search: Dict[str, Any],
min_web_results: int,
dry_run: bool
) -> Optional[ConsolidationResult]:
"""
Process a single search query for knowledge consolidation.
"""
search_id = search['id']
query = search['query']
user = search['user']
web_count = search.get('web_count', 0)
logger.info(f"Processing: '{query}' (user: {user}, web: {web_count})")
# Skip if insufficient web results
if web_count < min_web_results:
logger.info(f"Skipping - insufficient web results ({web_count} < {min_web_results})")
return None
# Get web results from SearchQuery
web_results = await self._get_web_results(search_id)
if not web_results:
logger.info("No web results found in database")
return None
logger.info(f"Retrieved {len(web_results)} web results")
# Analyze web results with Ollama for novel information
analysis = await self._analyze_web_results(
query=query,
web_results=web_results,
keywords=search.get('keywords', []),
user=user
)
if not analysis or not analysis.get('has_novel_info'):
logger.info("No novel information found")
return ConsolidationResult(
search_id=search_id,
query=query
)
# Extract consolidation actions
pages_to_create = analysis.get('new_pages', [])
pages_to_update = analysis.get('update_pages', [])
new_entities = analysis.get('new_entities', [])
logger.info(
f"Analysis: {len(pages_to_create)} new pages, "
f"{len(pages_to_update)} updates, {len(new_entities)} entities"
)
if dry_run:
logger.info("[DRY RUN] Would create/update pages and entities")
return ConsolidationResult(
search_id=search_id,
query=query,
pages_created=len(pages_to_create),
pages_updated=len(pages_to_update),
entities_added=len(new_entities)
)
# Create/update wiki pages
pages_created = 0
pages_updated = 0
entities_added = 0
# Create new pages
for page_data in pages_to_create:
try:
await self._create_or_consolidate_page(
user=user,
title=page_data.get('title'),
path=page_data.get('path'),
summary=page_data.get('summary'),
source_query=query,
web_results=web_results
)
pages_created += 1
logger.info(f"Created page: {page_data.get('title')}")
except Exception as e:
logger.error(f"Failed to create page {page_data.get('title')}: {e}")
# Update existing pages
for page_data in pages_to_update:
try:
await self._update_page_with_facts(
title=page_data.get('title'),
new_facts=page_data.get('new_facts', []),
source_url=page_data.get('source_url'),
user=user
)
pages_updated += 1
logger.info(f"Updated page: {page_data.get('title')}")
except Exception as e:
logger.error(f"Failed to update page {page_data.get('title')}: {e}")
# Add new entities to graph
for entity_data in new_entities:
try:
await self._add_entity_to_graph(
user=user,
entity_name=entity_data.get('name'),
entity_type=entity_data.get('type'),
description=entity_data.get('description'),
source_search_id=search_id
)
entities_added += 1
logger.info(f"Added entity: {entity_data.get('name')}")
except Exception as e:
logger.error(f"Failed to add entity {entity_data.get('name')}: {e}")
return ConsolidationResult(
search_id=search_id,
query=query,
pages_created=pages_created,
pages_updated=pages_updated,
entities_added=entities_added
)
async def _get_web_results(self, search_id: str) -> List[Dict[str, Any]]:
"""Get web results for a search from Neo4j."""
query = """
MATCH (sq:SearchQuery {id: $search_id})-[f:FOUND]->(wr:WebResult)
RETURN wr.url as url,
wr.title as title,
wr.content as content,
f.rank as rank,
f.rrf_score as rrf_score
ORDER BY f.rank
LIMIT 20
"""
try:
results = await self.neo4j.execute_query(query, {"search_id": search_id})
web_results = []
for record in results:
web_results.append({
'url': record['url'],
'title': record['title'],
'content': record['content'],
'rank': record['rank'],
'rrf_score': record['rrf_score']
})
return web_results
except Exception as e:
logger.error(f"Failed to get web results: {e}")
return []
async def _analyze_web_results(
self,
query: str,
web_results: List[Dict[str, Any]],
keywords: List[str],
user: str = "jpmschweitzer"
) -> Optional[Dict[str, Any]]:
"""
Analyze web results with Ollama for novel information.
Returns analysis with has_novel_info, new_pages, update_pages, new_entities.
"""
# Fetch existing taxonomy structure for this user
try:
taxonomy_structure = await self.wiki.get_taxonomy_structure(f"users/{user}")
existing_paths_info = self._format_taxonomy_for_prompt(taxonomy_structure)
logger.info(f"Fetched taxonomy with {len(taxonomy_structure)} categories for user {user}")
except Exception as e:
logger.warning(f"Failed to fetch taxonomy structure: {e}")
existing_paths_info = ""
# Build analysis prompt
web_summary = "\n\n".join([
f"[{i+1}] {r['title']}\n{r['url']}\n{r['content'][:300]}..."
for i, r in enumerate(web_results[:5])
])
prompt = f"""You are a Librarian helping build a personal knowledge base and extended memory system.
Analyze these web search results for information worth documenting in our personal wiki.
Query: "{query}"
Keywords: {', '.join(keywords) if keywords else 'none'}
Web Results:
{web_summary}
This is a PERSONAL knowledge base using Schema.org-aligned taxonomy that captures:
- People: Family members, friends, colleagues, public figures (Schema.org: Person)
- Companies: Businesses, organizations, institutions (Schema.org: Organization)
- Places: Locations, restaurants, travel destinations (Schema.org: Place)
- Entertainment: Books, movies, TV, music, games (Schema.org: CreativeWork)
- Recipes: Food, cooking techniques, ingredients (Schema.org: CreativeWork/Recipe)
- Products: Purchased items, gear, tools, equipment (Schema.org: Product)
- Technology: Software, applications, infrastructure (Schema.org: SoftwareApplication)
- Health: Medical info, fitness, wellness (Schema.org: MedicalEntity)
- Events: Concerts, travel, appointments, important dates (Schema.org: Event)
- Hobbies: Personal interests, activities, pastimes (Custom extension)
- Projects: Work projects, personal projects (Schema.org: Project)
- Reference: General knowledge, how-tos (Custom extension)
ANALYSIS STEPS:
1. Read each web result carefully for substantive, factual content
2. Identify genuinely novel information not likely already known
3. Match topics to appropriate taxonomy categories
4. Generate valid paths following the exact format below
RULES:
- Do NOT suggest pages for topics with insufficient information in results
- Do NOT invent entities not explicitly mentioned in results
- Do NOT suggest paths that don't match the taxonomy exactly
- Do NOT suggest generic or vague page topics
- Be CONSERVATIVE - fewer high-quality suggestions is better than many low-quality ones
- ONLY suggest documentation for substantive, specific information
**CRITICAL: Use ONLY these Schema.org-aligned path prefixes (case-sensitive):**
- People: `people/<name>` (Schema.org: Person)
- Companies: `companies/<company-name>` (Schema.org: Organization)
- Places: `places/<location>` (Schema.org: Place)
- Entertainment (Schema.org: CreativeWork):
- Books: `entertainment/books/<title>`
- Movies: `entertainment/movies/<title>`
- TV: `entertainment/tv/<title>`
- Music: `entertainment/music/<artist-or-album>`
- Games: `entertainment/games/<title>`
- Recipes: `recipes/<cuisine-or-category>/<dish>` (Schema.org: Recipe)
- Products: `products/<category>/<product-name>` (Schema.org: Product)
- Technology: `technology/<category>/<topic>` (Schema.org: SoftwareApplication)
- Health: `health/<category>/<topic>` (Schema.org: MedicalEntity)
- Events: `events/<event-type>/<event-name>` (Schema.org: Event)
- Hobbies: `hobbies/<hobby-name>` (Custom extension)
- Projects: `projects/<project-name>` (Schema.org: Project)
- Reference: `reference/<category>/<topic>` (Custom extension)
**Path Rules:**
- Use lowercase with hyphens (kebab-case): "machine-learning" not "Machine_Learning"
- Keep paths 2-3 levels deep maximum
- Be consistent with existing paths when possible
{existing_paths_info}
Return ONLY valid JSON:
{{
"has_novel_info": true,
"new_pages": [
{{"title": "Page Title", "path": "companies/example-company", "summary": "What information to include"}}
],
"update_pages": [
{{"title": "Existing Page", "new_facts": ["fact 1"], "source_url": "url"}}
],
"new_entities": [
{{"name": "Entity Name", "type": "person/place/thing/concept/recipe/media", "description": "Brief description"}}
]
}}
JSON:"""
try:
# Call Ollama for analysis (temperature=0.0 for consistent classification)
response = await self.ollama.generate_text(
prompt=prompt,
model=self.settings.ollama_model,
stream=False,
temperature=0.0
)
if not response:
logger.warning("Empty response from Ollama")
return None
# Extract JSON from response
response_clean = response.strip()
if '{' in response_clean:
json_start = response_clean.find('{')
json_end = response_clean.rfind('}') + 1
response_clean = response_clean[json_start:json_end]
analysis = json.loads(response_clean)
return analysis
except json.JSONDecodeError as e:
logger.error(f"Failed to parse Ollama response as JSON: {e}")
logger.debug(f"Response was: {response[:500]}")
return None
except Exception as e:
logger.error(f"Analysis failed: {e}", exc_info=True)
return None
def _format_taxonomy_for_prompt(self, taxonomy: Dict[str, List[str]]) -> str:
"""
Format taxonomy structure for inclusion in LLM prompt.
Args:
taxonomy: Dict mapping categories to subcategories
Returns:
Formatted string showing existing paths
"""
if not taxonomy:
return ""
lines = ["**Existing paths in your wiki (PREFER these over creating new ones):**"]
for category, subcategories in taxonomy.items():
if subcategories:
lines.append(f"- {category}/")
for sub in subcategories:
lines.append(f" - {category}/{sub}/")
else:
lines.append(f"- {category}/")
lines.append("")
lines.append("**IMPORTANT:** If a suitable existing path exists, use it instead of creating a new category.")
lines.append("Example: NATO should go in `reference/political-entities/` not a new `reference/military-alliances/`")
return "\n".join(lines)
async def _mark_search_processed(self, search_id: str):
"""Mark SearchQuery node as processed."""
query = """
MATCH (sq:SearchQuery {id: $search_id})
SET sq.processed = true,
sq.processed_at = datetime()
RETURN sq.id
"""
try:
await self.neo4j.execute_query(query, {"search_id": search_id})
logger.debug(f"Marked search {search_id} as processed")
except Exception as e:
logger.error(f"Failed to mark search as processed: {e}")
async def _apply_bidirectional_entity_linking(
self,
page_id: int,
page_title: str,
user: str
) -> Dict[str, int]:
"""
Apply bidirectional entity linking after page creation/update.
This runs AFTER ingestion so entities are extracted and in the graph.
Steps:
1. Link entities in the new page (forward links to existing entities)
2. Find pages that mention the new entity (reverse references)
3. Link entities in those pages (backward links to the new entity)
Args:
page_id: Wiki page ID
page_title: Page title (used to find reverse references)
user: User identifier
Returns:
Dict with link counts: {
"forward_links": int, # Links added to the new page
"backward_links": int, # Links added to other pages pointing to new page
"pages_updated": int # Number of other pages updated
}
"""
from src.core.multi_tenancy import get_neo4j_user_base_label
forward_links = 0
backward_links = 0
pages_updated = 0
try:
# Import here to avoid circular dependency
from src.routers.entity_linking import link_entities_in_page, EntityLinkingRequest
from src.core.dependencies import get_wiki_service, get_graph_service
wiki_service = get_wiki_service()
graph_service = get_graph_service()
# STEP 1: Forward linking - link entities in the new page
logger.info(f"Step 1/3: Linking entities in page {page_id} ('{page_title}')")
try:
forward_result = await link_entities_in_page(
request=EntityLinkingRequest(
user=user,
page_id=page_id,
create_relationships=True,
re_index_if_changed=False # Already indexed, no need to re-index
),
wiki_service=wiki_service,
graph_service=graph_service,
ingestion_service=self.ingestion_service,
api_key="" # Internal call, no auth needed
)
forward_links = forward_result.content_links_added
logger.info(f"Added {forward_links} forward links in page {page_id}")
except Exception as e:
logger.error(f"Failed to add forward links: {e}")
# STEP 2: Find reverse references - which pages mention this new entity?
logger.info(f"Step 2/3: Finding pages that mention '{page_title}'")
user_base_label = get_neo4j_user_base_label(user)
# Query to find documents that mention entities with this page's title
reverse_query = f"""
// Find entities with the same name as the page title
MATCH (e:{user_base_label})
WHERE toLower(e.name) = toLower($title)
AND NOT e:Document
// Find documents that mention those entities
MATCH (d:Document)-[r:MENTIONS]->(e)
WHERE d.page_id <> $page_id // Exclude the page itself
RETURN DISTINCT d.page_id as page_id, d.title as title
LIMIT 50
"""
try:
reverse_refs = await self.neo4j.execute_query(
reverse_query,
{"title": page_title, "page_id": page_id}
)
logger.info(f"Found {len(reverse_refs)} pages that mention '{page_title}'")
except Exception as e:
logger.error(f"Failed to find reverse references: {e}")
reverse_refs = []
# STEP 3: Backward linking - add links in those pages to the new entity
if reverse_refs:
logger.info(f"Step 3/3: Adding backward links in {len(reverse_refs)} pages")
for ref in reverse_refs:
try:
backward_result = await link_entities_in_page(
request=EntityLinkingRequest(
user=user,
page_id=ref['page_id'],
create_relationships=False, # Relationships already exist
re_index_if_changed=False # Don't re-index for link updates
),
wiki_service=wiki_service,
graph_service=graph_service,
ingestion_service=self.ingestion_service,
api_key=""
)
if backward_result.content_links_added > 0:
backward_links += backward_result.content_links_added
pages_updated += 1
logger.info(
f"Added {backward_result.content_links_added} links "
f"in page {ref['page_id']} ('{ref['title']}')"
)
except Exception as e:
logger.error(f"Failed to add backward links in page {ref['page_id']}: {e}")
else:
logger.info("Step 3/3: No reverse references found, skipping backward linking")
return {
"forward_links": forward_links,
"backward_links": backward_links,
"pages_updated": pages_updated
}
except Exception as e:
logger.error(f"Bidirectional entity linking failed: {e}", exc_info=True)
return {
"forward_links": 0,
"backward_links": 0,
"pages_updated": 0
}
async def _create_or_consolidate_page(
self,
user: str,
title: str,
path: str,
summary: str,
source_query: str,
web_results: List[Dict[str, Any]]
):
"""
Create wiki page or consolidate with existing synonym page.
Uses WikiPageWriter for intelligent LLM-based content generation:
- For new pages: Holistic structured content creation
- For existing pages: Zero-loss reconstruction with conflict detection
"""
# Normalize path to user namespace
if not path.startswith(f"users/{user}"):
path = f"users/{user}/{path.lstrip('/')}"
# Format web results as source information
source_information = [
{
'title': r['title'],
'url': r['url'],
'content': r['content']
}
for r in web_results[:5] # Top 5 web results
]
# Search for existing pages with similar titles (synonym consolidation)
existing_pages = await self.wiki.search_pages(title, path_prefix=f"users/{user}")
if existing_pages:
# Page exists - reconstruct with new information using LLM
logger.info(f"Found existing page for '{title}', will reconstruct with new info")
page_id = existing_pages[0]['id']
# Get current content
existing_page = await self.wiki.get_page(page_id)
if existing_page:
# Build new information text from summary and web results
new_information = f"{summary}\n\n"
for r in web_results[:3]:
new_information += f"- {r['title']}: {r['content'][:200]}...\n"
# Use WikiPageWriter to reconstruct with LLM
reconstructed_content, conflicts = await self.wiki_page_writer.reconstruct_page(
title=title,
existing_content=existing_page['content'],
new_information=new_information,
new_sources=source_information,
detect_conflicts=True
)
if conflicts:
logger.warning(
f"Detected {len(conflicts)} conflicts when updating '{title}' - "
"LLM chose most authoritative sources"
)
await self.wiki.update_page(
page_id=page_id,
content=reconstructed_content
)
logger.info(f"Reconstructed existing page: {title}")
# Trigger ingestion to update vectors and graph
if self.ingestion_service:
try:
await self.ingestion_service.ingest_page(
page_id=page_id,
user=user,
force_refresh=True
)
logger.info(f"Ingested updated page {page_id} into knowledge base")
# Apply bidirectional entity linking after ingestion
link_stats = await self._apply_bidirectional_entity_linking(
page_id=page_id,
page_title=title,
user=user
)
logger.info(
f"Entity linking complete: {link_stats['forward_links']} forward links, "
f"{link_stats['backward_links']} backward links "
f"({link_stats['pages_updated']} pages updated)"
)
except Exception as e:
logger.error(f"Failed to ingest updated page {page_id}: {e}")
return
# Create new page with LLM-generated structured content
logger.info(f"Creating new page: {title}")
# Use WikiPageWriter to create structured content
content = await self.wiki_page_writer.create_page(
title=title,
topic_summary=summary,
source_information=source_information,
entities=None, # Could extract from keywords if available
related_docs=None
)
# Extract tags from path for dossier organization
path_parts = path.split('/')
tags = [part for part in path_parts if part and part not in ['users', user]]
created_page = await self.wiki.create_page(
path=path,
title=title,
content=content,
description=f"Consolidated from search: {source_query}",
tags=tags[:3], # Limit to 3 tags
is_published=True
)
page_id = created_page.get("id") if created_page else None
logger.info(f"Created new page: {path} (page_id: {page_id})")
# Trigger ingestion to update vectors and graph
if self.ingestion_service and page_id:
try:
await self.ingestion_service.ingest_page(
page_id=page_id,
user=user,
force_refresh=False # New page, no need to force
)
logger.info(f"Ingested new page {page_id} into knowledge base")
# Apply bidirectional entity linking after ingestion
link_stats = await self._apply_bidirectional_entity_linking(
page_id=page_id,
page_title=title,
user=user
)
logger.info(
f"Entity linking complete: {link_stats['forward_links']} forward links, "
f"{link_stats['backward_links']} backward links "
f"({link_stats['pages_updated']} pages updated)"
)
except Exception as e:
logger.error(f"Failed to ingest new page {page_id}: {e}")
async def _update_page_with_facts(
self,
title: str,
new_facts: List[str],
source_url: str,
user: str
):
"""
Update existing page with new facts using LLM reconstruction.
Uses WikiPageWriter to intelligently merge facts with zero loss.
"""
# Search for page
pages = await self.wiki.search_pages(title, path_prefix=f"users/{user}")
if not pages:
logger.warning(f"Page '{title}' not found for update")
return
page_id = pages[0]['id']
existing_page = await self.wiki.get_page(page_id)
if not existing_page:
return
# Build new information from facts
new_information = "\n".join([f"- {fact}" for fact in new_facts])
# Format source
source_information = [{
'title': source_url,
'url': source_url,
'content': new_information
}]
# Use WikiPageWriter to reconstruct with LLM
reconstructed_content, conflicts = await self.wiki_page_writer.reconstruct_page(
title=title,
existing_content=existing_page['content'],
new_information=new_information,
new_sources=source_information,
detect_conflicts=True
)
if conflicts:
logger.warning(
f"Detected {len(conflicts)} conflicts when updating '{title}' with new facts"
)
await self.wiki.update_page(
page_id=page_id,
content=reconstructed_content
)
# Trigger ingestion to update vectors and graph
logger.debug(f"ingestion_service available: {self.ingestion_service is not None}")
if self.ingestion_service:
try:
logger.info(f"Starting ingestion for updated page {page_id}")
await self.ingestion_service.ingest_page(
page_id=page_id,
user=user,
force_refresh=True
)
logger.info(f"Ingested updated page {page_id} into knowledge base")
# Apply bidirectional entity linking after ingestion
link_stats = await self._apply_bidirectional_entity_linking(
page_id=page_id,
page_title=title,
user=user
)
logger.info(
f"Entity linking complete: {link_stats['forward_links']} forward links, "
f"{link_stats['backward_links']} backward links "
f"({link_stats['pages_updated']} pages updated)"
)
except Exception as e:
logger.error(f"Failed to ingest updated page {page_id}: {e}")
async def _add_entity_to_graph(
self,
user: str,
entity_name: str,
entity_type: str,
description: str,
source_search_id: str
):
"""Add new entity to knowledge graph."""
from src.core.multi_tenancy import get_neo4j_user_base_label
user_base_label = get_neo4j_user_base_label(user)
# Create entity node with appropriate type label
type_label = entity_type.capitalize() if entity_type else "Entity"
query = f"""
MERGE (e:{user_base_label}:{type_label} {{name: $name}})
ON CREATE SET
e.description = $description,
e.created_at = datetime(),
e.source = 'librarian_consolidation',
e.source_search_id = $search_id
ON MATCH SET
e.updated_at = datetime()
RETURN e
"""
try:
await self.neo4j.execute_query(query, {
"name": entity_name,
"description": description,
"search_id": source_search_id
})
logger.debug(f"Added entity to graph: {entity_name} ({entity_type})")
except Exception as e:
logger.error(f"Failed to add entity to graph: {e}")