""" The Librarian - Expert agent for research and knowledge management. A PydanticAI agent that provides research assistance through the library-desk API, offering: - HybridRAG search across all knowledge sources - Wiki and document management - Semantic search and knowledge graph exploration """ from typing import Any, Optional from pydantic_ai import Agent from src.agents.librarian.tools import ( create_wiki_page, explore_knowledge_graph, find_related_entities, get_dossier_pages, get_wiki_page, hybrid_search, list_dossiers, read_url, read_urls_batch, search_web, search_wiki, semantic_search, smart_create_wiki_page, update_wiki_page, ) from src.core.config import config from src.core.logging_config import get_logger logger = get_logger(__name__) # Librarian system prompt LIBRARIAN_SYSTEM_PROMPT = """You are The Librarian, an expert research assistant in the Tatlock household. Your role is to help users find, understand, synthesize, and manage information from: - The personal wiki (Wiki.js) containing documentation and notes - The knowledge graph (Neo4j) with entities and relationships - Vector embeddings (Qdrant) for semantic search - Web search (SearXNG) for current information ## Your Personality - Scholarly and thorough in your research - Cite your sources and provide context - Organize information clearly - Suggest related topics when relevant - Acknowledge limitations when information is incomplete ## Your Tools ### Web Search & Content Extraction - **search_web**: Search the internet for current information (weather, news, facts) - Use for: weather forecasts, current events, recent developments, external facts - Returns extracted content from search results, not just snippets - **read_url**: Read and extract content from a specific URL - Use when: user provides a URL or you need to read a specific webpage - **read_urls_batch**: Read multiple URLs in parallel (up to 20) - Use for: comparing multiple sources, gathering info from several pages ### Internal Research Tools - **hybrid_search**: Your primary research tool - searches wiki, graph, and web at once - **search_wiki**: Find specific wiki pages by keyword - **semantic_search**: Find conceptually similar content - **explore_knowledge_graph** / **find_related_entities**: Discover connections - **list_dossiers** / **get_dossier_pages**: Browse knowledge collections ### Wiki Reading Tools - **get_wiki_page**: Read full content of a wiki page by ID - ALWAYS use this to fetch and read page content when summarizing - Use after search_wiki to get the full text of a specific page ### Wiki Writing Tools - **smart_create_wiki_page**: Create a page with automatic research (PREFERRED) - **This is the DEFAULT choice when user asks to create a wiki page about a topic** - When user says "Create a page about X" or "Add X to the wiki" without providing specific content, ALWAYS use this tool - Automatically researches the topic from wiki, graph, and web - Synthesizes content with proper source attribution - Creates bidirectional links in knowledge graph - **create_wiki_page**: Create a page with user-provided content - ONLY use when user provides specific text/content they want added verbatim - For simple notes, reminders, or quick additions with exact content - **update_wiki_page**: Update an existing page (partial updates) - Use when: "Update the page about X", "Fix this info", "Add to dossier" - First search_wiki to find the page, then get_wiki_page to read it - Only specify fields you want to change ## Research Approach 1. Start with hybrid_search for broad queries 2. Use search_wiki for specific document lookups 3. **ALWAYS use get_wiki_page to fetch full content** before summarizing a page 4. Use semantic_search when looking for conceptually similar content 5. Explore the knowledge graph to find connections between concepts 6. Synthesize and summarize findings clearly ## Writing Approach When asked to create or update wiki content: 1. **"Create a page about X" (no specific content provided)**: Use smart_create_wiki_page - This is the PREFERRED tool for topic-based page creation - It researches first and creates comprehensive, well-sourced content 2. **User provides exact text to add**: Use create_wiki_page with their content 3. **Updating existing pages**: - Search for the page with search_wiki - Fetch full content with get_wiki_page - Make edits and use update_wiki_page 4. **Organizing into dossiers**: Use update_wiki_page with just the tags field ## Response Format Your responses are returned to Tatlock (the butler) who will synthesize them into a final answer for the user. Keep this in mind: - Lead with the key findings or confirmation of action - Include relevant sources and citations - When summarizing wiki pages, fetch and read them first - Note any gaps in available information - Be concise but thorough - Tatlock will format the final response - Structure your findings clearly so they can be easily integrated with other responses """ # Lazy initialization to avoid connection issues during imports _librarian_agent: Optional[Agent[None, str]] = None def _create_librarian_agent() -> Agent[None, str]: """Create the Librarian PydanticAI agent.""" from pydantic_ai.models.openai import OpenAIChatModel from src.ollama.provider import get_ollama_provider # Create Ollama model with sanitized provider # (fixes 'content: null' issue with tool calls) model = OpenAIChatModel( model_name=config.OLLAMA_DEFAULT_MODEL, provider=get_ollama_provider(), ) agent: Agent[None, str] = Agent( model=model, system_prompt=LIBRARIAN_SYSTEM_PROMPT, retries=2, ) # Register research tools (internal knowledge) agent.tool_plain(hybrid_search) agent.tool_plain(search_wiki) agent.tool_plain(semantic_search) agent.tool_plain(list_dossiers) agent.tool_plain(get_dossier_pages) agent.tool_plain(explore_knowledge_graph) agent.tool_plain(find_related_entities) # Register web search & content extraction tools agent.tool_plain(search_web) agent.tool_plain(read_url) agent.tool_plain(read_urls_batch) # Register wiki read tools agent.tool_plain(get_wiki_page) # Register wiki write tools agent.tool_plain(create_wiki_page) agent.tool_plain(update_wiki_page) agent.tool_plain(smart_create_wiki_page) logger.info( "librarian_agent_created", model=config.OLLAMA_DEFAULT_MODEL, tool_count=14, # 7 research + 3 web + 1 wiki read + 3 wiki write ) return agent def get_librarian_agent() -> Agent[None, str]: """ Get the Librarian agent instance (lazy initialization). Returns: PydanticAI Agent configured for research tasks """ global _librarian_agent if _librarian_agent is None: _librarian_agent = _create_librarian_agent() return _librarian_agent async def run_librarian( task: str, context: str = "", message_history: Optional[list[Any]] = None, ) -> str: """ Execute a research task with The Librarian. This is the main entry point for delegating research tasks to The Librarian from Tatlock or other agents. Args: task: The research task or question context: Additional context from conversation message_history: Optional conversation history Returns: Research results and findings Example: result = await run_librarian( task="Find information about Docker networking", context="User is setting up a homelab", ) """ agent = get_librarian_agent() # Build prompt with context if provided prompt = task if context: prompt = f"Context: {context}\n\nTask: {task}" logger.info( "librarian_task_started", task=task[:100], has_context=bool(context), has_history=bool(message_history), ) try: result = await agent.run( prompt, message_history=message_history, ) logger.info( "librarian_task_completed", task=task[:50], output_length=len(result.output), ) return result.output except Exception as e: logger.error( "librarian_task_error", task=task[:50], error=str(e), exc_info=True, ) return f"The Librarian encountered an error: {str(e)}" async def run_librarian_stream( task: str, context: str = "", message_history: Optional[list[Any]] = None, ): """ Execute a research task with streaming output. Yields text deltas as The Librarian generates the response. Args: task: The research task or question context: Additional context from conversation message_history: Optional conversation history Yields: str: Text deltas from the response Example: async for delta in run_librarian_stream("Find Docker docs"): print(delta, end="", flush=True) """ agent = get_librarian_agent() # Build prompt with context if provided prompt = task if context: prompt = f"Context: {context}\n\nTask: {task}" logger.info( "librarian_stream_started", task=task[:100], ) try: async with agent.run_stream( prompt, message_history=message_history, ) as response: async for delta in response.stream_text(delta=True): yield delta logger.info("librarian_stream_completed", task=task[:50]) except Exception as e: logger.error( "librarian_stream_error", task=task[:50], error=str(e), exc_info=True, ) yield f"\n\nThe Librarian encountered an error: {str(e)}"