feat: migrate web search from tatlock_core to Librarian

Move web search functionality to The Librarian agent, integrating with
the library-desk /rag/search endpoint for enhanced search capabilities.

Changes:
- Add search_web, read_url, read_urls_batch tools to Librarian
- Add WebSearchResult, ContentExtractionResult models to client
- Add search_web, extract_content, extract_content_batch client methods
- Update Librarian capability with web/url/internet domains
- Remove search_web from tatlock_core tools and toolset
- Update Tatlock system prompt to delegate web search to Librarian
- Add comprehensive unit tests for new Librarian tools
- Clean up legacy src/agents/tools.py

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
This commit is contained in:
2025-12-15 18:30:16 +01:00
co-authored by Claude Opus 4.5
parent 3e432d662e
commit 100ebeae52
13 changed files with 1287 additions and 425 deletions
+7 -3
View File
@@ -21,10 +21,11 @@ LIBRARIAN_CAPABILITY = HouseholdCapability(
role="The Librarian",
category="research",
description=(
"Research and wiki management: can CREATE wiki pages about topics "
"Research, web search, and wiki management: can SEARCH the web for current "
"information, READ URLs/articles, CREATE wiki pages about topics "
"(with automatic HybridRAG research), UPDATE existing pages, "
"SEARCH wiki/knowledge graph/web, and synthesize information. "
"Use for: 'create a page about X', 'update wiki', 'find info on X'"
"and synthesize information from multiple sources. "
"Use for: 'search for X', 'what is X', 'create a page about X', 'read this URL'"
),
domains=[
"research",
@@ -33,6 +34,9 @@ LIBRARIAN_CAPABILITY = HouseholdCapability(
"wiki",
"documents",
"search",
"web",
"url",
"internet",
"synthesis",
"create",
"write",
+221
View File
@@ -98,6 +98,47 @@ class ResearchSummary(BaseModel):
timing_ms: int = 0
class WebSearchResult(BaseModel):
"""Result from web search via /rag/search."""
title: str
url: str
content: str = "" # Full extracted text via Trafilatura
snippet: str = "" # Original search engine snippet
source: str = "" # Domain name
published_date: Optional[str] = None
class WebSearchResponse(BaseModel):
"""Response from /rag/search endpoint."""
query: str
search_type: str
results: list[WebSearchResult] = Field(default_factory=list)
total_results: int = 0
search_time_ms: int = 0
sources_summary: str = "" # Pre-formatted markdown citations
class ContentExtractionResult(BaseModel):
"""Result from content extraction."""
url: str
title: Optional[str] = None
content: str = ""
author: Optional[str] = None
date: Optional[str] = None
language: Optional[str] = None
success: bool = True
error: Optional[str] = None
class BatchExtractionResponse(BaseModel):
"""Response from batch content extraction."""
results: list[ContentExtractionResult] = Field(default_factory=list)
total_urls: int = 0
successful: int = 0
failed: int = 0
extraction_time_ms: int = 0
class EntityLinking(BaseModel):
"""Entity linking results from smart-create."""
forward_links: int = 0
@@ -685,6 +726,186 @@ class LibraryDeskClient:
logger.warning("library_desk_health_check_failed", error=str(e))
return False
# ========================================================================
# RAG Search (Web Search with Content Extraction)
# ========================================================================
async def search_web(
self,
query: str,
user: str | None = None,
search_type: str = "web",
limit: int = 10,
) -> WebSearchResponse:
"""
Search the web and extract content from results.
Uses SearXNG for search and Trafilatura for content extraction.
Returns both snippets and full extracted text.
Args:
query: Search query (1-500 chars)
user: User identifier for tracking
search_type: "web", "news", or "images"
limit: Number of results (1-20)
Returns:
WebSearchResponse with results and pre-formatted sources
"""
user = user or get_user()
client = self._ensure_client()
payload = {
"query": query,
"search_type": search_type,
"limit": limit,
"user": user or "tatlock-librarian",
}
logger.info("library_desk_web_search", query=query, limit=limit)
response = await client.post("/rag/search", json=payload, timeout=30.0)
response.raise_for_status()
data = response.json()
results = [
WebSearchResult(
title=r.get("title", ""),
url=r.get("url", ""),
content=r.get("content", ""),
snippet=r.get("snippet", ""),
source=r.get("source", ""),
published_date=r.get("published_date"),
)
for r in data.get("results", [])
]
return WebSearchResponse(
query=data.get("query", query),
search_type=data.get("search_type", search_type),
results=results,
total_results=data.get("total_results", len(results)),
search_time_ms=data.get("search_time_ms", 0),
sources_summary=data.get("sources_summary", ""),
)
# ========================================================================
# Content Extraction
# ========================================================================
async def extract_content(
self,
url: str,
include_metadata: bool = True,
max_length: int = 5000,
) -> ContentExtractionResult:
"""
Extract main content from a URL.
Uses Trafilatura for intelligent content extraction,
removing boilerplate, ads, and navigation.
Note: Uses soft failure pattern - check result.success field.
Args:
url: URL to extract content from
include_metadata: Whether to extract author, date, etc.
max_length: Maximum content length
Returns:
ContentExtractionResult (check .success and .error fields)
"""
client = self._ensure_client()
payload = {
"url": url,
"include_metadata": include_metadata,
"max_length": max_length,
}
logger.debug("library_desk_extract_content", url=url)
response = await client.post("/content/extract", json=payload, timeout=30.0)
response.raise_for_status()
data = response.json()
result = data.get("result", {})
return ContentExtractionResult(
url=result.get("url", url),
title=result.get("title"),
content=result.get("content", ""),
author=result.get("author"),
date=result.get("date"),
language=result.get("language"),
success=result.get("success", False),
error=result.get("error"),
)
async def extract_content_batch(
self,
urls: list[str],
include_metadata: bool = True,
max_length: int = 2000,
) -> BatchExtractionResponse:
"""
Extract content from multiple URLs in parallel.
More efficient than sequential calls. Max 20 URLs per batch.
Note: Uses soft failure pattern - individual failures don't
throw errors, check each result's .success field.
Args:
urls: List of URLs to extract (max 20)
include_metadata: Whether to extract author, date, etc.
max_length: Maximum content length per URL
Returns:
BatchExtractionResponse with results and stats
"""
client = self._ensure_client()
payload = {
"urls": urls[:20], # Server limit
"include_metadata": include_metadata,
"max_length": max_length,
}
logger.info("library_desk_extract_batch", url_count=len(urls))
response = await client.post(
"/content/extract/batch",
json=payload,
timeout=60.0, # Longer timeout for batch
)
response.raise_for_status()
data = response.json()
results = [
ContentExtractionResult(
url=r.get("url", ""),
title=r.get("title"),
content=r.get("content", ""),
author=r.get("author"),
date=r.get("date"),
language=r.get("language"),
success=r.get("success", False),
error=r.get("error"),
)
for r in data.get("results", [])
]
return BatchExtractionResponse(
results=results,
total_urls=data.get("total_urls", len(urls)),
successful=data.get("successful", 0),
failed=data.get("failed", 0),
extraction_time_ms=data.get("extraction_time_ms", 0),
)
# Global client factory
async def get_library_client() -> LibraryDeskClient:
+238 -1
View File
@@ -432,6 +432,239 @@ async def find_related_entities(
return f"Error finding related entities: {str(e)}"
# ============================================================================
# Web Search & Content Extraction
# ============================================================================
async def search_web(
query: str,
limit: int = 10,
search_type: str = "web",
) -> str:
"""
Search the web and extract content from results.
This is the primary tool for finding current information online.
Results include both snippets and full extracted text from pages.
Search types:
- "web": General web search (default)
- "news": News articles
- "images": Image search
Args:
query: Search query (1-500 chars)
limit: Number of results (1-20, default: 10)
search_type: Type of search ("web", "news", or "images")
Returns:
Formatted search results with sources and extracted content
Examples:
search_web("Python 3.12 new features")
search_web("latest tech news", search_type="news", limit=5)
"""
try:
async with LibraryDeskClient() as client:
response = await client.search_web(
query=query,
limit=limit,
search_type=search_type,
)
if not response.results:
return f"No results found for '{query}'"
output_parts = [f"## Web Search: {query}\n"]
output_parts.append(f"*Found {response.total_results} results in {response.search_time_ms}ms*\n")
for i, result in enumerate(response.results, 1):
output_parts.append(f"### {i}. {result.title}")
output_parts.append(f"**Source:** {result.source}")
output_parts.append(f"**URL:** {result.url}")
if result.published_date:
output_parts.append(f"**Date:** {result.published_date}")
# Use full content if available, otherwise snippet
content = result.content or result.snippet
if content:
# Truncate for readability
if len(content) > 500:
content = content[:500] + "..."
output_parts.append(f"\n{content}")
output_parts.append("")
# Add pre-formatted sources for citations
if response.sources_summary:
output_parts.append("---")
output_parts.append(response.sources_summary)
logger.info(
"librarian_web_search",
query=query,
result_count=response.total_results,
search_type=search_type,
)
return "\n".join(output_parts)
except Exception as e:
logger.error("librarian_web_search_error", error=str(e), query=query)
return f"Error searching web: {str(e)}"
async def read_url(
url: str,
max_length: int = 5000,
) -> str:
"""
Read and extract the main content from a URL.
Use this when you have a specific URL to read, such as:
- A link the user provided
- A URL from search results you want to read in full
- Documentation or article pages
Extracts the main content, removing ads, navigation, and boilerplate.
Args:
url: The URL to read
max_length: Maximum content length (default: 5000)
Returns:
Extracted page content with metadata
Examples:
read_url("https://docs.python.org/3/library/asyncio.html")
read_url("https://example.com/article", max_length=10000)
"""
try:
async with LibraryDeskClient() as client:
result = await client.extract_content(
url=url,
include_metadata=True,
max_length=max_length,
)
if not result.success:
return f"Could not read page: {result.error or 'Unknown error'}"
output_parts = []
# Header with metadata
if result.title:
output_parts.append(f"# {result.title}")
else:
output_parts.append(f"# Content from {url}")
output_parts.append(f"**URL:** {url}")
if result.author:
output_parts.append(f"**Author:** {result.author}")
if result.date:
output_parts.append(f"**Date:** {result.date}")
if result.language and result.language != "en":
output_parts.append(f"**Language:** {result.language}")
output_parts.append("")
# Main content
if result.content:
output_parts.append(result.content)
else:
output_parts.append("(No content could be extracted)")
logger.info(
"librarian_read_url",
url=url,
content_length=len(result.content) if result.content else 0,
)
return "\n".join(output_parts)
except Exception as e:
logger.error("librarian_read_url_error", error=str(e), url=url)
return f"Error reading URL: {str(e)}"
async def read_urls_batch(
urls: list[str],
max_length: int = 2000,
) -> str:
"""
Read and extract content from multiple URLs in parallel.
More efficient than calling read_url multiple times.
Max 20 URLs per batch.
Note: Individual failures don't fail the entire batch -
failed URLs are reported but other content is still returned.
Args:
urls: List of URLs to read (max 20)
max_length: Maximum content length per URL (default: 2000)
Returns:
Extracted content from all successful URLs with failure report
Examples:
read_urls_batch(["https://example.com/1", "https://example.com/2"])
"""
try:
async with LibraryDeskClient() as client:
response = await client.extract_content_batch(
urls=urls,
include_metadata=True,
max_length=max_length,
)
output_parts = [
f"## Batch Content Extraction",
f"*Extracted {response.successful}/{response.total_urls} URLs in {response.extraction_time_ms}ms*\n",
]
# Show successful extractions
for result in response.results:
if result.success:
title = result.title or result.url
output_parts.append(f"### {title}")
output_parts.append(f"**URL:** {result.url}")
if result.content:
# Truncate for readability in batch mode
content = result.content
if len(content) > max_length:
content = content[:max_length] + "..."
output_parts.append(f"\n{content}")
output_parts.append("")
# Report failures
failed = [r for r in response.results if not r.success]
if failed:
output_parts.append("---")
output_parts.append("### Failed Extractions")
for result in failed:
output_parts.append(f"- {result.url}: {result.error}")
logger.info(
"librarian_read_urls_batch",
total=response.total_urls,
successful=response.successful,
failed=response.failed,
)
return "\n".join(output_parts)
except Exception as e:
logger.error("librarian_read_urls_batch_error", error=str(e))
return f"Error reading URLs: {str(e)}"
# ============================================================================
# Wiki Write Operations
# ============================================================================
@@ -685,7 +918,7 @@ async def smart_create_wiki_page(
# All tools available to The Librarian
LIBRARIAN_TOOLS = [
# Research tools
# Research tools (internal knowledge)
hybrid_search,
search_wiki,
get_wiki_page,
@@ -694,6 +927,10 @@ LIBRARIAN_TOOLS = [
semantic_search,
explore_knowledge_graph,
find_related_entities,
# Web search & content extraction
search_web,
read_url,
read_urls_batch,
# Write tools
create_wiki_page,
update_wiki_page,
+7 -29
View File
@@ -17,7 +17,6 @@ from src.agents.tatlock_core.tools import (
get_current_datetime,
calculate_time_offset,
time_difference,
search_web,
)
from src.core.config import config
from src.core.logging_config import get_logger
@@ -76,18 +75,17 @@ You have direct access to several permanent tools that you should USE whenever a
- time_difference: Calculate the time between two dates
- Use these for ANY date/time queries - never guess at dates or times
3. **Web Search** (search_web): Search for current, volatile, or factual information
- Use this for ANY information that might be current, factual, or outside your training data
3. **Web Search** (via Librarian): For current, volatile, or factual information
- Delegate to the Librarian for web searches and research
- Examples: news, current events, recent developments, specific facts, technical documentation
- Always prefer searching over guessing or using potentially outdated knowledge
- For extensive research questions, note that this will later be delegated to the librarian
- Use: delegate_to_librarian(task="search the web for ...")
## Tool Usage Guidelines
- **Mathematics**: ALWAYS use the calculator tool, even for simple arithmetic
- **Dates/Times**: ALWAYS use the date/time tools, never guess or estimate
- **Current Information**: ALWAYS search for facts, news, or volatile information
- **Verification**: When facts are important, use search to verify rather than rely on memory alone
- **Current Information**: Delegate web searches to the Librarian
- **Verification**: When facts are important, delegate to Librarian for research
- When you use a tool, explain what you're doing in a butler-appropriate manner
- Present tool results naturally in your response
@@ -238,28 +236,8 @@ class TatlockAgent(AgentInterface):
ctx.deps.log_call(f"🕐 Calculating time difference between {date1_str} and {date2_str}")
return time_difference(date1_str, date2_str)
# Web search tool
@self._agent.tool
async def web_search(ctx: RunContext[ToolCallTracker], query: str, num_results: int = 5) -> str:
"""
Search the web using SearXNG for current information.
Use this tool for ANY information that might be:
- Current or time-sensitive (news, events, recent developments)
- Factual and verifiable (statistics, technical specs, definitions)
- Outside your training data or knowledge cutoff
Args:
query: Search query string
num_results: Number of results to return (default: 5, max: 10)
Returns:
Formatted search results with titles, URLs, and snippets
"""
# Log the search query to reasoning output
if ctx.deps:
ctx.deps.log_call(f"🔍 Searching for: '{query}'")
return await search_web(query, num_results)
# NOTE: Web search has been moved to The Librarian agent.
# Use delegate_to_librarian(task="search web for ...") for web search.
@property
def agent(self):
+2 -3
View File
@@ -1,7 +1,8 @@
"""
Tatlock's core tools package.
Provides calculator, date/time, and web search capabilities.
Provides calculator and date/time capabilities.
Web search has been moved to The Librarian agent.
Organized as a household member with toolset and capability registration.
"""
from .capability import TATLOCK_CORE_CAPABILITY, get_capability
@@ -10,7 +11,6 @@ from .tools import (
calculate,
calculate_time_offset,
get_current_datetime,
search_web,
time_difference,
)
@@ -20,7 +20,6 @@ __all__ = [
"get_current_datetime",
"calculate_time_offset",
"time_difference",
"search_web",
# Toolset
"tatlock_core_tools",
"get_core_tools",
+3 -3
View File
@@ -11,10 +11,10 @@ TATLOCK_CORE_CAPABILITY = HouseholdCapability(
name="tatlock_core",
role="Butler's Core Tools",
category="core",
description="Essential tools for computation, date/time operations, and web searches",
domains=["computation", "datetime", "information", "research"],
description="Essential tools for computation and date/time operations",
domains=["computation", "datetime", "math", "calculator"],
cost="low",
requires_network=True, # For web search
requires_network=False, # Web search moved to Librarian
)
+2 -93
View File
@@ -256,96 +256,5 @@ def time_difference(date1_str: str, date2_str: str = "now") -> str:
return f"Error calculating time difference: {str(e)}"
# ============================================================================
# SearXNG Search Tool
# ============================================================================
async def search_web(query: str, num_results: int = 5) -> str:
"""
Search the web using SearXNG.
Args:
query: Search query string
num_results: Number of results to return (default: 5, max: 10)
Returns:
Formatted search results as a string with titles, URLs, and snippets
Examples:
search_web("Python async programming") -> "1. Title: ...\n URL: ...\n ..."
"""
try:
# Limit results
num_results = min(num_results, 10)
# Get SearXNG host with fallback logic
searxng_host = str(config.SEARXNG_HOST)
# Try production host first, fall back to localhost in development
hosts_to_try = [searxng_host]
if config.ENVIRONMENT.value == "development" and "localhost" not in searxng_host:
# Add localhost fallback for development
hosts_to_try.append("http://localhost:8087")
last_error = None
for host in hosts_to_try:
try:
logger.debug("searxng_search_attempt", host=host, query=query)
async with httpx.AsyncClient(timeout=config.SEARXNG_TIMEOUT) as client:
response = await client.get(
f"{host}/search",
params={
"q": query,
"format": "json",
"pageno": 1,
}
)
if response.status_code == 200:
data = response.json()
results = data.get("results", [])
if not results:
return f"No results found for '{query}'"
# Format results
formatted_results = []
for i, result in enumerate(results[:num_results], 1):
title = result.get("title", "No title")
url = result.get("url", "")
content = result.get("content", "No description available")
formatted_results.append(
f"{i}. {title}\n"
f" URL: {url}\n"
f" {content}\n"
)
logger.info(
"searxng_search_success",
host=host,
query=query,
result_count=len(results),
)
return "\n".join(formatted_results)
else:
last_error = f"SearXNG returned status {response.status_code}"
except httpx.ConnectError:
last_error = f"Cannot connect to SearXNG at {host}"
logger.warning("searxng_connection_failed", host=host)
continue
except Exception as e:
last_error = str(e)
logger.warning("searxng_error", host=host, error=str(e))
continue
# All hosts failed
logger.error("searxng_all_hosts_failed", error=last_error)
return f"Error searching: {last_error}. Please check that SearXNG is running."
except Exception as e:
logger.error("searxng_unexpected_error", error=str(e), exc_info=True)
return f"Error searching: {str(e)}"
# NOTE: Web search has been moved to The Librarian agent.
# Use delegate_to_librarian(task="search web for ...") for web search.
+2 -12
View File
@@ -55,17 +55,8 @@ time_difference_tool = Tool(
),
)
web_search_tool = Tool(
function=tools.search_web,
name="search_web",
description=(
"Search the web using SearXNG for current information. "
"Use this to find recent events, current data, or verify facts. "
"Returns formatted results with titles, URLs, and snippets. "
"Useful for information that may have changed since training data."
),
takes_ctx=False,
)
# NOTE: Web search has been moved to The Librarian agent.
# Use delegate_to_librarian(task="search web for ...") for web search.
# Combined toolset of all core tools
@@ -74,7 +65,6 @@ tatlock_core_tools = [
current_datetime_tool,
time_offset_tool,
time_difference_tool,
web_search_tool,
]
+3 -97
View File
@@ -4,20 +4,14 @@ Tatlock's permanent tools.
These tools are always available to the butler agent:
- Calculator: For all mathematical operations
- Date/Time toolkit: For current time and time calculations
- SearXNG search: For searching the web for current information
Note: Web search has been moved to The Librarian agent.
See src/agents/librarian/tools.py for search_web functionality.
"""
import logging
import math
import re
from datetime import datetime, timedelta
from typing import Any
import httpx
from src.core.config import config
logger = logging.getLogger(__name__)
# ============================================================================
@@ -256,91 +250,3 @@ def time_difference(date1_str: str, date2_str: str = "now") -> str:
except Exception as e:
return f"Error calculating time difference: {str(e)}"
# ============================================================================
# SearXNG Search Tool
# ============================================================================
async def search_web(query: str, num_results: int = 5) -> str:
"""
Search the web using SearXNG.
Args:
query: Search query string
num_results: Number of results to return (default: 5, max: 10)
Returns:
Formatted search results as a string with titles, URLs, and snippets
Examples:
search_web("Python async programming") -> "1. Title: ...\n URL: ...\n ..."
"""
try:
# Limit results
num_results = min(num_results, 10)
# Get SearXNG host with fallback logic
searxng_host = str(config.SEARXNG_HOST)
# Try production host first, fall back to localhost in development
hosts_to_try = [searxng_host]
if config.ENVIRONMENT.value == "development" and "localhost" not in searxng_host:
# Add localhost fallback for development
hosts_to_try.append("http://localhost:8087")
last_error = None
for host in hosts_to_try:
try:
logger.info(f"Attempting SearXNG search at {host}")
async with httpx.AsyncClient(timeout=config.SEARXNG_TIMEOUT) as client:
response = await client.get(
f"{host}/search",
params={
"q": query,
"format": "json",
"pageno": 1,
}
)
if response.status_code == 200:
data = response.json()
results = data.get("results", [])
if not results:
return f"No results found for '{query}'"
# Format results
formatted_results = []
for i, result in enumerate(results[:num_results], 1):
title = result.get("title", "No title")
url = result.get("url", "")
content = result.get("content", "No description available")
formatted_results.append(
f"{i}. {title}\n"
f" URL: {url}\n"
f" {content}\n"
)
return "\n".join(formatted_results)
else:
last_error = f"SearXNG returned status {response.status_code}"
except httpx.ConnectError:
last_error = f"Cannot connect to SearXNG at {host}"
logger.warning(f"SearXNG connection failed at {host}, trying next host if available")
continue
except Exception as e:
last_error = str(e)
logger.warning(f"SearXNG error at {host}: {e}")
continue
# All hosts failed
return f"Error searching: {last_error}. Please check that SearXNG is running."
except Exception as e:
logger.error(f"Unexpected error in search_web: {e}", exc_info=True)
return f"Error searching: {str(e)}"