diff --git a/services/search/core.py b/services/search/core.py index dc97bfb61..d6fce2846 100644 --- a/services/search/core.py +++ b/services/search/core.py @@ -3,11 +3,19 @@ import json import logging import re +import time import xml.etree.ElementTree as ET from concurrent.futures import ThreadPoolExecutor, as_completed +from contextlib import contextmanager +from contextvars import ContextVar from datetime import datetime, timedelta from typing import Dict, Any, Optional, List, Set from urllib.parse import urlparse +from src.constants import ( + ARXIV_API_URL, + OPENALEX_API_URL, + SCHOLARLY_LOOKUP_TOTAL_BUDGET, +) from src.search_passages import search_excerpt import httpx @@ -489,22 +497,80 @@ def _result_strongly_matches_title(title: str, result: dict) -> bool: return overlap >= (1.0 if len(wanted) == 2 else 0.8) +def _scholarly_user_agent() -> str: + """Identify this build to the scholarly APIs using the real app version.""" + from src.constants import APP_VERSION + + return f"Odysseus/{APP_VERSION} scholarly-title-resolver" + + +_scholarly_deadline: ContextVar[Optional[float]] = ContextVar( + "scholarly_deadline", default=None +) + + +@contextmanager +def _scholarly_budget(): + """Open one wall-clock budget shared by every hop of a lookup chain.""" + token = _scholarly_deadline.set( + time.monotonic() + SCHOLARLY_LOOKUP_TOTAL_BUDGET + ) + try: + yield + finally: + _scholarly_deadline.reset(token) + + +def _scholarly_api_get(url: str, params: dict) -> Optional[httpx.Response]: + """GET a scholarly metadata API under the shared outbound policy. + + Returns ``None`` when the URL fails the outbound check or the caller's + budget is already spent, so callers degrade to their next source instead + of raising. ``follow_redirects`` stays on because both APIs redirect to + canonical paths, which is exactly why the destination needs checking. + """ + from src.constants import SCHOLARLY_LOOKUP_TIMEOUT + from src.url_safety import check_outbound_url + + ok, reason = check_outbound_url(url, block_private=True) + if not ok: + logger.warning("Scholarly lookup blocked for %s: %s", url, reason) + return None + + timeout = SCHOLARLY_LOOKUP_TIMEOUT + deadline = _scholarly_deadline.get() + if deadline is not None: + remaining = deadline - time.monotonic() + if remaining <= 0: + logger.info("Scholarly lookup budget exhausted before %s", url) + return None + timeout = min(timeout, remaining) + + response = httpx.get( + url, + params=params, + headers={"User-Agent": _scholarly_user_agent()}, + timeout=timeout, + follow_redirects=True, + ) + response.raise_for_status() + return response + + def _arxiv_title_results(title: str, count: int = 3) -> list[dict]: """Resolve a paper title through arXiv's public Atom API.""" try: - response = httpx.get( - "https://export.arxiv.org/api/query", - params={ + response = _scholarly_api_get( + ARXIV_API_URL, + { "search_query": f'ti:"{title}"', "start": 0, "max_results": max(1, min(int(count), 5)), }, - headers={"User-Agent": "Odysseus/0.20 scholarly-title-resolver"}, - timeout=12.0, - follow_redirects=True, ) - response.raise_for_status() + if response is None: + return [] root = ET.fromstring(response.text) except Exception as exc: logger.info("arXiv title lookup failed for %r: %s", title, exc) @@ -541,20 +607,18 @@ def _openalex_title_results(title: str, count: int = 3) -> list[dict]: # OpenAlex treats a literal question mark as query syntax and returns # HTTP 400 for otherwise valid titles such as "How Far ... GPT-4V?". search_title = re.sub(r"[?]+", " ", str(title or "")).strip() - response = httpx.get( - "https://api.openalex.org/works", - params={ + response = _scholarly_api_get( + OPENALEX_API_URL, + { "search": search_title, "per-page": max(1, min(int(count), 5)), "select": ( "display_name,doi,primary_location,publication_year,type" ), }, - headers={"User-Agent": "Odysseus/0.20 scholarly-title-resolver"}, - timeout=12.0, - follow_redirects=True, ) - response.raise_for_status() + if response is None: + return [] payload = response.json() except Exception as exc: logger.info("OpenAlex title lookup failed for %r: %s", title, exc) @@ -597,8 +661,16 @@ def _openalex_title_results(title: str, count: int = 3) -> list[dict]: def _scholarly_title_results(title: str, count: int = 3) -> list[dict]: - """Retry a noisy scholarly query as a bare title, then use arXiv API.""" + """Retry a noisy scholarly query as a bare title, then use arXiv API. + The three hops share one wall-clock budget so a slow upstream cannot hold a + user-facing search open for the sum of every per-request timeout. + """ + with _scholarly_budget(): + return _scholarly_title_results_inner(title, count) + + +def _scholarly_title_results_inner(title: str, count: int) -> list[dict]: try: simplified = searxng_search_api(title, count=max(3, count)) except Exception as exc: @@ -621,10 +693,11 @@ def _direct_scholarly_title_results(title: str, count: int = 3) -> list[dict]: # OpenAlex typically resolves titles in under a second and often returns # the official arXiv landing page. The arXiv API remains the fallback. - openalex = _openalex_title_results(title, count) - if openalex: - return openalex - return _arxiv_title_results(title, count) + with _scholarly_budget(): + openalex = _openalex_title_results(title, count) + if openalex: + return openalex + return _arxiv_title_results(title, count) def _augment_scholarly_results(query: str, results: list[dict], count: int) -> list[dict]: diff --git a/src/constants.py b/src/constants.py index 2b4919d3d..d52249e25 100644 --- a/src/constants.py +++ b/src/constants.py @@ -140,6 +140,18 @@ LLM_HOSTS = [h.strip() for h in os.getenv("LLM_HOSTS", "").split(",") if h.strip OPENAI_API_KEY = os.getenv("OPENAI_API_KEY") SEARXNG_INSTANCE = os.getenv("SEARXNG_INSTANCE", "http://localhost:8080") +# Scholarly title resolution. These are the only third-party metadata APIs the +# search path calls directly, so they get named constants rather than literals +# repeated at each call site. The budget bounds the whole SearXNG -> OpenAlex -> +# arXiv chain: each hop used to get its own full timeout, so one scholarly query +# could stall a user-facing search for the sum of all three. +ARXIV_API_URL = "https://export.arxiv.org/api/query" +OPENALEX_API_URL = "https://api.openalex.org/works" +SCHOLARLY_LOOKUP_TIMEOUT = float(os.getenv("ODYSSEUS_SCHOLARLY_LOOKUP_TIMEOUT", "12")) +SCHOLARLY_LOOKUP_TOTAL_BUDGET = float( + os.getenv("ODYSSEUS_SCHOLARLY_LOOKUP_BUDGET", "20") +) + # Cleanup configuration CLEANUP_ENABLED = os.getenv("CLEANUP_ENABLED", "True").lower() == "true" CLEANUP_INTERVAL_HOURS = int(os.getenv("CLEANUP_INTERVAL_HOURS", "24"))