merge: reconcile PR 40 with current lab

Integrate lab fff55a78 into PR #40 (cc25d5ba). Lab's modular email
backend/frontend, modular settings, split stylesheets (static/style.css
stays deleted), procfs compatibility, and request-scoped TurnContract
authority win; PR #40's routing classifiers, editor/email/task features,
and style.css changes are ported into lab's module and stylesheet homes.

Integration fixes:
- settings/api.js imports ui.js under its canonical versioned URL
- browser observations keep legacy CAPTCHA/access-block evidence
- artifact turns do not re-trigger broad-web research recovery
- env reference documents PR test-tool variables; page regenerated

PR #40 defects surfaced by lab gates and fixed here:
- web_fetch generic schema drops top-level anyOf (OpenAI contract);
  the compact preview contract still requires url or urls
- get_weather registered as a brokered network read
- new lazy editor modules precached for offline use
- SearXNG pin mirrored into GPU standalone compose files
- image model picker again skips offline endpoints

Tests updated where PR #40 changed behaviour on purpose, and PR tests
moved onto lab's document_source helpers.
This commit is contained in:
Alexandre Teixeira
2026-10-01 05:03:58 +01:00
337 changed files with 90287 additions and 75556 deletions
+118 -19
View File
@@ -3,11 +3,19 @@
import json
import logging
import re
import time
import xml.etree.ElementTree as ET
from concurrent.futures import ThreadPoolExecutor, as_completed
from contextlib import contextmanager
from contextvars import ContextVar
from datetime import datetime, timedelta
from typing import Dict, Any, Optional, List, Set
from urllib.parse import urlparse
from src.constants import (
ARXIV_API_URL,
OPENALEX_API_URL,
SCHOLARLY_LOOKUP_TOTAL_BUDGET,
)
from src.search_passages import search_excerpt
import httpx
@@ -489,22 +497,106 @@ def _result_strongly_matches_title(title: str, result: dict) -> bool:
return overlap >= (1.0 if len(wanted) == 2 else 0.8)
def _scholarly_user_agent() -> str:
"""Identify this build to the scholarly APIs using the real app version."""
from src.constants import APP_VERSION
return f"Odysseus/{APP_VERSION} scholarly-title-resolver"
_scholarly_deadline: ContextVar[Optional[float]] = ContextVar(
"scholarly_deadline", default=None
)
@contextmanager
def _scholarly_budget():
"""Open one wall-clock budget shared by every hop of a lookup chain."""
token = _scholarly_deadline.set(
time.monotonic() + SCHOLARLY_LOOKUP_TOTAL_BUDGET
)
try:
yield
finally:
_scholarly_deadline.reset(token)
MAX_SCHOLARLY_REDIRECTS = 3
def _scholarly_api_get(url: str, params: dict) -> Optional[httpx.Response]:
"""GET a scholarly metadata API under the shared outbound policy.
Returns ``None`` when any destination URL fails the outbound check or the
caller's budget is already spent, so callers degrade to their next source
instead of raising. Bounded manual redirects ensure every hop passes
through ``check_outbound_url`` before the destination is contacted.
"""
from src.constants import SCHOLARLY_LOOKUP_TIMEOUT
from src.url_safety import check_outbound_url
current_url = url
current_params: Optional[dict] = params
for _ in range(MAX_SCHOLARLY_REDIRECTS + 1):
ok, reason = check_outbound_url(current_url, block_private=True)
if not ok:
logger.warning("Scholarly lookup blocked for %s: %s", current_url, reason)
return None
timeout = SCHOLARLY_LOOKUP_TIMEOUT
deadline = _scholarly_deadline.get()
if deadline is not None:
remaining = deadline - time.monotonic()
if remaining <= 0:
logger.info("Scholarly lookup budget exhausted before %s", current_url)
return None
timeout = min(timeout, remaining)
response = httpx.get(
current_url,
params=current_params,
headers={"User-Agent": _scholarly_user_agent()},
timeout=timeout,
follow_redirects=False,
)
is_redirect = getattr(response, "is_redirect", False) or (
getattr(response, "status_code", None) in (301, 302, 303, 307, 308)
)
if is_redirect:
headers = getattr(response, "headers", {})
location = headers.get("location")
if not location:
logger.warning(
"Scholarly redirect missing Location header from %s", current_url
)
return None
current_url = str(httpx.URL(str(response.url)).join(location))
current_params = None
continue
response.raise_for_status()
return response
logger.warning("Scholarly lookup exceeded max redirects from %s", url)
return None
def _arxiv_title_results(title: str, count: int = 3) -> list[dict]:
"""Resolve a paper title through arXiv's public Atom API."""
try:
response = httpx.get(
"https://export.arxiv.org/api/query",
params={
response = _scholarly_api_get(
ARXIV_API_URL,
{
"search_query": f'ti:"{title}"',
"start": 0,
"max_results": max(1, min(int(count), 5)),
},
headers={"User-Agent": "Odysseus/0.20 scholarly-title-resolver"},
timeout=12.0,
follow_redirects=True,
)
response.raise_for_status()
if response is None:
return []
root = ET.fromstring(response.text)
except Exception as exc:
logger.info("arXiv title lookup failed for %r: %s", title, exc)
@@ -541,20 +633,18 @@ def _openalex_title_results(title: str, count: int = 3) -> list[dict]:
# OpenAlex treats a literal question mark as query syntax and returns
# HTTP 400 for otherwise valid titles such as "How Far ... GPT-4V?".
search_title = re.sub(r"[?]+", " ", str(title or "")).strip()
response = httpx.get(
"https://api.openalex.org/works",
params={
response = _scholarly_api_get(
OPENALEX_API_URL,
{
"search": search_title,
"per-page": max(1, min(int(count), 5)),
"select": (
"display_name,doi,primary_location,publication_year,type"
),
},
headers={"User-Agent": "Odysseus/0.20 scholarly-title-resolver"},
timeout=12.0,
follow_redirects=True,
)
response.raise_for_status()
if response is None:
return []
payload = response.json()
except Exception as exc:
logger.info("OpenAlex title lookup failed for %r: %s", title, exc)
@@ -597,8 +687,16 @@ def _openalex_title_results(title: str, count: int = 3) -> list[dict]:
def _scholarly_title_results(title: str, count: int = 3) -> list[dict]:
"""Retry a noisy scholarly query as a bare title, then use arXiv API."""
"""Retry a noisy scholarly query as a bare title, then use arXiv API.
The three hops share one wall-clock budget so a slow upstream cannot hold a
user-facing search open for the sum of every per-request timeout.
"""
with _scholarly_budget():
return _scholarly_title_results_inner(title, count)
def _scholarly_title_results_inner(title: str, count: int) -> list[dict]:
try:
simplified = searxng_search_api(title, count=max(3, count))
except Exception as exc:
@@ -621,10 +719,11 @@ def _direct_scholarly_title_results(title: str, count: int = 3) -> list[dict]:
# OpenAlex typically resolves titles in under a second and often returns
# the official arXiv landing page. The arXiv API remains the fallback.
openalex = _openalex_title_results(title, count)
if openalex:
return openalex
return _arxiv_title_results(title, count)
with _scholarly_budget():
openalex = _openalex_title_results(title, count)
if openalex:
return openalex
return _arxiv_title_results(title, count)
def _augment_scholarly_results(query: str, results: list[dict], count: int) -> list[dict]: