merge: reconcile PR 40 with current lab

Integrate lab fff55a78 into PR #40 (cc25d5ba). Lab's modular email
backend/frontend, modular settings, split stylesheets (static/style.css
stays deleted), procfs compatibility, and request-scoped TurnContract
authority win; PR #40's routing classifiers, editor/email/task features,
and style.css changes are ported into lab's module and stylesheet homes.

Integration fixes:
- settings/api.js imports ui.js under its canonical versioned URL
- browser observations keep legacy CAPTCHA/access-block evidence
- artifact turns do not re-trigger broad-web research recovery
- env reference documents PR test-tool variables; page regenerated

PR #40 defects surfaced by lab gates and fixed here:
- web_fetch generic schema drops top-level anyOf (OpenAI contract);
  the compact preview contract still requires url or urls
- get_weather registered as a brokered network read
- new lazy editor modules precached for offline use
- SearXNG pin mirrored into GPU standalone compose files
- image model picker again skips offline endpoints

Tests updated where PR #40 changed behaviour on purpose, and PR tests
moved onto lab's document_source helpers.
This commit is contained in:
Alexandre Teixeira
2026-10-01 05:03:58 +01:00
337 changed files with 90287 additions and 75556 deletions
+32 -8
View File
@@ -22473,7 +22473,7 @@ async def stream_agent_loop(
"manage_notes", "manage_calendar", "manage_tasks",
"ask_user", "update_plan",
}
elif _ody_doc_finetune_mode and route_tools is not None:
elif (_ody_doc_finetune_mode or doc_mode) and route_tools is not None:
if _prompt_active_document is not None:
route_tools = {
"edit_document", "update_document", "suggest_document",
@@ -22481,12 +22481,12 @@ async def stream_agent_loop(
}
else:
route_tools = {"create_document", "ask_user", "update_plan"}
elif _ody_notes_finetune_mode and route_tools is not None:
elif (_ody_notes_finetune_mode or notes_mode) and route_tools is not None:
route_tools = {
"manage_notes", "manage_calendar", "manage_tasks",
"ask_user", "update_plan",
}
elif _ody_general_no_tool_mode:
elif _ody_general_no_tool_mode or general_no_tool_mode:
route_tools = set()
else:
route_tools = _route_tui_local_workspace_tools(
@@ -22984,6 +22984,8 @@ async def stream_agent_loop(
# navigation tools. Do not let the general agent floor re-add bash
# after that narrow surface was selected.
and not (_low_signal_turn and workspace)
and not _ody_notes_finetune_mode
and not _ody_general_no_tool_mode
):
from src.turn_contract import CONTRACT_CORE_TOOLS
_core_agent_tools = set(CONTRACT_CORE_TOOLS)
@@ -23231,6 +23233,13 @@ async def stream_agent_loop(
_base_relevant_tools = set(_relevant_tools)
logger.info("[agent-intent] explicit plan request clamped to plan tools")
if _low_signal_turn and not workspace and not _terminal_agent_mode and _relevant_tools is not None:
# Retrieval and the core floor can surface file readers for a vague
# local-project hint even though no project has been selected.
_relevant_tools.difference_update(_DOMAIN_TOOL_MAP["files"])
if _base_relevant_tools is not None:
_base_relevant_tools.difference_update(_DOMAIN_TOOL_MAP["files"])
if _relevant_tools is not None:
logger.info("[agent-intent] selected_tools=%s", sorted(_relevant_tools)[:50])
@@ -24225,6 +24234,7 @@ async def stream_agent_loop(
_failed_read_recovery_sent = False
_failed_read_recovery_instruction_sent = False
_post_effectful_mutation_done = False
_verified_coding_summary_emitted = False
_successful_mutation_signatures: set[tuple[str, str]] = set()
_single_execution_bound = _request_forbids_execution_retry(_last_user)
_execution_tool_attempts: dict[str, int] = {}
@@ -25874,9 +25884,17 @@ async def stream_agent_loop(
and not _approved_result_injected
and not _native_terminal_runtime
and not normalized_external_tool_schemas
# A one-tool shortcut cannot own a causal compound workflow. Let
# the agent consume the complete request-scoped tool surface.
and len(_caller_relevant_tools or ()) <= 1
# The explicit topic-bulk path below owns its search-then-bulk
# sequence. Other multi-tool requests need the agent's full route.
and (
len(_caller_relevant_tools or ()) <= 1
or (
_caller_relevant_tools == {
"mcp__email__search_emails", "mcp__email__bulk_email",
}
and _parse_qwen_explicit_email_topic_bulk_action_request(_last_user)
)
)
and not _request_has_compound_actions(_last_user)
# Sealed safe reads use the central required-operation path so
# execution and canonical rendering have the same owner.
@@ -33949,6 +33967,11 @@ async def stream_agent_loop(
_tui_bash_block_completed
and block.tool_type == "host_shell"
)
and not (
block.tool_type == "host_shell"
and _has_tui_host_bridge
and _post_effectful_mutation_done
)
):
_terminal_summary = _ody_qwen_terminal_tool_summary({
"tool": block.tool_type,
@@ -35383,10 +35406,11 @@ async def stream_agent_loop(
_post_effectful_mutation_done
and _post_edit_verification_completed
and _workspace_mutation_completion_authorized
and _deterministic_terminal_eligible
and (_deterministic_terminal_eligible or _tui_local_execution_turn)
):
if _tui_local_execution_turn or _qwen38_tool_router:
full_response = _tui_verified_coding_summary(tool_events)
_verified_coding_summary_emitted = True
yield f'data: {json.dumps({"type": "final_response", "content": full_response})}\n\n'
elif not full_response.strip() or full_response.strip().startswith("```"):
_verification_output = ""
@@ -36970,7 +36994,7 @@ async def stream_agent_loop(
_response_before_tool_summary = full_response
_action_summary_selected = False
if tool_events and _deterministic_terminal_eligible:
if tool_events and _deterministic_terminal_eligible and not _verified_coding_summary_emitted:
_multi_read_email_summaries = _email_read_summaries_from_tool_events(tool_events)
_multi_attachment_summaries = _email_attachment_summaries_from_tool_events(tool_events)
_bulk_email_state_summary = _email_state_bulk_terminal_summary(tool_events, user_text=_last_user)
+48 -1
View File
@@ -231,6 +231,7 @@ def _wrap_workspace_namespace(
cwd: str,
*,
chdir: str = "/workspace",
interpreter_prefix: str | None = None,
) -> str | None:
"""Run a shell command with the active workspace mounted at /workspace.
@@ -254,8 +255,53 @@ def _wrap_workspace_namespace(
"--dir", "/tmp", "--tmpfs", "/tmp",
"--dev-bind", "/dev", "/dev", "--proc", "/proc",
"--dir", "/workspace", "--bind", cwd, "/workspace",
"--chdir", chdir, "/bin/bash", "-lc", content,
]
# setup-python installs interpreters under /opt, and local CI virtualenvs
# can live under /tmp. Those paths are hidden by the private root/tmpfs.
# Expose only the active interpreter environment, read-only, so Python
# tools keep their installed packages without exposing the host /tmp.
if interpreter_prefix:
prefix = os.path.abspath(interpreter_prefix)
resolved_prefix = os.path.realpath(prefix)
mounted_roots = ("/usr", "/home", "/mnt")
reserved_roots = {
"/", "/tmp", "/var", "/opt", "/etc", "/workspace",
"/root", "/run", "/proc", "/dev", "/sys", *mounted_roots,
}
already_visible = any(
prefix == root or prefix.startswith(root + os.sep)
for root in mounted_roots
)
# A prefix is trusted only when it names a specific interpreter tree.
# In particular, never overlay the private root, tmpfs, or workspace
# with a broad host directory. Reject symlinked prefixes too: bwrap
# would otherwise bind the resolved source at a different destination.
has_environment_layout = (
os.path.isfile(os.path.join(prefix, "pyvenv.cfg"))
or (
os.path.isfile(os.path.join(prefix, "bin", "python"))
and os.path.isdir(os.path.join(
prefix, "lib", f"python{sys.version_info.major}.{sys.version_info.minor}",
))
)
)
if (
not already_visible
and prefix == resolved_prefix
and prefix not in reserved_roots
and len(prefix.split(os.sep)) >= 3
and os.path.isdir(prefix)
and has_environment_layout
):
parents = []
parent = os.path.dirname(prefix)
while parent not in ("/", "/tmp", "/etc", "/workspace", *mounted_roots):
parents.append(parent)
parent = os.path.dirname(parent)
for directory in reversed(parents):
args.extend(("--dir", directory))
args.extend(("--ro-bind", prefix, prefix))
args.extend(("--chdir", chdir, "/bin/bash", "-lc", content))
return shlex.join(args)
@@ -940,6 +986,7 @@ class PythonTool:
python_command,
agent_cwd(),
chdir="/workspace",
interpreter_prefix=sys.prefix,
)
if needs_virtual_namespace
else None
+71 -14
View File
@@ -18,6 +18,7 @@ import urllib.request
from pathlib import Path
from typing import Dict, Any
from core import platform_compat
from src.constants import MAX_OUTPUT_CHARS
PDF_EXTRACT_MAX_BYTES = 80_000_000
@@ -119,6 +120,47 @@ def _browser_pid_file_candidates(
)
return list(dict.fromkeys(candidates))
# Linux exposes one command line per pid under /proc; macOS and Windows do not.
# Kept as a module attribute so the procfs-dependent paths stay testable on a
# host that has no procfs, and on one that does.
def _process_command_line(pid: int) -> str | None:
"""Command line of a running process, or ``None`` when it cannot be read.
``None`` means "this host cannot tell", not "the process is gone". Off
Linux there is no procfs to read a command line from, so callers must not
treat it as proof that the process exited.
"""
try:
return (platform_compat.PROC_ROOT / str(pid) / "cmdline").read_bytes().replace(
b"\0", b" "
).decode("utf-8", errors="replace")
except (OSError, UnicodeError):
return None
def _process_is_alive(pid: int) -> bool:
"""Whether a pid currently exists.
Delegates to ``core.platform_compat.pid_alive`` rather than probing with
``os.kill(pid, 0)`` directly. That probe is POSIX-only: CPython's Windows
``os.kill`` calls ``TerminateProcess(handle, sig)`` for any signal other
than CTRL_C / CTRL_BREAK, so it would *kill* the daemon it is asked about.
Windows is also where there is no procfs, which is precisely when this
function gets called at all.
``pid_alive`` reads False for a pid that ``os.kill`` reports with
``PermissionError`` — a live process owned by another user. Neither caller
here wants a different answer: the sweep only unlinks a pid file it wrote
itself, and treating somebody else's pid as "not our daemon" is the safe
reading in both.
"""
return platform_compat.pid_alive(pid)
_SCHOLARLY_METADATA_CUE_RE = re.compile(
r"\b(?:accept(?:ed|ance)?|publish(?:ed|ing|cation)?|venue|conference|"
r"journal|proceedings|doi)\b",
@@ -2366,8 +2408,14 @@ class PrivateBrowserTool:
except OSError:
return
profile_prefix = str(tmpdir / "agent-browser-chrome-")
if not platform_compat.has_procfs():
# Without procfs there is no way to match a reparented Chrome by
# its command line, and the sweep is an optimisation rather than a
# correctness requirement. Leave those trees to the daemon's own
# lifecycle instead of failing the whole shutdown path.
return
pids: list[int] = []
for entry in Path("/proc").iterdir():
for entry in platform_compat.PROC_ROOT.iterdir():
if not entry.name.isdigit():
continue
try:
@@ -2397,16 +2445,18 @@ class PrivateBrowserTool:
for pid_file in pid_files:
try:
pid = int(pid_file.read_text().strip())
command_line = (Path("/proc") / str(pid) / "cmdline").read_bytes().replace(
b"\0", b" "
).decode("utf-8", errors="replace")
except FileNotFoundError:
# The daemon may have exited between writing its pid file and
# this cleanup pass. The exact file is still ours to remove.
with contextlib.suppress(FileNotFoundError, PermissionError, OSError):
pid_file.unlink()
except (OSError, ValueError):
continue
except (OSError, UnicodeError, ValueError):
command_line = _process_command_line(pid)
if command_line is None:
# Either the daemon exited between writing its pid file and
# this pass, or this host has no procfs to ask. Only the first
# justifies forgetting the pid file. Without procfs we cannot
# confirm the process is ours, so we neither kill it nor drop
# the record that would let a later pass find it.
if not _process_is_alive(pid):
with contextlib.suppress(FileNotFoundError, PermissionError, OSError):
pid_file.unlink()
continue
if "agent-browser" in command_line:
with contextlib.suppress(ProcessLookupError, PermissionError, OSError):
@@ -2433,10 +2483,17 @@ class PrivateBrowserTool:
for pid_file in _browser_pid_file_candidates(runtime_dir, namespace, session_id):
try:
pid = int(pid_file.read_text().strip())
command_line = (Path("/proc") / str(pid) / "cmdline").read_bytes().replace(
b"\0", b" "
).decode("utf-8", errors="replace")
except (FileNotFoundError, OSError, UnicodeError, ValueError):
except (OSError, ValueError):
continue
command_line = _process_command_line(pid)
if command_line is None:
# Without procfs we can only tell that something with this pid
# is alive, not that it is agent-browser. The pid file is our
# own namespaced one, so treat a live pid as a match: answering
# "no daemon" here is what lets `close` bootstrap a fresh one
# and wait on its browser forever.
if _process_is_alive(pid):
return True
continue
if "agent-browser" in command_line:
return True
+3 -1
View File
@@ -17,7 +17,9 @@ def compact_browser_observation(value, budget=8000):
notices.append('Exit code: ' + str(item['exit_code']))
if item.get('success') is False:
notices.append('Browser command failed.')
snapshot = item.get('snapshot')
snapshot = item.get('snapshot') or item.get('text')
if item.get('title'):
notices.append('Title: ' + str(item['title']))
url = item.get('url') or item.get('origin')
if isinstance(snapshot, str) and snapshot.strip():
# Snapshot text already contains labels and refs in DOM order.
+4 -2
View File
@@ -2319,7 +2319,7 @@ def compact_schemas(schemas, *, model=None):
'Saved notes. Create a todo in ONE add call: note_type="checklist", '
'checklist_items=[{text,done:false}], title only if requested (otherwise auto-dated). '
'Keep tasks and stated times in item text, never title. No time conversion. '
'Freeform body: content. Existing note: update+id, never add. '
'Freeform body: content. add creates a new note; update with id edits an existing note, never add. '
'list supports label/archived; search by topic; view by id. Delete only on request. '
'due_date sets a reminder, not an item time.'
)
@@ -5339,6 +5339,8 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
required_artifacts = runtime_required_artifacts(
direct_user_text, client_runtime_context,
) if native_workspace_enabled else tuple()
if required_artifacts:
web_briefing_target = False
yield event({'type': 'turn_contract', **turn_contract.audit(), 'schema_mode': 'compact_contract_v5',
'native_workspace': native_workspace_enabled,
'required_artifacts': list(required_artifacts),
@@ -7186,7 +7188,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
browser_access_blocked = (
canonical(actual_tool) == 'private_browser'
and not failed
and browser_observation_access_blocked(output)
and browser_observation_access_blocked(result.get('output') or result)
)
browser_page_missing = (
canonical(actual_tool) == 'private_browser'
+10
View File
@@ -140,6 +140,16 @@ LLM_HOSTS = [h.strip() for h in os.getenv("LLM_HOSTS", "").split(",") if h.strip
OPENAI_API_KEY = os.getenv("OPENAI_API_KEY")
SEARXNG_INSTANCE = os.getenv("SEARXNG_INSTANCE", "http://localhost:8080")
# Scholarly title resolution. These are the only third-party metadata APIs the
# search path calls directly, so they get named constants rather than literals
# repeated at each call site. The budget bounds the whole SearXNG -> OpenAlex ->
# arXiv chain: each hop used to get its own full timeout, so one scholarly query
# could stall a user-facing search for the sum of all three.
ARXIV_API_URL = "https://export.arxiv.org/api/query"
OPENALEX_API_URL = "https://api.openalex.org/works"
SCHOLARLY_LOOKUP_TIMEOUT = 12.0
SCHOLARLY_LOOKUP_TOTAL_BUDGET = 20.0
# Cleanup configuration
CLEANUP_ENABLED = os.getenv("CLEANUP_ENABLED", "True").lower() == "true"
CLEANUP_INTERVAL_HOURS = int(os.getenv("CLEANUP_INTERVAL_HOURS", "24"))
+4 -1
View File
@@ -38,7 +38,10 @@ def discover_tailscale_hosts() -> List[str]:
global _hosts_cache, _hosts_cache_time
now = time.time()
if _hosts_cache and (now - _hosts_cache_time) < _HOSTS_CACHE_TTL:
# Gate on the timestamp, not the list: a successful query that found no
# eligible peers is a real answer, and testing the list's truthiness made
# that case re-run `tailscale status` (up to a 5s timeout) on every call.
if _hosts_cache_time and (now - _hosts_cache_time) < _HOSTS_CACHE_TTL:
return list(_hosts_cache)
hosts = []
+15 -4
View File
@@ -97,19 +97,30 @@ async def _cached(key: Tuple, ttl: float, fetch: Callable[[], Awaitable[Any]]) -
pending = fut
owner = True
if not owner:
return await pending
# A cancelled waiter must not cancel the shared Future for the owner
# and every other waiter.
return await asyncio.shield(pending)
try:
val = await fetch()
async with _shared_cache_lock:
_shared_cache[key] = (time.monotonic() + ttl, val)
_shared_cache_pending.pop(key, None)
pending.set_result(val)
return val
except asyncio.CancelledError:
# Cancellation is a BaseException on supported Python versions, so it
# bypasses the Exception handler below. Wake all current waiters while
# allowing a later caller to retry the fetch.
pending.cancel()
raise
except Exception as e:
async with _shared_cache_lock:
_shared_cache_pending.pop(key, None)
pending.set_exception(e)
raise
finally:
# Keep this cleanup synchronous so a second cancellation cannot
# interrupt it and leave a permanently pending Future behind. All
# access runs on the scheduler's event-loop thread.
if _shared_cache_pending.get(key) is pending:
_shared_cache_pending.pop(key, None)
def compute_next_run(schedule: str, scheduled_time: str,
+1 -1
View File
@@ -101,7 +101,7 @@ _register(
result_integrity=ResultIntegrity.WORKSPACE_UNTRUSTED,
)
_register(
{"private_browser", "web_search", "youtube_tool"},
{"get_weather", "private_browser", "web_search", "youtube_tool"},
ToolEffect.BROKERED_NETWORK_READ,
result_integrity=ResultIntegrity.EXTERNAL_UNTRUSTED,
)
-1
View File
@@ -369,7 +369,6 @@ FUNCTION_TOOL_SCHEMAS = [
"full": {"type": "boolean", "description": "Raise the download budget to the hard cap for large pages/files. Use only after a result reported partial content."},
"query": {"type": "string", "description": "Optional comma-separated terms used to select matching passages/pages from long documents or PDFs, for example 'DocVQA, ChartQA, TextVQA, Qwen2.5-VL-72B'."}
},
"anyOf": [{"required": ["url"]}, {"required": ["urls"]}],
"required": []
}
}
+12 -7
View File
@@ -17,6 +17,7 @@ import re
from typing import Any, Dict, List, Optional
from fastapi import HTTPException
from core import platform_compat
from routes._validators import validate_remote_host, validate_ssh_port
from src.tools._common import _parse_tool_args
@@ -676,17 +677,17 @@ def _scan_running_model_processes() -> List[Dict[str, Any]]:
a dict shaped like a cookbook task so the caller can merge cleanly.
"""
import os
if not os.path.isdir("/proc"):
if not platform_compat.has_procfs():
return []
proc_root = platform_compat.PROC_ROOT
out: List[Dict[str, Any]] = []
seen_keys = set()
try:
for pid_dir in os.listdir("/proc"):
for pid_dir in os.listdir(proc_root):
if not pid_dir.isdigit():
continue
try:
with open(f"/proc/{pid_dir}/cmdline", "rb") as f:
raw = f.read()
raw = (proc_root / pid_dir / "cmdline").read_bytes()
except (OSError, PermissionError):
continue
if not raw:
@@ -1124,12 +1125,16 @@ async def _cookbook_kill_session(session_id: str, *, remote_host: str = "",
import signal
tracked_cmd = str((matched.get("payload") or {}).get("_cmd") or "").strip()
matched_pids: list[int] = []
if tracked_cmd:
for pid_name in os.listdir("/proc"):
# No procfs means no way to match a survivor by its command line.
# The tmux kill above already stopped the session, so skip the
# sweep instead of failing a stop that worked.
if tracked_cmd and platform_compat.has_procfs():
proc_root = platform_compat.PROC_ROOT
for pid_name in os.listdir(proc_root):
if not pid_name.isdigit() or int(pid_name) == os.getpid():
continue
try:
raw = open(f"/proc/{pid_name}/cmdline", "rb").read()
raw = (proc_root / pid_name / "cmdline").read_bytes()
process_cmd = raw.replace(b"\x00", b" ").decode("utf-8", errors="replace").strip()
except (OSError, PermissionError):
continue
+13 -10
View File
@@ -955,6 +955,15 @@ def selected_tools_for_request(message: str) -> frozenset[str] | None:
# Content words such as "reviews", "which", "highlights", or
# "final" must not become a public-Web lookup operation.
return None
if re.fullmatch(
_REQUEST_PREFIX + r"(?:which\s+search\s+(?:backend|provider)\s+am\s+i\s+on"
r"(?:\s+right\s+now)?|what\s+(?:default\s+)?time\s+filter\s+is\s+"
r"my\s+search\s+set\s+to(?:\s+by\s+default)?|show\s+me\s+the\s+whole\s+"
r"search\s+(?:settings?\s+)?group)[?!.]*",
text,
re.I,
):
return frozenset({"manage_settings"})
web_lookup_fallback = False
if (
re.search(r"\b(?:look\s*up|search|find)\b", text, re.I)
@@ -969,6 +978,7 @@ def selected_tools_for_request(message: str) -> frozenset[str] | None:
text,
re.I,
)
and not re.search(r"\b(?:inbox|emails?|mails?|calendar|meetings?|my\s+notes?)\b", text, re.I)
):
# Current lookups need discovery before navigation. Letting the model
# begin on an arbitrary browser page can ground an answer in stale or
@@ -983,7 +993,7 @@ def selected_tools_for_request(message: str) -> frozenset[str] | None:
r"compare|pros?|cons?|opinions?|thoughts?|about)\b",
text,
re.I,
):
) and not re.search(r"\b(?:inbox|emails?|mails?|calendar|meetings?|my\s+notes?)\b", text, re.I):
# Product/service review requests are current public-web lookups even
# when the user does not say "search". Route them to web_search before
# the model sees a schema; otherwise a no-tool contract invites raw
@@ -1145,15 +1155,6 @@ def selected_tools_for_request(message: str) -> frozenset[str] | None:
re.I,
):
return frozenset({"web_search"})
if re.fullmatch(
_REQUEST_PREFIX + r"(?:which\s+search\s+(?:backend|provider)\s+am\s+i\s+on"
r"(?:\s+right\s+now)?|what\s+(?:default\s+)?time\s+filter\s+is\s+"
r"my\s+search\s+set\s+to(?:\s+by\s+default)?|show\s+me\s+the\s+whole\s+"
r"search\s+(?:settings?\s+)?group)[?!.]*",
text,
re.I,
):
return frozenset({"manage_settings"})
if re.fullmatch(
_REQUEST_PREFIX + r"(?:is\s+there\s+)?anything\s+new\s+(?:in|on|about)\s+"
r"[^?!.]{2,160}\b(?:today|this\s+(?:week|month|year)|recently)[?!.]*",
@@ -4572,6 +4573,8 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum
established_family = immediately_established_family(text, history)
if established_family and not newly_named_families:
return frozenset({established_family})
if selected_tools_for_request(raw_text) == frozenset({"manage_settings"}):
return frozenset({"cookbook_admin"})
concrete_urls = re.findall(r"\bhttps?://[^\s<>\"']+", raw_text, re.I)
workspace_media = re.search(
r"(?:file://)?/workspace/[^\s`\"']+\."
+48
View File
@@ -16,6 +16,12 @@ break the primary use case. What it *always* rejects:
For exposed multi-tenant deployments, set ``EMBEDDING_BLOCK_PRIVATE_IPS=true`` to
additionally reject all private and loopback targets (full SSRF lockdown).
On a DNS64/NAT64 network an IPv4-only host resolves to the RFC 6052 Well-Known
Prefix ``64:ff9b::/96``. Such an address is decoded to the IPv4 destination the
translator will actually contact, and that destination is then judged under the
strict policy — so the prefix reaches public IPv4 but never tunnels to loopback,
private, shared or link-local space.
"""
import ipaddress
@@ -33,6 +39,39 @@ ALLOWED_SCHEMES = ("http", "https")
# versions for other special ranges.
_SHARED_ADDRESS_SPACE_V4 = ipaddress.ip_network("100.64.0.0/10")
# RFC 6052 §2.1 Well-Known Prefix for IPv4/IPv6 address translation (NAT64).
# An address inside exactly this /96 is not a destination in its own right: the
# low 32 bits carry the IPv4 address the translator will actually contact. On a
# DNS64/NAT64 network every public IPv4-only host resolves this way, so judging
# the outer IPv6 (which CPython reports as ``is_reserved``) would reject the
# whole public internet while telling us nothing about the real target.
#
# RFC 6052 §3.1 allows the Well-Known Prefix to represent *only* globally
# routable IPv4. The embedded destination is therefore always evaluated under
# the strict policy, whatever ``block_private`` the caller passed: the prefix
# must never become a path to loopback, private, shared, link-local, multicast,
# unspecified or otherwise non-global space.
#
# Deliberately exact. Network-specific prefixes carry locally assigned meaning
# and are NOT decoded here — notably 64:ff9b:1::/48 (RFC 8215), which this /96
# membership test excludes and which stays rejected as reserved.
_NAT64_WELL_KNOWN_PREFIX_V6 = ipaddress.ip_network("64:ff9b::/96")
def _nat64_well_known_embedded_ipv4(
ip: ipaddress._BaseAddress,
) -> Optional[ipaddress.IPv4Address]:
"""Return the IPv4 target embedded in an RFC 6052 Well-Known-Prefix address.
``None`` when ``ip`` is not inside ``64:ff9b::/96``, i.e. when no IPv4
destination may be inferred from it.
"""
if not isinstance(ip, ipaddress.IPv6Address):
return None
if ip not in _NAT64_WELL_KNOWN_PREFIX_V6:
return None
return ipaddress.IPv4Address(int(ip) & 0xFFFFFFFF)
def _default_resolver(host: str) -> List[str]:
"""Resolve a hostname to the list of IP strings it maps to (A + AAAA)."""
@@ -44,6 +83,15 @@ def _classify(ip: ipaddress._BaseAddress, *, block_private: bool) -> Optional[st
# IPv4-mapped IPv6 (e.g. ::ffff:169.254.169.254) — judge the embedded v4.
if isinstance(ip, ipaddress.IPv6Address) and ip.ipv4_mapped is not None:
ip = ip.ipv4_mapped
else:
# RFC 6052 Well-Known Prefix — judge the IPv4 destination the NAT64
# translator will contact, always under the strict policy.
translated = _nat64_well_known_embedded_ipv4(ip)
if translated is not None:
reason = _classify(translated, block_private=True)
if reason:
return f"NAT64 translated destination blocked: {reason}"
return None
if ip.is_link_local:
return f"link-local address blocked (SSRF metadata risk): {ip}"
if ip.is_multicast or ip.is_reserved or ip.is_unspecified: