feat(runtime): bind browser resources to authority

This commit is contained in:
Alexandre Teixeira
2026-10-02 18:54:09 +01:00
parent b648f9ddbe
commit e175bea752
27 changed files with 2120 additions and 3153 deletions
+2 -3
View File
@@ -7507,10 +7507,9 @@ Get current conditions and a three-day forecast using Open-Meteo. Use this for w
"private_browser": """\
```private_browser
{"action": "open", "url": "https://example.com"}
{"action": "session_info"}
```
Private browser automation through Odysseus' agent-browser wrapper. Actions include open/read/snapshot/find/evaluate/click/fill/press/wait/screenshot/close/batch. For find, pass visible text in `find`. For evaluate, pass JavaScript in `script`. Use ONLY for specific pages that need JavaScript, login/session state, clicking, forms, waiting, screenshots, or rendered DOM inspection. For open-ended search use `web_search`. For ordinary URL reading use `web_fetch`.
After opening a page, call `snapshot` before interacting, then use the returned element refs such as `@e12` as `target`; target is a selector/ref, never guessed visible text. Prefer one `batch` for known consecutive steps, e.g. `[["open","https://example.com"],["snapshot"]]`. Batch commands must be non-empty.""",
Registered browser session metadata only: session_info. Page/document reads and effects are unavailable because the configured local producer cannot guarantee captured-target binding. Do not send batches, raw commands, flags, URLs or guessed page handles. Use web_search/web_fetch for supported web access.""",
"youtube_tool": """\
```youtube_tool
+40 -9
View File
@@ -14,6 +14,7 @@ from uuid import uuid4
from src.agent_runtime.resources import (
FilesystemRoot, ExternalResource, NativeBackendResource, OwnedScope,
ProcessLaunchScope, ProcessResource, BackgroundJobResource,
BrowserSessionResource, BrowserPageResource,
backend_from_dict, intersect_roots, seal_owned_scopes,
)
from src.tool_policy import ToolPolicy, build_effective_tool_policy
@@ -124,6 +125,8 @@ class RequestAuthority:
launch_scopes: tuple[ProcessLaunchScope, ...] | None = None
process_resources: tuple[ProcessResource, ...] = ()
job_resources: tuple[BackgroundJobResource, ...] | None = None
browser_sessions: tuple[BrowserSessionResource, ...] | None = None
browser_pages: tuple[BrowserPageResource, ...] | None = None
def __post_init__(self):
if (not isinstance(self.request_id, str) or not self.request_id
@@ -178,11 +181,24 @@ class RequestAuthority:
raise ValueError("Job resource thread changed")
if any(r.thread_id != (self.session_id or "request:" + self.request_id) for r in self.process_resources):
raise ValueError("Process resource thread changed")
from src.browser_identity import seal_browser_resources
sessions, pages = seal_browser_resources(self) if self.browser_sessions is None or self.browser_pages is None else ((), ())
if self.browser_sessions is None:
object.__setattr__(self, "browser_sessions", sessions)
if self.browser_pages is None:
object.__setattr__(self, "browser_pages", pages)
for values, kind in ((self.browser_sessions, BrowserSessionResource), (self.browser_pages, BrowserPageResource)):
if not isinstance(values, tuple) or any(not isinstance(r, kind) for r in values):
raise ValueError("Malformed browser resource scope")
for r in values:
session = r.session if isinstance(r, BrowserPageResource) else r
if (session.owner, session.thread_id) != (self.owner, self.session_id):
raise ValueError("Browser owner/thread binding changed")
@classmethod
def empty(cls, *, owner=None, session_id=None, workspace=None):
return cls(uuid4().hex, _owner(owner), str(session_id or ""), str(workspace or ""),
resource_roots=(), backend_resources=(), owned_scopes=(), launch_scopes=(), job_resources=())
resource_roots=(), backend_resources=(), owned_scopes=(), launch_scopes=(), job_resources=(), browser_sessions=(), browser_pages=())
def bound_to(self, *, owner=None, session_id=None, workspace=None):
return (self.owner == _owner(owner) and self.session_id == str(session_id or "")
@@ -211,6 +227,7 @@ class RequestAuthority:
backends = ()
owned = ()
launches = processes = jobs = ()
browser_sessions = browser_pages = ()
if (self.owner, self.session_id, self.workspace) == (child.owner, child.session_id, child.workspace):
theirs = {g.tool: g for g in child.grants}
grants = [g.intersect(theirs[g.tool]) for g in self.grants if g.tool in theirs]
@@ -222,11 +239,15 @@ class RequestAuthority:
launches = intersect_launch_scopes(self.launch_scopes, child.launch_scopes)
processes = intersect_observed(self.process_resources, child.process_resources, lambda r: r.validate())
jobs = intersect_observed(self.job_resources, child.job_resources, validate_job)
from src.browser_identity import intersect_browser
browser_sessions, browser_pages = intersect_browser(self.browser_sessions, self.browser_pages,
child.browser_sessions, child.browser_pages)
return replace(self, grants=tuple(grants), denied=self.denied | child.denied,
block_all=self.block_all or child.block_all,
disable_mcp=self.disable_mcp or child.disable_mcp, inherited=True,
resource_roots=roots, backend_resources=backends, owned_scopes=owned,
launch_scopes=launches, process_resources=processes, job_resources=jobs)
launch_scopes=launches, process_resources=processes, job_resources=jobs,
browser_sessions=browser_sessions, browser_pages=browser_pages)
def continuation(self, *, owner=None, session_id=None):
"""A server continuation may rebind a session, never change owner/grants."""
@@ -236,10 +257,12 @@ class RequestAuthority:
return replace(self, session_id=rebound, inherited=True,
owned_scopes=tuple(replace(s, thread_id=rebound) for s in self.owned_scopes) if rebound else (),
process_resources=tuple(r for r in self.process_resources if r.thread_id == rebound),
job_resources=tuple(r for r in self.job_resources if r.thread_id == rebound))
job_resources=tuple(r for r in self.job_resources if r.thread_id == rebound),
browser_sessions=tuple(r for r in self.browser_sessions if r.thread_id == rebound),
browser_pages=tuple(r for r in self.browser_pages if r.session.thread_id == rebound))
def to_dict(self):
return {"version": 4, "request_id": self.request_id, "owner": self.owner,
return {"version": 5, "request_id": self.request_id, "owner": self.owner,
"session_id": self.session_id, "workspace": self.workspace,
"grants": [{"tool": g.tool,
"actions": None if g.actions is None else sorted(g.actions),
@@ -251,12 +274,14 @@ class RequestAuthority:
"owned_scopes": [s.to_dict() for s in self.owned_scopes],
"launch_scopes": [s.to_dict() for s in self.launch_scopes],
"process_resources": [r.to_dict() for r in self.process_resources],
"job_resources": [r.to_dict() for r in self.job_resources]}
"job_resources": [r.to_dict() for r in self.job_resources],
"browser_sessions": [r.to_dict() for r in self.browser_sessions],
"browser_pages": [r.to_dict() for r in self.browser_pages]}
@classmethod
def from_dict(cls, value):
if (not isinstance(value, dict) or type(value.get("version")) is not int
or value["version"] not in {1, 2, 3, 4}):
or value["version"] not in {1, 2, 3, 4, 5}):
raise ValueError("Unsupported authority snapshot")
def limits(value):
if value is None:
@@ -275,6 +300,8 @@ class RequestAuthority:
raise ValueError("Malformed process resource snapshot")
if not isinstance(backends, list) or not isinstance(owned, list):
raise ValueError("Malformed request resource scope snapshot")
if value["version"] >= 5 and any(not isinstance(value.get(name), list) for name in ("browser_sessions", "browser_pages")):
raise ValueError("Malformed browser resource scope snapshot")
return cls(value["request_id"], value["owner"], value["session_id"], value["workspace"],
tuple(OperationGrant(g["tool"], limits(g["actions"]), limits(g["inputs"]))
for g in value["grants"]), limits(value["denied"]),
@@ -283,11 +310,13 @@ class RequestAuthority:
tuple(backend_from_dict(r) for r in backends), tuple(OwnedScope.from_dict(s) for s in owned),
tuple(ProcessLaunchScope.from_dict(s) for s in process_fields["launch_scopes"]),
tuple(ProcessResource.from_dict(r) for r in process_fields["process_resources"]),
tuple(BackgroundJobResource.from_dict(r) for r in process_fields["job_resources"]))
tuple(BackgroundJobResource.from_dict(r) for r in process_fields["job_resources"]),
tuple(BrowserSessionResource.from_dict(r) for r in value["browser_sessions"]) if value["version"] >= 5 else (),
tuple(BrowserPageResource.from_dict(r) for r in value["browser_pages"]) if value["version"] >= 5 else ())
_BROWSER_READ_ACTIONS = frozenset({"open", "navigate", "snapshot", "text", "read", "find",
"screenshot", "scroll", "back", "forward", "wait", "status", "close", "tabs"})
"screenshot", "scroll", "back", "forward", "wait", "status", "close", "tabs", "session_info"})
@dataclass(frozen=True)
@@ -505,7 +534,9 @@ def seal_task_authority(prompt, task_type, action, *, owner=None, parent_authori
owned_scopes=parent.owned_scopes,
launch_scopes=parent.launch_scopes,
process_resources=parent.process_resources,
job_resources=parent.job_resources))
job_resources=parent.job_resources,
browser_sessions=parent.browser_sessions,
browser_pages=parent.browser_pages))
return _json({"task_input": [prompt, task_type, action], "authority": authority.to_dict()})
+3
View File
@@ -347,8 +347,11 @@ def guard_launch_workspace(root):
They do not claim freedom from concurrent link replacement after checking.
"""
from src import bg_jobs, containment, constants
from src import browser_identity
from src.agent_runtime.resources import _control_plane_path
control = (Path(bg_jobs._STORE), Path(bg_jobs._JOBS_DIR), containment._store_path(), _LAUNCH_DIR,
Path(constants.BROWSER_RESOURCES_DIR),
browser_identity.STATE_ROOT,
Path(constants.APP_DB), Path(constants.AUTH_FILE), Path(constants.SETTINGS_FILE))
base = Path(root.path)
if any(Path(p).resolve().is_relative_to(base) for p in control):
+115 -30
View File
@@ -37,7 +37,11 @@ def _control_plane_path(path):
"SETTINGS_FILE", "SESSIONS_FILE", "USER_PREFS_FILE", "VAULT_FILE",
"SCHEDULED_EMAILS_DB", "EMAIL_CACHE_DB", "MEMORY_FILE", "INTEGRATIONS_FILE",
)}
job_dirs = {canonical_root(constants.BG_JOBS_DIR), canonical_root(constants.PROCESS_RESOURCES_DIR)}
job_dirs = {canonical_root(constants.BG_JOBS_DIR), canonical_root(constants.PROCESS_RESOURCES_DIR),
canonical_root(constants.BROWSER_RESOURCES_DIR)}
browser = sys.modules.get("src.browser_identity")
if browser is not None:
job_dirs.add(canonical_root(browser.STATE_ROOT))
processes = sys.modules.get("src.agent_runtime.process_resources")
if processes is not None:
job_dirs.add(canonical_root(processes._LAUNCH_DIR))
@@ -73,7 +77,7 @@ def _control_plane_path(path):
return True
if jobs.exists():
# Uninspectable state fails closed; hardlinks retain object identity.
protected.update(canonical_root(p) for p in jobs.iterdir())
protected.update(canonical_root(p) for p in jobs.rglob("*") if p.is_file())
protected.update(canonical_root(getattr(constants, name) + suffix)
for name in ("APP_DB", "SCHEDULED_EMAILS_DB", "EMAIL_CACHE_DB")
for suffix in ("-wal", "-shm", "-journal"))
@@ -106,6 +110,115 @@ class ResourceIdentityError(ValueError):
"""An observed execution resource has changed or cannot be resolved."""
@dataclass(frozen=True)
class BrowserSessionObservation:
producer_namespace: str
producer_version: str
platform: str
binary_sha256: str
configuration_digest: str
session_key: str
daemon: "ProcessIdentity"
browser_instance_digest: str
session_incarnation: str
def __post_init__(self):
from src.process_lifecycle import ProcessIdentity
from src.browser_identity import PRODUCER_HASHES, incarnation
if (self.producer_namespace != "native:agent-browser"
or self.producer_version != "0.35.0"
or PRODUCER_HASHES.get(self.platform) != self.binary_sha256
or not isinstance(self.daemon, ProcessIdentity)
or type(self.daemon.pid) is not int or self.daemon.pid <= 0
or (self.daemon.pgid is not None and (type(self.daemon.pgid) is not int or self.daemon.pgid <= 0))):
raise ValueError("Unsupported browser producer observation")
import re
_text(self.daemon.start_token, "daemon incarnation")
if not re.fullmatch(r"ody-[a-f0-9]{24}", self.session_key):
raise ValueError("Malformed browser session selector")
for value in (self.configuration_digest, self.browser_instance_digest, self.session_incarnation):
if not re.fullmatch(r"[a-f0-9]{64}", value):
raise ValueError("Malformed browser digest")
if incarnation(self) != self.session_incarnation:
raise ValueError("Browser incarnation digest changed")
def to_dict(self):
return {**asdict(self), "daemon": self.daemon.to_record()}
@classmethod
def from_dict(cls, value):
from src.process_lifecycle import ProcessIdentity
if not isinstance(value, dict) or set(value) != set(cls.__dataclass_fields__):
raise ValueError("Malformed browser observation snapshot")
daemon = value["daemon"]
if not isinstance(daemon, dict) or set(daemon) != {"pid", "start_token", "pgid"}:
raise ValueError("Malformed browser daemon observation")
return cls(**{**value, "daemon": ProcessIdentity(**daemon)})
@dataclass(frozen=True)
class BrowserSessionResource:
owner: str
thread_id: str
observation: BrowserSessionObservation
def __post_init__(self):
_text(self.owner, "browser owner")
_text(self.thread_id, "browser thread")
if not isinstance(self.observation, BrowserSessionObservation):
raise ValueError("Missing browser session observation")
def validate(self):
from src.browser_identity import validate_session
validate_session(self)
def to_dict(self):
return {"owner": self.owner, "thread_id": self.thread_id, "observation": self.observation.to_dict()}
@classmethod
def from_dict(cls, value):
if not isinstance(value, dict) or set(value) != {"owner", "thread_id", "observation"}:
raise ValueError("Malformed browser resource snapshot")
return cls(value["owner"], value["thread_id"], BrowserSessionObservation.from_dict(value["observation"]))
@dataclass(frozen=True)
class BrowserPageResource:
session: BrowserSessionResource
target_id: str
loader_id: str
resolved_alias: str = ""
observed_url: str = ""
scope: str = "document"
def __post_init__(self):
import re
if not isinstance(self.session, BrowserSessionResource) or not re.fullmatch(r"[A-F0-9]{32}", self.target_id):
raise ValueError("Malformed browser page identity")
if self.scope not in {"page", "document"}:
raise ValueError("Malformed browser page scope")
_text(self.loader_id, "document loader", optional=self.scope == "page")
_text(self.observed_url, "observed URL", optional=True)
if self.resolved_alias and not re.fullmatch(r"t[1-9][0-9]*", self.resolved_alias):
raise ValueError("Malformed browser alias metadata")
def authority_key(self):
return (self.session, self.target_id, self.loader_id if self.scope == "document" else None)
def validate(self):
from src.browser_identity import validate_page
validate_page(self)
def to_dict(self):
return {**asdict(self), "session": self.session.to_dict()}
@classmethod
def from_dict(cls, value):
if not isinstance(value, dict) or set(value) != set(cls.__dataclass_fields__):
raise ValueError("Malformed browser page snapshot")
return cls(**{**value, "session": BrowserSessionResource.from_dict(value["session"])})
@dataclass(frozen=True)
class FileObjectIdentity:
device: int
@@ -439,34 +552,6 @@ class BackgroundJobResource:
return cls(**{**value, "processes": tuple(ProcessResource.from_dict(p) for p in value["processes"])})
@dataclass(frozen=True)
class BrowserProducer:
namespace: str
owner: str
thread_id: str
session_id: str
incarnation: str
def __post_init__(self):
for name in ("namespace", "owner", "thread_id", "session_id", "incarnation"):
_text(getattr(self, name), name)
@dataclass(frozen=True)
class BrowserPageResource:
producer: BrowserProducer
page_id: str
navigation_generation: int
observed_url: str
def __post_init__(self):
if (not isinstance(self.producer, BrowserProducer)
or type(self.navigation_generation) is not int or self.navigation_generation < 0):
raise ValueError("Malformed browser page identity")
_text(self.page_id, "page")
_text(self.observed_url, "observed URL")
@dataclass(frozen=True)
class ExternalResource:
namespace: str
File diff suppressed because it is too large Load Diff
+649
View File
@@ -0,0 +1,649 @@
"""Trusted browser observations. No page execution capability is available.
0.35.0 local-launch CLI drops pin flags on `session info`; live Docker probes
proved destroyed-target retargeting. Observations are not permission to run a
page command. The future producer must atomically enforce expected identities.
"""
from __future__ import annotations
import asyncio
import base64
from contextlib import contextmanager
from contextvars import ContextVar
from dataclasses import dataclass, replace, field
import hashlib
import json
import os
from pathlib import Path
import platform
import re
import struct
import tempfile
from typing import Any
from urllib.parse import urlsplit
from src.agent_runtime.resources import (
BrowserPageResource, BrowserSessionObservation, BrowserSessionResource,
NativeBackendResource, ResourceIdentityError,
)
from src.process_lifecycle import ProcessIdentity, observe
from src.constants import BROWSER_RESOURCES_DIR
PRODUCER_VERSION = "0.35.0"
PRODUCER_HASHES = {
"linux-x64": "b7a28c3a43a7008dd02585e2e60c391c08983f7a099149caed63c9f13f57b752",
"linux-arm64": "92cd7d0897837ac648b9a6ab1965c69c5920e0f54df57e4295cdb1143b0541c8",
}
# Explicit release installation paths; PATH and npm caches are never searched.
PRODUCER_ROOT = Path("/usr/local/lib/node_modules/agent-browser/bin")
STATE_ROOT = Path(BROWSER_RESOURCES_DIR)
CLIENT_DEADLINE_S = 20 # Below 0.35.0's source-verified 30s read/resend floor.
CDP_DEADLINE_S = 3
CDP_METHODS = frozenset({"Target.getTargets", "Target.getTargetInfo", "Target.attachToTarget",
"Page.getFrameTree", "Target.detachFromTarget"})
PAGE_ACTIONS = frozenset({"open", "read", "snapshot", "find", "evaluate", "click", "fill",
"press", "scroll", "wait", "screenshot", "navigate", "reload", "back", "forward",
"select_page", "close_page", "network", "console", "new_page", "tabs"})
SESSION_ACTIONS = frozenset({"session_info"})
PAGE_FAILURE = "browser_page_authority_unavailable"
_ACTIVE = ContextVar("browser_resource_operation", default=None)
_REGISTRY: dict[tuple[str, str], "RegisteredBrowser"] = {}
def digest(domain, value):
return hashlib.sha256((domain + "\0" + json.dumps(value, sort_keys=True, separators=(",", ":"))).encode()).hexdigest()
def incarnation(observation):
values = observation.to_dict() if hasattr(observation, "to_dict") else dict(observation)
values.pop("session_incarnation", None)
return digest("odysseus.browser.session.v1", values)
def browser_digest(url):
# Never include the capability URL, raw GUID or exceptions containing them
# in results/logs/persisted records.
if not isinstance(url, str) or not re.fullmatch(
r"ws://127\.0\.0\.1:[1-9][0-9]{0,4}/devtools/browser/[a-f0-9]{8}(?:-[a-f0-9]{4}){3}-[a-f0-9]{12}", url):
raise ResourceIdentityError("Unverifiable browser endpoint")
parsed = urlsplit(url)
if parsed.port is None or parsed.port > 65535:
raise ResourceIdentityError("Invalid browser endpoint port")
return digest("odysseus.browser.guid.v1", parsed.path.rsplit("/", 1)[-1])
def page_unavailable():
return {"error": "The configured producer cannot guarantee stable binding to the captured page in local-launch mode.",
"exit_code": 1, "failure_kind": PAGE_FAILURE, "executed": False,
"retryable": False, "producer_capability_unavailable": True}
def parse_operation(content):
from src.agent_runtime.authority import ExactOperation
operation = ExactOperation.normalize("private_browser", content)
try:
args = json.loads(operation.input)
except (ValueError, TypeError):
raise ResourceIdentityError("Browser arguments require a JSON object") from None
if not isinstance(args, dict):
raise ResourceIdentityError("Browser arguments require a JSON object")
action = args.get("action")
if not isinstance(action, str) or action not in PAGE_ACTIONS | SESSION_ACTIONS | {"close"}:
raise ResourceIdentityError("Unsupported browser action; raw commands and batch are forbidden")
allowed = {"action", "page", "url", "selector", "target", "ref", "key", "direction", "amount",
"timeout_ms", "timeout_s", "text", "value", "script", "path", "find"}
if set(args) - allowed:
raise ResourceIdentityError("Browser flags, labels, configuration and raw targetIds are forbidden")
if "page" in args and (not isinstance(args["page"], str) or not re.fullmatch(r"t[1-9][0-9]*", args["page"])):
raise ResourceIdentityError("Browser page selector must be tN")
if action in SESSION_ACTIONS and set(args) != {"action"}:
raise ResourceIdentityError("Session metadata takes no page or CLI arguments")
for key, value in args.items():
if isinstance(value, str) and ("\0" in value or value.lstrip().startswith("-")):
raise ResourceIdentityError("Model values cannot become browser flags")
return operation, args
def native_browser(operation, backend):
return operation.tool == "private_browser" and isinstance(backend, NativeBackendResource)
@dataclass(frozen=True)
class TrustedProducer:
path: Path
platform: str
binary_sha256: str
def validate(self):
if (self.path != PRODUCER_ROOT / ("agent-browser-" + self.platform)
or self.path.is_symlink() or not self.path.is_file()
or self.path.stat().st_mode & 0o022
or self.path.stat().st_uid != os.getuid() and self.path.stat().st_uid != 0
or hashlib.sha256(self.path.read_bytes()).hexdigest() != PRODUCER_HASHES.get(self.platform)):
raise ResourceIdentityError("Browser producer is not an allowlisted release binary")
async def trusted_producer():
machine = {"x86_64": "x64", "aarch64": "arm64"}.get(platform.machine())
key = platform.system().lower() + "-" + str(machine)
if key not in PRODUCER_HASHES:
raise ResourceIdentityError("Unsupported browser producer platform")
producer = TrustedProducer(PRODUCER_ROOT / ("agent-browser-" + key), key, PRODUCER_HASHES[key])
producer.validate()
stdout, _ = await run_client([str(producer.path), "--version"], env={"PATH": "/usr/bin:/bin"}, cwd="/")
if stdout.strip() != "agent-browser " + PRODUCER_VERSION:
raise ResourceIdentityError("Unsupported browser producer version")
return producer
async def run_client(argv, *, env, cwd):
"""One bounded invocation, never retry. Timeout/cancellation kills the client.
Internal immediate EOF/reset retries cannot be eliminated by an outer
deadline. Consequently no effect is authorized by this client wrapper.
"""
process = None
# Files avoid detached daemon pipe inheritance keeping communicate alive.
with tempfile.TemporaryFile() as out, tempfile.TemporaryFile() as err:
spawn = None
try:
spawn = asyncio.create_task(asyncio.create_subprocess_exec(*argv, stdout=out, stderr=err,
stdin=asyncio.subprocess.DEVNULL, env=env, cwd=cwd, start_new_session=True))
process = await asyncio.shield(spawn)
await asyncio.wait_for(process.wait(), CLIENT_DEADLINE_S)
if process.returncode != 0:
raise ResourceIdentityError("Browser producer command failed")
out.seek(0); err.seek(0)
raw = out.read(1024 * 1024 + 1)
if len(raw) > 1024 * 1024:
raise ResourceIdentityError("Oversized producer response")
return raw.decode("utf-8", errors="strict"), ""
except (asyncio.TimeoutError, asyncio.CancelledError):
if process is None and spawn is not None:
process = await asyncio.shield(spawn)
if process is not None and process.returncode is None:
process.kill()
await asyncio.shield(process.wait())
raise
def response(raw):
from src.agent_runtime.authority import _pairs, _invalid_constant
try:
value = json.loads(raw, object_pairs_hook=_pairs, parse_constant=_invalid_constant)
except (ValueError, TypeError):
raise ResourceIdentityError("Malformed browser producer response") from None
if (not isinstance(value, dict) or set(value) - {"success", "data", "error"} or value.get("success") is not True
or value.get("error") is not None or not isinstance(value.get("data"), dict)):
raise ResourceIdentityError("Unsuccessful browser producer response")
return value["data"]
@dataclass
class RegisteredBrowser:
owner: str
thread_id: str
producer: TrustedProducer
key: str
cwd: Path
env: dict[str, str]
config: Path
config_identity: tuple[int, int]
lock: asyncio.Lock
session: BrowserSessionResource | None = None
pages: tuple[BrowserPageResource, ...] = ()
# A successful pin flag is NOT evidence this producer has armed its manager.
pin_armed_for: str | None = None
_endpoint: str = field(default="", repr=False) # In memory only, never a snapshot.
def validate_config(self):
self.producer.validate()
expected = owned_environment(self.cwd, self.key)
if self.env != expected or self.config != self.cwd / "config.json":
raise ResourceIdentityError("Browser producer configuration changed")
info = self.config.lstat()
if (self.cwd.is_symlink() or self.cwd.stat().st_mode & 0o077
or self.config.is_symlink() or info.st_mode & 0o077
or (info.st_dev, info.st_ino) != self.config_identity or self.config.read_text() != "{}"):
raise ResourceIdentityError("Browser owned configuration changed")
async def command(self, *args):
self.validate_config()
raw, _ = await run_client([str(self.producer.path), "--config", str(self.config),
"--session", self.key, "--json", *args], env=self.env, cwd=self.cwd)
return response(raw)
def invalidate(self):
self.session = None
self.pages = ()
self.pin_armed_for = None
self._endpoint = ""
def owned_environment(cwd, key):
# No ambient AGENT_BROWSER_*, XDG, proxy, provider, CDP, profile or state.
return {"PATH": "/usr/bin:/bin", "HOME": str(cwd), "TMPDIR": str(cwd / "tmp"),
"AGENT_BROWSER_SOCKET_DIR": str(cwd / "runtime"),
"AGENT_BROWSER_EXECUTABLE_PATH": "/usr/bin/chromium",
"AGENT_BROWSER_IDLE_TIMEOUT_MS": "300000"}
async def register_producer(owner, thread_id):
"""Server-only registration, not model discovery, restoration or lookup.
Does not launch a daemon/browser. A future trusted launch producer must
populate this exact owned runtime; legacy lifecycle entries are not adopted.
"""
if not isinstance(owner, str) or not owner or not isinstance(thread_id, str) or not thread_id:
raise ResourceIdentityError("Browser application ownership is required")
if (owner, thread_id) in _REGISTRY:
raise ResourceIdentityError("Browser producer is already registered")
producer = await trusted_producer()
key = "ody-" + digest("odysseus.browser.selector.v1", [owner, thread_id])[:24]
STATE_ROOT.mkdir(parents=True, exist_ok=True, mode=0o700)
cwd = STATE_ROOT / key
cwd.mkdir(mode=0o700) # Existing unregistered state is not authoritative.
for directory in ("tmp", "runtime"):
(cwd / directory).mkdir(mode=0o700)
config = cwd / "config.json"
with config.open("x") as f:
os.chmod(config, 0o600)
f.write("{}")
f.flush(); os.fsync(f.fileno())
info = config.stat()
record = RegisteredBrowser(owner, thread_id, producer, key, cwd, owned_environment(cwd, key),
config, (info.st_dev, info.st_ino), asyncio.Lock())
record.validate_config()
_REGISTRY[(owner, thread_id)] = record
return record
def registered(owner, thread_id):
return _REGISTRY.get((owner, thread_id)) # Lookup never creates a session.
def daemon_observation(record, info):
required = {"session", "active", "version", "pid", "runtimeError", "socketDir", "namespace", "runtime"}
if (not isinstance(info, dict) or not required <= info.keys()
or info.get("session") != record.key or info.get("active") is not True
or info.get("version") != PRODUCER_VERSION or info.get("runtimeError") is not None
or info.get("socketDir") != record.env["AGENT_BROWSER_SOCKET_DIR"]
or info.get("namespace") is not None):
raise ResourceIdentityError("Unregistered browser daemon")
runtime = info.get("runtime")
pid = info.get("pid")
required_runtime = {"backgroundPid", "session", "engine", "browserLaunched",
"compatibilityStatus", "socketDir", "restoreKey"}
if (type(pid) is not int or pid <= 0 or not isinstance(runtime, dict)
or not required_runtime <= runtime.keys()
or runtime.get("backgroundPid") != pid or runtime.get("session") != record.key
or runtime.get("engine") != "chrome" or runtime.get("browserLaunched") is not True
or runtime.get("compatibilityStatus") != "current"
or runtime.get("socketDir") != info["socketDir"] or runtime.get("restoreKey") is not None):
raise ResourceIdentityError("Malformed browser lifecycle observation")
def executable(candidate):
return Path(f"/proc/{candidate}/exe").resolve(strict=True)
seen = observe(pid, executable)
if seen is None or seen.facts != record.producer.path or not seen.identity.owned():
raise ResourceIdentityError("Daemon does not match the trusted binary incarnation")
return seen.identity
class CDPSidecar:
"""Minimal loopback websocket client for the five identity-only methods."""
def __init__(self, url):
browser_digest(url)
self._url = url # Ephemeral capability; never repr/serialize/log.
self._counter = 0
async def __aenter__(self):
url = urlsplit(self._url)
self.reader, self.writer = await asyncio.wait_for(asyncio.open_connection(url.hostname, url.port), CDP_DEADLINE_S)
key = base64.b64encode(os.urandom(16)).decode()
request = f"GET {url.path} HTTP/1.1\r\nHost: 127.0.0.1:{url.port}\r\nUpgrade: websocket\r\nConnection: Upgrade\r\nSec-WebSocket-Key: {key}\r\nSec-WebSocket-Version: 13\r\n\r\n"
try:
self.writer.write(request.encode())
await asyncio.wait_for(self.writer.drain(), CDP_DEADLINE_S)
header = await asyncio.wait_for(self.reader.readuntil(b"\r\n\r\n"), CDP_DEADLINE_S)
accept = base64.b64encode(hashlib.sha1((key + "258EAFA5-E914-47DA-95CA-C5AB0DC85B11").encode()).digest())
headers = dict(line.split(b":", 1) for line in header.split(b"\r\n")[1:] if b":" in line)
if not header.startswith(b"HTTP/1.1 101 ") or not any(k.lower() == b"sec-websocket-accept" and v.strip() == accept for k, v in headers.items()):
raise ResourceIdentityError("Invalid CDP websocket handshake")
return self
except BaseException:
self.writer.close()
raise
async def __aexit__(self, *args):
self.writer.close()
try:
await asyncio.wait_for(self.writer.wait_closed(), CDP_DEADLINE_S)
finally:
self._url = ""
async def _send(self, payload, opcode=1):
mask = os.urandom(4)
size = len(payload)
if size > 65535 or opcode in {9, 10} and size > 125:
raise ResourceIdentityError("Oversized CDP observation request")
length = bytes([0x80 | size]) if size < 126 else b"\xfe" + struct.pack("!H", size)
self.writer.write(bytes([0x80 | opcode]) + length + mask + bytes(b ^ mask[i % 4] for i, b in enumerate(payload)))
await self.writer.drain()
async def _message(self):
chunks = bytearray()
for _ in range(64):
first, second = await self.reader.readexactly(2)
if second & 0x80 or first & 0x70:
raise ResourceIdentityError("Invalid CDP websocket frame")
size = second & 127
if size in {126, 127}:
size = struct.unpack("!H" if size == 126 else "!Q", await self.reader.readexactly(2 if size == 126 else 8))[0]
if size + len(chunks) > 1024 * 1024:
raise ResourceIdentityError("Oversized CDP response")
payload = await self.reader.readexactly(size)
opcode = first & 15
if opcode == 9:
await self._send(payload, 10)
continue
if opcode not in {0, 1}:
raise ResourceIdentityError("Unexpected CDP websocket opcode")
chunks.extend(payload)
if first & 0x80:
from src.agent_runtime.authority import _pairs, _invalid_constant
return json.loads(chunks, object_pairs_hook=_pairs, parse_constant=_invalid_constant)
raise ResourceIdentityError("Unbounded CDP websocket response")
async def call(self, method, params=None, session_id=None):
if method not in CDP_METHODS:
raise ResourceIdentityError("CDP method is outside the identity allowlist")
self._counter += 1
message = {"id": self._counter, "method": method, "params": params or {}}
if session_id is not None:
message["sessionId"] = session_id
async def exchange():
await self._send(json.dumps(message).encode())
for _ in range(32):
result = await self._message()
if not isinstance(result, dict):
raise ResourceIdentityError("Malformed CDP identity envelope")
if "id" in result and type(result["id"]) is not int:
raise ResourceIdentityError("Malformed CDP response identity")
if result.get("id") == self._counter:
if "error" in result or not isinstance(result.get("result"), dict):
raise ResourceIdentityError("Unverifiable CDP identity response")
return result["result"]
raise ResourceIdentityError("Unbounded CDP event stream")
try:
return await asyncio.wait_for(exchange(), CDP_DEADLINE_S)
except (OSError, ValueError, asyncio.TimeoutError, asyncio.IncompleteReadError):
raise ResourceIdentityError("CDP identity observation unavailable") from None
def tabs_schema(data):
tabs = data.get("tabs")
if not isinstance(tabs, list):
raise ResourceIdentityError("Missing producer tab inventory")
aliases, targets = set(), set()
for row in tabs:
if (not isinstance(row, dict) or set(row) != {"tabId", "targetId", "label", "title", "url", "type", "active"}
or not isinstance(row.get("tabId"), str)
or not re.fullmatch(r"t[1-9][0-9]*", row["tabId"])
or not isinstance(row.get("targetId"), str) or not re.fullmatch(r"[A-F0-9]{32}", row["targetId"])
or row.get("label") is not None or row.get("type") != "page"
or type(row.get("active")) is not bool or not isinstance(row.get("url"), str)
or not isinstance(row.get("title"), str)
or row["tabId"] in aliases or row["targetId"] in targets):
raise ResourceIdentityError("Malformed, labelled or ambiguous producer page")
aliases.add(row["tabId"]); targets.add(row["targetId"])
return tabs
async def observe_registered(record, alias=None):
"""Observe only an existing registered producer; never auto-launch/rearm.
get cdp-url can launch when cold, so it is preceded by strict active runtime
validation and followed by launch metadata rejection. No result reaches the
model if the trusted observation cannot be established.
"""
try:
async with record.lock:
return await _observe_registered_locked(record, alias)
except BaseException:
record.invalidate()
raise
async def _observe_registered_locked(record, alias):
try:
first = daemon_observation(record, await record.command("session", "info"))
endpoint = await record.command("get", "cdp-url")
lifecycle = endpoint.get("lifecycle")
if (not isinstance(lifecycle, dict) or any(lifecycle.get(k) is not False for k in
("launched", "relaunchedBrowser", "restartedBackground"))):
raise ResourceIdentityError("Unexpected browser lifecycle launch")
url = endpoint.get("cdpUrl")
browser = browser_digest(url)
values = dict(producer_namespace="native:agent-browser", producer_version=PRODUCER_VERSION,
platform=record.producer.platform, binary_sha256=record.producer.binary_sha256,
configuration_digest=digest("odysseus.browser.config.v1", [record.env, str(record.cwd), "{}"]),
session_key=record.key, daemon=first.to_record(), browser_instance_digest=browser)
observation = BrowserSessionObservation(**{**values, "daemon": first, "session_incarnation": incarnation(values)})
session = BrowserSessionResource(record.owner, record.thread_id, observation)
rows = tabs_schema(await record.command("tab", "list"))
pages = []
async with CDPSidecar(url) as cdp:
targets = (await cdp.call("Target.getTargets")).get("targetInfos")
if not isinstance(targets, list):
raise ResourceIdentityError("Missing CDP target inventory")
for row in rows:
# Never select a page by targetId: even read dispatch is disabled.
target = row["targetId"]
if not any(t.get("targetId") == target and t.get("type") == "page" for t in targets if isinstance(t, dict)):
raise ResourceIdentityError("Producer/CDP target disagreement")
attached = await cdp.call("Target.attachToTarget", {"targetId": target, "flatten": True})
sid = attached.get("sessionId")
if not isinstance(sid, str) or not sid:
raise ResourceIdentityError("Missing CDP observation session")
try:
tree = await cdp.call("Page.getFrameTree", session_id=sid)
frame = tree.get("frameTree", {}).get("frame", {})
if frame.get("id") != target or not isinstance(frame.get("loaderId"), str) or not frame["loaderId"]:
raise ResourceIdentityError("Unsupported main-frame/document invariant")
pages.append(BrowserPageResource(session, target, frame["loaderId"], row["tabId"], row["url"]))
info = (await cdp.call("Target.getTargetInfo", {"targetId": target})).get("targetInfo", {})
if info.get("targetId") != target or info.get("type") != "page":
raise ResourceIdentityError("Page disappeared during observation")
finally:
await cdp.call("Target.detachFromTarget", {"sessionId": sid})
last = daemon_observation(record, await record.command("session", "info"))
final = await record.command("get", "cdp-url")
if first != last or not first.owned() or browser_digest(final.get("cdpUrl")) != browser:
raise ResourceIdentityError("Browser incarnation changed during observation")
final_lifecycle = final.get("lifecycle", {})
if any(final_lifecycle.get(k) is not False for k in ("launched", "relaunchedBrowser", "restartedBackground")):
raise ResourceIdentityError("Unexpected browser replacement")
if record.session != session:
record.invalidate()
record.session, record.pages = session, tuple(pages)
record._endpoint = url
if alias is not None:
match = [p for p in pages if p.resolved_alias == alias]
if len(match) != 1:
raise ResourceIdentityError("Unresolved browser alias")
return match[0]
return session
except BaseException:
record.invalidate()
raise
def validate_session(resource):
record = registered(resource.owner, resource.thread_id)
if record is None or record.session != resource or not resource.observation.daemon.owned():
raise ResourceIdentityError("Browser observation is stale, replaced or unregistered")
record.validate_config()
def validate_page(resource):
resource.session.validate()
record = registered(resource.session.owner, resource.session.thread_id)
if not any(p.target_id == resource.target_id and (resource.scope == "page" or p.loader_id == resource.loader_id) for p in record.pages):
raise ResourceIdentityError("Browser page/document observation changed")
def seal_browser_resources(authority):
record = registered(authority.owner, authority.session_id)
if record is None or record.session is None or not any(g.tool == "private_browser" for g in authority.grants):
return (), ()
try:
record.session.validate()
except ResourceIdentityError:
return (), ()
return (record.session,), record.pages
def intersect_browser(parent_sessions, parent_pages, child_sessions, child_pages):
# Validate old observations before considering anything newly observed.
for item in (*parent_sessions, *parent_pages, *child_sessions, *child_pages):
item.validate()
sessions = tuple(s for s in parent_sessions if s in child_sessions)
pages = []
for p in parent_pages:
for c in child_pages:
if p.session == c.session and p.target_id == c.target_id and (p.scope == "page" or p.loader_id == c.loader_id):
pages.append(c if p.scope == "page" else replace(c, loader_id=p.loader_id, scope="document"))
return sessions, tuple(pages)
@dataclass(frozen=True)
class BoundBrowserOperation:
operation: Any
request_id: str
owner: str
thread_id: str
session: BrowserSessionResource
page: BrowserPageResource | None = None
exact_approval: Any = None
def validate(self):
if (self.session.owner, self.session.thread_id) != (self.owner, self.thread_id) or not self.request_id:
raise ResourceIdentityError("Browser application binding changed")
operation, args = parse_operation(self.operation.input)
if operation != self.operation or self.operation.tool != "private_browser":
raise ResourceIdentityError("Browser normalized operation changed")
self.session.validate()
if self.page is not None:
if self.page.session != self.session:
raise ResourceIdentityError("Browser page/session binding changed")
self.page.validate()
if args["action"] not in SESSION_ACTIONS and self.page is None:
raise ResourceIdentityError("Missing proposal-bound page observation")
def to_dict(self):
return {"operation": {"tool": self.operation.tool, "input": self.operation.input,
"action": self.operation.action, "transport_tool": self.operation.transport_tool},
"request_id": self.request_id, "owner": self.owner, "thread_id": self.thread_id,
"session": self.session.to_dict(), "page": self.page.to_dict() if self.page else None}
def resolve_browser_operation(authority, operation, *, approved=None, exact_admission=False):
_, args = parse_operation(operation.input)
if approved is not None:
bound = approved
if (bound.operation != operation or (bound.request_id, bound.owner, bound.thread_id) !=
(authority.request_id, authority.owner, authority.session_id)):
raise ResourceIdentityError("Approved browser operation binding changed")
else:
record = registered(authority.owner, authority.session_id)
if record is None or record.session is None:
raise ResourceIdentityError("No admitted browser session observation")
page = None
if args["action"] not in SESSION_ACTIONS:
alias = args.get("page")
matches = [p for p in record.pages if alias and p.resolved_alias == alias]
if len(matches) != 1:
raise ResourceIdentityError("An observed tN selector is required")
page = matches[0] # Alias is audit metadata after this single resolution.
bound = BoundBrowserOperation(operation, authority.request_id, authority.owner,
authority.session_id, record.session, page)
bound.validate()
if not (approved is not None and exact_admission and not authority.inherited):
if bound.page is None and bound.session not in authority.browser_sessions:
raise ResourceIdentityError("Browser session is outside admitted scope")
if bound.page is not None and not any(p.session == bound.page.session and p.target_id == bound.page.target_id
and (p.scope == "page" or p.loader_id == bound.page.loader_id) for p in authority.browser_pages):
raise ResourceIdentityError("Browser page/document is outside admitted scope")
return bound
async def revalidate_browser_operation(bound):
bound.validate()
record = registered(bound.owner, bound.thread_id)
async with record.lock:
try:
# The existing capability connects to the captured browser only.
# Never issue get cdp-url here: its CLI can auto-launch a replacement.
if daemon_observation(record, await record.command("session", "info")) != bound.session.observation.daemon:
raise ResourceIdentityError("Browser proposal daemon replaced")
if browser_digest(record._endpoint) != bound.session.observation.browser_instance_digest:
raise ResourceIdentityError("Browser proposal incarnation replaced")
async with CDPSidecar(record._endpoint) as cdp:
await cdp.call("Target.getTargets")
bound.validate()
except BaseException:
record.invalidate()
raise
@contextmanager
def bind_browser_operation(bound):
if bound is not None:
bound.validate()
token = _ACTIVE.set(bound)
try:
yield bound
finally:
_ACTIVE.reset(token)
async def execute_browser(content, ctx):
try:
operation, args = parse_operation(content)
# Unconditional capability denial, before producer selection, alias
# lookup, spawning, approval claims or any page-specific data read.
if args["action"] not in SESSION_ACTIONS:
return page_unavailable()
from src.agent_runtime.authority import active_request_authority
authority, bound = active_request_authority(), _ACTIVE.get()
if authority is None or bound is None or bound.operation != operation:
raise ResourceIdentityError("Browser producer requires a normalized resource-bound operation")
if (authority.owner, authority.request_id, authority.session_id) != (bound.owner, bound.request_id, bound.thread_id):
raise ResourceIdentityError("Browser caller authority changed")
if (str(ctx.get("owner") or "").casefold(), str(ctx.get("session_id") or "")) != (bound.owner, bound.thread_id):
raise ResourceIdentityError("Browser producer caller changed")
if not authority.permits(operation):
approval = bound.exact_approval
if (authority.inherited or approval is None or not approval._claimed
or approval.pending.browser_operation is None or approval.pending.browser_operation.to_dict() != bound.to_dict()):
raise ResourceIdentityError("Browser operation lacks exact admission")
bound.validate()
record = registered(bound.owner, bound.thread_id)
await revalidate_browser_operation(bound)
async with record.lock:
bound.validate()
# Metadata only. Never return URL/title/content, raw CDP capability,
# or producer lifecycle data as semantic verification.
output = {"session_incarnation": bound.session.observation.session_incarnation,
"producer_version": PRODUCER_VERSION}
return {"output": json.dumps(output), "exit_code": 0, "executed": True,
"browser_page_operations_supported": False}
except asyncio.CancelledError:
record = registered(str(ctx.get("owner") or "").casefold(), str(ctx.get("session_id") or ""))
if record is not None:
record.invalidate()
raise
except Exception:
# No raw producer/CDP exception text: it can contain capability URLs.
return {"error": "Trusted browser session metadata is unavailable.", "exit_code": 1,
"executed": False, "retryable": False, "failure_kind": "browser_session_authority_unavailable"}
+7 -43
View File
@@ -158,7 +158,7 @@ SAFE_ACTIONS = {
'manage_contact': frozenset({'list', 'search', 'find'}),
'private_browser': frozenset({
'open', 'read', 'snapshot', 'find', 'evaluate', 'click', 'fill', 'press',
'scroll', 'wait', 'screenshot', 'close', 'batch',
'scroll', 'wait', 'screenshot', 'close', 'session_info',
}),
# These UI effects are reversible. A model switch is additionally bound
# below to explicit user wording; keep toggle mutation, mode changes, and
@@ -2472,31 +2472,16 @@ def compact_schemas(schemas, *, model=None):
properties['code']['description'] = 'Valid Python source code to execute once.'
elif function.get('name') == 'private_browser':
function['description'] = (
'Browse and interact with websites. First open then snapshot the page. '
'Use returned element refs (such as @e1) for fill/click; never guess selectors. '
'press uses a keyboard key such as Enter on the focused element. '
'To search a site, fill its search field and submit, then snapshot results. '
'find only locates one existing page element/text; it does not search the site. '
'To list links, headings, or controls, use snapshot and read its returned DOM.'
'Registered session_info metadata only. Page/document reads and effects are '
'unavailable because the producer cannot atomically bind a captured page. '
'No batch, raw commands, flags, labels or current-tab selectors.'
)
for name in ('target', 'selector'):
if isinstance(properties.get(name), dict):
properties[name]['description'] = (
'For click/fill/read/wait: snapshot ref such as @e2 or CSS selector, not visible text.'
'Disabled page operation: ref such as @e2 or CSS selector, not visible text.'
)
if isinstance(properties.get('key'), dict):
properties['key']['description'] = 'For press: keyboard key such as Enter on the currently focused element.'
commands = properties.get('commands')
if isinstance(commands, dict):
commands['description'] = (
'For action=batch, an array of command arrays such as '
'[["open","https://example.com"],["snapshot"]].'
)
commands['items'] = {
'type': 'array',
'items': {'type': 'string'},
'minItems': 1,
}
properties.pop('commands', None)
elif function.get('name') == 'ui_control':
function['description'] = (
'Control the UI. Themes: get_theme reads current saved colors and available names; '
@@ -2672,28 +2657,6 @@ def normalize_preview_function_args(name, args, *, user_text=''):
# is a lossless completion of an explicit field, not inferred content.
args['content'] += '\n'
tool_type, normalized = normalize_native_function_args(name, args)
if (
tool_type == 'private_browser'
and str(normalized.get('action') or '').casefold() == 'open'
and str(normalized.get('url') or '').startswith(('http://', 'https://'))
):
# Opening a page invalidates old element references. The compact
# model commonly emits only ``open`` and then answers from the title,
# leaving a later conversational turn with no refs it can safely
# click. Make the transport honor the browser schema's documented
# open-then-snapshot contract in one atomic call. This is generic DOM
# grounding, not a rule for any particular site or link label.
normalized = {
'action': 'batch',
'commands': [
['open', normalized['url']],
['snapshot'],
],
**(
{'timeout_ms': normalized['timeout_ms']}
if normalized.get('timeout_ms') is not None else {}
),
}
if (
tool_type == 'inspect_media'
and str(normalized.get('sampling') or '').casefold() == 'overview'
@@ -2703,6 +2666,7 @@ def normalize_preview_function_args(name, args, *, user_text=''):
# eight observations per native sheet. Avoid the tool's broader
# default, which would require lossy second-stage sheet packing.
normalized['frames'] = 24
return tool_type, normalized
+1
View File
@@ -90,6 +90,7 @@ RAG_DIR = os.path.join(DATA_DIR, "rag")
CHROMA_DIR = os.path.join(DATA_DIR, "chroma")
BG_JOBS_DIR = os.path.join(DATA_DIR, "bg_jobs")
PROCESS_RESOURCES_DIR = os.path.join(DATA_DIR, "process_resources")
BROWSER_RESOURCES_DIR = os.path.join(DATA_DIR, "browser_resources")
DEEP_RESEARCH_DIR = os.path.join(DATA_DIR, "deep_research")
MCP_OAUTH_DIR = os.path.join(DATA_DIR, "mcp_oauth")
GENERATED_IMAGES_DIR = os.path.join(DATA_DIR, "generated_images")
+13
View File
@@ -32,6 +32,7 @@ if TYPE_CHECKING:
from src.agent_runtime.remote_resources import BoundBackendOperation
from src.agent_runtime.owned_resources import BoundOwnedOperation
from src.agent_runtime.process_resources import BoundProcessOperation
from src.browser_identity import BoundBrowserOperation
DEFAULT_APPROVAL_TTL_SECONDS = 10 * 60
@@ -129,6 +130,7 @@ def _binding_payload(
backend_operation=None,
owned_operation=None,
process_operation=None,
browser_operation=None,
) -> dict[str, Any]:
return {
"owner": _normalized_owner(owner),
@@ -154,6 +156,7 @@ def _binding_payload(
"backend_operation": backend_operation.to_dict() if backend_operation is not None else None,
"owned_operation": owned_operation.to_dict() if owned_operation is not None else None,
"process_operation": process_operation.to_dict() if process_operation is not None else None,
"browser_operation": browser_operation.to_dict() if browser_operation is not None else None,
}
@@ -188,6 +191,7 @@ class PendingToolApproval:
backend_operation: BoundBackendOperation | None = None
owned_operation: BoundOwnedOperation | None = None
process_operation: BoundProcessOperation | None = None
browser_operation: BoundBrowserOperation | None = None
def public_payload(self, *, reason: str | None = None) -> dict[str, Any]:
return {
@@ -301,6 +305,7 @@ class ExactToolApproval:
backend_operation=self.pending.backend_operation,
owned_operation=self.pending.owned_operation,
process_operation=self.pending.process_operation,
browser_operation=self.pending.browser_operation,
)
return _canonical_digest(expected) == self.pending.digest
@@ -397,6 +402,7 @@ class ToolApprovalStore:
backend_operation = None
owned_operation = None
process_operation = None
browser_operation = None
from src.agent_runtime.remote_resources import BoundBackendOperation, resolve_backend
from src.agent_runtime.owned_resources import needs_owned_binding, resolve_owned_operation
from src.agent_runtime.resources import NativeBackendResource
@@ -412,6 +418,11 @@ class ToolApprovalStore:
from src.agent_runtime.process_resources import needs_process_binding, resolve_process_operation
if request_authority is not None and needs_process_binding(operation, backend):
process_operation = resolve_process_operation(request_authority, operation, backend)
from src.browser_identity import native_browser, resolve_browser_operation
if native_browser(operation, backend):
if request_authority is None:
raise ValueError("Browser approval requires originating resource authority")
browser_operation = resolve_browser_operation(request_authority, operation)
if isinstance(backend, NativeBackendResource) and needs_owned_binding(operation):
resolved_owned = resolve_owned_operation(operation, owner=_normalized_owner(owner),
thread_id=str(session_id or ""), request_id=backend_operation.request_id,
@@ -460,6 +471,7 @@ class ToolApprovalStore:
backend_operation=backend_operation,
owned_operation=owned_operation,
process_operation=process_operation,
browser_operation=browser_operation,
)
pending = PendingToolApproval(
approval_id=secrets.token_urlsafe(32),
@@ -488,6 +500,7 @@ class ToolApprovalStore:
backend_operation=backend_operation,
owned_operation=owned_operation,
process_operation=process_operation,
browser_operation=browser_operation,
)
with self._lock:
self._purge_expired_locked(now)
+35 -1
View File
@@ -1327,6 +1327,10 @@ from src.agent_runtime.authority import (
from src.agent_runtime.process_resources import (
active_process_operation, bind_process_operation, needs_process_binding, resolve_process_operation,
)
from src.browser_identity import (
native_browser, parse_operation as parse_browser_operation, SESSION_ACTIONS,
page_unavailable, resolve_browser_operation, bind_browser_operation, revalidate_browser_operation,
)
@record_action
@@ -1411,6 +1415,22 @@ async def execute_tool_block(
}
transport = operation.transport_tool
if operation.tool == "private_browser":
try:
_, browser_args = parse_browser_operation(operation.input)
except (ValueError, TypeError):
return f"{transport}: UNSUPPORTED", {**page_unavailable(), "error": "Browser raw commands, flags and batches are unsupported."}
if browser_args["action"] not in SESSION_ACTIONS:
return f"{transport}: UNSUPPORTED", page_unavailable()
# Raw global Playwright MCP has no authoritative session/page observation.
# Its transport process and remote backend identity cannot substitute for it.
if transport.startswith("mcp__") and transport.rsplit("__", 1)[-1] in {
"browser_click", "browser_fill_form", "browser_type", "browser_press_key", "browser_evaluate",
"browser_navigate", "browser_navigate_back", "browser_snapshot", "browser_take_screenshot",
"browser_wait_for", "browser_tabs", "browser_close", "browser_run_code", "browser_network_requests",
"browser_console_messages", "browser_drag", "browser_hover", "browser_select_option",
"browser_file_upload", "browser_handle_dialog", "browser_resize", "browser_install"}:
return f"{transport}: UNSUPPORTED", page_unavailable()
try:
pending = exact_approval.pending if exact_approval is not None else None
if pending is not None and pending.backend_operation is None:
@@ -1420,8 +1440,20 @@ async def execute_tool_block(
approved=pending.backend_operation if pending is not None else None,
exact_admission=exact_admission)
external_resource_call = isinstance(backend_operation.resource, ExternalResource)
if operation.tool == "private_browser" and external_resource_call:
raise ResourceIdentityError("External backend cannot supply native browser session authority")
owned_operation = None
process_operation = None
browser_operation = None
if native_browser(operation, backend_operation.resource):
_, browser_args = parse_browser_operation(operation.input)
if browser_args["action"] not in SESSION_ACTIONS:
return f"{transport}: UNSUPPORTED", page_unavailable()
if pending is not None and pending.browser_operation is None:
raise ResourceIdentityError("Approved action has no sealed browser identity")
browser_operation = resolve_browser_operation(authority, operation,
approved=pending.browser_operation if pending is not None else None, exact_admission=exact_admission)
await revalidate_browser_operation(browser_operation)
if needs_process_binding(operation, backend_operation.resource):
if pending is not None and pending.process_operation is None:
raise ResourceIdentityError("Approved action has no sealed process/job identity")
@@ -1555,11 +1587,13 @@ async def execute_tool_block(
backend_operation.validate(client_runtime_context)
if process_operation is not None and approval_claimed:
process_operation = replace(process_operation, exact_approval=exact_approval)
if browser_operation is not None and approval_claimed:
browser_operation = replace(browser_operation, exact_approval=exact_approval)
normalized = resource_operation or owned_operation
sealed_document = owned_operation or (exact_approval.pending if approval_claimed else None)
with (bind_request_authority(authority), bind_resource_operation(resource_operation),
bind_backend_operation(backend_operation), bind_owned_operation(owned_operation),
bind_process_operation(process_operation)):
bind_process_operation(process_operation), bind_browser_operation(browser_operation)):
output = await _execute_tool_block_impl(
ToolBlock(transport, normalized.execution_input) if normalized is not None else block,
session_id=session_id,
+2 -2
View File
@@ -111,8 +111,8 @@ BUILTIN_TOOL_DESCRIPTIONS: Dict[str, str] = {
"get_weather": "Get current weather and a three-day forecast for a city or place from Open-Meteo without an API key. Use for weather lookups before web_search.",
"web_fetch": "Fetch and read the text content of a specific URL/website the user names (e.g. 'check example.com', 'open this link'). Use when you have a concrete URL; for open-ended lookups use web_search instead.",
"pdf_extract": "Extract focused, source-attributed passages and exact table values from an online PDF or task-local /workspace/*.pdf. Use for arXiv papers, reports, manuals, PDF tables, evaluation metrics, and multi-document PDF extraction. Prefer this over Python requests, curl, downloading, pdftotext, or guessing. Include target model names, metrics, and table headings in query.",
"youtube_tool": "Read YouTube-specific data without fighting the JS page: video comments, transcripts, metadata, or latest video from a channel. Use for YouTube comments/transcript/channel latest-video tasks; use private_browser only for visual site interaction.",
"private_browser": "Private browser automation through Odysseus' agent-browser wrapper. Use only for specific pages that need JavaScript, login/session state, clicking, filling forms, waiting, screenshots, or rendered DOM inspection. For open-ended search use web_search; for ordinary URL reading use web_fetch.",
"youtube_tool": "Read YouTube-specific data without fighting the JS page: video comments, transcripts, metadata, or latest video from a channel. Use for YouTube comments/transcript/channel latest-video tasks.",
"private_browser": "Trusted metadata for an existing server-registered browser session only. Page/document reads and interactions are unavailable because the configured producer cannot guarantee exact target binding. No model batch or raw browser commands. Use web_search or web_fetch for supported web access.",
"inspect_media": "Inspect local workspace images, SVGs, videos, and PDF pages with the current multimodal model. Samples bounded timestamped video frames uniformly, at scene cuts, or from temporally diverse motion peaks; renders SVG to PNG; exports stills or clips; concatenates ranges; changes clip speed while preserving audio pitch; and renders query-relevant PDF pages. Prefer these native operations over raw ffmpeg. Increase max_dimension only for small visual details; saved exports keep source quality.",
"extract_text": "Extract exact visible text, confidence, and pixel centers from a local workspace image with Odysseus local OCR. Use for screenshots, scans, labels, numbers, receipts, and text-location tasks; use inspect_media for general visual understanding.",
"transcribe_media": "Transcribe dialogue, narration, names, and spoken timing from a local audio or video file with Odysseus local Whisper. Returns [START --> END] TEXT segments and always persists them to a workspace text file. For a named chapter, question, scene, or topic, locate its boundaries and restrict filtering to that interval. This handles audio speech; combine with inspect_media for audiovisual tasks or visually burned-in subtitles.",
+18 -25
View File
@@ -393,35 +393,28 @@ FUNCTION_TOOL_SCHEMAS = [
"type": "function",
"function": {
"name": "private_browser",
"description": "Private browser automation through Odysseus' agent-browser wrapper. After open, snapshot the page and interact with returned element refs such as @e12; click/fill target is a selector or element ref, never guessed visible text. Prefer one batch for known consecutive steps, such as open plus snapshot. Use only when a specific page needs JavaScript, login/session state, interaction, or rendered DOM. For open-ended search use web_search; for reading a normal URL use web_fetch.",
"description": "Trusted browser session metadata only. Page/document operations are unavailable because the local producer cannot atomically bind a captured target. No batch or raw CLI flags. Use web_search/web_fetch for supported web access.",
"parameters": {
"type": "object",
"properties": {
"action": {"type": "string", "enum": ["open", "read", "snapshot", "find", "evaluate", "click", "fill", "press", "scroll", "wait", "screenshot", "close", "batch"]},
"url": {"type": "string", "description": "Required URL for open; optional URL for read (omit to read the current page)"},
"selector": {"type": "string", "description": "Element ref or selector for read/click/fill/wait"},
"target": {"type": "string", "description": "Element ref returned by snapshot (preferred, e.g. @e12) or CSS selector for read/click/fill/wait; never a guessed visible label; top or bottom for scroll"},
"key": {"type": "string", "description": "Key name for press action, e.g. Enter"},
"direction": {"type": "string", "enum": ["up", "down", "left", "right"], "description": "Direction for scroll action"},
"amount": {"type": "integer", "minimum": 1, "description": "Optional scroll distance in pixels; default 300"},
"text": {"type": "string", "description": "Text for fill action"},
"value": {"type": "string", "description": "Alternative text/value for fill action"},
"find": {"type": "string", "description": "Visible text to locate for find action"},
"script": {"type": "string", "description": "JavaScript expression for evaluate action"},
"path": {"type": "string", "description": "Optional output path for screenshot"},
"commands": {
"type": "array",
"description": "Non-empty batch commands as arrays, e.g. [[\"open\", \"https://example.com\"], [\"snapshot\"]]. Do not send an empty batch; use action=snapshot for current page state.",
"items": {
"oneOf": [
{"type": "array", "items": {"type": "string"}},
{"type": "object"},
]
},
},
"timeout_ms": {"type": "integer", "description": "Optional operation timeout, max 120000; for action=wait without a selector, this is the wait duration"}
"action": {"type": "string", "enum": ["session_info", "tabs", "open", "read", "snapshot", "find", "evaluate", "click", "fill", "press", "scroll", "wait", "screenshot", "close", "navigate", "reload", "back", "forward", "select_page", "close_page", "network", "console", "new_page"]},
"page": {"type": "string", "pattern": "^t[1-9][0-9]*$", "description": "Observed alias only; page commands remain disabled for the current producer."},
"url": {"type": "string"},
"selector": {"type": "string"},
"target": {"type": "string"},
"ref": {"type": "string"},
"key": {"type": "string"},
"direction": {"type": "string"},
"text": {"type": "string"},
"value": {"type": "string"},
"script": {"type": "string"},
"path": {"type": "string"},
"find": {"type": "string"},
"amount": {"type": "integer"},
"timeout_ms": {"type": "integer", "minimum": 0, "maximum": 20000}
},
"required": ["action"]
"required": ["action"],
"additionalProperties": False
}
}
},