From c57aa7ad5c0083b70726352c253bcbba2e37a75b Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Fri, 2 Oct 2026 19:45:39 +0100 Subject: [PATCH 01/15] feat(runtime): recreate Wave 4 effect contracts on exact Wave 3 resources Recreate (rather than cherry-pick 9012e208) the effect/provenance foundation. The historical types used opaque string resource keys, a may_have_changed flag defaulting to no impact, and a single status mixing execution and verification. Claims, outcomes and observations now reference only typed Wave 3 identities (filesystem root/inode/ancestor chain, process PID+start token, job generation, owned revision, external incarnation, browser session incarnation). Browser page resources are refused. Known no-op is limited to refusal before invocation; unknown scope stays conservative; verification is derived from fresh, complete, post-settlement readback checked against an explicit predicate, and unknown execution with matching state is reported as observed state without causation. Cleanup is recorded separately from effect outcome. --- src/agent_runtime/effects.py | 800 +++++++++++++++++++++++++++++++ tests/test_effects_foundation.py | 466 ++++++++++++++++++ 2 files changed, 1266 insertions(+) create mode 100644 src/agent_runtime/effects.py create mode 100644 tests/test_effects_foundation.py diff --git a/src/agent_runtime/effects.py b/src/agent_runtime/effects.py new file mode 100644 index 000000000..e69572eec --- /dev/null +++ b/src/agent_runtime/effects.py @@ -0,0 +1,800 @@ +"""Wave 4 effect claims, outcomes, observations and verification. + +This module consumes exact Wave 3 resource identities. It never resolves a +selector, discovers an alias, grants an operation or performs I/O. A +``ResourceRef`` can only be built from an already-admitted typed Wave 3 resource +object; names, paths, PIDs, URLs, labels and dictionaries are not accepted. + +Facts are kept separate: + +* a claim records intent and scope before backend invocation, not dispatch; +* an outcome records what the executor reported, not the resulting state; +* an observation records state seen through an admitted mechanism; +* verification is derived from fresh, relevant, complete observations made + after the effect settled, and never from receipts or acknowledgements. + +History is append-only. Invalidation and freshness are computed from the +ordered record history; earlier records are never rewritten. Refresh is a new +observation. Unknown scope is conservative, never "no impact". +""" +from __future__ import annotations + +from dataclasses import dataclass +from enum import Enum +from pathlib import PurePosixPath +import hashlib +import json +import re +from typing import Any, Iterable, Mapping + + +def _sha(value: Any) -> str: + return hashlib.sha256(json.dumps(value, sort_keys=True, separators=(",", ":"), + ensure_ascii=False, default=str).encode()).hexdigest() + + +def _text(value: Any, label: str, *, optional: bool = False) -> None: + if (not isinstance(value, str) or (not value and not optional) + or any(c in value for c in ("\0", "\n", "\r"))): + raise ValueError(f"Invalid effect {label}") + + +def _position(value: Any) -> None: + if type(value) is not int or value < 0: + raise ValueError("Effect history position must be a nonnegative integer") + + +_SHA256 = re.compile(r"[a-f0-9]{64}") + + +# --------------------------------------------------------------------------- +# Exact resource references (Wave 3 consumption only) +# --------------------------------------------------------------------------- + +class ResourceKind(str, Enum): + FILESYSTEM = "filesystem" + PROCESS = "process" + BACKGROUND_JOB = "background_job" + OWNED = "owned" + EXTERNAL = "external" + BROWSER_SESSION = "browser_session" + + +@dataclass(frozen=True) +class ResourceRef: + """Historical reference to one exact admitted Wave 3 resource. + + ``location`` identifies where the resource lives (including the identity of + its sealed root/namespace); ``incarnation`` identifies the object observed + there when the reference was taken. Replacement keeps the location and + changes the incarnation, so evidence never transfers to a replacement. + ``snapshot_sha256`` digests the full Wave 3 snapshot for audit. A ref is not + authority: it is not accepted by any dispatcher, resolver or grant. + """ + + kind: ResourceKind + role: str + location: tuple[str, ...] + incarnation: str + snapshot_sha256: str + + def __post_init__(self) -> None: + if not isinstance(self.kind, ResourceKind): + raise ValueError("Unsupported effect resource kind") + _text(self.role, "resource role") + _text(self.incarnation, "resource incarnation", optional=True) + if (not isinstance(self.location, tuple) or len(self.location) < 2 + or any(not isinstance(part, str) or any(c in part for c in ("\0", "\n", "\r")) + for part in self.location) + or self.location[0] != self.kind.value): + raise ValueError("Malformed effect resource location") + if not _SHA256.fullmatch(self.snapshot_sha256 or ""): + raise ValueError("Malformed effect resource snapshot digest") + + @property + def location_key(self) -> str: + return _sha(list(self.location)) + + def same_location(self, other: "ResourceRef") -> bool: + return self.kind is other.kind and self.location == other.location + + def overlaps(self, other: "ResourceRef") -> bool: + """Conservative relevance between two exact references. + + Filesystem relevance is ancestor-or-self within one sealed root + identity: a mutation of ``d/x`` invalidates a listing of ``d`` and a + replacement of ``d`` invalidates observations of ``d/x``. Other kinds + only overlap at the same exact location. No alias discovery is done. + """ + if self.kind is not other.kind: + return False + if self.kind is not ResourceKind.FILESYSTEM: + return self.location == other.location + if self.location[:-1] != other.location[:-1]: + return False + left, right = PurePosixPath(self.location[-1]), PurePosixPath(other.location[-1]) + return left == right or left.is_relative_to(right) or right.is_relative_to(left) + + def to_dict(self) -> dict[str, Any]: + return {"kind": self.kind.value, "role": self.role, "location": list(self.location), + "incarnation": self.incarnation, "snapshot_sha256": self.snapshot_sha256} + + @classmethod + def from_dict(cls, value: Any) -> "ResourceRef": + """Reload a persisted historical reference. This creates no authority.""" + if (not isinstance(value, dict) + or set(value) != {"kind", "role", "location", "incarnation", "snapshot_sha256"} + or not isinstance(value["location"], list)): + raise ValueError("Malformed persisted effect resource reference") + return cls(ResourceKind(value["kind"]), value["role"], tuple(value["location"]), + value["incarnation"], value["snapshot_sha256"]) + + +def resource_ref(resource: Any, role: str) -> ResourceRef: + """Reference an exact typed Wave 3 resource; anything else is refused. + + Browser page/document resources are refused: Wave 3 fails closed for page + authority and Wave 4 must not promote page observations into identity. + """ + from src.agent_runtime import resources as wave3 + if isinstance(resource, wave3.BrowserPageResource): + raise TypeError("Browser page resources are not effect-bindable") + if isinstance(resource, wave3.FilesystemResource): + root = resource.root + location = ("filesystem", root.scope.value, root.owner, root.path, + str(root.identity.device), str(root.identity.inode), resource.path) + chain = [[a.path, a.identity.device, a.identity.inode] for a in resource.ancestors] + identity = resource.identity + incarnation = ("absent:" + _sha(chain) if identity is None else + f"{identity.kind}:{identity.device}:{identity.inode}:" + _sha(chain)) + return ResourceRef(ResourceKind.FILESYSTEM, role, location, incarnation, _sha(resource.to_dict())) + if isinstance(resource, wave3.ProcessResource): + ident = resource.identity + location = ("process", resource.namespace, resource.owner, resource.request_id, resource.thread_id, + str(ident.pid), ident.start_token, resource.role) + return ResourceRef(ResourceKind.PROCESS, role, location, ident.start_token, _sha(resource.to_dict())) + if isinstance(resource, wave3.BackgroundJobResource): + location = ("background_job", resource.namespace, resource.owner, resource.request_id, + resource.thread_id, resource.job_id, resource.generation) + return ResourceRef(ResourceKind.BACKGROUND_JOB, role, location, resource.generation, + _sha(resource.to_dict())) + if isinstance(resource, wave3.OwnedResource): + location = ("owned", resource.namespace, resource.owner, resource.thread_id, + resource.collection, resource.record_id) + return ResourceRef(ResourceKind.OWNED, role, location, resource.revision, _sha(resource.to_dict())) + if isinstance(resource, wave3.ExternalResource): + location = ("external", resource.namespace, resource.owner, resource.endpoint_id, + resource.server_id, resource.tool_id) + return ResourceRef(ResourceKind.EXTERNAL, role, location, resource.incarnation, _sha(resource.to_dict())) + if isinstance(resource, wave3.BrowserSessionResource): + observation = resource.observation + location = ("browser_session", resource.owner, resource.thread_id, observation.session_key) + return ResourceRef(ResourceKind.BROWSER_SESSION, role, location, observation.session_incarnation, + _sha(resource.to_dict())) + raise TypeError("Effect scope requires an exact Wave 3 resource identity") + + +def bound_filesystem_refs(bound: Any) -> tuple[ResourceRef, ...]: + """References for an admitted ``BoundFilesystemOperation``'s exact bindings.""" + from src.agent_runtime.resource_binding import BoundFilesystemOperation + if not isinstance(bound, BoundFilesystemOperation): + raise TypeError("Filesystem effect scope requires a server-owned bound operation") + return tuple(resource_ref(binding.resource, binding.role) for binding in bound.bindings) + + +# --------------------------------------------------------------------------- +# Claims +# --------------------------------------------------------------------------- + +@dataclass(frozen=True) +class OperationRef: + """Final normalized operation reference; not a second normalization API.""" + + tool: str + action: str + input_sha256: str + request_id: str = "" + + def __post_init__(self) -> None: + _text(self.tool, "operation tool") + _text(self.action, "operation action", optional=True) + _text(self.request_id, "operation request", optional=True) + if not _SHA256.fullmatch(self.input_sha256 or ""): + raise ValueError("Malformed operation input digest") + + @classmethod + def from_exact(cls, operation: Any, execution_input: str | None = None, request_id: str = "") -> "OperationRef": + from src.agent_runtime.authority import ExactOperation + if not isinstance(operation, ExactOperation): + raise TypeError("Effect claims require the admitted exact operation") + body = operation.input if execution_input is None else execution_input + return cls(str(operation.tool), str(operation.action or ""), _sha(body), request_id or "") + + def to_dict(self) -> dict[str, Any]: + return {"tool": self.tool, "action": self.action, "input_sha256": self.input_sha256, + "request_id": self.request_id} + + @classmethod + def from_dict(cls, value: Any) -> "OperationRef": + if not isinstance(value, dict) or set(value) != {"tool", "action", "input_sha256", "request_id"}: + raise ValueError("Malformed persisted operation reference") + return cls(**value) + + +class Predicate(str, Enum): + EXISTS = "exists" + ABSENT = "absent" + CONTENT_SHA256 = "content_sha256" + # The observed content digest differs from ``expected`` (the pre-state). + CONTENT_CHANGED = "content_changed" + + +_PREDICATE_KINDS = { + Predicate.EXISTS: {ResourceKind.FILESYSTEM, ResourceKind.OWNED, ResourceKind.EXTERNAL}, + Predicate.ABSENT: {ResourceKind.FILESYSTEM, ResourceKind.OWNED, ResourceKind.EXTERNAL}, + Predicate.CONTENT_SHA256: {ResourceKind.FILESYSTEM, ResourceKind.OWNED, ResourceKind.EXTERNAL}, + Predicate.CONTENT_CHANGED: {ResourceKind.FILESYSTEM, ResourceKind.OWNED, ResourceKind.EXTERNAL}, +} + + +@dataclass(frozen=True) +class Postcondition: + """An explicit requested post-state predicate on one exact claimed target.""" + + target: ResourceRef + predicate: Predicate + expected: str = "" + + def __post_init__(self) -> None: + if not isinstance(self.target, ResourceRef) or not isinstance(self.predicate, Predicate): + raise ValueError("Malformed postcondition") + if self.target.kind not in _PREDICATE_KINDS[self.predicate]: + raise ValueError("Predicate is not supported for this resource kind") + needs_digest = self.predicate in {Predicate.CONTENT_SHA256, Predicate.CONTENT_CHANGED} + if needs_digest != bool(_SHA256.fullmatch(self.expected or "")) or (not needs_digest and self.expected): + raise ValueError("Malformed postcondition expectation") + + def to_dict(self) -> dict[str, Any]: + return {"target": self.target.to_dict(), "predicate": self.predicate.value, "expected": self.expected} + + @classmethod + def from_dict(cls, value: Any) -> "Postcondition": + if not isinstance(value, dict) or set(value) != {"target", "predicate", "expected"}: + raise ValueError("Malformed persisted postcondition") + return cls(ResourceRef.from_dict(value["target"]), Predicate(value["predicate"]), value["expected"]) + + +@dataclass(frozen=True) +class EffectClaim: + """Server-owned claim, persisted before backend invocation. + + The claim states intent and scope; it is not evidence that dispatch, the + backend operation, or any mutation happened. ``impact_scope`` holds the + exact admitted bindings the operation may change; empty means unknown + scope, never no impact. ``dependencies`` are resources the predicate + relies on without being mutation targets. + """ + + effect_id: str + run_id: str + action_id: str + sequence: int + operation: OperationRef + impact_scope: tuple[ResourceRef, ...] = () + dependencies: tuple[ResourceRef, ...] = () + obligations: tuple[Postcondition, ...] = () + parent_run_id: str = "" + external: bool = False + + def __post_init__(self) -> None: + for name in ("effect_id", "run_id", "action_id"): + _text(getattr(self, name), name) + _text(self.parent_run_id, "parent run", optional=True) + _position(self.sequence) + if not isinstance(self.operation, OperationRef) or type(self.external) is not bool: + raise ValueError("Malformed effect claim") + for name in ("impact_scope", "dependencies"): + refs = getattr(self, name) + if not isinstance(refs, tuple) or any(not isinstance(r, ResourceRef) for r in refs): + raise ValueError("Effect scope must be exact resource references") + if (not isinstance(self.obligations, tuple) + or any(not isinstance(o, Postcondition) for o in self.obligations)): + raise ValueError("Malformed effect obligations") + for obligation in self.obligations: + if not any(obligation.target == ref for ref in self.impact_scope): + raise ValueError("Postcondition target must be a claimed impact binding") + + @property + def unknown_scope(self) -> bool: + return not self.impact_scope + + def to_dict(self) -> dict[str, Any]: + return {"effect_id": self.effect_id, "run_id": self.run_id, "action_id": self.action_id, + "sequence": self.sequence, "operation": self.operation.to_dict(), + "impact_scope": [r.to_dict() for r in self.impact_scope], + "dependencies": [r.to_dict() for r in self.dependencies], + "obligations": [o.to_dict() for o in self.obligations], + "parent_run_id": self.parent_run_id, "external": self.external} + + @classmethod + def from_dict(cls, value: Any) -> "EffectClaim": + keys = {"effect_id", "run_id", "action_id", "sequence", "operation", "impact_scope", + "dependencies", "obligations", "parent_run_id", "external"} + if not isinstance(value, dict) or set(value) != keys or any( + not isinstance(value[k], list) for k in ("impact_scope", "dependencies", "obligations")): + raise ValueError("Malformed persisted effect claim") + return cls(value["effect_id"], value["run_id"], value["action_id"], value["sequence"], + OperationRef.from_dict(value["operation"]), + tuple(ResourceRef.from_dict(r) for r in value["impact_scope"]), + tuple(ResourceRef.from_dict(r) for r in value["dependencies"]), + tuple(Postcondition.from_dict(o) for o in value["obligations"]), + value["parent_run_id"], value["external"]) + + +# --------------------------------------------------------------------------- +# Outcomes +# --------------------------------------------------------------------------- + +class ExecutionOutcome(str, Enum): + NOT_EXECUTED = "not_executed" # refused before backend invocation + ATTEMPTED = "attempted" # claimed; no settled outcome yet + REPORTED_SUCCESS = "reported_success" # executor reported success; not post-state + FAILED = "failed" + TIMED_OUT = "timed_out" + CANCELLED = "cancelled" + RUNNING = "running" # admitted/background; not completed work + INTERRUPTED = "interrupted" # unknown: lost, crashed or replayed + + +class Impact(str, Enum): + NONE = "none" # known no-op: the backend was never invoked + POSSIBLE = "possible" # may have changed state, including partially + CHANGED = "changed" # a trusted before/after capture differs + + +class CleanupState(str, Enum): + NOT_APPLICABLE = "not_applicable" + VERIFIED = "verified" + FAILED = "failed" + UNKNOWN = "unknown" + + +_SETTLED = {ExecutionOutcome.NOT_EXECUTED, ExecutionOutcome.REPORTED_SUCCESS, ExecutionOutcome.FAILED, + ExecutionOutcome.TIMED_OUT, ExecutionOutcome.CANCELLED, ExecutionOutcome.INTERRUPTED} + + +@dataclass(frozen=True) +class ProducerFacts: + """Bounded typed producer facts; arbitrary returned data is never kept. + + These are execution/lifecycle facts reported by a server producer. None of + them is a post-state observation. + """ + + exit_code: int | None = None + timed_out: bool = False + output_truncated: bool = False + failure_kind: str = "" + job_state: str = "" + remote_acknowledged: bool = False + external: bool = False + + def __post_init__(self) -> None: + if self.exit_code is not None and type(self.exit_code) is not int: + raise ValueError("Malformed producer exit code") + for name in ("timed_out", "output_truncated", "remote_acknowledged", "external"): + if type(getattr(self, name)) is not bool: + raise ValueError("Malformed producer flag") + for name in ("failure_kind", "job_state"): + value = getattr(self, name) + _text(value, name, optional=True) + if len(value) > 64 or (value and not re.fullmatch(r"[a-z0-9_.:-]+", value)): + raise ValueError("Malformed producer label") + + def to_dict(self) -> dict[str, Any]: + return {"exit_code": self.exit_code, "timed_out": self.timed_out, + "output_truncated": self.output_truncated, "failure_kind": self.failure_kind, + "job_state": self.job_state, "remote_acknowledged": self.remote_acknowledged, + "external": self.external} + + @classmethod + def from_dict(cls, value: Any) -> "ProducerFacts": + if not isinstance(value, dict) or set(value) != set(cls.__dataclass_fields__): + raise ValueError("Malformed persisted producer facts") + return cls(**value) + + +def _label(value: Any) -> str: + text = value.strip().lower() if isinstance(value, str) else "" + return text if len(text) <= 64 and re.fullmatch(r"[a-z0-9_.:-]+", text) else "" + + +def producer_facts(result: Any) -> ProducerFacts: + """Project a dispatcher result into typed facts without trusting its shape. + + Only exact scalar types are copied. Anything else becomes the default, so a + forged or malformed dictionary can only lose information, not add trust. + """ + if not isinstance(result, Mapping): + return ProducerFacts() + code = result.get("exit_code") + containment = result.get("containment") + external = isinstance(containment, Mapping) and containment.get("external") is True + job = result.get("status") if isinstance(result.get("job_id"), str) else "" + return ProducerFacts( + exit_code=code if type(code) is int else None, + timed_out=result.get("timed_out") is True or _label(result.get("failure_kind")) == "timeout", + output_truncated=result.get("output_truncated") is True or result.get("truncated") is True, + failure_kind=_label(result.get("failure_kind")), + job_state=_label(job), + external=external, + ) + + +@dataclass(frozen=True) +class EffectOutcome: + """Append-only execution outcome for one claim. + + ``impact`` must not claim no change for anything that reached a backend. + ``cleanup`` is recorded separately: cleanup success is not business-effect + success and cleanup failure does not erase an achieved effect. + """ + + effect_id: str + sequence: int + execution: ExecutionOutcome + impact: Impact + facts: ProducerFacts = ProducerFacts() + cleanup: CleanupState = CleanupState.NOT_APPLICABLE + execution_id: str = "" + replayed: bool = False + + def __post_init__(self) -> None: + _text(self.effect_id, "effect identifier") + _text(self.execution_id, "execution identifier", optional=True) + _position(self.sequence) + if (not isinstance(self.execution, ExecutionOutcome) or not isinstance(self.impact, Impact) + or not isinstance(self.facts, ProducerFacts) or not isinstance(self.cleanup, CleanupState) + or type(self.replayed) is not bool): + raise ValueError("Malformed effect outcome") + if self.execution is ExecutionOutcome.ATTEMPTED: + raise ValueError("ATTEMPTED is derived from a claim without an outcome") + if (self.impact is Impact.NONE) != (self.execution is ExecutionOutcome.NOT_EXECUTED): + raise ValueError("Only a refused, never-invoked operation is a known no-op") + if self.execution is ExecutionOutcome.NOT_EXECUTED and self.execution_id: + raise ValueError("A refused operation has no execution identity") + + def to_dict(self) -> dict[str, Any]: + return {"effect_id": self.effect_id, "sequence": self.sequence, "execution": self.execution.value, + "impact": self.impact.value, "facts": self.facts.to_dict(), "cleanup": self.cleanup.value, + "execution_id": self.execution_id, "replayed": self.replayed} + + @classmethod + def from_dict(cls, value: Any) -> "EffectOutcome": + if not isinstance(value, dict) or set(value) != set(cls.__dataclass_fields__): + raise ValueError("Malformed persisted effect outcome") + return cls(value["effect_id"], value["sequence"], ExecutionOutcome(value["execution"]), + Impact(value["impact"]), ProducerFacts.from_dict(value["facts"]), + CleanupState(value["cleanup"]), value["execution_id"], value["replayed"]) + + +# --------------------------------------------------------------------------- +# Observations +# --------------------------------------------------------------------------- + +class ObservationMechanism(str, Enum): + FILESYSTEM_READ = "filesystem_read" # admitted read of the exact binding + OWNED_RECORD_READ = "owned_record_read" # admitted owner-scoped readback + REMOTE_READBACK = "remote_readback" # admitted independent remote query + PROCESS_OWNERSHIP = "process_ownership" # lifecycle owner's verdict + JOB_STATE = "job_state" # background job record transition + BROWSER_SESSION = "browser_session" # session lifecycle metadata only + # The following are never post-state verification. + EXECUTION_RECEIPT = "execution_receipt" + REMOTE_ACKNOWLEDGEMENT = "remote_acknowledgement" + + +class Coverage(str, Enum): + COMPLETE = "complete" + PARTIAL = "partial" + + +# Mechanisms able to decide a postcondition for each resource kind. Process, +# job and browser-session observations are lifecycle facts: they can make +# earlier evidence stale but cannot verify a file/record/remote predicate. +_VERIFYING = { + ResourceKind.FILESYSTEM: {ObservationMechanism.FILESYSTEM_READ}, + ResourceKind.OWNED: {ObservationMechanism.OWNED_RECORD_READ}, + ResourceKind.EXTERNAL: {ObservationMechanism.REMOTE_READBACK}, +} +_ADMITTED_READS = {ObservationMechanism.FILESYSTEM_READ, ObservationMechanism.OWNED_RECORD_READ, + ObservationMechanism.REMOTE_READBACK} + + +@dataclass(frozen=True) +class Observation: + """State seen through one mechanism for one exact resource. + + ``exists``/``content_sha256`` are what the mechanism saw; ``None``/empty + means not observed. A PARTIAL observation (offset/limit/truncated read, + listing, existence-only probe) never decides a whole-content predicate. + Admitted reads must name the journal action that performed them. + """ + + observation_id: str + sequence: int + resource: ResourceRef + mechanism: ObservationMechanism + coverage: Coverage + source_action_id: str = "" + source_execution_id: str = "" + exists: bool | None = None + content_sha256: str = "" + evidence_event_id: str = "" + + def __post_init__(self) -> None: + _text(self.observation_id, "observation identifier") + for name in ("source_action_id", "source_execution_id", "evidence_event_id"): + _text(getattr(self, name), name, optional=True) + _position(self.sequence) + if (not isinstance(self.resource, ResourceRef) or not isinstance(self.mechanism, ObservationMechanism) + or not isinstance(self.coverage, Coverage) + or (self.exists is not None and type(self.exists) is not bool)): + raise ValueError("Malformed observation") + if self.content_sha256 and (not _SHA256.fullmatch(self.content_sha256) or self.exists is not True): + raise ValueError("Malformed observed content digest") + if self.mechanism in _ADMITTED_READS and not self.source_action_id: + raise ValueError("Readback observations require the admitted action that performed them") + + def to_dict(self) -> dict[str, Any]: + return {"observation_id": self.observation_id, "sequence": self.sequence, + "resource": self.resource.to_dict(), "mechanism": self.mechanism.value, + "coverage": self.coverage.value, "source_action_id": self.source_action_id, + "source_execution_id": self.source_execution_id, "exists": self.exists, + "content_sha256": self.content_sha256, "evidence_event_id": self.evidence_event_id} + + @classmethod + def from_dict(cls, value: Any) -> "Observation": + if not isinstance(value, dict) or set(value) != set(cls.__dataclass_fields__): + raise ValueError("Malformed persisted observation") + return cls(**{**value, "resource": ResourceRef.from_dict(value["resource"]), + "mechanism": ObservationMechanism(value["mechanism"]), + "coverage": Coverage(value["coverage"])}) + + +def predicate_holds(postcondition: Postcondition, observation: Observation) -> bool | None: + """Decide one predicate from one observation; ``None`` means undecidable. + + The check is performed here from the observed state, so no adapter can + attest verification by labelling an unrelated read. + """ + target = postcondition.target + if (not observation.resource.same_location(target) + or observation.mechanism not in _VERIFYING.get(target.kind, set())): + return None + predicate = postcondition.predicate + if predicate is Predicate.ABSENT: + return None if observation.exists is None else not observation.exists + if predicate is Predicate.EXISTS: + return observation.exists + if observation.exists is False: + return False + if observation.coverage is not Coverage.COMPLETE or not observation.content_sha256: + return None + if predicate is Predicate.CONTENT_SHA256: + return observation.content_sha256 == postcondition.expected + return observation.content_sha256 != postcondition.expected + + +# --------------------------------------------------------------------------- +# History, invalidation and freshness +# --------------------------------------------------------------------------- + +class Freshness(str, Enum): + FRESH = "fresh" + STALE = "stale" # a later possible mutation or replacement overlaps + UNSETTLED = "unsettled" # an overlapping effect was still in flight + + +@dataclass(frozen=True) +class EffectHistory: + """An immutable, totally ordered view of one effect log. + + Sequences are unique positions in one log. Duplicate positions are rejected + rather than ordered arbitrarily. + """ + + claims: tuple[EffectClaim, ...] = () + outcomes: tuple[EffectOutcome, ...] = () + observations: tuple[Observation, ...] = () + + def __post_init__(self) -> None: + positions = [r.sequence for r in (*self.claims, *self.outcomes, *self.observations)] + if len(positions) != len(set(positions)): + raise ValueError("Effect history positions must be unique") + ids = [c.effect_id for c in self.claims] + if len(ids) != len(set(ids)): + raise ValueError("Effect claims must have unique identifiers") + claim_at = {c.effect_id: c.sequence for c in self.claims} + settled: set[str] = set() + for outcome in sorted(self.outcomes, key=lambda o: o.sequence): + if outcome.effect_id not in claim_at or outcome.sequence <= claim_at[outcome.effect_id]: + raise ValueError("Outcome must follow its claim in one history") + # A RUNNING effect may later settle (background continuation or + # replay interruption); a settled outcome is never replaced. + if outcome.effect_id in settled: + raise ValueError("A settled effect outcome cannot be replaced") + if outcome.execution is not ExecutionOutcome.RUNNING: + settled.add(outcome.effect_id) + + def claim(self, effect_id: str) -> EffectClaim | None: + return next((c for c in self.claims if c.effect_id == effect_id), None) + + def latest_outcome(self, effect_id: str, before: int | None = None) -> EffectOutcome | None: + matching = [o for o in self.outcomes if o.effect_id == effect_id + and (before is None or o.sequence < before)] + return max(matching, key=lambda o: o.sequence) if matching else None + + def execution(self, effect_id: str, before: int | None = None) -> ExecutionOutcome: + outcome = self.latest_outcome(effect_id, before) + return ExecutionOutcome.ATTEMPTED if outcome is None else outcome.execution + + +def _claim_touches(claim: EffectClaim, resource: ResourceRef) -> bool: + return claim.unknown_scope or any(ref.overlaps(resource) for ref in claim.impact_scope) + + +def invalidated_by(observation: Observation, history: EffectHistory) -> tuple[str, ...]: + """Identifiers of later records that make ``observation`` stale. + + Any later claim that may touch the resource invalidates it once the claim + exists (it may already be executing), unless it settled as a known no-op. + A later observation of the same location with a different incarnation + reveals replacement. Execution receipts are never invalidated: they remain + historical execution facts. + """ + if observation.mechanism in {ObservationMechanism.EXECUTION_RECEIPT, + ObservationMechanism.REMOTE_ACKNOWLEDGEMENT}: + return () + reasons: list[str] = [] + for claim in history.claims: + if claim.sequence <= observation.sequence or not _claim_touches(claim, observation.resource): + continue + outcome = history.latest_outcome(claim.effect_id) + if outcome is not None and outcome.impact is Impact.NONE: + continue + reasons.append(claim.effect_id) + for later in history.observations: + if (later.sequence > observation.sequence and later.resource.same_location(observation.resource) + and later.resource.incarnation != observation.resource.incarnation): + reasons.append(later.observation_id) + return tuple(dict.fromkeys(reasons)) + + +def freshness(observation: Observation, history: EffectHistory) -> Freshness: + if invalidated_by(observation, history): + return Freshness.STALE + for claim in history.claims: + if claim.sequence < observation.sequence and _claim_touches(claim, observation.resource): + state = history.execution(claim.effect_id, before=observation.sequence) + if state in {ExecutionOutcome.ATTEMPTED, ExecutionOutcome.RUNNING}: + return Freshness.UNSETTLED + return Freshness.FRESH + + +# --------------------------------------------------------------------------- +# Verification +# --------------------------------------------------------------------------- + +class EffectVerdict(str, Enum): + NOT_EXECUTED = "not_executed" + PENDING = "pending" # attempted/running; not settled + VERIFIED = "verified" # reported success + fresh matching post-state + STATE_OBSERVED = "state_observed" # matching post-state; causality unknown + UNVERIFIED = "unverified" # no adequate fresh evidence + CONTRADICTED = "contradicted" # latest fresh check shows the predicate false + FAILED = "failed" # execution failed; never effect success + + +_VERDICT_RANK = {EffectVerdict.FAILED: 0, EffectVerdict.CONTRADICTED: 1, EffectVerdict.PENDING: 2, + EffectVerdict.UNVERIFIED: 3, EffectVerdict.NOT_EXECUTED: 4, + EffectVerdict.STATE_OBSERVED: 5, EffectVerdict.VERIFIED: 6} + + +@dataclass(frozen=True) +class EffectAssessment: + effect_id: str + action_id: str + execution: ExecutionOutcome + impact: Impact | None + verdict: EffectVerdict + reason: str + cleanup: CleanupState = CleanupState.NOT_APPLICABLE + observation_ids: tuple[str, ...] = () + targets: tuple[ResourceRef, ...] = () + + @property + def unresolved_impact(self) -> bool: + """Resources may have changed in a way no fresh evidence has settled.""" + return (self.impact is not Impact.NONE + and self.execution is not ExecutionOutcome.REPORTED_SUCCESS + and self.verdict not in {EffectVerdict.STATE_OBSERVED, EffectVerdict.CONTRADICTED}) + + def to_dict(self) -> dict[str, Any]: + return {"effect_id": self.effect_id, "action_id": self.action_id, "execution": self.execution.value, + "impact": None if self.impact is None else self.impact.value, "verdict": self.verdict.value, + "reason": self.reason, "cleanup": self.cleanup.value, + "observation_ids": list(self.observation_ids), + "targets": [t.to_dict() for t in self.targets]} + + +def _assess_obligation(claim: EffectClaim, settled: EffectOutcome, obligation: Postcondition, + history: EffectHistory) -> tuple[EffectVerdict, str, str]: + candidates = [o for o in history.observations + if o.sequence > settled.sequence and o.resource.same_location(obligation.target) + and o.mechanism in _VERIFYING.get(obligation.target.kind, set())] + if not candidates: + return EffectVerdict.UNVERIFIED, "no authorized post-settlement observation of the target", "" + # The newest check wins. A newer partial or failed check never falls back + # to an earlier complete one. + latest = max(candidates, key=lambda o: o.sequence) + state = freshness(latest, history) + if state is not Freshness.FRESH: + return EffectVerdict.UNVERIFIED, f"the latest target observation is {state.value}", latest.observation_id + holds = predicate_holds(obligation, latest) + if holds is None: + return EffectVerdict.UNVERIFIED, "the latest observation does not decide the postcondition", latest.observation_id + if not holds: + return EffectVerdict.CONTRADICTED, "the latest fresh observation contradicts the postcondition", latest.observation_id + execution = settled.execution + if execution is ExecutionOutcome.FAILED: + return EffectVerdict.FAILED, "execution failed; matching state is not attributed to it", latest.observation_id + if execution is ExecutionOutcome.REPORTED_SUCCESS: + return EffectVerdict.VERIFIED, "fresh authorized observation matches the postcondition", latest.observation_id + return (EffectVerdict.STATE_OBSERVED, + "state matches, but this execution's outcome is unknown; causality is not established", + latest.observation_id) + + +def assess(claim: EffectClaim, history: EffectHistory) -> EffectAssessment: + """Derive a claim's verdict from the append-only history.""" + settled = history.latest_outcome(claim.effect_id) + targets = tuple(o.target for o in claim.obligations) + if settled is None: + return EffectAssessment(claim.effect_id, claim.action_id, ExecutionOutcome.ATTEMPTED, None, + EffectVerdict.PENDING, "no settled execution outcome", targets=targets) + base = dict(effect_id=claim.effect_id, action_id=claim.action_id, execution=settled.execution, + impact=settled.impact, cleanup=settled.cleanup, targets=targets) + if settled.execution is ExecutionOutcome.NOT_EXECUTED: + return EffectAssessment(**base, verdict=EffectVerdict.NOT_EXECUTED, reason="refused before invocation") + if settled.execution is ExecutionOutcome.RUNNING: + return EffectAssessment(**base, verdict=EffectVerdict.PENDING, + reason="background execution has not settled") + if not claim.obligations: + verdict = EffectVerdict.FAILED if settled.execution is ExecutionOutcome.FAILED else EffectVerdict.UNVERIFIED + return EffectAssessment(**base, verdict=verdict, reason="no explicit postcondition obligation") + results = [_assess_obligation(claim, settled, o, history) for o in claim.obligations] + worst = min(results, key=lambda r: _VERDICT_RANK[r[0]]) + if settled.execution is ExecutionOutcome.FAILED and worst[0] is not EffectVerdict.CONTRADICTED: + worst = (EffectVerdict.FAILED, worst[1] if worst[0] is EffectVerdict.FAILED else + "execution failed and may have partially changed the target", worst[2]) + return EffectAssessment(**base, verdict=worst[0], reason=worst[1], + observation_ids=tuple(dict.fromkeys(r[2] for r in results if r[2]))) + + +def assess_all(history: EffectHistory) -> tuple[EffectAssessment, ...]: + return tuple(assess(claim, history) for claim in sorted(history.claims, key=lambda c: c.sequence)) + + +def replay_interrupted(history: EffectHistory, next_sequence: int) -> tuple[EffectOutcome, ...]: + """Outcomes to append for claims that never settled before a reload. + + Unknown remains unknown: the backend may or may not have been invoked, so + impact is POSSIBLE. Running background effects are left to their own + lifecycle owner and are not converted here. + """ + _position(next_sequence) + pending = [c for c in sorted(history.claims, key=lambda c: c.sequence) + if history.latest_outcome(c.effect_id) is None] + return tuple(EffectOutcome(c.effect_id, next_sequence + i, ExecutionOutcome.INTERRUPTED, + Impact.POSSIBLE, replayed=True) for i, c in enumerate(pending)) diff --git a/tests/test_effects_foundation.py b/tests/test_effects_foundation.py new file mode 100644 index 000000000..7d0976243 --- /dev/null +++ b/tests/test_effects_foundation.py @@ -0,0 +1,466 @@ +"""Wave 4 effect semantics against exact Wave 3 resource identities.""" +from __future__ import annotations + +import hashlib +import os + +import pytest + +from src.agent_runtime import effects as fx +from src.agent_runtime.authority import ExactOperation +from src.agent_runtime.resources import ( + BrowserPageResource, ExternalResource, FilesystemResource, FilesystemRoot, OwnedResource, ProcessResource, +) +from src.process_lifecycle import ProcessIdentity + + +def sha(data: bytes) -> str: + return hashlib.sha256(data).hexdigest() + + +@pytest.fixture +def root(tmp_path): + workspace = tmp_path / "ws" + workspace.mkdir() + return FilesystemRoot.seal(str(workspace)) + + +def fs_ref(root, name, role="target", *, missing=False): + resource = FilesystemResource.resolve(root, os.path.join(root.path, name), allow_missing=missing) + return fx.resource_ref(resource, role) + + +def op(tool="write_file"): + return fx.OperationRef("write_file" if tool == "write_file" else tool, "", "0" * 64) + + +class Log: + """Test helper assigning one total order, like the runtime effect log.""" + + def __init__(self): + self.claims, self.outcomes, self.observations, self.seq = [], [], [], 0 + + def _next(self): + self.seq += 1 + return self.seq + + def claim(self, scope=(), obligations=(), effect_id=None, dependencies=()): + claim = fx.EffectClaim(effect_id or f"e{len(self.claims) + 1}", "run", f"a{len(self.claims) + 1}", + self._next(), op(), tuple(scope), tuple(dependencies), tuple(obligations)) + self.claims.append(claim) + return claim + + def outcome(self, claim, execution=fx.ExecutionOutcome.REPORTED_SUCCESS, impact=fx.Impact.POSSIBLE, **kw): + outcome = fx.EffectOutcome(claim.effect_id, self._next(), execution, impact, + execution_id="" if impact is fx.Impact.NONE else claim.action_id + ":x", **kw) + self.outcomes.append(outcome) + return outcome + + def observe(self, resource, *, exists=True, digest="", coverage=fx.Coverage.COMPLETE, + mechanism=fx.ObservationMechanism.FILESYSTEM_READ): + observation = fx.Observation(f"o{len(self.observations) + 1}", self._next(), resource, mechanism, coverage, + source_action_id="read-action", exists=exists, content_sha256=digest) + self.observations.append(observation) + return observation + + @property + def history(self): + return fx.EffectHistory(tuple(self.claims), tuple(self.outcomes), tuple(self.observations)) + + +def content(target, body=b"hello"): + return fx.Postcondition(target, fx.Predicate.CONTENT_SHA256, sha(body)) + + +# -- exact resource references -------------------------------------------- + +def test_refs_only_accept_typed_wave3_resources(root): + for forged in ({"kind": "filesystem", "path": "/etc/passwd"}, "/workspace/a.txt", 1234, + ("filesystem", "x")): + with pytest.raises(TypeError): + fx.resource_ref(forged, "target") + page = object.__new__(BrowserPageResource) + with pytest.raises(TypeError, match="page"): + fx.resource_ref(page, "target") + + +def test_filesystem_replacement_changes_incarnation_not_location(root): + path = os.path.join(root.path, "a.txt") + with open(path, "w") as handle: + handle.write("one") + before = fs_ref(root, "a.txt") + os.replace(_write(root, "tmp", "two"), path) + after = fs_ref(root, "a.txt") + assert before.same_location(after) + assert before.incarnation != after.incarnation + assert fs_ref(root, "missing.txt", missing=True).incarnation.startswith("absent:") + + +def _write(root, name, text): + path = os.path.join(root.path, name) + with open(path, "w") as handle: + handle.write(text) + return path + + +def test_filesystem_overlap_is_ancestor_or_self_within_one_sealed_root(root, tmp_path): + os.mkdir(os.path.join(root.path, "d")) + _write(root, "d/x.txt", "x") + _write(root, "dx.txt", "x") + directory = fs_ref(root, "d", "search_root") + child = fs_ref(root, "d/x.txt") + sibling = fs_ref(root, "dx.txt") + assert directory.overlaps(child) and child.overlaps(directory) + assert not sibling.overlaps(directory) + other_dir = tmp_path / "other" + other_dir.mkdir() + (other_dir / "d").mkdir() + other = FilesystemRoot.seal(str(other_dir)) + assert not fx.resource_ref(FilesystemResource.resolve(other, str(other_dir / "d")), "target").overlaps(directory) + + +def test_process_pid_reuse_is_a_different_location(): + first = ProcessResource("native:containment", "u", "r", "t", ProcessIdentity(4242, "boot:1:100", None), "leader") + reused = ProcessResource("native:containment", "u", "r", "t", ProcessIdentity(4242, "boot:1:999", None), "leader") + a, b = fx.resource_ref(first, "subject"), fx.resource_ref(reused, "subject") + assert not a.same_location(b) and not a.overlaps(b) + + +def test_owned_revision_is_incarnation_and_external_never_contained(): + v1 = fx.resource_ref(OwnedResource("notes", "u", "t", "notes", "n1", "rev-1"), "target") + v2 = fx.resource_ref(OwnedResource("notes", "u", "t", "notes", "n1", "rev-2"), "target") + assert v1.same_location(v2) and v1.incarnation != v2.incarnation + remote = fx.resource_ref(ExternalResource("mcp", "ep", "srv", "tool", "inc-1"), "target") + assert remote.kind is fx.ResourceKind.EXTERNAL + + +def test_ref_round_trip_is_historical_and_strict(root): + ref = fs_ref(root, "a.txt", missing=True) + assert fx.ResourceRef.from_dict(ref.to_dict()) == ref + with pytest.raises(ValueError): + fx.ResourceRef.from_dict({**ref.to_dict(), "extra": 1}) + with pytest.raises(ValueError): + fx.ResourceRef.from_dict({**ref.to_dict(), "location": ["owned", "x"]}) + + +def test_operation_ref_requires_admitted_exact_operation(): + with pytest.raises(TypeError): + fx.OperationRef.from_exact({"tool": "write_file"}) + exact = ExactOperation.normalize("write_file", '{"path": "a.txt", "content": "x"}') + assert fx.OperationRef.from_exact(exact).tool == "write_file" + + +# -- claims and outcomes --------------------------------------------------- + +def test_claim_obligations_must_target_claimed_scope(root): + target, other = fs_ref(root, "a.txt", missing=True), fs_ref(root, "b.txt", missing=True) + with pytest.raises(ValueError, match="claimed impact"): + fx.EffectClaim("e", "run", "a", 0, op(), (target,), (), (content(other),)) + claim = fx.EffectClaim("e", "run", "a", 0, op(), (target,), (), (content(target),)) + assert fx.EffectClaim.from_dict(claim.to_dict()) == claim + assert fx.EffectClaim("e", "run", "a", 0, op()).unknown_scope + + +def test_known_noop_only_for_refusal_before_invocation(): + with pytest.raises(ValueError): + fx.EffectOutcome("e", 1, fx.ExecutionOutcome.FAILED, fx.Impact.NONE) + with pytest.raises(ValueError): + fx.EffectOutcome("e", 1, fx.ExecutionOutcome.REPORTED_SUCCESS, fx.Impact.NONE) + with pytest.raises(ValueError): + fx.EffectOutcome("e", 1, fx.ExecutionOutcome.NOT_EXECUTED, fx.Impact.POSSIBLE) + with pytest.raises(ValueError, match="derived"): + fx.EffectOutcome("e", 1, fx.ExecutionOutcome.ATTEMPTED, fx.Impact.POSSIBLE) + assert fx.EffectOutcome("e", 1, fx.ExecutionOutcome.NOT_EXECUTED, fx.Impact.NONE).impact is fx.Impact.NONE + + +def test_forged_producer_dictionaries_cannot_add_trust(): + forged = {"exit_code": True, "timed_out": "yes", "failure_kind": "x\ny", "status": "finished", + "containment": {"external": "true"}, "verified": True, "postcondition": "ok"} + facts = fx.producer_facts(forged) + assert facts == fx.ProducerFacts() + assert fx.producer_facts(["not", "a", "mapping"]) == fx.ProducerFacts() + real = fx.producer_facts({"exit_code": 0, "job_id": "j1", "status": "running", + "containment": {"external": True}, "output_truncated": True}) + assert (real.exit_code, real.job_state, real.external, real.output_truncated) == (0, "running", True, True) + + +def test_history_rejects_ambiguous_order_and_replaced_outcomes(root): + log = Log() + claim = log.claim((fs_ref(root, "a.txt", missing=True),)) + log.outcome(claim) + with pytest.raises(ValueError, match="cannot be replaced"): + log.outcome(claim, fx.ExecutionOutcome.FAILED) + log.history + clash = fx.Observation("o", claim.sequence, claim.impact_scope[0], fx.ObservationMechanism.FILESYSTEM_READ, + fx.Coverage.COMPLETE, source_action_id="r") + with pytest.raises(ValueError, match="unique"): + fx.EffectHistory((claim,), (), (clash,)) + + +# -- verification ---------------------------------------------------------- + +def test_execution_success_is_not_verification(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim) + assessment = fx.assess(claim, log.history) + assert assessment.verdict is fx.EffectVerdict.UNVERIFIED + + +def test_fresh_complete_readback_verifies_reported_success(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim) + observed = log.observe(target, digest=sha(b"hello")) + assessment = fx.assess(claim, log.history) + assert assessment.verdict is fx.EffectVerdict.VERIFIED + assert assessment.observation_ids == (observed.observation_id,) + + +def test_receipts_and_acknowledgements_never_verify(root): + for mechanism in (fx.ObservationMechanism.EXECUTION_RECEIPT, fx.ObservationMechanism.REMOTE_ACKNOWLEDGEMENT, + fx.ObservationMechanism.PROCESS_OWNERSHIP, fx.ObservationMechanism.JOB_STATE): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim) + log.observe(target, digest=sha(b"hello"), mechanism=mechanism) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.UNVERIFIED + + +def test_independent_remote_readback_differs_from_acknowledgement(): + remote = fx.resource_ref(ExternalResource("mcp", "ep", "srv", "tool", "inc"), "target") + log = Log() + claim = log.claim((remote,), (fx.Postcondition(remote, fx.Predicate.EXISTS),)) + log.outcome(claim, facts=fx.ProducerFacts(exit_code=0, remote_acknowledged=True, external=True)) + log.observe(remote, mechanism=fx.ObservationMechanism.REMOTE_ACKNOWLEDGEMENT) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.UNVERIFIED + log.observe(remote, mechanism=fx.ObservationMechanism.REMOTE_READBACK) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.VERIFIED + + +def test_verifier_before_mutation_does_not_count(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + log.observe(target, digest=sha(b"hello")) + claim = log.claim((target,), (content(target),)) + log.outcome(claim) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.UNVERIFIED + + +def test_observation_while_effect_in_flight_is_unsettled(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + early = log.observe(target, digest=sha(b"hello")) + log.outcome(claim) + assert fx.freshness(early, log.history) is fx.Freshness.UNSETTLED + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.UNVERIFIED + + +def test_stale_evidence_after_later_mutation_is_preserved_but_stale(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim) + observed = log.observe(target, digest=sha(b"hello")) + later = log.claim((target,)) + log.outcome(later, fx.ExecutionOutcome.FAILED) + history = log.history + assert observed in history.observations # history is never rewritten + assert fx.invalidated_by(observed, history) == (later.effect_id,) + assert fx.assess(claim, history).verdict is fx.EffectVerdict.UNVERIFIED + + +def test_concurrent_mutation_between_effect_and_verification(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim) + racing = log.claim((target,)) + log.observe(target, digest=sha(b"hello")) + log.outcome(racing) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.UNVERIFIED + + +def test_unknown_mutation_scope_invalidates_everything_earlier(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim) + observed = log.observe(target, digest=sha(b"hello")) + unknown = log.claim(()) + log.outcome(unknown, fx.ExecutionOutcome.INTERRUPTED) + assert fx.invalidated_by(observed, log.history) == (unknown.effect_id,) + + +def test_refused_operation_is_a_known_noop_and_preserves_freshness(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim) + log.observe(target, digest=sha(b"hello")) + refused = log.claim((target,)) + log.outcome(refused, fx.ExecutionOutcome.NOT_EXECUTED, fx.Impact.NONE) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.VERIFIED + assert fx.assess(refused, log.history).verdict is fx.EffectVerdict.NOT_EXECUTED + + +def test_unrelated_resource_mutation_does_not_invalidate(root): + log = Log() + target, other = fs_ref(root, "a.txt", missing=True), fs_ref(root, "b.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim) + log.observe(target, digest=sha(b"hello")) + log.outcome(log.claim((other,))) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.VERIFIED + + +def test_replacement_revealed_by_later_observation_makes_earlier_stale(root): + _write(root, "a.txt", "hello") + original = fs_ref(root, "a.txt") + log = Log() + claim = log.claim((original,), (content(original),)) + log.outcome(claim) + first = log.observe(original, digest=sha(b"hello")) + os.replace(_write(root, "tmp", "hello"), os.path.join(root.path, "a.txt")) + replacement = fs_ref(root, "a.txt") + second = log.observe(replacement, digest=sha(b"hello")) + assert fx.invalidated_by(first, log.history) == (second.observation_id,) + # The latest check is of the replacement: same bytes, still fresh. + assert fx.freshness(second, log.history) is fx.Freshness.FRESH + + +def test_partial_read_cannot_verify_whole_content_and_blocks_fallback(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim) + log.observe(target, digest=sha(b"hello")) + log.observe(target, coverage=fx.Coverage.PARTIAL) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.UNVERIFIED + # A partial read can still decide existence. + exists = fx.Postcondition(target, fx.Predicate.EXISTS) + log2 = Log() + claim2 = log2.claim((target,), (exists,)) + log2.outcome(claim2) + log2.observe(target, coverage=fx.Coverage.PARTIAL) + assert fx.assess(claim2, log2.history).verdict is fx.EffectVerdict.VERIFIED + + +def test_partial_verifier_coverage_of_multiple_obligations(root): + log = Log() + a, b = fs_ref(root, "a.txt", missing=True), fs_ref(root, "b.txt", missing=True) + claim = log.claim((a, b), (content(a), content(b))) + log.outcome(claim) + log.observe(a, digest=sha(b"hello")) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.UNVERIFIED + log.observe(b, digest=sha(b"hello")) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.VERIFIED + + +def test_contradicting_fresh_observation(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim) + log.observe(target, digest=sha(b"other")) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.CONTRADICTED + + +def test_unknown_execution_matching_state_is_not_causation(root): + for execution in (fx.ExecutionOutcome.INTERRUPTED, fx.ExecutionOutcome.TIMED_OUT, + fx.ExecutionOutcome.CANCELLED): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim, execution) + log.observe(target, digest=sha(b"hello")) + assessment = fx.assess(claim, log.history) + assert assessment.verdict is fx.EffectVerdict.STATE_OBSERVED + assert "causality" in assessment.reason + + +def test_failed_execution_never_becomes_success(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim, fx.ExecutionOutcome.FAILED) + assessment = fx.assess(claim, log.history) + assert assessment.verdict is fx.EffectVerdict.FAILED and assessment.unresolved_impact + log.observe(target, digest=sha(b"hello")) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.FAILED + + +def test_unknown_partial_effect_has_unresolved_impact(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim, fx.ExecutionOutcome.TIMED_OUT, facts=fx.ProducerFacts(timed_out=True)) + assessment = fx.assess(claim, log.history) + assert assessment.verdict is fx.EffectVerdict.UNVERIFIED and assessment.unresolved_impact + + +def test_cleanup_failure_is_preserved_separately_from_effect(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim, cleanup=fx.CleanupState.FAILED) + log.observe(target, digest=sha(b"hello")) + assessment = fx.assess(claim, log.history) + assert assessment.verdict is fx.EffectVerdict.VERIFIED + assert assessment.cleanup is fx.CleanupState.FAILED + + +def test_background_running_is_pending_until_settled(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + log.outcome(claim, fx.ExecutionOutcome.RUNNING, facts=fx.ProducerFacts(job_state="running")) + log.observe(target, digest=sha(b"hello")) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.PENDING + log.outcome(claim) + # The observation predates settlement; a new one is required. + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.UNVERIFIED + log.observe(target, digest=sha(b"hello")) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.VERIFIED + + +def test_interrupted_claim_replays_as_unknown_never_success(root): + log = Log() + target = fs_ref(root, "a.txt", missing=True) + claim = log.claim((target,), (content(target),)) + assert fx.assess(claim, log.history).verdict is fx.EffectVerdict.PENDING + appended = fx.replay_interrupted(log.history, log.seq + 1) + assert [o.execution for o in appended] == [fx.ExecutionOutcome.INTERRUPTED] + assert appended[0].impact is fx.Impact.POSSIBLE and appended[0].replayed + history = fx.EffectHistory((claim,), appended, ()) + assert fx.assess(claim, history).unresolved_impact + assert fx.replay_interrupted(history, 99) == () + + +def test_stale_owned_revision_does_not_verify(): + v1 = fx.resource_ref(OwnedResource("notes", "u", "t", "notes", "n1", "rev-1"), "target") + v2 = fx.resource_ref(OwnedResource("notes", "u", "t", "notes", "n1", "rev-2"), "target") + log = Log() + claim = log.claim((v1,), (fx.Postcondition(v1, fx.Predicate.EXISTS),)) + log.outcome(claim) + first = log.observe(v1, mechanism=fx.ObservationMechanism.OWNED_RECORD_READ) + second = log.observe(v2, mechanism=fx.ObservationMechanism.OWNED_RECORD_READ) + # The old revision's readback never inherits freshness once a different + # revision of the same display ID is observed. + assert fx.invalidated_by(first, log.history) == (second.observation_id,) + assert fx.freshness(first, log.history) is fx.Freshness.STALE + # Verification rests only on the newest readback, of the current revision. + assert fx.assess(claim, log.history).observation_ids == (second.observation_id,) + + +def test_readbacks_require_admitted_source_action(root): + target = fs_ref(root, "a.txt", missing=True) + with pytest.raises(ValueError, match="admitted action"): + fx.Observation("o", 1, target, fx.ObservationMechanism.FILESYSTEM_READ, fx.Coverage.COMPLETE) + with pytest.raises(ValueError): + fx.Observation("o", 1, target, fx.ObservationMechanism.FILESYSTEM_READ, fx.Coverage.COMPLETE, + source_action_id="a", exists=False, content_sha256=sha(b"x")) From 9c4ed242962f5072f3d399437a8439c93182db28 Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Fri, 2 Oct 2026 20:02:31 +0100 Subject: [PATCH 02/15] feat(runtime): add durable append-only effect log with interrupted replay Claims are fsynced to a per-lineage JSONL log before a caller may invoke a backend; a persistence failure raises EffectPersistenceError (a ResourceIdentityError) so dispatch fails closed. Outcomes and observations are appended; nothing is rewritten. Reload validates every record strictly, ignores only a torn final write, and fails closed on corruption, forgery or hardlink aliasing. recover_interrupted appends INTERRUPTED/possible-impact outcomes for claims that never settled and leaves RUNNING background effects alone. The effect store is added to Wave 3 control-plane paths (prefix check only; the log itself refuses aliased files), so filesystem tools cannot forge it. Tests redirect the store to a session tmp directory. --- src/agent_runtime/effect_log.py | 177 +++++++++++++++++++++++ src/agent_runtime/resources.py | 8 + tests/conftest.py | 22 +++ tests/test_effect_journal_persistence.py | 164 +++++++++++++++++++++ 4 files changed, 371 insertions(+) create mode 100644 src/agent_runtime/effect_log.py create mode 100644 tests/test_effect_journal_persistence.py diff --git a/src/agent_runtime/effect_log.py b/src/agent_runtime/effect_log.py new file mode 100644 index 000000000..34ead8372 --- /dev/null +++ b/src/agent_runtime/effect_log.py @@ -0,0 +1,177 @@ +"""Durable append-only effect log for one root run lineage. + +This is the Wave 4 semantic store: claims, outcomes and observations only. It +is not a resource database, a process/containment store or an authority source. +A claim is fsynced before the backend is invoked; if that fails, the caller must +refuse the invocation. Later records are appended; nothing is rewritten. + +On reload, a claim without a settled outcome becomes an appended INTERRUPTED +outcome with possible impact. Reload never manufactures success and never +upgrades an old report to fresh state. +""" +from __future__ import annotations + +import json +import os +from pathlib import Path +import re +import stat +import threading +from typing import Any + +from src.constants import DATA_DIR +from src.agent_runtime.effects import ( + EffectAssessment, EffectClaim, EffectHistory, EffectOutcome, Observation, assess_all, replay_interrupted, +) +from src.agent_runtime.resources import ResourceIdentityError + + +EFFECTS_DIR = os.path.join(DATA_DIR, "effects") +_RUN_ID = re.compile(r"[a-f0-9]{32}") +_TYPES = {"claim": EffectClaim, "outcome": EffectOutcome, "observation": Observation} +_VERSION = 1 + + +class EffectPersistenceError(ResourceIdentityError): + """A pre-invocation claim could not be made durable; do not invoke.""" + + +def effects_dir() -> Path: + return Path(EFFECTS_DIR) + + +class EffectLog: + def __init__(self, run_id: str, *, durable: bool = True, directory: str | os.PathLike | None = None) -> None: + if not isinstance(run_id, str) or not _RUN_ID.fullmatch(run_id): + raise ValueError("Effect log requires a server-generated run identifier") + self.run_id = run_id + self.path = (Path(directory) if directory is not None else effects_dir()) / f"{run_id}.jsonl" if durable else None + self._claims: list[EffectClaim] = [] + self._outcomes: list[EffectOutcome] = [] + self._observations: list[Observation] = [] + self._sequence = 0 + # A non-claim record failed to persist. In-memory history stays + # truthful for this process; replay may lack the later record. + self.degraded = False + self._lock = threading.RLock() + + # -- persistence ------------------------------------------------------- + + def _write(self, kind: str, record: Any) -> None: + assert self.path is not None + line = json.dumps({"v": _VERSION, "type": kind, "record": record.to_dict()}, + sort_keys=True, separators=(",", ":"), ensure_ascii=False) + "\n" + self.path.parent.mkdir(mode=0o700, parents=True, exist_ok=True) + flags = os.O_WRONLY | os.O_APPEND | os.O_CREAT | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0) + descriptor = os.open(self.path, flags, 0o600) + try: + info = os.fstat(descriptor) + if info.st_nlink != 1 or not stat.S_ISREG(info.st_mode): + raise OSError("Effect log is aliased") + data = line.encode("utf-8") + while data: + written = os.write(descriptor, data) + data = data[written:] + os.fsync(descriptor) + finally: + os.close(descriptor) + + def _append(self, kind: str, build, *, required: bool): + with self._lock: + record = build(self._sequence + 1) + # Read-only runs need no durable file: replay concerns claims, and + # observations matter on disk only alongside them. + if self.path is not None and (kind != "observation" or self._claims): + try: + self._write(kind, record) + except OSError as error: + if required: + raise EffectPersistenceError("Effect claim could not be persisted durably") from error + self.degraded = True + self._sequence = record.sequence + {"claim": self._claims, "outcome": self._outcomes, "observation": self._observations}[kind].append(record) + return record + + # -- records ----------------------------------------------------------- + + def claim(self, **fields: Any) -> EffectClaim: + """Persist a claim before invocation; raises if it is not durable.""" + run_id = fields.pop("run_id", self.run_id) + return self._append("claim", lambda seq: EffectClaim(sequence=seq, run_id=run_id, **fields), required=True) + + def outcome(self, **fields: Any) -> EffectOutcome: + return self._append("outcome", lambda seq: EffectOutcome(sequence=seq, **fields), required=False) + + def observe(self, **fields: Any) -> Observation: + return self._append("observation", lambda seq: Observation(sequence=seq, **fields), required=False) + + def history(self) -> EffectHistory: + with self._lock: + return EffectHistory(tuple(self._claims), tuple(self._outcomes), tuple(self._observations)) + + def assessments(self) -> tuple[EffectAssessment, ...]: + return assess_all(self.history()) + + # -- replay ------------------------------------------------------------ + + @classmethod + def load(cls, run_id: str, *, directory: str | os.PathLike | None = None) -> "EffectLog": + """Reload a persisted log. A malformed record fails closed. + + A torn final line (no newline) is the only tolerated damage: it was a + write interrupted by a crash, so its claim never returned to a caller + and no backend invocation followed it. + """ + log = cls(run_id, directory=directory) + assert log.path is not None + try: + descriptor = os.open(log.path, os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0)) + except FileNotFoundError: + return log + except OSError as error: + raise EffectPersistenceError("Effect log is unreadable") from error + try: + info = os.fstat(descriptor) + if info.st_nlink != 1 or not stat.S_ISREG(info.st_mode): + raise EffectPersistenceError("Effect log is aliased") + with os.fdopen(descriptor, "rb") as stream: + descriptor = None + raw = stream.read() + except OSError as error: + raise EffectPersistenceError("Effect log is unreadable") from error + finally: + if descriptor is not None: + os.close(descriptor) + lines = raw.split(b"\n") + if lines and lines[-1] == b"": + lines.pop() + elif lines: + lines.pop() # torn final write + records: dict[str, list] = {"claim": [], "outcome": [], "observation": []} + for line in lines: + try: + entry = json.loads(line.decode("utf-8")) + if (not isinstance(entry, dict) or set(entry) != {"v", "type", "record"} + or entry["v"] != _VERSION or entry["type"] not in _TYPES): + raise ValueError("unsupported effect record") + records[entry["type"]].append(_TYPES[entry["type"]].from_dict(entry["record"])) + except (ValueError, TypeError, KeyError, UnicodeDecodeError) as error: + raise EffectPersistenceError("Effect log is corrupt") from error + try: + history = EffectHistory(tuple(records["claim"]), tuple(records["outcome"]), tuple(records["observation"])) + except ValueError as error: + raise EffectPersistenceError("Effect log history is inconsistent") from error + log._claims, log._outcomes, log._observations = (list(history.claims), list(history.outcomes), + list(history.observations)) + log._sequence = max((r.sequence for r in (*history.claims, *history.outcomes, *history.observations)), + default=0) + return log + + def recover_interrupted(self) -> tuple[EffectOutcome, ...]: + """Append INTERRUPTED outcomes for claims that never settled.""" + with self._lock: + pending = replay_interrupted(self.history(), self._sequence + 1) + for outcome in pending: + self._append("outcome", lambda seq, o=outcome: EffectOutcome( + o.effect_id, seq, o.execution, o.impact, replayed=True), required=False) + return tuple(self._outcomes[-len(pending):]) if pending else () diff --git a/src/agent_runtime/resources.py b/src/agent_runtime/resources.py index 8513d087f..66461744b 100644 --- a/src/agent_runtime/resources.py +++ b/src/agent_runtime/resources.py @@ -45,6 +45,14 @@ def _control_plane_path(path): processes = sys.modules.get("src.agent_runtime.process_resources") if processes is not None: job_dirs.add(canonical_root(processes._LAUNCH_DIR)) + # Durable effect claims/outcomes/observations are server evidence state. + # The log refuses hardlinked files itself, so a prefix check suffices. + effect_dirs = {canonical_root(os.path.join(constants.DATA_DIR, "effects"))} + effect_log = sys.modules.get("src.agent_runtime.effect_log") + if effect_log is not None: + effect_dirs.add(canonical_root(effect_log.EFFECTS_DIR)) + if any(Path(path).is_relative_to(directory) for directory in effect_dirs): + return True # Producers may have configured paths different from the default constants. # Inspect already-loaded server metadata without initializing a store here. bg = sys.modules.get("src.bg_jobs") diff --git a/tests/conftest.py b/tests/conftest.py index 4f3eb2c07..0915d519c 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -227,6 +227,28 @@ def _serve_test_static(): server.server_close() +@pytest.fixture(scope="session") +def _effects_store_root(tmp_path_factory): + return tmp_path_factory.mktemp("effects") + + +@pytest.fixture(autouse=True) +def _isolated_effects_store(_effects_store_root): + """Keep durable effect claims out of the developer's real data directory. + + Restored manually: requesting the shared ``monkeypatch`` here would move + its teardown after ``_no_leaked_module_stubs`` and misreport test stubs. + """ + from src.agent_runtime import effect_log + + previous = effect_log.EFFECTS_DIR + effect_log.EFFECTS_DIR = str(_effects_store_root) + try: + yield + finally: + effect_log.EFFECTS_DIR = previous + + @pytest.fixture(autouse=True) def _no_leaked_module_stubs(): """Fail the test that leaves a bare ``src.*``/``core.*`` stub behind. diff --git a/tests/test_effect_journal_persistence.py b/tests/test_effect_journal_persistence.py new file mode 100644 index 000000000..dea9125c7 --- /dev/null +++ b/tests/test_effect_journal_persistence.py @@ -0,0 +1,164 @@ +"""Durable Wave 4 effect log: pre-invocation claims, append-only replay.""" +from __future__ import annotations + +import json +import os + +import pytest + +from src.agent_runtime import effects as fx +from src.agent_runtime.effect_log import EffectLog, EffectPersistenceError +from src.agent_runtime.resources import FilesystemResource, FilesystemRoot, OwnedResource + + +RUN = "a" * 32 + + +@pytest.fixture +def target(tmp_path): + workspace = tmp_path / "ws" + workspace.mkdir() + root = FilesystemRoot.seal(str(workspace)) + return fx.resource_ref(FilesystemResource.resolve(root, str(workspace / "a.txt"), allow_missing=True), "destination") + + +def claim(log, target, effect_id="e1", action_id="act-1"): + return log.claim(effect_id=effect_id, action_id=action_id, operation=fx.OperationRef("write_file", "", "0" * 64), + impact_scope=(target,), obligations=(fx.Postcondition(target, fx.Predicate.EXISTS),)) + + +def records(path): + return [json.loads(line) for line in path.read_text().splitlines()] + + +def test_claim_is_fsynced_to_disk_before_returning(tmp_path, target, monkeypatch): + synced = [] + real_fsync = os.fsync + monkeypatch.setattr(os, "fsync", lambda fd: (synced.append(fd), real_fsync(fd))) + log = EffectLog(RUN, directory=tmp_path / "fx") + made = claim(log, target) + assert synced, "claim must be fsynced before the caller can invoke a backend" + on_disk = records(log.path) + assert [r["type"] for r in on_disk] == ["claim"] + assert fx.EffectClaim.from_dict(on_disk[0]["record"]) == made + assert oct(log.path.stat().st_mode & 0o777) == "0o600" + + +def test_claim_persistence_failure_raises_and_records_nothing(tmp_path, target): + blocker = tmp_path / "not-a-directory" + blocker.write_text("x") + log = EffectLog(RUN, directory=blocker) + with pytest.raises(EffectPersistenceError): + claim(log, target) + assert log.history().claims == () + + +def test_non_claim_failure_degrades_without_losing_in_memory_truth(tmp_path, target, monkeypatch): + log = EffectLog(RUN, directory=tmp_path / "fx") + made = claim(log, target) + monkeypatch.setattr(log, "_write", lambda kind, record: (_ for _ in ()).throw(OSError("disk full"))) + log.outcome(effect_id=made.effect_id, execution=fx.ExecutionOutcome.REPORTED_SUCCESS, impact=fx.Impact.POSSIBLE) + assert log.degraded + assert fx.assess(made, log.history()).execution is fx.ExecutionOutcome.REPORTED_SUCCESS + # Replay only sees the durable claim: it stays unknown, never success. + reloaded = EffectLog.load(RUN, directory=tmp_path / "fx") + assert fx.assess(made, reloaded.history()).verdict is fx.EffectVerdict.PENDING + + +def test_replay_after_restart_marks_unsettled_claims_interrupted(tmp_path, target): + directory = tmp_path / "fx" + log = EffectLog(RUN, directory=directory) + settled = claim(log, target, "e1", "a1") + log.outcome(effect_id="e1", execution=fx.ExecutionOutcome.REPORTED_SUCCESS, impact=fx.Impact.POSSIBLE, + execution_id="a1:x") + log.observe(observation_id="o1", resource=target, mechanism=fx.ObservationMechanism.FILESYSTEM_READ, + coverage=fx.Coverage.COMPLETE, source_action_id="r1", exists=True) + pending = claim(log, target, "e2", "a2") + del log # process "crashes" before e2 settles + + reloaded = EffectLog.load(RUN, directory=directory) + assert fx.assess(settled, reloaded.history()).verdict is fx.EffectVerdict.UNVERIFIED # e2 made o1 stale + appended = reloaded.recover_interrupted() + assert [(o.effect_id, o.execution, o.impact, o.replayed) for o in appended] == [ + ("e2", fx.ExecutionOutcome.INTERRUPTED, fx.Impact.POSSIBLE, True)] + assessment = fx.assess(pending, reloaded.history()) + assert assessment.verdict is fx.EffectVerdict.UNVERIFIED and assessment.unresolved_impact + # Recovery is append-only and idempotent across another restart. + again = EffectLog.load(RUN, directory=directory) + assert again.recover_interrupted() == () + assert [r["type"] for r in records(again.path)] == ["claim", "outcome", "observation", "claim", "outcome"] + + +def test_running_background_claim_is_not_converted_by_replay(tmp_path, target): + log = EffectLog(RUN, directory=tmp_path / "fx") + made = claim(log, target) + log.outcome(effect_id=made.effect_id, execution=fx.ExecutionOutcome.RUNNING, impact=fx.Impact.POSSIBLE) + reloaded = EffectLog.load(RUN, directory=tmp_path / "fx") + assert reloaded.recover_interrupted() == () + assert fx.assess(made, reloaded.history()).verdict is fx.EffectVerdict.PENDING + + +def test_torn_final_write_is_ignored_but_corruption_fails_closed(tmp_path, target): + directory = tmp_path / "fx" + log = EffectLog(RUN, directory=directory) + claim(log, target) + with open(log.path, "ab") as stream: + stream.write(b'{"v":1,"type":"outcome","rec') # crash mid-append + assert len(EffectLog.load(RUN, directory=directory).history().claims) == 1 + with open(log.path, "ab") as stream: + stream.write(b'\n{"v":1,"type":"outcome","record":{"forged":true}}\n') + with pytest.raises(EffectPersistenceError): + EffectLog.load(RUN, directory=directory) + + +def test_forged_success_record_cannot_be_replayed_into_verification(tmp_path, target): + directory = tmp_path / "fx" + log = EffectLog(RUN, directory=directory) + made = claim(log, target) + forged = {"v": 1, "type": "outcome", "record": {**fx.EffectOutcome( + made.effect_id, 2, fx.ExecutionOutcome.FAILED, fx.Impact.POSSIBLE).to_dict(), "execution": "verified"}} + with open(log.path, "a") as stream: + stream.write(json.dumps(forged) + "\n") + with pytest.raises(EffectPersistenceError): + EffectLog.load(RUN, directory=directory) + + +def test_hardlinked_log_is_refused(tmp_path, target): + directory = tmp_path / "fx" + log = EffectLog(RUN, directory=directory) + claim(log, target) + os.link(log.path, tmp_path / "alias.jsonl") + with pytest.raises(EffectPersistenceError): + EffectLog.load(RUN, directory=directory) + with pytest.raises(EffectPersistenceError): + claim(log, target, "e2", "a2") + + +def test_read_only_runs_write_no_file(tmp_path, target): + log = EffectLog(RUN, directory=tmp_path / "fx") + log.observe(observation_id="o1", resource=target, mechanism=fx.ObservationMechanism.FILESYSTEM_READ, + coverage=fx.Coverage.PARTIAL, source_action_id="r1", exists=True) + assert not log.path.exists() and len(log.history().observations) == 1 + + +def test_run_identifier_must_be_server_generated(tmp_path): + for forged in ("../escape", "", "A" * 32, "a" * 31): + with pytest.raises(ValueError): + EffectLog(forged, directory=tmp_path) + + +def test_effect_store_is_server_control_state(tmp_path): + from src.agent_runtime import effect_log + store = effect_log.effects_dir() + store.mkdir(parents=True, exist_ok=True) + root = FilesystemRoot.seal(str(store.parent)) + with pytest.raises(ValueError, match="sensitive"): + FilesystemResource.resolve(root, str(store / ("b" * 32 + ".jsonl")), allow_missing=True) + + +def test_owned_revision_scope_round_trips(tmp_path): + record = fx.resource_ref(OwnedResource("notes", "u", "t", "notes", "n1", "rev-1"), "record") + log = EffectLog(RUN, directory=tmp_path / "fx") + log.claim(effect_id="e1", action_id="a1", operation=fx.OperationRef("manage_notes", "", "0" * 64), + impact_scope=(record,)) + assert EffectLog.load(RUN, directory=tmp_path / "fx").history().claims[0].impact_scope == (record,) From 8402c388b47cbd26ae79a91a4dee051d80d59b0b Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Fri, 2 Oct 2026 20:09:35 +0100 Subject: [PATCH 03/15] feat(runtime): claim effects before dispatch and gate completion on them Dispatcher seam: mark_dispatch, which runs inside the live Wave 3 binding scope immediately before backend invocation, now durably claims a possible effect before execution_id is assigned. If the claim cannot be persisted the action stays undispatched and the dispatcher returns BLOCKED; dispatched() closes the never-awaited coroutine. record_action appends the outcome (including cancellation/interruption) and admitted-read observations before the receipt reduction drops producer facts. Adapters consume only the bound operations the dispatcher admitted: filesystem bindings give exact scope and predicates (write_file content digest after fence unwrapping, apply_patch add/delete, edit existence); bash/python launches have unknown scope with the launch generation as lineage; job kills scope the exact job and its processes; owned operations scope their exact revisioned records; external backends are claimed as external and never verified by acknowledgement; browser session_info yields session lifecycle observations only, and a page binding is never effect scope. Complete read_file re-reads the exact bound source to digest it; offset/limit, truncation, extraction and listings are partial. Background launches stay RUNNING until an admitted read of the exact job generation (via a durable launch index, across continuation runs) reports settlement. Producer seams: typed job lifecycle facts on manage_bg_jobs reads/kills, a structured timed_out flag on containment timeouts, and mutation_attempted on in-place write_file/edit_file failures after truncation. Completion: the existing EvidenceLedger consumes effect assessments through a single helper used for the decision, ask_user and prose filtering. A required artifact is unsettled by a later unresolved effect that may have touched it, a fresh contradicting readback fails the decision, and partial reads no longer count as artifact validation. Ordinary conversation and read-only turns are unchanged; no second completion policy is introduced. --- src/agent_evidence.py | 99 +++++- src/agent_runtime/completion.py | 20 +- src/agent_runtime/effect_adapters.py | 370 +++++++++++++++++++++ src/agent_runtime/effect_log.py | 61 ++++ src/agent_runtime/effects.py | 20 +- src/agent_runtime/journal.py | 76 ++++- src/agent_tools/bg_job_tools.py | 20 +- src/agent_tools/filesystem_tools.py | 22 +- src/agent_tools/subprocess_tools.py | 2 +- tests/test_effect_resource_bindings.py | 227 +++++++++++++ tests/test_effect_verification_adapters.py | 331 ++++++++++++++++++ 11 files changed, 1231 insertions(+), 17 deletions(-) create mode 100644 src/agent_runtime/effect_adapters.py create mode 100644 tests/test_effect_resource_bindings.py create mode 100644 tests/test_effect_verification_adapters.py diff --git a/src/agent_evidence.py b/src/agent_evidence.py index 54d14367c..29daebebe 100644 --- a/src/agent_evidence.py +++ b/src/agent_evidence.py @@ -632,6 +632,84 @@ class EvidenceLedger: # Retain receipt command identity privately for presentation matching; # model prose and client dictionaries never populate this evidence. self._verifier_commands: dict[str, tuple[str, ...]] = {} + # Wave 4 effect assessments from the run's journal, plus the journal + # order of actions so receipt evidence and effects share one ordering. + self.effects: list[dict[str, Any]] = [] + self._action_order: dict[str, int] = {} + + def record_effects(self, entries: Iterable[Mapping[str, Any]], action_order: Mapping[str, int], + partial_reads: Iterable[str] = ()) -> None: + """Consume server-derived effect assessments (never model/client data). + + ``partial_reads`` names read actions whose admitted observation was + partial (offset/limit, truncation or extraction): such a read cannot + validate omitted content, so its validation event is not authoritative. + """ + from dataclasses import replace + from src.agent_runtime.effects import EffectAssessment + self.effects = [dict(entry) for entry in entries + if isinstance(entry, Mapping) and isinstance(entry.get("assessment"), EffectAssessment)] + self._action_order = {str(k): v for k, v in action_order.items() if type(v) is int} + partial = set(partial_reads) + self.events = [replace(event, authoritative=False, detail="partial read; omitted content is unvalidated") + if event.kind == EvidenceKind.ARTIFACT_VALIDATION and event.tool == "read_file" + and event.action_id in partial else event for event in self.events] + + def _last_success_ordinal(self, required: str) -> int: + return max((self._action_order.get(event.action_id, 0) for event in self.events + if event.kind == EvidenceKind.ARTIFACT_MUTATION and event.authoritative and event.success + and _artifact_path_matches_required(event.artifact_path, required, self.requirements.workspace_root)), + default=0) + + def _later_effects(self, required: str) -> list[tuple[dict[str, Any], bool]]: + """Effects after the artifact's last successful mutation, with targeting.""" + floor = self._last_success_ordinal(required) + later = [] + for entry in self.effects: + ordinal = entry.get("ordinal") + if type(ordinal) is not int or ordinal <= floor: + continue + explicit = any(_artifact_path_matches_required(path, required, self.requirements.workspace_root) + for path in entry.get("paths") or ()) + if explicit or entry.get("unknown_scope"): + later.append((entry, explicit)) + return later + + def _effect_unsettled(self, required: str) -> bool: + """A later operation may have partially changed this artifact. + + Explicit targets are unsettled by unknown/timed-out/cancelled outcomes + and by failures after the producer reached its mutation stage (atomic + refusals keep the earlier artifact). Unknown-scope effects are + unsettled when nothing captured their settlement: cancellation, + interruption, or failed process teardown; settled shell/Python changes + are already tracked through artifact version capture. + """ + from src.agent_runtime.effects import CleanupState, ExecutionOutcome + unknown = {ExecutionOutcome.ATTEMPTED, ExecutionOutcome.INTERRUPTED, ExecutionOutcome.CANCELLED} + for entry, explicit in self._later_effects(required): + assessment = entry["assessment"] + if not assessment.unresolved_impact: + continue + if assessment.execution in unknown or assessment.execution is ExecutionOutcome.RUNNING: + return True + if explicit and (assessment.execution is ExecutionOutcome.TIMED_OUT + or (assessment.execution is ExecutionOutcome.FAILED and entry.get("mutation_attempted"))): + return True + if not explicit and assessment.cleanup is CleanupState.FAILED: + return True + return False + + def _effect_contradicted(self, required: str) -> str: + """The latest effect targeting the artifact, if fresh readback contradicts it.""" + from src.agent_runtime.effects import EffectVerdict + targeting = [entry for entry in self.effects if type(entry.get("ordinal")) is int + and any(_artifact_path_matches_required(path, required, self.requirements.workspace_root) + for path in entry.get("paths") or ())] + if not targeting: + return "" + latest = max(targeting, key=lambda entry: entry["ordinal"])["assessment"] + return latest.effect_id if latest.verdict is EffectVerdict.CONTRADICTED else "" @classmethod def from_tool_events( @@ -827,6 +905,8 @@ class EvidenceLedger: and matching[-1].tool in {'bash', 'python'}) if not successful or destructive_failure: return False + if self.effects and (self._effect_unsettled(path) or self._effect_contradicted(path)): + return False return True def record_media_ingress(self, metadata: Mapping[str, Any]) -> None: @@ -897,8 +977,16 @@ class EvidenceLedger: 'artifact content changed after verification', (latest_verifier.event_id,)) + for required in self.requirements.required_artifacts if self.effects else (): + contradicted = self._effect_contradicted(required) + if contradicted: + return CompletionDecision(CompletionStatus.FAILED, False, + "fresh readback contradicts the requested artifact content", + (), (required,)) + satisfied_ids: list[str] = [] missing: list[str] = [] + unsettled: list[str] = [] workspace_root = str(self.requirements.workspace_root or "").strip() for required in self.requirements.required_artifacts: matches = [ @@ -934,15 +1022,20 @@ class EvidenceLedger: filesystem_missing = True if latest_success is None or destructive_failure or filesystem_missing: missing.append(required) + elif self.effects and self._effect_unsettled(required): + # Earlier success is historical; a later possible change to + # this artifact has no settled evidence. + unsettled.append(required) else: satisfied_ids.append(latest_success.event_id) - if missing: + if missing or unsettled: return CompletionDecision( CompletionStatus.BLOCKED, False, - "required artifacts lack successful mutation evidence", + "required artifacts lack successful mutation evidence" if missing else + "a later operation may have changed a required artifact without settled evidence", tuple(satisfied_ids), - tuple(missing), + tuple([*missing, *unsettled]), ) latest_mutation_index = max( diff --git a/src/agent_runtime/completion.py b/src/agent_runtime/completion.py index b8b80700c..943074df3 100644 --- a/src/agent_runtime/completion.py +++ b/src/agent_runtime/completion.py @@ -20,9 +20,19 @@ from src.agent_evidence import ( requirements_from_runtime_context, _execution_obligation, _unquoted_statements, _ARTIFACT_PATH, ) +from .effect_log import EffectLog from .journal import ActionJournal, bind_journal, current_journal +def _ledger(journal: ActionJournal, requirements) -> EvidenceLedger: + """The single evidence view used for the decision and the prose filter.""" + ledger = EvidenceLedger.from_tool_events(journal.evidence_events(), requirements) + ledger.record_effects(journal.effect_entries(), + {action.action_id: index for index, action in enumerate(journal.actions, 1)}, + journal.partial_reads()) + return ledger + + _TEST_CLAIM = re.compile( r'\b(?:(?:all\s+)?(?:tests?|checks?|verification|suite)\s+(?:have\s+|has\s+|now\s+|are\s+|is\s+)*(?:passed|passing|successful|green)|' r'(?:passed|passing)\s+(?:all\s+)?(?:the\s+)?tests?|\d+\s+passed)\b', re.I) @@ -192,6 +202,10 @@ def with_completion_gate(func): journal = ActionJournal( workspace=requirements.workspace_root, observed_artifacts=requirements.required_artifacts, parent_run_id=bound.get('_parent_run_id') or (parent.run_id if parent is not None else None)) + # One durable effect log per run lineage gives child effects and parent + # observations a single total order for invalidation. + journal.effects = (parent.effects if parent is not None and parent.effects is not None + else EffectLog(journal.run_id)) answer_events: list[dict] = [] metrics_events: list[dict] = [] answer = '' @@ -241,7 +255,7 @@ def with_completion_gate(func): awaiting = True payload = data.get('data') or {} if isinstance(payload.get('question'), str): - current = EvidenceLedger.from_tool_events(journal.evidence_events(), requirements) + current = _ledger(journal, requirements) question, why = completion_answer(payload['question'], current, current.evaluate(awaiting_user=True)) if why: data = {**data, 'data': {**payload, 'question': question}} @@ -290,7 +304,7 @@ def with_completion_gate(func): presentation_replaced = True answer = terminal_answer answer_events = [event for event in answer_events if event.get('thinking') is True] - ledger = EvidenceLedger.from_tool_events(journal.evidence_events(), requirements) + ledger = _ledger(journal, requirements) decision = ledger.evaluate(exhausted=exhausted, awaiting_user=awaiting) if provider_error: decision = replace(decision, status=CompletionStatus.FAILED, @@ -334,6 +348,8 @@ def with_completion_gate(func): metadata.update(completion_decision=decision.to_dict(), evidence_events=ledger.to_list(), action_receipts=journal.to_list(), completion_requirements=requirements.to_dict(), run_id=journal.run_id, parent_run_id=journal.parent_run_id) + if ledger.effects: + metadata['effect_assessments'] = [entry['assessment'].to_dict() for entry in ledger.effects] metadata['completion_gate'] = { 'buffer_seconds': released_at - first_answer_at if first_answer_at is not None else 0, 'first_visible_answer_seconds': released_at - started, diff --git a/src/agent_runtime/effect_adapters.py b/src/agent_runtime/effect_adapters.py new file mode 100644 index 000000000..bd8ba3d50 --- /dev/null +++ b/src/agent_runtime/effect_adapters.py @@ -0,0 +1,370 @@ +"""Server-boundary adapters from admitted Wave 3 bindings to effect records. + +Runs only inside the dispatcher's existing admission scope: the bindings read +here are the contextvars the dispatcher bound after authority, resource and +approval checks. Nothing here admits, resolves, broadens or re-derives a +resource. Observations are recorded only for operations that were themselves +admitted reads of the exact bound resource; evidence bookkeeping never performs +a read that the operation was not already admitted to perform. +""" +from __future__ import annotations + +import asyncio +from dataclasses import dataclass, field +import hashlib +import json +import logging +import os +import stat +from typing import Any + +from src.agent_runtime.effects import ( + CleanupState, Coverage, EffectClaim, ExecutionOutcome, Impact, ObservationMechanism, OperationRef, + Postcondition, Predicate, ProducerFacts, ResourceKind, ResourceRef, producer_facts, resource_ref, +) + + +_FILESYSTEM_READS = frozenset({"read_file", "ls", "glob", "grep"}) +_JOB_READS = frozenset({"list", "ls", "jobs", "output", "get", "read", "tail", "status", "show"}) +_OWNED_READS = frozenset({"vault_get", "vault_search", "list_sessions", "search_chats"}) +_JOB_SETTLED = {"done", "failed"} +logger = logging.getLogger(__name__) + + +@dataclass +class DispatchCapture: + """The admitted bindings that were live when the backend was invoked.""" + + filesystem: Any = None + owned: Any = None + process: Any = None + backend: Any = None + browser: Any = None + claim: EffectClaim | None = None + read_only: bool = False + paths: tuple[str, ...] = field(default_factory=tuple) + + +def capture_dispatch() -> DispatchCapture: + from src.agent_runtime.owned_resources import active_owned_operation + from src.agent_runtime.process_resources import active_process_operation + from src.agent_runtime.remote_resources import active_backend_operation + from src.agent_runtime.resource_binding import active_resource_operation + import sys + browser_module = sys.modules.get("src.browser_identity") + browser = browser_module._ACTIVE.get() if browser_module is not None else None + return DispatchCapture(active_resource_operation(), active_owned_operation(), active_process_operation(), + active_backend_operation(), browser) + + +def _exact_operation(capture: DispatchCapture): + for bound in (capture.filesystem, capture.owned, capture.process, capture.browser): + if bound is not None: + return bound.operation, getattr(bound, "execution_input", None), getattr(bound, "request_id", "") + return None, None, "" + + +def _operation(capture: DispatchCapture, action: Any) -> OperationRef: + operation, execution_input, request_id = _exact_operation(capture) + if operation is not None: + return OperationRef.from_exact(operation, execution_input, request_id) + backend = capture.backend + # Unbound tools still name their final normalized dispatcher input. + digest = hashlib.sha256(str(action.arguments).encode("utf-8", errors="replace")).hexdigest() + return OperationRef(str(action.tool) or "unknown", "", digest, + getattr(backend, "request_id", "") if backend is not None else "") + + +def _write_file_digest(execution_input: str, path: str) -> str: + """The exact bytes WriteFileTool commits for this admitted input, or ''.""" + from src.agent_tools.filesystem_tools import _unwrap_fenced_source_body + try: + args = json.loads(execution_input) + except (TypeError, ValueError): + return "" + body = args.get("content") if isinstance(args, dict) else None + if not isinstance(body, str) or os.linesep != "\n": + return "" + return hashlib.sha256(_unwrap_fenced_source_body(body, path).encode("utf-8")).hexdigest() + + +def _filesystem_scope(bound: Any) -> tuple[tuple[ResourceRef, ...], tuple[Postcondition, ...]]: + from src.agent_tools.filesystem_tools import _parse_agent_patch + tool = bound.operation.tool + refs = tuple(resource_ref(b.resource, b.role) for b in bound.bindings) + obligations: list[Postcondition] = [] + if tool == "write_file": + target = refs[0] + expected = _write_file_digest(bound.execution_input, bound.bindings[0].resource.path) + obligations.append(Postcondition(target, Predicate.CONTENT_SHA256, expected) if expected + else Postcondition(target, Predicate.EXISTS)) + elif tool == "edit_file": + obligations.append(Postcondition(refs[0], Predicate.EXISTS)) + elif tool == "apply_patch": + ops = _parse_agent_patch(json.loads(bound.execution_input)["patch_text"]) + for op, ref in zip(ops, refs): + if op["kind"] == "add": + digest = hashlib.sha256(op["content"].encode("utf-8")).hexdigest() + obligations.append(Postcondition(ref, Predicate.CONTENT_SHA256, digest)) + elif op["kind"] == "delete": + obligations.append(Postcondition(ref, Predicate.ABSENT)) + else: + obligations.append(Postcondition(ref, Predicate.EXISTS)) + return refs, tuple(obligations) + + +def classify(capture: DispatchCapture) -> dict[str, Any] | None: + """Claim scope for the captured bindings, or None for an admitted read. + + Unbound operations get an unknown-scope claim: they may change anything. + """ + impact: tuple[ResourceRef, ...] = () + dependencies: tuple[ResourceRef, ...] = () + obligations: tuple[Postcondition, ...] = () + external = False + if capture.browser is not None: + # Wave 3 admits only session metadata. A page binding is never + # effect-bindable; leave its scope unknown rather than infer it. + if capture.browser.page is None: + return None + elif capture.filesystem is not None: + if capture.filesystem.operation.tool in _FILESYSTEM_READS: + return None + impact, obligations = _filesystem_scope(capture.filesystem) + elif capture.process is not None: + bound = capture.process + if bound.launch is not None: + # An arbitrary command has unknown impact scope; the exact launch + # reservation is kept only as lineage for background settlement. + dependencies = (resource_ref(bound.launch, "launch"),) + else: + action = str(json.loads(bound.operation.input or "{}").get("action", "list")).strip().lower() + if action in _JOB_READS: + return None + impact = tuple(resource_ref(job, "job") for job in bound.jobs) + tuple( + resource_ref(process, "process") for job in bound.jobs for process in job.processes) + tuple( + resource_ref(process, "process") for process in bound.processes) + elif capture.owned is not None: + if capture.owned.operation.tool in _OWNED_READS: + return None + impact = tuple(resource_ref(r, "record") for r in capture.owned.resources) + dependencies = tuple(resource_ref(a.file, "attachment") for a in capture.owned.attachments) + if capture.backend is not None: + from src.agent_runtime.resources import ExternalResource + if isinstance(capture.backend.resource, ExternalResource): + external = True + impact = (*impact, resource_ref(capture.backend.resource, "backend")) + return {"impact_scope": impact, "dependencies": dependencies, "obligations": obligations, "external": external} + + +def begin_effect(journal: Any, action: Any) -> DispatchCapture: + """Capture bindings and durably claim a possible effect before invocation.""" + capture = capture_dispatch() + log = journal.effects + try: + scope = classify(capture) + except Exception: # noqa: BLE001 - classification never blocks dispatch + # An unclassifiable admitted operation may change anything. + logger.warning("Effect scope classification failed; claiming unknown scope", exc_info=True) + scope = {"impact_scope": (), "dependencies": (), "obligations": (), "external": False} + if scope is None: + capture.read_only = True + else: + capture.claim = log.claim(effect_id=action.action_id + ":effect", run_id=journal.run_id, + action_id=action.action_id, operation=_operation(capture, action), + parent_run_id=journal.parent_run_id or "", **scope) + capture.paths = tuple(ref.location[-1] for ref in capture.claim.impact_scope + if ref.kind is ResourceKind.FILESYSTEM) + for ref in capture.claim.dependencies: + if ref.kind is ResourceKind.PROCESS_LAUNCH: + try: + log.index_launch(ref.incarnation, capture.claim.effect_id) + except (OSError, ValueError): + # Without the index a later turn cannot settle this + # launch: it stays running/unknown, never successful. + logger.warning("Background launch lineage was not indexed", exc_info=True) + return capture + + +def _execution(result: Any, facts: ProducerFacts) -> ExecutionOutcome: + if not isinstance(result, dict): + return ExecutionOutcome.INTERRUPTED + if facts.timed_out: + return ExecutionOutcome.TIMED_OUT + if isinstance(result.get("bg_job_id"), str) and facts.exit_code == 0: + return ExecutionOutcome.RUNNING + if result.get("detached") is True or result.get("status") == "running" or result.get("running") is True: + return ExecutionOutcome.RUNNING + denied = bool(result.get("blocked") or result.get("approval_required") + or facts.failure_kind.endswith("_denied")) + if facts.exit_code == 0 and not result.get("error") and not denied: + return ExecutionOutcome.REPORTED_SUCCESS + return ExecutionOutcome.FAILED + + +def _cleanup(result: Any, facts: ProducerFacts) -> CleanupState: + if not isinstance(result, dict): + return CleanupState.UNKNOWN + if facts.failure_kind == "process_teardown_failed": + return CleanupState.FAILED + teardown = result.get("teardown") + if isinstance(teardown, dict) and type(teardown.get("dead")) is bool: + return CleanupState.VERIFIED if teardown["dead"] else CleanupState.FAILED + if facts.external: + # External execution reports no locally observed teardown. + return CleanupState.UNKNOWN + return CleanupState.NOT_APPLICABLE + + +def settle_effect(journal: Any, action: Any, capture: DispatchCapture | None, *, + result: Any = None, error: BaseException | None = None) -> None: + """Append the outcome and any admitted-read observations for one action.""" + if capture is None: + return + log = journal.effects + if capture.claim is not None: + if error is not None: + execution = (ExecutionOutcome.CANCELLED if isinstance(error, asyncio.CancelledError) + else ExecutionOutcome.INTERRUPTED) + facts, cleanup = ProducerFacts(), CleanupState.UNKNOWN + else: + facts = producer_facts(result) + if capture.backend is not None and capture.claim.external: + facts = ProducerFacts(**{**facts.to_dict(), "external": True, + "remote_acknowledged": facts.exit_code == 0}) + execution, cleanup = _execution(result, facts), _cleanup(result, facts) + log.outcome(effect_id=capture.claim.effect_id, execution=execution, impact=Impact.POSSIBLE, + facts=facts, cleanup=cleanup, execution_id=action.execution_id or "") + if (execution is ExecutionOutcome.REPORTED_SUCCESS and capture.process is not None + and capture.process.launch is None): + _settle_background(log, capture, result) # e.g. an exact kill + return + if error is not None or not isinstance(result, dict) or result.get("exit_code") != 0 or result.get("error"): + return + for fields in _observations(capture, action, result): + log.observe(**fields) + if capture.process is not None and capture.process.launch is None: + _settle_background(log, capture, result) + + +# -- observations ------------------------------------------------------------ + +def _read_whole(resource: Any, limit: int) -> bytes | None: + """Re-read the exact admitted source binding; None if it is not stable.""" + flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0) + try: + resource.validate() + descriptor = os.open(resource.path, flags) + with os.fdopen(descriptor, "rb") as stream: + info = os.fstat(stream.fileno()) + identity = resource.identity + if (not stat.S_ISREG(info.st_mode) or identity is None + or (info.st_dev, info.st_ino) != (identity.device, identity.inode)): + return None + data = stream.read(limit + 1) + resource.validate() + except (OSError, ValueError): + return None + return data + + +def _file_observation(capture: DispatchCapture, action: Any) -> dict[str, Any] | None: + from src.agent_tools import filesystem_tools as producer + bound = capture.filesystem + binding = bound.bindings[0] + resource = binding.resource + args = json.loads(bound.execution_input) + partial = bool(args.get("offset") or args.get("limit")) or ( + os.path.splitext(resource.path)[1].lower() in producer._STRUCTURED_DOCUMENT_SUFFIXES) + data = _read_whole(resource, producer.MAX_READ_CHARS * 4) + if data is None: + return None + if len(data) > producer.MAX_READ_CHARS * 4 or len(data.decode("utf-8", errors="replace")) > producer.MAX_READ_CHARS: + partial = True # the producer truncated what it read + complete = not partial + return dict(observation_id=action.action_id + ":observation", resource=resource_ref(resource, binding.role), + mechanism=ObservationMechanism.FILESYSTEM_READ, + coverage=Coverage.COMPLETE if complete else Coverage.PARTIAL, + source_action_id=action.action_id, source_execution_id=action.execution_id or "", + exists=True, content_sha256=hashlib.sha256(data).hexdigest() if complete else "") + + +def _observations(capture: DispatchCapture, action: Any, result: dict) -> list[dict[str, Any]]: + base = dict(source_action_id=action.action_id, source_execution_id=action.execution_id or "") + if capture.browser is not None and capture.browser.page is None: + # Session lifecycle metadata only; never page/document state. + return [dict(observation_id=action.action_id + ":observation", + resource=resource_ref(capture.browser.session, "session"), + mechanism=ObservationMechanism.BROWSER_SESSION, coverage=Coverage.PARTIAL, + exists=True, **base)] + if capture.filesystem is not None: + tool = capture.filesystem.operation.tool + if tool == "read_file": + observation = _file_observation(capture, action) + return [observation] if observation else [] + if tool in _FILESYSTEM_READS: + # Listings/searches are partial: they cannot decide content. + return [dict(observation_id=f"{action.action_id}:observation:{i}", resource=resource_ref(b.resource, b.role), + mechanism=ObservationMechanism.FILESYSTEM_READ, coverage=Coverage.PARTIAL, exists=True, **base) + for i, b in enumerate(capture.filesystem.bindings)] + if capture.owned is not None and capture.owned.operation.tool in _OWNED_READS: + return [dict(observation_id=f"{action.action_id}:observation:{i}", resource=resource_ref(r, "record"), + mechanism=ObservationMechanism.OWNED_RECORD_READ, coverage=Coverage.PARTIAL, exists=True, **base) + for i, r in enumerate(capture.owned.resources) if r.record_id != "*"] + if capture.process is not None and capture.process.launch is None: + job = result.get("job") + if isinstance(job, dict) and len(capture.process.jobs) == 1: + return [dict(observation_id=action.action_id + ":observation", + resource=resource_ref(capture.process.jobs[0], "job"), + mechanism=ObservationMechanism.JOB_STATE, coverage=Coverage.PARTIAL, exists=True, **base)] + return [] + + +def _settle_background(log: Any, capture: DispatchCapture, result: dict) -> None: + """Settle a RUNNING launch claim from an admitted read of its exact job. + + Linkage is the Wave 3 launch generation plus owner/request/thread, already + validated by ``job_from_record`` at admission. Job completion is execution + evidence for that claim; it verifies no postcondition. + """ + job_facts = result.get("job") + if not isinstance(job_facts, dict) or len(capture.process.jobs) != 1: + return + status = job_facts.get("status") + if status not in _JOB_SETTLED: + return + job = capture.process.jobs[0] + lineage = ("process_launch", "native:containment", job.owner, job.request_id, job.thread_id, job.generation) + owner = log if any(any(ref.kind is ResourceKind.PROCESS_LAUNCH and ref.location == lineage + for ref in c.dependencies) for c in log.history().claims) else None + if owner is None and log.path is not None: + # Background continuation: the launch was claimed by an earlier run. + from src.agent_runtime.effect_log import EffectLog, EffectPersistenceError + indexed = EffectLog.launch_owner(job.generation, directory=log.path.parent) + if indexed is not None: + try: + owner = EffectLog.open(indexed[0], directory=log.path.parent) + except (EffectPersistenceError, ValueError): + owner = None + if owner is None: + return + history = owner.history() + for claim in history.claims: + if not any(ref.kind is ResourceKind.PROCESS_LAUNCH and ref.location == lineage for ref in claim.dependencies): + continue + latest = history.latest_outcome(claim.effect_id) + if latest is None or latest.execution is not ExecutionOutcome.RUNNING: + continue + code = job_facts.get("exit_code") + code = code if type(code) is int else None + if job_facts.get("timed_out") is True: + execution = ExecutionOutcome.TIMED_OUT + elif job_facts.get("killed") is True: + execution = ExecutionOutcome.CANCELLED + elif status == "done" and code == 0 and job_facts.get("died") is not True: + execution = ExecutionOutcome.REPORTED_SUCCESS + else: + execution = ExecutionOutcome.FAILED + facts = ProducerFacts(exit_code=code, timed_out=job_facts.get("timed_out") is True, job_state=status) + owner.outcome(effect_id=claim.effect_id, execution=execution, impact=Impact.POSSIBLE, facts=facts, + cleanup=CleanupState.UNKNOWN, execution_id=latest.execution_id) diff --git a/src/agent_runtime/effect_log.py b/src/agent_runtime/effect_log.py index 34ead8372..7addcb0b1 100644 --- a/src/agent_runtime/effect_log.py +++ b/src/agent_runtime/effect_log.py @@ -17,6 +17,7 @@ from pathlib import Path import re import stat import threading +import weakref from typing import Any from src.constants import DATA_DIR @@ -41,11 +42,17 @@ def effects_dir() -> Path: class EffectLog: + # Logs still owned by a live run in this process. A later turn appends to + # the same object rather than a second copy with its own sequence counter. + _LIVE: "weakref.WeakValueDictionary[tuple[str, str], EffectLog]" = weakref.WeakValueDictionary() + def __init__(self, run_id: str, *, durable: bool = True, directory: str | os.PathLike | None = None) -> None: if not isinstance(run_id, str) or not _RUN_ID.fullmatch(run_id): raise ValueError("Effect log requires a server-generated run identifier") self.run_id = run_id self.path = (Path(directory) if directory is not None else effects_dir()) / f"{run_id}.jsonl" if durable else None + if self.path is not None: + self._LIVE[(str(self.path.parent), run_id)] = self self._claims: list[EffectClaim] = [] self._outcomes: list[EffectOutcome] = [] self._observations: list[Observation] = [] @@ -167,6 +174,60 @@ class EffectLog: default=0) return log + @classmethod + def open(cls, run_id: str, *, directory: str | os.PathLike | None = None) -> "EffectLog": + """The live log for a run, or its replayed durable history. + + A log that is not live belongs to a finished or crashed run, so its + unsettled claims are recovered as interrupted before any append. + """ + base = Path(directory) if directory is not None else effects_dir() + live = cls._LIVE.get((str(base), run_id)) + if live is not None: + return live + log = cls.load(run_id, directory=base) + log.recover_interrupted() + return log + + # -- background launch lineage ------------------------------------------ + + def index_launch(self, generation: str, effect_id: str) -> None: + """Durably map an exact Wave 3 launch generation to its claim.""" + if self.path is None: + return + if not _RUN_ID.fullmatch(generation or ""): + raise ValueError("Malformed launch generation") + target = self.path.parent / f"launch-{generation}.json" + temporary = target.with_suffix(".tmp") + data = json.dumps({"run_id": self.run_id, "effect_id": effect_id}, sort_keys=True).encode() + flags = os.O_WRONLY | os.O_CREAT | os.O_TRUNC | getattr(os, "O_NOFOLLOW", 0) + descriptor = os.open(temporary, flags, 0o600) + try: + os.write(descriptor, data) + os.fsync(descriptor) + finally: + os.close(descriptor) + os.replace(temporary, target) + + @staticmethod + def launch_owner(generation: str, *, directory: str | os.PathLike | None = None) -> tuple[str, str] | None: + if not _RUN_ID.fullmatch(generation or ""): + return None + base = Path(directory) if directory is not None else effects_dir() + try: + descriptor = os.open(base / f"launch-{generation}.json", os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0)) + with os.fdopen(descriptor, "rb") as stream: + if os.fstat(stream.fileno()).st_nlink != 1: + return None + value = json.loads(stream.read(4096)) + except (OSError, ValueError): + return None + if (not isinstance(value, dict) or set(value) != {"run_id", "effect_id"} + or not isinstance(value["run_id"], str) or not _RUN_ID.fullmatch(value["run_id"]) + or not isinstance(value["effect_id"], str)): + return None + return value["run_id"], value["effect_id"] + def recover_interrupted(self) -> tuple[EffectOutcome, ...]: """Append INTERRUPTED outcomes for claims that never settled.""" with self._lock: diff --git a/src/agent_runtime/effects.py b/src/agent_runtime/effects.py index e69572eec..994557645 100644 --- a/src/agent_runtime/effects.py +++ b/src/agent_runtime/effects.py @@ -54,6 +54,7 @@ _SHA256 = re.compile(r"[a-f0-9]{64}") class ResourceKind(str, Enum): FILESYSTEM = "filesystem" PROCESS = "process" + PROCESS_LAUNCH = "process_launch" BACKGROUND_JOB = "background_job" OWNED = "owned" EXTERNAL = "external" @@ -108,6 +109,11 @@ class ResourceRef: """ if self.kind is not other.kind: return False + if self.kind is ResourceKind.OWNED: + # A collection binding ("*") covers every record it can create, + # list or change; specific records only overlap themselves. + return self.location[:-1] == other.location[:-1] and ( + self.location[-1] == other.location[-1] or "*" in (self.location[-1], other.location[-1])) if self.kind is not ResourceKind.FILESYSTEM: return self.location == other.location if self.location[:-1] != other.location[:-1]: @@ -153,6 +159,13 @@ def resource_ref(resource: Any, role: str) -> ResourceRef: location = ("process", resource.namespace, resource.owner, resource.request_id, resource.thread_id, str(ident.pid), ident.start_token, resource.role) return ResourceRef(ResourceKind.PROCESS, role, location, ident.start_token, _sha(resource.to_dict())) + if isinstance(resource, wave3.ProcessLaunchResource): + # The reservation generation is the exact launch -> job linkage that + # Wave 3 validates in ``job_from_record``. + location = ("process_launch", resource.namespace, resource.owner, resource.request_id, + resource.thread_id, resource.generation) + return ResourceRef(ResourceKind.PROCESS_LAUNCH, role, location, resource.generation, + _sha(resource.to_dict())) if isinstance(resource, wave3.BackgroundJobResource): location = ("background_job", resource.namespace, resource.owner, resource.request_id, resource.thread_id, resource.job_id, resource.generation) @@ -378,11 +391,13 @@ class ProducerFacts: job_state: str = "" remote_acknowledged: bool = False external: bool = False + # The producer reached its mutation stage before reporting failure. + mutation_attempted: bool = False def __post_init__(self) -> None: if self.exit_code is not None and type(self.exit_code) is not int: raise ValueError("Malformed producer exit code") - for name in ("timed_out", "output_truncated", "remote_acknowledged", "external"): + for name in ("timed_out", "output_truncated", "remote_acknowledged", "external", "mutation_attempted"): if type(getattr(self, name)) is not bool: raise ValueError("Malformed producer flag") for name in ("failure_kind", "job_state"): @@ -395,7 +410,7 @@ class ProducerFacts: return {"exit_code": self.exit_code, "timed_out": self.timed_out, "output_truncated": self.output_truncated, "failure_kind": self.failure_kind, "job_state": self.job_state, "remote_acknowledged": self.remote_acknowledged, - "external": self.external} + "external": self.external, "mutation_attempted": self.mutation_attempted} @classmethod def from_dict(cls, value: Any) -> "ProducerFacts": @@ -428,6 +443,7 @@ def producer_facts(result: Any) -> ProducerFacts: failure_kind=_label(result.get("failure_kind")), job_state=_label(job), external=external, + mutation_attempted=result.get("mutation_attempted") is True, ) diff --git a/src/agent_runtime/journal.py b/src/agent_runtime/journal.py index 1d2c8456b..200718bf5 100644 --- a/src/agent_runtime/journal.py +++ b/src/agent_runtime/journal.py @@ -7,6 +7,7 @@ from copy import deepcopy from dataclasses import dataclass, field, asdict from functools import wraps from inspect import signature +import logging from typing import Any from uuid import uuid4 @@ -67,6 +68,43 @@ class ActionJournal: workspace: str = '' observed_artifacts: tuple[str, ...] = () parent_run_id: str | None = None + # Durable Wave 4 effect log, shared across one run lineage; None disables. + effects: Any = field(default=None, repr=False, compare=False) + _dispatches: dict[str, Any] = field(default_factory=dict, repr=False, compare=False) + + def effect_entries(self) -> list[dict[str, Any]]: + """Effect assessments ordered against this journal's actions. + + Ordinal is the 1-based position of the action in this journal, so the + ledger can compare effects with receipt-derived evidence. Effects from + other journals in the lineage carry no ordinal here. + """ + if self.effects is None: + return [] + order = {action.action_id: index for index, action in enumerate(self.actions, 1)} + changes = {action.action_id: action.artifact_changes for action in self.actions} + history = self.effects.history() + entries = [] + for assessment in self.effects.assessments(): + claim = history.claim(assessment.effect_id) + outcome = history.latest_outcome(assessment.effect_id) + entries.append({ + 'ordinal': order.get(assessment.action_id), 'assessment': assessment, + 'tool': claim.operation.tool, 'unknown_scope': claim.unknown_scope, + 'paths': tuple(ref.location[-1] for ref in claim.impact_scope if ref.kind.value == 'filesystem'), + 'mutation_attempted': bool(outcome and outcome.facts.mutation_attempted), + 'artifact_changes': changes.get(assessment.action_id), + }) + return entries + + def partial_reads(self) -> tuple[str, ...]: + """Read actions in this journal whose admitted observation was partial.""" + if self.effects is None: + return () + mine = {action.action_id for action in self.actions} + return tuple(o.source_action_id for o in self.effects.history().observations + if o.source_action_id in mine and o.mechanism.value == 'filesystem_read' + and o.coverage.value == 'partial') def capture_versions(self, action: ActionReceipt) -> None: if self.workspace: @@ -138,6 +176,16 @@ def mark_authorized() -> None: def mark_dispatch() -> None: action = _ACTION.get() if action is not None and action.execution_id is None: + journal = _JOURNAL.get() + if journal is not None and journal.effects is not None: + # Durable claim first. If it cannot be persisted this raises and + # the action stays undispatched: the backend is never invoked. + from .effect_adapters import begin_effect + capture = begin_effect(journal, action) + journal._dispatches[action.action_id] = capture + if capture.claim is not None: + action.transition('effect_claimed', effect_id=capture.claim.effect_id, + sequence=capture.claim.sequence) mark_authorized() action.execution_id = action.action_id + ':execution:1' action.transition('dispatched', execution_id=action.execution_id) @@ -145,10 +193,32 @@ def mark_dispatch() -> None: async def dispatched(operation): """Record an actual backend invocation, distinct from router admission.""" - mark_dispatch() + try: + mark_dispatch() + except BaseException: + close = getattr(operation, 'close', None) + if close is not None: + close() # never invoked; do not leave an un-awaited coroutine + raise return await operation +def _settle(journal: ActionJournal | None, action: ActionReceipt, **outcome: Any) -> None: + if journal is None or journal.effects is None: + return + capture = journal._dispatches.pop(action.action_id, None) + if capture is None: + return + from .effect_adapters import settle_effect + try: + settle_effect(journal, action, capture, **outcome) + except Exception: # noqa: BLE001 - bookkeeping must not alter the tool result + # The claim stays unsettled (ATTEMPTED), which assesses as pending + # with possible impact: conservative, never a manufactured success. + journal.effects.degraded = True + logging.getLogger(__name__).warning('Effect outcome could not be recorded', exc_info=True) + + def mark_operation_started(backend: str, **details: Any) -> None: action = _ACTION.get() if action is not None: @@ -197,10 +267,14 @@ def record_action(func): action.finish({**result, 'blocked': True}) else: action.finish(result) + # Structured producer facts are projected here, before the + # receipt reduction drops them. + _settle(journal, action, result=result) return description, result except BaseException as exc: if action is not None: action.transition('interrupted', category=type(exc).__name__) + _settle(current_journal(), action, error=exc) raise finally: _ACTION.reset(token) diff --git a/src/agent_tools/bg_job_tools.py b/src/agent_tools/bg_job_tools.py index 8d3fd5c1a..0f7260025 100644 --- a/src/agent_tools/bg_job_tools.py +++ b/src/agent_tools/bg_job_tools.py @@ -44,6 +44,19 @@ def _status_label(rec: Dict[str, Any]) -> str: return status +def _job_facts(rec: Dict[str, Any]) -> Dict[str, Any]: + """Typed lifecycle facts from the exact admitted job record. + + Execution evidence only: completion of a job is not verification of any + filesystem, service or external state its command was meant to change. + """ + code = rec.get("exit_code") + return {"status": rec.get("status") if rec.get("status") in {"running", "done", "failed"} else "unknown", + "exit_code": code if type(code) is int else None, + "timed_out": rec.get("timed_out") is True, "killed": rec.get("killed") is True, + "died": rec.get("died") is True} + + def _row(rec: Dict[str, Any]) -> str: cmd = (rec.get("command") or "").strip().splitlines()[0][:80] return f"[{rec.get('id')}] {_status_label(rec)} | {_age(rec)} | {cmd}" @@ -101,17 +114,20 @@ class ManageBgJobsTool: if action in _KILL_ACTIONS: if rec.get("status") != "running": - return {"output": f"Job `{job_id}` already {_status_label(rec)}; nothing to kill.", "exit_code": 0} + return {"output": f"Job `{job_id}` already {_status_label(rec)}; nothing to kill.", "exit_code": 0, + "job": _job_facts(rec)} killed = bg_jobs.kill(job_id, expected=resource) if not killed or not killed.get("killed"): return {"error": f"Could not verify termination of background job `{job_id}`.", "exit_code": 1, "teardown": (killed or {}).get("teardown")} - return {"output": f"Killed background job `{job_id}` ({(killed or {}).get('command', '').splitlines()[0][:80]}).", "exit_code": 0} + return {"output": f"Killed background job `{job_id}` ({(killed or {}).get('command', '').splitlines()[0][:80]}).", "exit_code": 0, + "job": _job_facts(killed)} out = rec.get("output") or "(no output yet)" return { "output": f"Job `{job_id}` [{_status_label(rec)}, {_age(rec)}]\nCommand: {rec.get('command')}\n\nOutput:\n{out}", "exit_code": 0, + "job": _job_facts(rec), } return {"error": f"manage_bg_jobs: unknown action '{action}'. Use list, output, or kill.", "exit_code": 1} diff --git a/src/agent_tools/filesystem_tools.py b/src/agent_tools/filesystem_tools.py index 89d161d43..53854da61 100644 --- a/src/agent_tools/filesystem_tools.py +++ b/src/agent_tools/filesystem_tools.py @@ -156,20 +156,24 @@ class EditFileTool: if count > 1 and not replace_all: return original, None, f"not_unique:{count}" updated = original.replace(old, new) if replace_all else original.replace(old, new, 1) + attempted.append(True) with open(path, "w", encoding="utf-8", newline="") as f: f.write(updated) return original, updated, "ok" + # In-place rewrite: a failure after truncation may leave partial bytes. + attempted = [] + partial = lambda: {"mutation_attempted": True} if attempted else {} try: original, updated, status = await asyncio.to_thread(_apply) except FileNotFoundError: - return {"error": f"edit_file: {path}: not found (use write_file to create it)", "exit_code": 1} + return {"error": f"edit_file: {path}: not found (use write_file to create it)", "exit_code": 1, **partial()} except (IsADirectoryError, UnicodeDecodeError): - return {"error": f"edit_file: {path}: not an editable text file", "exit_code": 1} + return {"error": f"edit_file: {path}: not an editable text file", "exit_code": 1, **partial()} except PermissionError: - return {"error": f"edit_file: {path}: permission denied", "exit_code": 1} + return {"error": f"edit_file: {path}: permission denied", "exit_code": 1, **partial()} except OSError as e: - return {"error": f"edit_file: {path}: {e}", "exit_code": 1} + return {"error": f"edit_file: {path}: {e}", "exit_code": 1, **partial()} if status == "not_found": return {"error": f"edit_file: old_string not found in {path}. Read the file and match it exactly.", "exit_code": 1} @@ -332,6 +336,9 @@ class WriteFileTool: "exit_code": 1, "binary_artifact_preserved": target_existed, } + # This writer truncates in place. Once that stage is reached, a failure + # may leave a partial file; report it so effect evidence stays honest. + attempted = [] try: def _write(): old = "" @@ -343,14 +350,17 @@ class WriteFileTool: d = os.path.dirname(path) if d: os.makedirs(d, exist_ok=True) + attempted.append(True) with open(path, "w", encoding="utf-8") as f: f.write(body) return old, len(body) old_content, size = await asyncio.to_thread(_write) except PermissionError: - return {"error": f"write_file: {path}: permission denied", "exit_code": 1} + return {"error": f"write_file: {path}: permission denied", "exit_code": 1, + **({"mutation_attempted": True} if attempted else {})} except OSError as e: - return {"error": f"write_file: {path}: {e}", "exit_code": 1} + return {"error": f"write_file: {path}: {e}", "exit_code": 1, + **({"mutation_attempted": True} if attempted else {})} diff = _unified_diff(old_content, body, path) result = { "output": f"Wrote {size} bytes to {_display_tool_path(path)}", diff --git a/src/agent_tools/subprocess_tools.py b/src/agent_tools/subprocess_tools.py index 399755caf..fc34eb046 100644 --- a/src/agent_tools/subprocess_tools.py +++ b/src/agent_tools/subprocess_tools.py @@ -571,7 +571,7 @@ async def _run_owned_command(command, ctx: dict, *, tool: str, timeout: int, arg "stderr": _truncate(result.stderr, MAX_OUTPUT_CHARS)} if result.timed_out: return {**common, "error": f"{tool}: timed out after {timeout}s; process tree terminated.{capture_note}", - "exit_code": 124, "stdout": _truncate(result.stdout, MAX_OUTPUT_CHARS), + "exit_code": 124, "timed_out": True, "stdout": _truncate(result.stdout, MAX_OUTPUT_CHARS), "stderr": _truncate(result.stderr, MAX_OUTPUT_CHARS)} if tool == "python": child_failure = _python_child_runtime_failure(result.stdout, result.stderr, result.exit_code) diff --git a/tests/test_effect_resource_bindings.py b/tests/test_effect_resource_bindings.py new file mode 100644 index 000000000..23a378577 --- /dev/null +++ b/tests/test_effect_resource_bindings.py @@ -0,0 +1,227 @@ +"""Wave 4 effects through the real dispatcher and exact Wave 3 filesystem bindings.""" +from __future__ import annotations + +import asyncio +import hashlib +import json +import os + +import pytest + +from src import tool_execution +from src.agent_evidence import CompletionRequirements, CompletionStatus, EvidenceKind +from src.agent_runtime import effects as fx +from src.agent_runtime.authority import OperationGrant, RequestAuthority +from src.agent_runtime.completion import _ledger, completion_answer +from src.agent_runtime.effect_log import EffectLog +from src.agent_runtime.journal import ActionJournal, bind_journal +from src.agent_tools import TOOL_HANDLERS +from src.tool_capabilities import ToolRunSecurityContext +from src.tool_types import ToolBlock + + +@pytest.fixture +def ws(tmp_path, monkeypatch): + work = tmp_path / "ws" + work.mkdir() + monkeypatch.setattr(tool_execution, "_owner_is_admin", lambda owner: True) + return work + + +@pytest.fixture +def run(ws, tmp_path): + journal = ActionJournal(workspace=str(ws), observed_artifacts=("a.txt",)) + journal.effects = EffectLog(journal.run_id, directory=tmp_path / "fx") + authority = RequestAuthority("request", "alice", "thread", str(ws), tuple( + OperationGrant(tool) for tool in ("write_file", "read_file", "edit_file", "apply_patch", "ls"))) + + async def call(tool, args): + content = args if isinstance(args, str) else json.dumps(args) + with bind_journal(journal): + return await tool_execution.execute_tool_block( + ToolBlock(tool, content), owner="alice", session_id="thread", workspace=str(ws), + security_context=ToolRunSecurityContext(external_untrusted_context_seen=False), + request_authority=authority) + + def go(tool, args): + return asyncio.run(call(tool, args)) + + go.journal = journal + return go + + +def sha(text): + return hashlib.sha256(text.encode()).hexdigest() + + +def verdicts(journal): + return [a.verdict for a in journal.effects.assessments()] + + +def ledger(journal, ws): + return _ledger(journal, CompletionRequirements(required_artifacts=("a.txt",), workspace_root=str(ws))) + + +def records(journal): + return [json.loads(line)["type"] for line in journal.effects.path.read_text().splitlines()] + + +def test_claim_is_durable_before_the_producer_runs(run, monkeypatch): + seen = [] + original = TOOL_HANDLERS["write_file"] + + async def spy(content, ctx): + seen.append(records(run.journal)) + return await original(content, ctx) + + monkeypatch.setitem(TOOL_HANDLERS, "write_file", spy) + _, result = run("write_file", {"path": "a.txt", "content": "hello\n"}) + assert result["exit_code"] == 0 + assert seen == [["claim"]], "the claim must be on disk, with no outcome, at backend invocation" + claim = run.journal.effects.history().claims[0] + assert [ref.role for ref in claim.impact_scope] == ["destination"] + assert claim.obligations[0].predicate is fx.Predicate.CONTENT_SHA256 + assert claim.obligations[0].expected == sha("hello\n") + receipt = run.journal.actions[0] + stages = [t["stage"] for t in receipt.transitions] + assert stages.index("effect_claimed") < stages.index("dispatched") + + +def test_persistence_failure_refuses_invocation(run, monkeypatch, tmp_path): + called = [] + monkeypatch.setitem(TOOL_HANDLERS, "write_file", lambda content, ctx: called.append(1)) + blocker = tmp_path / "blocker" + blocker.write_text("x") + run.journal.effects = EffectLog(run.journal.run_id, directory=blocker) + description, result = run("write_file", {"path": "a.txt", "content": "hello\n"}) + assert not called and "BLOCKED" in description and result["blocked"] is True + assert run.journal.actions[0].execution_id is None + assert not (tmp_path / "ws" / "a.txt").exists() + + +def test_execution_success_then_complete_readback_verifies(run): + run("write_file", {"path": "a.txt", "content": "hello\n"}) + assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED] + run("read_file", {"path": "a.txt"}) + assert verdicts(run.journal) == [fx.EffectVerdict.VERIFIED] + observation = run.journal.effects.history().observations[0] + assert observation.source_action_id == run.journal.actions[1].action_id + assert observation.content_sha256 == sha("hello\n") + + +def test_partial_read_neither_verifies_nor_validates(run, ws): + run("write_file", {"path": "a.txt", "content": "one\ntwo\n"}) + run("read_file", {"path": "a.txt", "offset": 1, "limit": 1}) + assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED] + current = ledger(run.journal, ws) + validations = [e for e in current.events if e.kind == EvidenceKind.ARTIFACT_VALIDATION] + assert validations and not any(e.authoritative for e in validations) + + +def test_later_mutation_makes_earlier_verification_stale(run): + run("write_file", {"path": "a.txt", "content": "one\n"}) + run("read_file", {"path": "a.txt"}) + run("write_file", {"path": "a.txt", "content": "two\n"}) + assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED, fx.EffectVerdict.UNVERIFIED] + run("read_file", {"path": "a.txt"}) + assert verdicts(run.journal) == [fx.EffectVerdict.CONTRADICTED, fx.EffectVerdict.VERIFIED] + + +def test_unrecorded_change_is_contradicted_and_fails_completion(run, ws): + run("write_file", {"path": "a.txt", "content": "hello\n"}) + (ws / "a.txt").write_text("tampered\n") + run("read_file", {"path": "a.txt"}) + assert verdicts(run.journal) == [fx.EffectVerdict.CONTRADICTED] + decision = ledger(run.journal, ws).evaluate() + assert decision.status == CompletionStatus.FAILED and decision.missing_artifacts == ("a.txt",) + + +def test_cancelled_write_unsettles_an_earlier_success(run, ws, monkeypatch): + run("write_file", {"path": "a.txt", "content": "hello\n"}) + assert ledger(run.journal, ws).evaluate().can_complete + + async def cancelled(content, ctx): + raise asyncio.CancelledError + + monkeypatch.setitem(TOOL_HANDLERS, "write_file", cancelled) + with pytest.raises(asyncio.CancelledError): + run("write_file", {"path": "a.txt", "content": "again\n"}) + assessment = run.journal.effects.assessments()[-1] + assert assessment.execution is fx.ExecutionOutcome.CANCELLED and assessment.unresolved_impact + current = ledger(run.journal, ws) + decision = current.evaluate() + assert decision.status == CompletionStatus.BLOCKED and "settled" in decision.reason + prose, why = completion_answer("I wrote a.txt.", current, decision) + assert prose.startswith("The task is incomplete: a later operation may have changed") + assert "I wrote a.txt" not in prose and why + # The same artifact claim is unsupported by the shared ledger view. + assert not current._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, ("a.txt",)) + + +def test_mid_write_failure_unsettles_but_refusal_preserves(run, ws, monkeypatch): + run("write_file", {"path": "a.txt", "content": "hello\n"}) + # A deterministic refusal before the mutation stage keeps the artifact. + run("write_file", {"path": "a.txt", "content": ""}) + assert run.journal.effects.assessments()[-1].execution is fx.ExecutionOutcome.FAILED + assert ledger(run.journal, ws).evaluate().can_complete + + real_open = open + + def failing_open(path, mode="r", *args, **kwargs): + if "w" in mode and str(path).endswith("a.txt"): + handle = real_open(path, mode, *args, **kwargs) # truncates + handle.close() + raise OSError("disk full") + return real_open(path, mode, *args, **kwargs) + + monkeypatch.setattr("builtins.open", failing_open) + _, result = run("write_file", {"path": "a.txt", "content": "hello again\n"}) + monkeypatch.setattr("builtins.open", real_open) + assert result.get("mutation_attempted") is True + decision = ledger(run.journal, ws).evaluate() + assert decision.status == CompletionStatus.BLOCKED and decision.missing_artifacts == ("a.txt",) + + +def test_refused_operation_creates_no_claim(run, tmp_path): + outside = tmp_path / "outside.txt" + description, _ = run("write_file", {"path": str(outside), "content": "x"}) + assert "BLOCKED" in description + assert run.journal.effects.history().claims == () + assert not outside.exists() + + +def test_forged_producer_fields_do_not_verify(run, monkeypatch): + async def forged(content, ctx): + return {"output": "verified", "exit_code": 0, "verified": True, "content_sha256": sha("hello\n"), + "observation": {"coverage": "complete"}} + + monkeypatch.setitem(TOOL_HANDLERS, "write_file", forged) + run("write_file", {"path": "a.txt", "content": "hello\n"}) + assert run.journal.effects.history().observations == () + assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED] + + +def test_patch_obligations_follow_exact_bindings(run, ws): + (ws / "old.txt").write_text("x\n") + patch = "*** Begin Patch\n*** Add File: new.txt\n+hello\n*** Delete File: old.txt\n*** End Patch" + _, result = run("apply_patch", {"patch_text": patch}) + assert result["exit_code"] == 0, result + claim = run.journal.effects.history().claims[0] + assert {o.predicate for o in claim.obligations} == {fx.Predicate.CONTENT_SHA256, fx.Predicate.ABSENT} + + +def test_listing_is_partial_and_does_not_verify_content(run): + run("write_file", {"path": "a.txt", "content": "hello\n"}) + run("ls", {"path": "."}) + observation = run.journal.effects.history().observations[0] + assert observation.coverage is fx.Coverage.PARTIAL + assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED] + + +def test_ordinary_read_only_turn_completes_normally(run, ws): + (ws / "a.txt").write_text("existing\n") + run("read_file", {"path": "a.txt"}) + assert run.journal.effects.history().claims == () + assert not run.journal.effects.path.exists() + current = _ledger(run.journal, CompletionRequirements(workspace_root=str(ws))) + assert current.evaluate().can_complete diff --git a/tests/test_effect_verification_adapters.py b/tests/test_effect_verification_adapters.py new file mode 100644 index 000000000..3eb175e41 --- /dev/null +++ b/tests/test_effect_verification_adapters.py @@ -0,0 +1,331 @@ +"""Wave 4 adapters for process/background, owned, external and browser bindings. + +These drive ``begin_effect``/``settle_effect`` with real Wave 3 bound-operation +objects. The dispatcher's contextvar capture is replaced by the same objects so +that each producer family can be exercised without its live backend. +""" +from __future__ import annotations + +import asyncio +import gc +import json + +import pytest + +from src import browser_identity +from src.agent_evidence import CompletionRequirements, CompletionStatus +from src.agent_runtime import effect_adapters as adapters +from src.agent_runtime import effects as fx +from src.agent_runtime.authority import ExactOperation +from src.agent_runtime.completion import _ledger +from src.agent_runtime.effect_log import EffectLog +from src.agent_runtime.journal import ActionJournal +from src.agent_runtime.owned_resources import BoundOwnedOperation +from src.agent_runtime.process_resources import BoundProcessOperation, digest as process_digest +from src.agent_runtime.remote_resources import BoundBackendOperation +from src.agent_runtime.resources import ( + BackgroundJobResource, BrowserPageResource, BrowserSessionObservation, BrowserSessionResource, ExternalResource, + FilesystemRoot, NativeBackendResource, OwnedResource, ProcessLaunchResource, ProcessLaunchScope, ProcessResource, +) +from src.process_lifecycle import ProcessIdentity +from src.tool_types import ToolBlock + + +GENERATION = "c" * 32 + + +@pytest.fixture +def store(tmp_path): + return tmp_path / "fx" + + +def journal_for(store, parent=None): + journal = ActionJournal(parent_run_id=parent.run_id if parent else None) + journal.effects = parent.effects if parent else EffectLog(journal.run_id, directory=store) + return journal + + +def act(journal, monkeypatch, capture, tool="bash", content="{}", *, result=None, error=None): + """One admitted action: claim at dispatch, then settle with a producer result.""" + action = journal.propose(ToolBlock(tool, content)) + monkeypatch.setattr(adapters, "capture_dispatch", lambda: capture) + captured = adapters.begin_effect(journal, action) + action.execution_id = action.action_id + ":execution:1" + if result is not None: + action.finish(result) + adapters.settle_effect(journal, action, captured, result=result, error=error) + return action, captured + + +def launch_capture(tmp_path, generation=GENERATION): + workspace = tmp_path / "ws" + workspace.mkdir(exist_ok=True) + operation = ExactOperation.normalize("bash", "#!bg\nsleep 1") + scope = ProcessLaunchScope(NativeBackendResource("bash"), FilesystemRoot.seal(str(workspace)), frozenset({"filesystem"})) + launch = ProcessLaunchResource("native:containment", "alice", "request", "thread", generation, "bash", + process_digest(operation.input), scope, "b" * 64) + return adapters.DispatchCapture(process=BoundProcessOperation(operation, "request", "alice", "thread", launch)) + + +def job_capture(action="status", generation=GENERATION, job_id="job1"): + supervisor = ProcessResource("native:bg_jobs", "alice", "request", "thread", + ProcessIdentity(4242, "boot:1:100", None), "supervisor", job_id, "cont-1") + job = BackgroundJobResource("native:bg_jobs", job_id, generation, "alice", "request", "thread", "cont-1", + (supervisor,)) + operation = ExactOperation.normalize("manage_bg_jobs", json.dumps({"action": action, "job_id": job_id})) + return adapters.DispatchCapture(process=BoundProcessOperation(operation, "request", "alice", "thread", jobs=(job,))) + + +def job_result(status, exit_code=None, **flags): + return {"output": "Job report says everything succeeded and was verified.", "exit_code": 0, + "job": {"status": status, "exit_code": exit_code, "timed_out": False, "killed": False, + "died": False, **flags}} + + +# -- process / background ---------------------------------------------------- + +def test_process_exit_is_execution_evidence_not_a_postcondition(tmp_path, store, monkeypatch): + journal = journal_for(store) + act(journal, monkeypatch, launch_capture(tmp_path), "bash", "ls", + result={"output": "ok", "exit_code": 0, "teardown": {"dead": True}}) + claim = journal.effects.history().claims[0] + assert claim.unknown_scope, "an arbitrary command has unknown impact scope" + assert [ref.kind for ref in claim.dependencies] == [fx.ResourceKind.PROCESS_LAUNCH] + assessment = journal.effects.assessments()[0] + assert (assessment.execution, assessment.verdict, assessment.cleanup) == ( + fx.ExecutionOutcome.REPORTED_SUCCESS, fx.EffectVerdict.UNVERIFIED, fx.CleanupState.VERIFIED) + + +@pytest.mark.parametrize("result,execution,cleanup", [ + ({"error": "timed out", "exit_code": 124, "timed_out": True, "teardown": {"dead": True}}, + fx.ExecutionOutcome.TIMED_OUT, fx.CleanupState.VERIFIED), + ({"error": "teardown", "exit_code": 1, "failure_kind": "process_teardown_failed", "teardown": {"dead": False}}, + fx.ExecutionOutcome.FAILED, fx.CleanupState.FAILED), + ({"output": "", "exit_code": 0, "status": "running", "detached": True, "containment": {"external": True}}, + fx.ExecutionOutcome.RUNNING, fx.CleanupState.UNKNOWN), +]) +def test_process_outcomes_are_preserved_separately(tmp_path, store, monkeypatch, result, execution, cleanup): + journal = journal_for(store) + act(journal, monkeypatch, launch_capture(tmp_path), "bash", "x", result=result) + assessment = journal.effects.assessments()[0] + assert (assessment.execution, assessment.cleanup) == (execution, cleanup) + assert assessment.unresolved_impact + + +def test_cleanup_failure_after_command_unsettles_required_artifact(tmp_path, store, monkeypatch): + workspace = tmp_path / "ws" + journal = journal_for(store) + journal.workspace, journal.observed_artifacts = str(workspace), ("out.txt",) + write = journal.propose(ToolBlock("write_file", json.dumps({"path": "out.txt", "content": "x"}))) + write.execution_id = write.action_id + ":execution:1" + write.finish({"output": "Wrote", "exit_code": 0}) + (workspace).mkdir(exist_ok=True) + (workspace / "out.txt").write_text("x") + requirements = CompletionRequirements(required_artifacts=("out.txt",), workspace_root=str(workspace)) + assert _ledger(journal, requirements).evaluate().can_complete + act(journal, monkeypatch, launch_capture(tmp_path), "bash", "x", + result={"error": "teardown", "exit_code": 1, "failure_kind": "process_teardown_failed", + "teardown": {"dead": False}}) + decision = _ledger(journal, requirements).evaluate() + assert decision.status == CompletionStatus.BLOCKED and decision.missing_artifacts == ("out.txt",) + + +def test_background_launch_is_running_not_completed_work(tmp_path, store, monkeypatch): + journal = journal_for(store) + act(journal, monkeypatch, launch_capture(tmp_path), "bash", "#!bg\nsleep 1", + result={"output": "Started background job `job1`.", "exit_code": 0, "bg_job_id": "job1"}) + assessment = journal.effects.assessments()[0] + assert (assessment.execution, assessment.verdict) == (fx.ExecutionOutcome.RUNNING, fx.EffectVerdict.PENDING) + + +def test_exact_job_read_settles_launch_across_a_continuation_run(tmp_path, store, monkeypatch): + first = journal_for(store) + act(first, monkeypatch, launch_capture(tmp_path), "bash", "#!bg\nsleep 1", + result={"output": "Started", "exit_code": 0, "bg_job_id": "job1"}) + launch_effect = first.effects.history().claims[0] + # A still-running job does not settle anything. + second = journal_for(store) + act(second, monkeypatch, job_capture(), "manage_bg_jobs", "{}", result=job_result("running")) + assert fx.assess(launch_effect, first.effects.history()).verdict is fx.EffectVerdict.PENDING + assert second.effects.history().observations[0].mechanism is fx.ObservationMechanism.JOB_STATE + + del first + gc.collect() # the launching run is gone: settle through its durable log + third = journal_for(store) + act(third, monkeypatch, job_capture(), "manage_bg_jobs", "{}", result=job_result("done", 0)) + reloaded = EffectLog.load(launch_effect.run_id, directory=store) + assessment = fx.assess(launch_effect, reloaded.history()) + # Delivered completion is execution evidence; the job's report prose is + # attributed content and verifies nothing. + assert (assessment.execution, assessment.verdict) == (fx.ExecutionOutcome.REPORTED_SUCCESS, + fx.EffectVerdict.UNVERIFIED) + + +def test_job_linkage_requires_the_exact_generation(tmp_path, store, monkeypatch): + journal = journal_for(store) + act(journal, monkeypatch, launch_capture(tmp_path), "bash", "#!bg\nsleep 1", + result={"output": "Started", "exit_code": 0, "bg_job_id": "job1"}) + # Same display job id, different launch generation: a replacement job. + act(journal, monkeypatch, job_capture(generation="d" * 32), "manage_bg_jobs", "{}", + result=job_result("done", 0)) + assert journal.effects.assessments()[0].execution is fx.ExecutionOutcome.RUNNING + + +def test_killed_job_settles_as_cancelled(tmp_path, store, monkeypatch): + journal = journal_for(store) + act(journal, monkeypatch, launch_capture(tmp_path), "bash", "#!bg\nsleep 1", + result={"output": "Started", "exit_code": 0, "bg_job_id": "job1"}) + act(journal, monkeypatch, job_capture("kill"), "manage_bg_jobs", "{}", + result={**job_result("failed", -9, killed=True), "output": "Killed"}) + kill_claim = journal.effects.history().claims[1] + assert [ref.kind for ref in kill_claim.impact_scope] == [fx.ResourceKind.BACKGROUND_JOB, fx.ResourceKind.PROCESS] + assert journal.effects.assessments()[0].execution is fx.ExecutionOutcome.CANCELLED + + +def test_running_background_work_unsettles_later_required_artifact(tmp_path, store, monkeypatch): + workspace = tmp_path / "ws" + workspace.mkdir() + (workspace / "out.txt").write_text("x") + journal = journal_for(store) + journal.workspace = str(workspace) + write = journal.propose(ToolBlock("write_file", json.dumps({"path": "out.txt", "content": "x"}))) + write.execution_id = write.action_id + ":execution:1" + write.finish({"output": "Wrote", "exit_code": 0}) + act(journal, monkeypatch, launch_capture(tmp_path), "bash", "#!bg\nsleep 1", + result={"output": "Started", "exit_code": 0, "bg_job_id": "job1"}) + requirements = CompletionRequirements(required_artifacts=("out.txt",), workspace_root=str(workspace)) + assert _ledger(journal, requirements).evaluate().status == CompletionStatus.BLOCKED + + +# -- owned records ------------------------------------------------------------ + +def owned_capture(tool, payload, record): + operation = ExactOperation.normalize(tool, json.dumps(payload)) + return adapters.DispatchCapture(owned=BoundOwnedOperation(operation, operation.input, "request", "alice", + "thread", (record,))) + + +def test_owned_mutation_claims_exact_record_and_stays_unverified(store, monkeypatch): + record = OwnedResource("notes", "alice", "thread", "notes", "n1", "rev-1") + journal = journal_for(store) + act(journal, monkeypatch, owned_capture("manage_notes", {"action": "update", "id": "n1"}, record), + "manage_notes", result={"output": "Note updated and verified.", "exit_code": 0}) + claim = journal.effects.history().claims[0] + assert claim.impact_scope == (fx.resource_ref(record, "record"),) + assert journal.effects.assessments()[0].verdict is fx.EffectVerdict.UNVERIFIED + + +def test_same_display_id_new_revision_does_not_inherit_freshness(store, monkeypatch): + journal = journal_for(store) + old = OwnedResource("vault", "alice", "thread", "vault", "rec", "rev-1") + new = OwnedResource("vault", "alice", "thread", "vault", "rec", "rev-2") + act(journal, monkeypatch, owned_capture("vault_get", {"id": "rec"}, old), "vault_get", + result={"output": "secret", "exit_code": 0}) + act(journal, monkeypatch, owned_capture("vault_get", {"id": "rec"}, new), "vault_get", + result={"output": "secret", "exit_code": 0}) + history = journal.effects.history() + first, second = history.observations + assert history.claims == () + assert fx.freshness(first, history) is fx.Freshness.STALE + assert fx.freshness(second, history) is fx.Freshness.FRESH + + +# -- external / MCP ------------------------------------------------------------- + +def test_remote_success_is_acknowledgement_not_state(store, monkeypatch): + remote = ExternalResource("mcp", "endpoint", "server", "tool", "inc-1") + bound = BoundBackendOperation(remote, "request", "alice", "thread", "mcp__server__tool", "{}") + journal = journal_for(store) + act(journal, monkeypatch, adapters.DispatchCapture(backend=bound), "mcp__server__tool", + result={"output": "Successfully created and verified the record.", "exit_code": 0}) + claim = journal.effects.history().claims[0] + assert claim.external and claim.impact_scope == (fx.resource_ref(remote, "backend"),) + outcome = journal.effects.history().outcomes[0] + assert outcome.facts.remote_acknowledged and outcome.facts.external + assert outcome.cleanup is fx.CleanupState.UNKNOWN + assert journal.effects.history().observations == () + assert journal.effects.assessments()[0].verdict is fx.EffectVerdict.UNVERIFIED + + +def test_remote_failure_after_send_may_have_changed_state(store, monkeypatch): + remote = ExternalResource("mcp", "endpoint", "server", "tool", "inc-1") + bound = BoundBackendOperation(remote, "request", "alice", "thread", "mcp__server__tool", "{}") + journal = journal_for(store) + act(journal, monkeypatch, adapters.DispatchCapture(backend=bound), "mcp__server__tool", + error=TimeoutError("transport closed after send")) + assessment = journal.effects.assessments()[0] + assert assessment.execution is fx.ExecutionOutcome.INTERRUPTED and assessment.unresolved_impact + + +# -- browser session metadata only -------------------------------------------- + +def session(monkeypatch, incarnation_seed="1"): + monkeypatch.setattr(browser_identity, "PRODUCER_HASHES", {"linux-x64": "e" * 64}) + values = {"producer_namespace": "native:agent-browser", "producer_version": "0.35.0", "platform": "linux-x64", + "binary_sha256": "e" * 64, "configuration_digest": "1" * 64, "session_key": "ody-" + "a" * 24, + "daemon": {"pid": 4321, "start_token": "boot:" + incarnation_seed, "pgid": 4321}, + "browser_instance_digest": incarnation_seed * 64} + observation = BrowserSessionObservation(**{**values, "daemon": ProcessIdentity(4321, "boot:" + incarnation_seed, 4321), + "session_incarnation": browser_identity.incarnation(values)}) + return BrowserSessionResource("alice", "thread", observation) + + +def test_browser_session_info_is_lifecycle_observation_only(store, monkeypatch): + journal = journal_for(store) + operation = ExactOperation.normalize("private_browser", json.dumps({"action": "session_info"})) + first = browser_identity.BoundBrowserOperation(operation, "request", "alice", "thread", session(monkeypatch, "1")) + replaced = browser_identity.BoundBrowserOperation(operation, "request", "alice", "thread", session(monkeypatch, "2")) + for bound in (first, replaced): + act(journal, monkeypatch, adapters.DispatchCapture(browser=bound), "private_browser", + result={"output": "{}", "exit_code": 0, "executed": True, "browser_page_operations_supported": False}) + history = journal.effects.history() + assert history.claims == () + assert {o.mechanism for o in history.observations} == {fx.ObservationMechanism.BROWSER_SESSION} + # Session replacement never transfers freshness to the new session. + assert fx.freshness(history.observations[0], history) is fx.Freshness.STALE + # A session observation decides no file/record/remote postcondition. + assert all(o.coverage is fx.Coverage.PARTIAL for o in history.observations) + + +def test_browser_page_binding_never_becomes_effect_scope(store, monkeypatch): + journal = journal_for(store) + owner_session = session(monkeypatch) + page = BrowserPageResource(owner_session, "A" * 32, "loader") + operation = ExactOperation.normalize("private_browser", json.dumps({"action": "session_info"})) + bound = browser_identity.BoundBrowserOperation(operation, "request", "alice", "thread", owner_session, page) + act(journal, monkeypatch, adapters.DispatchCapture(browser=bound), "private_browser", + result={"output": "{}", "exit_code": 0}) + claim = journal.effects.history().claims[0] + assert claim.unknown_scope and journal.effects.history().observations == () + with pytest.raises(TypeError): + fx.resource_ref(page, "target") + + +# -- lineage -------------------------------------------------------------------- + +def test_child_effects_share_lineage_order_and_invalidate_parent_evidence(tmp_path, store, monkeypatch): + parent = journal_for(store) + old = OwnedResource("vault", "alice", "thread", "vault", "rec", "rev-1") + act(parent, monkeypatch, owned_capture("vault_get", {"id": "rec"}, old), "vault_get", + result={"output": "x", "exit_code": 0}) + child = journal_for(store, parent=parent) + act(child, monkeypatch, launch_capture(tmp_path), "bash", "x", result={"output": "", "exit_code": 0}) + history = parent.effects.history() + claim = history.claims[0] + assert (claim.run_id, claim.parent_run_id) == (child.run_id, parent.run_id) + # The child's unknown-scope command may have changed the parent's record. + assert fx.freshness(history.observations[0], history) is fx.Freshness.STALE + + +def test_classification_failure_claims_unknown_scope(store, monkeypatch): + journal = journal_for(store) + monkeypatch.setattr(adapters, "classify", lambda capture: (_ for _ in ()).throw(KeyError("bug"))) + act(journal, monkeypatch, adapters.DispatchCapture(), "anything", result={"output": "", "exit_code": 0}) + assert journal.effects.history().claims[0].unknown_scope + + +def test_cancellation_is_recorded_without_inventing_a_result(tmp_path, store, monkeypatch): + journal = journal_for(store) + act(journal, monkeypatch, launch_capture(tmp_path), "bash", "x", error=asyncio.CancelledError()) + assessment = journal.effects.assessments()[0] + assert (assessment.execution, assessment.cleanup) == (fx.ExecutionOutcome.CANCELLED, fx.CleanupState.UNKNOWN) From 3953ea244463775e68d2f1eb795c5df18a21baaa Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Fri, 2 Oct 2026 20:10:46 +0100 Subject: [PATCH 04/15] feat(runtime): settle background launch effects from validated job lifecycle The background monitor already validates the exact Wave 3 job linkage (job_from_record + validate_job) before continuing a session. At that point it now records the job's settlement against the durable launch claim through the launch-generation index, using typed lifecycle facts from the server-owned record. Settlement is idempotent across deferred retries, is execution evidence only, and never reads the delivered output: the injected report stays attributed content. Failure to record leaves the claim running/unknown and never blocks the follow-up. --- src/agent_runtime/effect_adapters.py | 28 ++++++++++++++++------ src/agent_tools/bg_job_tools.py | 8 +++---- src/bg_monitor.py | 17 +++++++++++++ tests/test_effect_verification_adapters.py | 22 +++++++++++++++++ 4 files changed, 64 insertions(+), 11 deletions(-) diff --git a/src/agent_runtime/effect_adapters.py b/src/agent_runtime/effect_adapters.py index bd8ba3d50..34f1fbfc7 100644 --- a/src/agent_runtime/effect_adapters.py +++ b/src/agent_runtime/effect_adapters.py @@ -330,20 +330,34 @@ def _settle_background(log: Any, capture: DispatchCapture, result: dict) -> None job_facts = result.get("job") if not isinstance(job_facts, dict) or len(capture.process.jobs) != 1: return + settle_background_job(capture.process.jobs[0], job_facts, log=log) + + +def settle_background_job(job: Any, job_facts: Any, *, log: Any = None) -> None: + """Settle the RUNNING launch claim of one exact, Wave 3-validated job. + + ``job`` must be a ``BackgroundJobResource`` the caller obtained through + Wave 3 validation (an admitted job read, or the monitor's + ``job_from_record``/``validate_job``). ``job_facts`` are typed lifecycle + facts from that server-owned record; delivered output is never consulted. + """ + from src.agent_runtime.effect_log import EffectLog, EffectPersistenceError, effects_dir + from src.agent_runtime.resources import BackgroundJobResource + if not isinstance(job, BackgroundJobResource) or not isinstance(job_facts, dict): + return status = job_facts.get("status") if status not in _JOB_SETTLED: return - job = capture.process.jobs[0] lineage = ("process_launch", "native:containment", job.owner, job.request_id, job.thread_id, job.generation) - owner = log if any(any(ref.kind is ResourceKind.PROCESS_LAUNCH and ref.location == lineage - for ref in c.dependencies) for c in log.history().claims) else None - if owner is None and log.path is not None: + owner = log if log is not None and any(any(ref.kind is ResourceKind.PROCESS_LAUNCH and ref.location == lineage + for ref in c.dependencies) for c in log.history().claims) else None + if owner is None: # Background continuation: the launch was claimed by an earlier run. - from src.agent_runtime.effect_log import EffectLog, EffectPersistenceError - indexed = EffectLog.launch_owner(job.generation, directory=log.path.parent) + directory = log.path.parent if log is not None and log.path is not None else effects_dir() + indexed = EffectLog.launch_owner(job.generation, directory=directory) if indexed is not None: try: - owner = EffectLog.open(indexed[0], directory=log.path.parent) + owner = EffectLog.open(indexed[0], directory=directory) except (EffectPersistenceError, ValueError): owner = None if owner is None: diff --git a/src/agent_tools/bg_job_tools.py b/src/agent_tools/bg_job_tools.py index 0f7260025..abab4d7b5 100644 --- a/src/agent_tools/bg_job_tools.py +++ b/src/agent_tools/bg_job_tools.py @@ -44,7 +44,7 @@ def _status_label(rec: Dict[str, Any]) -> str: return status -def _job_facts(rec: Dict[str, Any]) -> Dict[str, Any]: +def job_lifecycle_facts(rec: Dict[str, Any]) -> Dict[str, Any]: """Typed lifecycle facts from the exact admitted job record. Execution evidence only: completion of a job is not verification of any @@ -115,19 +115,19 @@ class ManageBgJobsTool: if action in _KILL_ACTIONS: if rec.get("status") != "running": return {"output": f"Job `{job_id}` already {_status_label(rec)}; nothing to kill.", "exit_code": 0, - "job": _job_facts(rec)} + "job": job_lifecycle_facts(rec)} killed = bg_jobs.kill(job_id, expected=resource) if not killed or not killed.get("killed"): return {"error": f"Could not verify termination of background job `{job_id}`.", "exit_code": 1, "teardown": (killed or {}).get("teardown")} return {"output": f"Killed background job `{job_id}` ({(killed or {}).get('command', '').splitlines()[0][:80]}).", "exit_code": 0, - "job": _job_facts(killed)} + "job": job_lifecycle_facts(killed)} out = rec.get("output") or "(no output yet)" return { "output": f"Job `{job_id}` [{_status_label(rec)}, {_age(rec)}]\nCommand: {rec.get('command')}\n\nOutput:\n{out}", "exit_code": 0, - "job": _job_facts(rec), + "job": job_lifecycle_facts(rec), } return {"error": f"manage_bg_jobs: unknown action '{action}'. Use list, output, or kill.", "exit_code": 1} diff --git a/src/bg_monitor.py b/src/bg_monitor.py index d8e3288ea..790783bc9 100644 --- a/src/bg_monitor.py +++ b/src/bg_monitor.py @@ -36,6 +36,22 @@ def _background_result_message(rec): return untrusted_context_message("background job output", inject) +def _settle_launch_effect(resource, rec): + """Record the exact job's settlement against its durable launch claim. + + Uses only the Wave 3-validated job identity and typed lifecycle facts from + the server-owned record. Settlement is execution evidence; the delivered + output remains attributed content and verifies nothing. Best-effort: a + failure leaves the claim running/unknown and never blocks the follow-up. + """ + try: + from src.agent_runtime.effect_adapters import settle_background_job + from src.agent_tools.bg_job_tools import job_lifecycle_facts + settle_background_job(resource, job_lifecycle_facts(rec)) + except Exception as error: # noqa: BLE001 + logger.warning("bg-followup: effect settlement for %s was not recorded: %s", rec.get("id"), error) + + async def _drain_agent(sess, messages, request_authority=None): """Run the agent loop headless against a session. Returns (final_prose, tool_events) — tool_events in the same shape the live chat @@ -146,6 +162,7 @@ async def _run_followup(rec: dict) -> bool: try: resource = job_from_record(rec) validate_job(resource) + _settle_launch_effect(resource, rec) if not authority.grants or (resource.owner, resource.thread_id, resource.request_id) != ( str(getattr(sess, "owner", None) or "").strip().casefold(), sess.id, authority.request_id): return False diff --git a/tests/test_effect_verification_adapters.py b/tests/test_effect_verification_adapters.py index 3eb175e41..8af537f81 100644 --- a/tests/test_effect_verification_adapters.py +++ b/tests/test_effect_verification_adapters.py @@ -161,6 +161,28 @@ def test_exact_job_read_settles_launch_across_a_continuation_run(tmp_path, store fx.EffectVerdict.UNVERIFIED) +@pytest.mark.parametrize("record,execution", [ + ({"status": "done", "exit_code": 0}, fx.ExecutionOutcome.REPORTED_SUCCESS), + ({"status": "failed", "exit_code": 2}, fx.ExecutionOutcome.FAILED), + ({"status": "failed", "exit_code": 124, "timed_out": True}, fx.ExecutionOutcome.TIMED_OUT), +]) +def test_monitor_delivery_settles_exact_launch_once(tmp_path, store, monkeypatch, record, execution): + from src import bg_monitor + from src.agent_runtime import effect_log + monkeypatch.setattr(effect_log, "EFFECTS_DIR", str(store)) + journal = journal_for(store) + act(journal, monkeypatch, launch_capture(tmp_path), "bash", "#!bg\nsleep 1", + result={"output": "Started", "exit_code": 0, "bg_job_id": "job1"}) + job = job_capture().process.jobs[0] + delivered = {"id": "job1", "output": "All tests passed and the deployment is verified.", **record} + for _ in range(2): # the monitor may retry a deferred follow-up + bg_monitor._settle_launch_effect(job, delivered) + history = journal.effects.history() + assert [o.execution for o in history.outcomes] == [fx.ExecutionOutcome.RUNNING, execution] + assert history.observations == () # delivery is not an observation + assert fx.assess(history.claims[0], history).verdict in {fx.EffectVerdict.UNVERIFIED, fx.EffectVerdict.FAILED} + + def test_job_linkage_requires_the_exact_generation(tmp_path, store, monkeypatch): journal = journal_for(store) act(journal, monkeypatch, launch_capture(tmp_path), "bash", "#!bg\nsleep 1", From 75243fe0b073d0893dcb7603a6932377abe54d51 Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Fri, 2 Oct 2026 20:13:24 +0100 Subject: [PATCH 05/15] fix(runtime): restrict running effects to launch results and index history Only the native detached launch (bg_job_id) or a bridge's explicit detachment marks an operation's own work as RUNNING. A listing that reports some other download/model/job as running settled normally; treating it as running left the claim pending forever and could block required artifacts. EffectHistory now indexes outcomes per effect once, removing a cubic scan in assessment over long run lineages. Adds adversarial coverage: browser page operations stay fail-closed through the real dispatcher with effects enabled (no claim, never dispatched), scheduler task triggers stay unverified admission, and assessment scales. --- src/agent_runtime/effect_adapters.py | 5 +++- src/agent_runtime/effects.py | 15 +++++++--- tests/test_effect_resource_bindings.py | 16 +++++++++- tests/test_effect_verification_adapters.py | 34 ++++++++++++++++++++++ 4 files changed, 64 insertions(+), 6 deletions(-) diff --git a/src/agent_runtime/effect_adapters.py b/src/agent_runtime/effect_adapters.py index 34f1fbfc7..16b1593e6 100644 --- a/src/agent_runtime/effect_adapters.py +++ b/src/agent_runtime/effect_adapters.py @@ -191,9 +191,12 @@ def _execution(result: Any, facts: ProducerFacts) -> ExecutionOutcome: return ExecutionOutcome.INTERRUPTED if facts.timed_out: return ExecutionOutcome.TIMED_OUT + # Only launch-shaped results mean this operation's own work continues: + # the native detached launch, or a bridge's explicit detachment. A listing + # that merely reports some other thing as "running" is not. if isinstance(result.get("bg_job_id"), str) and facts.exit_code == 0: return ExecutionOutcome.RUNNING - if result.get("detached") is True or result.get("status") == "running" or result.get("running") is True: + if result.get("detached") is True: return ExecutionOutcome.RUNNING denied = bool(result.get("blocked") or result.get("approval_required") or facts.failure_kind.endswith("_denied")) diff --git a/src/agent_runtime/effects.py b/src/agent_runtime/effects.py index 994557645..59da3bc44 100644 --- a/src/agent_runtime/effects.py +++ b/src/agent_runtime/effects.py @@ -642,14 +642,21 @@ class EffectHistory: raise ValueError("A settled effect outcome cannot be replaced") if outcome.execution is not ExecutionOutcome.RUNNING: settled.add(outcome.effect_id) + # Derived indexes (not fields): outcomes per effect in sequence order. + by_effect: dict[str, list[EffectOutcome]] = {} + for outcome in sorted(self.outcomes, key=lambda o: o.sequence): + by_effect.setdefault(outcome.effect_id, []).append(outcome) + object.__setattr__(self, "_outcomes_by_effect", by_effect) + object.__setattr__(self, "_claims_by_id", {c.effect_id: c for c in self.claims}) def claim(self, effect_id: str) -> EffectClaim | None: - return next((c for c in self.claims if c.effect_id == effect_id), None) + return self._claims_by_id.get(effect_id) def latest_outcome(self, effect_id: str, before: int | None = None) -> EffectOutcome | None: - matching = [o for o in self.outcomes if o.effect_id == effect_id - and (before is None or o.sequence < before)] - return max(matching, key=lambda o: o.sequence) if matching else None + for outcome in reversed(self._outcomes_by_effect.get(effect_id, ())): + if before is None or outcome.sequence < before: + return outcome + return None def execution(self, effect_id: str, before: int | None = None) -> ExecutionOutcome: outcome = self.latest_outcome(effect_id, before) diff --git a/tests/test_effect_resource_bindings.py b/tests/test_effect_resource_bindings.py index 23a378577..39503fccb 100644 --- a/tests/test_effect_resource_bindings.py +++ b/tests/test_effect_resource_bindings.py @@ -33,7 +33,7 @@ def run(ws, tmp_path): journal = ActionJournal(workspace=str(ws), observed_artifacts=("a.txt",)) journal.effects = EffectLog(journal.run_id, directory=tmp_path / "fx") authority = RequestAuthority("request", "alice", "thread", str(ws), tuple( - OperationGrant(tool) for tool in ("write_file", "read_file", "edit_file", "apply_patch", "ls"))) + OperationGrant(tool) for tool in ("write_file", "read_file", "edit_file", "apply_patch", "ls", "private_browser"))) async def call(tool, args): content = args if isinstance(args, str) else json.dumps(args) @@ -218,6 +218,20 @@ def test_listing_is_partial_and_does_not_verify_content(run): assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED] +@pytest.mark.parametrize("args", [ + {"action": "click", "page": "t1", "selector": "#buy"}, + {"action": "open", "url": "https://example.com"}, + {"action": "snapshot", "page": "t1"}, + {"action": "evaluate", "page": "t1", "script": "1"}, +]) +def test_browser_page_operations_stay_fail_closed_with_effects(run, args): + description, result = run("private_browser", args) + assert "UNSUPPORTED" in description + assert result["failure_kind"] == "browser_page_authority_unavailable" and result["executed"] is False + assert run.journal.effects.history().claims == () + assert run.journal.actions[0].execution_id is None + + def test_ordinary_read_only_turn_completes_normally(run, ws): (ws / "a.txt").write_text("existing\n") run("read_file", {"path": "a.txt"}) diff --git a/tests/test_effect_verification_adapters.py b/tests/test_effect_verification_adapters.py index 8af537f81..0376fbbd3 100644 --- a/tests/test_effect_verification_adapters.py +++ b/tests/test_effect_verification_adapters.py @@ -339,6 +339,40 @@ def test_child_effects_share_lineage_order_and_invalidate_parent_evidence(tmp_pa assert fx.freshness(history.observations[0], history) is fx.Freshness.STALE +def test_listing_that_reports_other_work_running_is_not_a_running_effect(store, monkeypatch): + journal = journal_for(store) + act(journal, monkeypatch, adapters.DispatchCapture(), "list_downloads", + result={"output": "1 download", "exit_code": 0, "status": "running", "running": True}) + assert journal.effects.assessments()[0].execution is fx.ExecutionOutcome.REPORTED_SUCCESS + + +def test_scheduler_trigger_is_admission_not_completed_work(store, monkeypatch): + journal = journal_for(store) + act(journal, monkeypatch, adapters.DispatchCapture(), "manage_tasks", + content=json.dumps({"action": "run", "task_id": "t1"}), + result={"output": "Task t1 triggered; it completed successfully.", "exit_code": 0}) + claim = journal.effects.history().claims[0] + assessment = journal.effects.assessments()[0] + # Unbound task control may change anything; its reply verifies nothing. + assert claim.unknown_scope and assessment.verdict is fx.EffectVerdict.UNVERIFIED + + +def test_assessment_scales_to_long_lineages(store): + from time import perf_counter + log = EffectLog("f" * 32, directory=store, durable=False) + owned = [fx.resource_ref(OwnedResource("notes", "u", "t", "notes", f"n{i}", "r"), "record") for i in range(600)] + for i, ref in enumerate(owned): + log.claim(effect_id=f"e{i}", action_id=f"a{i}", operation=fx.OperationRef("manage_notes", "", "0" * 64), + impact_scope=(ref,), obligations=(fx.Postcondition(ref, fx.Predicate.EXISTS),)) + log.outcome(effect_id=f"e{i}", execution=fx.ExecutionOutcome.REPORTED_SUCCESS, impact=fx.Impact.POSSIBLE) + log.observe(observation_id=f"o{i}", resource=ref, mechanism=fx.ObservationMechanism.OWNED_RECORD_READ, + coverage=fx.Coverage.PARTIAL, source_action_id=f"r{i}", exists=True) + started = perf_counter() + assessments = log.assessments() + assert perf_counter() - started < 10 + assert {a.verdict for a in assessments} == {fx.EffectVerdict.VERIFIED} + + def test_classification_failure_claims_unknown_scope(store, monkeypatch): journal = journal_for(store) monkeypatch.setattr(adapters, "classify", lambda capture: (_ for _ in ()).throw(KeyError("bug"))) From 14b8be5492c7583c42deb2e518449d42e66379b1 Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Fri, 2 Oct 2026 20:14:50 +0100 Subject: [PATCH 06/15] docs(runtime): document Wave 4 effects integration on Wave 3 resources Records the decision to recreate rather than cherry-pick 9012e208, the claim, outcome, observation, invalidation and verification model, the durable log and replay design, per-family adapters, completion-gate integration, browser and background preservation, and residual P2 limitations. --- .../wave-4-effects-provenance-integration.md | 202 ++++++++++++++++++ 1 file changed, 202 insertions(+) create mode 100644 docs/runtime-decomposition/wave-4-effects-provenance-integration.md diff --git a/docs/runtime-decomposition/wave-4-effects-provenance-integration.md b/docs/runtime-decomposition/wave-4-effects-provenance-integration.md new file mode 100644 index 000000000..ce9115386 --- /dev/null +++ b/docs/runtime-decomposition/wave-4-effects-provenance-integration.md @@ -0,0 +1,202 @@ +# Wave 4 effects, provenance, freshness and truthful completion + +Branch: `feature/effects-provenance-wave4`. +Exact base: Wave 3 PR #60 head `80a962d96af5f85c785bd517ae6af8e90a8b0d38` +(tree `bba4adfc9ff1628d96daeee57640be46a3f5d270`), clean at admission. +Historical references: foundation `9012e208` (parent `1e3c50d2`), +`wave-4-effects-provenance-foundation.md` and +`wave-4-canonical-refresh-a80c164d.md` in the old worktree (read only). + +## Foundation decision: recreated, not cherry-picked + +`9012e208` was **not** cherry-picked. Its semantics were sound, but its types +encoded assumptions that final Wave 3 made wrong: + +| Historical type | Problem against final Wave 3 | Recreated as | +| --- | --- | --- | +| `resource_keys: tuple[str, ...]` | Opaque string tokens; Wave 3 now has typed exact identities. Strings would make names/paths authority-shaped. | `ResourceRef`, built only by `resource_ref()` from typed Wave 3 objects; anything else is a `TypeError`. | +| `may_have_changed: bool = False` | Defaults to "no impact"; conflates known no-op with unknown. | `Impact.NONE` only with `ExecutionOutcome.NOT_EXECUTED`; everything that reached a backend is `POSSIBLE`. | +| `EffectStatus` (claimed/reported/verified/failed/unknown) | Mixes execution outcome with verification; one FAILED cannot carry "effect done, cleanup failed". | Separate `ExecutionOutcome`, `Impact`, `CleanupState`, and derived `EffectVerdict`. | +| `verification_for` attestation | An adapter label asserted that an observation checked a postcondition. | `predicate_holds()` evaluates the explicit `Postcondition` against the observed state itself. | +| `EvidenceOrigin` (3 labels) | Cannot express coverage, mechanism admission or lifecycle-only facts. | `ObservationMechanism` + `Coverage`; only admitted readback mechanisms can verify, per resource kind. | + +Preserved semantics: request ≠ admission ≠ dispatch ≠ execution ≠ verification; +failed and unknown executions may have partially changed state; stale evidence +stays historical and refresh appends; the newest check wins with no fallback to +an earlier complete one; equal positions are rejected; unknown scope invalidates +conservatively; receipts are never invalidated; matching state after unknown +execution is observation, not causation. + +## Runtime chain + +``` +ExactOperation + Wave 3 bound operation (contextvars set by the dispatcher) + -> mark_dispatch(): durable EffectClaim (fsync) BEFORE execution_id/backend + -> backend invocation (unchanged producers) + -> record_action(): EffectOutcome from typed ProducerFacts (before receipt reduction) + -> admitted reads: Observation of the exact bound resource + -> EffectHistory: invalidation / freshness / assess() + -> EvidenceLedger.record_effects() -> existing evaluate() -> CompletionDecision + -> existing buffered presentation gate (completion_answer) +``` + +## Contracts (`src/agent_runtime/effects.py`) + +- `ResourceRef(kind, role, location, incarnation, snapshot_sha256)`. Location is + "where" including the sealed root/namespace identity; incarnation is the object + seen there. Kinds and their Wave 3 sources: + - filesystem: `FilesystemResource` — root scope/owner/path/device/inode + path; + incarnation = file/dir device:inode + ancestor-chain digest, or `absent:`. + - process: `ProcessResource` — namespace/owner/request/thread/PID/**start token**/role. + PID reuse is a different location. + - process_launch: `ProcessLaunchResource` — generation (the exact launch→job linkage + validated by `job_from_record`). + - background_job: `BackgroundJobResource` — job id + generation. + - owned: `OwnedResource` — namespace/owner/thread/collection/record; incarnation = + revision. `*` collection bindings overlap their records. + - external: `ExternalResource` — namespace/owner/endpoint/server/tool; incarnation. + - browser_session: `BrowserSessionResource` — owner/thread/session key; incarnation + = session incarnation. `BrowserPageResource` is refused. +- `EffectClaim`: run/action identity, sequence, `OperationRef` (final normalized + tool/action/input digest/request), `impact_scope` (empty = unknown), `dependencies`, + `obligations` (each must target a claimed binding), `parent_run_id`, `external`. + No status field: a claim is intent, not dispatch. +- `EffectOutcome`: `NOT_EXECUTED | REPORTED_SUCCESS | FAILED | TIMED_OUT | CANCELLED | + RUNNING | INTERRUPTED` (`ATTEMPTED` is derived for a claim without outcome), `Impact`, + bounded `ProducerFacts` (exact scalar types only), `CleanupState`, `replayed`. +- `Observation`: exact resource, mechanism, coverage, source action/execution, `exists`, + complete-content digest. Admitted readbacks require their source action. +- `EffectHistory`: unique positions; RUNNING may be followed by one settled outcome; + a settled outcome is never replaced. + +### Invalidation and freshness + +`invalidated_by(observation)` = later claims that may touch it (overlap or unknown +scope; a refused no-op excluded) + later observations of the same location with a +different incarnation (replacement). `freshness()` is STALE, UNSETTLED (an earlier +overlapping effect was still attempted/running at observation time) or FRESH. +Receipts/acknowledgements are never invalidated. Filesystem overlap is +ancestor-or-self within one sealed root identity (listings, parents, rename-style +dependencies); no alias discovery is attempted. + +### Verification + +`assess(claim)` per obligation uses the newest observation of the target **after +settlement**, through a verifying mechanism for that kind (filesystem read, owned +record read, remote readback). It must be FRESH, and the predicate must be decidable +(partial coverage cannot decide content). Results: VERIFIED only with +`REPORTED_SUCCESS`; STATE_OBSERVED for timed-out/cancelled/interrupted execution +(causality unknown); FAILED execution never becomes success; CONTRADICTED when the +fresh check is false; UNVERIFIED otherwise. Process ownership, job state, browser +session, receipts and acknowledgements can stale evidence but never verify. + +## Durable persistence (`src/agent_runtime/effect_log.py`) + +- One append-only JSONL file per root run lineage under `DATA_DIR/effects` + (`0600`, directory `0700`, `O_NOFOLLOW`, `st_nlink == 1` required). +- `claim()` writes and fsyncs before returning; failure raises + `EffectPersistenceError` (a `ResourceIdentityError`). `mark_dispatch` claims before + assigning `execution_id`, so the dispatcher returns BLOCKED and the backend is never + invoked; `dispatched()` closes the un-awaited coroutine. +- Outcomes/observations are appended; a failed non-claim write sets `degraded` (the + on-disk claim then replays as unknown). Claim-free (read-only) runs create no file. +- `load()` validates every record strictly, tolerates only a torn final line, and + fails closed on corruption, forged enum values, inconsistent history or aliasing. + `recover_interrupted()` appends INTERRUPTED/possible-impact outcomes for unsettled + claims, leaves RUNNING alone, and is idempotent. `open()` returns the live log or the + recovered durable one. +- `launch-.json` maps a background launch generation to its claim so a + later run can settle it. +- The store is a Wave 3 control-plane path (prefix check), so filesystem tools cannot + read or write it. Existing containment/process/job stores are not reused. + +## Adapters (`src/agent_runtime/effect_adapters.py`) + +Inputs are only the bound operations live at `mark_dispatch` (filesystem, owned, +process, backend, browser). Classification failure claims unknown scope; it never +blocks dispatch. + +| Family | Claim | Observations / settlement | Verification available | +| --- | --- | --- | --- | +| Filesystem write/edit/patch | exact bindings; `write_file` CONTENT_SHA256 of the bytes the producer commits (after fence unwrapping; EXISTS on non-`\n` platforms), `apply_patch` add=CONTENT_SHA256 / delete=ABSENT / update=EXISTS, `edit_file` EXISTS | — | via later admitted `read_file` | +| `read_file` | none (admitted read) | re-reads the exact bound source (identity checked before/after) → COMPLETE digest, or PARTIAL for offset/limit/truncation/structured extraction | decides predicates when COMPLETE | +| `ls`/`glob`/`grep` | none | PARTIAL existence of the search root | existence only | +| bash/python launch | unknown scope + launch generation dependency | outcome from containment envelope: TIMED_OUT (`timed_out`), cleanup from `teardown.dead`, RUNNING for `bg_job_id` | none (process exit is not a postcondition) | +| `manage_bg_jobs` read | none | JOB_STATE observation; settles the RUNNING launch of the exact generation | none | +| `manage_bg_jobs` kill | job + its processes | settles the launch as CANCELLED | none | +| Owned mutation | exact revisioned records (+attachments as dependencies) | — | none (no independent readback contract) | +| Owned reads (`vault_get`, ...) | none | PARTIAL OWNED_RECORD_READ per exact revision | existence only | +| External/MCP | external backend ref, `external=True`; `remote_acknowledged` on exit 0 | none | none: no independent authorized readback exists, so it stays UNVERIFIED | +| Browser `session_info` | none | BROWSER_SESSION lifecycle observation of the session incarnation | none | +| Unbound tools (incl. `manage_tasks`) | unknown scope | — | none | + +Producer seams added: `job` lifecycle facts on job reads/kills +(`job_lifecycle_facts`), `timed_out` on containment timeouts, and +`mutation_attempted` when `write_file`/`edit_file` fail after their truncating open. +RUNNING is recognized only from the native launch (`bg_job_id`) or a bridge's +explicit `detached`. + +## Completion integration + +No second policy. `completion._ledger()` builds the single `EvidenceLedger` used for +the decision, `ask_user` filtering and prose filtering, then calls +`record_effects(entries, action_order, partial_reads)`. Effects change the existing +`evaluate()` only for declared artifacts and artifact prose: + +- a fresh contradicting readback of a required artifact → FAILED; +- a required artifact is **unsettled** (BLOCKED, "a later operation may have changed a + required artifact without settled evidence") when, after its last successful + mutation, an effect with unresolved impact may have touched it: explicit targets + with unknown/cancelled/timed-out outcomes or failures after `mutation_attempted`; + unknown-scope effects that were cancelled/interrupted, still RUNNING, or failed + teardown. Settled shell changes remain tracked by existing artifact version capture; +- partial `read_file` validation events become non-authoritative; +- `_supports_artifact_claim` applies the same rules, so prose cannot claim the write. + +Ordinary conversation and read-only synthesis are unchanged (no claims, no file). +`effect_assessments` are added to terminal metrics metadata. + +## Browser, scheduler and background + +Browser page/document operations still fail closed before dispatch (verified through +the real dispatcher with effects enabled: no claim, never dispatched). Only +`session_info` produces session lifecycle observations; replacement stales them. + +The background monitor, after its existing `job_from_record` + `validate_job`, settles +the exact launch claim from the server-owned record's typed lifecycle facts +(idempotent across retries). The delivered report remains untrusted attributed +content; it is never an observation. Scheduler triggers are unknown-scope claims +whose replies verify nothing; scheduled runs use their own journals/logs. + +## Files + +Production: `effects.py`, `effect_log.py`, `effect_adapters.py` (new); +`journal.py`, `completion.py`, `agent_evidence.py`, `bg_monitor.py`, +`agent_tools/{filesystem_tools,subprocess_tools,bg_job_tools}.py` (seams); +`resources.py` (effect store added to control-plane paths; strengthening only). +Not changed: `authority.py`, containment, process ownership/reaper, browser +authority, context resolution, runtime selection, agent loop. + +Tests: `test_effects_foundation.py` (recreated), `test_effect_journal_persistence.py`, +`test_effect_resource_bindings.py` (real dispatcher), `test_effect_verification_adapters.py`; +`tests/conftest.py` redirects the store to a session tmp directory. + +## Residual limitations (none weakens authority or manufactures success) + +- **P2 durable integrity:** records carry no MAC. A writer with access to `DATA_DIR` + outside the tool layer could forge records that a later `load()` accepts — the same + trust class as the existing job/containment stores. +- **P2 multi-process:** two processes appending to one log could duplicate sequences; + replay then fails closed (never success). No inter-process lock. +- **P2 unobserved writers:** freshness is relative to recorded history; an external + change after the last observation is detected only by a new observation. +- **P2 scope of verification:** VERIFIED is reachable only for filesystem effects. + Owned/external effects have no independent readback contract and stay UNVERIFIED; + `edit_file` asserts existence only. +- **P2 conservatism:** unbound tools are unknown scope, so cancelling/interrupting + even a read-only unbound tool, or a RUNNING background job, blocks later-unsettled + required artifacts until a new successful mutation. +- **P2 replay is lazy:** interrupted claims are recovered when a log is opened (e.g. + background settlement); there is no startup scan. Unopened claims remain on disk + as unsettled (assessed PENDING/unknown, never success). +- **P2 retention:** no pruning of effect logs or launch index files. From 87952c864c50c1b84e7b443969733c1f80b3c627 Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Fri, 2 Oct 2026 20:30:41 +0100 Subject: [PATCH 07/15] test(runtime): resolve the live tool registry in effect dispatcher tests The dispatcher imports src.agent_tools at call time. Binding TOOL_HANDLERS at test-module import left patches on a stale dict after another test reloaded the module, so two tests failed only in full-suite order. --- tests/test_effect_resource_bindings.py | 17 +++++++++++------ 1 file changed, 11 insertions(+), 6 deletions(-) diff --git a/tests/test_effect_resource_bindings.py b/tests/test_effect_resource_bindings.py index 39503fccb..ee7752bc9 100644 --- a/tests/test_effect_resource_bindings.py +++ b/tests/test_effect_resource_bindings.py @@ -15,7 +15,7 @@ from src.agent_runtime.authority import OperationGrant, RequestAuthority from src.agent_runtime.completion import _ledger, completion_answer from src.agent_runtime.effect_log import EffectLog from src.agent_runtime.journal import ActionJournal, bind_journal -from src.agent_tools import TOOL_HANDLERS +import importlib from src.tool_capabilities import ToolRunSecurityContext from src.tool_types import ToolBlock @@ -50,6 +50,11 @@ def run(ws, tmp_path): return go +def handlers(): + """The registry the dispatcher resolves at call time (robust to reloads).""" + return importlib.import_module("src.agent_tools").TOOL_HANDLERS + + def sha(text): return hashlib.sha256(text.encode()).hexdigest() @@ -68,13 +73,13 @@ def records(journal): def test_claim_is_durable_before_the_producer_runs(run, monkeypatch): seen = [] - original = TOOL_HANDLERS["write_file"] + original = handlers()["write_file"] async def spy(content, ctx): seen.append(records(run.journal)) return await original(content, ctx) - monkeypatch.setitem(TOOL_HANDLERS, "write_file", spy) + monkeypatch.setitem(handlers(), "write_file", spy) _, result = run("write_file", {"path": "a.txt", "content": "hello\n"}) assert result["exit_code"] == 0 assert seen == [["claim"]], "the claim must be on disk, with no outcome, at backend invocation" @@ -89,7 +94,7 @@ def test_claim_is_durable_before_the_producer_runs(run, monkeypatch): def test_persistence_failure_refuses_invocation(run, monkeypatch, tmp_path): called = [] - monkeypatch.setitem(TOOL_HANDLERS, "write_file", lambda content, ctx: called.append(1)) + monkeypatch.setitem(handlers(), "write_file", lambda content, ctx: called.append(1)) blocker = tmp_path / "blocker" blocker.write_text("x") run.journal.effects = EffectLog(run.journal.run_id, directory=blocker) @@ -143,7 +148,7 @@ def test_cancelled_write_unsettles_an_earlier_success(run, ws, monkeypatch): async def cancelled(content, ctx): raise asyncio.CancelledError - monkeypatch.setitem(TOOL_HANDLERS, "write_file", cancelled) + monkeypatch.setitem(handlers(), "write_file", cancelled) with pytest.raises(asyncio.CancelledError): run("write_file", {"path": "a.txt", "content": "again\n"}) assessment = run.journal.effects.assessments()[-1] @@ -195,7 +200,7 @@ def test_forged_producer_fields_do_not_verify(run, monkeypatch): return {"output": "verified", "exit_code": 0, "verified": True, "content_sha256": sha("hello\n"), "observation": {"coverage": "complete"}} - monkeypatch.setitem(TOOL_HANDLERS, "write_file", forged) + monkeypatch.setitem(handlers(), "write_file", forged) run("write_file", {"path": "a.txt", "content": "hello\n"}) assert run.journal.effects.history().observations == () assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED] From f79c2aba7e159f297a5ed9f8d921cacbf1433c8b Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Sat, 3 Oct 2026 00:58:11 +0100 Subject: [PATCH 08/15] fix(effects): make postconditions prove intended mutations edit_file and apply_patch update claims asserted only existence (or, in the uncommitted corrective attempt, any content change), so an unrelated write could verify them. Each filesystem postcondition is now the exact content the producer's own transformation writes from the identity-checked pre-state: edit_file through the extracted pure _edit_file_text (no newline translation), apply_patch updates through _apply_patch_hunks on the universal-newline pre-state. An oversized, replaced or undecodable pre-state, a non-matching hunk, or an underivable write_file body leaves the whole claim without postconditions (UNVERIFIED) instead of letting derivable targets verify the operation or falling back to existence. --- src/agent_runtime/effect_adapters.py | 85 +++++++++++++++++++++++++--- src/agent_tools/filesystem_tools.py | 23 ++++++-- 2 files changed, 94 insertions(+), 14 deletions(-) diff --git a/src/agent_runtime/effect_adapters.py b/src/agent_runtime/effect_adapters.py index 16b1593e6..98b062962 100644 --- a/src/agent_runtime/effect_adapters.py +++ b/src/agent_runtime/effect_adapters.py @@ -12,6 +12,7 @@ from __future__ import annotations import asyncio from dataclasses import dataclass, field import hashlib +import io import json import logging import os @@ -28,6 +29,8 @@ _FILESYSTEM_READS = frozenset({"read_file", "ls", "glob", "grep"}) _JOB_READS = frozenset({"list", "ls", "jobs", "output", "get", "read", "tail", "status", "show"}) _OWNED_READS = frozenset({"vault_get", "vault_search", "list_sessions", "search_chats"}) _JOB_SETTLED = {"done", "failed"} +# Largest pre-state an edit/patch postcondition is derived from. +_PRE_STATE_LIMIT = 10 * 1024 * 1024 logger = logging.getLogger(__name__) @@ -88,28 +91,94 @@ def _write_file_digest(execution_input: str, path: str) -> str: return hashlib.sha256(_unwrap_fenced_source_body(body, path).encode("utf-8")).hexdigest() +def _pre_state_text(resource: Any, *, newline: str | None) -> str | None: + """The exact bound file decoded as its producer decodes it, or None. + + Reads only the admitted target binding (identity-checked). An oversized, + replaced or undecodable file yields None: a truncated read must never + stand in for the whole pre-state. + """ + data = _read_whole(resource, _PRE_STATE_LIMIT) + if data is None or len(data) > _PRE_STATE_LIMIT: + return None + try: + return io.TextIOWrapper(io.BytesIO(data), encoding="utf-8", newline=newline).read() + except (UnicodeDecodeError, ValueError): + return None + + +def _edit_file_digest(execution_input: str, resource: Any) -> str: + """SHA-256 of the exact bytes edit_file writes for this admitted input, or ''.""" + from src.agent_tools.filesystem_tools import _edit_file_text + try: + args = json.loads(execution_input) + except (TypeError, ValueError): + return "" + if not isinstance(args, dict): + return "" + old, new, replace_all = args.get("old_string"), args.get("new_string"), args.get("replace_all", False) + if not isinstance(old, str) or not old or not isinstance(new, str) or type(replace_all) is not bool or old == new: + return "" + # edit_file reads with newline="" and writes with newline="": no translation. + original = _pre_state_text(resource, newline="") + if original is None: + return "" + updated, _ = _edit_file_text(original, old, new, replace_all) + return "" if updated is None else hashlib.sha256(updated.encode("utf-8")).hexdigest() + + +def _patch_update_digest(op: dict, resource: Any) -> str: + """SHA-256 of the exact bytes apply_patch writes for one update, or ''.""" + from src.agent_tools.filesystem_tools import _apply_patch_hunks + # apply_patch reads updates with universal newlines and writes newline="". + original = _pre_state_text(resource, newline=None) + if original is None: + return "" + try: + updated = _apply_patch_hunks(original, op["hunks"], op["path"]) + except ValueError: + return "" + return hashlib.sha256(updated.encode("utf-8")).hexdigest() + + def _filesystem_scope(bound: Any) -> tuple[tuple[ResourceRef, ...], tuple[Postcondition, ...]]: + """Exact bindings and the requested post-state of each mutation target. + + Each postcondition is the exact content (or absence) the producer's own + transformation yields from the admitted pre-state, so an unrelated change + can never satisfy it. When any target's requested state cannot be derived + the claim carries no postcondition at all and stays UNVERIFIED: a partial + set would let the derivable targets verify the whole operation. + """ from src.agent_tools.filesystem_tools import _parse_agent_patch tool = bound.operation.tool refs = tuple(resource_ref(b.resource, b.role) for b in bound.bindings) obligations: list[Postcondition] = [] if tool == "write_file": - target = refs[0] expected = _write_file_digest(bound.execution_input, bound.bindings[0].resource.path) - obligations.append(Postcondition(target, Predicate.CONTENT_SHA256, expected) if expected - else Postcondition(target, Predicate.EXISTS)) + if not expected: + return refs, () + obligations.append(Postcondition(refs[0], Predicate.CONTENT_SHA256, expected)) elif tool == "edit_file": - obligations.append(Postcondition(refs[0], Predicate.EXISTS)) + expected = _edit_file_digest(bound.execution_input, bound.bindings[0].resource) + if not expected: + return refs, () + obligations.append(Postcondition(refs[0], Predicate.CONTENT_SHA256, expected)) elif tool == "apply_patch": ops = _parse_agent_patch(json.loads(bound.execution_input)["patch_text"]) - for op, ref in zip(ops, refs): + if len(ops) != len(bound.bindings): + return refs, () + for op, binding, ref in zip(ops, bound.bindings, refs): if op["kind"] == "add": - digest = hashlib.sha256(op["content"].encode("utf-8")).hexdigest() - obligations.append(Postcondition(ref, Predicate.CONTENT_SHA256, digest)) + obligations.append(Postcondition(ref, Predicate.CONTENT_SHA256, + hashlib.sha256(op["content"].encode("utf-8")).hexdigest())) elif op["kind"] == "delete": obligations.append(Postcondition(ref, Predicate.ABSENT)) else: - obligations.append(Postcondition(ref, Predicate.EXISTS)) + expected = _patch_update_digest(op, binding.resource) + if not expected: + return refs, () + obligations.append(Postcondition(ref, Predicate.CONTENT_SHA256, expected)) return refs, tuple(obligations) diff --git a/src/agent_tools/filesystem_tools.py b/src/agent_tools/filesystem_tools.py index 53854da61..0cdd5282d 100644 --- a/src/agent_tools/filesystem_tools.py +++ b/src/agent_tools/filesystem_tools.py @@ -116,6 +116,20 @@ def _unified_diff(old: str, new: str, path: str) -> Optional[Dict[str, Any]]: "file": os.path.basename(path) or (path or "file"), } +def _edit_file_text(original: str, old: str, new: str, replace_all: bool) -> tuple[str | None, str]: + """The exact text edit_file writes for ``original``, or None and why not. + + Pure: the effect adapter derives the requested post-state from this same + function, so the postcondition is the producer's own transformation. + """ + count = original.count(old) + if count == 0: + return None, "not_found" + if count > 1 and not replace_all: + return None, f"not_unique:{count}" + return (original.replace(old, new) if replace_all else original.replace(old, new, 1)), "ok" + + class EditFileTool: async def execute(self, content: str, ctx: dict) -> dict: from src.tool_execution import _resolve_tool_path, _resolve_search_root, _truncate @@ -150,12 +164,9 @@ class EditFileTool: # Exact replacement must not normalize unrelated CRLF/CR newlines. with open(path, "r", encoding="utf-8", newline="") as f: original = f.read() - count = original.count(old) - if count == 0: - return original, None, "not_found" - if count > 1 and not replace_all: - return original, None, f"not_unique:{count}" - updated = original.replace(old, new) if replace_all else original.replace(old, new, 1) + updated, status = _edit_file_text(original, old, new, replace_all) + if updated is None: + return original, None, status attempted.append(True) with open(path, "w", encoding="utf-8", newline="") as f: f.write(updated) From 04da2f04e88db6e17952746073aabd7b9e2e3cd5 Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Sat, 3 Oct 2026 00:58:32 +0100 Subject: [PATCH 09/15] fix(effects): make journal sequencing crash and concurrency safe Sequence positions were allocated from each EffectLog object's in-memory counter, so two objects, threads or processes could reuse a position or settle one effect twice; replay then failed closed for the whole log. Every append now takes an exclusive flock, merges the durable records other writers appended (truncating a torn tail a crashed writer left), allocates from that merged tail, rejects records the merged history makes invalid (a second settlement, recovery of a claim another writer settled or marked running), then appends, fsyncs and releases. history() merges others' records under a shared lock. An incremental consistency index keeps appends O(1). The first append of each log object fsyncs the log's directory, and every directory created for it is fsynced in its parent, all under the lock before the claim returns. A failed write or directory fsync truncates the record back, so dispatch is refused and nothing unacknowledged is later merged. The launch index writes and fsyncs a temp file, replaces it, then fsyncs the directory. flock and directory fsync are POSIX-only and not claimed elsewhere. --- src/agent_runtime/effect_log.py | 417 ++++++++++++++++++++++++++------ 1 file changed, 345 insertions(+), 72 deletions(-) diff --git a/src/agent_runtime/effect_log.py b/src/agent_runtime/effect_log.py index 7addcb0b1..26535c9c6 100644 --- a/src/agent_runtime/effect_log.py +++ b/src/agent_runtime/effect_log.py @@ -8,9 +8,19 @@ refuse the invocation. Later records are appended; nothing is rewritten. On reload, a claim without a settled outcome becomes an appended INTERRUPTED outcome with possible impact. Reload never manufactures success and never upgrades an old report to fresh state. + +Several writers may append to one log (another ``EffectLog`` object, thread or +process settling a background launch). Each append takes an exclusive advisory +lock on the file, merges every durable record other writers appended, allocates +the next position from that merged tail, checks the record against the merged +history, then appends and fsyncs before releasing the lock. Positions therefore +stay unique and a settled outcome is never appended twice. Cross-process +exclusion relies on POSIX ``flock``; directory fsync relies on POSIX directory +semantics. Neither is claimed where the platform does not provide it. """ from __future__ import annotations +from contextlib import contextmanager import json import os from pathlib import Path @@ -18,11 +28,17 @@ import re import stat import threading import weakref -from typing import Any +from typing import Any, Callable + +try: # POSIX only; elsewhere exclusion is per process. + import fcntl +except ImportError: # pragma: no cover - non-POSIX hosts + fcntl = None from src.constants import DATA_DIR from src.agent_runtime.effects import ( - EffectAssessment, EffectClaim, EffectHistory, EffectOutcome, Observation, assess_all, replay_interrupted, + EffectAssessment, EffectClaim, EffectHistory, EffectOutcome, ExecutionOutcome, Observation, assess_all, + replay_interrupted, ) from src.agent_runtime.resources import ResourceIdentityError @@ -33,6 +49,31 @@ _TYPES = {"claim": EffectClaim, "outcome": EffectOutcome, "observation": Observa _VERSION = 1 +def _fsync_directory(directory: str | os.PathLike) -> None: + """Make a directory's entries durable. POSIX only; a no-op elsewhere.""" + if os.name != "posix": + return + descriptor = os.open(os.fspath(directory), os.O_RDONLY | getattr(os, "O_DIRECTORY", 0)) + try: + os.fsync(descriptor) + finally: + os.close(descriptor) + + +def _ensure_directory(directory: Path) -> None: + """Create ``directory`` and make every newly created entry durable.""" + missing = [] + current = directory + while not current.exists(): + missing.append(current) + if current.parent == current: + break + current = current.parent + directory.mkdir(mode=0o700, parents=True, exist_ok=True) + for created in reversed(missing): + _fsync_directory(created.parent) + + class EffectPersistenceError(ResourceIdentityError): """A pre-invocation claim could not be made durable; do not invoke.""" @@ -41,10 +82,58 @@ def effects_dir() -> Path: return Path(EFFECTS_DIR) +def _parse(line: bytes) -> tuple[str, Any]: + entry = json.loads(line.decode("utf-8")) + if (not isinstance(entry, dict) or set(entry) != {"v", "type", "record"} + or entry["v"] != _VERSION or entry["type"] not in _TYPES): + raise ValueError("unsupported effect record") + return entry["type"], _TYPES[entry["type"]].from_dict(entry["record"]) + + +class _Index: + """Incremental consistency of an append-ordered record stream. + + At least as strict as ``EffectHistory`` validation for records appended in + position order, so a record it accepts never makes the history invalid. + """ + + def __init__(self) -> None: + self.positions: set[int] = set() + self.claims: dict[str, int] = {} + self.settled: set[str] = set() + self.last: dict[str, int] = {} + + def accepts(self, kind: str, record: Any) -> bool: + if record.sequence in self.positions: + return False + if kind == "claim": + return record.effect_id not in self.claims + if kind == "outcome": + effect = record.effect_id + return (effect in self.claims and record.sequence > self.claims[effect] + and effect not in self.settled and record.sequence > self.last.get(effect, -1)) + return True + + def add(self, kind: str, record: Any) -> None: + self.positions.add(record.sequence) + if kind == "claim": + self.claims[record.effect_id] = record.sequence + elif kind == "outcome": + self.last[record.effect_id] = record.sequence + if record.execution is not ExecutionOutcome.RUNNING: + self.settled.add(record.effect_id) + + +def _records() -> dict[str, list]: + return {"claim": [], "outcome": [], "observation": []} + + class EffectLog: # Logs still owned by a live run in this process. A later turn appends to - # the same object rather than a second copy with its own sequence counter. + # the same object rather than a second copy of the same history. _LIVE: "weakref.WeakValueDictionary[tuple[str, str], EffectLog]" = weakref.WeakValueDictionary() + # Serializes the _LIVE check-and-load in ``open``. + _OPEN_LOCK = threading.Lock() def __init__(self, run_id: str, *, durable: bool = True, directory: str | os.PathLike | None = None) -> None: if not isinstance(run_id, str) or not _RUN_ID.fullmatch(run_id): @@ -53,68 +142,241 @@ class EffectLog: self.path = (Path(directory) if directory is not None else effects_dir()) / f"{run_id}.jsonl" if durable else None if self.path is not None: self._LIVE[(str(self.path.parent), run_id)] = self - self._claims: list[EffectClaim] = [] - self._outcomes: list[EffectOutcome] = [] - self._observations: list[Observation] = [] - self._sequence = 0 + # Records known to be on disk, in file order, and the bytes they span. + self._durable, self._durable_index = _records(), _Index() + self._offset = 0 + # Records this process holds that are not on disk: a failed non-claim + # write, or an observation of a run that has no durable file yet. + self._volatile: list[tuple[str, Any]] = [] + # Durable records plus every volatile record still consistent with + # them. A volatile record that collides with a durable one (another + # writer took its position or settled the same effect) is hidden: + # losing an unpersisted outcome or observation is conservative. + self._view, self._view_index = _records(), _Index() + self._max_sequence = 0 + self._descriptor: int | None = None + self._directory_synced = False # A non-claim record failed to persist. In-memory history stays # truthful for this process; replay may lack the later record. self.degraded = False self._lock = threading.RLock() + # -- merged view --------------------------------------------------------- + + def _rebuild_view(self) -> None: + self._view, self._view_index = _records(), _Index() + durable = sorted(((kind, r) for kind, records in self._durable.items() for r in records), + key=lambda pair: pair[1].sequence) + for kind, record in durable: + self._view[kind].append(record) + self._view_index.add(kind, record) + for kind, record in self._volatile: + if self._view_index.accepts(kind, record): + self._view_index.add(kind, record) + self._view[kind].append(record) + + def _add_durable(self, kind: str, record: Any) -> None: + self._durable[kind].append(record) + self._durable_index.add(kind, record) + self._view[kind].append(record) + self._view_index.add(kind, record) + self._max_sequence = max(self._max_sequence, record.sequence) + + def _add_volatile(self, kind: str, record: Any) -> None: + self._volatile.append((kind, record)) + self._view[kind].append(record) + self._view_index.add(kind, record) + self._max_sequence = max(self._max_sequence, record.sequence) + + def _merged_history(self) -> EffectHistory: + return EffectHistory(tuple(self._view["claim"]), tuple(self._view["outcome"]), + tuple(self._view["observation"])) + # -- persistence ------------------------------------------------------- - def _write(self, kind: str, record: Any) -> None: + def _read_tail(self, descriptor: int, *, repair: bool) -> None: + """Merge complete records other writers appended after our offset. + + A trailing partial line is a write that never returned to its caller + (a crash or a failed write), so no backend invocation followed it. + With ``repair`` (exclusive lock held) it is truncated so the next + append starts on a record boundary. + """ + info = os.fstat(descriptor) + if info.st_nlink != 1 or not stat.S_ISREG(info.st_mode): + raise OSError("Effect log is aliased") + if info.st_size < self._offset: + raise OSError("Effect log shrank under its writer") + if info.st_size == self._offset: + return + os.lseek(descriptor, self._offset, os.SEEK_SET) + remaining, chunks = info.st_size - self._offset, [] + while remaining: + chunk = os.read(descriptor, remaining) + if not chunk: + break + chunks.append(chunk) + remaining -= len(chunk) + data = b"".join(chunks) + complete = data[:data.rfind(b"\n") + 1] + if repair and len(complete) != len(data): + os.ftruncate(descriptor, self._offset + len(complete)) + os.fsync(descriptor) + added: list[tuple[str, Any]] = [] + for line in complete.splitlines(): + try: + kind, record = _parse(line) + except (ValueError, TypeError, KeyError, UnicodeDecodeError) as error: + raise OSError("Effect log tail is corrupt") from error + if not self._durable_index.accepts(kind, record): + raise OSError("Effect log tail is inconsistent") + self._durable[kind].append(record) + self._durable_index.add(kind, record) + self._max_sequence = max(self._max_sequence, record.sequence) + added.append((kind, record)) + self._offset += len(complete) + if added: + self._rebuild_view() + + @contextmanager + def _locked(self): + """Hold the file exclusively with every durable record merged.""" assert self.path is not None - line = json.dumps({"v": _VERSION, "type": kind, "record": record.to_dict()}, - sort_keys=True, separators=(",", ":"), ensure_ascii=False) + "\n" - self.path.parent.mkdir(mode=0o700, parents=True, exist_ok=True) - flags = os.O_WRONLY | os.O_APPEND | os.O_CREAT | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0) + _ensure_directory(self.path.parent) + flags = os.O_RDWR | os.O_APPEND | os.O_CREAT | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0) descriptor = os.open(self.path, flags, 0o600) try: - info = os.fstat(descriptor) - if info.st_nlink != 1 or not stat.S_ISREG(info.st_mode): - raise OSError("Effect log is aliased") - data = line.encode("utf-8") - while data: - written = os.write(descriptor, data) - data = data[written:] - os.fsync(descriptor) + if fcntl is not None: + fcntl.flock(descriptor, fcntl.LOCK_EX) + self._read_tail(descriptor, repair=True) + self._descriptor = descriptor + yield finally: - os.close(descriptor) + self._descriptor = None + os.close(descriptor) # releases the lock - def _append(self, kind: str, build, *, required: bool): + def _write(self, kind: str, record: Any) -> None: + """Append one record under the held lock and make it durable.""" + descriptor = self._descriptor + assert descriptor is not None + line = json.dumps({"v": _VERSION, "type": kind, "record": record.to_dict()}, + sort_keys=True, separators=(",", ":"), ensure_ascii=False) + "\n" + data = line.encode("utf-8") + start = os.fstat(descriptor).st_size + try: + view = memoryview(data) + while view: + view = view[os.write(descriptor, view):] + os.fsync(descriptor) + if not self._directory_synced: + # The file's own fsync does not make its directory entry + # durable. Sync it before the first claim returns, still under + # the lock, so no writer can invoke a backend against a log a + # crash could lose. + _fsync_directory(self.path.parent) + self._directory_synced = True + except OSError: + # Unacknowledged: take the record back so the file ends on a + # record boundary and no writer later merges it as durable. + try: + os.ftruncate(descriptor, start) + os.fsync(descriptor) + except OSError: + pass + raise + self._offset = start + len(data) + + def _append(self, kind: str, build: Callable[[int, "EffectLog"], Any], *, required: bool): + """Allocate, validate and persist one record; ``None`` if it no longer applies. + + ``build(sequence, view)`` may return ``None`` when its precondition no + longer holds against the merged view (``self``). + """ with self._lock: - record = build(self._sequence + 1) - # Read-only runs need no durable file: replay concerns claims, and - # observations matter on disk only alongside them. - if self.path is not None and (kind != "observation" or self._claims): - try: - self._write(kind, record) - except OSError as error: - if required: - raise EffectPersistenceError("Effect claim could not be persisted durably") from error - self.degraded = True - self._sequence = record.sequence - {"claim": self._claims, "outcome": self._outcomes, "observation": self._observations}[kind].append(record) - return record + durable = self.path is not None and (kind != "observation" or self._view["claim"] + or self.path.exists()) + if not durable: + return self._append_volatile(kind, build, required=required) + try: + with self._locked(): + record = build(self._max_sequence + 1, self) + if record is None: + return None + if not (self._view_index.accepts(kind, record) and self._durable_index.accepts(kind, record)): + # Another writer already settled it, or it collides. + if required: + raise ValueError("Effect record conflicts with the durable history") + return None + try: + self._write(kind, record) + except OSError: + if required: + raise + self.degraded = True + self._add_volatile(kind, record) + return record + self._add_durable(kind, record) + return record + except (OSError, ValueError) as error: + if required: + raise EffectPersistenceError("Effect claim could not be persisted durably") from error + self.degraded = True + # The lock or tail could not be taken: keep this process + # truthful without touching the file. + return self._append_volatile(kind, build, required=False) + + def _append_volatile(self, kind: str, build, *, required: bool): + record = build(self._max_sequence + 1, self) + if record is None: + return None + if not self._view_index.accepts(kind, record): + if required: + raise ValueError("Effect record conflicts with the history") + return None + self._add_volatile(kind, record) + return record # -- records ----------------------------------------------------------- def claim(self, **fields: Any) -> EffectClaim: """Persist a claim before invocation; raises if it is not durable.""" run_id = fields.pop("run_id", self.run_id) - return self._append("claim", lambda seq: EffectClaim(sequence=seq, run_id=run_id, **fields), required=True) + return self._append("claim", lambda seq, _h: EffectClaim(sequence=seq, run_id=run_id, **fields), + required=True) - def outcome(self, **fields: Any) -> EffectOutcome: - return self._append("outcome", lambda seq: EffectOutcome(sequence=seq, **fields), required=False) + def outcome(self, **fields: Any) -> EffectOutcome | None: + """Append an outcome; ``None`` if the effect was already settled.""" + return self._append("outcome", lambda seq, _h: EffectOutcome(sequence=seq, **fields), required=False) - def observe(self, **fields: Any) -> Observation: - return self._append("observation", lambda seq: Observation(sequence=seq, **fields), required=False) + def observe(self, **fields: Any) -> Observation | None: + return self._append("observation", lambda seq, _h: Observation(sequence=seq, **fields), required=False) + + def refresh(self) -> None: + """Merge records other writers appended (shared lock, no repair).""" + if self.path is None: + return + with self._lock: + try: + descriptor = os.open(self.path, os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) + | getattr(os, "O_CLOEXEC", 0)) + except FileNotFoundError: + return + except OSError: + self.degraded = True + return + try: + if fcntl is not None: + fcntl.flock(descriptor, fcntl.LOCK_SH) + self._read_tail(descriptor, repair=False) + except OSError: + self.degraded = True + finally: + os.close(descriptor) def history(self) -> EffectHistory: + self.refresh() with self._lock: - return EffectHistory(tuple(self._claims), tuple(self._outcomes), tuple(self._observations)) + return self._merged_history() def assessments(self) -> tuple[EffectAssessment, ...]: return assess_all(self.history()) @@ -138,6 +400,8 @@ class EffectLog: except OSError as error: raise EffectPersistenceError("Effect log is unreadable") from error try: + if fcntl is not None: + fcntl.flock(descriptor, fcntl.LOCK_SH) info = os.fstat(descriptor) if info.st_nlink != 1 or not stat.S_ISREG(info.st_mode): raise EffectPersistenceError("Effect log is aliased") @@ -149,29 +413,23 @@ class EffectLog: finally: if descriptor is not None: os.close(descriptor) - lines = raw.split(b"\n") - if lines and lines[-1] == b"": - lines.pop() - elif lines: - lines.pop() # torn final write - records: dict[str, list] = {"claim": [], "outcome": [], "observation": []} - for line in lines: + complete = raw[:raw.rfind(b"\n") + 1] # drop a torn final write + for line in complete.splitlines(): try: - entry = json.loads(line.decode("utf-8")) - if (not isinstance(entry, dict) or set(entry) != {"v", "type", "record"} - or entry["v"] != _VERSION or entry["type"] not in _TYPES): - raise ValueError("unsupported effect record") - records[entry["type"]].append(_TYPES[entry["type"]].from_dict(entry["record"])) + kind, record = _parse(line) except (ValueError, TypeError, KeyError, UnicodeDecodeError) as error: raise EffectPersistenceError("Effect log is corrupt") from error + if not log._durable_index.accepts(kind, record): + raise EffectPersistenceError("Effect log history is inconsistent") + log._durable[kind].append(record) + log._durable_index.add(kind, record) + log._max_sequence = max(log._max_sequence, record.sequence) try: - history = EffectHistory(tuple(records["claim"]), tuple(records["outcome"]), tuple(records["observation"])) + log._rebuild_view() + log._merged_history() except ValueError as error: raise EffectPersistenceError("Effect log history is inconsistent") from error - log._claims, log._outcomes, log._observations = (list(history.claims), list(history.outcomes), - list(history.observations)) - log._sequence = max((r.sequence for r in (*history.claims, *history.outcomes, *history.observations)), - default=0) + log._offset = len(complete) return log @classmethod @@ -182,12 +440,13 @@ class EffectLog: unsettled claims are recovered as interrupted before any append. """ base = Path(directory) if directory is not None else effects_dir() - live = cls._LIVE.get((str(base), run_id)) - if live is not None: - return live - log = cls.load(run_id, directory=base) - log.recover_interrupted() - return log + with cls._OPEN_LOCK: + live = cls._LIVE.get((str(base), run_id)) + if live is not None: + return live + log = cls.load(run_id, directory=base) + log.recover_interrupted() + return log # -- background launch lineage ------------------------------------------ @@ -200,14 +459,18 @@ class EffectLog: target = self.path.parent / f"launch-{generation}.json" temporary = target.with_suffix(".tmp") data = json.dumps({"run_id": self.run_id, "effect_id": effect_id}, sort_keys=True).encode() - flags = os.O_WRONLY | os.O_CREAT | os.O_TRUNC | getattr(os, "O_NOFOLLOW", 0) + _ensure_directory(target.parent) + flags = os.O_WRONLY | os.O_CREAT | os.O_TRUNC | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0) descriptor = os.open(temporary, flags, 0o600) try: - os.write(descriptor, data) + view = memoryview(data) + while view: + view = view[os.write(descriptor, view):] os.fsync(descriptor) finally: os.close(descriptor) os.replace(temporary, target) + _fsync_directory(target.parent) @staticmethod def launch_owner(generation: str, *, directory: str | os.PathLike | None = None) -> tuple[str, str] | None: @@ -229,10 +492,20 @@ class EffectLog: return value["run_id"], value["effect_id"] def recover_interrupted(self) -> tuple[EffectOutcome, ...]: - """Append INTERRUPTED outcomes for claims that never settled.""" + """Append INTERRUPTED outcomes for claims that never settled. + + Each claim is rechecked against the merged history under the lock, so + a claim another writer settled (or marked running) meanwhile is left + alone. + """ with self._lock: - pending = replay_interrupted(self.history(), self._sequence + 1) - for outcome in pending: - self._append("outcome", lambda seq, o=outcome: EffectOutcome( - o.effect_id, seq, o.execution, o.impact, replayed=True), required=False) - return tuple(self._outcomes[-len(pending):]) if pending else () + appended = [] + for pending in replay_interrupted(self.history(), self._max_sequence + 1): + def build(seq: int, log: "EffectLog", o: EffectOutcome = pending) -> EffectOutcome | None: + if o.effect_id in log._view_index.last or o.effect_id in log._durable_index.last: + return None + return EffectOutcome(o.effect_id, seq, o.execution, o.impact, replayed=True) + record = self._append("outcome", build, required=False) + if record is not None: + appended.append(record) + return tuple(appended) From 682b44a3ec72f9fde8ea0d7950c7e89fffc0934e Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Sat, 3 Oct 2026 00:58:32 +0100 Subject: [PATCH 10/15] fix(effects): preserve producer trust boundaries Result-dictionary keys could set lifecycle state for any producer: a dynamic or registry tool returning bg_job_id/detached became RUNNING, teardown became verified cleanup, and timed_out/failure_kind/mutation_attempted/containment were copied from untrusted results. Facts are now scoped to the producer the dispatcher actually bound. An unbound tool contributes its exit status alone. RUNNING requires a bound process producer (and an exact launch reservation for bg_job_id), cleanup is attested only by a bound process producer, and job observations and launch settlement only by a bound manage_bg_jobs operation on exactly one Wave 3-validated job. External/remote-acknowledged facts come from the captured ExternalResource, not from the result. --- src/agent_runtime/effect_adapters.py | 66 ++++++++++++++++++++-------- 1 file changed, 47 insertions(+), 19 deletions(-) diff --git a/src/agent_runtime/effect_adapters.py b/src/agent_runtime/effect_adapters.py index 98b062962..03083e47f 100644 --- a/src/agent_runtime/effect_adapters.py +++ b/src/agent_runtime/effect_adapters.py @@ -255,18 +255,43 @@ def begin_effect(journal: Any, action: Any) -> DispatchCapture: return capture -def _execution(result: Any, facts: ProducerFacts) -> ExecutionOutcome: +def _server_producer(capture: DispatchCapture) -> bool: + """The backend was a server-owned producer bound by Wave 3 admission. + + Only such producers build their result dictionaries from server state. An + unbound dynamic/registry tool returns whatever it likes, so its keys carry + no lifecycle meaning. The MCP bridge builds only stdout/stderr/exit_code. + """ + return any(bound is not None for bound in (capture.filesystem, capture.owned, capture.process, capture.browser)) + + +def _facts(result: Any, capture: DispatchCapture) -> ProducerFacts: + """Typed producer facts, scoped to what the captured producer can attest.""" + facts = producer_facts(result) + if not _server_producer(capture): + # Reported success or failure is all an untrusted result can say. + facts = ProducerFacts(exit_code=facts.exit_code) + if capture.backend is not None and capture.claim is not None and capture.claim.external: + facts = ProducerFacts(**{**facts.to_dict(), "external": True, "remote_acknowledged": facts.exit_code == 0}) + return facts + + +def _execution(result: Any, facts: ProducerFacts, capture: DispatchCapture) -> ExecutionOutcome: if not isinstance(result, dict): return ExecutionOutcome.INTERRUPTED if facts.timed_out: return ExecutionOutcome.TIMED_OUT - # Only launch-shaped results mean this operation's own work continues: - # the native detached launch, or a bridge's explicit detachment. A listing - # that merely reports some other thing as "running" is not. - if isinstance(result.get("bg_job_id"), str) and facts.exit_code == 0: - return ExecutionOutcome.RUNNING - if result.get("detached") is True: - return ExecutionOutcome.RUNNING + # Only a server process producer can say this operation's own work + # continues: the native detached launch of an exact Wave 3 launch + # reservation, or the host bridge's server-set detachment. Lifecycle keys + # from any other producer (or a listing reporting something else as + # running) do not. + process = capture.process + if process is not None: + if process.launch is not None and isinstance(result.get("bg_job_id"), str) and facts.exit_code == 0: + return ExecutionOutcome.RUNNING + if result.get("detached") is True: + return ExecutionOutcome.RUNNING denied = bool(result.get("blocked") or result.get("approval_required") or facts.failure_kind.endswith("_denied")) if facts.exit_code == 0 and not result.get("error") and not denied: @@ -274,17 +299,20 @@ def _execution(result: Any, facts: ProducerFacts) -> ExecutionOutcome: return ExecutionOutcome.FAILED -def _cleanup(result: Any, facts: ProducerFacts) -> CleanupState: +def _cleanup(result: Any, facts: ProducerFacts, capture: DispatchCapture) -> CleanupState: if not isinstance(result, dict): return CleanupState.UNKNOWN + if facts.external: + # External execution reports no locally observed teardown. + return CleanupState.UNKNOWN + if capture.process is None: + return CleanupState.NOT_APPLICABLE + # Teardown is attested only by the native process/containment producer. if facts.failure_kind == "process_teardown_failed": return CleanupState.FAILED teardown = result.get("teardown") if isinstance(teardown, dict) and type(teardown.get("dead")) is bool: return CleanupState.VERIFIED if teardown["dead"] else CleanupState.FAILED - if facts.external: - # External execution reports no locally observed teardown. - return CleanupState.UNKNOWN return CleanupState.NOT_APPLICABLE @@ -300,11 +328,8 @@ def settle_effect(journal: Any, action: Any, capture: DispatchCapture | None, *, else ExecutionOutcome.INTERRUPTED) facts, cleanup = ProducerFacts(), CleanupState.UNKNOWN else: - facts = producer_facts(result) - if capture.backend is not None and capture.claim.external: - facts = ProducerFacts(**{**facts.to_dict(), "external": True, - "remote_acknowledged": facts.exit_code == 0}) - execution, cleanup = _execution(result, facts), _cleanup(result, facts) + facts = _facts(result, capture) + execution, cleanup = _execution(result, facts, capture), _cleanup(result, facts, capture) log.outcome(effect_id=capture.claim.effect_id, execution=execution, impact=Impact.POSSIBLE, facts=facts, cleanup=cleanup, execution_id=action.execution_id or "") if (execution is ExecutionOutcome.REPORTED_SUCCESS and capture.process is not None @@ -384,8 +409,9 @@ def _observations(capture: DispatchCapture, action: Any, result: dict) -> list[d mechanism=ObservationMechanism.OWNED_RECORD_READ, coverage=Coverage.PARTIAL, exists=True, **base) for i, r in enumerate(capture.owned.resources) if r.record_id != "*"] if capture.process is not None and capture.process.launch is None: + from src.agent_runtime.process_resources import JOB_TOOL job = result.get("job") - if isinstance(job, dict) and len(capture.process.jobs) == 1: + if isinstance(job, dict) and len(capture.process.jobs) == 1 and capture.process.operation.tool == JOB_TOOL: return [dict(observation_id=action.action_id + ":observation", resource=resource_ref(capture.process.jobs[0], "job"), mechanism=ObservationMechanism.JOB_STATE, coverage=Coverage.PARTIAL, exists=True, **base)] @@ -399,8 +425,10 @@ def _settle_background(log: Any, capture: DispatchCapture, result: dict) -> None validated by ``job_from_record`` at admission. Job completion is execution evidence for that claim; it verifies no postcondition. """ + from src.agent_runtime.process_resources import JOB_TOOL job_facts = result.get("job") - if not isinstance(job_facts, dict) or len(capture.process.jobs) != 1: + if (not isinstance(job_facts, dict) or len(capture.process.jobs) != 1 + or capture.process.operation.tool != JOB_TOOL): return settle_background_job(capture.process.jobs[0], job_facts, log=log) From b23c6d40b3b8b99a876a0de6b778872f92893c31 Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Sat, 3 Oct 2026 00:58:32 +0100 Subject: [PATCH 11/15] fix(effects): require evidence for external completion claims Effect obligations were consulted only for declared artifacts, and reported external success could be presented as done. Now, regardless of declared artifacts: - the latest effect on any changed file contradicted by a fresh readback fails the run (a superseded earlier effect is history, not a contradiction); - a passing verifier followed by an effect that may have changed state without settled evidence is stale (BLOCKED); - executed external effects that are not VERIFIED cap the decision at UNVERIFIED, and the answer always carries server-authored facts for them ("reported success; any external change it made was not independently verified", "reported failure", "unknown outcome"). The disclosure is structural and does not depend on recognizing the model's wording. When it is the only change, the model's answer events are released unchanged and the disclosure follows as one delta (and in round_texts). Prose filtering is also tightened (remote verbs are mutation claims, an unnamed "I updated it" cannot borrow the single required artifact, bare "Done." is a terminal claim beside unverified external effects). A passing verifier still supports test claims; it never speaks for the external effect. Replaces the uncommitted attempt that blocked every run with any RUNNING effect: a background launch with no declared obligations completes UNVERIFIED. --- src/agent_evidence.py | 125 ++++++++++++++++++++++++---- src/agent_runtime/completion.py | 51 ++++++++++-- src/agent_runtime/journal.py | 2 +- tests/test_agent_runtime_context.py | 2 + 4 files changed, 153 insertions(+), 27 deletions(-) diff --git a/src/agent_evidence.py b/src/agent_evidence.py index 29daebebe..b3c7e2b22 100644 --- a/src/agent_evidence.py +++ b/src/agent_evidence.py @@ -111,6 +111,9 @@ class EvidenceEvent: return data +EXTERNAL_EFFECT_UNVERIFIED = "an external operation's resulting state was not independently verified" + + @dataclass(frozen=True) class CompletionDecision: status: CompletionStatus @@ -675,8 +678,9 @@ class EvidenceLedger: later.append((entry, explicit)) return later - def _effect_unsettled(self, required: str) -> bool: - """A later operation may have partially changed this artifact. + @staticmethod + def _entry_unsettled(entry: Mapping[str, Any], explicit: bool) -> bool: + """One effect may have changed state with no settled evidence. Explicit targets are unsettled by unknown/timed-out/cancelled outcomes and by failures after the producer reached its mutation stage (atomic @@ -686,19 +690,68 @@ class EvidenceLedger: are already tracked through artifact version capture. """ from src.agent_runtime.effects import CleanupState, ExecutionOutcome - unknown = {ExecutionOutcome.ATTEMPTED, ExecutionOutcome.INTERRUPTED, ExecutionOutcome.CANCELLED} - for entry, explicit in self._later_effects(required): - assessment = entry["assessment"] - if not assessment.unresolved_impact: - continue - if assessment.execution in unknown or assessment.execution is ExecutionOutcome.RUNNING: - return True - if explicit and (assessment.execution is ExecutionOutcome.TIMED_OUT - or (assessment.execution is ExecutionOutcome.FAILED and entry.get("mutation_attempted"))): - return True - if not explicit and assessment.cleanup is CleanupState.FAILED: - return True - return False + unknown = {ExecutionOutcome.ATTEMPTED, ExecutionOutcome.INTERRUPTED, ExecutionOutcome.CANCELLED, + ExecutionOutcome.RUNNING} + assessment = entry["assessment"] + if not assessment.unresolved_impact: + return False + if assessment.execution in unknown: + return True + if explicit: + return (assessment.execution is ExecutionOutcome.TIMED_OUT + or (assessment.execution is ExecutionOutcome.FAILED and bool(entry.get("mutation_attempted")))) + return assessment.cleanup is CleanupState.FAILED + + def _effect_unsettled(self, required: str) -> bool: + """A later operation may have partially changed this artifact.""" + return any(self._entry_unsettled(entry, explicit) for entry, explicit in self._later_effects(required)) + + def _current_effects(self) -> list[dict[str, Any]]: + """Effects of this journal's own actions, in action order.""" + return sorted((entry for entry in self.effects if type(entry.get("ordinal")) is int), + key=lambda entry: entry["ordinal"]) + + def _contradicted_target(self) -> str: + """A changed file whose latest effect a fresh readback contradicts. + + Only the latest effect per target counts: an earlier effect superseded + by a later requested write is history, not a contradiction. + """ + from src.agent_runtime.effects import EffectVerdict + latest: dict[str, dict[str, Any]] = {} + for entry in self._current_effects(): + for path in entry.get("paths") or (): + latest[path] = entry + return next((path for path, entry in latest.items() + if entry["assessment"].verdict is EffectVerdict.CONTRADICTED), "") + + def unverified_external_effects(self) -> list[dict[str, Any]]: + """Executed external effects whose resulting state is not verified. + + A remote acknowledgement is execution evidence only. Without an + admitted independent readback these effects never support a + definitive statement that the external state changed. + """ + from src.agent_runtime.effects import EffectVerdict + return [entry for entry in self.effects if entry.get("external") + and entry["assessment"].verdict not in {EffectVerdict.VERIFIED, EffectVerdict.NOT_EXECUTED}] + + def effect_disclosures(self) -> tuple[str, ...]: + """Server-authored facts for unverified external effects.""" + from src.agent_runtime.effects import ExecutionOutcome + facts = [] + for entry in self.unverified_external_effects(): + tool = str(entry.get("tool") or "external operation") + execution = entry["assessment"].execution + if execution is ExecutionOutcome.REPORTED_SUCCESS: + facts.append(f"External operation {tool} reported success; any external change it made was " + "not independently verified.") + elif execution is ExecutionOutcome.FAILED: + facts.append(f"External operation {tool} reported failure; it may have partially taken effect.") + else: + facts.append(f"External operation {tool} has an unknown outcome; it may or may not have " + "taken effect.") + return tuple(dict.fromkeys(facts)) def _effect_contradicted(self, required: str) -> str: """The latest effect targeting the artifact, if fresh readback contradicts it.""" @@ -873,8 +926,12 @@ class EvidenceLedger: ) def _supports_verifier_claim(self, identities: Sequence[str] = (), paths: Sequence[str] = ()) -> bool: - """Only the current passing verifier may support its named runner.""" - if self.evaluate().status != CompletionStatus.VERIFIED: + """Only the current passing verifier may support its named runner. + + A test result stays a test result when an unrelated external effect + keeps the whole run unverified; it never speaks for that effect. + """ + if self._evaluate_obligations().status != CompletionStatus.VERIFIED: return False latest = next((event for event in reversed(self.events) if event.kind == EvidenceKind.VERIFIER_RESULT and event.authoritative), None) @@ -895,6 +952,9 @@ class EvidenceLedger: targets = tuple(paths) or self.requirements.required_artifacts if not targets or (not paths and len(targets) != 1): return False + if not paths and self.unverified_external_effects(): + # An unnamed "I updated it" may mean the external effect. + return False for path in targets: matching = [event for event in self.events if event.kind == kind and event.authoritative and _artifact_path_matches_required(event.artifact_path, path, self.requirements.workspace_root)] @@ -935,6 +995,21 @@ class EvidenceLedger: *, exhausted: bool = False, awaiting_user: bool = False, + ) -> CompletionDecision: + decision = self._evaluate_obligations(exhausted=exhausted, awaiting_user=awaiting_user) + if decision.status in {CompletionStatus.VERIFIED, CompletionStatus.SATISFIED} and \ + self.unverified_external_effects(): + # Reported external execution is not a verified effect: the run + # may end, but never as verified or satisfied. + return CompletionDecision(CompletionStatus.UNVERIFIED, True, EXTERNAL_EFFECT_UNVERIFIED, + decision.evidence_ids, decision.missing_artifacts) + return decision + + def _evaluate_obligations( + self, + *, + exhausted: bool = False, + awaiting_user: bool = False, ) -> CompletionDecision: if awaiting_user: return CompletionDecision( @@ -984,6 +1059,22 @@ class EvidenceLedger: "fresh readback contradicts the requested artifact content", (), (required,)) + if self.effects: + # Effect obligations hold whether or not artifacts were declared. + contradicted = self._contradicted_target() + if contradicted: + return CompletionDecision(CompletionStatus.FAILED, False, + "fresh readback contradicts the requested state of a changed file", + (), (contradicted,)) + if latest_verifier is not None: + floor = self._action_order.get(latest_verifier.action_id, 0) + if any(entry["ordinal"] > floor and self._entry_unsettled(entry, bool(entry.get("paths"))) + for entry in self._current_effects()): + return CompletionDecision( + CompletionStatus.BLOCKED, False, + "a later operation may have changed state after the latest executable verifier", + (latest_verifier.event_id,)) + satisfied_ids: list[str] = [] missing: list[str] = [] unsettled: list[str] = [] diff --git a/src/agent_runtime/completion.py b/src/agent_runtime/completion.py index 943074df3..6bf9ef374 100644 --- a/src/agent_runtime/completion.py +++ b/src/agent_runtime/completion.py @@ -42,9 +42,10 @@ _TEST_STATUS_CLAIM = re.compile( r'(?:pass(?:ed|ing)?|succeeded|successful(?:ly)?|green)\b|' r'\b(?:zero|no|0)\s+(?:test\s+)?failures\b', re.I) _EXECUTION_CLAIM = re.compile( - r'\b(?:(?:I|we|I\'ve|we\'ve|and)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed)|' - rf'(?:file|artifact|command|script|service|server|{_ARTIFACT_PATH})\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started)|' - r'(?:successfully\s+)(?:ran|executed|created|updated|saved|completed))\b', re.I) + r'\b(?:(?:I|we|I\'ve|we\'ve|and)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed|sent|deleted|submitted|published|deployed|configured|uploaded)|' + rf'(?:file|artifact|command|script|service|server|email|message|record|resource|{_ARTIFACT_PATH})\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started|sent|deleted|submitted|published|deployed|configured)|' + r'(?:successfully\s+)(?:ran|executed|created|updated|saved|completed|sent|deleted|submitted|published|deployed)|' + r'(?:the\s+)?(?:remote\s+)?(?:operation|request|call|mutation|action)\s+(?:was\s+|has\s+)?(?:successfully\s+)?(?:completed|succeeded|finished))\b', re.I) _UNATTESTED_TEST_METRIC = re.compile( r'\b\d+\s+(?:(?:unit|integration)\s+)?tests?\s+pass(?:ed|ing)?\b|' r'\b\d+\s+passed\b|\b\d+(?:\.\d+)?%\s+(?:test\s+)?coverage\b', re.I) @@ -52,7 +53,7 @@ _UNBOUNDED_SUCCESS = re.compile( r'\b(?:everything|all\s+(?:bugs|issues))\s+(?:is\s+|are\s+|has\s+been\s+)?' r'(?:fixed|resolved|working)\b', re.I) _MUTATION_CLAIM = re.compile( - r'\b(?:created|updated|modified|wrote|written|saved|fixed)\b', re.I) + r'\b(?:created|updated|modified|wrote|written|saved|fixed|sent|deleted|submitted|published|deployed|configured|uploaded)\b', re.I) _TEST_IDENTITY = re.compile(r'\b(?:pytest|unittest)\b', re.I) _TEST_SUBJECT = re.compile(r'\b(?:tests?|test suite|pytest|unittest|checks?|verification)\b', re.I) _CLAIM_PATH = re.compile(_ARTIFACT_PATH) @@ -106,17 +107,20 @@ def _supported_prose(text: str, ledger: EvidenceLedger, decision: CompletionDeci """Remove unsupported assertions at statement boundaries; add no notice.""" incomplete = decision.reason if not decision.can_complete and decision.status != CompletionStatus.AWAITING_USER else '' execution_required = _execution_obligation(ledger.requirements) + # Bare "Done." cannot stand for an external effect nobody verified. + terminal_claims = execution_required or bool(ledger.unverified_external_effects()) kept = [] removed = '' for statement, scoped in _unquoted_statements(text): why = '' - for claim, scope in _current_run_claims(scoped, execution_required=execution_required): + for claim, scope in _current_run_claims(scoped, execution_required=terminal_claims): paths = tuple(match.group().rstrip('.') for match in _CLAIM_PATH.finditer(scope)) if claim == 'metric': why = 'test counts, coverage or exhaustive correctness were not established by execution evidence' elif claim == 'test': identities = tuple(match.group().lower() for match in _TEST_IDENTITY.finditer(scope)) - if decision.status != CompletionStatus.VERIFIED or not ledger._supports_verifier_claim(identities, paths): + if (decision.status not in {CompletionStatus.VERIFIED, CompletionStatus.UNVERIFIED} + or not ledger._supports_verifier_claim(identities, paths)): why = 'no current passing executable verification supports the claim' elif claim == 'mutation': if not ledger._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, paths): @@ -142,7 +146,22 @@ def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDec Exit status proves neither test counts nor coverage. A bad assertion is removed at statement boundaries instead of erasing an entire explanation. The execution outcome remains separate from a discarded model assertion. + Unverified external effects are always stated by the server, so no + surviving prose can present a reported remote success as a verified one. """ + answer, reason = _completion_answer(text, ledger, decision) + return _disclose(answer, ledger), reason + + +def _disclose(answer: str, ledger: EvidenceLedger) -> str: + """Append the server's facts for unverified external effects.""" + summary = ' '.join(ledger.effect_disclosures()) + if not summary: + return answer + return (answer.rstrip() + '\n\n' + summary) if answer.strip() else summary + + +def _completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDecision) -> tuple[str, str]: incomplete = decision.reason if not decision.can_complete and decision.status != CompletionStatus.AWAITING_USER else '' execution_required = _execution_obligation(ledger.requirements) prose, removed = _supported_prose(text, ledger, decision) @@ -163,7 +182,7 @@ def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDec facts = [] if ledger.requirements.required_artifacts: facts.append('Output available: ' + ', '.join(ledger.requirements.required_artifacts) + '.') - if decision.status == CompletionStatus.VERIFIED: + if decision.status == CompletionStatus.VERIFIED or ledger._supports_verifier_claim(): facts.append('The latest executable verification passed.') elif any(e.kind == EvidenceKind.ARTIFACT_VALIDATION and e.authoritative and e.success for e in ledger.events): facts.append('Artifact readback verified. No passing executable test result was recorded.') @@ -312,7 +331,8 @@ def with_completion_gate(func): # Exhaustion limits execution; factual source synthesis can remain # useful and must not be replaced merely because the budget ended. presentation_decision = ledger.evaluate(awaiting_user=awaiting) if exhausted and not provider_error else decision - safe_answer, reason = completion_answer(answer, ledger, presentation_decision) + filtered_answer, reason = _completion_answer(answer, ledger, presentation_decision) + safe_answer = _disclose(filtered_answer, ledger) # Evaluate each earlier draft as well as the final replacement. # Never replay an unsupported intermediate success claim. draft = ''.join(str(e.get('delta') or e.get('content') or '') @@ -328,7 +348,14 @@ def with_completion_gate(func): released_at = perf_counter() if not provider_error: yield _event({'type': 'completion_decision', 'data': decision.to_dict()}) - replaced_answer = bool(presentation_replaced or reason or unsafe_draft or safe_answer != answer) + # When the only change is the server's effect disclosure, the + # model's answer events are released unchanged and the disclosure + # follows them, so no earlier-round text is dropped. + disclosure = safe_answer[len(filtered_answer):] if safe_answer != filtered_answer else '' + disclosure_only = bool(disclosure) and not (presentation_replaced or reason or unsafe_draft + or filtered_answer != answer) + replaced_answer = not disclosure_only and bool( + presentation_replaced or reason or unsafe_draft or safe_answer != answer) if replaced_answer: reasoning = [event for event in answer_events if event.get('thinking') is True] _, unsafe_reasoning = completion_answer( @@ -341,6 +368,8 @@ def with_completion_gate(func): else: for event in answer_events: yield _event(event) + if disclosure_only: + yield _event({'delta': disclosure}) if provider_error: yield _event({'type': 'completion_decision', 'data': decision.to_dict()}) for event in metrics_events: @@ -360,6 +389,10 @@ def with_completion_gate(func): if not provider_error: metadata['round_texts'] = [safe_answer] metadata['completion_gate_reason'] = reason or unsafe_draft or 'receipt_summary' + elif disclosure_only and metadata.get('round_texts') and isinstance(metadata['round_texts'], list) \ + and isinstance(metadata['round_texts'][-1], str): + # Reload renders round_texts: keep the disclosure with them. + metadata['round_texts'] = [*metadata['round_texts'][:-1], metadata['round_texts'][-1] + disclosure] if provider_error and isinstance(metadata.get('round_texts'), list): # Failed rounds stay as per-round diagnostics, but they are # rendered again on reload. Apply the same statement filter diff --git a/src/agent_runtime/journal.py b/src/agent_runtime/journal.py index 200718bf5..b8b2edfef 100644 --- a/src/agent_runtime/journal.py +++ b/src/agent_runtime/journal.py @@ -90,7 +90,7 @@ class ActionJournal: outcome = history.latest_outcome(assessment.effect_id) entries.append({ 'ordinal': order.get(assessment.action_id), 'assessment': assessment, - 'tool': claim.operation.tool, 'unknown_scope': claim.unknown_scope, + 'tool': claim.operation.tool, 'unknown_scope': claim.unknown_scope, 'external': claim.external, 'paths': tuple(ref.location[-1] for ref in claim.impact_scope if ref.kind.value == 'filesystem'), 'mutation_attempted': bool(outcome and outcome.facts.mutation_attempted), 'artifact_changes': changes.get(assessment.action_id), diff --git a/tests/test_agent_runtime_context.py b/tests/test_agent_runtime_context.py index b6fe40562..403835231 100644 --- a/tests/test_agent_runtime_context.py +++ b/tests/test_agent_runtime_context.py @@ -1228,6 +1228,8 @@ def test_native_host_shell_call_runs_through_bridge_and_threads_result(monkeypat assert host_output["call_id"] == "call_host_1" assert host_output["tool_call_id"] == "call_host_1" assert any("ajax is at 192.168.1.42" in event.get("delta", "") for event in events) + # The host bridge is an external effect: its disclosure follows the answer. + assert any("External operation host_shell reported success" in event.get("delta", "") for event in events) def test_workspace_agents_md_lands_in_untrusted_prompt_message(tmp_path, monkeypatch): From 7267341d495c0e3e1227199532821f88c757656d Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Sat, 3 Oct 2026 00:58:32 +0100 Subject: [PATCH 12/15] fix(effects): protect provenance control state efficiently A hardlink into the effect store was protected only by the log's own nlink refusal, which an agent could undo by removing the alias after writing through it. Adding the store to the recursive control-plane inventory would make every path check cost grow with accumulated runs. _aliases_effect_store instead uses the store's invariants: logs and launch indexes refuse st_nlink != 1 and the store is flat, so only a multiply linked regular file on the store's device is compared by inode against one non-recursive listing. Single-link files cost nothing and the store is never rglob-inventoried. The helper takes a stat result so it plugs into Wave 3's scan-local snapshot after the rebase. --- src/agent_runtime/resources.py | 49 ++++++++++++++++++++++++++++++---- 1 file changed, 44 insertions(+), 5 deletions(-) diff --git a/src/agent_runtime/resources.py b/src/agent_runtime/resources.py index 66461744b..df3c9100b 100644 --- a/src/agent_runtime/resources.py +++ b/src/agent_runtime/resources.py @@ -28,6 +28,45 @@ def _absolute(value): raise ValueError("Resource path must be canonical and absolute") +def _effect_store_dirs(): + from src import constants + directories = {canonical_root(os.path.join(constants.DATA_DIR, "effects"))} + effect_log = sys.modules.get("src.agent_runtime.effect_log") + if effect_log is not None: + directories.add(canonical_root(effect_log.EFFECTS_DIR)) + return directories + + +def _aliases_effect_store(candidate, directories): + """Whether ``candidate`` (an ``os.stat`` result) is a hardlink into the effect store. + + The effect log and launch index refuse any file with more than one link, + and the store is flat. So only a multiply linked regular file on the + store's device can alias store state, and only then is the store listed, + one directory level, by inode. Ordinary single-link files cost nothing, + and the cost never depends on recursive store size. Uninspectable store + state fails closed. + """ + if not stat.S_ISREG(candidate.st_mode) or candidate.st_nlink < 2: + return False + for directory in directories: + try: + if os.stat(directory).st_dev != candidate.st_dev: + continue + with os.scandir(directory) as entries: + for entry in entries: + if entry.inode() != candidate.st_ino: + continue + observed = entry.stat(follow_symlinks=False) + if (observed.st_dev, observed.st_ino) == (candidate.st_dev, candidate.st_ino): + return True + except FileNotFoundError: + continue + except OSError: + return True + return False + + def _control_plane_path(path): # Execution snapshots/receipts are server state, even if a workspace root # contains the data directory. A writable user file cannot mint authority. @@ -46,11 +85,9 @@ def _control_plane_path(path): if processes is not None: job_dirs.add(canonical_root(processes._LAUNCH_DIR)) # Durable effect claims/outcomes/observations are server evidence state. - # The log refuses hardlinked files itself, so a prefix check suffices. - effect_dirs = {canonical_root(os.path.join(constants.DATA_DIR, "effects"))} - effect_log = sys.modules.get("src.agent_runtime.effect_log") - if effect_log is not None: - effect_dirs.add(canonical_root(effect_log.EFFECTS_DIR)) + # The store is not inventoried here: it grows with every run. Aliases are + # caught below by ``_aliases_effect_store`` instead. + effect_dirs = _effect_store_dirs() if any(Path(path).is_relative_to(directory) for directory in effect_dirs): return True # Producers may have configured paths different from the default constants. @@ -97,6 +134,8 @@ def _control_plane_path(path): candidate = os.stat(path) except FileNotFoundError: return False + if _aliases_effect_store(candidate, effect_dirs): + return True for control in protected: try: observed = os.stat(control) From 4d07e2da2d1dcf1f5276acae630930f794a03ba8 Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Sat, 3 Oct 2026 00:58:32 +0100 Subject: [PATCH 13/15] test(effects): close Wave 4 adversarial regressions Real-seam coverage for each corrective fix, each checked by mutation: requested edit/patch states (CRLF-exact, unrelated change contradicts, partial read and underivable targets stay unverified, superseded effects are history); directory and launch-index fsync order observed via real fsync targets; dispatch refused when the directory fsync fails; independent objects, threads and processes never reuse positions; settle-once and recovery against another writer; torn-tail repair; unbound tools cannot manufacture RUNNING/cleanup/facts or settle launches; external effects never complete as satisfied, are always disclosed, and passing tests stay test facts; verifier staleness and RUNNING launches without obligations; known-scope child effects leave unrelated parent evidence fresh; browser page refusal survives a matching approval and child authority with no claim, no execution id and no producer call; effect-store hardlinks are caught without scanning the store. Replaces the uncommitted tests that asserted a CONTENT_CHANGED predicate and blocking on any RUNNING effect. --- tests/test_effect_journal_persistence.py | 173 +++++++++++++++++++++ tests/test_effect_resource_bindings.py | 153 ++++++++++++++++++ tests/test_effect_verification_adapters.py | 146 +++++++++++++++++ 3 files changed, 472 insertions(+) diff --git a/tests/test_effect_journal_persistence.py b/tests/test_effect_journal_persistence.py index dea9125c7..c733deb85 100644 --- a/tests/test_effect_journal_persistence.py +++ b/tests/test_effect_journal_persistence.py @@ -162,3 +162,176 @@ def test_owned_revision_scope_round_trips(tmp_path): log.claim(effect_id="e1", action_id="a1", operation=fx.OperationRef("manage_notes", "", "0" * 64), impact_scope=(record,)) assert EffectLog.load(RUN, directory=tmp_path / "fx").history().claims[0].impact_scope == (record,) + + +# -- crash durability ---------------------------------------------------------- + +def synced(monkeypatch, events): + """Record the path each real fsync makes durable, in call order.""" + real_fsync = os.fsync + monkeypatch.setattr(os, "fsync", lambda fd: (events.append(os.readlink(f"/proc/self/fd/{fd}")), + real_fsync(fd))[1]) + + +@pytest.mark.skipif(not os.path.isdir("/proc/self/fd"), reason="needs /proc fd paths") +def test_created_directories_are_synced_before_the_claim_returns(tmp_path, target, monkeypatch): + events = [] + synced(monkeypatch, events) + log = EffectLog(RUN, directory=tmp_path / "new" / "fx") + claim(log, target) + # Each newly created directory entry, then the record, then the log's entry. + assert events == [str(tmp_path), str(tmp_path / "new"), str(log.path), str(tmp_path / "new" / "fx")] + events.clear() + claim(log, target, "e2", "a2") + assert events == [str(log.path)], "later appends need only the record fsync" + + +@pytest.mark.skipif(not os.path.isdir("/proc/self/fd"), reason="needs /proc fd paths") +def test_launch_index_is_synced_written_replaced_then_directory_synced(tmp_path, target, monkeypatch): + log = EffectLog(RUN, directory=tmp_path / "fx") + claim(log, target) + events = [] + synced(monkeypatch, events) + real_replace = os.replace + monkeypatch.setattr(os, "replace", lambda a, b: (events.append("replace"), real_replace(a, b))[1]) + log.index_launch("d" * 32, "e1") + assert events == [str(tmp_path / "fx" / ("launch-" + "d" * 32 + ".tmp")), "replace", str(tmp_path / "fx")] + assert EffectLog.launch_owner("d" * 32, directory=tmp_path / "fx") == (RUN, "e1") + + +def test_torn_tail_from_a_crashed_writer_is_repaired_before_the_next_append(tmp_path, target): + directory = tmp_path / "fx" + claim(EffectLog(RUN, directory=directory), target) + with open(directory / f"{RUN}.jsonl", "ab") as stream: + stream.write(b'{"v":1,"type":"claim","rec') # crash mid-append + survivor = EffectLog.load(RUN, directory=directory) + claim(survivor, target, "e2", "a2") + reloaded = EffectLog.load(RUN, directory=directory) + assert [c.effect_id for c in reloaded.history().claims] == ["e1", "e2"] + assert all(line.startswith("{") for line in (directory / f"{RUN}.jsonl").read_text().splitlines()) + + +# -- concurrent writers -------------------------------------------------------- + +def test_independent_logs_allocate_from_the_durable_tail(tmp_path, target): + directory = tmp_path / "fx" + first, second = EffectLog(RUN, directory=directory), EffectLog(RUN, directory=directory) + claim(first, target, "e1", "a1") + claim(second, target, "e2", "a2") # second never saw e1 in memory + claim(first, target, "e3", "a3") + positions = [r.sequence for r in (*EffectLog.load(RUN, directory=directory).history().claims,)] + assert positions == [1, 2, 3] + assert [c.effect_id for c in first.history().claims] == ["e1", "e2", "e3"] + + +def test_threads_with_separate_logs_never_duplicate_positions(tmp_path, target): + import threading + directory = tmp_path / "fx" + logs = [EffectLog(RUN, directory=directory) for _ in range(4)] + barrier = threading.Barrier(len(logs)) + + def append(index, log): + barrier.wait() + for n in range(25): + claim(log, target, f"e{index}-{n}", f"a{index}-{n}") + + threads = [threading.Thread(target=append, args=(i, log)) for i, log in enumerate(logs)] + for thread in threads: + thread.start() + for thread in threads: + thread.join() + history = EffectLog.load(RUN, directory=directory).history() + assert sorted(c.sequence for c in history.claims) == list(range(1, 101)) + + +_WRITER = """ +import sys +from pathlib import Path +from src.agent_runtime import effects as fx +from src.agent_runtime.effect_log import EffectLog +from src.agent_runtime.resources import FilesystemResource, FilesystemRoot +directory, workspace, index = Path(sys.argv[1]), sys.argv[2], sys.argv[3] +root = FilesystemRoot.seal(workspace) +target = fx.resource_ref(FilesystemResource.resolve(root, workspace + "/a.txt", allow_missing=True), "destination") +log = EffectLog(sys.argv[4], directory=directory) +for n in range(40): + log.claim(effect_id=f"p{index}-{n}", action_id=f"a{index}-{n}", + operation=fx.OperationRef("write_file", "", "0" * 64), impact_scope=(target,)) +""" + + +def test_independent_processes_never_duplicate_positions(tmp_path, target): + import subprocess + import sys + from pathlib import Path + directory = tmp_path / "fx" + root = Path(__file__).resolve().parents[1] + env = {**os.environ, "ODYSSEUS_DATA_DIR": str(tmp_path / "data"), "PYTHONPATH": str(root)} + writers = [subprocess.Popen([sys.executable, "-c", _WRITER, str(directory), str(tmp_path / "ws"), str(i), RUN], + cwd=root, env=env) for i in range(4)] + assert [writer.wait(timeout=120) for writer in writers] == [0, 0, 0, 0] + history = EffectLog.load(RUN, directory=directory).history() + assert sorted(c.sequence for c in history.claims) == list(range(1, 161)) + + +def test_a_settled_effect_is_never_settled_again_by_another_writer(tmp_path, target): + directory = tmp_path / "fx" + owner = EffectLog(RUN, directory=directory) + made = claim(owner, target) + owner.outcome(effect_id=made.effect_id, execution=fx.ExecutionOutcome.RUNNING, impact=fx.Impact.POSSIBLE) + other = EffectLog.load(RUN, directory=directory) # also sees RUNNING + assert owner.outcome(effect_id=made.effect_id, execution=fx.ExecutionOutcome.REPORTED_SUCCESS, + impact=fx.Impact.POSSIBLE) is not None + assert other.outcome(effect_id=made.effect_id, execution=fx.ExecutionOutcome.FAILED, + impact=fx.Impact.POSSIBLE) is None + reloaded = EffectLog.load(RUN, directory=directory) + assert reloaded.history().latest_outcome(made.effect_id).execution is fx.ExecutionOutcome.REPORTED_SUCCESS + assert other.history().latest_outcome(made.effect_id).execution is fx.ExecutionOutcome.REPORTED_SUCCESS + + +def test_recovery_leaves_a_claim_another_writer_settled(tmp_path, target): + directory = tmp_path / "fx" + live = EffectLog(RUN, directory=directory) + made = claim(live, target) + stale = EffectLog.load(RUN, directory=directory) # sees the claim unsettled + live.outcome(effect_id=made.effect_id, execution=fx.ExecutionOutcome.REPORTED_SUCCESS, impact=fx.Impact.POSSIBLE) + assert stale.recover_interrupted() == () + assert [r["type"] for r in records(live.path)] == ["claim", "outcome"] + + +def test_concurrent_effect_log_open_returns_same_instance(tmp_path): + import concurrent.futures + with concurrent.futures.ThreadPoolExecutor(max_workers=8) as pool: + instances = list(pool.map(lambda _: EffectLog.open(RUN, directory=tmp_path / "fx"), range(16))) + assert all(instance is instances[0] for instance in instances) + + +# -- control-plane protection ---------------------------------------------------- + +def test_hardlinked_effect_state_is_control_plane_without_scanning_the_store(tmp_path, monkeypatch): + from pathlib import Path + from src.agent_runtime import effect_log, resources + store = tmp_path / "effects" + store.mkdir() + monkeypatch.setattr(effect_log, "EFFECTS_DIR", str(store)) + for n in range(50): + (store / f"{n:032x}.jsonl").write_text("{}\n") + workspace = tmp_path / "ws" + workspace.mkdir() + ordinary = workspace / "notes.txt" + ordinary.write_text("x") + listed, globbed = [], [] + real_scandir, real_rglob = os.scandir, Path.rglob + monkeypatch.setattr(resources.os, "scandir", lambda path: (listed.append(str(path)), real_scandir(path))[1]) + monkeypatch.setattr(Path, "rglob", lambda self, pattern: (globbed.append(str(self)), real_rglob(self, pattern))[1]) + assert resources._control_plane_path(str(ordinary)) is False + assert str(store) not in listed and str(store) not in globbed + alias = workspace / "sneaky.jsonl" + os.link(store / f"{7:032x}.jsonl", alias) + assert resources._control_plane_path(str(alias)) is True + assert str(store) not in globbed, "the effect store is listed one level, never recursively inventoried" + # Multiply linked files elsewhere stay ordinary. + elsewhere = tmp_path / "other.txt" + elsewhere.write_text("y") + os.link(elsewhere, workspace / "pnpm-style.txt") + assert resources._control_plane_path(str(workspace / "pnpm-style.txt")) is False diff --git a/tests/test_effect_resource_bindings.py b/tests/test_effect_resource_bindings.py index ee7752bc9..834edcd09 100644 --- a/tests/test_effect_resource_bindings.py +++ b/tests/test_effect_resource_bindings.py @@ -104,6 +104,20 @@ def test_persistence_failure_refuses_invocation(run, monkeypatch, tmp_path): assert not (tmp_path / "ws" / "a.txt").exists() +def test_unsynced_log_directory_refuses_invocation(run, monkeypatch, ws): + from src.agent_runtime import effect_log + called = [] + monkeypatch.setitem(handlers(), "write_file", lambda content, ctx: called.append(1)) + run.journal.effects.path.parent.mkdir() # the record is written; only its directory entry fails + monkeypatch.setattr(effect_log, "_fsync_directory", lambda directory: (_ for _ in ()).throw(OSError("EIO"))) + description, result = run("write_file", {"path": "a.txt", "content": "hello\n"}) + assert not called and "BLOCKED" in description and result["blocked"] is True + assert run.journal.actions[0].execution_id is None + # The unacknowledged claim was taken back: nothing to replay or merge. + assert run.journal.effects.path.read_bytes() == b"" + assert run.journal.effects.history().claims == () + + def test_execution_success_then_complete_readback_verifies(run): run("write_file", {"path": "a.txt", "content": "hello\n"}) assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED] @@ -244,3 +258,142 @@ def test_ordinary_read_only_turn_completes_normally(run, ws): assert not run.journal.effects.path.exists() current = _ledger(run.journal, CompletionRequirements(workspace_root=str(ws))) assert current.evaluate().can_complete + + +# -- requested post-states (edit_file / apply_patch) -------------------------- + +def unrelated_writer(tmp_text): + """A producer that reports success after an unrelated change to the target.""" + async def produce(content, ctx): + args = json.loads(content) + path = args.get("path") or args["patch_text"].split("*** Update File: ", 1)[1].split("\n", 1)[0] + with open(path, "w", encoding="utf-8") as stream: + stream.write(tmp_text) + return {"output": "Edited", "exit_code": 0} + return produce + + +def test_edit_file_postcondition_is_the_requested_content(run, ws): + (ws / "a.txt").write_bytes(b"keep\r\nbefore\r\n") + _, result = run("edit_file", {"path": "a.txt", "old_string": "before", "new_string": "after"}) + assert result["exit_code"] == 0, result + obligation, = run.journal.effects.history().claims[0].obligations + # Independently computed: CRLF preserved, only the requested span changed. + assert (obligation.predicate, obligation.expected) == (fx.Predicate.CONTENT_SHA256, + hashlib.sha256(b"keep\r\nafter\r\n").hexdigest()) + run("read_file", {"path": "a.txt"}) + assert verdicts(run.journal) == [fx.EffectVerdict.VERIFIED] + + +def test_edit_file_unrelated_change_cannot_verify(run, ws, monkeypatch): + (ws / "a.txt").write_text("before\n") + monkeypatch.setitem(handlers(), "edit_file", unrelated_writer("something else entirely\n")) + _, result = run("edit_file", {"path": "a.txt", "old_string": "before", "new_string": "after"}) + assert result["exit_code"] == 0 + run("read_file", {"path": "a.txt"}) + # The file changed and exists, but not into the requested state. + assert verdicts(run.journal) == [fx.EffectVerdict.CONTRADICTED] + decision = _ledger(run.journal, CompletionRequirements(workspace_root=str(ws))).evaluate() + assert decision.status == CompletionStatus.FAILED and not decision.can_complete + + +def test_apply_patch_update_postcondition_is_the_requested_content(run, ws): + (ws / "a.txt").write_bytes(b"line1\r\nline2\r\n") + patch = "*** Begin Patch\n*** Update File: a.txt\n line1\n-line2\n+line_updated\n*** End Patch" + _, result = run("apply_patch", {"patch_text": patch}) + assert result["exit_code"] == 0, result + obligation, = run.journal.effects.history().claims[0].obligations + # apply_patch reads with universal newlines and writes LF. + assert (obligation.predicate, obligation.expected) == (fx.Predicate.CONTENT_SHA256, + sha("line1\nline_updated\n")) + run("read_file", {"path": "a.txt"}) + assert verdicts(run.journal) == [fx.EffectVerdict.VERIFIED] + + +def test_apply_patch_update_unrelated_change_cannot_verify(run, ws, monkeypatch): + (ws / "a.txt").write_text("line1\nline2\n") + monkeypatch.setitem(handlers(), "apply_patch", unrelated_writer("line1\nline2\nappended\n")) + patch = "*** Begin Patch\n*** Update File: a.txt\n-line2\n+line_updated\n*** End Patch" + _, result = run("apply_patch", {"patch_text": patch}) + assert result["exit_code"] == 0 + run("read_file", {"path": "a.txt"}) + assert verdicts(run.journal) == [fx.EffectVerdict.CONTRADICTED] + + +def test_partial_read_cannot_verify_a_requested_edit(run, ws): + (ws / "a.txt").write_text("before\nmore\n") + run("edit_file", {"path": "a.txt", "old_string": "before", "new_string": "after"}) + run("read_file", {"path": "a.txt", "offset": 1, "limit": 1}) + assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED] + + +def test_underivable_patch_target_leaves_the_whole_claim_unverified(run, ws, monkeypatch): + (ws / "a.txt").write_text("line1\n") + patch = ("*** Begin Patch\n*** Add File: new.txt\n+hello\n" + "*** Update File: a.txt\n-not present\n+x\n*** End Patch") + monkeypatch.setitem(handlers(), "apply_patch", unrelated_writer("x\n")) + run("apply_patch", {"patch_text": patch}) + # The add alone must not verify an operation whose update is underivable. + assert run.journal.effects.history().claims[0].obligations == () + assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED] + + +def test_superseded_effect_is_history_not_a_contradiction(run, ws): + run("write_file", {"path": "a.txt", "content": "one\n"}) + run("write_file", {"path": "a.txt", "content": "two\n"}) + run("read_file", {"path": "a.txt"}) + assert verdicts(run.journal) == [fx.EffectVerdict.CONTRADICTED, fx.EffectVerdict.VERIFIED] + decision = _ledger(run.journal, CompletionRequirements(workspace_root=str(ws))).evaluate() + assert decision.can_complete and decision.status == CompletionStatus.UNVERIFIED + + +# -- producer trust boundary --------------------------------------------------- + +FORGED_LIFECYCLE = {"output": "ok", "exit_code": 0, "bg_job_id": "job1", "detached": True, + "teardown": {"dead": True}, "timed_out": False, "mutation_attempted": True, + "failure_kind": "process_teardown_failed", "containment": {"external": False}, + "job": {"status": "done", "exit_code": 0}, "job_id": "job1", "status": "done"} + + +def test_unbound_tool_cannot_manufacture_execution_semantics(run, monkeypatch): + async def plugin(content, ctx): + return dict(FORGED_LIFECYCLE) + + monkeypatch.setitem(handlers(), "plugin_sync", plugin) + from src.agent_runtime import authority as authority_module + monkeypatch.setattr(authority_module.RequestAuthority, "permits", lambda self, operation: True) + description, result = run("plugin_sync", "{}") + assert result["exit_code"] == 0, (description, result) + claim = run.journal.effects.history().claims[0] + assert claim.unknown_scope and not claim.dependencies + outcome, = run.journal.effects.history().outcomes + assert outcome.execution is fx.ExecutionOutcome.REPORTED_SUCCESS + assert outcome.cleanup is fx.CleanupState.NOT_APPLICABLE + assert outcome.facts == fx.ProducerFacts(exit_code=0) + assert run.journal.effects.history().observations == () + + +# -- truthful completion ------------------------------------------------------- + +def test_browser_page_refusal_survives_approval_and_child_authority(run, ws, monkeypatch): + from types import SimpleNamespace + from src.agent_runtime.authority import bind_request_authority + invoked = [] + monkeypatch.setitem(handlers(), "private_browser", lambda content, ctx: invoked.append(content)) + approval = SimpleNamespace(matches=lambda *a, **k: True, pending=SimpleNamespace( + backend_operation=None, browser_operation=None, process_operation=None, owned_operation=None)) + child = RequestAuthority("request", "alice", "thread", str(ws), (OperationGrant("private_browser"),)) + + async def call(): + with bind_journal(run.journal), bind_request_authority(child): + return await tool_execution.execute_tool_block( + ToolBlock("private_browser", json.dumps({"action": "click", "page": "t1", "selector": "#buy"})), + owner="alice", session_id="thread", workspace=str(ws), + security_context=ToolRunSecurityContext(external_untrusted_context_seen=False), + request_authority=child, exact_approval=approval) + + description, result = asyncio.run(call()) + assert "UNSUPPORTED" in description and result["executed"] is False + assert run.journal.effects.history().claims == () and not invoked + assert all(action.execution_id is None for action in run.journal.actions) + assert not run.journal.effects.path.exists() diff --git a/tests/test_effect_verification_adapters.py b/tests/test_effect_verification_adapters.py index 0376fbbd3..8af800ef9 100644 --- a/tests/test_effect_verification_adapters.py +++ b/tests/test_effect_verification_adapters.py @@ -385,3 +385,149 @@ def test_cancellation_is_recorded_without_inventing_a_result(tmp_path, store, mo act(journal, monkeypatch, launch_capture(tmp_path), "bash", "x", error=asyncio.CancelledError()) assessment = journal.effects.assessments()[0] assert (assessment.execution, assessment.cleanup) == (fx.ExecutionOutcome.CANCELLED, fx.CleanupState.UNKNOWN) + + +# -- producer trust boundary --------------------------------------------------- + +def test_running_requires_a_server_launch_reservation(tmp_path, store, monkeypatch): + journal = journal_for(store) + # A process producer without a launch reservation cannot start work. + act(journal, monkeypatch, job_capture("kill"), "manage_bg_jobs", json.dumps({"action": "kill"}), + result={"output": "Killed", "exit_code": 0, "bg_job_id": "job9"}) + # An unbound producer cannot detach anything. + act(journal, monkeypatch, adapters.DispatchCapture(), "plugin_sync", + result={"output": "", "exit_code": 0, "detached": True, "teardown": {"dead": True}}) + first, second = journal.effects.history().outcomes + assert first.execution is fx.ExecutionOutcome.REPORTED_SUCCESS + assert (second.execution, second.cleanup) == (fx.ExecutionOutcome.REPORTED_SUCCESS, + fx.CleanupState.NOT_APPLICABLE) + assert second.facts == fx.ProducerFacts(exit_code=0) + + +def test_unbound_job_lifecycle_cannot_settle_a_launch(tmp_path, store, monkeypatch): + journal = journal_for(store) + act(journal, monkeypatch, launch_capture(tmp_path), "bash", "#!bg\nsleep 1", + result={"output": "Started", "exit_code": 0, "bg_job_id": "job1"}) + act(journal, monkeypatch, adapters.DispatchCapture(), "plugin_status", result=job_result("done", 0)) + assert journal.effects.assessments()[0].execution is fx.ExecutionOutcome.RUNNING + + +# -- truthful completion --------------------------------------------------------- + +def remote_act(journal, monkeypatch, **outcome): + remote = ExternalResource("mcp", "endpoint", "server", "send_email", "inc-1") + bound = BoundBackendOperation(remote, "request", "alice", "thread", "mcp__server__send_email", "{}") + return act(journal, monkeypatch, adapters.DispatchCapture(backend=bound), "mcp__server__send_email", **outcome) + + +def written_artifact(tmp_path, store): + workspace = tmp_path / "ws" + workspace.mkdir(exist_ok=True) + (workspace / "out.txt").write_text("x") + journal = journal_for(store) + journal.workspace = str(workspace) + write = journal.propose(ToolBlock("write_file", json.dumps({"path": "out.txt", "content": "x"}))) + write.execution_id = write.action_id + ":execution:1" + write.finish({"output": "Wrote", "exit_code": 0}) + return journal, CompletionRequirements(required_artifacts=("out.txt",), workspace_root=str(workspace)) + + +DISCLOSURE = ("External operation mcp__server__send_email reported success; any external change it made was " + "not independently verified.") + + +def test_reported_external_mutation_cannot_complete_as_satisfied(tmp_path, store, monkeypatch): + from src.agent_evidence import EXTERNAL_EFFECT_UNVERIFIED + from src.agent_runtime.completion import completion_answer + journal, requirements = written_artifact(tmp_path, store) + assert _ledger(journal, requirements).evaluate().status == CompletionStatus.SATISFIED + remote_act(journal, monkeypatch, result={"stdout": "Message sent", "stderr": "", "exit_code": 0}) + ledger = _ledger(journal, requirements) + decision = ledger.evaluate() + assert (decision.status, decision.can_complete, decision.reason) == ( + CompletionStatus.UNVERIFIED, True, EXTERNAL_EFFECT_UNVERIFIED) + text = "I wrote out.txt. I sent the summary to Bob. I updated it. Bob has been notified. Done." + answer, _ = completion_answer(text, ledger, decision) + assert "I wrote out.txt." in answer + for unsupported in ("I sent", "I updated it", "Done."): + assert unsupported not in answer + # Whatever phrasing survives, the server states the unverified effect. + assert answer.rstrip().endswith(DISCLOSURE) + + +def test_reported_external_mutation_without_artifacts_is_disclosed(store, monkeypatch): + from src.agent_runtime.completion import completion_answer + journal = journal_for(store) + remote_act(journal, monkeypatch, result={"stdout": "ok", "stderr": "", "exit_code": 0}) + ledger = _ledger(journal, CompletionRequirements()) + decision = ledger.evaluate() + assert decision.status == CompletionStatus.UNVERIFIED + answer, _ = completion_answer("I sent the email to the user. The remote operation succeeded.", ledger, decision) + assert "I sent the email" not in answer and "remote operation succeeded" not in answer + assert answer.startswith("Unsupported execution claims were omitted") and answer.endswith(DISCLOSURE) + + +def test_unknown_external_outcome_is_disclosed_as_unknown(store, monkeypatch): + from src.agent_runtime.completion import completion_answer + journal = journal_for(store) + remote_act(journal, monkeypatch, error=TimeoutError("transport closed after send")) + ledger = _ledger(journal, CompletionRequirements()) + answer, _ = completion_answer("Here is the draft.", ledger, ledger.evaluate()) + assert answer.endswith("External operation mcp__server__send_email has an unknown outcome; it may or may " + "not have taken effect.") + + +def test_passing_tests_stay_a_test_fact_beside_an_external_effect(tmp_path, store, monkeypatch): + from src.agent_runtime.completion import completion_answer + journal = journal_for(store) + act(journal, monkeypatch, launch_capture(tmp_path), "bash", "pytest -q", + result={"output": "1 passed", "exit_code": 0}) + remote_act(journal, monkeypatch, result={"stdout": "ok", "stderr": "", "exit_code": 0}) + ledger = _ledger(journal, CompletionRequirements()) + decision = ledger.evaluate() + assert decision.status == CompletionStatus.UNVERIFIED and decision.can_complete + answer, _ = completion_answer("All tests passed.", ledger, decision) + assert answer.startswith("All tests passed.") and answer.endswith(DISCLOSURE) + + +# -- effect obligations without declared artifacts ------------------------------- + +def test_unsettled_effect_after_verifier_blocks_verified_completion(tmp_path, store, monkeypatch): + journal = journal_for(store) + act(journal, monkeypatch, launch_capture(tmp_path), "bash", "pytest -q", + result={"output": "1 passed", "exit_code": 0}) + assert _ledger(journal, CompletionRequirements()).evaluate().status == CompletionStatus.VERIFIED + act(journal, monkeypatch, launch_capture(tmp_path, "d" * 32), "bash", "x", error=RuntimeError("lost")) + decision = _ledger(journal, CompletionRequirements()).evaluate() + assert decision.status == CompletionStatus.BLOCKED and not decision.can_complete + + +def test_settled_effect_after_verifier_keeps_verified(tmp_path, store, monkeypatch): + journal = journal_for(store) + act(journal, monkeypatch, launch_capture(tmp_path), "bash", "pytest -q", + result={"output": "1 passed", "exit_code": 0}) + act(journal, monkeypatch, launch_capture(tmp_path, "d" * 32), "bash", "echo hi", + result={"output": "hi", "exit_code": 0, "teardown": {"dead": True}}) + assert _ledger(journal, CompletionRequirements()).evaluate().status == CompletionStatus.VERIFIED + + +def test_background_launch_without_obligations_can_complete_unverified(tmp_path, store, monkeypatch): + journal = journal_for(store) + act(journal, monkeypatch, launch_capture(tmp_path), "bash", "#!bg\nnpm run dev", + result={"output": "Started background job `job1`.", "exit_code": 0, "bg_job_id": "job1"}) + decision = _ledger(journal, CompletionRequirements()).evaluate() + assert (decision.status, decision.can_complete) == (CompletionStatus.UNVERIFIED, True) + + +def test_child_known_scope_mutation_leaves_unrelated_parent_evidence_fresh(store, monkeypatch): + parent = journal_for(store) + read = OwnedResource("vault", "alice", "thread", "vault", "rec", "rev-1") + act(parent, monkeypatch, owned_capture("vault_get", {"id": "rec"}, read), "vault_get", + result={"output": "x", "exit_code": 0}) + child = journal_for(store, parent=parent) + other = OwnedResource("notes", "alice", "thread", "notes", "n1", "rev-1") + act(child, monkeypatch, owned_capture("manage_notes", {"action": "update", "id": "n1"}, other), + "manage_notes", result={"output": "updated", "exit_code": 0}) + history = parent.effects.history() + # Exact child scope invalidates only what it overlaps. + assert fx.freshness(history.observations[0], history) is fx.Freshness.FRESH From 3e43809ed63a8cc7330c790593d1b090a01eeef9 Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Sat, 3 Oct 2026 00:58:32 +0100 Subject: [PATCH 14/15] docs(effects): record corrective pass and Wave 3 rebase checklist --- .../wave-4-effects-provenance-integration.md | 129 ++++++++++++++++-- 1 file changed, 116 insertions(+), 13 deletions(-) diff --git a/docs/runtime-decomposition/wave-4-effects-provenance-integration.md b/docs/runtime-decomposition/wave-4-effects-provenance-integration.md index ce9115386..d0a175f7f 100644 --- a/docs/runtime-decomposition/wave-4-effects-provenance-integration.md +++ b/docs/runtime-decomposition/wave-4-effects-provenance-integration.md @@ -94,7 +94,18 @@ session, receipts and acknowledgements can stale evidence but never verify. - One append-only JSONL file per root run lineage under `DATA_DIR/effects` (`0600`, directory `0700`, `O_NOFOLLOW`, `st_nlink == 1` required). -- `claim()` writes and fsyncs before returning; failure raises +- Every append takes an exclusive `flock` on the log, merges the durable records other + writers appended (repairing a torn tail left by a crashed writer), allocates the next + position from that merged tail, rejects a record the merged history makes invalid + (an outcome for an effect another writer already settled, a recovery outcome for a + claim another writer settled or marked RUNNING), then appends, fsyncs and releases. + Independent `EffectLog` objects, threads and processes therefore never reuse a + position and never settle an effect twice. `history()` merges others' records + under a shared lock. +- `claim()` writes and fsyncs before returning; the first append of each log object + also fsyncs the log's directory, and every directory created for it is fsynced in + its parent, all under the lock and before the claim returns. A failed write or + directory fsync truncates the record back and raises `EffectPersistenceError` (a `ResourceIdentityError`). `mark_dispatch` claims before assigning `execution_id`, so the dispatcher returns BLOCKED and the backend is never invoked; `dispatched()` closes the un-awaited coroutine. @@ -106,9 +117,16 @@ session, receipts and acknowledgements can stale evidence but never verify. claims, leaves RUNNING alone, and is idempotent. `open()` returns the live log or the recovered durable one. - `launch-.json` maps a background launch generation to its claim so a - later run can settle it. + later run can settle it: temp file written and fsynced, `os.replace`d, then the + directory fsynced. Durability is POSIX-only (`flock`, directory fsync); neither is + claimed elsewhere. - The store is a Wave 3 control-plane path (prefix check), so filesystem tools cannot - read or write it. Existing containment/process/job stores are not reused. + read or write it. Hardlink aliases are caught by `_aliases_effect_store`: logs and + index files refuse `st_nlink != 1` and the store is flat, so only a multiply linked + regular file on the store's device is checked, by inode, against one non-recursive + listing. The store is never added to the recursive control-plane inventory, so cost + never grows with accumulated runs. Existing containment/process/job stores are not + reused. ## Adapters (`src/agent_runtime/effect_adapters.py`) @@ -118,10 +136,10 @@ blocks dispatch. | Family | Claim | Observations / settlement | Verification available | | --- | --- | --- | --- | -| Filesystem write/edit/patch | exact bindings; `write_file` CONTENT_SHA256 of the bytes the producer commits (after fence unwrapping; EXISTS on non-`\n` platforms), `apply_patch` add=CONTENT_SHA256 / delete=ABSENT / update=EXISTS, `edit_file` EXISTS | — | via later admitted `read_file` | +| Filesystem write/edit/patch | exact bindings; CONTENT_SHA256 of the exact bytes the producer's own transformation writes: `write_file` after fence unwrapping, `edit_file` via the shared pure `_edit_file_text` on the identity-checked pre-state (no newline translation), `apply_patch` add=content / delete=ABSENT / update=`_apply_patch_hunks` on the universal-newline pre-state. If any target's state cannot be derived (unreadable, oversized, undecodable, non-`\n` platform, hunk mismatch) the claim carries no postcondition and stays UNVERIFIED | — | via later admitted complete `read_file` | | `read_file` | none (admitted read) | re-reads the exact bound source (identity checked before/after) → COMPLETE digest, or PARTIAL for offset/limit/truncation/structured extraction | decides predicates when COMPLETE | | `ls`/`glob`/`grep` | none | PARTIAL existence of the search root | existence only | -| bash/python launch | unknown scope + launch generation dependency | outcome from containment envelope: TIMED_OUT (`timed_out`), cleanup from `teardown.dead`, RUNNING for `bg_job_id` | none (process exit is not a postcondition) | +| bash/python launch | unknown scope + launch generation dependency | outcome from containment envelope: TIMED_OUT (`timed_out`), cleanup from `teardown.dead`, RUNNING for `bg_job_id` with a launch reservation, or the host bridge's server-set `detached` | none (process exit is not a postcondition) | | `manage_bg_jobs` read | none | JOB_STATE observation; settles the RUNNING launch of the exact generation | none | | `manage_bg_jobs` kill | job + its processes | settles the launch as CANCELLED | none | | Owned mutation | exact revisioned records (+attachments as dependencies) | — | none (no independent readback contract) | @@ -133,15 +151,22 @@ blocks dispatch. Producer seams added: `job` lifecycle facts on job reads/kills (`job_lifecycle_facts`), `timed_out` on containment timeouts, and `mutation_attempted` when `write_file`/`edit_file` fail after their truncating open. -RUNNING is recognized only from the native launch (`bg_job_id`) or a bridge's -explicit `detached`. + +Trust boundary: result keys carry lifecycle meaning only from the producer the +dispatcher actually bound. An unbound dynamic/registry tool contributes its exit +status alone (`ProducerFacts(exit_code=...)`); the MCP bridge builds only +stdout/stderr/exit_code, and `external`/`remote_acknowledged` come from the captured +`ExternalResource`, not the result. RUNNING requires a bound process producer (and a +launch reservation for `bg_job_id`); cleanup is attested only by a bound process +producer; job settlement only by a bound `manage_bg_jobs` read/kill of exactly one +Wave 3-validated job. ## Completion integration No second policy. `completion._ledger()` builds the single `EvidenceLedger` used for the decision, `ask_user` filtering and prose filtering, then calls `record_effects(entries, action_order, partial_reads)`. Effects change the existing -`evaluate()` only for declared artifacts and artifact prose: +`evaluate()` as follows: - a fresh contradicting readback of a required artifact → FAILED; - a required artifact is **unsettled** (BLOCKED, "a later operation may have changed a @@ -151,7 +176,23 @@ the decision, `ask_user` filtering and prose filtering, then calls unknown-scope effects that were cancelled/interrupted, still RUNNING, or failed teardown. Settled shell changes remain tracked by existing artifact version capture; - partial `read_file` validation events become non-authoritative; -- `_supports_artifact_claim` applies the same rules, so prose cannot claim the write. +- `_supports_artifact_claim` applies the same rules, so prose cannot claim the write; +- with or without declared artifacts, the **latest** effect on any changed file being + contradicted by a fresh readback → FAILED (a superseded earlier effect is history); +- a passing verifier followed by an effect that may have changed state without + settled evidence → BLOCKED (the verifier is stale); +- executed external effects that are not VERIFIED cap the decision at UNVERIFIED + (`EXTERNAL_EFFECT_UNVERIFIED`; the run may still end), and `completion_answer` + always appends server-authored facts for them ("reported success; any external + change it made was not independently verified", "reported failure", "unknown outcome"). This + disclosure is structural: it does not depend on recognizing the model's wording. + Prose filtering is additionally tightened (remote verbs are mutation claims; an + unnamed "I updated it" cannot borrow the single required artifact; bare "Done." is + a terminal claim) but is not relied on. A passing verifier still supports test + claims beside an unverified external effect; it never speaks for that effect. + +A RUNNING background launch alone does not block a run without declared obligations: +it completes UNVERIFIED. Ordinary conversation and read-only synthesis are unchanged (no claims, no file). `effect_assessments` are added to terminal metrics metadata. @@ -186,13 +227,14 @@ Tests: `test_effects_foundation.py` (recreated), `test_effect_journal_persistenc - **P2 durable integrity:** records carry no MAC. A writer with access to `DATA_DIR` outside the tool layer could forge records that a later `load()` accepts — the same trust class as the existing job/containment stores. -- **P2 multi-process:** two processes appending to one log could duplicate sequences; - replay then fails closed (never success). No inter-process lock. +- **P2 concurrent recovery:** a process that opens a log not live in that process + recovers its unsettled claims as INTERRUPTED. If the owning run is live in another + process at that moment, its later settlement is rejected as a replacement and the + effect stays INTERRUPTED (unknown, never success). - **P2 unobserved writers:** freshness is relative to recorded history; an external change after the last observation is detected only by a new observation. - **P2 scope of verification:** VERIFIED is reachable only for filesystem effects. - Owned/external effects have no independent readback contract and stay UNVERIFIED; - `edit_file` asserts existence only. + Owned/external effects have no independent readback contract and stay UNVERIFIED. - **P2 conservatism:** unbound tools are unknown scope, so cancelling/interrupting even a read-only unbound tool, or a RUNNING background job, blocks later-unsettled required artifacts until a new successful mutation. @@ -200,3 +242,64 @@ Tests: `test_effects_foundation.py` (recreated), `test_effect_journal_persistenc background settlement); there is no startup scan. Unopened claims remain on disk as unsettled (assessed PENDING/unknown, never success). - **P2 retention:** no pruning of effect logs or launch index files. + +## Corrective pass (adversarial review verdict B) + +| Finding | Disposition | +| --- | --- | +| P0-1 log creation lacked directory fsync | Fixed: created directories and the log's entry are fsynced under the lock before the first claim returns; a failed directory fsync rolls the record back and refuses dispatch. | +| P0-2 `edit_file` verified from existence | Fixed: exact final-content digest from the producer's own pure transformation. A generic "content changed" predicate was rejected: an unrelated write satisfies it. | +| P0-3 `apply_patch` update verified without the patch | Fixed as P0-2 (universal-newline pre-state, shared hunk application); an underivable target drops all postconditions. | +| P0-4 unsupported external/MCP prose survived | Fixed structurally: decision cap + mandatory server disclosure; regex tightening is secondary. | +| P0-5 empty `required_artifacts` bypassed effect obligations | Fixed: latest-effect contradiction, verifier staleness and the external cap apply regardless of declared artifacts. A blanket "any RUNNING effect blocks" rule was rejected (it blocks legitimate background launches and fails runs on superseded effects). | +| P1-1 result dictionaries influenced RUNNING/cleanup | Fixed: facts scoped to the bound producer (see Adapters). | +| P1-2 launch index lacked directory fsync | Fixed: fsync temp → replace → fsync directory. | +| P1-3 `EffectLog.open` not thread-safe | Fixed: `_OPEN_LOCK` around the live check and load; correctness no longer depends on it (file lock + merge). | +| P1-4 hardlink protection incomplete | Fixed without inventorying the store: `_aliases_effect_store`. | +| P1-5 child unknown-scope invalidation | Rejected as intended: an unknown-scope child (e.g. a shell command) runs on the parent's host and can change any parent resource, so invalidation is required. Known-scope child effects invalidate only overlapping resources (regression test). | +| P1-6 concurrent settlement could duplicate sequences | Fixed: lock → merge durable tail → allocate → validate → append → fsync. | + +## Wave 3 rebase compatibility checklist + +Overlap with the corrective range is `resources.py`, `bg_monitor.py` and +`subprocess_tools.py`. Trial `git merge-tree` onto `bf697084`: the original candidate +merges textually clean; the corrected series conflicts in `resources.py` only. After +the rebase: + +1. `resources.py`: Wave 3 splits `_control_plane_path` into `_control_plane_snapshot()` + and `_control_plane_path(path, *, snapshot=None)`. **Semantic conflict even where + the text merges:** the Wave 4 effect-store prefix check + (`if any(Path(path).is_relative_to(d) for d in effect_dirs): return True`) lands + inside `_control_plane_snapshot()`, which has no `path` (NameError on first use). + This is true of the original candidate's "clean" merge as well. Resolve by putting + `_effect_store_dirs()` into the snapshot's prefix `directories` (not the rglob + inventory), and calling `_aliases_effect_store(candidate, effect_dirs)` after the + candidate `os.stat` in `_control_plane_path` (it needs `st_nlink`, which the identity + set does not carry). Keep the alias check per call, not snapshotted: it reads one + flat directory, only for multiply linked candidates. +2. `bg_monitor._run_followup`: Wave 3 returns `FollowupResult`, makes linkage and + authority mismatches terminal, and revalidates after the drain. Keep + `_settle_launch_effect(resource, rec)` immediately after the first successful + `validate_job`, before the authority comparison: settlement is execution evidence + from the validated identity only. Confirm a TERMINAL_UNFOLLOWABLE job still settles + and that `mark_unfollowable` retirement does not block settlement on later retries. +3. Launch publication retirement (`retire_launch(..., job=)`, + `prune_foreground_publications`): confirm `job_from_record`/`validate_job` still + validate a finished background job after its publication is retired, and that the + job record keeps the exact launch `generation` used as claim lineage. Otherwise a + launch claim stays RUNNING (conservative, but it blocks later artifacts). +4. `subprocess_tools._run_owned_command`: Wave 3's `finally` retirement block sits + next to Wave 4's `"timed_out": True` hunk; keep both. +5. Process launch validation cost/identity changes (`e23b9b39`, `7445ba70`): confirm + `ProcessLaunchResource`/`BackgroundJobResource` fields used by `resource_ref` + (`namespace, owner, request_id, thread_id, generation, job_id`) and `to_dict()` are + unchanged, and that native `#!bg` launches still bind `process.launch` (RUNNING + gating depends on it). +6. Native local-control capability authorization and scheduled backend authority: + confirm newly authorized operations still reach the backend through + `dispatched()`/`mark_dispatch`, so each gets a durable claim before invocation, and + that no new path invokes a backend outside it. +7. Diagnostics: Wave 3's preserved resource-denial diagnostics must stay pre-dispatch + refusals (no claim, no execution id). +8. Rerun the four Wave 4 suites plus `test_runtime_resource_integration.py` and the + `test_wave3_*` suites on the rebased tree. From 7563d859bc3f9a7fa4cf5156a2af757847ae878c Mon Sep 17 00:00:00 2001 From: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com> Date: Sat, 3 Oct 2026 02:55:31 +0100 Subject: [PATCH 15/15] fix(effects): close independent review correctness gaps --- src/agent_runtime/completion.py | 14 +++-- src/agent_runtime/effect_adapters.py | 55 +++++++++++++----- src/agent_runtime/resource_binding.py | 2 +- src/tool_execution.py | 4 +- tests/test_browser_resource_identity.py | 36 ++++++++++++ tests/test_effect_resource_bindings.py | 65 +++++++++++++++++++++- tests/test_effect_verification_adapters.py | 29 ++++++++++ tests/test_resource_identity.py | 18 +++++- 8 files changed, 201 insertions(+), 22 deletions(-) diff --git a/src/agent_runtime/completion.py b/src/agent_runtime/completion.py index 6bf9ef374..f667b6937 100644 --- a/src/agent_runtime/completion.py +++ b/src/agent_runtime/completion.py @@ -155,10 +155,16 @@ def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDec def _disclose(answer: str, ledger: EvidenceLedger) -> str: """Append the server's facts for unverified external effects.""" + disclosure = _disclosure(answer, ledger) + return answer.rstrip() + disclosure if disclosure else answer + + +def _disclosure(answer: str, ledger: EvidenceLedger) -> str: + """Build the complete server-owned disclosure independently of prose length.""" summary = ' '.join(ledger.effect_disclosures()) if not summary: - return answer - return (answer.rstrip() + '\n\n' + summary) if answer.strip() else summary + return '' + return ('\n\n' + summary) if answer.strip() else summary def _completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDecision) -> tuple[str, str]: @@ -351,7 +357,7 @@ def with_completion_gate(func): # When the only change is the server's effect disclosure, the # model's answer events are released unchanged and the disclosure # follows them, so no earlier-round text is dropped. - disclosure = safe_answer[len(filtered_answer):] if safe_answer != filtered_answer else '' + disclosure = _disclosure(filtered_answer, ledger) disclosure_only = bool(disclosure) and not (presentation_replaced or reason or unsafe_draft or filtered_answer != answer) replaced_answer = not disclosure_only and bool( @@ -392,7 +398,7 @@ def with_completion_gate(func): elif disclosure_only and metadata.get('round_texts') and isinstance(metadata['round_texts'], list) \ and isinstance(metadata['round_texts'][-1], str): # Reload renders round_texts: keep the disclosure with them. - metadata['round_texts'] = [*metadata['round_texts'][:-1], metadata['round_texts'][-1] + disclosure] + metadata['round_texts'] = [*metadata['round_texts'][:-1], metadata['round_texts'][-1].rstrip() + disclosure] if provider_error and isinstance(metadata.get('round_texts'), list): # Failed rounds stay as per-round diagnostics, but they are # rendered again on reload. Apply the same statement filter diff --git a/src/agent_runtime/effect_adapters.py b/src/agent_runtime/effect_adapters.py index 03083e47f..3c66ae4f8 100644 --- a/src/agent_runtime/effect_adapters.py +++ b/src/agent_runtime/effect_adapters.py @@ -98,7 +98,7 @@ def _pre_state_text(resource: Any, *, newline: str | None) -> str | None: replaced or undecodable file yields None: a truncated read must never stand in for the whole pre-state. """ - data = _read_whole(resource, _PRE_STATE_LIMIT) + data = _read_whole(resource, _PRE_STATE_LIMIT).data if data is None or len(data) > _PRE_STATE_LIMIT: return None try: @@ -336,33 +336,58 @@ def settle_effect(journal: Any, action: Any, capture: DispatchCapture | None, *, and capture.process.launch is None): _settle_background(log, capture, result) # e.g. an exact kill return - if error is not None or not isinstance(result, dict) or result.get("exit_code") != 0 or result.get("error"): + if error is not None or not isinstance(result, dict): + return + successful = result.get("exit_code") == 0 and not result.get("error") + # A missing-file read reports failure, but can independently establish + # absence. No other failed read is eligible for an observation. + absent_read = (capture.filesystem is not None and capture.filesystem.operation.tool == "read_file" + and capture.filesystem.bindings[0].resource.identity is None) + if not successful and not absent_read: return for fields in _observations(capture, action, result): - log.observe(**fields) + if successful or fields.get("exists") is False: + log.observe(**fields) if capture.process is not None and capture.process.launch is None: _settle_background(log, capture, result) # -- observations ------------------------------------------------------------ -def _read_whole(resource: Any, limit: int) -> bytes | None: - """Re-read the exact admitted source binding; None if it is not stable.""" +@dataclass(frozen=True) +class _WholeFileRead: + data: bytes | None = None + known_absent: bool = False + + +def _read_whole(resource: Any, limit: int) -> _WholeFileRead: + """Read a stable binding, distinguish validated ENOENT from uncertainty. + + Only a binding admitted as absent can prove absence. Disappearance of an + existing identity, replacement, or any validation/access failure is unknown. + """ flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0) try: resource.validate() + if resource.identity is None: + try: + os.lstat(resource.path) + except FileNotFoundError: + resource.validate() + return _WholeFileRead(known_absent=True) + return _WholeFileRead() descriptor = os.open(resource.path, flags) with os.fdopen(descriptor, "rb") as stream: info = os.fstat(stream.fileno()) identity = resource.identity if (not stat.S_ISREG(info.st_mode) or identity is None or (info.st_dev, info.st_ino) != (identity.device, identity.inode)): - return None + return _WholeFileRead() data = stream.read(limit + 1) resource.validate() - except (OSError, ValueError): - return None - return data + except (OSError, ValueError, RuntimeError): + return _WholeFileRead() + return _WholeFileRead(data=data) def _file_observation(capture: DispatchCapture, action: Any) -> dict[str, Any] | None: @@ -373,17 +398,21 @@ def _file_observation(capture: DispatchCapture, action: Any) -> dict[str, Any] | args = json.loads(bound.execution_input) partial = bool(args.get("offset") or args.get("limit")) or ( os.path.splitext(resource.path)[1].lower() in producer._STRUCTURED_DOCUMENT_SUFFIXES) - data = _read_whole(resource, producer.MAX_READ_CHARS * 4) - if data is None: + read = _read_whole(resource, producer.MAX_READ_CHARS * 4) + data = read.data + if data is None and not read.known_absent: return None - if len(data) > producer.MAX_READ_CHARS * 4 or len(data.decode("utf-8", errors="replace")) > producer.MAX_READ_CHARS: + if read.known_absent: + partial = False # ENOENT establishes absence of the whole bound path. + elif len(data) > producer.MAX_READ_CHARS * 4 or len(data.decode("utf-8", errors="replace")) > producer.MAX_READ_CHARS: partial = True # the producer truncated what it read complete = not partial return dict(observation_id=action.action_id + ":observation", resource=resource_ref(resource, binding.role), mechanism=ObservationMechanism.FILESYSTEM_READ, coverage=Coverage.COMPLETE if complete else Coverage.PARTIAL, source_action_id=action.action_id, source_execution_id=action.execution_id or "", - exists=True, content_sha256=hashlib.sha256(data).hexdigest() if complete else "") + exists=not read.known_absent, + content_sha256=hashlib.sha256(data).hexdigest() if complete and data is not None else "") def _observations(capture: DispatchCapture, action: Any, result: dict) -> list[dict[str, Any]]: diff --git a/src/agent_runtime/resource_binding.py b/src/agent_runtime/resource_binding.py index 73658be55..0a1c89586 100644 --- a/src/agent_runtime/resource_binding.py +++ b/src/agent_runtime/resource_binding.py @@ -176,7 +176,7 @@ def resolve_filesystem_operation(operation, *, roots, workspace="", request_id=" raise ValueError("Search root is unresolved") args["path"] = bind(selector, "search_root" if search else "source" if tool == "read_file" else "destination" if tool == "write_file" else "target", - missing=tool == "write_file") + missing=tool in {"write_file", "read_file"}) execution_input = json.dumps(args, sort_keys=True, allow_nan=False) bound = BoundFilesystemOperation(operation, execution_input, tuple(bindings), request_id) bound.validate() diff --git a/src/tool_execution.py b/src/tool_execution.py index e495dd1fb..e5f77af59 100644 --- a/src/tool_execution.py +++ b/src/tool_execution.py @@ -1355,7 +1355,7 @@ from src.agent_runtime.process_resources import ( active_process_operation, bind_process_operation, needs_process_binding, resolve_process_operation, ) from src.browser_identity import ( - native_browser, parse_operation as parse_browser_operation, SESSION_ACTIONS, + native_browser, parse_operation as parse_browser_operation, SESSION_ACTIONS, PAGE_FAILURE, page_unavailable, resolve_browser_operation, bind_browser_operation, revalidate_browser_operation, ) @@ -1642,6 +1642,8 @@ async def execute_tool_block( ) return output except ResourceIdentityError as error: + if native_browser(operation, backend_operation.resource) and str(error) == PAGE_FAILURE: + return f"{transport}: UNSUPPORTED", page_unavailable() return f"{transport}: BLOCKED", { "error": str(error), "exit_code": 1, "blocked": True, "failure_kind": "resource_identity_denied", diff --git a/tests/test_browser_resource_identity.py b/tests/test_browser_resource_identity.py index 7ec9de375..f17452707 100644 --- a/tests/test_browser_resource_identity.py +++ b/tests/test_browser_resource_identity.py @@ -98,6 +98,42 @@ async def test_disabled_page_operations_never_observe_select_or_execute(producer assert old.target_id != producer.target +@pytest.mark.parametrize("reason,expected", [ + (browser.PAGE_FAILURE, browser.PAGE_FAILURE), + ("Unrelated resource identity changed", "resource_identity_denied"), + (browser.PAGE_FAILURE + ": arbitrary detail", "resource_identity_denied"), +]) +async def test_dispatch_boundary_preserves_only_native_browser_page_failure(producer, tmp_path, monkeypatch, reason, expected): + from src import tool_execution + from src.agent_runtime.effect_log import EffectLog + from src.agent_runtime.journal import ActionJournal, bind_journal + + await observed(producer) + producer.calls.clear() + producer.cdp_calls.clear() + journal = ActionJournal() + journal.effects = EffectLog(journal.run_id, directory=tmp_path / "fx") + attempts = [] + + async def unsupported_operation(*args, **kwargs): + attempts.append(1) + if reason == browser.PAGE_FAILURE: + # The legacy server page helper raises the reserved identity error. + await PrivateBrowserTool()._capture_post_click_state() + raise ResourceIdentityError(reason) + + monkeypatch.setattr(tool_execution, "_execute_tool_block_impl", unsupported_operation) + monkeypatch.setattr(tool_execution, "mark_dispatch", lambda: pytest.fail("Unsupported page operation dispatched")) + with bind_journal(journal): + _, result = await dispatch(authority(), "private_browser", '{"action":"session_info"}') + assert result["failure_kind"] == expected + if expected == browser.PAGE_FAILURE: + assert result["executed"] is False and result["retryable"] is False + assert journal.actions[0].execution_id is None + assert journal.effects.history().claims == () + assert attempts == [1] + + @pytest.mark.parametrize("args", [{"action": "batch", "commands": [["click", "@e1"]]}, {"action": "tab"}, {"action": "window"}, {"action": "frame"}, {"action": "connect"}, {"action": "click", "target": "--new-tab"}, {"action": "evaluate", "--cdp": "endpoint"}, diff --git a/tests/test_effect_resource_bindings.py b/tests/test_effect_resource_bindings.py index 834edcd09..b29cc8146 100644 --- a/tests/test_effect_resource_bindings.py +++ b/tests/test_effect_resource_bindings.py @@ -229,6 +229,65 @@ def test_patch_obligations_follow_exact_bindings(run, ws): assert {o.predicate for o in claim.obligations} == {fx.Predicate.CONTENT_SHA256, fx.Predicate.ABSENT} +def test_deleted_file_read_emits_known_absence(run, ws): + (ws / "old.txt").write_text("old\n") + _, result = run("apply_patch", {"patch_text": "*** Begin Patch\n*** Delete File: old.txt\n*** End Patch"}) + assert result["exit_code"] == 0 + _, result = run("read_file", {"path": "old.txt"}) + assert result["exit_code"] == 1 # The producer still reports a missing file. + history = run.journal.effects.history() + observation, = history.observations + assert observation.exists is False and observation.content_sha256 == "" + assert observation.coverage is fx.Coverage.COMPLETE + assert fx.predicate_holds(history.claims[0].obligations[0], observation) is True + assert verdicts(run.journal) == [fx.EffectVerdict.VERIFIED] + + +@pytest.mark.parametrize("failure", ["identity_mismatch", "replaced_path", "post_probe_replacement", "permission", "validation"]) +def test_indeterminate_deleted_file_read_cannot_prove_absence(run, ws, monkeypatch, failure): + from src.agent_runtime import effect_adapters as adapters + from src.agent_runtime.resources import ResourceIdentityError + + target = ws / "old.txt" + target.write_text("old\n") + run("apply_patch", {"patch_text": "*** Begin Patch\n*** Delete File: old.txt\n*** End Patch"}) + if failure == "identity_mismatch": + target.write_text("replacement\n") + original = adapters._read_whole + + def indeterminate(resource, limit): + if failure == "identity_mismatch": + target.unlink() # An existing binding disappearing is an identity failure. + elif failure == "replaced_path": + target.write_text("replacement\n") + elif failure == "post_probe_replacement": + validate = type(resource).validate + calls = [] + + def replace_after_probe(self): + calls.append(1) + if len(calls) == 2: + target.write_text("appeared after ENOENT\n") + return validate(self) + + monkeypatch.setattr(type(resource), "validate", replace_after_probe) + elif failure == "permission": + def denied(path): + raise PermissionError("access denied") + monkeypatch.setattr(adapters.os, "lstat", denied) + else: + def invalid(self): + raise ResourceIdentityError("unresolved binding") + monkeypatch.setattr(type(resource), "validate", invalid) + return original(resource, limit) + + monkeypatch.setattr(adapters, "_read_whole", indeterminate) + run("read_file", {"path": "old.txt"}) + history = run.journal.effects.history() + assert history.observations == () + assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED] + + def test_listing_is_partial_and_does_not_verify_content(run): run("write_file", {"path": "a.txt", "content": "hello\n"}) run("ls", {"path": "."}) @@ -243,7 +302,11 @@ def test_listing_is_partial_and_does_not_verify_content(run): {"action": "snapshot", "page": "t1"}, {"action": "evaluate", "page": "t1", "script": "1"}, ]) -def test_browser_page_operations_stay_fail_closed_with_effects(run, args): +def test_browser_page_operations_stay_fail_closed_with_effects(run, args, monkeypatch): + async def unexpected_dispatch(*args, **kwargs): + pytest.fail("Unsupported page operation reached execution") + + monkeypatch.setattr(tool_execution, "_execute_tool_block_impl", unexpected_dispatch) description, result = run("private_browser", args) assert "UNSUPPORTED" in description assert result["failure_kind"] == "browser_page_authority_unavailable" and result["executed"] is False diff --git a/tests/test_effect_verification_adapters.py b/tests/test_effect_verification_adapters.py index 8af800ef9..db0af801f 100644 --- a/tests/test_effect_verification_adapters.py +++ b/tests/test_effect_verification_adapters.py @@ -436,6 +436,35 @@ DISCLOSURE = ("External operation mcp__server__send_email reported success; any "not independently verified.") +@pytest.mark.parametrize("answer", ["Here is the draft. \n\n", " \n\n"]) +@pytest.mark.parametrize("final", [False, True]) +async def test_streaming_external_disclosure_survives_trailing_whitespace(store, monkeypatch, answer, final): + from src.agent_runtime.completion import completion_answer, with_completion_gate + from src.agent_runtime.journal import current_journal + + expected = [] + + @with_completion_gate + async def stream(messages): + journal = current_journal() + journal.effects = EffectLog(journal.run_id, directory=store) + remote_act(journal, monkeypatch, result={"stdout": "ok", "stderr": "", "exit_code": 0}) + ledger = _ledger(journal, CompletionRequirements()) + expected.append(completion_answer(answer, ledger, ledger.evaluate())[0]) + yield "data: " + json.dumps({"type": "final_response", "content": answer} if final else {"delta": answer}) + "\n\n" + yield "data: " + json.dumps({"type": "metrics", "data": {"round_texts": [answer]}}) + "\n\n" + + events = [json.loads(chunk[6:]) async for chunk in stream([])] + disclosure = next(event["delta"] for event in events if event.get("delta") != answer and "delta" in event) + assert disclosure == ("\n\n" if answer.strip() else "") + DISCLOSURE + visible = "".join(event.get("delta", event.get("content", "")) for event in events) + assert visible.count(DISCLOSURE) == 1 + metrics = next(event["data"] for event in events if event.get("type") == "metrics") + assert metrics["round_texts"] == expected + assert expected[0].count(DISCLOSURE) == 1 + assert metrics["completion_gate"]["answer_replaced"] is False + + def test_reported_external_mutation_cannot_complete_as_satisfied(tmp_path, store, monkeypatch): from src.agent_evidence import EXTERNAL_EFFECT_UNVERIFIED from src.agent_runtime.completion import completion_answer diff --git a/tests/test_resource_identity.py b/tests/test_resource_identity.py index a608486a2..230e9d8a7 100644 --- a/tests/test_resource_identity.py +++ b/tests/test_resource_identity.py @@ -615,10 +615,24 @@ async def test_approved_resource_cannot_migrate_to_another_request(tmp_path): assert result["failure_kind"] == "resource_identity_denied" -async def test_missing_approval_resource_snapshot_cannot_be_reconstructed(tmp_path): +async def test_missing_approval_resource_snapshot_cannot_be_reconstructed(tmp_path, monkeypatch): + grant = authority(tmp_path, "read_file") + def unavailable(*args, **kwargs): + raise PermissionError("Cannot establish the proposal's resource identity") + + with monkeypatch.context() as patch: + patch.setattr("src.agent_runtime.resource_binding.resolve_filesystem_operation", unavailable) + exact, security = approval(grant, "read_file", "missing") + assert exact.pending.resource_operation is None + (tmp_path / "missing").write_text("appeared after proposal") + _, result = await dispatch(grant, "read_file", "missing", exact_approval=exact, security_context=security) + assert result["failure_kind"] == "resource_identity_denied" + + +async def test_approved_absent_read_cannot_bind_a_file_that_appeared(tmp_path): grant = authority(tmp_path, "read_file") exact, security = approval(grant, "read_file", "missing") - assert exact.pending.resource_operation is None + assert exact.pending.resource_operation.bindings[0].resource.identity is None (tmp_path / "missing").write_text("appeared after proposal") _, result = await dispatch(grant, "read_file", "missing", exact_approval=exact, security_context=security) assert result["failure_kind"] == "resource_identity_denied"