fix(runtime): restrict running effects to launch results and index history

Only the native detached launch (bg_job_id) or a bridge's explicit detachment
marks an operation's own work as RUNNING. A listing that reports some other
download/model/job as running settled normally; treating it as running left
the claim pending forever and could block required artifacts.

EffectHistory now indexes outcomes per effect once, removing a cubic scan in
assessment over long run lineages.

Adds adversarial coverage: browser page operations stay fail-closed through
the real dispatcher with effects enabled (no claim, never dispatched),
scheduler task triggers stay unverified admission, and assessment scales.
This commit is contained in:
Alexandre Teixeira
2026-10-02 20:13:24 +01:00
parent 3953ea2444
commit 75243fe0b0
4 changed files with 64 additions and 6 deletions
+4 -1
View File
@@ -191,9 +191,12 @@ def _execution(result: Any, facts: ProducerFacts) -> ExecutionOutcome:
return ExecutionOutcome.INTERRUPTED
if facts.timed_out:
return ExecutionOutcome.TIMED_OUT
# Only launch-shaped results mean this operation's own work continues:
# the native detached launch, or a bridge's explicit detachment. A listing
# that merely reports some other thing as "running" is not.
if isinstance(result.get("bg_job_id"), str) and facts.exit_code == 0:
return ExecutionOutcome.RUNNING
if result.get("detached") is True or result.get("status") == "running" or result.get("running") is True:
if result.get("detached") is True:
return ExecutionOutcome.RUNNING
denied = bool(result.get("blocked") or result.get("approval_required")
or facts.failure_kind.endswith("_denied"))
+11 -4
View File
@@ -642,14 +642,21 @@ class EffectHistory:
raise ValueError("A settled effect outcome cannot be replaced")
if outcome.execution is not ExecutionOutcome.RUNNING:
settled.add(outcome.effect_id)
# Derived indexes (not fields): outcomes per effect in sequence order.
by_effect: dict[str, list[EffectOutcome]] = {}
for outcome in sorted(self.outcomes, key=lambda o: o.sequence):
by_effect.setdefault(outcome.effect_id, []).append(outcome)
object.__setattr__(self, "_outcomes_by_effect", by_effect)
object.__setattr__(self, "_claims_by_id", {c.effect_id: c for c in self.claims})
def claim(self, effect_id: str) -> EffectClaim | None:
return next((c for c in self.claims if c.effect_id == effect_id), None)
return self._claims_by_id.get(effect_id)
def latest_outcome(self, effect_id: str, before: int | None = None) -> EffectOutcome | None:
matching = [o for o in self.outcomes if o.effect_id == effect_id
and (before is None or o.sequence < before)]
return max(matching, key=lambda o: o.sequence) if matching else None
for outcome in reversed(self._outcomes_by_effect.get(effect_id, ())):
if before is None or outcome.sequence < before:
return outcome
return None
def execution(self, effect_id: str, before: int | None = None) -> ExecutionOutcome:
outcome = self.latest_outcome(effect_id, before)
+15 -1
View File
@@ -33,7 +33,7 @@ def run(ws, tmp_path):
journal = ActionJournal(workspace=str(ws), observed_artifacts=("a.txt",))
journal.effects = EffectLog(journal.run_id, directory=tmp_path / "fx")
authority = RequestAuthority("request", "alice", "thread", str(ws), tuple(
OperationGrant(tool) for tool in ("write_file", "read_file", "edit_file", "apply_patch", "ls")))
OperationGrant(tool) for tool in ("write_file", "read_file", "edit_file", "apply_patch", "ls", "private_browser")))
async def call(tool, args):
content = args if isinstance(args, str) else json.dumps(args)
@@ -218,6 +218,20 @@ def test_listing_is_partial_and_does_not_verify_content(run):
assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED]
@pytest.mark.parametrize("args", [
{"action": "click", "page": "t1", "selector": "#buy"},
{"action": "open", "url": "https://example.com"},
{"action": "snapshot", "page": "t1"},
{"action": "evaluate", "page": "t1", "script": "1"},
])
def test_browser_page_operations_stay_fail_closed_with_effects(run, args):
description, result = run("private_browser", args)
assert "UNSUPPORTED" in description
assert result["failure_kind"] == "browser_page_authority_unavailable" and result["executed"] is False
assert run.journal.effects.history().claims == ()
assert run.journal.actions[0].execution_id is None
def test_ordinary_read_only_turn_completes_normally(run, ws):
(ws / "a.txt").write_text("existing\n")
run("read_file", {"path": "a.txt"})
@@ -339,6 +339,40 @@ def test_child_effects_share_lineage_order_and_invalidate_parent_evidence(tmp_pa
assert fx.freshness(history.observations[0], history) is fx.Freshness.STALE
def test_listing_that_reports_other_work_running_is_not_a_running_effect(store, monkeypatch):
journal = journal_for(store)
act(journal, monkeypatch, adapters.DispatchCapture(), "list_downloads",
result={"output": "1 download", "exit_code": 0, "status": "running", "running": True})
assert journal.effects.assessments()[0].execution is fx.ExecutionOutcome.REPORTED_SUCCESS
def test_scheduler_trigger_is_admission_not_completed_work(store, monkeypatch):
journal = journal_for(store)
act(journal, monkeypatch, adapters.DispatchCapture(), "manage_tasks",
content=json.dumps({"action": "run", "task_id": "t1"}),
result={"output": "Task t1 triggered; it completed successfully.", "exit_code": 0})
claim = journal.effects.history().claims[0]
assessment = journal.effects.assessments()[0]
# Unbound task control may change anything; its reply verifies nothing.
assert claim.unknown_scope and assessment.verdict is fx.EffectVerdict.UNVERIFIED
def test_assessment_scales_to_long_lineages(store):
from time import perf_counter
log = EffectLog("f" * 32, directory=store, durable=False)
owned = [fx.resource_ref(OwnedResource("notes", "u", "t", "notes", f"n{i}", "r"), "record") for i in range(600)]
for i, ref in enumerate(owned):
log.claim(effect_id=f"e{i}", action_id=f"a{i}", operation=fx.OperationRef("manage_notes", "", "0" * 64),
impact_scope=(ref,), obligations=(fx.Postcondition(ref, fx.Predicate.EXISTS),))
log.outcome(effect_id=f"e{i}", execution=fx.ExecutionOutcome.REPORTED_SUCCESS, impact=fx.Impact.POSSIBLE)
log.observe(observation_id=f"o{i}", resource=ref, mechanism=fx.ObservationMechanism.OWNED_RECORD_READ,
coverage=fx.Coverage.PARTIAL, source_action_id=f"r{i}", exists=True)
started = perf_counter()
assessments = log.assessments()
assert perf_counter() - started < 10
assert {a.verdict for a in assessments} == {fx.EffectVerdict.VERIFIED}
def test_classification_failure_claims_unknown_scope(store, monkeypatch):
journal = journal_for(store)
monkeypatch.setattr(adapters, "classify", lambda capture: (_ for _ in ()).throw(KeyError("bug")))