diff --git a/src/agent_runtime/effect_adapters.py b/src/agent_runtime/effect_adapters.py index 34f1fbfc7..16b1593e6 100644 --- a/src/agent_runtime/effect_adapters.py +++ b/src/agent_runtime/effect_adapters.py @@ -191,9 +191,12 @@ def _execution(result: Any, facts: ProducerFacts) -> ExecutionOutcome: return ExecutionOutcome.INTERRUPTED if facts.timed_out: return ExecutionOutcome.TIMED_OUT + # Only launch-shaped results mean this operation's own work continues: + # the native detached launch, or a bridge's explicit detachment. A listing + # that merely reports some other thing as "running" is not. if isinstance(result.get("bg_job_id"), str) and facts.exit_code == 0: return ExecutionOutcome.RUNNING - if result.get("detached") is True or result.get("status") == "running" or result.get("running") is True: + if result.get("detached") is True: return ExecutionOutcome.RUNNING denied = bool(result.get("blocked") or result.get("approval_required") or facts.failure_kind.endswith("_denied")) diff --git a/src/agent_runtime/effects.py b/src/agent_runtime/effects.py index 994557645..59da3bc44 100644 --- a/src/agent_runtime/effects.py +++ b/src/agent_runtime/effects.py @@ -642,14 +642,21 @@ class EffectHistory: raise ValueError("A settled effect outcome cannot be replaced") if outcome.execution is not ExecutionOutcome.RUNNING: settled.add(outcome.effect_id) + # Derived indexes (not fields): outcomes per effect in sequence order. + by_effect: dict[str, list[EffectOutcome]] = {} + for outcome in sorted(self.outcomes, key=lambda o: o.sequence): + by_effect.setdefault(outcome.effect_id, []).append(outcome) + object.__setattr__(self, "_outcomes_by_effect", by_effect) + object.__setattr__(self, "_claims_by_id", {c.effect_id: c for c in self.claims}) def claim(self, effect_id: str) -> EffectClaim | None: - return next((c for c in self.claims if c.effect_id == effect_id), None) + return self._claims_by_id.get(effect_id) def latest_outcome(self, effect_id: str, before: int | None = None) -> EffectOutcome | None: - matching = [o for o in self.outcomes if o.effect_id == effect_id - and (before is None or o.sequence < before)] - return max(matching, key=lambda o: o.sequence) if matching else None + for outcome in reversed(self._outcomes_by_effect.get(effect_id, ())): + if before is None or outcome.sequence < before: + return outcome + return None def execution(self, effect_id: str, before: int | None = None) -> ExecutionOutcome: outcome = self.latest_outcome(effect_id, before) diff --git a/tests/test_effect_resource_bindings.py b/tests/test_effect_resource_bindings.py index 23a378577..39503fccb 100644 --- a/tests/test_effect_resource_bindings.py +++ b/tests/test_effect_resource_bindings.py @@ -33,7 +33,7 @@ def run(ws, tmp_path): journal = ActionJournal(workspace=str(ws), observed_artifacts=("a.txt",)) journal.effects = EffectLog(journal.run_id, directory=tmp_path / "fx") authority = RequestAuthority("request", "alice", "thread", str(ws), tuple( - OperationGrant(tool) for tool in ("write_file", "read_file", "edit_file", "apply_patch", "ls"))) + OperationGrant(tool) for tool in ("write_file", "read_file", "edit_file", "apply_patch", "ls", "private_browser"))) async def call(tool, args): content = args if isinstance(args, str) else json.dumps(args) @@ -218,6 +218,20 @@ def test_listing_is_partial_and_does_not_verify_content(run): assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED] +@pytest.mark.parametrize("args", [ + {"action": "click", "page": "t1", "selector": "#buy"}, + {"action": "open", "url": "https://example.com"}, + {"action": "snapshot", "page": "t1"}, + {"action": "evaluate", "page": "t1", "script": "1"}, +]) +def test_browser_page_operations_stay_fail_closed_with_effects(run, args): + description, result = run("private_browser", args) + assert "UNSUPPORTED" in description + assert result["failure_kind"] == "browser_page_authority_unavailable" and result["executed"] is False + assert run.journal.effects.history().claims == () + assert run.journal.actions[0].execution_id is None + + def test_ordinary_read_only_turn_completes_normally(run, ws): (ws / "a.txt").write_text("existing\n") run("read_file", {"path": "a.txt"}) diff --git a/tests/test_effect_verification_adapters.py b/tests/test_effect_verification_adapters.py index 8af537f81..0376fbbd3 100644 --- a/tests/test_effect_verification_adapters.py +++ b/tests/test_effect_verification_adapters.py @@ -339,6 +339,40 @@ def test_child_effects_share_lineage_order_and_invalidate_parent_evidence(tmp_pa assert fx.freshness(history.observations[0], history) is fx.Freshness.STALE +def test_listing_that_reports_other_work_running_is_not_a_running_effect(store, monkeypatch): + journal = journal_for(store) + act(journal, monkeypatch, adapters.DispatchCapture(), "list_downloads", + result={"output": "1 download", "exit_code": 0, "status": "running", "running": True}) + assert journal.effects.assessments()[0].execution is fx.ExecutionOutcome.REPORTED_SUCCESS + + +def test_scheduler_trigger_is_admission_not_completed_work(store, monkeypatch): + journal = journal_for(store) + act(journal, monkeypatch, adapters.DispatchCapture(), "manage_tasks", + content=json.dumps({"action": "run", "task_id": "t1"}), + result={"output": "Task t1 triggered; it completed successfully.", "exit_code": 0}) + claim = journal.effects.history().claims[0] + assessment = journal.effects.assessments()[0] + # Unbound task control may change anything; its reply verifies nothing. + assert claim.unknown_scope and assessment.verdict is fx.EffectVerdict.UNVERIFIED + + +def test_assessment_scales_to_long_lineages(store): + from time import perf_counter + log = EffectLog("f" * 32, directory=store, durable=False) + owned = [fx.resource_ref(OwnedResource("notes", "u", "t", "notes", f"n{i}", "r"), "record") for i in range(600)] + for i, ref in enumerate(owned): + log.claim(effect_id=f"e{i}", action_id=f"a{i}", operation=fx.OperationRef("manage_notes", "", "0" * 64), + impact_scope=(ref,), obligations=(fx.Postcondition(ref, fx.Predicate.EXISTS),)) + log.outcome(effect_id=f"e{i}", execution=fx.ExecutionOutcome.REPORTED_SUCCESS, impact=fx.Impact.POSSIBLE) + log.observe(observation_id=f"o{i}", resource=ref, mechanism=fx.ObservationMechanism.OWNED_RECORD_READ, + coverage=fx.Coverage.PARTIAL, source_action_id=f"r{i}", exists=True) + started = perf_counter() + assessments = log.assessments() + assert perf_counter() - started < 10 + assert {a.verdict for a in assessments} == {fx.EffectVerdict.VERIFIED} + + def test_classification_failure_claims_unknown_scope(store, monkeypatch): journal = journal_for(store) monkeypatch.setattr(adapters, "classify", lambda capture: (_ for _ in ()).throw(KeyError("bug")))