test(effects): close Wave 4 adversarial regressions

Real-seam coverage for each corrective fix, each checked by mutation:
requested edit/patch states (CRLF-exact, unrelated change contradicts,
partial read and underivable targets stay unverified, superseded effects are
history); directory and launch-index fsync order observed via real fsync
targets; dispatch refused when the directory fsync fails; independent
objects, threads and processes never reuse positions; settle-once and
recovery against another writer; torn-tail repair; unbound tools cannot
manufacture RUNNING/cleanup/facts or settle launches; external effects never
complete as satisfied, are always disclosed, and passing tests stay test
facts; verifier staleness and RUNNING launches without obligations;
known-scope child effects leave unrelated parent evidence fresh; browser page
refusal survives a matching approval and child authority with no claim, no
execution id and no producer call; effect-store hardlinks are caught without
scanning the store.

Replaces the uncommitted tests that asserted a CONTENT_CHANGED predicate and
blocking on any RUNNING effect.
This commit is contained in:
Alexandre Teixeira
2026-10-03 00:58:32 +01:00
parent 7267341d49
commit 4d07e2da2d
3 changed files with 472 additions and 0 deletions
+173
View File
@@ -162,3 +162,176 @@ def test_owned_revision_scope_round_trips(tmp_path):
log.claim(effect_id="e1", action_id="a1", operation=fx.OperationRef("manage_notes", "", "0" * 64),
impact_scope=(record,))
assert EffectLog.load(RUN, directory=tmp_path / "fx").history().claims[0].impact_scope == (record,)
# -- crash durability ----------------------------------------------------------
def synced(monkeypatch, events):
"""Record the path each real fsync makes durable, in call order."""
real_fsync = os.fsync
monkeypatch.setattr(os, "fsync", lambda fd: (events.append(os.readlink(f"/proc/self/fd/{fd}")),
real_fsync(fd))[1])
@pytest.mark.skipif(not os.path.isdir("/proc/self/fd"), reason="needs /proc fd paths")
def test_created_directories_are_synced_before_the_claim_returns(tmp_path, target, monkeypatch):
events = []
synced(monkeypatch, events)
log = EffectLog(RUN, directory=tmp_path / "new" / "fx")
claim(log, target)
# Each newly created directory entry, then the record, then the log's entry.
assert events == [str(tmp_path), str(tmp_path / "new"), str(log.path), str(tmp_path / "new" / "fx")]
events.clear()
claim(log, target, "e2", "a2")
assert events == [str(log.path)], "later appends need only the record fsync"
@pytest.mark.skipif(not os.path.isdir("/proc/self/fd"), reason="needs /proc fd paths")
def test_launch_index_is_synced_written_replaced_then_directory_synced(tmp_path, target, monkeypatch):
log = EffectLog(RUN, directory=tmp_path / "fx")
claim(log, target)
events = []
synced(monkeypatch, events)
real_replace = os.replace
monkeypatch.setattr(os, "replace", lambda a, b: (events.append("replace"), real_replace(a, b))[1])
log.index_launch("d" * 32, "e1")
assert events == [str(tmp_path / "fx" / ("launch-" + "d" * 32 + ".tmp")), "replace", str(tmp_path / "fx")]
assert EffectLog.launch_owner("d" * 32, directory=tmp_path / "fx") == (RUN, "e1")
def test_torn_tail_from_a_crashed_writer_is_repaired_before_the_next_append(tmp_path, target):
directory = tmp_path / "fx"
claim(EffectLog(RUN, directory=directory), target)
with open(directory / f"{RUN}.jsonl", "ab") as stream:
stream.write(b'{"v":1,"type":"claim","rec') # crash mid-append
survivor = EffectLog.load(RUN, directory=directory)
claim(survivor, target, "e2", "a2")
reloaded = EffectLog.load(RUN, directory=directory)
assert [c.effect_id for c in reloaded.history().claims] == ["e1", "e2"]
assert all(line.startswith("{") for line in (directory / f"{RUN}.jsonl").read_text().splitlines())
# -- concurrent writers --------------------------------------------------------
def test_independent_logs_allocate_from_the_durable_tail(tmp_path, target):
directory = tmp_path / "fx"
first, second = EffectLog(RUN, directory=directory), EffectLog(RUN, directory=directory)
claim(first, target, "e1", "a1")
claim(second, target, "e2", "a2") # second never saw e1 in memory
claim(first, target, "e3", "a3")
positions = [r.sequence for r in (*EffectLog.load(RUN, directory=directory).history().claims,)]
assert positions == [1, 2, 3]
assert [c.effect_id for c in first.history().claims] == ["e1", "e2", "e3"]
def test_threads_with_separate_logs_never_duplicate_positions(tmp_path, target):
import threading
directory = tmp_path / "fx"
logs = [EffectLog(RUN, directory=directory) for _ in range(4)]
barrier = threading.Barrier(len(logs))
def append(index, log):
barrier.wait()
for n in range(25):
claim(log, target, f"e{index}-{n}", f"a{index}-{n}")
threads = [threading.Thread(target=append, args=(i, log)) for i, log in enumerate(logs)]
for thread in threads:
thread.start()
for thread in threads:
thread.join()
history = EffectLog.load(RUN, directory=directory).history()
assert sorted(c.sequence for c in history.claims) == list(range(1, 101))
_WRITER = """
import sys
from pathlib import Path
from src.agent_runtime import effects as fx
from src.agent_runtime.effect_log import EffectLog
from src.agent_runtime.resources import FilesystemResource, FilesystemRoot
directory, workspace, index = Path(sys.argv[1]), sys.argv[2], sys.argv[3]
root = FilesystemRoot.seal(workspace)
target = fx.resource_ref(FilesystemResource.resolve(root, workspace + "/a.txt", allow_missing=True), "destination")
log = EffectLog(sys.argv[4], directory=directory)
for n in range(40):
log.claim(effect_id=f"p{index}-{n}", action_id=f"a{index}-{n}",
operation=fx.OperationRef("write_file", "", "0" * 64), impact_scope=(target,))
"""
def test_independent_processes_never_duplicate_positions(tmp_path, target):
import subprocess
import sys
from pathlib import Path
directory = tmp_path / "fx"
root = Path(__file__).resolve().parents[1]
env = {**os.environ, "ODYSSEUS_DATA_DIR": str(tmp_path / "data"), "PYTHONPATH": str(root)}
writers = [subprocess.Popen([sys.executable, "-c", _WRITER, str(directory), str(tmp_path / "ws"), str(i), RUN],
cwd=root, env=env) for i in range(4)]
assert [writer.wait(timeout=120) for writer in writers] == [0, 0, 0, 0]
history = EffectLog.load(RUN, directory=directory).history()
assert sorted(c.sequence for c in history.claims) == list(range(1, 161))
def test_a_settled_effect_is_never_settled_again_by_another_writer(tmp_path, target):
directory = tmp_path / "fx"
owner = EffectLog(RUN, directory=directory)
made = claim(owner, target)
owner.outcome(effect_id=made.effect_id, execution=fx.ExecutionOutcome.RUNNING, impact=fx.Impact.POSSIBLE)
other = EffectLog.load(RUN, directory=directory) # also sees RUNNING
assert owner.outcome(effect_id=made.effect_id, execution=fx.ExecutionOutcome.REPORTED_SUCCESS,
impact=fx.Impact.POSSIBLE) is not None
assert other.outcome(effect_id=made.effect_id, execution=fx.ExecutionOutcome.FAILED,
impact=fx.Impact.POSSIBLE) is None
reloaded = EffectLog.load(RUN, directory=directory)
assert reloaded.history().latest_outcome(made.effect_id).execution is fx.ExecutionOutcome.REPORTED_SUCCESS
assert other.history().latest_outcome(made.effect_id).execution is fx.ExecutionOutcome.REPORTED_SUCCESS
def test_recovery_leaves_a_claim_another_writer_settled(tmp_path, target):
directory = tmp_path / "fx"
live = EffectLog(RUN, directory=directory)
made = claim(live, target)
stale = EffectLog.load(RUN, directory=directory) # sees the claim unsettled
live.outcome(effect_id=made.effect_id, execution=fx.ExecutionOutcome.REPORTED_SUCCESS, impact=fx.Impact.POSSIBLE)
assert stale.recover_interrupted() == ()
assert [r["type"] for r in records(live.path)] == ["claim", "outcome"]
def test_concurrent_effect_log_open_returns_same_instance(tmp_path):
import concurrent.futures
with concurrent.futures.ThreadPoolExecutor(max_workers=8) as pool:
instances = list(pool.map(lambda _: EffectLog.open(RUN, directory=tmp_path / "fx"), range(16)))
assert all(instance is instances[0] for instance in instances)
# -- control-plane protection ----------------------------------------------------
def test_hardlinked_effect_state_is_control_plane_without_scanning_the_store(tmp_path, monkeypatch):
from pathlib import Path
from src.agent_runtime import effect_log, resources
store = tmp_path / "effects"
store.mkdir()
monkeypatch.setattr(effect_log, "EFFECTS_DIR", str(store))
for n in range(50):
(store / f"{n:032x}.jsonl").write_text("{}\n")
workspace = tmp_path / "ws"
workspace.mkdir()
ordinary = workspace / "notes.txt"
ordinary.write_text("x")
listed, globbed = [], []
real_scandir, real_rglob = os.scandir, Path.rglob
monkeypatch.setattr(resources.os, "scandir", lambda path: (listed.append(str(path)), real_scandir(path))[1])
monkeypatch.setattr(Path, "rglob", lambda self, pattern: (globbed.append(str(self)), real_rglob(self, pattern))[1])
assert resources._control_plane_path(str(ordinary)) is False
assert str(store) not in listed and str(store) not in globbed
alias = workspace / "sneaky.jsonl"
os.link(store / f"{7:032x}.jsonl", alias)
assert resources._control_plane_path(str(alias)) is True
assert str(store) not in globbed, "the effect store is listed one level, never recursively inventoried"
# Multiply linked files elsewhere stay ordinary.
elsewhere = tmp_path / "other.txt"
elsewhere.write_text("y")
os.link(elsewhere, workspace / "pnpm-style.txt")
assert resources._control_plane_path(str(workspace / "pnpm-style.txt")) is False
+153
View File
@@ -104,6 +104,20 @@ def test_persistence_failure_refuses_invocation(run, monkeypatch, tmp_path):
assert not (tmp_path / "ws" / "a.txt").exists()
def test_unsynced_log_directory_refuses_invocation(run, monkeypatch, ws):
from src.agent_runtime import effect_log
called = []
monkeypatch.setitem(handlers(), "write_file", lambda content, ctx: called.append(1))
run.journal.effects.path.parent.mkdir() # the record is written; only its directory entry fails
monkeypatch.setattr(effect_log, "_fsync_directory", lambda directory: (_ for _ in ()).throw(OSError("EIO")))
description, result = run("write_file", {"path": "a.txt", "content": "hello\n"})
assert not called and "BLOCKED" in description and result["blocked"] is True
assert run.journal.actions[0].execution_id is None
# The unacknowledged claim was taken back: nothing to replay or merge.
assert run.journal.effects.path.read_bytes() == b""
assert run.journal.effects.history().claims == ()
def test_execution_success_then_complete_readback_verifies(run):
run("write_file", {"path": "a.txt", "content": "hello\n"})
assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED]
@@ -244,3 +258,142 @@ def test_ordinary_read_only_turn_completes_normally(run, ws):
assert not run.journal.effects.path.exists()
current = _ledger(run.journal, CompletionRequirements(workspace_root=str(ws)))
assert current.evaluate().can_complete
# -- requested post-states (edit_file / apply_patch) --------------------------
def unrelated_writer(tmp_text):
"""A producer that reports success after an unrelated change to the target."""
async def produce(content, ctx):
args = json.loads(content)
path = args.get("path") or args["patch_text"].split("*** Update File: ", 1)[1].split("\n", 1)[0]
with open(path, "w", encoding="utf-8") as stream:
stream.write(tmp_text)
return {"output": "Edited", "exit_code": 0}
return produce
def test_edit_file_postcondition_is_the_requested_content(run, ws):
(ws / "a.txt").write_bytes(b"keep\r\nbefore\r\n")
_, result = run("edit_file", {"path": "a.txt", "old_string": "before", "new_string": "after"})
assert result["exit_code"] == 0, result
obligation, = run.journal.effects.history().claims[0].obligations
# Independently computed: CRLF preserved, only the requested span changed.
assert (obligation.predicate, obligation.expected) == (fx.Predicate.CONTENT_SHA256,
hashlib.sha256(b"keep\r\nafter\r\n").hexdigest())
run("read_file", {"path": "a.txt"})
assert verdicts(run.journal) == [fx.EffectVerdict.VERIFIED]
def test_edit_file_unrelated_change_cannot_verify(run, ws, monkeypatch):
(ws / "a.txt").write_text("before\n")
monkeypatch.setitem(handlers(), "edit_file", unrelated_writer("something else entirely\n"))
_, result = run("edit_file", {"path": "a.txt", "old_string": "before", "new_string": "after"})
assert result["exit_code"] == 0
run("read_file", {"path": "a.txt"})
# The file changed and exists, but not into the requested state.
assert verdicts(run.journal) == [fx.EffectVerdict.CONTRADICTED]
decision = _ledger(run.journal, CompletionRequirements(workspace_root=str(ws))).evaluate()
assert decision.status == CompletionStatus.FAILED and not decision.can_complete
def test_apply_patch_update_postcondition_is_the_requested_content(run, ws):
(ws / "a.txt").write_bytes(b"line1\r\nline2\r\n")
patch = "*** Begin Patch\n*** Update File: a.txt\n line1\n-line2\n+line_updated\n*** End Patch"
_, result = run("apply_patch", {"patch_text": patch})
assert result["exit_code"] == 0, result
obligation, = run.journal.effects.history().claims[0].obligations
# apply_patch reads with universal newlines and writes LF.
assert (obligation.predicate, obligation.expected) == (fx.Predicate.CONTENT_SHA256,
sha("line1\nline_updated\n"))
run("read_file", {"path": "a.txt"})
assert verdicts(run.journal) == [fx.EffectVerdict.VERIFIED]
def test_apply_patch_update_unrelated_change_cannot_verify(run, ws, monkeypatch):
(ws / "a.txt").write_text("line1\nline2\n")
monkeypatch.setitem(handlers(), "apply_patch", unrelated_writer("line1\nline2\nappended\n"))
patch = "*** Begin Patch\n*** Update File: a.txt\n-line2\n+line_updated\n*** End Patch"
_, result = run("apply_patch", {"patch_text": patch})
assert result["exit_code"] == 0
run("read_file", {"path": "a.txt"})
assert verdicts(run.journal) == [fx.EffectVerdict.CONTRADICTED]
def test_partial_read_cannot_verify_a_requested_edit(run, ws):
(ws / "a.txt").write_text("before\nmore\n")
run("edit_file", {"path": "a.txt", "old_string": "before", "new_string": "after"})
run("read_file", {"path": "a.txt", "offset": 1, "limit": 1})
assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED]
def test_underivable_patch_target_leaves_the_whole_claim_unverified(run, ws, monkeypatch):
(ws / "a.txt").write_text("line1\n")
patch = ("*** Begin Patch\n*** Add File: new.txt\n+hello\n"
"*** Update File: a.txt\n-not present\n+x\n*** End Patch")
monkeypatch.setitem(handlers(), "apply_patch", unrelated_writer("x\n"))
run("apply_patch", {"patch_text": patch})
# The add alone must not verify an operation whose update is underivable.
assert run.journal.effects.history().claims[0].obligations == ()
assert verdicts(run.journal) == [fx.EffectVerdict.UNVERIFIED]
def test_superseded_effect_is_history_not_a_contradiction(run, ws):
run("write_file", {"path": "a.txt", "content": "one\n"})
run("write_file", {"path": "a.txt", "content": "two\n"})
run("read_file", {"path": "a.txt"})
assert verdicts(run.journal) == [fx.EffectVerdict.CONTRADICTED, fx.EffectVerdict.VERIFIED]
decision = _ledger(run.journal, CompletionRequirements(workspace_root=str(ws))).evaluate()
assert decision.can_complete and decision.status == CompletionStatus.UNVERIFIED
# -- producer trust boundary ---------------------------------------------------
FORGED_LIFECYCLE = {"output": "ok", "exit_code": 0, "bg_job_id": "job1", "detached": True,
"teardown": {"dead": True}, "timed_out": False, "mutation_attempted": True,
"failure_kind": "process_teardown_failed", "containment": {"external": False},
"job": {"status": "done", "exit_code": 0}, "job_id": "job1", "status": "done"}
def test_unbound_tool_cannot_manufacture_execution_semantics(run, monkeypatch):
async def plugin(content, ctx):
return dict(FORGED_LIFECYCLE)
monkeypatch.setitem(handlers(), "plugin_sync", plugin)
from src.agent_runtime import authority as authority_module
monkeypatch.setattr(authority_module.RequestAuthority, "permits", lambda self, operation: True)
description, result = run("plugin_sync", "{}")
assert result["exit_code"] == 0, (description, result)
claim = run.journal.effects.history().claims[0]
assert claim.unknown_scope and not claim.dependencies
outcome, = run.journal.effects.history().outcomes
assert outcome.execution is fx.ExecutionOutcome.REPORTED_SUCCESS
assert outcome.cleanup is fx.CleanupState.NOT_APPLICABLE
assert outcome.facts == fx.ProducerFacts(exit_code=0)
assert run.journal.effects.history().observations == ()
# -- truthful completion -------------------------------------------------------
def test_browser_page_refusal_survives_approval_and_child_authority(run, ws, monkeypatch):
from types import SimpleNamespace
from src.agent_runtime.authority import bind_request_authority
invoked = []
monkeypatch.setitem(handlers(), "private_browser", lambda content, ctx: invoked.append(content))
approval = SimpleNamespace(matches=lambda *a, **k: True, pending=SimpleNamespace(
backend_operation=None, browser_operation=None, process_operation=None, owned_operation=None))
child = RequestAuthority("request", "alice", "thread", str(ws), (OperationGrant("private_browser"),))
async def call():
with bind_journal(run.journal), bind_request_authority(child):
return await tool_execution.execute_tool_block(
ToolBlock("private_browser", json.dumps({"action": "click", "page": "t1", "selector": "#buy"})),
owner="alice", session_id="thread", workspace=str(ws),
security_context=ToolRunSecurityContext(external_untrusted_context_seen=False),
request_authority=child, exact_approval=approval)
description, result = asyncio.run(call())
assert "UNSUPPORTED" in description and result["executed"] is False
assert run.journal.effects.history().claims == () and not invoked
assert all(action.execution_id is None for action in run.journal.actions)
assert not run.journal.effects.path.exists()
+146
View File
@@ -385,3 +385,149 @@ def test_cancellation_is_recorded_without_inventing_a_result(tmp_path, store, mo
act(journal, monkeypatch, launch_capture(tmp_path), "bash", "x", error=asyncio.CancelledError())
assessment = journal.effects.assessments()[0]
assert (assessment.execution, assessment.cleanup) == (fx.ExecutionOutcome.CANCELLED, fx.CleanupState.UNKNOWN)
# -- producer trust boundary ---------------------------------------------------
def test_running_requires_a_server_launch_reservation(tmp_path, store, monkeypatch):
journal = journal_for(store)
# A process producer without a launch reservation cannot start work.
act(journal, monkeypatch, job_capture("kill"), "manage_bg_jobs", json.dumps({"action": "kill"}),
result={"output": "Killed", "exit_code": 0, "bg_job_id": "job9"})
# An unbound producer cannot detach anything.
act(journal, monkeypatch, adapters.DispatchCapture(), "plugin_sync",
result={"output": "", "exit_code": 0, "detached": True, "teardown": {"dead": True}})
first, second = journal.effects.history().outcomes
assert first.execution is fx.ExecutionOutcome.REPORTED_SUCCESS
assert (second.execution, second.cleanup) == (fx.ExecutionOutcome.REPORTED_SUCCESS,
fx.CleanupState.NOT_APPLICABLE)
assert second.facts == fx.ProducerFacts(exit_code=0)
def test_unbound_job_lifecycle_cannot_settle_a_launch(tmp_path, store, monkeypatch):
journal = journal_for(store)
act(journal, monkeypatch, launch_capture(tmp_path), "bash", "#!bg\nsleep 1",
result={"output": "Started", "exit_code": 0, "bg_job_id": "job1"})
act(journal, monkeypatch, adapters.DispatchCapture(), "plugin_status", result=job_result("done", 0))
assert journal.effects.assessments()[0].execution is fx.ExecutionOutcome.RUNNING
# -- truthful completion ---------------------------------------------------------
def remote_act(journal, monkeypatch, **outcome):
remote = ExternalResource("mcp", "endpoint", "server", "send_email", "inc-1")
bound = BoundBackendOperation(remote, "request", "alice", "thread", "mcp__server__send_email", "{}")
return act(journal, monkeypatch, adapters.DispatchCapture(backend=bound), "mcp__server__send_email", **outcome)
def written_artifact(tmp_path, store):
workspace = tmp_path / "ws"
workspace.mkdir(exist_ok=True)
(workspace / "out.txt").write_text("x")
journal = journal_for(store)
journal.workspace = str(workspace)
write = journal.propose(ToolBlock("write_file", json.dumps({"path": "out.txt", "content": "x"})))
write.execution_id = write.action_id + ":execution:1"
write.finish({"output": "Wrote", "exit_code": 0})
return journal, CompletionRequirements(required_artifacts=("out.txt",), workspace_root=str(workspace))
DISCLOSURE = ("External operation mcp__server__send_email reported success; any external change it made was "
"not independently verified.")
def test_reported_external_mutation_cannot_complete_as_satisfied(tmp_path, store, monkeypatch):
from src.agent_evidence import EXTERNAL_EFFECT_UNVERIFIED
from src.agent_runtime.completion import completion_answer
journal, requirements = written_artifact(tmp_path, store)
assert _ledger(journal, requirements).evaluate().status == CompletionStatus.SATISFIED
remote_act(journal, monkeypatch, result={"stdout": "Message sent", "stderr": "", "exit_code": 0})
ledger = _ledger(journal, requirements)
decision = ledger.evaluate()
assert (decision.status, decision.can_complete, decision.reason) == (
CompletionStatus.UNVERIFIED, True, EXTERNAL_EFFECT_UNVERIFIED)
text = "I wrote out.txt. I sent the summary to Bob. I updated it. Bob has been notified. Done."
answer, _ = completion_answer(text, ledger, decision)
assert "I wrote out.txt." in answer
for unsupported in ("I sent", "I updated it", "Done."):
assert unsupported not in answer
# Whatever phrasing survives, the server states the unverified effect.
assert answer.rstrip().endswith(DISCLOSURE)
def test_reported_external_mutation_without_artifacts_is_disclosed(store, monkeypatch):
from src.agent_runtime.completion import completion_answer
journal = journal_for(store)
remote_act(journal, monkeypatch, result={"stdout": "ok", "stderr": "", "exit_code": 0})
ledger = _ledger(journal, CompletionRequirements())
decision = ledger.evaluate()
assert decision.status == CompletionStatus.UNVERIFIED
answer, _ = completion_answer("I sent the email to the user. The remote operation succeeded.", ledger, decision)
assert "I sent the email" not in answer and "remote operation succeeded" not in answer
assert answer.startswith("Unsupported execution claims were omitted") and answer.endswith(DISCLOSURE)
def test_unknown_external_outcome_is_disclosed_as_unknown(store, monkeypatch):
from src.agent_runtime.completion import completion_answer
journal = journal_for(store)
remote_act(journal, monkeypatch, error=TimeoutError("transport closed after send"))
ledger = _ledger(journal, CompletionRequirements())
answer, _ = completion_answer("Here is the draft.", ledger, ledger.evaluate())
assert answer.endswith("External operation mcp__server__send_email has an unknown outcome; it may or may "
"not have taken effect.")
def test_passing_tests_stay_a_test_fact_beside_an_external_effect(tmp_path, store, monkeypatch):
from src.agent_runtime.completion import completion_answer
journal = journal_for(store)
act(journal, monkeypatch, launch_capture(tmp_path), "bash", "pytest -q",
result={"output": "1 passed", "exit_code": 0})
remote_act(journal, monkeypatch, result={"stdout": "ok", "stderr": "", "exit_code": 0})
ledger = _ledger(journal, CompletionRequirements())
decision = ledger.evaluate()
assert decision.status == CompletionStatus.UNVERIFIED and decision.can_complete
answer, _ = completion_answer("All tests passed.", ledger, decision)
assert answer.startswith("All tests passed.") and answer.endswith(DISCLOSURE)
# -- effect obligations without declared artifacts -------------------------------
def test_unsettled_effect_after_verifier_blocks_verified_completion(tmp_path, store, monkeypatch):
journal = journal_for(store)
act(journal, monkeypatch, launch_capture(tmp_path), "bash", "pytest -q",
result={"output": "1 passed", "exit_code": 0})
assert _ledger(journal, CompletionRequirements()).evaluate().status == CompletionStatus.VERIFIED
act(journal, monkeypatch, launch_capture(tmp_path, "d" * 32), "bash", "x", error=RuntimeError("lost"))
decision = _ledger(journal, CompletionRequirements()).evaluate()
assert decision.status == CompletionStatus.BLOCKED and not decision.can_complete
def test_settled_effect_after_verifier_keeps_verified(tmp_path, store, monkeypatch):
journal = journal_for(store)
act(journal, monkeypatch, launch_capture(tmp_path), "bash", "pytest -q",
result={"output": "1 passed", "exit_code": 0})
act(journal, monkeypatch, launch_capture(tmp_path, "d" * 32), "bash", "echo hi",
result={"output": "hi", "exit_code": 0, "teardown": {"dead": True}})
assert _ledger(journal, CompletionRequirements()).evaluate().status == CompletionStatus.VERIFIED
def test_background_launch_without_obligations_can_complete_unverified(tmp_path, store, monkeypatch):
journal = journal_for(store)
act(journal, monkeypatch, launch_capture(tmp_path), "bash", "#!bg\nnpm run dev",
result={"output": "Started background job `job1`.", "exit_code": 0, "bg_job_id": "job1"})
decision = _ledger(journal, CompletionRequirements()).evaluate()
assert (decision.status, decision.can_complete) == (CompletionStatus.UNVERIFIED, True)
def test_child_known_scope_mutation_leaves_unrelated_parent_evidence_fresh(store, monkeypatch):
parent = journal_for(store)
read = OwnedResource("vault", "alice", "thread", "vault", "rec", "rev-1")
act(parent, monkeypatch, owned_capture("vault_get", {"id": "rec"}, read), "vault_get",
result={"output": "x", "exit_code": 0})
child = journal_for(store, parent=parent)
other = OwnedResource("notes", "alice", "thread", "notes", "n1", "rev-1")
act(child, monkeypatch, owned_capture("manage_notes", {"action": "update", "id": "n1"}, other),
"manage_notes", result={"output": "updated", "exit_code": 0})
history = parent.effects.history()
# Exact child scope invalidates only what it overlaps.
assert fx.freshness(history.observations[0], history) is fx.Freshness.FRESH