Merge pull request #53 from pewdiepie-archdaemon/feature/runtime-containment

feat(runtime): enforce shared native execution containment
This commit is contained in:
Alexandre Teixeira
2026-10-02 00:44:43 +01:00
committed by GitHub
49 changed files with 7776 additions and 1011 deletions
+8
View File
@@ -1345,6 +1345,14 @@ async def _startup_event():
from src.cookbook_serve_lifecycle import cookbook_serve_lifecycle_loop
_startup_tasks.append(asyncio.create_task(cookbook_serve_lifecycle_loop()))
# Reconcile the processes a previous run left behind: tear down orphaned
# containment grants, and stop trusting background-job records whose pid the
# kernel has since reassigned. Runs once, and deliberately runs *here* —
# every record it sees predates this run, which is what makes "I cannot
# identify this process" a safe thing to act on. See src/process_reaper.py.
from src.process_reaper import reap_orphans_at_startup
_startup_tasks.append(asyncio.create_task(reap_orphans_at_startup()))
logger.info("Application startup complete")
async def _shutdown_event():
+40 -1
View File
@@ -16,9 +16,48 @@ from __future__ import annotations
import json
import os
import uuid
import functools
import threading
from typing import Any, Optional
_STORE_LOCKS: dict[str, threading.RLock] = {}
_STORE_LOCKS_GUARD = threading.Lock()
def store_transaction(path_factory):
"""Serialize a JSON read/modify/write across runtime threads and processes."""
def decorate(function):
@functools.wraps(function)
def locked(*args, **kwargs):
path = os.path.abspath(str(path_factory())) + ".lock"
with _STORE_LOCKS_GUARD:
lock = _STORE_LOCKS.setdefault(path, threading.RLock())
with lock:
os.makedirs(os.path.dirname(path), exist_ok=True)
with open(path, "a+b") as handle:
if os.name == "nt":
import msvcrt
if os.fstat(handle.fileno()).st_size == 0:
handle.write(b"0")
handle.flush()
handle.seek(0)
msvcrt.locking(handle.fileno(), msvcrt.LK_LOCK, 1)
else:
import fcntl
fcntl.flock(handle, fcntl.LOCK_EX)
try:
return function(*args, **kwargs)
finally:
if os.name == "nt":
handle.seek(0)
msvcrt.locking(handle.fileno(), msvcrt.LK_UNLCK, 1)
else:
fcntl.flock(handle, fcntl.LOCK_UN)
return locked
return decorate
def atomic_write_json(path: str, data: Any, *, indent: Optional[int] = None) -> None:
"""Atomically persist `data` as JSON at `path`.
@@ -64,4 +103,4 @@ def atomic_write_text(path: str, text: str) -> None:
try:
os.unlink(tmp)
except OSError:
pass
pass
+31 -34
View File
@@ -94,7 +94,13 @@ def pid_alive(pid: Optional[int]) -> bool:
the process it is checking. We instead open the process and read its exit
code via the Win32 API.
"""
if not pid:
if pid is None:
return False
try:
pid_int = int(pid)
except (TypeError, ValueError):
return False
if pid_int <= 0:
return False
if IS_WINDOWS:
import ctypes
@@ -104,54 +110,45 @@ def pid_alive(pid: Optional[int]) -> bool:
STILL_ACTIVE = 259
kernel32 = ctypes.windll.kernel32
handle = kernel32.OpenProcess(
PROCESS_QUERY_LIMITED_INFORMATION, False, int(pid)
PROCESS_QUERY_LIMITED_INFORMATION, False, pid_int
)
if not handle:
return False
return kernel32.GetLastError() != 87 # ERROR_INVALID_PARAMETER: PID absent
try:
code = wintypes.DWORD()
if kernel32.GetExitCodeProcess(handle, ctypes.byref(code)):
return code.value == STILL_ACTIVE
return False
return True # A failed probe does not establish death.
finally:
kernel32.CloseHandle(handle)
try:
os.kill(pid, 0)
os.kill(pid_int, 0)
return True
except (OSError, ProcessLookupError):
except ProcessLookupError:
return False
except OSError:
return True # EPERM and other inspection failures are not ESRCH.
def kill_process_tree(pid: Optional[int]) -> None:
"""Terminate ``pid`` and all of its descendants.
def kill_process_tree(pid: Optional[int], *, start_token=None, pgid=None, require_identity=False):
"""Use the runtime's shared escalating teardown and return verified death.
POSIX: signal the whole process group (``killpg``), falling back to a plain
``kill`` if the pid isn't a group leader.
Windows: ``taskkill /T /F`` walks and kills the child tree (there is no
process-group signalling).
Callers retaining durable PIDs must validate their recorded identity before
calling this compatibility entry point. Native grants retain identity at
spawn and use containment.release directly.
"""
if not pid:
return
if IS_WINDOWS:
try:
subprocess.run(
["taskkill", "/F", "/T", "/PID", str(pid)],
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
creationflags=getattr(subprocess, "CREATE_NO_WINDOW", 0),
)
except Exception:
pass
return
import signal
try:
os.killpg(os.getpgid(pid), signal.SIGTERM)
except Exception:
try:
os.kill(pid, signal.SIGTERM)
except Exception:
pass
from src import containment
if not pid or int(pid) <= 0:
return containment.ReleaseOutcome(dead=True, escalated=False)
spec = containment.ContainmentSpec(workspace=os.getcwd(), env={}, wall_clock_s=1,
required=frozenset())
grant = containment.ContainmentGrant(
id="", mechanism="windows_tree" if IS_WINDOWS else "process_group",
workspace=spec.workspace, enforced=frozenset(), degraded=(),
unenforced_required=(), owner="compatibility", mode=containment.CONTAINMENT_MODE,
spec=spec, pid=int(pid), pgid=pgid or containment._pgid_of(int(pid)),
)
return containment.release(grant, start_token=start_token, require_identity=require_identity)
# ── Shell / executable resolution ───────────────────────────────────────────
@@ -0,0 +1,128 @@
[
"tests/test_app_db_permissions.py::test_app_db_created_with_0600",
"tests/test_app_db_permissions.py::test_app_db_sidecars_relocked",
"tests/test_app_db_permissions.py::test_app_db_file_uri_created_with_0600",
"tests/test_app_db_permissions.py::test_app_db_localhost_file_uri_created_with_0600",
"tests/test_app_db_permissions.py::test_app_db_non_uri_mode_query_created_with_0600",
"tests/test_app_db_permissions.py::test_app_db_plain_file_uri_created_with_0600",
"tests/test_auth_config_lock_concurrency.py::TestConcurrentCreateUser::test_parallel_creates_no_lost_users",
"tests/test_auth_config_lock_concurrency.py::TestConcurrentCreateUser::test_parallel_creates_same_username_only_one_wins",
"tests/test_auth_config_lock_concurrency.py::TestConcurrentDeleteUser::test_parallel_deletes_no_corruption",
"tests/test_auth_config_lock_concurrency.py::TestConcurrentRenameUser::test_parallel_renames_no_lost_users",
"tests/test_auth_config_lock_concurrency.py::TestConcurrentMixedOperations::test_mixed_operations_no_corruption",
"tests/test_auth_config_lock_concurrency.py::TestDiskConsistency::test_file_always_valid_json_during_concurrent_ops",
"tests/test_auth_root_path.py::test_real_auth_middleware_uses_application_relative_path",
"tests/test_caldav_bidirectional_sync.py::test_event_to_ical_serializes_core_fields_and_rrule",
"tests/test_caldav_google_principal_url.py::test_google_sync_pulls_events_instead_of_empty",
"tests/test_caldav_writeback.py::test_build_ical_timed_event_has_core_fields",
"tests/test_caldav_writeback.py::test_build_ical_all_day_uses_date_values",
"tests/test_caldav_writeback.py::test_build_ical_includes_rrule",
"tests/test_caldav_writeback.py::test_push_create_calls_save_event",
"tests/test_caldav_writeback.py::test_push_update_overwrites_existing",
"tests/test_doc_library_open_orphaned.py::test_mobile_explicit_load_restores_full_editor_from_bottom_dock",
"tests/test_document_followup_integrity.py::test_unavailable_active_target_never_falls_back_to_other_document[deleted-document-edit_document]",
"tests/test_document_followup_integrity.py::test_unavailable_active_target_never_falls_back_to_other_document[deleted-document-update_document]",
"tests/test_document_followup_integrity.py::test_unavailable_active_target_never_falls_back_to_other_document[foreign-document-edit_document]",
"tests/test_document_followup_integrity.py::test_unavailable_active_target_never_falls_back_to_other_document[foreign-document-update_document]",
"tests/test_document_followup_integrity.py::test_targeted_edit_and_undo_preserve_other_occurrences",
"tests/test_document_followup_integrity.py::test_no_target_legacy_fallback_still_scopes_to_owner",
"tests/test_document_followup_integrity.py::test_invalid_multi_edit_saves_only_exact_matches_and_reports_remainder",
"tests/test_document_followup_integrity.py::test_batch_with_only_bad_anchors_reports_all_without_saving",
"tests/test_document_followup_integrity.py::test_long_proofreading_batch_saves_safe_matches_and_identifies_remainder",
"tests/test_document_followup_integrity.py::test_inline_suggestion_is_reviewable_then_applies_only_its_target",
"tests/test_document_followup_integrity.py::test_whole_document_update_persists_exact_replacement",
"tests/test_document_followup_integrity.py::test_ambiguous_or_partial_word_edits_do_not_mutate[alpha-beta]",
"tests/test_document_followup_integrity.py::test_ambiguous_or_partial_word_edits_do_not_mutate[vio-new]",
"tests/test_document_followup_integrity.py::test_ambiguous_or_partial_word_edits_do_not_mutate[tha-that]",
"tests/test_document_followup_integrity.py::test_explicit_replace_all_corrects_every_occurrence",
"tests/test_document_followup_integrity.py::test_replace_all_cannot_change_fragments_of_correct_words",
"tests/test_document_followup_integrity.py::test_ambiguous_suggestion_returns_exact_recovery_anchors",
"tests/test_document_followup_integrity.py::test_mixed_suggestion_batch_queues_valid_items_and_reports_bad_anchors",
"tests/test_document_history_controls.py::test_mobile_rich_text_history_state_and_document_switch",
"tests/test_document_library_mobile_footer.py::test_mobile_open_in_new_chat_copies_to_materialized_session",
"tests/test_document_module_api.py::test_default_export_surface_is_complete_and_callable",
"tests/test_document_module_api.py::test_named_exports_survive_and_stay_callable",
"tests/test_document_module_api.py::test_window_bridge_is_the_default_export",
"tests/test_document_outline.py::test_outline_jumps_in_markdown_and_rich_text_and_fits_mobile",
"tests/test_document_rich_checklist_enter.py::test_enter_creates_unchecked_task_and_empty_enter_exits_cleanly",
"tests/test_document_rich_color_reset_and_contrast.py::test_rich_colors_follow_theme_and_undo_as_one_edit",
"tests/test_document_rich_docx_export.py::test_browser_word_export_contains_native_rich_docx_ooxml",
"tests/test_document_rich_docx_export.py::test_browser_markdown_word_export_keeps_heading_and_inline_formatting",
"tests/test_document_rich_find_boundaries.py::test_find_rejects_cross_block_matches_but_supports_inline_matches_and_replacement",
"tests/test_document_rich_font_color_controls.py::test_numeric_font_size_and_custom_colors_work_on_desktop_and_mobile",
"tests/test_document_rich_heading_enter.py::test_mobile_heading_enter_exits_cleanly_and_is_one_step_undoable",
"tests/test_document_rich_heading_enter.py::test_heading_enter_preserves_shift_middle_and_empty_heading_semantics",
"tests/test_document_rich_image_caption.py::test_mobile_image_caption_survives_resize_history_and_empty_removal",
"tests/test_document_rich_input_rules.py::test_typing_markers_converts_blocks_and_preserves_following_text",
"tests/test_document_rich_keyboard_shortcuts.py::test_rich_document_shortcuts_work_at_desktop_and_mobile_widths",
"tests/test_document_rich_selection_toolbar.py::test_selection_toolbar_formats_and_stays_inside_desktop_and_mobile_viewports",
"tests/test_document_rich_slash_menu.py::test_slash_menu_filters_converts_blocks_inserts_tables_and_fits_mobile",
"tests/test_document_rich_smart_link_paste.py::test_rich_url_paste_links_selections_and_plain_urls_without_unsafe_autolinks",
"tests/test_document_rich_structure_tools.py::test_mobile_headings_page_break_history_and_persistence",
"tests/test_document_rich_table_cell_alignment.py::test_mobile_table_cell_alignment_tracks_state_and_native_history",
"tests/test_document_rich_table_header_preservation.py::test_mobile_structural_edits_preserve_header_modes_and_history",
"tests/test_document_rich_table_headers.py::test_mobile_header_row_and_column_toggle_independently_with_undo",
"tests/test_document_rich_table_merge_split.py::test_mobile_merge_split_round_trip_preserves_headers_formatting_and_history",
"tests/test_document_rich_table_tab_history.py::test_mobile_table_tab_navigation_row_creation_and_history",
"tests/test_document_rich_toolbar_menus.py::test_mobile_toolbar_uses_native_momentum_and_distinct_activation_tokens",
"tests/test_document_rich_toolbar_menus.py::test_mobile_toolbar_menu_preserves_selection_and_restores_focus",
"tests/test_document_rich_toolbar_menus.py::test_rich_toolbar_menus_track_live_formatting_values",
"tests/test_document_save_shortcut.py::test_ctrl_s_saves_rich_text_immediately_once_and_updates_status",
"tests/test_document_save_status.py::test_save_status_is_dirty_race_safe_and_reports_failures",
"tests/test_document_toolbar_order.py::test_rich_toolbar_rendered_order_is_stable_on_desktop_and_mobile",
"tests/test_email_library_module_graph_js.py::test_every_package_module_evaluates_on_its_own_in_a_browser",
"tests/test_email_library_module_graph_js.py::test_wrapper_and_entry_module_hand_out_the_same_functions",
"tests/test_email_package_compatibility.py::test_legacy_email_modules_alias_canonical_module_objects",
"tests/test_escape_inner_layers.py::test_rich_escape_closes_toolbar_then_selection_badge",
"tests/test_escape_inner_layers.py::test_email_escape_closes_inner_states_without_closing_library",
"tests/test_extract_text_tool.py::test_extract_text_renders_and_ocr_scans_pdf_pages",
"tests/test_history_resume_rendering_js.py::test_history_resume_rendering_browser_suite",
"tests/test_image_provider_transport.py::test_image_provider_protocol[https://openrouter.ai/api/v1-True]",
"tests/test_image_provider_transport.py::test_image_provider_protocol[https://openrouter.ai/api/v1-False]",
"tests/test_image_provider_transport.py::test_image_provider_protocol[https://api.openai.com/v1-True]",
"tests/test_image_provider_transport.py::test_image_provider_protocol[https://api.openai.com/v1-False]",
"tests/test_live_fallback_round_attribution.py::test_detached_resume_reconciles_canonical_terminal_failures",
"tests/test_live_fallback_round_attribution.py::test_detached_resume_surfaces_fallback_then_provider_alias_without_reload",
"tests/test_live_fallback_round_attribution.py::test_detached_resume_renders_preoutput_error_without_empty_reload",
"tests/test_manage_tasks_cron.py::test_cron_create_edit_resume_and_invalid_edit_rollback",
"tests/test_manage_tasks_cron.py::test_named_weekdays_create_and_edit_preserve_actual_clock",
"tests/test_manage_tasks_cron.py::test_time_only_edit_changes_cron_clock_not_calendar_fields[15 9 * * 1,3,5]",
"tests/test_manage_tasks_cron.py::test_time_only_edit_changes_cron_clock_not_calendar_fields[15 9 15 * *]",
"tests/test_manage_tasks_cron.py::test_time_only_edit_changes_cron_clock_not_calendar_fields[0,30 8-10 * * 2,4]",
"tests/test_manage_tasks_cron.py::test_invalid_cron_retime_rolls_back_all_edits",
"tests/test_reserved_username_admin_escalation.py::test_rename_into_reserved_username_is_blocked[internal-tool]",
"tests/test_reserved_username_admin_escalation.py::test_rename_into_reserved_username_is_blocked[api]",
"tests/test_reserved_username_admin_escalation.py::test_rename_into_reserved_username_is_blocked[demo]",
"tests/test_reserved_username_admin_escalation.py::test_rename_into_reserved_username_is_blocked[system]",
"tests/test_reserved_username_admin_escalation.py::test_rename_into_reserved_username_is_blocked[__odysseus_local__]",
"tests/test_reserved_username_admin_escalation.py::test_normal_usernames_still_allowed",
"tests/test_review_calendar_invitation.py::test_reschedule_and_cancellation_target_same_event",
"tests/test_review_calendar_invitation.py::test_cancellation_before_invite_does_not_create_event",
"tests/test_review_calendar_invitation.py::test_same_ics_uid_is_scoped_to_owner",
"tests/test_review_calendar_invitation.py::test_attendee_reply_does_not_create_event",
"tests/test_review_calendar_invitation.py::test_overlapping_revisions_do_not_race",
"tests/test_review_calendar_invitation.py::test_same_title_time_does_not_link_different_senders",
"tests/test_review_calendar_invitation.py::test_occurrence_reschedule_excludes_original_without_moving_series",
"tests/test_review_calendar_invitation.py::test_occurrence_cancellation_before_series_is_preserved",
"tests/test_review_calendar_invitation.py::test_series_cancellation_also_cancels_detached_events",
"tests/test_review_document_conversion.py::test_imported_office_document_is_owned_at_first_commit",
"tests/test_review_endpoint_credentials.py::test_credential_resolution_is_exact_and_owner_scoped[alice-https://api.example.test/v1/chat/completions-Bearer alice-secret-task]",
"tests/test_review_endpoint_credentials.py::test_credential_resolution_is_exact_and_owner_scoped[alice-https://api.example.test/v1/chat/completions-Bearer alice-secret-skill]",
"tests/test_review_endpoint_credentials.py::test_credential_resolution_is_exact_and_owner_scoped[bob-https://api.example.test/v1/chat/completions-None-task]",
"tests/test_review_endpoint_credentials.py::test_credential_resolution_is_exact_and_owner_scoped[bob-https://api.example.test/v1/chat/completions-None-skill]",
"tests/test_review_endpoint_credentials.py::test_credential_resolution_is_exact_and_owner_scoped[alice-https://api.example.test.evil.test/v1-None-task]",
"tests/test_review_endpoint_credentials.py::test_credential_resolution_is_exact_and_owner_scoped[alice-https://api.example.test.evil.test/v1-None-skill]",
"tests/test_review_endpoint_credentials.py::test_credential_resolution_is_exact_and_owner_scoped[alice-https://evil.test/https://api.example.test/v1-None-task]",
"tests/test_review_endpoint_credentials.py::test_credential_resolution_is_exact_and_owner_scoped[alice-https://evil.test/https://api.example.test/v1-None-skill]",
"tests/test_setup_admin_user.py::test_create_default_admin_normalizes_env_username",
"tests/test_setup_admin_user.py::test_main_loads_admin_password_from_env_file",
"tests/test_turn_rendering_js.py::test_turn_rendering_browser_suite",
"tests/test_research_endpoint_owner_scope.py::test_endpoint_id_rejects_another_owners_private_endpoint",
"tests/test_research_endpoint_owner_scope.py::test_endpoint_id_returns_callers_own_endpoint",
"tests/test_research_endpoint_owner_scope.py::test_endpoint_id_allows_legacy_null_owner_shared_row",
"tests/test_research_endpoint_owner_scope.py::test_endpoint_id_skips_disabled_even_when_owned",
"tests/test_research_endpoint_owner_scope.py::test_fallback_never_picks_another_owners_endpoint",
"tests/test_research_endpoint_owner_scope.py::test_fallback_returns_none_when_only_others_endpoints",
"tests/test_research_endpoint_owner_scope.py::test_null_owner_is_legacy_single_user_noop",
"tests/test_research_endpoint_owner_scope.py::test_runtime_resolution_uses_provider_auth_for_chatgpt_subscription"
]
@@ -0,0 +1,2 @@
added 4 packages in 560ms
@@ -0,0 +1,256 @@
# Wave 3-S delivery record
Branch: `feature/runtime-containment`. The final production/delivery commit
contains namespace-init verification, this record and validation evidence;
its exact HEAD is in the delivery message. All commits are local. No push,
PR, merge into lab, branch switch,
reset, rebase, merge abort, cleanup, or other Odysseus worktree mutation occurred.
## Reconciliation
| Revision | Exact commit |
| --- | --- |
| Original containment head | `8e101fdcb8e775105bd4297298be580988bc7ad0` |
| Frozen integration lab | `1e3c50d2dd66484dd515c8caff3614e4ee9cea20` |
| Merge base | `d6c3c98c75e03f70c05ebe4058c6fa12e0395f62` |
| Reconciliation checkpoint | `083a573f7eab63d014331e669178cc367c22a2c8` |
The checkpoint has exactly the original containment head and frozen lab as its
two parents. The in-progress merge was recovered, not restarted. Its only
unmerged path was `website/configuration-reference.md`. All three conflict
stages were inspected; regenerating the reference from the merged sources
preserved containment references and newer lab references together.
Automatic merges of `src/agent_tools/subprocess_tools.py`,
`src/tool_execution.py`, and `tests/test_agent_bash_windows.py` preserved the
Windows Bash environment/cwd/capture contract and authority before dispatch.
The checkpoint also corrected two test assumptions: exact result equality after
adding containment metadata, and an approval-test database stub that needed to
be isolated to that test. Reconciliation validation passed 1,224 tests before
the merge was committed.
RequestAuthority, SemanticIntent, ExactOperation, OperationGrant, TurnContract,
approval policy, and trusted/untrusted request boundaries were preserved.
Since reconciliation, `src/agent_runtime/authority.py`, `src/turn_contract.py`,
and `src/tool_approvals.py` have no changes. The edits to tool execution pass the
existing trusted environment into the contained background launcher and report
its refusal; authority evaluation and background authority sealing retain their
original ordering and owner.
## Subsequent commits
| Commit | Change |
| --- | --- |
| `5bb1326183306e8341d3ca1e6e6f31e4bf9cb0b3` | ODY-152: shared native execution, capture, persistence and teardown |
| `765d79cadf3113e973048ff2e04b0c51d64a88b6` | ODY-143: unconditional native Python containment |
| `f48931407a81bac138cd231d95b95ec0b326ad5b` | Correct the Python namespace test's outside-sibling fixture |
| `127f9b0836456cd95ac8fe4bd5a7ee0c238d8f0d` | ODY-145: contained detached Bash supervisor |
| `f63d333a61404656885be9546e5102f46c248b1c` | ODY-147: retire automatic tmux sessions and reap verified legacy sessions |
| `929987dde7920afb90f0590c24474ae3fa2b4e58` | ODY-150: replace pane capture with bounded, explicit output capture |
| `865968c8d5c0ff72c3faeeaa993705064dca33d9` | ODY-141 LAST: functional namespaces, readiness, cancellation and enforcement |
| `a655abf69839f5a83f14bd48675a9fb178a9b028` | Release and report a background supervisor's failed initialization |
| Commit containing this record | Verify namespace-init death, pin the probed binary, make completed release idempotent, and record final validation |
## Item status
| Item | Status and evidence |
| --- | --- |
| ODY-152 | Implemented. Native tools, detached jobs and compatibility callers use shared containment/teardown; transactional stores preserve concurrent job receipts. |
| ODY-143 | Implemented. Every native Python execution takes the shared boundary, independent of source content. Final-expression output and configured imports remain supported. |
| ODY-145 | Implemented. `#!bg` acquires the same required dimensions before supervisor launch; the supervisor receives the command only after durable ownership/job recording. |
| ODY-147 | Implemented. Chat IDs no longer create tmux shells. Legacy cleanup checks launcher, runtime HOME, session generation, server/pane lineage and start tokens. Ambiguous sessions remain unsignalled and reported. |
| ODY-150 | Implemented. Native Bash no longer reads a 2,000-line pane. A 3,002-line result is complete; actual byte/presentation truncation has metadata and a visible notice. |
| ODY-141 | Implemented last. Shipped mode is enforcing. Missing required dimensions or failed namespace initialization refuse execution deterministically. No tool/configuration host-access mode was introduced. |
## Final containment architecture
`agent_spec` fixes the required dimensions from trusted runtime configuration;
tool text cannot weaken them. `acquire` selects capabilities without examining
the command. Installed bubblewrap must pass a functional PID/mount namespace
probe. Launch uses the absolute trusted binary path, so the execution environment
cannot substitute a workspace binary through PATH. `run` checks the declared mechanism's dimensions again, establishes the
namespace, and consumes a private readiness receipt before acknowledging the
trusted wrapper and starting model code. Bind/setup failure cannot produce a
successful containment result.
The shared bubblewrap recipe uses a private root, private PID namespace, private
`/proc` and devices, read-only system/interpreter mounts, private `/tmp`, and
writable workspace mounts. Extras are mounted before the workspace, so a
read-only ancestor cannot hide its writable workspace bind. Active Python
environments under `/home` are bound explicitly rather than assumed visible.
The compatibility namespace builder also uses this shared recipe.
Spawn is shielded until its process handle is recovered. Timeout, initialization
failure, clean exit and cancellation converge on shared teardown. Repeated
cancellation cannot interrupt TERM, bounded wait, KILL and death verification.
Bubblewrap's separate info pipe records the namespace's PID 1 before model
execution starts. Linux held owners and namespace init use pidfds when available.
Release verifies death of both, including init's kernel cleanup of descendants
that used `setsid()` or double-fork/session escape. Outer-owner exit alone cannot
claim whole-tree death. The receipt retains a live/unverifiable init after failed
signals; recovered teardown validates its start identity before signalling it.
Completed release is idempotent and cannot signal a reused PID; a released grant
cannot execute again. The namespace target uses the same escalating teardown
primitive, not a second escalation implementation.
Detached jobs run a trusted supervisor, not model code outside the boundary.
Its child executes through `containment.run`; completion metadata is published
before the exit receipt. Failed log initialization releases an unstarted grant
and still publishes failure metadata when those destinations are available.
An owned live supervisor remains responsible across server restart; killing a
job validates ownership and checks actual teardown before claiming it was killed.
Process ownership compares PID plus start identity. Linux tokens now include
boot identity, preventing a receipt from matching the same start tick after a
reboot. Recovered teardown validates identity and the recorded PGID before
signals, including again before escalation. EPERM means unknown/live, never
verified death. A gone leader with a populated but unowned group is retained as
a failed cleanup rather than signalled. Foreign/unverifiable receipts remain
visible. JSON read/modify/write operations are serialized across processes.
`src/path_confinement.py` remains the centralized canonical path boundary for
in-process tools. It was preserved rather than replaced by a second policy.
## Explicit dimensions
| Dimension | Native contract |
| --- | --- |
| Filesystem | Required. Functional mount namespace and the trusted workspace/mount recipe. No alias-rewrite fallback in shipped enforcement. |
| Process tree | Required. Private PID namespace and parent-death semantics. Process groups and Windows taskkill do **not** advertise this dimension. |
| Wall clock | Required. Startup/readiness, stdin backpressure and child waiting share the execution timeout; teardown then has bounded escalation waits. |
| Network | Inherited by default, explicitly reported, not isolated. Explicit `none` requests add a real network namespace or refuse at initialization. Loopback sidecars remain reachable by default. |
| Memory | Optional existing Linux RLIMIT_AS hook when the requested hard limit can be applied. No generic resource authority was added. |
| Process count | Optional existing RLIMIT_NPROC hook where supported and not root. This is a user-level limit, not a per-grant quota. |
| Output | Bounded bytes per stream, fully drained to avoid pipe deadlock; UTF-8 decoding spans chunks. Truncation is visible and reported. Presentation caps also carry a notice. |
## Production and test inventory
Production changes after the reconciliation checkpoint:
```text
core/atomic_io.py
core/platform_compat.py
src/agent_tools/bg_job_tools.py
src/agent_tools/subprocess_tools.py
src/bg_jobs.py
src/containment.py
src/containment_worker.py
src/process_ownership.py
src/process_reaper.py
src/tool_execution.py
website/configuration-reference.md
```
Tests changed or added after reconciliation:
```text
tests/containment_helpers.py
tests/test_agent_bash_tmux_env.py
tests/test_agent_bash_windows.py
tests/test_agent_tmux_retirement.py
tests/test_background_containment.py
tests/test_bg_job_tools.py
tests/test_containment_contract.py
tests/test_containment_enforcement.py
tests/test_containment_process_tree.py
tests/test_execution_filesystem_boundary.py
tests/test_native_execution_containment.py
tests/test_orphan_reaping.py
tests/test_process_ownership.py
tests/test_workspace_artifact_tool_floor.py
tests/test_workspace_confine.py
```
The reconciliation commit additionally imports the frozen lab's production/test
changes, including its authority and PTY changes; these are distinct from the
Wave 3-S edits above. `git diff --name-only
8e101fdcb8e775105bd4297298be580988bc7ad0
083a573f7eab63d014331e669178cc367c22a2c8` gives that exact inventory.
The only additional test edits made while reconciling were the Windows result
assertion and `tests/test_tool_approvals.py`'s isolated stub.
## Validation
| Check | Result |
| --- | --- |
| Reconciliation overlap | 1,224 passed |
| ODY-152 focused | 193 passed, 2 skipped |
| ODY-143 focused, corrected sibling fixture | 186 passed |
| ODY-145 focused | 205 passed, 1 skipped |
| ODY-147 focused, including private real tmux server | 71 passed |
| ODY-150 focused | 64 passed |
| ODY-141 focused | 306 passed, 1 skipped |
| Final containment/path/background/authority/PTY/Windows overlap | 657 passed, 2 skipped |
| Supervisor follow-up plus containment/authority/bridge/PTY/Windows tests | 426 passed, 1 skipped |
| Namespace-init ownership/teardown follow-up | 626 passed, 2 skipped |
| Final delivery containment/background/authority/turn-contract/PTY/Windows overlap | 1,608 passed, 2 skipped |
| Full Python suite, single completed run | 11,727 passed; 118 failed; 8 errors; 68 skipped; 2 xfailed; 6 subtests passed; 182 warnings |
| Exact failed/error nodes after environment repair | All 126 passed; 4 deprecation warnings |
| `compileall app.py core routes src tests` | Passed, including final production revision |
| JS/MJS syntax | Not applicable: no JS/MJS changed from the original containment head; affected browser tests were exercised by targeted recovery. |
| Whitespace, conflict markers and unmerged paths | Checked at reconciliation and delivery; no remaining conflict markers or unmerged paths. Captured log trailing whitespace normalized for the final diff check. |
Counts overlap and must not be summed. The initial system-Python full attempt
stopped at collection with 16 missing-dependency errors and ran no tests. It is
preserved as `validation/wave-3-s-full-collection.txt`. An isolated ignored
`.venv` with system packages was created in this worktree. Missing test/runtime
dependencies from `requirements.txt` were installed there; `npm ci` used the
existing lockfile in this worktree. No package manifest or lockfile was changed.
The completed full run is preserved as `validation/wave-3-s-full.txt`; it was
**not green**. Its failures included missing bcrypt/calendar/cron/PDF-rendering
dependencies, import mocks following failed ORM pre-import, and absent Node
test packages. Repairing those dependencies and executing exactly its 126
failed/error node IDs produced 126 passes. The full suite was not repeated, in
accordance with the one-run instruction. This proves targeted recovery, not a
new all-green full run in the repaired environment. The final supervisor and
namespace-init fixes were validated by focused follow-ups after that full run.
Focused commands and summaries are retained under `validation/wave-3-s-*`.
Real tests cover private PID namespaces, a hidden host sibling, sidecar
connectivity, explicit network isolation or deterministic refusal, escaped
session death on timeout and clean parent exit, startup failure, stdin closure,
cancellation during spawn, repeated cancellation during escalation, denied
namespace-init signals after owner death, recovered/reused init identities,
idempotent release, the old PATH substitution and its pinned-path fix, concurrent
job recording, server restart ownership, verified legacy tmux cleanup and
output above 2,000 lines. Existing request-authority and #44/#45 regression
tests passed in the overlap runs.
## Limits, concerns and independent review
No unresolved P0/P1 was observed in the tested Wave 3-S native execution paths.
The implementation and focused Wave 3-S validation are complete. The original
full-run failure result remains part of the delivery evidence.
Platform support is deliberately truthful. Native required containment refuses
on macOS/Windows without a suitable mechanism and on Docker/Linux where
bubblewrap is missing or namespace creation is blocked. Windows Bash contract
tests used platform simulation; no real Windows/macOS machine was validated.
Installing bubblewrap alone does not establish Docker namespace support.
Network egress/LAN access remains inherited by default. Existing externally
owned Wave 2 bridges are not attested as locally contained by this work.
P2 follow-up concerns: independently validate the entire suite in the repaired
environment/CI; adversarially review identity/token and PGID races in recovered
or legacy processes that lack a retained kernel handle; inspect migration of
older identity receipts and ambiguous legacy sessions. Token granularity remains
finite (Linux clock ticks, macOS seconds); boot identity removes cross-boot
matches, not every inspection-to-signal race. Failed/unverifiable receipts are
kept visible rather than expired as if teardown succeeded. Remote bridge
containment claims require an independent assessment of the remote owner.
Maestrum was used for bounded read review. An earlier audit identified the
functional namespace, session escape and cancellation gaps that were verified
and addressed. Its suggestion to signal a group after losing leader identity
was rejected; retaining uncertain receipts is deliberate. Its store-lock claim
did not account for the current transactional writer decorators. The final
review of `865968c8d5c0ff72c3faeeaa993705064dca33d9` failed before any worker ran
because Maestrum placement selected an unrecognized model. The current
orchestrate-work skill assigns placement/retries to Maestrum and directs failed
work to targeted local inspection; no native worker fallback was used. Final
independent adversarial review remains outstanding, especially for detached
supervisor cancellation and recovered ownership under hostile timing.
Work stops at Wave 3-S. No subsequent authority, provenance/egress, browser,
generic lifecycle or decomposition wave was started.
+2 -6
View File
@@ -13,6 +13,7 @@ from pydantic import BaseModel
from core.database import Document, DocumentVersion
from core.database import Session as DbSession
from src.auth_helpers import _auth_disabled
from src.path_confinement import is_inside
from src.upload_handler import UploadHandler
logger = logging.getLogger(__name__)
@@ -136,12 +137,7 @@ _PDF_RENDER_SCALE = 2.0
def _upload_path_inside(upload_dir: str, path: str) -> bool:
base = os.path.realpath(upload_dir)
p = os.path.realpath(path)
try:
return os.path.commonpath([base, p]) == base
except Exception:
return False
return is_inside(upload_dir, path)
def _resolve_user_upload_path(
+7 -3
View File
@@ -39,6 +39,7 @@ from email.mime.multipart import MIMEMultipart
from fastapi import APIRouter, Query, UploadFile, File, BackgroundTasks, HTTPException, Depends, Request
from fastapi.responses import FileResponse, StreamingResponse
from src.constants import DATA_DIR
from src.path_confinement import confine
from src.llm_core import llm_call_async
from src.upload_limits import read_upload_limited, EMAIL_COMPOSE_UPLOAD_MAX_BYTES
@@ -4199,9 +4200,12 @@ def setup_email_routes():
return {"error": f"Attachment index {index} not found"}
from pathlib import Path as _Path
target_root = os.path.abspath(str(target_dir))
filepath_str = os.path.abspath(str(filepath))
if os.path.commonpath([target_root, filepath_str]) != target_root:
# realpath, not abspath: abspath only folds `..`, so a symlink
# written into the extraction directory would have passed this
# check and then been read through.
try:
filepath_str = confine(str(target_dir), str(filepath))
except (ValueError, OSError):
logger.warning("Rejected attachment path outside extraction dir: %s", filepath)
return {"error": "Invalid attachment path"}
filepath = _Path(filepath_str)
+3 -5
View File
@@ -21,6 +21,7 @@ from src.upload_limits import (
GALLERY_TRANSFORM_UPLOAD_MAX_BYTES,
)
from src.constants import GENERATED_IMAGES_DIR
from src.path_confinement import confine
from src.optional_deps import patch_realesrgan_torchvision_compat
from routes.gallery.gallery_helpers import (
@@ -235,12 +236,9 @@ def _gallery_image_path(filename: str) -> Path:
raise HTTPException(400, "Unsafe gallery filename")
safe_name = _sanitize_gallery_filename(filename)
original = str(filename or "")
root = GALLERY_IMAGE_DIR.resolve()
path = (GALLERY_IMAGE_DIR / safe_name).resolve()
try:
if os.path.commonpath([str(root), str(path)]) != str(root):
raise ValueError
except Exception:
path = Path(confine(GALLERY_IMAGE_DIR, safe_name, allow_root=False))
except (ValueError, OSError):
raise HTTPException(400, "Unsafe gallery filename")
if safe_name != original:
raise HTTPException(400, "Unsafe gallery filename")
+14 -28
View File
@@ -13,6 +13,7 @@ from core.constants import BASE_DIR, PERSONAL_DIR, PERSONAL_UPLOADS_DIR
from src.rag_singleton import get_rag_manager
from src.auth_helpers import require_privilege, require_user
from core.middleware import require_admin
from src.path_confinement import confine
from src.upload_handler import secure_filename
from src.upload_limits import PERSONAL_UPLOAD_MAX_BYTES
@@ -23,10 +24,7 @@ logger = logging.getLogger(__name__)
def _personal_upload_dir_for_owner(owner: str | None, *, create: bool = True) -> str:
"""Return the per-owner upload directory used for direct RAG uploads."""
owner_segment = secure_filename((owner or "local").strip())[:80] or "local"
upload_dir = os.path.abspath(os.path.join(UPLOADS_DIR, owner_segment))
base_abs = os.path.abspath(UPLOADS_DIR)
if os.path.commonpath([upload_dir, base_abs]) != base_abs:
raise ValueError("Unsafe upload owner path")
upload_dir = confine(UPLOADS_DIR, owner_segment, allow_root=False)
if create:
os.makedirs(upload_dir, exist_ok=True)
return upload_dir
@@ -41,10 +39,7 @@ def _unique_personal_upload_path(upload_dir: str, original_name: str | None) ->
stem, ext = os.path.splitext(safe_name)
stem = (stem or "upload")[:80]
filename = f"{stem}-{uuid.uuid4().hex[:10]}{ext.lower()}"
file_path = os.path.abspath(os.path.join(upload_dir, filename))
upload_abs = os.path.abspath(upload_dir)
if os.path.commonpath([file_path, upload_abs]) != upload_abs:
raise ValueError("Unsafe upload filename")
file_path = confine(upload_dir, filename, allow_root=False)
return file_path, filename, safe_name
@@ -167,19 +162,10 @@ def setup_personal_routes(personal_docs_manager, rag_manager, rag_available):
if not directory:
raise HTTPException(400, "Directory path is required")
# realpath (not abspath) so a symlink inside PERSONAL_DIR that points
# outside it is resolved before the commonpath confinement check below;
# abspath only normalises `..` and would let such a symlink escape.
base_abs = os.path.realpath(PERSONAL_DIR)
candidate = directory if os.path.isabs(directory) else os.path.join(base_abs, directory)
resolved = os.path.realpath(candidate)
try:
in_base = os.path.commonpath([resolved, base_abs]) == base_abs
except ValueError:
in_base = False
if not in_base:
return confine(PERSONAL_DIR, directory)
except (ValueError, OSError):
raise HTTPException(403, "Directory must be inside personal documents")
return resolved
@router.get("")
def api_personal_list(owner: str = Depends(require_user), _admin: None = Depends(require_admin)):
@@ -425,17 +411,17 @@ def setup_personal_routes(personal_docs_manager, rag_manager, rag_available):
# Scope to the per-owner subdir, not the shared uploads root, so one
# admin can't delete another user's personal files by path.
deleted_from_disk = False
# allow_root=False: the per-owner upload directory itself is
# never a deletion target, only files under it.
try:
abs_target = os.path.realpath(filepath)
base_abs = os.path.realpath(_personal_upload_dir_for_owner(owner, create=False))
in_uploads = (
abs_target == base_abs
or os.path.commonpath([abs_target, base_abs]) == base_abs
abs_target = confine(
_personal_upload_dir_for_owner(owner, create=False),
filepath,
allow_root=False,
)
except ValueError:
# commonpath raises on mixed drives / non-comparable paths
in_uploads = False
if in_uploads and abs_target != base_abs:
except (ValueError, OSError):
abs_target = ""
if abs_target:
try:
os.remove(abs_target)
deleted_from_disk = True
+2 -4
View File
@@ -24,6 +24,7 @@ from core.database import (
from src.auth_helpers import effective_user
from src.attachment_refs import attachment_refs_from_metadata
from src.constants import GENERATED_IMAGES_DIR
from src.path_confinement import is_inside
from src.upload_handler import (
UploadCleanupSafetyError,
count_recent_uploads,
@@ -152,10 +153,7 @@ def setup_upload_routes(upload_handler):
return os.path.realpath(getattr(upload_handler, "upload_dir", UPLOAD_DIR))
def _path_inside_upload_dir(path: str) -> bool:
try:
return os.path.commonpath([_upload_root(), os.path.realpath(path)]) == _upload_root()
except Exception:
return False
return is_inside(_upload_root(), path)
def _resolve_upload_path(file_id: str) -> str:
from src.constants import UPLOAD_DIR
+10 -3
View File
@@ -25,6 +25,8 @@ import os
import time
from typing import Dict, Iterable, List, Optional
from src.path_confinement import confine
from .skill_format import Skill, slugify
logger = logging.getLogger(__name__)
@@ -644,9 +646,14 @@ class SkillsManager:
or (sk.source == "builtin" and not (sk.owner or ""))
):
continue
base = os.path.realpath(os.path.dirname(path))
target = os.path.realpath(os.path.join(base, ref_path))
if os.path.commonpath([base, target]) != base or target == os.path.dirname(path):
# allow_root=False refuses the skill directory itself. The old
# guard compared a realpath-ed target against a raw dirname, so on
# a host where the skills tree is reached through a symlink (macOS
# /tmp -> /private/tmp) the two sides never matched and the guard
# could not fire.
try:
target = confine(os.path.dirname(path), ref_path, allow_root=False)
except (ValueError, OSError):
return None
if not os.path.isfile(target):
return None
+3
View File
@@ -87,6 +87,9 @@ class ManageBgJobsTool:
if rec.get("status") != "running":
return {"output": f"Job `{job_id}` already {_status_label(rec)}; nothing to kill.", "exit_code": 0}
killed = bg_jobs.kill(job_id)
if not killed or not killed.get("killed"):
return {"error": f"Could not verify termination of background job `{job_id}`.",
"exit_code": 1, "teardown": (killed or {}).get("teardown")}
return {"output": f"Killed background job `{job_id}` ({(killed or {}).get('command', '').splitlines()[0][:80]}).", "exit_code": 0}
out = rec.get("output") or "(no output yet)"
+2 -7
View File
@@ -9,6 +9,7 @@ import tempfile
from typing import Optional, Dict, Any, Tuple, List
from src.constants import MAX_READ_CHARS, MAX_DIFF_LINES, MAX_OUTPUT_CHARS
from src.path_confinement import is_inside
_CODENAV_SKIP_DIRS = frozenset({
".git", ".hg", ".svn", "node_modules", "venv", ".venv", "__pycache__",
@@ -675,13 +676,7 @@ class GlobTool:
# confinement that _resolve_search_root applies to the root.
# An escaping literal falls through to the walk, which only ever
# yields paths under base.
nbase = os.path.normcase(rbase)
try:
inside = cand == rbase or os.path.commonpath(
[os.path.normcase(cand), nbase]
) == nbase
except ValueError:
inside = False
inside = is_inside(rbase, cand)
# A literal that names a deny-listed sensitive file (.env,
# .ssh/id_rsa, …) falls through to the walk, which skips it —
# otherwise glob would surface secret paths that read_file /
+324 -476
View File
@@ -1,5 +1,6 @@
import asyncio
import ast
import logging
import os
import re
import shlex
@@ -8,15 +9,16 @@ import shutil
import subprocess
import sys
import time
import collections
import json
from typing import Optional, Callable, Awaitable, Tuple, Dict
from typing import Optional
from urllib.parse import urlparse
import httpx
from src.constants import MAX_OUTPUT_CHARS
from src.agent_runtime.journal import mark_operation_started
from src import containment
from src.constants import AGENT_ISOLATED_TMP_DIRNAME, MAX_OUTPUT_CHARS, WORKSPACE_MOUNT
logger = logging.getLogger(__name__)
# Agent shell calls must fail fast enough for the loop to recover and choose a
# better tool. A one-hour default can pin an entire benchmark worker on an
@@ -25,9 +27,6 @@ from src.agent_runtime.journal import mark_operation_started
DEFAULT_BASH_TIMEOUT = 120
DEFAULT_PYTHON_TIMEOUT = 60 * 60
PROGRESS_INTERVAL_S = 2.0
PROGRESS_TAIL_LINES = 12
TMUX_CAPTURE_LINES = 2000
_HOST_SHELL_BRIDGE_HOSTS = {"127.0.0.1", "localhost", "::1", "host.docker.internal"}
IS_WINDOWS = sys.platform.startswith("win")
_HOST_SHELL_CANCEL_TASKS: set[asyncio.Task] = set()
@@ -217,11 +216,6 @@ def is_host_shell_bridge_url_allowed(url: str) -> bool:
return True
def _tmux_session_name(session_id: Optional[str]) -> str:
raw = re.sub(r"[^A-Za-z0-9_.-]+", "-", str(session_id or "default")).strip("-")
return f"ody-agent-{raw[:80] or 'default'}"
def _replace_workspace_alias(content: str, cwd: str) -> str:
"""Map virtual /workspace paths without corrupting absolute host paths."""
return re.sub(
@@ -231,11 +225,221 @@ def _replace_workspace_alias(content: str, cwd: str) -> str:
)
#: Roots the namespace argv mounts itself. A host path under one of these is
#: already reachable inside the namespace, so it needs no bind and must not get
#: a ``--dir`` chain: mkdir inside a read-only bind fails and takes the whole
#: namespace with it.
_NAMESPACE_MOUNTED_ROOTS = ("/usr", "/home", "/mnt")
#: Destinations a bind must never overlay. Replacing the private root, the
#: private /tmp or the workspace mount with a host directory undoes the
#: namespace from inside the argv that builds it.
_NAMESPACE_RESERVED_DESTS = frozenset({
"/", "/tmp", "/var", "/opt", "/etc", WORKSPACE_MOUNT,
"/root", "/run", "/proc", "/dev", "/sys", *_NAMESPACE_MOUNTED_ROOTS,
})
def _namespace_visible_without_bind(path: str) -> bool:
"""True when ``path`` is already reachable through a root the argv mounts."""
return any(
path == root or path.startswith(root + os.sep)
for root in _NAMESPACE_MOUNTED_ROOTS
)
def _namespace_dir_chain(path: str) -> list[str]:
"""``--dir`` args for every ancestor of ``path`` the argv has to create.
bwrap mounts into a tmpfs root, so a bind destination's parents have to
exist before the bind. Returns nothing when the parents already exist by
virtue of a mount the argv made — creating a directory inside a read-only
bind is an error, not a no-op.
"""
if _namespace_visible_without_bind(path):
return []
parents: list[str] = []
parent = os.path.dirname(path)
while parent not in ("/", "", "/tmp", "/etc", WORKSPACE_MOUNT, *_NAMESPACE_MOUNTED_ROOTS):
parents.append(parent)
parent = os.path.dirname(parent)
args: list[str] = []
for directory in reversed(parents):
args.extend(("--dir", directory))
return args
def _isolated_tmp_dir(cwd: str) -> str:
"""The workspace-local stand-in for the host ``/tmp``.
Creation is best-effort: the source tree is read-only in Docker and a
workspace can be mounted read-only, and a command that mentions ``/tmp/``
must not die with an OSError traceback because a scratch directory could
not be made. The rewrite still points at the workspace, so a command that
really needs to write there fails on its own terms, inside the boundary,
with its own error message.
"""
path = os.path.join(cwd, AGENT_ISOLATED_TMP_DIRNAME)
try:
os.makedirs(path, exist_ok=True)
except OSError:
pass
return path
def _execution_boundary(
cwd: str, *, wall_clock_s: int = DEFAULT_BASH_TIMEOUT,
) -> "containment.ContainmentProbe":
"""What this host can actually enforce for an agent command in ``cwd``.
The single place the shell and Python tools ask. Both used to decide for
themselves, by testing whether a namespace wrapper came back non-None, and
both then fell through to a regex if it had not — so "was that command
confined" had no answer and no field in the result. Routing the question
through :mod:`src.containment` means one mechanism table, one answer, and a
``containment`` block in the tool result either way.
``network`` is left inherited on purpose: ``--unshare-net`` was measured to
cut the loopback sidecars this product depends on (ChromaDB on 8100), and
the Dockerfile installs ``nmap``/``iproute2``/``dnsutils`` because
Docker-hosted agents are expected to do LAN work. It is a reported
dimension here, not an enforced one.
"""
try:
return containment.probe(
containment.agent_spec(
workspace=cwd,
env={},
wall_clock_s=wall_clock_s,
max_output_bytes=MAX_OUTPUT_CHARS,
)
)
except ValueError as exc:
# A workspace that is not a usable directory is a caller bug to
# containment, which raises rather than reporting. Here it must not
# take out the tool, and it is still a containment failure: nothing can
# be confined to a directory that is not there. Fail closed. The reason
# goes in the message rather than a traceback -- this is a known shape,
# not an unexpected exception.
logger.warning(
"execution boundary: cannot probe containment for workspace %r (%s); "
"treating every required dimension as unenforced",
cwd, exc,
)
return containment.ContainmentProbe(
mechanism="none",
enforced=frozenset(),
degraded=(),
unenforced_required=tuple(sorted(containment.DEFAULT_REQUIRED)),
mode=containment.CONTAINMENT_MODE,
)
#: What the fallback actually is, named so it cannot be mistaken for a
#: mechanism. ``_replace_workspace_alias`` rewrites the literal token
#: ``/workspace`` to the real path in the command string; a command that never
#: mentions ``/workspace`` is untouched by it and runs on the host unrestricted.
ALIAS_REWRITE_MECHANISM = "workspace_alias_rewrite"
#: Guards the one-per-process fallback warning below. Module state, because the
#: fact it reports is a property of the host rather than of a command.
_ALIAS_FALLBACK_LOGGED = False
def _filesystem_boundary_block(mechanism: str, mode: str, *, confined: bool) -> dict:
"""The ``containment`` block for a spawn these tools still build themselves.
Reports the **filesystem dimension only**, deliberately. The probe knows
this host could also give a process group and a real wall clock, but
Compatibility namespace previews assemble their own ``create_subprocess_*``
call and pass neither ``start_new_session`` nor a group-wide kill, so
listing those dimensions here would be the false claim
:mod:`src.containment` calls worse than an honest absence. They arrive when
this spawn path moves onto :func:`containment.run`, not before.
"""
return {
"mechanism": mechanism,
"mode": mode,
"enforced": [containment.FILESYSTEM] if confined else [],
"unenforced_required": [] if confined else [containment.FILESYSTEM],
"contained": confined,
"executed": True,
# Names the scope of the claim, so "process_tree is absent from
# enforced" reads as "not reported here" rather than "not enforced".
"reported_dimensions": [containment.FILESYSTEM],
}
def _contained_command(
content: str,
cwd: str,
*,
chdir: str = WORKSPACE_MOUNT,
interpreter_prefix: str | None = None,
) -> tuple[str, dict, bool]:
"""Resolve ``content`` into the strongest form this host can run.
Returns ``(command, containment_block, confined)``. The caller spawns
``command``, copies ``containment_block`` into its result verbatim, and
refuses instead when ``confined`` is false under enforcing mode.
This replaces ``namespaced or _replace_workspace_alias(...)``, the line this
ticket exists to delete. The two branches it chose between are not
comparable — one is a mount namespace, the other is a regex — and choosing
the second silently means an uncontained host execution reads in the
transcript exactly like a contained one. The fallback still happens under
an explicit report-only diagnostic mode; the difference is that it is now
recorded in the result.
:raises containment.ContainmentUnavailable: filesystem containment could
not be established and the mode is enforcing. The command is not run.
"""
probe = _execution_boundary(cwd)
wrapped = _wrap_workspace_namespace(
content, cwd, chdir=chdir, interpreter_prefix=interpreter_prefix,
)
# The probe's filesystem answer and the wrapper's None/not-None answer rest
# on the same functional namespace probe, so they agree
# by construction. `wrapped` is still what decides, because it is what
# actually runs: a probe that said yes to a wrapper that declined would be
# the same false claim in the other direction.
if wrapped is not None:
return wrapped, _filesystem_boundary_block(
probe.mechanism, probe.mode, confined=True,
), True
if probe.mode == containment.MODE_ENFORCING:
raise containment.ContainmentUnavailable(
frozenset({containment.FILESYSTEM}), ALIAS_REWRITE_MECHANISM,
)
# Once per process, not once per command. The host's ability to establish a
# namespace does not change between calls, so a per-call warning would
# drown the log on every macOS install while adding nothing — and the
# per-call fact is already in the result block, which is where a reader
# looking at one command will look.
global _ALIAS_FALLBACK_LOGGED
if not _ALIAS_FALLBACK_LOGGED:
_ALIAS_FALLBACK_LOGGED = True
logger.warning(
"execution boundary: no filesystem containment is available on this "
"host (mechanism %r); agent commands fall back to the %s, which is "
"a path rewrite and not a boundary. Reported per command in the "
"result's containment block.",
probe.mechanism, ALIAS_REWRITE_MECHANISM,
)
return (
_replace_workspace_alias(content, cwd),
_filesystem_boundary_block(
ALIAS_REWRITE_MECHANISM, probe.mode, confined=False,
),
False,
)
def _wrap_workspace_namespace(
content: str,
cwd: str,
*,
chdir: str = "/workspace",
chdir: str = WORKSPACE_MOUNT,
interpreter_prefix: str | None = None,
) -> str | None:
"""Run a shell command with the active workspace mounted at /workspace.
@@ -245,22 +449,9 @@ def _wrap_workspace_namespace(
bubblewrap namespace preserves that public contract for each concurrent
agent without creating a process-global /workspace symlink.
"""
if IS_WINDOWS or not shutil.which("bwrap"):
if IS_WINDOWS or not containment._bwrap_available():
return None
args = [
"bwrap", "--die-with-parent", "--new-session", "--tmpfs", "/",
"--dir", "/usr", "--ro-bind", "/usr", "/usr",
"--symlink", "usr/bin", "/bin",
"--symlink", "usr/lib", "/lib",
"--symlink", "usr/lib64", "/lib64",
"--symlink", "usr/bin", "/sbin",
"--dir", "/etc", "--ro-bind", "/etc", "/etc",
"--dir", "/home", "--bind", "/home", "/home",
"--dir", "/mnt", "--bind", "/mnt", "/mnt",
"--dir", "/tmp", "--tmpfs", "/tmp",
"--dev-bind", "/dev", "/dev", "--proc", "/proc",
"--dir", "/workspace", "--bind", cwd, "/workspace",
]
readonly = [path for path in ("/home", "/mnt") if os.path.isdir(path)]
# setup-python installs interpreters under /opt, and local CI virtualenvs
# can live under /tmp. Those paths are hidden by the private root/tmpfs.
# Expose only the active interpreter environment, read-only, so Python
@@ -268,15 +459,7 @@ def _wrap_workspace_namespace(
if interpreter_prefix:
prefix = os.path.abspath(interpreter_prefix)
resolved_prefix = os.path.realpath(prefix)
mounted_roots = ("/usr", "/home", "/mnt")
reserved_roots = {
"/", "/tmp", "/var", "/opt", "/etc", "/workspace",
"/root", "/run", "/proc", "/dev", "/sys", *mounted_roots,
}
already_visible = any(
prefix == root or prefix.startswith(root + os.sep)
for root in mounted_roots
)
already_visible = _namespace_visible_without_bind(prefix)
# A prefix is trusted only when it names a specific interpreter tree.
# In particular, never overlay the private root, tmpfs, or workspace
# with a broad host directory. Reject symlinked prefixes too: bwrap
@@ -293,304 +476,91 @@ def _wrap_workspace_namespace(
if (
not already_visible
and prefix == resolved_prefix
and prefix not in reserved_roots
and prefix not in _NAMESPACE_RESERVED_DESTS
and len(prefix.split(os.sep)) >= 3
and os.path.isdir(prefix)
and has_environment_layout
):
parents = []
parent = os.path.dirname(prefix)
while parent not in ("/", "/tmp", "/etc", "/workspace", *mounted_roots):
parents.append(parent)
parent = os.path.dirname(parent)
for directory in reversed(parents):
args.extend(("--dir", directory))
args.extend(("--ro-bind", prefix, prefix))
args.extend(("--chdir", chdir, "/bin/bash", "-lc", content))
readonly.append(prefix)
spec = containment.ContainmentSpec(
workspace=cwd, env={}, wall_clock_s=DEFAULT_BASH_TIMEOUT,
readonly_extra=tuple(readonly),
)
args = containment._bwrap_prefix(spec)
args[-1] = chdir
args.extend(("/bin/bash", "-lc", content))
return shlex.join(args)
async def _run_exec(*args: str, timeout: float = 10) -> Tuple[str, str, int]:
proc = await asyncio.create_subprocess_exec(
*args,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
def _owned_spec(cwd: str, env: Optional[dict], timeout: int, readonly_extra: tuple = ()) -> containment.ContainmentSpec:
"""Server-defined boundary shared by the native execution tools."""
readonly = []
for prefix in (sys.prefix, sys.base_prefix):
prefix = os.path.realpath(prefix)
visible = any(prefix == root or prefix.startswith(root + os.sep) for root in ("/usr", "/etc"))
if not visible and prefix not in _NAMESPACE_RESERVED_DESTS:
readonly.append(prefix)
return containment.agent_spec(
cwd, dict(os.environ if env is None else env), timeout,
readonly_extra=tuple(dict.fromkeys([*readonly, *readonly_extra])),
)
async def _run_owned_command(command, ctx: dict, *, tool: str, timeout: int, argv: bool = False,
readonly_extra: tuple = ()) -> dict:
from src.tool_execution import agent_cwd, _truncate
grant = None
try:
out_b, err_b = await asyncio.wait_for(proc.communicate(), timeout=timeout)
except asyncio.TimeoutError:
try:
proc.kill()
except Exception:
pass
return "", "timeout", 124
return (
out_b.decode("utf-8", errors="replace"),
err_b.decode("utf-8", errors="replace"),
proc.returncode or 0,
)
async def _tmux_has_session(name: str) -> bool:
_, _, rc = await _run_exec("tmux", "has-session", "-t", name, timeout=3)
return rc == 0
async def _tmux_capture(name: str) -> str:
out, _, _ = await _run_exec(
"tmux", "capture-pane", "-p", "-J", "-S", f"-{TMUX_CAPTURE_LINES}", "-t", name,
timeout=5,
)
return out
async def _tmux_send_line(name: str, line: str) -> None:
if line:
await _run_exec("tmux", "send-keys", "-t", name, "-l", line, timeout=5)
await _run_exec("tmux", "send-keys", "-t", name, "C-m", timeout=5)
async def _ensure_tmux_session(name: str, cwd: str, env: Optional[dict]) -> None:
# tmux creates child panes from the long-lived server environment, not
# necessarily from the app process that issued ``new-session``. On hosts
# where tmux predates the Odysseus virtualenv this silently resolves
# ``python`` to the system interpreter, losing plotting/PDF dependencies
# and prompting futile pip-install loops. Reassert the small execution
# environment on both new and reused panes.
forwarded_env = {
key: str(env[key])
for key in ("PATH", "VIRTUAL_ENV", "HOME", "TMPDIR")
if env and env.get(key)
}
if await _tmux_has_session(name):
if forwarded_env:
exports = " ".join(
f"{key}={shlex.quote(value)}" for key, value in forwarded_env.items()
)
await _tmux_send_line(name, f"export {exports}")
await _run_exec("tmux", "send-keys", "-t", name, "stty -echo", "C-m", timeout=5)
return
env_args = [f"{key}={value}" for key, value in forwarded_env.items()]
await _run_exec(
"tmux", "new-session", "-d", "-s", name, "-c", cwd,
"env",
*env_args,
f"TERM={env.get('TERM', 'xterm-256color') if env else 'xterm-256color'}",
f"COLUMNS={env.get('COLUMNS', '120') if env else '120'}",
f"LINES={env.get('LINES', '40') if env else '40'}",
"/bin/bash",
"--noprofile",
"--norc",
timeout=10,
)
if not await _tmux_has_session(name):
raise RuntimeError(f"failed to create tmux session {name}")
await _run_exec("tmux", "send-keys", "-t", name, "stty -echo", "C-m", timeout=5)
def _output_after_marker(capture: str, start_marker: str, end_marker: str) -> Tuple[str, bool]:
lines = capture.splitlines()
start_idx = -1
for idx, line in enumerate(lines):
if line.strip() == start_marker:
start_idx = idx
if start_idx < 0:
return capture, False
end_idx = -1
for idx in range(start_idx + 1, len(lines)):
if lines[idx].strip().startswith(end_marker):
end_idx = idx
if end_idx < 0:
return "\n".join(lines[start_idx + 1:]), False
return "\n".join(lines[start_idx + 1:end_idx]), True
def _extract_marker_rc(capture: str, end_marker: str) -> int:
for line in reversed(capture.splitlines()):
stripped = line.strip()
if stripped.startswith(end_marker):
suffix = stripped[len(end_marker):].strip()
if suffix.isdigit():
return int(suffix)
return 0
async def _run_tmux_bash(
content: str,
*,
session_id: str,
cwd: str,
env: Optional[dict],
timeout: float,
progress_cb: Optional[Callable[[Dict], Awaitable[None]]] = None,
) -> Tuple[str, str, Optional[int], bool]:
name = _tmux_session_name(session_id)
await _ensure_tmux_session(name, cwd, env)
stamp = f"{int(time.time() * 1000)}-{abs(hash(content)) % 1000000}"
start_marker = f"__ODYSSEUS_CMD_START_{stamp}__"
end_prefix = f"__ODYSSEUS_CMD_END_{stamp}__:"
# Execute each tool call in a non-interactive child shell. The tmux pane
# is deliberately persistent, but handing its terminal stdin to commands
# lets programs such as ffmpeg block forever on overwrite prompts. EOF is
# the deterministic behavior expected from an agent tool invocation.
child_command = f"/bin/bash -lc {shlex.quote(content)} </dev/null"
wrapped = (
f"printf '\\n{start_marker}\\n'\n"
f"{child_command}\n"
f"__ody_rc=$?\n"
f"printf '\\n{end_prefix}%s\\n' \"$__ody_rc\"\n"
)
for line in wrapped.splitlines():
await _tmux_send_line(name, line)
started = time.time()
last_tail = ""
while True:
capture = await _tmux_capture(name)
body, done = _output_after_marker(capture, start_marker, end_prefix)
tail = "\n".join(body.splitlines()[-PROGRESS_TAIL_LINES:])
if progress_cb and tail != last_tail:
last_tail = tail
try:
await progress_cb({
"elapsed_s": round(time.time() - started, 1),
"tail": tail,
"tmux_session": name,
})
except Exception:
pass
if done:
rc = _extract_marker_rc(capture, end_prefix)
cleaned = _clean_tmux_command_output(body, wrapped)
return cleaned, "", rc, False
if time.time() - started > timeout:
try:
await _run_exec("tmux", "send-keys", "-t", name, "C-c", timeout=3)
except Exception:
pass
# Ctrl-C targets the pane's foreground process group, but a child
# can outlive its wrapper shell and become an orphan. Destroy this
# task-scoped session as the timeout boundary; the next tool call
# recreates it through _ensure_tmux_session.
try:
await _run_exec("tmux", "kill-session", "-t", name, timeout=3)
except Exception:
pass
cleaned = _clean_tmux_command_output(body, wrapped)
return cleaned, "", 124, True
await asyncio.sleep(0.5)
def _clean_tmux_command_output(text: str, wrapped_command: str) -> str:
lines = text.splitlines()
wrapped_lines = {ln.rstrip() for ln in wrapped_command.splitlines() if ln.strip()}
cleaned = []
for line in lines:
raw = line.rstrip()
stripped = raw.strip()
if not stripped:
cleaned.append(raw)
continue
if stripped in wrapped_lines:
continue
if stripped.startswith("__ody_rc=") or stripped.startswith("printf "):
continue
if re.fullmatch(r"(?:bash|sh)-[\d.]+\$ ?", stripped):
continue
if re.fullmatch(r"[\w.@:/~+-]+[#$] ?", stripped):
continue
cleaned.append(raw)
return "\n".join(cleaned).strip()
async def _run_subprocess_streaming(
proc: asyncio.subprocess.Process,
*,
timeout: float,
progress_cb: Optional[Callable[[Dict], Awaitable[None]]] = None,
) -> Tuple[str, str, Optional[int], bool]:
started = time.time()
stdout_full: list[str] = []
stderr_full: list[str] = []
tail = collections.deque(maxlen=PROGRESS_TAIL_LINES)
async def _reader(stream, full_buf, label: str):
if stream is None:
return
while True:
line = await stream.readline()
if not line:
break
decoded = line.decode("utf-8", errors="replace").rstrip("\n")
full_buf.append(decoded)
if label == "err":
tail.append(f"! {decoded}")
grant = containment.acquire(
_owned_spec(agent_cwd(), ctx.get("subproc_env"), timeout, readonly_extra),
owner=str(ctx.get("session_id") or ctx.get("owner") or tool),
)
if containment.FILESYSTEM not in grant.enforced:
if argv:
command = [*command[:-1], _replace_workspace_alias(command[-1], grant.workspace)]
else:
tail.append(decoded)
command = _replace_workspace_alias(command, grant.workspace)
result = await containment.run(grant, command, argv=argv, progress_cb=ctx.get("progress_cb"))
except containment.ContainmentUnavailable as exc:
return containment.unavailable_tool_result(exc, tool=tool)
except (OSError, RuntimeError, ValueError) as exc:
boundary = grant.to_dict() if grant else {}
boundary["executed"] = bool(getattr(exc, "containment_executed", False))
if not getattr(exc, "containment_established", False):
boundary.update(contained=False, enforced=[])
return {"error": f"{tool}: execution failed: {exc}", "exit_code": 1,
"containment": boundary}
async def _progress_emitter():
await asyncio.sleep(PROGRESS_INTERVAL_S)
while True:
if progress_cb:
try:
await progress_cb({
"elapsed_s": round(time.time() - started, 1),
"tail": "\n".join(list(tail)),
})
except Exception:
pass
await asyncio.sleep(PROGRESS_INTERVAL_S)
boundary = result.grant.to_dict()
boundary["executed"] = True
teardown = result.release.to_dict() if result.release else {"dead": False}
output = result.stdout.rstrip()
if result.stderr.rstrip():
output = (output + "\nSTDERR: " + result.stderr.rstrip()).strip()
truncated = result.output_truncated or len(output) > MAX_OUTPUT_CHARS
capture_note = " Captured output was truncated." if truncated else ""
common = {"containment": boundary, "teardown": teardown, "output_truncated": truncated}
if not teardown["dead"]:
return {**common, "error": f"{tool}: process teardown could not verify death.{capture_note}",
"failure_kind": "process_teardown_failed", "exit_code": 1,
"stdout": _truncate(result.stdout, MAX_OUTPUT_CHARS),
"stderr": _truncate(result.stderr, MAX_OUTPUT_CHARS)}
if result.timed_out:
return {**common, "error": f"{tool}: timed out after {timeout}s; process tree terminated.{capture_note}",
"exit_code": 124, "stdout": _truncate(result.stdout, MAX_OUTPUT_CHARS),
"stderr": _truncate(result.stderr, MAX_OUTPUT_CHARS)}
if tool == "python":
child_failure = _python_child_runtime_failure(result.stdout, result.stderr, result.exit_code)
if child_failure:
return {**common, "error": _truncate("python: a child operation failed despite a zero Python exit status:\n" + child_failure, MAX_OUTPUT_CHARS),
"exit_code": 1, "stderr": _truncate(result.stderr, MAX_OUTPUT_CHARS)}
if truncated:
note = "\n…[output truncated by containment capture limit]…"
output = output[:MAX_OUTPUT_CHARS - len(note)] + note
return {**common, "output": _truncate(output, MAX_OUTPUT_CHARS) or "(no output)",
"exit_code": result.exit_code if result.exit_code is not None else 1}
rd_out = asyncio.create_task(_reader(proc.stdout, stdout_full, "out"))
rd_err = asyncio.create_task(_reader(proc.stderr, stderr_full, "err"))
prog_task = asyncio.create_task(_progress_emitter()) if progress_cb else None
timed_out = False
try:
await asyncio.wait_for(proc.wait(), timeout=timeout)
except asyncio.TimeoutError:
timed_out = True
try:
proc.kill()
except Exception:
pass
try:
await asyncio.wait_for(proc.wait(), timeout=2)
except Exception:
pass
except asyncio.CancelledError:
try:
proc.kill()
except Exception:
pass
try:
await asyncio.wait_for(proc.wait(), timeout=2)
except Exception:
pass
for t in (rd_out, rd_err):
t.cancel()
if prog_task is not None:
prog_task.cancel()
raise
finally:
if prog_task is not None and not prog_task.done():
prog_task.cancel()
try:
await prog_task
except (asyncio.CancelledError, Exception):
pass
for t in (rd_out, rd_err):
try:
await asyncio.wait_for(t, timeout=1)
except Exception:
pass
return (
"\n".join(stdout_full),
"\n".join(stderr_full),
proc.returncode,
timed_out,
)
class BashTool:
async def execute(self, content: str, ctx: dict) -> dict:
@@ -639,76 +609,10 @@ class BashTool:
),
"exit_code": 1,
}
isolated_tmp = os.path.join(agent_cwd(), ".tmp")
if "/tmp/" in content:
os.makedirs(isolated_tmp, exist_ok=True)
isolated_tmp = _isolated_tmp_dir(agent_cwd())
content = content.replace("/tmp/", isolated_tmp.rstrip("/") + "/")
namespaced = _wrap_workspace_namespace(content, agent_cwd())
content = namespaced or _replace_workspace_alias(content, agent_cwd())
progress_cb = ctx.get("progress_cb")
_subproc_env = ctx.get("subproc_env")
session_id = ctx.get("session_id")
if not IS_WINDOWS and session_id and shutil.which("tmux"):
stdout, stderr, rc, timed_out = await _run_tmux_bash(
content,
session_id=str(session_id),
cwd=agent_cwd(),
env=_subproc_env,
timeout=DEFAULT_BASH_TIMEOUT,
progress_cb=progress_cb,
)
if timed_out:
return {
"error": f"bash: timed out after {DEFAULT_BASH_TIMEOUT}s — terminated task shell session",
"exit_code": 124,
"stdout": _truncate(stdout, MAX_OUTPUT_CHARS),
"stderr": _truncate(stderr, MAX_OUTPUT_CHARS),
"tmux_session": _tmux_session_name(str(session_id)),
}
output = stdout.rstrip()
err = stderr.rstrip()
if err:
output = (output + "\nSTDERR: " + err).strip() if output else "STDERR: " + err
return {
"output": _truncate(output, MAX_OUTPUT_CHARS) or "(no output)",
"exit_code": rc or 0,
"tmux_session": _tmux_session_name(str(session_id)),
}
try:
if IS_WINDOWS:
proc = await _create_bash_subprocess(
content,
cwd=agent_cwd(),
env=_subproc_env,
)
else:
# Preserve the existing captured POSIX path; the structural
# helper is primarily needed to avoid cmd.exe on Windows.
proc = await asyncio.create_subprocess_shell(
content,
stdin=asyncio.subprocess.DEVNULL,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
env=_subproc_env,
cwd=agent_cwd(),
)
except RuntimeError as exc:
return {"error": str(exc), "exit_code": 1}
mark_operation_started('subprocess', pid=proc.pid)
stdout, stderr, rc, timed_out = await _run_subprocess_streaming(
proc,
timeout=DEFAULT_BASH_TIMEOUT,
progress_cb=progress_cb,
)
if timed_out:
return {"error": f"bash: timed out after {DEFAULT_BASH_TIMEOUT}s — process killed", "exit_code": 124, "stdout": _truncate(stdout, MAX_OUTPUT_CHARS), "stderr": _truncate(stderr, MAX_OUTPUT_CHARS)}
output = stdout.rstrip()
err = stderr.rstrip()
if err:
output = (output + "\nSTDERR: " + err).strip() if output else "STDERR: " + err
output = _truncate(output, MAX_OUTPUT_CHARS)
return {"output": output or "(no output)", "exit_code": rc or 0}
return await _run_owned_command(content, ctx, tool="bash", timeout=DEFAULT_BASH_TIMEOUT)
class HostShellTool:
async def execute(self, content: str, ctx: dict) -> dict:
@@ -749,6 +653,16 @@ class HostShellTool:
requested_timeout = 30
timeout = max(1, min(requested_timeout, 120))
from src import containment
from src.tool_execution import agent_cwd
sanitized_endpoint = f"{parsed.scheme}://{parsed.netloc}{parsed.path}" if parsed.scheme and parsed.netloc else "host_shell_bridge"
owner = str(ctx.get("session_id") or ctx.get("owner") or "host_shell")
spec = _owned_spec(agent_cwd(), ctx.get("subproc_env"), timeout)
grant = containment.declare_external_bridge(spec, owner=owner, endpoint=sanitized_endpoint)
boundary = grant.to_dict()
boundary["executed"] = False
request_body: dict[str, object] = {"timeout": timeout}
request_id = ""
if job_id:
@@ -772,6 +686,8 @@ class HostShellTool:
return {
"error": f"host_shell: bridge returned HTTP {resp.status_code}",
"exit_code": 1,
"host_bridge": "tui",
"containment": boundary,
}
data = resp.json()
@@ -798,6 +714,8 @@ class HostShellTool:
return {
"error": f"host_shell: bridge returned HTTP {poll.status_code}",
"exit_code": 1,
"host_bridge": "tui",
"containment": boundary,
}
data = poll.json()
if not isinstance(data, dict):
@@ -828,15 +746,16 @@ class HostShellTool:
task.add_done_callback(_HOST_SHELL_CANCEL_TASKS.discard)
raise
except Exception as e:
return {"error": f"host_shell: bridge call failed: {e}", "exit_code": 1}
return {"error": f"host_shell: bridge call failed: {e}", "exit_code": 1, "containment": boundary}
if not isinstance(data, dict):
return {"error": "host_shell: bridge returned invalid payload", "exit_code": 1}
return {"error": "host_shell: bridge returned invalid payload", "exit_code": 1, "containment": boundary}
if data.get("error"):
return {
"error": _truncate(str(data["error"]), MAX_OUTPUT_CHARS),
"exit_code": 1,
"host_bridge": "tui",
"containment": boundary,
}
stdout = str(data.get("stdout") or data.get("output") or "")
stderr = str(data.get("stderr") or "")
@@ -850,15 +769,18 @@ class HostShellTool:
"error": "host_shell: bridge returned an invalid exit_code",
"exit_code": 1,
"host_bridge": "tui",
"containment": boundary,
}
exit_code = raw_exit_code
output = stdout.rstrip()
if stderr.strip():
output = (output + "\nSTDERR: " + stderr.strip()).strip() if output else "STDERR: " + stderr.strip()
boundary["executed"] = True
result = {
"output": _truncate(output, MAX_OUTPUT_CHARS) or "(no output)",
"exit_code": exit_code,
"host_bridge": "tui",
"containment": boundary,
}
for key in ("detached", "job_id", "status", "running", "finished", "cwd"):
if key in data:
@@ -957,92 +879,18 @@ class PythonTool:
),
"exit_code": 1,
}
# Only create a mount namespace when the submitted code actually
# relies on the public virtual path. Ordinary Python probes and
# scripts should retain the real workspace as os.getcwd(); wrapping
# every invocation would make that stable contract appear as
# ``/workspace`` instead.
needs_virtual_namespace = bool(
"/workspace" in content
or re.search(r"\b(?:runpy\.run_path|exec\s*\(|importlib\.)", content)
)
isolated_tmp = os.path.join(agent_cwd(), ".tmp")
if "/tmp/" in content:
os.makedirs(isolated_tmp, exist_ok=True)
isolated_tmp = _isolated_tmp_dir(agent_cwd())
content = content.replace("/tmp/", isolated_tmp.rstrip("/") + "/")
progress_cb = ctx.get("progress_cb")
_subproc_env = ctx.get("subproc_env")
# Generated scripts commonly contain the public `/workspace/...`
# paths shown in the tool contract. Rewriting the inline `-c` body
# cannot repair paths embedded in a script loaded via `runpy`, and a
# process-global `/workspace` symlink would break concurrent tasks.
# Give Python the same per-task namespace Bash receives so both inline
# code and loaded scripts see the stable virtual workspace root.
namespaced_content = _python_with_configured_import_paths(
content = _python_with_configured_import_paths(
_python_with_visible_final_expression(content), _subproc_env
)
python_command = shlex.join((sys.executable or "python", "-I", "-c", namespaced_content))
# Code that explicitly uses the public /workspace path runs inside a
# namespace whose stable cwd is that same bind. Host workspaces under
# /tmp or another unbound parent are intentionally invisible by their
# real path inside the namespace; trying to chdir there makes otherwise
# valid native Python fail before execution.
namespaced = (
_wrap_workspace_namespace(
python_command,
agent_cwd(),
chdir="/workspace",
interpreter_prefix=sys.prefix,
)
if needs_virtual_namespace
else None
# All Python code acquires the same server-defined boundary, including
# arithmetic and ordinary imports. Source text never selects a scope.
raw_paths = str((_subproc_env or {}).get("ODYSSEUS_PYTHON_TOOL_SITE_PACKAGES", ""))
roots = tuple(path for path in raw_paths.split(os.pathsep) if path and os.path.isabs(path))
return await _run_owned_command(
[sys.executable or "python", "-I", "-c", content], ctx,
tool="python", timeout=DEFAULT_PYTHON_TIMEOUT, argv=True, readonly_extra=roots,
)
if namespaced:
proc = await asyncio.create_subprocess_exec(
"/bin/bash", "-lc", namespaced,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
env=_subproc_env,
cwd=agent_cwd(),
)
else:
# Platforms without a usable namespace still receive the same
# alias contract through a conservative source rewrite.
content = _python_with_configured_import_paths(
_python_with_visible_final_expression(
_replace_workspace_alias(content, agent_cwd())
),
_subproc_env,
)
proc = await asyncio.create_subprocess_exec(
(sys.executable or "python"), "-I", "-c", content,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
env=_subproc_env,
cwd=agent_cwd(),
)
mark_operation_started('subprocess', pid=proc.pid)
stdout, stderr, rc, timed_out = await _run_subprocess_streaming(
proc,
timeout=DEFAULT_PYTHON_TIMEOUT,
progress_cb=progress_cb,
)
if timed_out:
return {"error": f"python: timed out after {DEFAULT_PYTHON_TIMEOUT}s — process killed", "exit_code": 124, "stdout": _truncate(stdout, MAX_OUTPUT_CHARS), "stderr": _truncate(stderr, MAX_OUTPUT_CHARS)}
child_failure = _python_child_runtime_failure(stdout, stderr, rc)
if child_failure:
return {
"error": _truncate(
"python: a child operation failed despite a zero Python exit "
"status:\n" + child_failure,
MAX_OUTPUT_CHARS,
),
"exit_code": 1,
"stderr": _truncate(stderr, MAX_OUTPUT_CHARS),
}
output = stdout.rstrip()
err = stderr.rstrip()
if err:
output = (output + "\nSTDERR: " + err).strip() if output else "STDERR: " + err
output = _truncate(output, MAX_OUTPUT_CHARS)
return {"output": output or "(no output)", "exit_code": rc or 0}
+3 -8
View File
@@ -7,6 +7,8 @@ from fastapi import HTTPException
from fastapi.responses import HTMLResponse
from starlette.requests import Request
from src.path_confinement import is_inside
logger = logging.getLogger(__name__)
def read_if_exists(path: str) -> str:
@@ -51,11 +53,4 @@ def serve_html_with_nonce(request: Request, file_path: str) -> HTMLResponse:
def inside_base_dir(base_dir: str, path: str) -> bool:
"""Check if path is inside base directory."""
if not isinstance(base_dir, str) or not isinstance(path, str):
return False
base = os.path.realpath(base_dir)
p = os.path.realpath(path)
try:
return os.path.commonpath([base, p]) == base
except Exception:
return False
return is_inside(base_dir, path)
+177 -71
View File
@@ -22,22 +22,21 @@ from __future__ import annotations
import json
import os
import shlex
import sys
import subprocess
import time
import uuid
from pathlib import Path
from typing import Any, Dict, List, Optional
from core.atomic_io import atomic_write_json
from core.atomic_io import atomic_write_json, store_transaction
from core.platform_compat import (
detached_popen_kwargs,
find_bash,
git_bash_path,
kill_process_tree,
pid_alive,
)
from src import process_ownership
from src.constants import BG_JOBS_DIR, BG_JOBS_FILE
_JOBS_DIR = Path(BG_JOBS_DIR)
@@ -52,6 +51,7 @@ _MAX_OUTPUT_CHARS = 16000
# files) is kept before pruning, so neither the store nor data/bg_jobs/ grows
# without bound. The agent has already consumed the result by then.
_RETENTION_S = 3600 # 1 hour after follow-up
_LIVE_PROCS: dict[int, subprocess.Popen] = {}
def _load() -> Dict[str, Dict[str, Any]]:
@@ -78,66 +78,50 @@ def _pid_alive(pid: Optional[int]) -> bool:
return pid_alive(pid)
@store_transaction(lambda: _STORE)
def launch(command: str, session_id: str, cwd: Optional[str] = None,
max_runtime_s: int = DEFAULT_MAX_RUNTIME_S) -> Dict[str, Any]:
max_runtime_s: int = DEFAULT_MAX_RUNTIME_S, env: Optional[dict] = None) -> Dict[str, Any]:
"""Launch `command` detached. Returns the job record (status='running').
Output + the final exit code are written to files so status survives a
server restart. The process is put in its own session (setsid) so it
outlives the request/stream that started it.
A trusted detached supervisor owns the shared containment runner, output,
wall clock and exit metadata, independently of the request/server lifetime.
"""
_JOBS_DIR.mkdir(parents=True, exist_ok=True)
job_id = uuid.uuid4().hex[:12]
log_path = _JOBS_DIR / f"{job_id}.log"
exit_path = _JOBS_DIR / f"{job_id}.exit"
# The user command goes in its OWN script file, run as a child `bash`. This
# is what isolates it: an `exit` inside it only ends that child (so the
# wrapper still records the exit code), and — unlike textually wrapping the
# command in `( … )` — the wrapper can't be broken by an unbalanced paren or
# a trailing line-continuation in the command. `$?` is the child's real
# exit status.
bash = find_bash()
if bash:
# POSIX, or Windows with Git Bash/WSL. The user command goes in its OWN
# script file, run as a child `bash` — an `exit` inside it only ends
# that child (so the wrapper still records the exit code), and an
# unbalanced paren / trailing line-continuation in the command can't
# break the wrapper. `$?` is the child's real exit status. Paths are
# emitted as POSIX (forward-slash) + shell-quoted so Git Bash on Windows
# handles drive paths and spaces correctly.
cmd_path = _JOBS_DIR / f"{job_id}.cmd.sh"
cmd_path.write_text(command + "\n", encoding="utf-8")
lp, xp, cp = (shlex.quote(git_bash_path(p)) for p in (log_path, exit_path, cmd_path))
script_path = _JOBS_DIR / f"{job_id}.sh"
script_path.write_text(
f"bash {cp} > {lp} 2>&1\n"
f"echo $? > {xp}\n",
encoding="utf-8",
)
argv = [bash, str(script_path)]
else:
# Windows without any bash installed: cmd.exe wrapper. The command runs
# in its own child .cmd so %ERRORLEVEL% is the command's real exit code.
child_path = _JOBS_DIR / f"{job_id}.child.cmd"
child_path.write_text("@echo off\r\n" + command + "\r\n", encoding="utf-8")
script_path = _JOBS_DIR / f"{job_id}.cmd"
script_path.write_text(
"@echo off\r\n"
f'call "{child_path}" > "{log_path}" 2>&1\r\n'
f'echo %ERRORLEVEL%> "{exit_path}"\r\n',
encoding="utf-8",
)
argv = [os.environ.get("ComSpec", "cmd.exe"), "/c", str(script_path)]
proc = subprocess.Popen(
argv,
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
stdin=subprocess.DEVNULL,
cwd=cwd or None,
**detached_popen_kwargs(), # detach from the request lifecycle (setsid / DETACHED_PROCESS)
)
from src import containment
from src.agent_tools.subprocess_tools import _owned_spec, _replace_workspace_alias
spec = _owned_spec(cwd or os.getcwd(), env, max_runtime_s)
grant = containment.acquire(spec, owner=f"bg:{session_id}")
bounded_command = command
if containment.FILESYSTEM not in grant.enforced:
bounded_command = _replace_workspace_alias(command, grant.workspace)
result_path = _JOBS_DIR / f"{job_id}.result.json"
payload = {
"store_path": str(containment._store_path().resolve()),
"grant": {**grant.to_dict(), "owner": grant.owner},
"spec": {
"workspace": spec.workspace, "env": dict(spec.env), "wall_clock_s": spec.wall_clock_s,
"required": sorted(spec.required), "network": spec.network,
"readonly_extra": list(spec.readonly_extra), "writable_extra": list(spec.writable_extra),
"max_output_bytes": spec.max_output_bytes,
},
"command": bounded_command, "log_path": str(log_path.resolve()),
"result_path": str(result_path.resolve()), "exit_path": str(exit_path.resolve()),
}
try:
with open(log_path, "ab") as bootstrap_log:
proc = subprocess.Popen(
[sys.executable, str(Path(containment.__file__).with_name("containment_worker.py"))],
stdin=subprocess.PIPE, stdout=subprocess.DEVNULL, stderr=bootstrap_log,
cwd=str(Path(containment.__file__).resolve().parent.parent),
**detached_popen_kwargs(),
)
except BaseException:
containment.release(grant, grace_s=0)
raise
rec = {
"id": job_id,
@@ -152,10 +136,31 @@ def launch(command: str, session_id: str, cwd: Optional[str] = None,
"followed_up": False, # has the agent been re-invoked with the result?
"log_path": str(log_path),
"exit_path": str(exit_path),
"result_path": str(result_path),
"containment_id": grant.id,
"containment": {**grant.to_dict(), "contained": False, "enforced": [], "pending": True, "executed": False},
"pgid": None if os.name == "nt" else proc.pid,
# Identity, not just a slot. The pid above is reused by the kernel, and
# this record outlives the process and the server; the token is what a
# later run compares before it signals anything. See
# src/process_ownership.py.
"start_token": process_ownership.capture(proc.pid)["start_token"],
}
jobs = _load()
jobs[job_id] = rec
_save(jobs)
try:
containment._update_record(grant.id, lifetime="background", supervisor_pid=proc.pid,
supervisor_token=rec["start_token"])
jobs = _load()
jobs[job_id] = rec
_save(jobs)
# The supervisor cannot execute until the identity and job record are durable.
proc.stdin.write(json.dumps(payload).encode("utf-8"))
proc.stdin.close()
except BaseException:
kill_process_tree(proc.pid)
proc.wait(timeout=5)
containment.release(grant, grace_s=0)
raise
_LIVE_PROCS[proc.pid] = proc
return rec
@@ -188,10 +193,14 @@ def _prune(jobs: Dict[str, Dict[str, Any]], now: float) -> bool:
return bool(stale)
@store_transaction(lambda: _STORE)
def refresh() -> Dict[str, Dict[str, Any]]:
"""Reconcile every running job against disk. Marks done/failed (incl.
timeout). Idempotent — safe to call from a poll loop. Returns the store."""
jobs = _load()
for pid, proc in list(_LIVE_PROCS.items()):
if proc.poll() is not None:
_LIVE_PROCS.pop(pid, None)
changed = False
now = time.time()
for rec in jobs.values():
@@ -206,13 +215,24 @@ def refresh() -> Dict[str, Dict[str, Any]]:
rec["exit_code"] = code
rec["status"] = "done" if code == 0 else "failed"
rec["ended_at"] = now
if rec.get("result_path"):
try:
report = json.loads(Path(rec["result_path"]).read_text(encoding="utf-8"))
rec.update(report)
except (OSError, ValueError):
rec["status"], rec["exit_code"] = "failed", 1
rec["result_unavailable"] = True
changed = True
elif (now - rec.get("started_at", now)) > rec.get("max_runtime_s", DEFAULT_MAX_RUNTIME_S):
# Runaway / stuck — reap it but STILL surface a follow-up.
_kill(rec.get("pid"))
rec["status"] = "failed"
rec["exit_code"] = -1
rec["ended_at"] = now
outcome = _kill_record(rec)
rec["teardown"] = outcome.to_dict()
if outcome.dead:
rec["status"] = "failed"
rec["exit_code"] = -1
rec["ended_at"] = now
else:
rec["kill_failed"] = True
rec["timed_out"] = True
changed = True
elif not _pid_alive(rec.get("pid")) and not exit_path.exists():
@@ -230,9 +250,33 @@ def refresh() -> Dict[str, Dict[str, Any]]:
return jobs
def _kill(pid: Optional[int]) -> None:
def _kill(pid: Optional[int], **kwargs):
# Cross-platform process-tree teardown (POSIX killpg / Windows taskkill /T).
kill_process_tree(pid)
return kill_process_tree(pid, **kwargs)
def _kill_record(rec):
from src import containment
verdict = process_ownership.verify(rec.get("pid"), rec.get("start_token"))
if verdict in (process_ownership.FOREIGN, process_ownership.UNVERIFIABLE):
return containment.ReleaseOutcome(dead=False, escalated=False, ownership=verdict)
if rec.get("containment_id"):
record = containment._load_records().get(rec["containment_id"])
if record and record.get("pid"):
outcome = containment.reap_record(record)
if not outcome.dead:
return outcome
outcome = _kill(rec.get("pid"), start_token=rec.get("start_token"),
pgid=rec.get("pgid"), require_identity=True)
proc = _LIVE_PROCS.get(rec.get("pid"))
if proc and outcome.dead:
proc.wait(timeout=5)
_LIVE_PROCS.pop(proc.pid, None)
if outcome.dead and rec.get("containment_id"):
record = containment._load_records().get(rec["containment_id"])
if record and not record.get("pid"):
containment.reap_record(record)
return outcome
def pending_followups() -> List[Dict[str, Any]]:
@@ -243,6 +287,7 @@ def pending_followups() -> List[Dict[str, Any]]:
if r.get("status") in ("done", "failed") and not r.get("followed_up")]
@store_transaction(lambda: _STORE)
def mark_followed_up(job_id: str) -> None:
jobs = _load()
if job_id in jobs:
@@ -263,6 +308,7 @@ def list_for_session(session_id: str) -> List[Dict[str, Any]]:
return [r for r in refresh().values() if r.get("session_id") == session_id]
@store_transaction(lambda: _STORE)
def kill(job_id: str) -> Optional[Dict[str, Any]]:
"""Terminate a running job's process tree and mark it killed. Returns the
updated record, or None if the id is unknown. Idempotent: a job that already
@@ -273,20 +319,80 @@ def kill(job_id: str) -> Optional[Dict[str, Any]]:
if rec is None:
return None
if rec.get("status") == "running":
_kill(rec.get("pid"))
rec["status"] = "failed"
rec["exit_code"] = -1
rec["ended_at"] = time.time()
rec["killed"] = True
rec["followed_up"] = True
outcome = _kill_record(rec)
rec["teardown"] = outcome.to_dict()
if outcome.dead:
rec["status"] = "failed"
rec["exit_code"] = -1
rec["ended_at"] = time.time()
rec["killed"] = True
rec["followed_up"] = True
else:
rec["kill_failed"] = True
_save(jobs)
return rec
@store_transaction(lambda: _STORE)
def disown_unverified() -> Dict[str, Any]:
"""Stop tracking running jobs whose process can no longer be proven ours.
Called once at startup by :mod:`src.process_reaper`, never from the poll
loop — every record it sees was written by an earlier run, which is what
makes "unidentifiable" a statement about a previous run's child rather than
about a job this run just launched.
Signals nothing. A detached job is meant to survive a restart, so a job that
verifies as ours is left alone and its result is still collected. What is
corrected is the record that would otherwise be signalled later on a pid the
kernel has reassigned: the max-runtime branch of :func:`refresh` sends
SIGTERM then SIGKILL to ``rec["pid"]`` an hour in, and on a reused pid that
lands on a bystander.
Fail closed: a job that cannot be verified is retired too, not kept.
Retiring loses a result, which is visible; keeping it leaves a pid this
server will eventually signal without knowing what it is pointing at, which
is not.
"""
jobs = _load()
report = {"seen": 0, "retired": 0, "kept": 0}
changed = False
now = time.time()
for rec in jobs.values():
if rec.get("status") != "running":
continue
report["seen"] += 1
verdict = process_ownership.verify(rec.get("pid"), rec.get("start_token"))
if verdict in (process_ownership.OWNED, process_ownership.GONE):
# OWNED: still ours, still running, still watched. GONE: refresh()
# already turns an absent process into a "died" record, and it may
# yet find an exit-code file the job wrote before it went.
report["kept"] += 1
continue
rec["status"] = "failed"
rec["exit_code"] = -1
rec["ended_at"] = now
rec["ownership_lost"] = verdict
# followed_up stays False: the agent asked for this job and is owed an
# answer, even when the answer is that we lost track of it.
report["retired"] += 1
changed = True
if changed:
_save(jobs)
return report
def result_text(rec: Dict[str, Any]) -> str:
"""Human/agent-readable summary of a finished job, for the follow-up."""
out = _read_output(rec)
if rec.get("killed"):
if rec.get("ownership_lost"):
head = (
"Background job was abandoned across a server restart: its process "
f"could not be identified as ours ({rec.get('ownership_lost')}), so it was "
"neither waited on nor signalled. Any output below is what it had "
"written by then; if the work matters, re-run it."
)
elif rec.get("killed"):
head = "Background job was killed."
elif rec.get("timed_out"):
head = f"Background job timed out after {rec.get('max_runtime_s')}s."
+12
View File
@@ -75,6 +75,7 @@ APP_KEY_FILE = os.path.join(DATA_DIR, ".app_key")
EMBEDDING_ENDPOINT_FILE = os.path.join(DATA_DIR, "embedding_endpoint.json")
COOKBOOK_STATE_FILE = os.path.join(DATA_DIR, "cookbook_state.json")
BG_JOBS_FILE = os.path.join(DATA_DIR, "bg_jobs.json")
CONTAINMENT_STATE_FILE = os.path.join(DATA_DIR, "containment_grants.json")
VAULT_FILE = os.path.join(DATA_DIR, "vault.json")
TIDY_CALENDAR_STATE_FILE = os.path.join(DATA_DIR, "tidy_calendar_state.json")
SKILLS_FILE = os.path.join(DATA_DIR, "skills.json")
@@ -154,6 +155,17 @@ SCHOLARLY_LOOKUP_TOTAL_BUDGET = 20.0
CLEANUP_ENABLED = os.getenv("CLEANUP_ENABLED", "True").lower() == "true"
CLEANUP_INTERVAL_HOURS = int(os.getenv("CLEANUP_INTERVAL_HOURS", "24"))
# Agent workspace
# The stable virtual root the tool contract promises an agent, independent of
# where the workspace physically lives. Both the mount namespace and the
# path resolvers map it to the active workspace, so it is the one absolute path
# a contained command may assume.
WORKSPACE_MOUNT = "/workspace"
# Scratch directory inside the workspace that agent shell commands get in place
# of the host /tmp. A dirname rather than a path: the workspace is dynamic, so
# the full path is only knowable per turn.
AGENT_ISOLATED_TMP_DIRNAME = ".tmp"
# Auth policy
PASSWORD_MIN_LENGTH = 8
+1808
View File
File diff suppressed because it is too large Load Diff
+84
View File
@@ -0,0 +1,84 @@
"""Trusted detached supervisor; command execution stays in containment.run."""
from __future__ import annotations
import asyncio
import json
import signal
import sys
import types
from pathlib import Path
# Launch by absolute script path, so a task workspace cannot shadow src.
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
# This supervisor needs atomic I/O and platform primitives, not core's chat
# facade (auth, database, LLM startup). Keep that facade out of the detached
# process without changing the application's normal imports.
core_package = types.ModuleType("core")
core_package.__path__ = [str(Path(__file__).resolve().parent.parent / "core")]
sys.modules["core"] = core_package
from core.atomic_io import atomic_write_json, atomic_write_text
from src import containment
async def supervise(payload: dict) -> None:
containment._store_path = lambda: Path(payload["store_path"])
data = payload["spec"]
data["required"] = frozenset(data["required"])
spec = containment.ContainmentSpec(**data)
info = payload["grant"]
grant = containment.ContainmentGrant(
id=info["id"], mechanism=info["mechanism"], workspace=spec.workspace,
enforced=frozenset(info["enforced"]), degraded=tuple(info["degraded"]),
unenforced_required=tuple(info["unenforced_required"]), owner=info["owner"],
mode=info["mode"], spec=spec,
)
task = asyncio.current_task()
loop = asyncio.get_running_loop()
if sys.platform != "win32":
loop.add_signal_handler(signal.SIGTERM, task.cancel)
loop.add_signal_handler(signal.SIGINT, task.cancel)
try:
with open(payload["log_path"], "w", encoding="utf-8") as log:
def capture(text):
log.write(text)
log.flush()
result = await containment.run(grant, payload["command"], output_cb=capture)
output = ""
code = 124 if result.timed_out else result.exit_code
if not result.release or not result.release.dead:
code = 1
report = {"containment": result.grant.to_dict(),
"teardown": result.release.to_dict() if result.release else {"dead": False},
"output_truncated": result.output_truncated,
"timed_out": result.timed_out}
report["containment"]["executed"] = True
if result.output_truncated:
output = "\n…[output truncated by containment capture limit]…\n"
except BaseException as exc:
record = containment._load_records().get(grant.id, {})
if not record.get("pid") and not record.get("release"):
containment.release(grant, grace_s=0)
record = containment._load_records().get(grant.id, {})
output, code = f"background execution failed: {type(exc).__name__}: {exc}\n", 1
report = {"containment": grant.to_dict(), "teardown": record.get("release") or {"dead": False},
"output_truncated": False}
report["containment"]["executed"] = bool(record.get("execution_started"))
if not record.get("containment_ready"):
report["containment"].update(contained=False, enforced=[])
if isinstance(exc, containment.ContainmentUnavailable):
report.update(containment.unavailable_tool_result(exc, tool="bash"))
if output:
try:
with open(payload["log_path"], "a", encoding="utf-8") as log:
log.write(output)
except OSError:
# A failed log initialization must not hide completion metadata.
sys.stderr.write(output)
atomic_write_json(payload["result_path"], report)
# Publish completion last: refresh must never see an exit without metadata.
atomic_write_text(payload["exit_path"], str(code if code is not None else 1))
if __name__ == "__main__":
asyncio.run(supervise(json.load(sys.stdin)))
+3 -6
View File
@@ -1,10 +1,10 @@
import os
import re
from pathlib import Path
from fastapi import HTTPException
from src.constants import GENERATED_IMAGES_DIR
from src.path_confinement import confine
GENERATED_IMAGE_DIR = Path(GENERATED_IMAGES_DIR)
@@ -20,12 +20,9 @@ GENERATED_IMAGE_HEADERS = {
def resolve_generated_image_path(filename: str) -> Path:
if not isinstance(filename, str) or not GENERATED_IMAGE_RE.fullmatch(filename):
raise HTTPException(status_code=400, detail="Invalid filename")
root = GENERATED_IMAGE_DIR.resolve()
path = (GENERATED_IMAGE_DIR / filename).resolve()
try:
if os.path.commonpath([str(root), str(path)]) != str(root):
raise ValueError
except Exception:
path = Path(confine(GENERATED_IMAGE_DIR, filename, allow_root=False))
except (ValueError, OSError):
raise HTTPException(status_code=400, detail="Invalid filename")
if not path.exists():
raise HTTPException(status_code=404, detail="Image not found")
+188
View File
@@ -0,0 +1,188 @@
"""The filesystem confinement boundary. One implementation, every call site.
"Is this path inside that root" is asked in twenty places in this tree, and
twenty times it is answered by a locally written ``realpath`` +
``os.path.commonpath`` pair. Each one is defensible on its own. Together they
are the problem: the boundary has no single definition, so a site that gets a
detail wrong is wrong *alone*, and a site added tomorrow starts from whichever
neighbour its author happened to copy.
The details that differ between those copies, and what this module settles:
**Both sides get canonicalized.** Comparing a ``realpath``-ed candidate against
a root that was only ``abspath``-ed is the bug class that has already cost this
project real time: on macOS ``/tmp`` is a symlink to ``/private/tmp``, so the
two sides disagree about a path neither of them is wrong about. It reads as an
escape and refuses a legitimate access. Canonicalizing one side is worse than
canonicalizing neither.
**``commonpath``, never ``startswith``.** ``/a/bc`` begins with ``/a/b`` and is
not inside it.
**Case folding is the filesystem's business, not the comparison's.**
``os.path.normcase`` lowercases on Windows and is the identity everywhere else
— including macOS, whose default filesystem is case-insensitive while its
``realpath`` preserves case. So normcase alone does not make the comparison
agree with the filesystem on macOS, and :func:`is_inside` does not pretend
otherwise: it answers about the canonical path, which is the question a
confinement check should be asking. Where a caller needs to match the
filesystem's own folding it must compare real paths of real files, not strings.
**A relative candidate joins the root, never the process cwd.** ``abspath`` of a
relative path silently uses ``os.getcwd()``, which is whatever the server
happens to be running in. A confinement helper that does that is resolving
against the wrong base before it even starts comparing.
**NUL and newline are rejected, not caught.** Several of the copies wrap the
whole comparison in ``except Exception: return False``, which turns a malformed
path into "outside" — the safe answer, reached by accident. Here it is a
``ValueError`` with a reason.
**``commonpath`` raising means outside.** It raises across Windows drive letters
and for mixed absolute/relative inputs. Both mean the candidate is not under the
root, so the refusal is deliberate rather than incidental.
What this module does *not* do: decide whether a path is sensitive (``.ssh``,
``id_rsa``, …). That is a separate deny list applied inside an allowed root, and
it lives with the callers that own it — ``src/tool_execution`` for the agent
tools. Confinement answers "inside the root"; it does not answer "allowed".
Relationship to :mod:`src.containment`: that module is the boundary for *where a
process runs*; this one is the boundary for *which paths a path check accepts*.
A contained process is restricted by a mount namespace, which this module cannot
express and does not try to; an in-process read of a model-supplied path is
restricted by this module, which a namespace does not see.
"""
from __future__ import annotations
import os
__all__ = [
"PathEscape",
"canonical_root",
"confine",
"is_inside",
]
class PathEscape(ValueError):
"""A candidate path does not resolve inside the root it was checked against.
A subclass of :class:`ValueError` so the call sites this replaces — which
raise ``ValueError`` and are caught as such by their callers and their
tests — keep behaving the way they did.
"""
def __init__(self, root: str, candidate: str, reason: str = "") -> None:
self.root = str(root)
self.candidate = str(candidate)
self.reason = str(reason or "outside the allowed root")
super().__init__(
f"path {self.candidate!r} is {self.reason} ({self.root})"
)
def _reject_unusable(value: str, *, label: str) -> str:
"""Normalize a path argument to ``str``, refusing the unusable shapes.
``\\x00`` is refused here because the OS layer raises on it much later and
from somewhere unhelpful, and because a broad ``except Exception`` around
the comparison would otherwise record it as an ordinary escape. Newlines
are refused for the same reason the workspace-mount parser refuses them:
a path carrying one has been built by splitting something that was not a
path list.
"""
if value is None:
raise ValueError(f"{label} is required")
if isinstance(value, os.PathLike):
value = os.fspath(value)
if not isinstance(value, str):
raise ValueError(f"{label} must be a path, got {type(value).__name__}")
text = value.strip()
if not text:
raise ValueError(f"{label} is required")
if "\x00" in text:
raise ValueError(f"{label} must not contain NUL")
if "\n" in text or "\r" in text:
raise ValueError(f"{label} must not contain a newline")
return text
def canonical_root(root) -> str:
"""The canonical form of a confinement root.
Exposed because a caller that holds a root across several checks should
canonicalize it once, and because a caller comparing two paths itself needs
the same canonical form this module compares against — a realpath-ed value
tested against a raw one is the asymmetry this module exists to remove.
"""
text = _reject_unusable(root, label="root")
return os.path.realpath(os.path.expanduser(text))
def _canonical_candidate(root: str, candidate) -> str:
"""Canonicalize ``candidate``, resolving a relative path under ``root``.
``realpath`` is deliberately the non-strict kind: a final component that
does not exist yet is normalized rather than refused, because a write target
is a legitimate thing to confine. Everything that *does* exist is resolved,
so a symlink anywhere in the chain — including the final component — is
followed before the comparison rather than after the open.
"""
text = _reject_unusable(candidate, label="path")
expanded = os.path.expanduser(text)
if not os.path.isabs(expanded):
expanded = os.path.join(root, expanded)
return os.path.realpath(expanded)
def is_inside(root, candidate, *, allow_root: bool = True) -> bool:
"""True when ``candidate`` resolves inside ``root``.
The boolean form, for call sites whose contract is a predicate. A malformed
argument is ``False`` here rather than a raise, because a predicate that
raises is the reason those call sites wrapped themselves in
``except Exception`` in the first place. Use :func:`confine` where the
caller wants the resolved path and a reason for the refusal.
``allow_root=False`` excludes the root itself, for a caller whose operation
is only meaningful on something *under* the root — deleting a file, say,
where the root is the directory it must not be.
"""
try:
confine(root, candidate, allow_root=allow_root)
return True
except (ValueError, OSError):
return False
def confine(root, candidate, *, allow_root: bool = True) -> str:
"""Resolve ``candidate`` inside ``root``, or raise.
Returns the canonical absolute path, which is what the caller should then
open: resolving and then opening the *original* string re-introduces the
symlink race the resolution just closed.
:raises ValueError: either argument is unusable as a path.
:raises PathEscape: the candidate resolves outside the root.
"""
base = canonical_root(root)
resolved = _canonical_candidate(base, candidate)
if resolved == base:
if allow_root:
return resolved
raise PathEscape(base, candidate, "the root itself, not a path inside it")
# normcase folds case on Windows and is the identity elsewhere; it is
# applied to both sides or to neither, which is the whole point.
try:
common = os.path.commonpath([os.path.normcase(resolved), os.path.normcase(base)])
except ValueError:
# Different Windows drives, or mixed absolute/relative. Both mean the
# candidate is not under the root.
raise PathEscape(base, candidate) from None
if common != os.path.normcase(base):
raise PathEscape(base, candidate)
return resolved
+414
View File
@@ -0,0 +1,414 @@
"""Process identity: is this pid still the process we started?
A recorded pid is not an identity. The kernel reuses pids, and every store in
this tree that remembers a process — ``data/bg_jobs.json``,
``data/containment_grants.json``, the Cookbook's task list — outlives the
process that wrote it, by design: those records exist so a restart does not lose
a job. The combination is the defect this module closes. A record that says
``pid 4242`` and a live ``pid 4242`` are not the same claim, and signalling the
second because the first was written is how a teardown kills a stranger.
That is not hypothetical here. ODY-86 was pid files unlinked while the daemons
they named were still live, with ownership never verified; the Cookbook survivor
sweep still terminates *any* process whose command line matches a tracked one,
which is a different spelling of the same mistake.
**The identity is (pid, start token).** A pid identifies a slot; the start token
identifies which process is occupying it. The kernel will not reissue a pid to a
process that started earlier, so comparing the token recorded at launch with the
token read now answers "is this still ours" without a handle, a lock file or a
supervisor.
Four verdicts, and the fourth is the point
------------------------------------------
:data:`OWNED`, :data:`GONE` and :data:`FOREIGN` are the answers. The fourth,
:data:`UNVERIFIABLE`, is what this host could not determine — no procfs, no
``ps``, a probe that raised, or a record written before anything recorded a
token. It is deliberately **not** collapsed into either "ours" (which would
signal strangers) or "gone" (which would abandon live processes).
Process inspection has broken off Linux four times in this tree — ODY-70, -86,
-94, -99 — every time because an inspection mechanism that was absent read as a
successful answer. So :data:`UNVERIFIABLE` is a containment failure and callers
must treat it as one: do not signal, and do not report a teardown that was not
performed. Refusing to act is the only honest option when you cannot tell what
you would be acting on.
Token granularity, stated because it bounds the guarantee
---------------------------------------------------------
======== ============================= ===============
Host Source Resolution
======== ============================= ===============
Linux boot ID + stat field 22 ~10 ms (1 tick)
macOS ``ps -o lstart=`` 1 s
Windows ``GetProcessTimes`` 100 ns
======== ============================= ===============
A pid recycled *within one token tick* is indistinguishable from the original.
On Linux and Windows that window is too small to hit in practice. On macOS it is
one second, which a pid wrap could theoretically land inside — so the token
narrows the risk by many orders of magnitude there without eliminating it. It is
a strictly better claim than the pid alone, which is the comparison that
matters; it is not a proof of identity and this module does not claim one.
"""
from __future__ import annotations
import logging
import os
import shutil
import subprocess
from typing import Any, Iterable, Mapping, NamedTuple, Optional
from core.platform_compat import IS_WINDOWS, PROC_ROOT, has_procfs
logger = logging.getLogger(__name__)
# ── Verdicts ────────────────────────────────────────────────────────────────
#: The pid is running and is the same process the token was taken from.
OWNED = "owned"
#: No process holds the pid. Nothing to signal and nothing to reap.
GONE = "gone"
#: A process holds the pid, and it is **not** ours — the pid was recycled.
#: Never signal a foreign pid; that is the defect, not the fix.
FOREIGN = "foreign"
#: This host could not answer. A containment failure, not a default.
UNVERIFIABLE = "unverifiable"
#: Verdicts that permit a signal. Exactly one.
SIGNALLABLE = frozenset({OWNED})
# ── Inspection mechanisms ───────────────────────────────────────────────────
MECHANISM_PROCFS = "procfs"
MECHANISM_PS = "ps"
MECHANISM_WIN32 = "win32"
#: No way to inspect processes on this host. Every verdict becomes
#: UNVERIFIABLE, which is the honest answer and not a permissive one.
MECHANISM_NONE = "none"
# The failure path only: a wedged `ps` must never hold up a teardown decision.
_PS_TIMEOUT_S = 5
#: Field 22 of ``/proc/<pid>/stat`` (1-indexed) is the process start time in
#: clock ticks since boot. Fields 1 and 2 are skipped by splitting on the last
#: ``)`` first, because a comm can itself contain spaces and parentheses.
_PROC_STAT_STARTTIME_INDEX = 19
class InspectionUnavailable(RuntimeError):
"""This host offers no way to inspect a process.
Raised by the probes rather than returned, so a caller that forgets to
handle it fails loudly instead of silently reading an absent mechanism as
"the process is gone". :func:`verify` catches it and reports
:data:`UNVERIFIABLE`.
"""
def __init__(self, what: str) -> None:
super().__init__(f"process inspection unavailable: cannot read {what}")
self.what = what
def inspection_mechanism() -> str:
"""Which mechanism this host can answer identity questions with.
Probed per call rather than cached at import: the tests substitute
``PROC_ROOT`` to exercise both branches on either kind of host, and a cached
answer would pin whichever host happened to import the module first.
"""
if IS_WINDOWS:
return MECHANISM_WIN32
if has_procfs():
return MECHANISM_PROCFS
if shutil.which("ps"):
return MECHANISM_PS
return MECHANISM_NONE
def inspection_available() -> bool:
return inspection_mechanism() != MECHANISM_NONE
# ── Start tokens ────────────────────────────────────────────────────────────
def _procfs_token(pid: int) -> Optional[str]:
try:
raw = (PROC_ROOT / str(pid) / "stat").read_text(encoding="utf-8", errors="replace")
except (FileNotFoundError, ProcessLookupError):
return None
except (OSError, PermissionError) as exc:
# The pid exists but is not readable. "I cannot tell" is not "it is
# gone", so this must not return None.
raise InspectionUnavailable(f"/proc/{pid}/stat ({exc})") from exc
# comm is parenthesised and may contain spaces and ')' — split past the last.
_, _, rest = raw.rpartition(")")
fields = rest.split()
try:
ticks = fields[_PROC_STAT_STARTTIME_INDEX]
except IndexError:
raise InspectionUnavailable(f"/proc/{pid}/stat (unexpected layout)") from None
try:
boot = (PROC_ROOT / "sys/kernel/random/boot_id").read_text(encoding="ascii").strip()
except OSError as exc:
raise InspectionUnavailable(f"boot identity ({exc})") from exc
if not boot:
raise InspectionUnavailable("boot identity (empty)")
# A persisted PID/start-tick pair can recur after reboot. Bind it to the
# boot as well; older receipts cannot authorize a signal on a new boot.
return f"procfs:{boot}:{ticks}"
def _ps_token(pid: int) -> Optional[str]:
try:
completed = subprocess.run(
["ps", "-p", str(pid), "-o", "lstart="],
stdout=subprocess.PIPE,
stderr=subprocess.DEVNULL,
timeout=_PS_TIMEOUT_S,
text=True,
)
except (OSError, subprocess.SubprocessError) as exc:
raise InspectionUnavailable(f"ps -p {pid} ({exc})") from exc
value = (completed.stdout or "").strip()
if completed.returncode != 0:
# ps exits non-zero for a pid that does not exist. With no output that
# is an absent process; with output it is a mechanism that misbehaved.
if not value:
return None
raise InspectionUnavailable(f"ps -p {pid} (exit {completed.returncode})")
if not value:
return None
return f"ps:{' '.join(value.split())}"
def _win32_token(pid: int) -> Optional[str]:
import ctypes
from ctypes import wintypes
PROCESS_QUERY_LIMITED_INFORMATION = 0x1000
kernel32 = ctypes.windll.kernel32
handle = kernel32.OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, False, int(pid))
if not handle:
return None
try:
creation = wintypes.FILETIME()
exit_time = wintypes.FILETIME()
kernel_time = wintypes.FILETIME()
user_time = wintypes.FILETIME()
ok = kernel32.GetProcessTimes(
handle,
ctypes.byref(creation),
ctypes.byref(exit_time),
ctypes.byref(kernel_time),
ctypes.byref(user_time),
)
if not ok:
raise InspectionUnavailable(f"GetProcessTimes({pid})")
stamp = (int(creation.dwHighDateTime) << 32) | int(creation.dwLowDateTime)
return f"win32:{stamp}"
finally:
kernel32.CloseHandle(handle)
def start_token(pid: Optional[int]) -> Optional[str]:
"""An opaque token identifying the process currently holding ``pid``.
Returns None when no process holds the pid. Raises
:class:`InspectionUnavailable` when this host cannot answer — never a
token, and never None, for a question it could not ask.
Record this at launch next to the pid. Compare it before signalling.
"""
if not pid:
return None
try:
pid = int(pid)
except (TypeError, ValueError):
return None
if pid <= 0:
return None
mechanism = inspection_mechanism()
if mechanism == MECHANISM_WIN32:
return _win32_token(pid)
if mechanism == MECHANISM_PROCFS:
return _procfs_token(pid)
if mechanism == MECHANISM_PS:
return _ps_token(pid)
raise InspectionUnavailable("process start time on this host")
def verify(pid: Optional[int], token: Optional[str]) -> str:
"""Is the process now holding ``pid`` the one ``token`` was taken from?
Returns :data:`OWNED`, :data:`GONE`, :data:`FOREIGN` or
:data:`UNVERIFIABLE`. Only :data:`OWNED` permits a signal.
A missing or empty ``token`` is :data:`UNVERIFIABLE`, not :data:`OWNED`:
a record that never captured an identity cannot establish one afterwards,
and treating "we did not write it down" as "it is ours" is precisely the
assumption that makes a recycled pid lethal.
"""
if not pid:
return GONE
if not token:
return UNVERIFIABLE
try:
current = start_token(pid)
except InspectionUnavailable as exc:
logger.warning("process_ownership: cannot verify pid %s: %s", pid, exc)
return UNVERIFIABLE
if current is None:
return GONE
return OWNED if current == str(token) else FOREIGN
def verify_record(
record: Mapping[str, Any], *, pid_key: str = "pid", token_key: str = "start_token",
) -> str:
""":func:`verify` against a stored record. Convenience for the reaper."""
return verify((record or {}).get(pid_key), (record or {}).get(token_key))
def capture(pid: Optional[int]) -> dict[str, Any]:
"""The identity fields to persist for a process at launch.
Always returns both keys, with ``start_token`` None when the host could not
produce one, so a record's shape never depends on the host and a later
reader can tell "no token" from "no field".
"""
try:
token = start_token(pid)
except InspectionUnavailable as exc:
logger.warning("process_ownership: launched pid %s without an identity: %s", pid, exc)
token = None
return {"pid": int(pid) if pid else None, "start_token": token}
# ── The process table ───────────────────────────────────────────────────────
class ProcessInfo(NamedTuple):
pid: int
ppid: int
command: str
#: Field 4 of ``/proc/<pid>/stat`` (1-indexed) is the parent pid; it lands at
#: index 1 of the fields that follow the comm's closing paren.
_PROC_STAT_PPID_INDEX = 1
def _procfs_process_table() -> dict[int, ProcessInfo]:
# Guarded here and not only in process_table(): a procfs scan whose
# existence check sits in a caller is one refactor away from being an
# unguarded scan, which is the defect tests/test_procfs_scan_guard.py pins.
if not has_procfs():
raise InspectionUnavailable(f"the process table via {PROC_ROOT}")
table: dict[int, ProcessInfo] = {}
for entry in os.listdir(PROC_ROOT):
if not entry.isdigit():
continue
pid = int(entry)
try:
raw = (PROC_ROOT / entry / "cmdline").read_bytes()
command = raw.replace(b"\x00", b" ").decode("utf-8", errors="replace").strip()
except (OSError, PermissionError):
continue
ppid = 0
try:
stat = (PROC_ROOT / entry / "stat").read_text(encoding="utf-8", errors="replace")
_, _, rest = stat.rpartition(")")
ppid = int(rest.split()[_PROC_STAT_PPID_INDEX])
except (OSError, PermissionError, IndexError, ValueError):
# A kernel thread or a pid that exited mid-walk. Keeping the row
# with ppid 0 is better than dropping it: a command-line match
# still works, only the descendant walk loses this link.
pass
if command:
table[pid] = ProcessInfo(pid=pid, ppid=ppid, command=command)
return table
def _ps_process_table() -> dict[int, ProcessInfo]:
try:
completed = subprocess.run(
# -ww defeats ps's default truncation to terminal width; without it
# a long serve command is clipped and no match can ever be exact.
["ps", "-axww", "-o", "pid=,ppid=,command="],
stdout=subprocess.PIPE,
stderr=subprocess.DEVNULL,
timeout=_PS_TIMEOUT_S,
text=True,
)
except (OSError, subprocess.SubprocessError) as exc:
raise InspectionUnavailable(f"ps -axww ({exc})") from exc
if completed.returncode != 0:
raise InspectionUnavailable(f"ps -axww (exit {completed.returncode})")
table: dict[int, ProcessInfo] = {}
for line in (completed.stdout or "").splitlines():
parts = line.strip().split(None, 2)
if len(parts) < 3 or not parts[0].isdigit() or not parts[1].isdigit():
continue
command = parts[2].strip()
if command:
pid = int(parts[0])
table[pid] = ProcessInfo(pid=pid, ppid=int(parts[1]), command=command)
return table
def process_table() -> dict[int, ProcessInfo]:
"""Every visible process, by pid, with its parent and full command line.
Raises :class:`InspectionUnavailable` when the host cannot enumerate
processes, so a caller reports that it could not look rather than reporting
that it found nothing. Those are different answers and this tree has
conflated them before (ODY-94).
``ps`` covers macOS and the BSDs, which have no procfs to walk — the reason
this exists rather than another ``/proc`` scan. procfs is preferred where
present because it needs no subprocess.
"""
mechanism = inspection_mechanism()
if mechanism == MECHANISM_PROCFS:
return _procfs_process_table()
if mechanism == MECHANISM_PS:
return _ps_process_table()
# Windows: tasklist cannot report a full command line without WMI, and a
# truncated one cannot be matched exactly. Claiming an empty table would
# read as "no survivors".
raise InspectionUnavailable(f"the process table via {mechanism}")
def command_lines() -> dict[int, str]:
"""Every visible pid mapped to its full command line."""
return {pid: info.command for pid, info in process_table().items()}
def descendants(
roots: "Iterable[int]", *, table: Optional[Mapping[int, ProcessInfo]] = None,
) -> list[int]:
"""Every process under ``roots``, roots included, breadth-first.
The point of taking several roots and one table is that the answer is a
*snapshot*: walking the tree one subprocess call at a time lets a child be
reparented between calls and vanish from the result. Callers that need to
act on a tree should capture it once, before they start tearing it down.
A pid that is its own parent, or a cycle the table reports, terminates the
walk rather than looping.
"""
rows = dict(table) if table is not None else process_table()
children: dict[int, list[int]] = {}
for info in rows.values():
children.setdefault(info.ppid, []).append(info.pid)
found: list[int] = []
seen: set[int] = set()
queue = [int(root) for root in roots if root]
while queue:
pid = queue.pop(0)
if pid in seen:
continue
seen.add(pid)
found.append(pid)
queue.extend(child for child in children.get(pid, ()) if child not in seen)
return found
+295
View File
@@ -0,0 +1,295 @@
"""Startup reconciliation for processes a previous run left behind.
Two stores in this tree outlive the process that wrote them, on purpose:
``data/containment_grants.json`` so a restart can reap rather than orphan, and
``data/bg_jobs.json`` so a restart never loses a detached job or its result.
Until now nothing read either of them at startup. A crashed or restarted server
therefore left every grant permanently "active" and every background job
permanently "running", and the first thing to touch one of those records was a
teardown aimed at a pid that had been reassigned in the meantime.
This module runs once, during startup, before anything of this run exists. That
timing is what makes its rules safe: every record it sees was written by an
earlier run, so "I cannot identify this process" is information about a previous
run's child and not about one of ours.
The two stores get **opposite** treatment, which is the whole reason this is a
module and not a loop:
* A **containment grant** is tied to a tool call that no longer has a caller.
A live process under an abandoned grant is by definition an orphan, so it is
torn down.
* A **background job** is detached deliberately and is documented to survive a
uvicorn restart. Killing one here would break the feature, so its record is
only corrected, never reaped. What gets fixed is identity: a job whose pid now
belongs to someone else is retired so that nothing later signals the stranger.
Fail closed in both: a signal requires a positive identity from
:mod:`src.process_ownership`, and every other verdict is recorded rather than
acted on. Containment that cannot identify its target is not containment, and
the honest failure is a visible orphan rather than a dead bystander.
"""
from __future__ import annotations
import logging
from typing import Any, Dict
from src import process_ownership
logger = logging.getLogger(__name__)
def reap_containment_grants() -> Dict[str, Any]:
"""Tear down or retire every grant a previous run left active.
Per grant: a verified live process is torn down through
:func:`src.containment.reap_record`; a grant whose process is gone is
dropped; a grant naming a pid that is now someone else's is dropped
*without a signal*, because the only thing left to do with it is stop
believing it. A grant that cannot be verified at all is **kept**, so the
orphan stays visible in ``active_grants()`` instead of being quietly
written off as handled.
"""
from src import containment
report: Dict[str, Any] = {
"seen": 0, "torn_down": 0, "already_gone": 0,
"foreign": 0, "unverifiable": 0, "failed": 0,
}
try:
records = containment.active_grants()
except Exception:
logger.warning("process_reaper: containment grant store unreadable", exc_info=True)
return report
for record in records:
report["seen"] += 1
grant_id = str(record.get("id") or "")
if record.get("external"):
# Nothing local ever ran, so there is nothing local to reap.
containment.forget(grant_id)
report["already_gone"] += 1
continue
if record.get("lifetime") == "background" and process_ownership.verify(
record.get("supervisor_pid"), record.get("supervisor_token"),
) == process_ownership.OWNED:
# Detached jobs deliberately survive a server restart. Their
# supervisor owns the wall clock and teardown, independently.
report["background_kept"] = report.get("background_kept", 0) + 1
continue
if record.get("lifetime") != "cleanup" and record.get("manager_pid") and process_ownership.verify(
record["manager_pid"], record.get("manager_token"),
) == process_ownership.OWNED:
report["manager_kept"] = report.get("manager_kept", 0) + 1
continue
verdict = process_ownership.verify_record(record)
if verdict == process_ownership.GONE:
if containment._group_present(record.get("pgid")):
# Leader death does not prove tree death. Without a surviving
# identity we cannot signal the group, so retain the evidence.
report["failed"] += 1
logger.error("process_reaper: grant %s leader is gone but group survives", grant_id)
continue
containment.forget(grant_id)
report["already_gone"] += 1
continue
if verdict == process_ownership.FOREIGN:
logger.warning(
"process_reaper: grant %s named pid %s, which now belongs to a "
"different process; dropping the record unsignalled",
grant_id, record.get("pid"),
)
containment.forget(grant_id)
report["foreign"] += 1
continue
if verdict == process_ownership.UNVERIFIABLE:
logger.error(
"process_reaper: grant %s (pid %s, owner %s) cannot be verified "
"via %s; leaving it active and unsignalled — this is a "
"containment failure, not a clean start",
grant_id, record.get("pid"), record.get("owner"),
process_ownership.inspection_mechanism(),
)
report["unverifiable"] += 1
continue
try:
outcome = containment.reap_record(record)
except Exception:
logger.warning("process_reaper: tearing down grant %s failed", grant_id, exc_info=True)
report["failed"] += 1
continue
if outcome.dead:
containment.forget(grant_id)
report["torn_down"] += 1
else:
logger.error(
"process_reaper: grant %s survived teardown; survivors=%s",
grant_id, list(outcome.survivors),
)
report["failed"] += 1
return report
def reap_bg_jobs() -> Dict[str, Any]:
"""Correct the identity of background jobs a previous run launched.
Deliberately kills nothing: a ``#!bg`` job is detached so that it outlives
the request *and* the server, and the store exists so its result is still
collected afterwards. The defect being closed is narrower — a record whose
pid has been reassigned will be signalled by the max-runtime reaper an hour
later, and that signal lands on whatever now holds the pid.
"""
from src import bg_jobs
try:
return bg_jobs.disown_unverified()
except Exception:
logger.warning("process_reaper: background job store unreadable", exc_info=True)
return {"seen": 0, "retired": 0, "kept": 0}
def reap_legacy_agent_tmux() -> Dict[str, Any]:
"""Retire this runtime's legacy agent shells; a name prefix is not ownership.
Match the original clean Bash launcher and this runtime's HOME marker on
every pane. Snapshot session/server identities and process start tokens
before teardown; ambiguous sessions remain visible and unsignalled.
"""
import os
import re
import shlex
import shutil
import subprocess
import uuid
from src import containment
from src.constants import DATA_DIR
report = {"seen": 0, "torn_down": 0, "unverifiable": 0, "failed": 0}
tmux = shutil.which("tmux")
if os.name == "nt" or not tmux:
return report
pattern = "#{session_id}\t#{session_name}\t#{session_created}\t#{pane_pid}\t#{pane_id}\t#{pid}\t#{pane_start_command}"
def snapshot():
result = subprocess.run([tmux, "list-panes", "-a", "-F", pattern],
capture_output=True, text=True, timeout=5)
if result.returncode:
if not result.stdout and any(message in result.stderr.lower() for message in ("no server", "no sessions", "error connecting")):
return {}
raise RuntimeError("tmux pane discovery failed")
sessions = {}
for line in result.stdout.splitlines():
fields = line.split("\t", 6)
if len(fields) != 7 or not fields[1].startswith("ody-agent-"):
continue
sessions.setdefault(fields[0], []).append(tuple(fields))
return {key: sorted(rows) for key, rows in sessions.items()}
def launcher_is_ours(command):
try:
argv = shlex.split(command)
except ValueError:
return False
if not argv or argv.pop(0) != "env":
return False
env = {}
while argv and "=" in argv[0]:
key, value = argv.pop(0).split("=", 1)
if key not in {"PATH", "VIRTUAL_ENV", "HOME", "TMPDIR", "TERM", "COLUMNS", "LINES"}:
return False
env[key] = value
return argv == ["/bin/bash", "--noprofile", "--norc"] and env.get("HOME") == DATA_DIR
try:
sessions = snapshot()
for session_id, panes in sessions.items():
report["seen"] += 1
if not re.fullmatch(r"\$\d+", session_id) or not all(launcher_is_ours(row[6]) for row in panes):
report["unverifiable"] += 1
continue
server_pid = int(panes[0][5])
server_token = process_ownership.start_token(server_pid)
roots = [int(row[3]) for row in panes]
table = process_ownership.process_table()
if not all(pid in table and table[pid].ppid == server_pid and shlex.split(table[pid].command) == [
"/bin/bash", "--noprofile", "--norc",
] for pid in roots):
# A stale pane PID can now name a bystander. Its parent and
# current launcher must still match the observed tmux server.
report["unverifiable"] += 1
continue
targets = process_ownership.descendants(roots, table=table)
identities = {pid: process_ownership.start_token(pid) for pid in targets}
if snapshot().get(session_id) != panes or process_ownership.verify(server_pid, server_token) != process_ownership.OWNED or any(
process_ownership.verify(pid, identities[pid]) != process_ownership.OWNED for pid in roots
):
report["unverifiable"] += 1
continue
# Persist every positively identified tree before touching it. A
# failed teardown then remains discoverable even if its pane dies.
tracked = []
for pid in reversed(targets):
if process_ownership.verify(pid, identities[pid]) != process_ownership.OWNED:
continue
spec = containment.ContainmentSpec(workspace=os.getcwd(), env={}, wall_clock_s=1,
required=frozenset())
grant = containment.ContainmentGrant(
id=uuid.uuid4().hex[:12], mechanism="process_group", workspace=spec.workspace,
enforced=frozenset(), degraded=(containment.FILESYSTEM, containment.PROCESS_TREE), unenforced_required=(),
owner=f"legacy-tmux:{session_id}", mode=containment.MODE_ENFORCING,
spec=spec, pid=pid, pgid=containment._pgid_of(pid),
)
containment._write_record(grant)
containment._update_record(grant.id, lifetime="cleanup", start_token=identities[pid])
receipt = containment._load_records().get(grant.id, {})
if receipt.get("start_token") != identities[pid] or receipt.get("lifetime") != "cleanup":
raise RuntimeError("legacy tmux cleanup receipt was not persisted")
tracked.append((grant, identities[pid]))
dead = True
for grant, token in tracked:
outcome = containment.release(grant, start_token=token, require_identity=True)
dead = dead and outcome.dead
remaining = snapshot().get(session_id)
if remaining and dead:
# Use the immutable tmux session id, not its reusable name.
if remaining != panes or process_ownership.verify(server_pid, server_token) != process_ownership.OWNED:
dead = False
else:
result = subprocess.run([tmux, "kill-session", "-t", session_id],
capture_output=True, timeout=5)
dead = result.returncode == 0 and session_id not in snapshot()
report["torn_down" if dead else "failed"] += 1
except Exception:
report["failed"] += 1
logger.warning("process_reaper: legacy agent tmux cleanup failed", exc_info=True)
if report["unverifiable"]:
logger.warning("process_reaper: left %s legacy tmux sessions without positive ownership", report["unverifiable"])
return report
def reap_orphans() -> Dict[str, Any]:
"""Run both reconciliations. Returns a report; raises nothing.
Blocking: a teardown escalates SIGTERM → grace → SIGKILL and waits for the
process to actually go. Call it off the event loop.
"""
report = {
"mechanism": process_ownership.inspection_mechanism(),
"grants": reap_containment_grants(),
"bg_jobs": reap_bg_jobs(),
"agent_tmux": reap_legacy_agent_tmux(),
}
if report["mechanism"] == process_ownership.MECHANISM_NONE:
logger.error(
"process_reaper: this host offers no process inspection; no orphan "
"from a previous run can be identified or reaped"
)
logger.info("process_reaper: startup reconciliation %s", report)
return report
async def reap_orphans_at_startup() -> Dict[str, Any]:
""":func:`reap_orphans` off the event loop, for an app startup task."""
import asyncio
return await asyncio.to_thread(reap_orphans)
+3 -7
View File
@@ -4,11 +4,11 @@ from __future__ import annotations
import json
import logging
import os
import re
from pathlib import Path
from src.constants import GENERATED_IMAGES_DIR
from src.path_confinement import confine
logger = logging.getLogger(__name__)
@@ -26,14 +26,10 @@ def _generated_image_path_for_cleanup(filename: str) -> Path | None:
name = Path(filename).name
if name != filename or name in {".", ".."}:
return None
root = Path(GENERATED_IMAGES_DIR).resolve()
path = (root / name).resolve()
try:
if os.path.commonpath([str(root), str(path)]) != str(root):
return None
except Exception:
return Path(confine(GENERATED_IMAGES_DIR, name, allow_root=False))
except (ValueError, OSError):
return None
return path
def _image_filename_from_url(url: str) -> str:
+57 -35
View File
@@ -34,7 +34,14 @@ from src.tool_capabilities import ToolRunSecurityContext, blocked_tool_result
from src.tool_approvals import ExactToolApproval
from src.tool_policy import ToolPolicy
from src.client_tool_contract import TUI_ROUTED_BRIDGE_TOOL_NAMES
from src.constants import MAX_OUTPUT_CHARS, MAX_READ_CHARS, MAX_DIFF_LINES, DATA_DIR
from src.constants import (
DATA_DIR,
MAX_DIFF_LINES,
MAX_OUTPUT_CHARS,
MAX_READ_CHARS,
WORKSPACE_MOUNT,
)
from src.path_confinement import canonical_root, confine, is_inside
from src.tool_utils import _truncate, get_mcp_manager
@@ -341,9 +348,24 @@ def _text_write_to_binary_artifact_result(content: str) -> tuple[str, Dict] | No
async def _route_tool_via_bridge(tool: str, content: str, session_id: Optional[str], client_runtime_context: Optional[Dict]):
import base64
from urllib.parse import urlparse
bridge = _client_bridge(client_runtime_context)
if bridge is None:
return tool, {"error": f"{tool}: TUI host bridge is not available", "exit_code": 1}
url = str(bridge.get("url") or "").strip()
parsed = urlparse(url)
sanitized_endpoint = f"{parsed.scheme}://{parsed.netloc}{parsed.path}" if parsed.scheme and parsed.netloc else "tui_bridge"
from src import containment
spec = containment.agent_spec(agent_cwd(), {}, int(_BRIDGE_TOOL_TIMEOUT_S))
grant = containment.declare_external_bridge(
spec,
owner=str(session_id or "tui_bridge"),
endpoint=sanitized_endpoint,
)
boundary = grant.to_dict()
boundary["executed"] = False
if tool == "bash":
from src.agent_tools.subprocess_tools import _host_shell_requires_detach, _host_shell_should_auto_poll
@@ -394,9 +416,13 @@ async def _route_tool_via_bridge(tool: str, content: str, session_id: Optional[s
"output": "host job still running; poll the returned job_id",
"exit_code": 0,
}
if isinstance(result, dict):
b = dict(boundary)
b["executed"] = (result.get("exit_code") == 0 or (isinstance(result.get("exit_code"), int) and not result.get("error")))
result["containment"] = b
return desc, result
if tool == "python":
return "python: (client)", await _bridge_post(
py_res = await _bridge_post(
bridge,
"/run",
{
@@ -407,6 +433,11 @@ async def _route_tool_via_bridge(tool: str, content: str, session_id: Optional[s
timeout_s=_BRIDGE_TOOL_TIMEOUT_S,
err_prefix="python",
)
if isinstance(py_res, dict):
b = dict(boundary)
b["executed"] = (py_res.get("exit_code") == 0 or (isinstance(py_res.get("exit_code"), int) and not py_res.get("error")))
py_res["containment"] = b
return "python: (client)", py_res
if tool == "grep":
stripped = content.strip()
try:
@@ -814,13 +845,7 @@ def _resolve_tool_path(raw_path: str) -> str:
)
for root in _tool_path_roots():
if resolved == root:
return resolved
try:
common = os.path.commonpath([resolved, root])
except ValueError:
continue
if common == root:
if is_inside(root, resolved):
return resolved
raise ValueError(
f"path '{raw_path}' is outside the allowed roots"
@@ -838,33 +863,27 @@ def _resolve_tool_path_in_workspace(workspace: str, raw_path: str) -> str:
"""
if raw_path is None or not str(raw_path).strip():
raise ValueError("path is required")
base = os.path.realpath(workspace)
base = canonical_root(workspace)
expanded = os.path.expanduser(str(raw_path).strip())
# `/workspace` is the stable user-facing agent root in tasks and docs.
# Native/manual installs may bind the request to another physical folder;
# resolve the alias inside that active workspace rather than rejecting it.
if expanded == "/workspace":
if expanded == WORKSPACE_MOUNT:
expanded = base
elif expanded.startswith("/workspace/"):
expanded = os.path.join(base, expanded.removeprefix("/workspace/"))
candidate = expanded if os.path.isabs(expanded) else os.path.join(base, expanded)
resolved = os.path.realpath(candidate)
elif expanded.startswith(WORKSPACE_MOUNT + "/"):
expanded = os.path.join(base, expanded.removeprefix(WORKSPACE_MOUNT + "/"))
try:
resolved = confine(base, expanded)
except (ValueError, OSError):
raise ValueError(f"path '{raw_path}' is outside the workspace ({workspace})")
# Confinement says "inside the root"; the deny list says "allowed". They
# are separate questions and this one stays here, with the policy that
# owns it.
if _is_sensitive_path(resolved):
raise ValueError(
f"path '{raw_path}' is inside a sensitive directory "
f"(e.g. .ssh, .gnupg) or matches a sensitive filename"
)
if resolved != base:
# normcase so containment holds on case-insensitive filesystems
# (Windows, default macOS): it lowercases on Windows and is a no-op on
# POSIX. commonpath raises ValueError across Windows drives (C: vs D:)
# or mixed abs/rel — both mean "outside", so the except rejects them.
nbase = os.path.normcase(base)
try:
if os.path.commonpath([os.path.normcase(resolved), nbase]) != nbase:
raise ValueError
except ValueError:
raise ValueError(f"path '{raw_path}' is outside the workspace ({workspace})")
return resolved
@@ -1192,6 +1211,10 @@ def _split_bg_marker(content: str):
return False, content
def _agent_subprocess_env() -> dict:
return {**os.environ, "TERM": "xterm-256color", "COLUMNS": "120", "LINES": "40", "HOME": _AGENT_WORKDIR}
async def _direct_fallback(
tool: str,
content: str,
@@ -1202,13 +1225,7 @@ async def _direct_fallback(
disabled_tools: Optional[set] = None,
tool_policy: Optional[ToolPolicy] = None,
) -> Optional[Dict]:
_subproc_env = {
**os.environ,
"TERM": "xterm-256color",
"COLUMNS": "120",
"LINES": "40",
"HOME": _AGENT_WORKDIR,
}
_subproc_env = _agent_subprocess_env()
try:
ctx = {
@@ -1668,7 +1685,11 @@ async def _execute_tool_block_impl(
if _is_bg and _bg_cmd:
from src import bg_jobs
mark_dispatch()
rec = bg_jobs.launch(_bg_cmd, session_id=session_id, cwd=agent_cwd())
from src import containment
try:
rec = bg_jobs.launch(_bg_cmd, session_id=session_id, cwd=agent_cwd(), env=_agent_subprocess_env())
except containment.ContainmentUnavailable as exc:
return "bash (background): containment unavailable", containment.unavailable_tool_result(exc, tool="bash")
# Only this server launch may seal detached-job authority; a
# handler/bridge output carrying a job id is not a grant source.
save_background_authority(rec["id"], active_request_authority())
@@ -1678,7 +1699,7 @@ async def _execute_tool_block_impl(
"output": (
f"Started background job `{rec['id']}`. It is running detached; "
f"do NOT wait for it or poll it. You will be automatically re-invoked "
f"with its full output when it finishes. Continue with other work, or "
f"with its captured output and any capture limit when it finishes. Continue with other work, or "
f"end your turn now and resume when the result arrives. If the user "
f"later asks to check progress or stop it, call the manage_bg_jobs "
f"tool yourself (output or kill); do not tell them to run a tool "
@@ -1686,6 +1707,7 @@ async def _execute_tool_block_impl(
),
"exit_code": 0,
"bg_job_id": rec["id"],
"containment": rec.get("containment"),
}
logger.info(f"Tool executed: {desc} -> bg job {rec['id']}")
return desc, result
+203 -30
View File
@@ -13,6 +13,7 @@ import asyncio
import contextlib
import json
import logging
import os
import re
from typing import Any, Dict, List, Optional
@@ -1024,6 +1025,189 @@ async def do_list_served_models(content: str, owner: Optional[str] = None) -> Di
return {"output": "\n".join(lines), "tasks": merged, "exit_code": 0}
# How long a tmux query may take before the stop gives up on identifying the
# session's processes and says so. The kill itself does not depend on it.
_TMUX_QUERY_TIMEOUT_S = 5
# Grace between SIGTERM and SIGKILL for a model server that ignored SIGHUP.
_SWEEP_GRACE_S = 2.0
_SWEEP_POLL_S = 0.05
async def _capture_session_processes(session_id: str) -> tuple[List[Dict[str, Any]], str]:
"""Snapshot the processes belonging to a local tmux session.
This is what gives the survivor sweep an *ownership* record rather than a
resemblance. tmux knows which pane hosts the session, the pane pid's
descendants are the processes that session started, and a start token taken
now is what lets the sweep prove, after the kill, that a pid it is about to
signal is still one of them.
Returns ``(records, note)``. An empty list with a note is the honest
outcome when the session cannot be enumerated — the note reaches the tool
result, because "I found no survivors" and "I could not look" are different
answers and the sweep used to give the first for both (ODY-94).
"""
from src import process_ownership
try:
proc = await asyncio.create_subprocess_exec(
"tmux", "list-panes", "-a", "-F", "#{session_name} #{pane_pid}",
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
)
try:
stdout, _stderr = await asyncio.wait_for(
proc.communicate(), timeout=_TMUX_QUERY_TIMEOUT_S
)
except asyncio.TimeoutError:
with contextlib.suppress(Exception):
proc.kill()
await proc.communicate()
return [], "; could not identify the session's processes (tmux timed out)"
except (OSError, FileNotFoundError) as exc:
return [], f"; could not identify the session's processes (tmux unavailable: {exc})"
if proc.returncode not in (0, None):
return [], "; could not identify the session's processes (tmux listed no panes)"
pane_pids: List[int] = []
for line in (stdout or b"").decode("utf-8", errors="replace").splitlines():
name, _, pid_text = line.strip().rpartition(" ")
if name == session_id and pid_text.isdigit():
pane_pids.append(int(pid_text))
if not pane_pids:
# The session is already gone, so nothing links a survivor to it. Said
# out loud rather than reported as a clean sweep.
return [], "; the session had no live pane, so its processes could not be identified"
def _snapshot() -> tuple[List[Dict[str, Any]], str]:
try:
table = process_ownership.process_table()
except process_ownership.InspectionUnavailable as exc:
return [], f"; could not identify the session's processes ({exc})"
own = {os.getpid(), os.getppid()}
records = []
for pid in process_ownership.descendants(pane_pids, table=table):
if pid in own:
continue
info = table.get(pid)
records.append({
"pid": pid,
"start_token": process_ownership.capture(pid)["start_token"],
"command": info.command if info else "",
})
return records, ""
return await asyncio.to_thread(_snapshot)
def _signal_owned(pid: int, token: Optional[str], sig: int) -> bool:
"""Signal ``pid`` only while it still verifies as the process we captured.
Re-verified immediately before every signal, including the escalation: the
gap between SIGTERM and SIGKILL is exactly long enough for the pid to be
freed and reissued, and a SIGKILL aimed at whatever landed in the slot is
the bug this sweep exists to stop committing.
"""
from src import process_ownership
if process_ownership.verify(pid, token) != process_ownership.OWNED:
return False
try:
os.kill(pid, sig)
return True
except (ProcessLookupError, PermissionError, OSError):
return False
def _sweep_session_survivors(
owned: List[Dict[str, Any]], tracked_cmd: str, capture_note: str,
) -> str:
"""Terminate the captured processes that outlived the tmux kill.
Blocking; call it off the event loop. Returns the note to append to the
tool result — the sweep's outcome is part of whether the stop worked, and
silence here is what let a half-stopped server read as stopped.
Only captured pids are signalled. A process that merely matches
``tracked_cmd`` is reported and left alone: the Cookbook composed that
command line, so an identical one may well be a server the user started by
hand, and killing it because it resembles ours is indistinguishable from
killing ours. Naming it lets whoever is reading decide.
"""
import signal as _signal
import time as _time
from src import process_ownership
if capture_note:
return capture_note
live = [rec for rec in owned
if process_ownership.verify(rec["pid"], rec["start_token"]) == process_ownership.OWNED]
killed: List[int] = []
survivors: List[int] = []
for rec in live:
pid, token = rec["pid"], rec["start_token"]
if not _signal_owned(pid, token, _signal.SIGTERM):
continue
deadline = _time.monotonic() + _SWEEP_GRACE_S
while _time.monotonic() < deadline:
if process_ownership.verify(pid, token) != process_ownership.OWNED:
break
_time.sleep(_SWEEP_POLL_S)
if process_ownership.verify(pid, token) == process_ownership.OWNED:
_signal_owned(pid, token, _signal.SIGKILL)
deadline = _time.monotonic() + 1.0
while _time.monotonic() < deadline:
if process_ownership.verify(pid, token) != process_ownership.OWNED:
break
_time.sleep(_SWEEP_POLL_S)
if process_ownership.verify(pid, token) == process_ownership.OWNED:
survivors.append(pid)
else:
killed.append(pid)
note = ""
if killed:
note += f"; killed {len(killed)} surviving process(es) owned by the session"
if survivors:
note += (
f"; {len(survivors)} process(es) survived SIGKILL and are still "
f"running (pid {', '.join(str(pid) for pid in survivors)})"
)
note += _unowned_match_note(tracked_cmd, {rec["pid"] for rec in owned})
return note
def _unowned_match_note(tracked_cmd: str, owned_pids: set) -> str:
"""Report, without signalling, processes that look like the tracked command.
The old sweep killed these. It could not tell them apart from the server it
started, and neither can this — so it names them instead. Reporting keeps
the information the old behaviour acted on while giving up the one thing it
was never entitled to do.
"""
from src import process_ownership
if not tracked_cmd:
return ""
try:
table = process_ownership.process_table()
except process_ownership.InspectionUnavailable:
return "; could not check for unowned processes matching the command"
strangers = sorted(
pid for pid, info in table.items()
if info.command == tracked_cmd and pid not in owned_pids and pid != os.getpid()
)
if not strangers:
return ""
return (
f"; note: {len(strangers)} other process(es) match this server's command "
f"line (pid {', '.join(str(pid) for pid in strangers)}) — not signalled, "
f"because nothing identifies them as started by this session"
)
async def _cookbook_kill_session(session_id: str, *, remote_host: str = "",
ssh_port: str = "", verb: str = "Stopped") -> Dict:
"""Kill a cookbook tmux session — remote-aware — AND mark the task
@@ -1077,6 +1261,16 @@ async def _cookbook_kill_session(session_id: str, *, remote_host: str = "",
cmd = f"tmux kill-session -t {shlex.quote(session_id)}"
target_label = session_id
# Capture what this session owns BEFORE the kill. Once tmux tears the
# session down the pane is gone, and with it the only evidence linking a
# surviving model server to the session that started it. A sweep that looks
# afterwards has nothing left but the command line, which identifies a
# *kind* of process and not one we started.
owned: List[Dict[str, Any]] = []
owned_note = ""
if not remote and isinstance(matched, dict):
owned, owned_note = await _capture_session_processes(session_id)
try:
if remote:
async with httpx.AsyncClient(timeout=15) as client:
@@ -1117,37 +1311,16 @@ async def _cookbook_kill_session(session_id: str, *, remote_host: str = "",
if kill_failed and not already_gone:
return {"error": f"Failed to {verb.lower()} {target_label}: {kill_err or 'kill-session returned non-zero'}", "exit_code": 1}
# Some model servers survive the tmux session's SIGHUP. For local
# tracked tasks only, terminate processes whose full command line
# exactly matches the command saved by the Cookbook launcher.
# Some model servers survive the tmux session's SIGHUP. Terminate the
# ones this session actually owns — captured above, each verified by
# identity at signal time — and report, without signalling, anything
# that merely looks like the tracked command.
sweep_note = ""
if not remote and isinstance(matched, dict):
import os
import signal
tracked_cmd = str((matched.get("payload") or {}).get("_cmd") or "").strip()
matched_pids: list[int] = []
# No procfs means no way to match a survivor by its command line.
# The tmux kill above already stopped the session, so skip the
# sweep instead of failing a stop that worked.
if tracked_cmd and platform_compat.has_procfs():
proc_root = platform_compat.PROC_ROOT
for pid_name in os.listdir(proc_root):
if not pid_name.isdigit() or int(pid_name) == os.getpid():
continue
try:
raw = (proc_root / pid_name / "cmdline").read_bytes()
process_cmd = raw.replace(b"\x00", b" ").decode("utf-8", errors="replace").strip()
except (OSError, PermissionError):
continue
if process_cmd == tracked_cmd:
matched_pids.append(int(pid_name))
with contextlib.suppress(ProcessLookupError, PermissionError):
os.kill(int(pid_name), signal.SIGTERM)
if matched_pids:
await asyncio.sleep(0.5)
for pid in matched_pids:
with contextlib.suppress(ProcessLookupError, PermissionError):
os.kill(pid, 0)
os.kill(pid, signal.SIGKILL)
sweep_note = await asyncio.to_thread(
_sweep_session_survivors, owned, tracked_cmd, owned_note,
)
# Update state: mark stopped (so the UI + list reflect reality).
if matched is not None:
@@ -1160,7 +1333,7 @@ async def _cookbook_kill_session(session_id: str, *, remote_host: str = "",
logger.debug(f"failed to mark {session_id} stopped in state: {e}")
suffix = " (was already gone)" if already_gone else ""
return {"output": f"{verb} {target_label}{suffix}", "exit_code": 0}
return {"output": f"{verb} {target_label}{suffix}{sweep_note}", "exit_code": 0}
except Exception as e:
return {"error": str(e), "exit_code": 1}
+3 -12
View File
@@ -13,6 +13,7 @@ from datetime import datetime, timedelta
from typing import Dict, Any, Optional
from fastapi import HTTPException, UploadFile
from src.path_confinement import is_inside
from src.upload_limits import format_byte_limit, get_chat_upload_max_bytes
@@ -256,12 +257,7 @@ class UploadHandler:
def inside_base_dir(self, path: str) -> bool:
"""Check if path is inside base directory"""
base = os.path.realpath(self.base_dir)
p = os.path.realpath(path)
try:
return os.path.commonpath([base, p]) == base
except Exception:
return False
return is_inside(self.base_dir, path)
def get_upload_dir(self):
"""Get date-based upload directory"""
@@ -684,12 +680,7 @@ class UploadHandler:
def _inside_upload_dir(self, path: str) -> bool:
"""Check if path is inside the upload directory."""
base = os.path.normcase(os.path.realpath(self.upload_dir))
p = os.path.normcase(os.path.realpath(path))
try:
return os.path.commonpath([base, p]) == base
except Exception:
return False
return is_inside(self.upload_dir, path)
def _atomic_write_json(
self,
+47
View File
@@ -0,0 +1,47 @@
"""Captured spawns exercise the production runner without signalling fake PIDs."""
import asyncio
import json
import os
from types import SimpleNamespace
from src import containment
def capture_owned_spawn(monkeypatch, tmp_path):
captured = {}
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_REPORT_ONLY)
monkeypatch.setattr(containment, "_store_path", lambda: tmp_path / "grants.json")
monkeypatch.setattr(containment, "_pgid_of", lambda pid: pid)
async def fake_exec(*argv, **kwargs):
captured.update(argv=argv, kwargs=kwargs, command=argv[-1])
if "--info-fd" in argv:
fd = int(argv[argv.index("--info-fd") + 1])
os.write(fd, json.dumps({"child-pid": 99999998}).encode())
stdout = asyncio.StreamReader()
if "ody-boundary" in argv:
stdout.feed_data((argv[argv.index("ody-boundary") + 1] + "\n").encode())
stdout.feed_data(b"ok")
stdout.feed_eof()
stderr = asyncio.StreamReader()
stderr.feed_eof()
async def wait():
return 0
async def drain():
return None
writer = SimpleNamespace(write=lambda data: None, drain=drain,
close=lambda: captured.update(stdin_closed=True))
return SimpleNamespace(pid=99999999, stdout=stdout, stderr=stderr,
stdin=writer, returncode=0, wait=wait)
async def release(*args, **kwargs):
return containment.ReleaseOutcome(dead=True, escalated=False)
monkeypatch.setattr(asyncio, "create_subprocess_exec", fake_exec)
real_capture = containment.process_ownership.capture
monkeypatch.setattr(containment.process_ownership, "capture", lambda pid:
{"pid": pid, "start_token": "captured-fake"} if pid in (99999998, 99999999) else real_capture(pid))
if hasattr(containment.os, "pidfd_open"):
monkeypatch.delattr(containment.os, "pidfd_open")
monkeypatch.setattr(containment, "_release_awaited", release)
return captured
+9 -184
View File
@@ -1,74 +1,4 @@
import asyncio
from types import SimpleNamespace
def test_new_tmux_session_forwards_runtime_python_environment(monkeypatch):
from src.agent_tools import subprocess_tools
calls = []
checks = 0
async def fake_has_session(_name):
nonlocal checks
checks += 1
return checks > 1
async def fake_run_exec(*args, **kwargs):
calls.append(args)
if args[:2] == ("tmux", "has-session"):
return "", "", 0
return "", "", 0
monkeypatch.setattr(subprocess_tools, "_tmux_has_session", fake_has_session)
monkeypatch.setattr(subprocess_tools, "_run_exec", fake_run_exec)
asyncio.run(subprocess_tools._ensure_tmux_session(
"ody-test",
"/workspace",
{
"PATH": "/opt/ody/bin:/usr/bin",
"VIRTUAL_ENV": "/opt/ody",
"HOME": "/workspace",
"TMPDIR": "/workspace/.tmp",
"SECRET": "must-not-forward",
},
))
new_session = next(call for call in calls if call[:2] == ("tmux", "new-session"))
assert "PATH=/opt/ody/bin:/usr/bin" in new_session
assert "VIRTUAL_ENV=/opt/ody" in new_session
assert "HOME=/workspace" in new_session
assert "TMPDIR=/workspace/.tmp" in new_session
assert not any("SECRET=" in arg for arg in new_session)
def test_reused_tmux_session_refreshes_runtime_python_environment(monkeypatch):
from src.agent_tools import subprocess_tools
sent = []
async def fake_has_session(_name):
return True
async def fake_send_line(name, line):
sent.append((name, line))
async def fake_run_exec(*_args, **_kwargs):
return "", "", 0
monkeypatch.setattr(subprocess_tools, "_tmux_has_session", fake_has_session)
monkeypatch.setattr(subprocess_tools, "_tmux_send_line", fake_send_line)
monkeypatch.setattr(subprocess_tools, "_run_exec", fake_run_exec)
asyncio.run(subprocess_tools._ensure_tmux_session(
"ody-existing",
"/workspace",
{"PATH": "/path with spaces/bin:/usr/bin", "VIRTUAL_ENV": "/path with spaces"},
))
assert sent == [(
"ody-existing",
"export PATH='/path with spaces/bin:/usr/bin' VIRTUAL_ENV='/path with spaces'",
)]
def test_workspace_alias_rewrite_does_not_duplicate_absolute_host_path():
@@ -97,116 +27,22 @@ def test_workspace_namespace_preserves_literal_paths_inside_scripts(tmp_path):
assert (Path(tmp_path) / "result.txt").read_text() == "ok"
def test_tmux_bash_runs_tool_command_with_closed_stdin(monkeypatch, tmp_path):
from src.agent_tools import subprocess_tools
sent = []
async def fake_ensure(*_args, **_kwargs):
return None
async def fake_send(_name, line):
sent.append(line)
async def fake_capture(_name):
start = next(line for line in sent if "__ODYSSEUS_CMD_START_" in line)
start = start.split("\\n")[1]
end_line = next(line for line in sent if "__ODYSSEUS_CMD_END_" in line)
end = end_line.split("\\n")[1].split("%s")[0]
return f"{start}\ngot-eof\n{end}0\n"
monkeypatch.setattr(subprocess_tools, "_ensure_tmux_session", fake_ensure)
monkeypatch.setattr(subprocess_tools, "_tmux_send_line", fake_send)
monkeypatch.setattr(subprocess_tools, "_tmux_capture", fake_capture)
monkeypatch.setattr(subprocess_tools.time, "time", lambda: 0.123456)
output, _stderr, rc, timed_out = asyncio.run(
subprocess_tools._run_tmux_bash(
"if read answer; then echo unexpected; else echo got-eof; fi",
session_id="test",
cwd=str(tmp_path),
env={},
timeout=2,
)
)
assert output == "got-eof"
assert rc == 0
assert timed_out is False
assert any("/bin/bash -lc" in line and "</dev/null" in line for line in sent)
def test_tmux_bash_timeout_destroys_session_process_tree(monkeypatch, tmp_path):
from src.agent_tools import subprocess_tools
exec_calls = []
sent = []
clock = [-2.0]
async def fake_ensure(*_args, **_kwargs):
return None
async def fake_send(name, line):
sent.append((name, line))
async def fake_capture(_name):
return "still running"
async def fake_run_exec(*args, **kwargs):
exec_calls.append(args)
return "", "", 0
async def fake_sleep(_seconds):
return None
monkeypatch.setattr(subprocess_tools, "_ensure_tmux_session", fake_ensure)
monkeypatch.setattr(subprocess_tools, "_tmux_send_line", fake_send)
monkeypatch.setattr(subprocess_tools, "_tmux_capture", fake_capture)
monkeypatch.setattr(subprocess_tools, "_run_exec", fake_run_exec)
monkeypatch.setattr(subprocess_tools.asyncio, "sleep", fake_sleep)
def fake_time():
clock[0] += 2.0
return clock[0]
monkeypatch.setattr(subprocess_tools.time, "time", fake_time)
_output, _stderr, rc, timed_out = asyncio.run(
subprocess_tools._run_tmux_bash(
"grep -r needle /large/tree",
session_id="timeout-test",
cwd=str(tmp_path),
env={},
timeout=1,
)
)
assert rc == 124 and timed_out is True
assert any(call[:3] == ("tmux", "send-keys", "-t") for call in exec_calls)
assert any(call[:3] == ("tmux", "kill-session", "-t") for call in exec_calls)
def test_direct_bash_subprocess_has_closed_stdin(monkeypatch, tmp_path):
from src.agent_tools import subprocess_tools
from src import tool_execution
captured = {}
sentinel = SimpleNamespace(pid=12345)
async def fake_create(command, **kwargs):
captured.update(kwargs)
return sentinel
async def fake_stream(proc, **_kwargs):
assert proc is sentinel
return "ok", "", 0, False
monkeypatch.setattr(asyncio, "create_subprocess_shell", fake_create)
monkeypatch.setattr(subprocess_tools, "_run_subprocess_streaming", fake_stream)
from tests.containment_helpers import capture_owned_spawn
captured = capture_owned_spawn(monkeypatch, tmp_path)
monkeypatch.setattr(tool_execution, "agent_cwd", lambda: str(tmp_path))
result = asyncio.run(subprocess_tools.BashTool().execute("echo ok", {}))
assert result["exit_code"] == 0
assert captured["stdin"] is asyncio.subprocess.DEVNULL
if "ody-boundary" in captured["argv"]:
assert captured["kwargs"]["stdin"] is asyncio.subprocess.PIPE
assert captured["stdin_closed"] is True
else:
assert captured["kwargs"]["stdin"] is asyncio.subprocess.DEVNULL
assert not (tmp_path / ".tmp").exists()
@@ -258,19 +94,8 @@ def test_bash_allows_unicode_ffmpeg_drawtext_with_explicit_fontfile(monkeypatch,
from src.agent_tools import subprocess_tools
from src import tool_execution
captured = {}
sentinel = SimpleNamespace(pid=12345)
async def fake_create(command, **kwargs):
captured["command"] = command
return sentinel
async def fake_stream(proc, **_kwargs):
assert proc is sentinel
return "ok", "", 0, False
monkeypatch.setattr(asyncio, "create_subprocess_shell", fake_create)
monkeypatch.setattr(subprocess_tools, "_run_subprocess_streaming", fake_stream)
from tests.containment_helpers import capture_owned_spawn
captured = capture_owned_spawn(monkeypatch, tmp_path)
monkeypatch.setattr(tool_execution, "agent_cwd", lambda: str(tmp_path))
command = (
+27 -31
View File
@@ -6,6 +6,8 @@ import pytest
from types import SimpleNamespace
from src.agent_tools import subprocess_tools
from src import containment
from tests.containment_helpers import capture_owned_spawn
@pytest.mark.asyncio
@@ -84,33 +86,26 @@ async def test_windows_bash_applies_the_subprocess_env(monkeypatch):
@pytest.mark.asyncio
async def test_windows_bash_tool_passes_ctx_env_through_to_the_child(monkeypatch):
captured = {}
async def test_windows_bash_tool_passes_ctx_env_through_to_the_child(monkeypatch, tmp_path):
captured = capture_owned_spawn(monkeypatch, tmp_path)
env = {"PATH": r"C:\Odysseus\venv\Scripts", "VIRTUAL_ENV": r"C:\Odysseus\venv"}
monkeypatch.setattr(subprocess_tools, "IS_WINDOWS", True)
monkeypatch.setattr(
subprocess_tools, "find_bash", lambda: r"C:\Program Files\Git\bin\bash.exe"
)
monkeypatch.setattr("src.tool_execution.agent_cwd", lambda: r"D:\Workspaces\Project")
async def fake_exec(*argv, **kwargs):
captured["argv"] = argv
captured["kwargs"] = kwargs
return SimpleNamespace(pid=4242)
async def fake_stream(_process, **_kwargs):
return "ok", "", 0, False
monkeypatch.setattr(subprocess_tools.asyncio, "create_subprocess_exec", fake_exec)
monkeypatch.setattr(subprocess_tools, "_run_subprocess_streaming", fake_stream)
monkeypatch.setattr(containment, "IS_WINDOWS", True)
monkeypatch.setattr(containment, "find_bash", lambda: r"C:\Program Files\Git\bin\bash.exe")
monkeypatch.setattr("src.tool_execution.agent_cwd", lambda: str(tmp_path))
result = await subprocess_tools.BashTool().execute(
"pwd",
{"subproc_env": env, "session_id": "chat-1"},
)
assert result == {"output": "ok", "exit_code": 0}
assert result["output"] == "ok"
assert result["exit_code"] == 0
assert "containment" in result
assert captured["kwargs"]["env"] == env
assert captured["kwargs"]["stdout"] == asyncio.subprocess.PIPE
assert captured["kwargs"]["stderr"] == asyncio.subprocess.PIPE
@@ -132,9 +127,13 @@ async def test_windows_bash_without_git_bash_fails_clearly(monkeypatch):
@pytest.mark.asyncio
async def test_bash_tool_returns_install_hint_when_git_bash_is_missing(monkeypatch):
async def test_bash_tool_returns_install_hint_when_git_bash_is_missing(monkeypatch, tmp_path):
capture_owned_spawn(monkeypatch, tmp_path)
monkeypatch.setattr(subprocess_tools, "IS_WINDOWS", True)
monkeypatch.setattr(subprocess_tools, "find_bash", lambda: None)
monkeypatch.setattr(containment, "IS_WINDOWS", True)
monkeypatch.setattr(containment, "find_bash", lambda: None)
monkeypatch.setattr("src.tool_execution.agent_cwd", lambda: str(tmp_path))
result = await subprocess_tools.BashTool().execute(
"pwd",
@@ -146,11 +145,13 @@ async def test_bash_tool_returns_install_hint_when_git_bash_is_missing(monkeypat
@pytest.mark.asyncio
async def test_windows_bash_does_not_use_a_stray_tmux_executable(monkeypatch):
captured = {}
workspace = r"D:\Workspaces\Project with spaces"
async def test_windows_bash_does_not_use_a_stray_tmux_executable(monkeypatch, tmp_path):
captured = capture_owned_spawn(monkeypatch, tmp_path)
workspace = str(tmp_path)
monkeypatch.setattr(subprocess_tools, "IS_WINDOWS", True)
monkeypatch.setattr(containment, "IS_WINDOWS", True)
monkeypatch.setattr(containment, "find_bash", lambda: r"C:\Program Files\Git\bin\bash.exe")
monkeypatch.setattr(
subprocess_tools.shutil,
"which",
@@ -161,24 +162,19 @@ async def test_windows_bash_does_not_use_a_stray_tmux_executable(monkeypatch):
async def fail_tmux(*_args, **_kwargs):
pytest.fail("native Windows must not enter the POSIX tmux path")
async def fake_create(command, **kwargs):
captured["command"] = command
captured["kwargs"] = kwargs
return SimpleNamespace(pid=12345)
async def fake_stream(_process, **_kwargs):
return "ok", "", 0, False
monkeypatch.setattr(subprocess_tools, "_run_tmux_bash", fail_tmux)
monkeypatch.setattr(subprocess_tools, "_create_bash_subprocess", fake_create)
monkeypatch.setattr(subprocess_tools, "_run_subprocess_streaming", fake_stream)
monkeypatch.setattr(subprocess_tools.asyncio, "create_subprocess_shell", fail_tmux)
result = await subprocess_tools.BashTool().execute(
"pwd",
{"subproc_env": {}, "session_id": "chat-1"},
)
assert result == {"output": "ok", "exit_code": 0}
assert result["output"] == "ok"
assert result["exit_code"] == 0
# Every bash result now carries the execution boundary it actually got.
# Asserting dict equality here would make that field impossible to add
# without touching a test about tmux, so the shape is asserted instead.
assert result["containment"]["network"] == "inherit"
assert captured["command"] == "pwd"
assert captured["kwargs"]["cwd"] == workspace
+134
View File
@@ -0,0 +1,134 @@
"""Native Bash does not resurrect tmux; legacy cleanup requires identity."""
import asyncio
from types import SimpleNamespace
import pytest
from src import containment, process_ownership, process_reaper, tool_execution
from src.agent_tools import subprocess_tools
from src.constants import DATA_DIR
from tests.containment_helpers import capture_owned_spawn
async def test_a_chat_session_always_uses_the_owned_runner(monkeypatch, tmp_path):
captured = capture_owned_spawn(monkeypatch, tmp_path)
monkeypatch.setattr(tool_execution, "agent_cwd", lambda: str(tmp_path))
original = subprocess_tools.shutil.which
monkeypatch.setattr(subprocess_tools.shutil, "which", lambda name: "/fake/tmux" if name == "tmux" else original(name))
async def forbidden(*args, **kwargs):
pytest.fail("native Bash resurrected a persistent tmux shell")
monkeypatch.setattr(subprocess_tools.asyncio, "create_subprocess_shell", forbidden)
result = await subprocess_tools.BashTool().execute("printf ok", {"session_id": "same-chat"})
assert result["output"] == "ok"
assert result["teardown"]["dead"] is True
assert "tmux_session" not in result
assert captured["kwargs"].get("start_new_session") is True
@pytest.fixture
def legacy(monkeypatch, tmp_path):
import shlex
import subprocess
import shutil
monkeypatch.setattr(containment, "_store_path", lambda: tmp_path / "grants.json")
monkeypatch.setattr(shutil, "which", lambda name: "/usr/bin/tmux" if name == "tmux" else None)
launcher = f"env HOME={shlex.quote(DATA_DIR)} TERM=xterm-256color /bin/bash --noprofile --norc"
row = f"$8\tody-agent-chat\t1234\t4200\t%9\t4100\t{launcher}\n"
state = {"rows": row, "calls": [], "released": []}
def run(argv, **kwargs):
state["calls"].append(argv)
if argv[1] == "list-panes":
return SimpleNamespace(returncode=0, stdout=state["rows"], stderr="")
if argv[1] == "kill-session":
state["rows"] = ""
return SimpleNamespace(returncode=0, stdout="", stderr="")
monkeypatch.setattr(subprocess, "run", run)
monkeypatch.setattr(process_ownership, "process_table", lambda: {
4200: process_ownership.ProcessInfo(4200, 4100, "/bin/bash --noprofile --norc"),
4201: process_ownership.ProcessInfo(4201, 4200, "sleep 60"),
})
monkeypatch.setattr(process_ownership, "start_token", lambda pid: f"token:{pid}")
monkeypatch.setattr(process_ownership, "verify", lambda pid, token: process_ownership.OWNED)
monkeypatch.setattr(containment, "_pgid_of", lambda pid: pid)
def release(grant, **kwargs):
state["released"].append((grant.pid, kwargs))
return containment.ReleaseOutcome(dead=True, escalated=False)
monkeypatch.setattr(containment, "release", release)
return state
def test_legacy_cleanup_checks_home_and_identity_and_kills_children_first(legacy):
report = process_reaper.reap_legacy_agent_tmux()
assert report["torn_down"] == 1
assert [pid for pid, _ in legacy["released"]] == [4201, 4200]
assert all(options["require_identity"] for _, options in legacy["released"])
assert legacy["calls"][-2][1:] == ["kill-session", "-t", "$8"]
def test_a_name_prefix_alone_never_authorizes_cleanup(legacy):
legacy["rows"] = "$8\tody-agent-chat\t1234\t4200\t%9\t4100\t/bin/bash\n"
report = process_reaper.reap_legacy_agent_tmux()
assert report["unverifiable"] == 1
assert legacy["released"] == []
assert all(call[1] != "kill-session" for call in legacy["calls"])
def test_a_recycled_pane_pid_is_never_signalled(legacy, monkeypatch):
monkeypatch.setattr(process_ownership, "verify", lambda pid, token: process_ownership.FOREIGN if pid == 4200 else process_ownership.OWNED)
report = process_reaper.reap_legacy_agent_tmux()
assert report["unverifiable"] == 1
assert legacy["released"] == []
def test_a_stale_pane_pid_pointing_at_another_parent_is_not_signalled(legacy, monkeypatch):
monkeypatch.setattr(process_ownership, "process_table", lambda: {
4200: process_ownership.ProcessInfo(4200, 9999, "/bin/bash --noprofile --norc"),
})
assert process_reaper.reap_legacy_agent_tmux()["unverifiable"] == 1
assert legacy["released"] == []
def test_a_session_changed_during_discovery_is_never_killed(legacy, monkeypatch):
original = process_ownership.process_table
def table():
legacy["rows"] = legacy["rows"].replace("1234", "5678")
return original()
monkeypatch.setattr(process_ownership, "process_table", table)
report = process_reaper.reap_legacy_agent_tmux()
assert report["unverifiable"] == 1
assert legacy["released"] == []
def test_startup_reaper_does_not_kill_a_current_runtime_grant(tmp_path, monkeypatch):
monkeypatch.setattr(containment, "_store_path", lambda: tmp_path / "grants.json")
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_REPORT_ONLY)
grant = containment.acquire(containment.agent_spec(str(tmp_path), {}, 1), owner="active-chat")
monkeypatch.setattr(containment, "reap_record", lambda record: pytest.fail("startup killed current execution"))
assert process_reaper.reap_containment_grants()["manager_kept"] == 1
containment.release(grant)
def test_legacy_cleanup_against_a_private_real_tmux_server(tmp_path, monkeypatch):
import os
import shlex
import shutil
import subprocess
real_tmux = shutil.which("tmux")
if os.name == "nt" or not real_tmux:
pytest.skip("requires POSIX tmux")
socket = str(tmp_path / "tmux.sock")
wrapper = tmp_path / "tmux"
wrapper.write_text(f"#!/bin/sh\nexec {shlex.quote(real_tmux)} -S {shlex.quote(socket)} \"$@\"\n")
wrapper.chmod(0o700)
subprocess.run([real_tmux, "-S", socket, "-f", "/dev/null", "new-session", "-d", "-s", "ody-agent-real",
"env", f"HOME={DATA_DIR}", "TERM=xterm-256color", "/bin/bash", "--noprofile", "--norc"], check=True)
original = shutil.which
monkeypatch.setattr(shutil, "which", lambda name: str(wrapper) if name == "tmux" else original(name))
monkeypatch.setattr(containment, "_store_path", lambda: tmp_path / "grants.json")
try:
report = process_reaper.reap_legacy_agent_tmux()
assert report["torn_down"] == 1, report
assert subprocess.run([real_tmux, "-S", socket, "has-session", "-t", "ody-agent-real"], capture_output=True).returncode != 0
assert containment.active_grants() == []
finally:
subprocess.run([real_tmux, "-S", socket, "kill-server"], capture_output=True)
+157
View File
@@ -0,0 +1,157 @@
"""Detached Bash uses the same boundary; restart and kill retain ownership."""
import asyncio
import os
import time
from collections import namedtuple
import pytest
from src import bg_jobs, containment, process_ownership, process_reaper, tool_execution
from src.tool_execution import NO_TOOL_SECURITY_CONTEXT
from tests.runtime_evidence_helpers import server_authorized_executor
@pytest.fixture
def jobs(tmp_path, monkeypatch):
monkeypatch.setattr(bg_jobs, "_JOBS_DIR", tmp_path / "jobs")
monkeypatch.setattr(bg_jobs, "_STORE", tmp_path / "jobs.json")
monkeypatch.setattr(containment, "_store_path", lambda: tmp_path / "grants.json")
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_REPORT_ONLY)
monkeypatch.setattr(containment, "MECHANISMS", tuple(m for m in containment.MECHANISMS if m.name == "process_group"))
monkeypatch.setattr(tool_execution, "_owner_is_admin", lambda owner: True)
launched = []
yield tmp_path, launched
for record in launched:
current = bg_jobs.get(record["id"])
if current and current["status"] == "running":
bg_jobs.kill(record["id"])
proc = bg_jobs._LIVE_PROCS.pop(record["pid"], None)
if proc:
proc.wait(timeout=8)
def finished(job_id):
deadline = time.monotonic() + 10
while time.monotonic() < deadline:
record = bg_jobs.get(job_id)
if record["status"] != "running":
return record
time.sleep(0.03)
pytest.fail("background job did not finish")
def test_detached_execution_owns_boundary_and_reports_death(jobs):
path, launched = jobs
record = bg_jobs.launch("printf captured", "chat", cwd=str(path))
launched.append(record)
result = finished(record["id"])
assert result["output"] == "captured"
assert result["exit_code"] == 0
assert result["containment"]["mechanism"] == "process_group"
assert result["containment"]["contained"] is False
assert result["teardown"]["dead"] is True
def test_supervisor_setup_failure_closes_unstarted_grant(jobs):
import json
import subprocess
import sys
from pathlib import Path
path, _ = jobs
spec = containment.agent_spec(str(path), dict(os.environ), 5)
grant = containment.acquire(spec, owner="failed-supervisor")
payload = {
"store_path": str(containment._store_path()),
"grant": {**grant.to_dict(), "owner": grant.owner},
"spec": {"workspace": str(path), "env": dict(spec.env), "wall_clock_s": 5,
"required": sorted(spec.required)},
"command": "printf effect > must-not-exist",
"log_path": str(path / "missing-directory" / "job.log"),
"result_path": str(path / "result.json"), "exit_path": str(path / "exit"),
}
worker = Path(containment.__file__).with_name("containment_worker.py")
result = subprocess.run([sys.executable, str(worker)], input=json.dumps(payload),
capture_output=True, text=True, timeout=10)
assert result.returncode == 0 # Supervisor publishes the failed job result.
assert "FileNotFoundError" in result.stderr
assert not (path / "must-not-exist").exists()
assert containment.active_grants() == []
assert (path / "exit").read_text() == "1"
report = json.loads((path / "result.json").read_text())
assert report["containment"]["executed"] is False
assert report["containment"]["contained"] is False
assert report["teardown"]["dead"] is True
async def test_bg_marker_refuses_without_spawning_and_authority_still_gates(jobs, monkeypatch):
path, _ = jobs
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_ENFORCING)
monkeypatch.setattr(bg_jobs.subprocess, "Popen", lambda *args, **kwargs: pytest.fail("uncontained bg spawn"))
block = namedtuple("Block", "tool_type content")("bash", "#!bg\nprintf unsafe")
execute = server_authorized_executor(tool_execution.execute_tool_block)
_, result = await execute(block, session_id="chat", owner="alice", workspace=str(path),
security_context=NO_TOOL_SECURITY_CONTEXT)
assert result["containment"]["executed"] is False
assert "bg_job_id" not in result
_, denied = await tool_execution.execute_tool_block(
block, session_id="chat", owner="alice", workspace=str(path),
security_context=NO_TOOL_SECURITY_CONTEXT, request_authority=None,
)
assert denied["failure_kind"] == "request_authority_denied"
def test_detached_supervisor_enforces_timeout(jobs):
path, launched = jobs
record = bg_jobs.launch("sleep 60", "chat", cwd=str(path), max_runtime_s=1)
launched.append(record)
result = finished(record["id"])
assert result["timed_out"] is True
assert result["teardown"]["dead"] is True
def test_restart_keeps_verified_background_supervisor(jobs):
path, launched = jobs
record = bg_jobs.launch("sleep 60", "chat", cwd=str(path))
launched.append(record)
report = process_reaper.reap_containment_grants()
assert report["background_kept"] == 1
killed = bg_jobs.kill(record["id"])
assert killed["killed"] is True
assert killed["teardown"]["dead"] is True
def test_kill_never_marks_a_foreign_pid_killed(jobs, monkeypatch):
record = {"id": "stale", "status": "running", "pid": 12345, "start_token": "old",
"session_id": "chat", "started_at": time.time(), "exit_path": "missing"}
bg_jobs._save({"stale": record})
monkeypatch.setattr(process_ownership, "verify", lambda *args: process_ownership.FOREIGN)
monkeypatch.setattr(bg_jobs, "_kill", lambda *args, **kwargs: pytest.fail("foreign process signalled"))
result = bg_jobs.kill("stale")
assert result["status"] == "running"
assert result.get("killed") is not True
assert result["teardown"]["dead"] is False
def test_running_detached_output_and_concurrent_grants_are_preserved(jobs):
path, launched = jobs
for number in range(3):
launched.append(bg_jobs.launch(f"printf job-{number}; sleep 0.3", "chat", cwd=str(path)))
for number, record in enumerate(launched):
assert finished(record["id"])["output"] == f"job-{number}"
grants = containment._load_records()
assert {record["containment_id"] for record in launched} <= grants.keys()
assert all(grants[record["containment_id"]]["release"]["dead"] for record in launched)
def test_detached_output_is_available_while_running(jobs):
path, launched = jobs
record = bg_jobs.launch("printf progress; sleep 5", "chat", cwd=str(path))
launched.append(record)
deadline = time.monotonic() + 3
while time.monotonic() < deadline:
current = bg_jobs.get(record["id"])
if "progress" in current["output"]:
assert current["status"] == "running"
return
time.sleep(0.03)
pytest.fail("detached stdout was unavailable until completion")
+6 -2
View File
@@ -11,7 +11,7 @@ import time
import pytest
from src import bg_jobs
from src import bg_jobs, containment, process_ownership
from src.agent_tools.bg_job_tools import ManageBgJobsTool
@@ -23,7 +23,11 @@ def store(tmp_path, monkeypatch):
monkeypatch.setattr(bg_jobs, "_JOBS_DIR", jobs_dir)
monkeypatch.setattr(bg_jobs, "_pid_alive", lambda pid: True)
killed: list = []
monkeypatch.setattr(bg_jobs, "_kill", lambda pid: killed.append(pid))
monkeypatch.setattr(process_ownership, "verify", lambda *args: process_ownership.OWNED)
def fake_kill(pid, **kwargs):
killed.append(pid)
return containment.ReleaseOutcome(dead=True, escalated=False)
monkeypatch.setattr(bg_jobs, "_kill", fake_kill)
return {"dir": jobs_dir, "killed": killed}
+544
View File
@@ -0,0 +1,544 @@
"""The containment API's own invariants.
These pin the contract rather than any one mechanism, so they run identically on
a host with bubblewrap and one without: every test substitutes
``containment.MECHANISMS`` with fake mechanisms whose availability and provided
dimensions are stated in the test. No real sandbox, no real process.
The one thing these tests must prove above all others: a request that cannot be
contained does not execute. That is asserted by recording every spawn attempt
and showing the list is empty.
"""
import asyncio
import pytest
from src import containment
@pytest.fixture(autouse=True)
def _isolated_store(tmp_path, monkeypatch):
"""Keep grant records out of ./data for every test in this module."""
store = tmp_path / "containment_grants.json"
monkeypatch.setattr(containment, "_store_path", lambda: store)
return store
@pytest.fixture
def workspace(tmp_path):
path = tmp_path / "ws"
path.mkdir()
return str(path)
@pytest.fixture
def no_spawn(monkeypatch):
"""Record spawn attempts and refuse them, so "did not execute" is provable."""
attempts = []
async def _refuse(*args, **kwargs):
attempts.append(args)
raise AssertionError("containment spawned a process it should not have")
monkeypatch.setattr(asyncio, "create_subprocess_exec", _refuse)
return attempts
def mechanism(name, rank, provides, *, available=True):
return containment.Mechanism(
name=name,
rank=rank,
available=lambda: available,
provides=lambda spec, _provides=frozenset(provides): _provides,
)
def install(monkeypatch, *mechanisms):
monkeypatch.setattr(containment, "MECHANISMS", tuple(mechanisms))
def enforcing(monkeypatch):
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_ENFORCING)
def report_only(monkeypatch):
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_REPORT_ONLY)
def spec_for(workspace, **kwargs):
kwargs.setdefault("env", {"PATH": "/usr/bin"})
kwargs.setdefault("wall_clock_s", 5)
return containment.ContainmentSpec(workspace=workspace, **kwargs)
ALL = tuple(sorted(containment.DIMENSIONS))
# ── The postcondition ───────────────────────────────────────────────────────
@pytest.mark.parametrize("provided", [
frozenset(containment.DEFAULT_REQUIRED),
containment.DIMENSIONS,
])
def test_required_is_always_a_subset_of_enforced(monkeypatch, workspace, provided):
"""spec.required <= grant.enforced, for any mechanism that can satisfy it."""
enforcing(monkeypatch)
install(monkeypatch, mechanism("fake", 10, provided))
grant = containment.acquire(spec_for(workspace), owner="session-1")
assert grant.spec.required <= grant.enforced
assert grant.contained is True
assert grant.unenforced_required == ()
def test_enforced_never_exceeds_what_the_spec_requested(monkeypatch, workspace):
"""A grant is never a superset of its spec: unrequested dimensions are not claimed."""
enforcing(monkeypatch)
install(monkeypatch, mechanism("generous", 10, containment.DIMENSIONS))
grant = containment.acquire(spec_for(workspace), owner="session-1")
# network/memory/process_count were not asked for, so they are not enforced
# even though the mechanism offers them.
assert grant.enforced == frozenset(containment.DEFAULT_REQUIRED)
assert containment.NETWORK not in grant.enforced
assert grant.degraded == ()
def test_enforced_is_always_within_the_known_dimension_set(monkeypatch, workspace):
"""A mechanism cannot invent a dimension the API does not define."""
enforcing(monkeypatch)
install(monkeypatch, mechanism(
"liar", 10, frozenset(containment.DEFAULT_REQUIRED) | {"telepathy"},
))
grant = containment.acquire(spec_for(workspace), owner="session-1")
assert grant.enforced <= containment.DIMENSIONS
# ── Deterministic failure: the request does not execute ─────────────────────
async def test_uncontainable_request_does_not_execute(monkeypatch, workspace, no_spawn):
"""The headline contract: no mechanism for a required dimension → no process.
Written as a call site would use the API — acquire, then run — so the proof
covers the whole path and not just the raising function.
"""
enforcing(monkeypatch)
# The only mechanism available cannot do filesystem, which is required.
install(monkeypatch, mechanism(
"group_only", 10, {containment.PROCESS_TREE, containment.WALL_CLOCK},
))
spec = containment.agent_spec(workspace, {"PATH": "/usr/bin"}, 5)
try:
grant = containment.acquire(spec, owner="session-1")
except containment.ContainmentUnavailable as exc:
result = containment.unavailable_tool_result(exc, tool="bash")
else: # pragma: no cover - the point of the test is that this is unreachable
result = await containment.run(grant, "echo hello")
assert no_spawn == [], "an uncontainable request reached a spawn"
assert result["exit_code"] == 1
assert "command not executed" in result["error"]
assert "filesystem" in result["error"]
assert result["containment"]["contained"] is False
assert result["containment"]["executed"] is False
assert result["containment"]["unenforced_required"] == ["filesystem"]
# A run that could not be contained is distinguishable from a contained run
# that failed: there is no output key at all, and `executed` is explicit.
assert "output" not in result
def test_unavailable_names_every_missing_dimension_and_the_mechanism_tried(
monkeypatch, workspace,
):
enforcing(monkeypatch)
install(monkeypatch, mechanism("group_only", 10, {containment.WALL_CLOCK}))
spec = containment.agent_spec(workspace, {}, 5)
with pytest.raises(containment.ContainmentUnavailable) as caught:
containment.acquire(spec, owner="session-1")
assert caught.value.missing == frozenset({
containment.FILESYSTEM, containment.PROCESS_TREE,
})
assert caught.value.mechanism_tried == "group_only"
assert "containment unavailable" in str(caught.value)
def test_no_mechanism_at_all_still_refuses_rather_than_running(
monkeypatch, workspace, no_spawn,
):
"""With nothing available there is no weaker thing to fall back to."""
enforcing(monkeypatch)
install(monkeypatch, mechanism("absent", 10, containment.DIMENSIONS, available=False))
with pytest.raises(containment.ContainmentUnavailable) as caught:
containment.acquire(containment.agent_spec(workspace, {}, 5), owner="s")
assert caught.value.mechanism_tried == "none"
assert no_spawn == []
async def test_a_forged_grant_cannot_buy_a_spawn(monkeypatch, workspace, no_spawn):
"""run() re-checks the postcondition at the point of effect.
A grant is a plain record, so a caller could construct one claiming
dimensions it does not have. run() refuses it rather than trusting the
record it was handed.
"""
enforcing(monkeypatch)
spec = containment.agent_spec(workspace, {}, 5)
forged = containment.ContainmentGrant(
id="forged",
mechanism="bubblewrap",
workspace=workspace,
enforced=frozenset({containment.WALL_CLOCK}),
degraded=(),
unenforced_required=(), # the lie: claims nothing is missing
owner="session-1",
mode=containment.MODE_ENFORCING,
spec=spec,
)
with pytest.raises(containment.ContainmentUnavailable):
await containment.run(forged, "echo hello")
assert no_spawn == []
# ── Best-effort dimensions degrade, they do not refuse ──────────────────────
def test_unavailable_best_effort_dimension_is_reported_not_refused(monkeypatch, workspace):
enforcing(monkeypatch)
install(monkeypatch, mechanism("fake", 10, containment.DEFAULT_REQUIRED))
spec = containment.agent_spec(
workspace, {}, 5, network=containment.NETWORK_NONE, max_memory_bytes=1 << 30,
)
grant = containment.acquire(spec, owner="session-1")
assert grant.contained is True
assert grant.degraded == (containment.MEMORY, containment.NETWORK)
# And it is reported as fact in the model-visible block, never as permission.
block = grant.to_dict()
assert block["degraded"] == ["memory", "network"]
assert block["contained"] is True
assert "env" not in block
def test_a_degraded_dimension_is_never_also_enforced(monkeypatch, workspace):
enforcing(monkeypatch)
install(monkeypatch, mechanism("fake", 10, containment.DEFAULT_REQUIRED))
spec = containment.agent_spec(workspace, {}, 5, max_processes=16)
grant = containment.acquire(spec, owner="session-1")
assert set(grant.degraded).isdisjoint(grant.enforced)
# ── Mechanism selection: strongest first, command-independent ───────────────
def test_selection_is_strongest_first(monkeypatch, workspace):
enforcing(monkeypatch)
install(
monkeypatch,
mechanism("weak", 10, containment.DEFAULT_REQUIRED),
mechanism("strong", 30, containment.DIMENSIONS),
mechanism("middle", 20, containment.DEFAULT_REQUIRED),
)
grant = containment.acquire(spec_for(workspace), owner="session-1")
assert grant.mechanism == "strong"
def test_selection_skips_unavailable_mechanisms(monkeypatch, workspace):
enforcing(monkeypatch)
install(
monkeypatch,
mechanism("strong", 30, containment.DIMENSIONS, available=False),
mechanism("weak", 10, containment.DEFAULT_REQUIRED),
)
grant = containment.acquire(spec_for(workspace), owner="session-1")
assert grant.mechanism == "weak"
def test_selection_never_substitutes_a_weaker_mechanism_for_a_required_dimension(
monkeypatch, workspace,
):
"""The strongest available mechanism is used, not the first that is "good enough"."""
enforcing(monkeypatch)
install(
monkeypatch,
mechanism("netcapable", 30, containment.DIMENSIONS),
mechanism("nonet", 20, containment.DEFAULT_REQUIRED),
)
spec = containment.ContainmentSpec(
workspace=workspace,
env={},
wall_clock_s=5,
required=frozenset(containment.DEFAULT_REQUIRED) | {containment.NETWORK},
network=containment.NETWORK_NONE,
)
grant = containment.acquire(spec, owner="session-1")
assert grant.mechanism == "netcapable"
assert containment.NETWORK in grant.enforced
def test_a_failing_availability_probe_is_treated_as_unavailable(monkeypatch, workspace):
enforcing(monkeypatch)
def _explode():
raise OSError("probe blew up")
install(
monkeypatch,
containment.Mechanism("broken", 30, _explode, lambda spec: containment.DIMENSIONS),
mechanism("weak", 10, containment.DEFAULT_REQUIRED),
)
grant = containment.acquire(spec_for(workspace), owner="session-1")
assert grant.mechanism == "weak"
@pytest.mark.parametrize("command", [
"echo hello",
"rm -rf / --no-preserve-root",
"cat /workspace/notes.txt # this command is safe, honestly",
])
def test_the_boundary_does_not_depend_on_the_command(monkeypatch, workspace, command):
"""Containment is established before any command text exists.
acquire() is not given the command, so no request text, tool argument or
model assertion can change the mechanism or widen the enforced set. The
parametrised commands are only here to show the API has nowhere to put them.
"""
enforcing(monkeypatch)
install(monkeypatch, mechanism("fake", 10, containment.DIMENSIONS))
spec = containment.agent_spec(workspace, {}, 5)
grant = containment.acquire(spec, owner="session-1")
assert grant.mechanism == "fake"
assert grant.enforced == frozenset(containment.DEFAULT_REQUIRED)
assert "command" not in grant.to_dict()
def test_agent_spec_cannot_be_given_a_weaker_required_set(workspace):
"""One factory for model-reachable spawns, so no call site can weaken it."""
spec = containment.agent_spec(
workspace, {}, 5, required=frozenset({containment.WALL_CLOCK}),
)
assert spec.required == containment.DEFAULT_REQUIRED
# ── Report-only mode ────────────────────────────────────────────────────────
def test_report_only_records_the_shortfall_instead_of_refusing(monkeypatch, workspace):
report_only(monkeypatch)
install(monkeypatch, mechanism(
"group_only", 10, {containment.PROCESS_TREE, containment.WALL_CLOCK},
))
grant = containment.acquire(containment.agent_spec(workspace, {}, 5), owner="s")
assert grant.unenforced_required == ("filesystem",)
assert grant.contained is False
assert grant.mode == containment.MODE_REPORT_ONLY
assert grant.to_dict()["unenforced_required"] == ["filesystem"]
def test_report_only_logs_the_shortfall_once_per_grant(monkeypatch, workspace, caplog):
report_only(monkeypatch)
install(monkeypatch, mechanism("group_only", 10, {containment.WALL_CLOCK}))
with caplog.at_level("WARNING", logger="src.containment"):
grant = containment.acquire(containment.agent_spec(workspace, {}, 5), owner="s")
messages = [record.getMessage() for record in caplog.records]
assert sum("NOT contained" in message for message in messages) == 1
assert grant.id in messages[0]
def test_the_two_modes_differ_only_in_whether_the_shortfall_refuses(monkeypatch, workspace):
install(monkeypatch, mechanism("group_only", 10, {containment.WALL_CLOCK}))
spec = containment.agent_spec(workspace, {}, 5)
report_only(monkeypatch)
reported = containment.acquire(spec, owner="s")
enforcing(monkeypatch)
with pytest.raises(containment.ContainmentUnavailable) as caught:
containment.acquire(spec, owner="s")
assert frozenset(reported.unenforced_required) == caught.value.missing
def test_enforcement_is_the_shipped_default():
assert containment.CONTAINMENT_MODE == containment.MODE_ENFORCING
# ── Spec validation: caller bugs raise in both modes ────────────────────────
@pytest.mark.parametrize("mode", [containment.MODE_ENFORCING, containment.MODE_REPORT_ONLY])
@pytest.mark.parametrize("overrides, fragment", [
({"required": frozenset({"telepathy"})}, "unknown required dimension"),
({"required": frozenset({containment.NETWORK})}, "does not request it"),
({"required": frozenset({containment.MEMORY})}, "does not request it"),
({"wall_clock_s": 0}, "must be positive"),
({"wall_clock_s": -1}, "must be positive"),
({"max_output_bytes": 0}, "max_output_bytes must be positive"),
({"max_memory_bytes": 0}, "max_memory_bytes must be a positive int"),
({"max_processes": -4}, "max_processes must be a positive int"),
({"network": "maybe"}, "network must be"),
({"env": {"A": 1}}, "env keys and values must be str"),
({"env": {"A": "x\x00y"}}, "must not contain NUL"),
({"writable_extra": ("relative/path",)}, "must be absolute"),
({"writable_extra": ("/",)}, "reserved path"),
({"readonly_extra": ("/workspace",)}, "reserved path"),
])
def test_a_malformed_spec_raises_in_both_modes(
monkeypatch, workspace, mode, overrides, fragment,
):
monkeypatch.setattr(containment, "CONTAINMENT_MODE", mode)
install(monkeypatch, mechanism("fake", 10, containment.DIMENSIONS))
spec = spec_for(workspace, **overrides)
with pytest.raises(ValueError, match=fragment):
containment.acquire(spec, owner="session-1")
@pytest.mark.parametrize("bad_workspace, fragment", [
("", "non-empty path"),
("relative/ws", "must be absolute"),
])
def test_a_malformed_workspace_raises(monkeypatch, bad_workspace, fragment):
install(monkeypatch, mechanism("fake", 10, containment.DIMENSIONS))
spec = containment.ContainmentSpec(
workspace=bad_workspace, env={}, wall_clock_s=5,
)
with pytest.raises(ValueError, match=fragment):
containment.acquire(spec, owner="session-1")
def test_a_workspace_that_is_not_a_directory_raises(monkeypatch, tmp_path):
install(monkeypatch, mechanism("fake", 10, containment.DIMENSIONS))
missing = tmp_path / "nope"
spec = containment.ContainmentSpec(workspace=str(missing), env={}, wall_clock_s=5)
with pytest.raises(ValueError, match="not a directory"):
containment.acquire(spec, owner="session-1")
def test_a_grant_without_an_owner_raises(monkeypatch, workspace):
install(monkeypatch, mechanism("fake", 10, containment.DIMENSIONS))
with pytest.raises(ValueError, match="needs an owner"):
containment.acquire(spec_for(workspace), owner=" ")
def test_the_child_environment_cannot_be_edited_after_acquire(monkeypatch, workspace):
"""env is part of the boundary, so the caller's dict is copied and frozen."""
enforcing(monkeypatch)
install(monkeypatch, mechanism("fake", 10, containment.DIMENSIONS))
caller_env = {"PATH": "/usr/bin"}
grant = containment.acquire(
spec_for(workspace, env=caller_env), owner="session-1",
)
caller_env["LD_PRELOAD"] = "/tmp/evil.so"
assert dict(grant.spec.env) == {"PATH": "/usr/bin"}
with pytest.raises(TypeError):
grant.spec.env["LD_PRELOAD"] = "/tmp/evil.so"
# ── Durable records: one owner, one record ─────────────────────────────────
def test_a_grant_is_recorded_with_its_owner_and_declared_limits(
monkeypatch, workspace, _isolated_store,
):
enforcing(monkeypatch)
install(monkeypatch, mechanism("fake", 10, containment.DIMENSIONS))
spec = containment.agent_spec(workspace, {}, 7, max_processes=8)
grant = containment.acquire(spec, owner="session-42")
active = containment.active_grants()
assert [record["id"] for record in active] == [grant.id]
record = active[0]
assert record["owner"] == "session-42"
assert record["wall_clock_s"] == 7
assert record["max_processes"] == 8
assert record["required"] == sorted(containment.DEFAULT_REQUIRED)
assert record["pid"] is None
def test_releasing_a_grant_with_no_process_clears_it_from_active(monkeypatch, workspace):
enforcing(monkeypatch)
install(monkeypatch, mechanism("fake", 10, containment.DIMENSIONS))
grant = containment.acquire(containment.agent_spec(workspace, {}, 5), owner="s")
outcome = containment.release(grant)
assert outcome.dead is True
assert outcome.escalated is False
assert outcome.survivors == ()
assert containment.active_grants() == []
def test_forget_drops_a_record(monkeypatch, workspace):
enforcing(monkeypatch)
install(monkeypatch, mechanism("fake", 10, containment.DIMENSIONS))
grant = containment.acquire(containment.agent_spec(workspace, {}, 5), owner="s")
containment.forget(grant.id)
assert containment.active_grants() == []
def test_an_unwritable_store_does_not_take_out_execution(monkeypatch, workspace, caplog):
"""The record is observability, not a containment dimension.
Refusing an authorized command because a journal file could not be written
would be a worse failure than running it, so this degrades loudly and the
grant still describes the boundary accurately.
"""
enforcing(monkeypatch)
install(monkeypatch, mechanism("fake", 10, containment.DIMENSIONS))
def _explode(*args, **kwargs):
raise OSError("read-only filesystem")
monkeypatch.setattr(containment, "atomic_write_json", _explode)
with caplog.at_level("WARNING", logger="src.containment"):
grant = containment.acquire(containment.agent_spec(workspace, {}, 5), owner="s")
assert grant.contained is True
assert any("could not persist" in record.getMessage() for record in caplog.records)
def test_a_corrupt_store_does_not_take_out_execution(monkeypatch, workspace, _isolated_store):
enforcing(monkeypatch)
install(monkeypatch, mechanism("fake", 10, containment.DIMENSIONS))
_isolated_store.write_text("{ this is not json", encoding="utf-8")
grant = containment.acquire(containment.agent_spec(workspace, {}, 5), owner="s")
assert [record["id"] for record in containment.active_grants()] == [grant.id]
# ── Execution that leaves the box is declared, not pretended ───────────────
async def test_an_external_bridge_grant_claims_nothing_and_cannot_be_run_locally(
monkeypatch, workspace, no_spawn,
):
enforcing(monkeypatch)
spec = containment.agent_spec(workspace, {}, 5)
grant = containment.declare_external_bridge(
spec, owner="session-1", endpoint="http://127.0.0.1:8777/exec",
)
assert grant.mechanism == "external_bridge"
assert grant.enforced == frozenset()
assert grant.external is True
assert grant.contained is False
assert grant.to_dict()["external"] is True
with pytest.raises(ValueError, match="does not own"):
await containment.run(grant, "echo hello")
assert no_spawn == []
@pytest.mark.parametrize("protected_dest", [
"/etc",
"/etc/ssl",
"/usr",
"/usr/local",
"/bin",
"/bin/sh",
"/sbin",
"/lib",
"/lib64",
"/proc",
"/proc/sys",
"/dev",
"/dev/shm",
"/sys",
"/root",
"/root/.ssh",
"/home",
"/workspace",
"/workspace/sub",
])
def test_writable_extra_rejects_protected_system_roots_and_descendants(monkeypatch, workspace, protected_dest):
install(monkeypatch, mechanism("fake", 10, containment.DIMENSIONS))
spec = spec_for(workspace, writable_extra=(protected_dest,))
with pytest.raises(ValueError, match="reserved path"):
containment.acquire(spec, owner="session-1")
def test_writable_extra_accepts_legitimate_scratch_destinations(monkeypatch, workspace):
install(monkeypatch, mechanism("fake", 10, containment.DIMENSIONS))
spec = spec_for(workspace, writable_extra=("/var/scratch", "/tmp/custom_scratch", "/home/testuser/scratch"))
grant = containment.acquire(spec, owner="session-1")
assert "/var/scratch" in grant.spec.writable_extra
assert "/tmp/custom_scratch" in grant.spec.writable_extra
assert "/home/testuser/scratch" in grant.spec.writable_extra
+379
View File
@@ -0,0 +1,379 @@
"""Final enforcement gate: real namespaces, no fallback, and owned cancellation."""
import asyncio
import os
import subprocess
import sys
from dataclasses import replace
import pytest
from src import containment, tool_execution
from src.agent_tools import subprocess_tools
@pytest.fixture
def workspace(tmp_path, monkeypatch):
path = tmp_path / "workspace"
path.mkdir()
monkeypatch.setattr(tool_execution, "agent_cwd", lambda: str(path))
monkeypatch.setattr(containment, "_store_path", lambda: tmp_path / "grants.json")
return path
@pytest.fixture
def namespaces(workspace):
if not containment._bwrap_available():
pytest.skip("functional bubblewrap PID/mount namespaces unavailable")
return workspace
def test_installed_but_nonfunctional_bwrap_is_not_available(monkeypatch):
calls = []
monkeypatch.setattr(containment, "IS_WINDOWS", False)
monkeypatch.setattr(containment.shutil, "which", lambda name: "/usr/bin/bwrap")
def blocked(argv, **kwargs):
calls.append((argv, kwargs))
return subprocess.CompletedProcess(argv, 1, b"", b"Operation not permitted")
monkeypatch.setattr(containment.subprocess, "run", blocked)
assert containment._bwrap_available() is False
assert calls[0][0][-1] == "/bin/true"
assert "--unshare-pid" in calls[0][0]
@pytest.mark.parametrize("tool,source", [
(subprocess_tools.BashTool, "echo forbidden"),
(subprocess_tools.PythonTool, "print(1 + 1)"),
])
async def test_shipped_mode_refuses_without_namespaces(tool, source, workspace, monkeypatch):
assert containment.CONTAINMENT_MODE == containment.MODE_ENFORCING
monkeypatch.setattr(containment, "MECHANISMS", tuple(
item for item in containment.MECHANISMS if item.name != "bubblewrap"
))
async def forbidden(*args, **kwargs):
pytest.fail("uncontained model command spawned")
monkeypatch.setattr(asyncio, "create_subprocess_exec", forbidden)
result = await tool().execute(source, {})
assert "containment unavailable" in result["error"]
assert result["containment"]["executed"] is False
assert result["containment"]["contained"] is False
assert {"filesystem", "process_tree"} <= set(result["containment"]["unenforced_required"])
def test_process_groups_and_taskkill_do_not_claim_tree_containment(workspace):
spec = containment.agent_spec(str(workspace), dict(os.environ), 5)
assert containment.PROCESS_TREE not in containment._posix_group_provides(spec)
assert containment.PROCESS_TREE not in containment._windows_provides(spec)
async def test_fully_overclaimed_group_grant_cannot_spawn(workspace, monkeypatch):
spec = containment.agent_spec(str(workspace), {}, 5)
forged = containment.ContainmentGrant(
id="forged-all", mechanism="process_group", workspace=str(workspace),
enforced=containment.DEFAULT_REQUIRED, degraded=(), unenforced_required=(),
owner="test", mode=containment.MODE_ENFORCING, spec=spec,
)
async def forbidden(*args, **kwargs):
pytest.fail("overclaimed grant reached a host spawn")
monkeypatch.setattr(asyncio, "create_subprocess_exec", forbidden)
with pytest.raises(containment.ContainmentUnavailable):
await containment.run(forged, "echo forbidden")
@pytest.mark.skipif(os.name == "nt", reason="POSIX namespace recipe")
def test_home_interpreter_is_bound_read_only(workspace, monkeypatch):
original = containment.shutil.which
monkeypatch.setattr(containment.shutil, "which", lambda name:
"/usr/bin/bwrap" if name == "bwrap" else original(name))
monkeypatch.setattr(sys, "prefix", "/home/test/venv")
spec = subprocess_tools._owned_spec(str(workspace), {}, 5)
assert "/home/test/venv" in spec.readonly_extra
argv = containment._bwrap_prefix(spec)
index = argv.index("/home/test/venv")
assert argv[index - 1] == "--ro-bind"
assert "--unshare-pid" in argv
assert "--dev-bind" not in argv
assert "--unshare-net" not in argv
async def test_actual_namespace_hides_host_pid_tree_and_sibling(namespaces):
sibling = namespaces.parent / "host-secret.txt"
sibling.write_text("original")
host_namespace = os.readlink("/proc/self/ns/pid")
result = await subprocess_tools.PythonTool().execute(
"import os\n"
"print(os.readlink('/proc/self/ns/pid'))\n"
f"print(os.path.exists({str(sibling)!r}))\n"
f"try:\n open({str(sibling)!r}, 'w').write('changed')\n"
"except OSError:\n pass\n", {},
)
assert result["exit_code"] == 0, result
lines = result["output"].splitlines()
assert lines[0] != host_namespace
assert lines[1] == "False"
assert sibling.read_text() == "original"
assert result["containment"]["contained"] is True
assert result["teardown"]["dead"] is True
@pytest.mark.skipif(not hasattr(os, "pidfd_open"), reason="Linux kernel handles")
async def test_dead_owner_cannot_hide_live_namespace_init(workspace, monkeypatch):
from types import SimpleNamespace
child = await asyncio.create_subprocess_exec(sys.executable, "-c", "import time; time.sleep(60)",
start_new_session=True)
spec = containment.ContainmentSpec(str(workspace), dict(os.environ), 5, required={containment.WALL_CLOCK})
grant = containment.acquire(spec, owner="init-life")
token = containment.process_ownership.start_token(child.pid)
live = replace(grant, mechanism="bubblewrap", pid=99999999, pgid=99999999)
async def wait():
return 0
owner = SimpleNamespace(returncode=0, wait=wait, _ody_namespace_pid=child.pid,
_ody_namespace_token=token, _ody_namespace_pidfd=os.pidfd_open(child.pid))
def denied(*args):
raise PermissionError("EPERM")
monkeypatch.setattr(containment.signal, "pidfd_send_signal", denied)
try:
result = await containment._release_awaited(live, owner, grace_s=0)
assert result.dead is False
assert child.pid in result.survivors
assert child.returncode is None
record = containment.active_grants()[0]
assert record["release"]["dead"] is False
assert record["released_at"] is None
finally:
child.kill()
await child.wait()
containment.release(live, grace_s=0)
@pytest.mark.skipif(not hasattr(os, "pidfd_open"), reason="Linux kernel handles")
@pytest.mark.parametrize("foreign", [False, True])
async def test_recovered_namespace_init_identity_survives_owner_exit(workspace, foreign):
child = await asyncio.create_subprocess_exec(sys.executable, "-c", "import time; time.sleep(60)",
start_new_session=True)
spec = containment.ContainmentSpec(str(workspace), dict(os.environ), 5, required={containment.WALL_CLOCK})
grant = containment.acquire(spec, owner="recovered-init")
token = "different-boot" if foreign else containment.process_ownership.start_token(child.pid)
containment._update_record(grant.id, pid=99999999, pgid=99999999, start_token="old-owner",
namespace_pid=child.pid, namespace_start_token=token, execution_started=True)
try:
result = containment.reap_record(containment.active_grants()[0], grace_s=0)
assert result.dead is True
if foreign:
assert child.returncode is None # Never signal a reused namespace PID.
else:
await asyncio.wait_for(child.wait(), 3)
assert child.returncode is not None
assert containment.active_grants() == []
finally:
if child.returncode is None:
child.kill()
await child.wait()
async def test_repeated_release_does_not_signal_reused_pid(workspace, monkeypatch):
spec = containment.ContainmentSpec(str(workspace), dict(os.environ), 5, required={containment.WALL_CLOCK})
grant = containment.acquire(spec, owner="completed")
assert containment.release(grant).dead
def forbidden(*args):
pytest.fail("completed grant signalled a reused slot")
monkeypatch.setattr(containment, "_signal_tree", forbidden)
assert containment.release(replace(grant, pid=12345678, pgid=12345678)).dead
async def forbidden_spawn(*args, **kwargs):
pytest.fail("released grant spawned another process")
monkeypatch.setattr(asyncio, "create_subprocess_exec", forbidden_spawn)
with pytest.raises(ValueError, match="released grant"):
await containment.run(grant, "echo forbidden")
async def test_default_namespace_preserves_loopback_sidecars(namespaces):
async def reply(reader, writer):
writer.write(b"sidecar\n")
await writer.drain()
writer.close()
await writer.wait_closed()
server = await asyncio.start_server(reply, "127.0.0.1", 0)
async with server:
port = server.sockets[0].getsockname()[1]
result = await subprocess_tools.PythonTool().execute(
f"import socket\ns=socket.create_connection(('127.0.0.1', {port}), timeout=2)\n"
"print(s.recv(100).decode().strip())\ns.close()", {},
)
assert result["output"] == "sidecar", result
assert result["containment"]["network"] == "inherit"
assert "network" not in result["containment"]["enforced"]
async def test_namespace_handshake_closes_model_stdin(namespaces):
result = await subprocess_tools.BashTool().execute(
"if read value; then echo unexpected; else echo closed; fi", {},
)
assert result["output"] == "closed", result
assert result["teardown"]["dead"] is True
@pytest.mark.parametrize("legacy_name", [True, False])
async def test_execution_environment_cannot_replace_probed_bwrap(namespaces, monkeypatch, legacy_name):
bin_dir = namespaces / "bin"
bin_dir.mkdir()
outside = namespaces.parent / "uncontained-effect"
impostor = bin_dir / "bwrap"
import shlex
impostor.write_text("#!/bin/sh\nprintf escaped > " + shlex.quote(str(outside)) + "\n")
impostor.chmod(0o700)
if legacy_name:
original = containment._bwrap_prefix
def bare_name(spec):
argv = original(spec)
argv[0] = "bwrap"
return argv
monkeypatch.setattr(containment, "_bwrap_prefix", bare_name)
result = await subprocess_tools.BashTool().execute("printf contained", {
"subproc_env": {**os.environ, "PATH": str(bin_dir) + os.pathsep + os.environ.get("PATH", "")},
})
if legacy_name:
# Reproduce the old mismatch: availability probed the host binary,
# while launch resolved a different binary through the child's PATH.
assert "containment unavailable" in result["error"]
assert outside.read_text() == "escaped"
else:
assert result["output"] == "contained", result
assert not outside.exists()
async def test_partial_initialization_reaps_before_model_code_starts(namespaces, monkeypatch):
from src.agent_runtime import journal
effect = namespaces / "must-not-exist"
def fail(*args, **kwargs):
raise RuntimeError("initialization failed")
monkeypatch.setattr(journal, "mark_operation_started", fail)
result = await subprocess_tools.PythonTool().execute(
f"open({str(effect)!r}, 'w').write('effect')", {},
)
assert result["exit_code"] == 1
assert result["containment"]["executed"] is False
assert result["containment"]["contained"] is False
assert not effect.exists()
assert containment.active_grants() == []
async def test_explicit_network_isolation_is_established_or_refused(namespaces):
spec = replace(subprocess_tools._owned_spec(str(namespaces), dict(os.environ), 5),
network=containment.NETWORK_NONE,
required=containment.DEFAULT_REQUIRED | {containment.NETWORK})
host_namespace = os.readlink("/proc/self/ns/net")
grant = containment.acquire(spec, owner="network-hook")
try:
result = await containment.run(grant, [sys.executable, "-c",
"import os; print(os.readlink('/proc/self/ns/net'))"], argv=True)
except containment.ContainmentUnavailable:
assert containment.active_grants() == []
else:
assert result.exit_code == 0
assert result.stdout.strip() != host_namespace
assert containment.NETWORK in result.grant.enforced
assert result.release.dead
@pytest.mark.parametrize("exit_parent", [False, True])
async def test_setsid_daemon_cannot_survive_namespace_death(namespaces, exit_parent):
heartbeat = namespaces / "heartbeat"
code = (
"import os,signal,time\n"
"pid=os.fork()\n"
"if pid:\n"
+ (" time.sleep(.2); os._exit(0)\n" if exit_parent else " time.sleep(60); os._exit(0)\n")
+ "os.setsid()\nsignal.signal(signal.SIGTERM, signal.SIG_IGN)\n"
"while True:\n"
f" open({str(heartbeat)!r}, 'w').write(str(time.monotonic_ns()))\n"
" time.sleep(.01)\n"
)
spec = subprocess_tools._owned_spec(str(namespaces), dict(os.environ), 1)
result = await containment.run(containment.acquire(spec, owner="setsid"),
[sys.executable, "-c", code], argv=True)
assert heartbeat.exists(), result.stderr
assert result.timed_out is (not exit_parent)
assert result.release.dead is True
last = heartbeat.read_text()
await asyncio.sleep(.15)
assert heartbeat.read_text() == last
assert containment.active_grants() == []
async def test_failed_namespace_initialization_does_not_claim_containment(workspace, monkeypatch):
from types import SimpleNamespace
monkeypatch.setattr(containment, "MECHANISMS", (containment.Mechanism(
"bubblewrap", 30, lambda: True, lambda spec: containment.DEFAULT_REQUIRED,
),))
async def fail(*args, **kwargs):
stdout, stderr = asyncio.StreamReader(), asyncio.StreamReader()
stdout.feed_eof()
stderr.feed_data(b"bwrap: bind failed\n")
stderr.feed_eof()
async def wait():
return 1
return SimpleNamespace(pid=99999999, stdout=stdout, stderr=stderr, returncode=1, wait=wait)
async def release(grant, proc, **kwargs):
outcome = containment.ReleaseOutcome(dead=True, escalated=False)
containment._finish_release(grant, outcome)
return outcome
monkeypatch.setattr(asyncio, "create_subprocess_exec", fail)
monkeypatch.setattr(containment, "_release_awaited", release)
result = await subprocess_tools.PythonTool().execute("print('never')", {})
assert "containment unavailable" in result["error"]
assert result["containment"]["executed"] is False
assert result["containment"]["enforced"] == []
assert containment.active_grants() == []
@pytest.mark.skipif(os.name == "nt", reason="real POSIX process")
async def test_cancellation_during_spawn_recovers_and_reaps_handle(workspace, monkeypatch):
monkeypatch.setattr(containment, "MECHANISMS", tuple(
item for item in containment.MECHANISMS if item.name == "process_group"
))
real_spawn = asyncio.create_subprocess_exec
spawned, return_handle = asyncio.Event(), asyncio.Event()
children = []
async def delayed_spawn(*args, **kwargs):
proc = await real_spawn(*args, **kwargs)
children.append(proc)
spawned.set()
await return_handle.wait()
return proc
monkeypatch.setattr(asyncio, "create_subprocess_exec", delayed_spawn)
spec = containment.ContainmentSpec(str(workspace), dict(os.environ), 10,
required={containment.WALL_CLOCK})
task = asyncio.create_task(containment.run(containment.acquire(spec, owner="spawn-cancel"),
[sys.executable, "-c", "import time; time.sleep(60)"], argv=True))
await asyncio.wait_for(spawned.wait(), 3)
task.cancel()
await asyncio.sleep(.01)
task.cancel()
return_handle.set()
with pytest.raises(asyncio.CancelledError):
await asyncio.wait_for(task, 8)
assert children[0].returncode is not None
assert containment.active_grants() == []
@pytest.mark.skipif(os.name == "nt", reason="real POSIX process")
async def test_repeated_cancellation_cannot_interrupt_kill_escalation(workspace, monkeypatch):
monkeypatch.setattr(containment, "MECHANISMS", tuple(
item for item in containment.MECHANISMS if item.name == "process_group"
))
ready = workspace / "ready"
spec = containment.ContainmentSpec(str(workspace), dict(os.environ), 10,
required={containment.WALL_CLOCK})
task = asyncio.create_task(containment.run(containment.acquire(spec, owner="cancel"),
[sys.executable, "-c", "import signal,time; signal.signal(signal.SIGTERM,signal.SIG_IGN); "
f"open({str(ready)!r},'w').write('ready'); time.sleep(60)"], argv=True))
for _ in range(100):
if ready.exists():
break
await asyncio.sleep(.02)
assert ready.exists()
task.cancel()
await asyncio.sleep(.1)
task.cancel()
with pytest.raises(asyncio.CancelledError):
await asyncio.wait_for(task, 8)
assert containment.active_grants() == []
+430
View File
@@ -0,0 +1,430 @@
"""Teardown through the containment boundary, against real processes.
The headline case is the one that fails on an unmodified baseline: a command
that backgrounds a grandchild and then times out leaves the grandchild running,
while the tool result claims "process killed". These tests pin that the boundary
signals the whole process group and verifies death before reporting it.
POSIX only — the Windows path walks the tree with ``taskkill /T /F`` and has no
host here to run on, which is stated in the PR rather than skipped silently.
"""
import os
import signal
import subprocess
import sys
import time
from dataclasses import replace
import pytest
from core.platform_compat import pid_alive
from src import containment
pytestmark = pytest.mark.skipif(
sys.platform.startswith("win"), reason="POSIX process groups; Windows path untested here"
)
@pytest.fixture(autouse=True)
def _isolated_store(tmp_path, monkeypatch):
store = tmp_path / "containment_grants.json"
monkeypatch.setattr(containment, "_store_path", lambda: store)
return store
@pytest.fixture(autouse=True)
def _real_process_group_mechanism(monkeypatch):
"""Pin the mechanism to the real POSIX process group.
Not a fake: this is the mechanism shipped in ``MECHANISMS``, selected by
name so the test behaves the same on a host that happens to have bubblewrap
installed. Filesystem containment is bubblewrap's job and is not what these
tests are about.
"""
selected = [item for item in containment.MECHANISMS if item.name == "process_group"]
assert selected, "process_group mechanism disappeared from MECHANISMS"
monkeypatch.setattr(containment, "MECHANISMS", tuple(selected))
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_ENFORCING)
@pytest.fixture
def workspace(tmp_path):
path = tmp_path / "ws"
path.mkdir()
return str(path)
def tree_spec(workspace, **kwargs):
"""Exercise group teardown without claiming prevention of session escape."""
kwargs.setdefault("env", {"PATH": "/usr/bin:/bin:/usr/sbin:/sbin"})
kwargs.setdefault("wall_clock_s", 1)
return containment.ContainmentSpec(
workspace=workspace,
required=frozenset({containment.WALL_CLOCK}),
**kwargs,
)
def read_pid(path, *, timeout=5.0):
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
try:
text = path.read_text(encoding="utf-8").strip()
except OSError:
text = ""
if text.isdigit():
return int(text)
time.sleep(0.02)
raise AssertionError(f"{path} never received a pid")
def gone(pid, *, timeout=5.0):
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
if not pid_alive(pid):
return True
time.sleep(0.02)
return not pid_alive(pid)
# ── The regression this lane exists to close ────────────────────────────────
async def test_a_timeout_leaves_no_surviving_grandchild(tmp_path, workspace):
"""A backgrounded grandchild does not survive the wall-clock kill.
On a baseline spawn site the wrapper shell is killed with ``proc.kill()``
and the grandchild keeps running, unowned and unreaped, while the tool
result says the process was killed.
"""
pidfile = tmp_path / "grandchild.pid"
command = f"bash -c 'sleep 60 & echo $! > {pidfile}'; sleep 60"
grant = containment.acquire(tree_spec(workspace), owner="session-1")
result = await containment.run(grant, command)
grandchild = read_pid(pidfile)
assert result.timed_out is True
assert gone(grandchild), f"grandchild {grandchild} survived the timeout kill"
assert result.release is not None
assert result.release.dead is True
assert result.release.survivors == ()
async def test_the_timeout_outcome_is_observed_not_asserted(tmp_path, workspace):
"""``dead`` reflects a verified empty process group, not a signal that was sent."""
pidfile = tmp_path / "child.pid"
command = f"echo $$ > {pidfile}; sleep 60"
grant = containment.acquire(tree_spec(workspace), owner="session-1")
result = await containment.run(grant, command)
leader = read_pid(pidfile)
assert result.timed_out is True
assert result.release.dead is True
assert gone(leader)
assert containment._group_present(result.grant.pgid) is False
async def test_a_clean_exit_tears_down_anything_left_behind(tmp_path, workspace):
"""A command that returns while leaving a background process does not leak it.
The leftover closes its inherited pipes (``>/dev/null 2>&1``) so the command
really does complete: a background process still holding the output pipes
keeps the grant open until the wall clock, which is the previous test's case
rather than this one's.
"""
pidfile = tmp_path / "leftover.pid"
command = f"sleep 60 >/dev/null 2>&1 & echo $! > {pidfile}; exit 0"
grant = containment.acquire(tree_spec(workspace, wall_clock_s=10), owner="session-1")
result = await containment.run(grant, command)
leftover = read_pid(pidfile)
assert result.timed_out is False
assert result.exit_code == 0
assert gone(leftover), f"background process {leftover} outlived its grant"
assert result.release.dead is True
# ── Escalation ──────────────────────────────────────────────────────────────
def test_release_escalates_to_sigkill_and_reports_only_verified_death(tmp_path, workspace):
"""A SIGTERM-ignoring tree is escalated, and ``dead`` is set only once gone."""
# The child announces itself only after installing the handler. Without that
# the test races process startup and sometimes measures a child that was
# still using the default SIGTERM disposition.
ready = tmp_path / "ignoring-sigterm"
code = (
"import signal, time\n"
"signal.signal(signal.SIGTERM, signal.SIG_IGN)\n"
f"open({str(ready)!r}, 'w').write('x')\n"
"time.sleep(60)\n"
)
proc = subprocess.Popen(
[sys.executable, "-c", code],
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
start_new_session=True,
)
try:
deadline = time.monotonic() + 10
while time.monotonic() < deadline and not ready.exists():
time.sleep(0.02)
assert ready.exists(), "child never installed its SIGTERM handler"
grant = containment.acquire(tree_spec(workspace), owner="session-1")
grant = replace(grant, pid=proc.pid, pgid=os.getpgid(proc.pid))
outcome = containment.release(grant, grace_s=0.3)
proc.wait(timeout=5)
assert outcome.escalated is True
assert outcome.dead is True
assert outcome.survivors == ()
finally:
if proc.poll() is None: # pragma: no cover - only on an unexpected failure
proc.kill()
proc.wait(timeout=5)
def test_release_does_not_escalate_a_cooperative_tree(workspace):
proc = subprocess.Popen(
[sys.executable, "-c", "import time; time.sleep(60)"],
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
start_new_session=True,
)
try:
grant = containment.acquire(tree_spec(workspace), owner="session-1")
grant = replace(grant, pid=proc.pid, pgid=os.getpgid(proc.pid))
outcome = containment.release(grant, grace_s=2.0)
proc.wait(timeout=5)
assert outcome.dead is True
assert outcome.escalated is False
finally:
if proc.poll() is None: # pragma: no cover
proc.kill()
proc.wait(timeout=5)
def test_a_surviving_tree_keeps_its_record_active(workspace, monkeypatch):
"""A record moves to released only on verified death.
With teardown unable to signal anything, ``release`` must report
``dead=False`` with the survivors named, and must not mark the grant
released — the inverse of marking a job killed without checking.
"""
proc = subprocess.Popen(
[sys.executable, "-c", "import time; time.sleep(60)"],
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
start_new_session=True,
)
try:
grant = containment.acquire(tree_spec(workspace), owner="session-1")
grant = replace(grant, pid=proc.pid, pgid=os.getpgid(proc.pid))
monkeypatch.setattr(containment, "_signal_tree", lambda *args, **kwargs: None)
outcome = containment.release(grant, grace_s=0.2)
assert outcome.dead is False
assert outcome.escalated is True
assert proc.pid in outcome.survivors
active = {record["id"] for record in containment.active_grants()}
assert grant.id in active
assert pid_alive(proc.pid) is True
finally:
proc.kill()
proc.wait(timeout=5)
def test_release_never_signals_the_servers_own_process_group(workspace, monkeypatch):
"""If setsid had not applied, killpg would take the server down with the child.
Driven by handing teardown our own group id, which is exactly the state a
failed setsid would leave behind.
"""
sent = []
monkeypatch.setattr(os, "killpg", lambda pgid, sig: sent.append((pgid, sig)))
monkeypatch.setattr(os, "kill", lambda pid, sig: sent.append(("pid", pid, sig)))
containment._signal_tree(os.getpid(), os.getpgid(0), signal.SIGTERM)
assert all(entry[0] != os.getpgid(0) for entry in sent), sent
assert sent == [("pid", os.getpid(), signal.SIGTERM)]
def test_our_own_group_is_never_reported_as_a_childs_group():
assert containment._group_present(os.getpgid(0)) is False
# ── run(): the contained happy path ─────────────────────────────────────────
async def test_run_returns_output_exit_code_and_a_released_record(workspace):
grant = containment.acquire(tree_spec(workspace, wall_clock_s=10), owner="session-9")
result = await containment.run(grant, "echo contained; exit 3")
assert result.stdout == "contained\n"
assert result.exit_code == 3
assert result.timed_out is False
assert result.output_truncated is False
assert result.grant.pid is not None
assert containment.active_grants() == []
async def test_run_executes_in_the_workspace_and_with_the_declared_env_only(workspace):
grant = containment.acquire(
tree_spec(workspace, wall_clock_s=10, env={"PATH": "/usr/bin:/bin", "MARK": "yes"}),
owner="session-9",
)
result = await containment.run(grant, 'pwd; echo "MARK=$MARK"; echo "HOME=${HOME:-unset}"')
assert result.exit_code == 0
assert os.path.realpath(workspace) == os.path.realpath(result.stdout.splitlines()[0])
assert "MARK=yes" in result.stdout
# The child env is exactly what the spec declared, never an implicit
# inherit, so the server's own environment does not leak into it.
assert "HOME=unset" in result.stdout
async def test_run_caps_output_and_says_so(workspace):
grant = containment.acquire(
tree_spec(workspace, wall_clock_s=10, max_output_bytes=16), owner="session-9",
)
result = await containment.run(grant, "printf 'x%.0s' $(seq 1 500); echo")
assert result.output_truncated is True
assert len(result.stdout.encode("utf-8")) <= 16
async def test_run_accepts_an_argv_command_without_a_shell(workspace):
grant = containment.acquire(tree_spec(workspace, wall_clock_s=10), owner="session-9")
result = await containment.run(
grant, [sys.executable, "-c", "print('argv path')"], argv=True,
)
assert result.exit_code == 0
assert result.stdout.strip() == "argv path"
async def test_run_feeds_stdin_when_given(workspace):
grant = containment.acquire(tree_spec(workspace, wall_clock_s=10), owner="session-9")
result = await containment.run(grant, "cat", stdin=b"piped\n")
assert result.exit_code == 0
assert result.stdout == "piped\n"
@pytest.mark.parametrize("command, argv", [(" ", False), ([], True)])
async def test_run_rejects_an_empty_command(workspace, command, argv):
grant = containment.acquire(tree_spec(workspace, wall_clock_s=10), owner="session-9")
with pytest.raises(ValueError, match="empty"):
await containment.run(grant, command, argv=argv)
# ── Resource limits, where the platform provides them ───────────────────────
async def test_a_process_count_limit_is_applied_to_the_child(workspace):
"""RLIMIT_NPROC is set in the child, so the ceiling is real where it is claimed."""
spec = containment.ContainmentSpec(
workspace=workspace,
env={"PATH": "/usr/bin:/bin"},
wall_clock_s=10,
required=frozenset({
containment.WALL_CLOCK, containment.PROCESS_COUNT,
}),
max_processes=64,
)
grant = containment.acquire(spec, owner="session-9")
assert containment.PROCESS_COUNT in grant.enforced
result = await containment.run(
grant,
[sys.executable, "-c",
"import resource; print(resource.getrlimit(resource.RLIMIT_NPROC))"],
argv=True,
)
assert result.exit_code == 0
assert result.stdout.strip() == "(64, 64)"
async def test_a_memory_limit_is_claimed_only_where_it_can_be_applied(workspace):
"""The claim and the reality agree, on whichever platform this runs.
macOS reports an infinite RLIMIT_AS hard limit and then refuses to lower it,
so `memory` must come back unenforced there rather than enforced-and-crashing.
Written to assert the consistency rather than the platform, so it is a real
test on Linux and a real test here.
"""
limit = 2 * 1024 * 1024 * 1024
spec = containment.ContainmentSpec(
workspace=workspace,
env={"PATH": "/usr/bin:/bin"},
wall_clock_s=10,
required=frozenset({containment.WALL_CLOCK}),
max_memory_bytes=limit,
)
grant = containment.acquire(spec, owner="session-9")
probe = [sys.executable, "-c",
"import resource; print(resource.getrlimit(resource.RLIMIT_AS)[0])"]
result = await containment.run(grant, probe, argv=True)
assert result.exit_code == 0, result.stderr
if containment.MEMORY in grant.enforced:
assert result.stdout.strip() == str(limit)
else:
assert containment.MEMORY in grant.degraded
assert result.stdout.strip() != str(limit)
def test_an_unenforceable_required_limit_refuses_instead_of_crashing_the_spawn(workspace):
"""A limit this platform cannot apply is refused at acquire, not in preexec_fn.
Skipped where the platform *can* apply it, since then there is nothing to
refuse.
"""
if containment._ADDRESS_SPACE_LIMIT_SUPPORTED:
pytest.skip("this platform can lower RLIMIT_AS, so there is no shortfall")
spec = containment.ContainmentSpec(
workspace=workspace,
env={"PATH": "/usr/bin:/bin"},
wall_clock_s=10,
required=frozenset({
containment.WALL_CLOCK, containment.MEMORY,
}),
max_memory_bytes=2 * 1024 * 1024 * 1024,
)
with pytest.raises(containment.ContainmentUnavailable) as caught:
containment.acquire(spec, owner="session-9")
assert caught.value.missing == frozenset({containment.MEMORY})
def test_pid_alive_rejects_non_process_pids(monkeypatch):
calls = []
real_kill = os.kill
def fake_kill(pid, sig):
calls.append((pid, sig))
return real_kill(pid, sig)
monkeypatch.setattr(os, "kill", fake_kill)
# Required: None, 0, and any negative integers must return False
assert pid_alive(None) is False
assert pid_alive(0) is False
assert pid_alive(-1) is False
assert pid_alive(-42) is False
assert pid_alive(-9999) is False
assert pid_alive("not_a_pid") is False
# The underlying process probe (os.kill) must NEVER be invoked for pid <= 0
assert calls == []
# Normal positive-PID behavior remains covered
my_pid = os.getpid()
assert pid_alive(my_pid) is True
assert (my_pid, 0) in calls
def test_pid_alive_retains_conservative_liveness_on_eperm(monkeypatch):
def fake_kill(pid, sig):
raise PermissionError(1, "Operation not permitted")
monkeypatch.setattr(os, "kill", fake_kill)
# Positive PID with EPERM is conservatively considered alive
assert pid_alive(1) is True
assert pid_alive(99999) is True
# But non-process values still immediately return False without calling probe
assert pid_alive(None) is False
assert pid_alive(0) is False
assert pid_alive(-1) is False
assert pid_alive(-100) is False
def test_pid_alive_reports_false_on_process_lookup_error(monkeypatch):
def fake_kill(pid, sig):
raise ProcessLookupError(3, "No such process")
monkeypatch.setattr(os, "kill", fake_kill)
assert pid_alive(99999999) is False
+191 -35
View File
@@ -1,13 +1,26 @@
"""Stopping a Cookbook server must succeed on a host with no procfs.
"""Stopping a Cookbook server, on a host with procfs and on one without.
The tmux kill is what actually stops the server; the pid sweep that follows
it only catches model servers that survive the session's SIGHUP. On macOS and
Windows there is no ``/proc`` to sweep, and letting that raise turned a
successful stop into a reported failure *and* skipped the state write that
marks the session stopped for the Cookbook UI.
The tmux kill is what actually stops the server; the pid sweep that follows it
only catches model servers that survive the session's SIGHUP. Two invariants
live here.
**The stop must not fail because the host cannot be inspected.** Letting a
procfs scan raise on macOS turned a successful stop into a reported failure and
skipped the state write that marks the session stopped for the Cookbook UI
(ODY-94). Skipping the sweep silently fixed the crash and left the other half:
the stop then claimed success without having looked at all. So the sweep now
runs through ``ps`` where there is no procfs, and says so when it cannot look.
**The sweep signals only processes the session owns.** It used to kill anything
whose full command line matched the tracked one. The Cookbook composed that
command line, so an identical one is just as likely to be a server the user
started by hand — killing it is indistinguishable from killing ours, which is
the "stop only what we started" failure. Ownership now comes from the tmux
pane's process tree, captured before the kill; a lookalike is reported instead.
"""
import asyncio
import json
import os
import signal
import pytest
@@ -67,20 +80,34 @@ def _install_httpx_client(monkeypatch, state):
return posts
def _install_successful_tmux_kill(monkeypatch):
"""Replace the real ``tmux kill-session`` with a process that succeeds."""
def _install_successful_tmux_kill(monkeypatch, panes=""):
"""Fake the two tmux calls a stop makes: list-panes, then kill-session.
``panes`` is the ``list-panes`` stdout, i.e. ``"<session> <pane_pid>"`` per
line — the stop reads it to learn which processes the session owns before
the kill destroys that link.
"""
calls = []
class FakeProc:
returncode = 0
def __init__(self, stdout=b""):
self._stdout = stdout
async def communicate(self):
return b"", b""
return self._stdout, b""
async def fake_exec(*argv, **kwargs):
assert argv[:2] == ("tmux", "kill-session")
calls.append(argv)
assert argv[0] == "tmux"
if argv[1] == "list-panes":
return FakeProc(panes.encode())
assert argv[1] == "kill-session"
return FakeProc()
monkeypatch.setattr(asyncio, "create_subprocess_exec", fake_exec)
return calls
def _stopped_statuses(posts, session_id):
@@ -92,63 +119,192 @@ def _stopped_statuses(posts, session_id):
return out
def _fake_table(monkeypatch, rows):
"""Substitute the process table. ``rows`` is {pid: (ppid, command)}.
Returns the live dict, so a test can model a process actually dying by
removing it: ``start_token`` reads from the same dict, and a pid that is no
longer in it has no token, which :func:`process_ownership.verify` reports as
``GONE``.
"""
from src import process_ownership
table = {
pid: process_ownership.ProcessInfo(pid=pid, ppid=ppid, command=command)
for pid, (ppid, command) in rows.items()
}
monkeypatch.setattr(process_ownership, "process_table", lambda: dict(table))
# Identity is what authorises a signal, so every pid in the fake table has
# one. A pid absent from the table has no token and cannot be signalled.
monkeypatch.setattr(
process_ownership, "start_token",
lambda pid: f"token:{pid}" if int(pid or 0) in table else None,
)
return table
def _install_effective_kill(monkeypatch, table):
"""Record signals, and let SIGTERM actually remove the process.
Keeps the sweep off its escalation path, which would otherwise spend the
full SIGTERM grace plus the SIGKILL confirmation window on every pid.
"""
signalled = []
def _kill(pid, sig):
signalled.append((pid, sig))
table.pop(int(pid), None)
monkeypatch.setattr(os, "kill", _kill)
return signalled
@pytest.mark.asyncio
async def test_stop_marks_session_stopped_when_the_host_has_no_procfs(
monkeypatch, tmp_path
):
"""The ODY-94 regression: no procfs must not turn a working stop into a failure."""
state = _tracked_state()
posts = _install_httpx_client(monkeypatch, state)
_install_successful_tmux_kill(monkeypatch)
_install_successful_tmux_kill(monkeypatch, panes="serve-abc123 900\n")
monkeypatch.setattr(platform_compat, "PROC_ROOT", tmp_path / "no-procfs")
import os
def _unexpected_listdir(*args, **kwargs):
raise AssertionError("the pid sweep must not run without procfs")
monkeypatch.setattr(os, "listdir", _unexpected_listdir)
# ps is the mechanism on a procfs-less host; the sweep goes through it
# instead of being skipped.
_fake_table(monkeypatch, {900: (1, "bash")})
result = await tools.do_stop_served_model(
json.dumps({"session_id": "serve-abc123"})
)
assert result == {"output": "Stopped server serve-abc123", "exit_code": 0}
assert result["exit_code"] == 0
assert result["output"].startswith("Stopped server serve-abc123")
assert _stopped_statuses(posts, "serve-abc123") == ["stopped"]
@pytest.mark.asyncio
async def test_stop_sweeps_surviving_pids_when_procfs_is_present(
async def test_stop_says_so_when_the_session_cannot_be_inspected(
monkeypatch, tmp_path
):
tracked_cmd = "python -m vllm.entrypoints.openai.api_server --model org/model"
state = _tracked_state(cmd=tracked_cmd)
"""A sweep that could not look must not read as a sweep that found nothing.
This is the half of ODY-94 that the procfs guard left behind: skipping the
sweep stopped the crash and still reported plain success.
"""
from src import process_ownership
state = _tracked_state()
posts = _install_httpx_client(monkeypatch, state)
_install_successful_tmux_kill(monkeypatch)
_install_successful_tmux_kill(monkeypatch, panes="serve-abc123 900\n")
monkeypatch.setattr(platform_compat, "PROC_ROOT", tmp_path / "no-procfs")
proc = tmp_path / "proc"
def _no_inspection():
raise process_ownership.InspectionUnavailable("the process table")
def _write_pid(pid, cmdline):
entry = proc / pid
entry.mkdir(parents=True)
(entry / "cmdline").write_bytes(cmdline.replace(" ", "\0").encode())
_write_pid("101", tracked_cmd)
_write_pid("202", "python -m http.server")
(proc / "self").mkdir()
monkeypatch.setattr(platform_compat, "PROC_ROOT", proc)
monkeypatch.setattr(process_ownership, "process_table", _no_inspection)
signalled = []
import os
monkeypatch.setattr(os, "kill", lambda pid, sig: signalled.append((pid, sig)))
result = await tools.do_stop_served_model(
json.dumps({"session_id": "serve-abc123"})
)
assert result["exit_code"] == 0
assert "could not identify the session's processes" in result["output"]
assert signalled == []
assert _stopped_statuses(posts, "serve-abc123") == ["stopped"]
@pytest.mark.asyncio
async def test_stop_kills_the_sessions_own_survivor(monkeypatch, tmp_path):
"""A process under the session's pane is ours, so it gets signalled."""
tracked_cmd = "python -m vllm.entrypoints.openai.api_server --model org/model"
state = _tracked_state(cmd=tracked_cmd)
posts = _install_httpx_client(monkeypatch, state)
_install_successful_tmux_kill(monkeypatch, panes="serve-abc123 900\n")
table = _fake_table(monkeypatch, {
900: (1, "bash"), # the pane shell
101: (900, tracked_cmd), # the model server it started — ours
})
signalled = _install_effective_kill(monkeypatch, table)
result = await tools.do_stop_served_model(
json.dumps({"session_id": "serve-abc123"})
)
assert result["exit_code"] == 0
assert (101, signal.SIGTERM) in signalled
assert "killed 2 surviving process(es)" in result["output"]
assert _stopped_statuses(posts, "serve-abc123") == ["stopped"]
@pytest.mark.asyncio
async def test_stop_reports_a_command_line_lookalike_without_signalling_it(
monkeypatch, tmp_path
):
"""The headline change: matching the command line is not owning the process.
pid 202 runs exactly the tracked command but descends from nothing this
session started — a server the user launched by hand looks precisely like
this. The old sweep killed it.
"""
tracked_cmd = "python -m vllm.entrypoints.openai.api_server --model org/model"
state = _tracked_state(cmd=tracked_cmd)
posts = _install_httpx_client(monkeypatch, state)
_install_successful_tmux_kill(monkeypatch, panes="serve-abc123 900\n")
table = _fake_table(monkeypatch, {
900: (1, "bash"),
202: (1, tracked_cmd), # same command, different lineage
})
signalled = _install_effective_kill(monkeypatch, table)
result = await tools.do_stop_served_model(
json.dumps({"session_id": "serve-abc123"})
)
assert result["exit_code"] == 0
assert not any(pid == 202 for pid, _sig in signalled)
# Reported rather than silently dropped: the old behaviour acted on this
# information, so giving it up entirely would be a regression of its own.
assert "202" in result["output"]
assert "not signalled" in result["output"]
assert _stopped_statuses(posts, "serve-abc123") == ["stopped"]
@pytest.mark.asyncio
async def test_stop_does_not_signal_a_pid_whose_identity_changed(
monkeypatch, tmp_path
):
"""Captured before the kill, recycled before the sweep: do not signal it."""
from src import process_ownership
tracked_cmd = "python -m vllm.entrypoints.openai.api_server --model org/model"
state = _tracked_state(cmd=tracked_cmd)
posts = _install_httpx_client(monkeypatch, state)
_install_successful_tmux_kill(monkeypatch, panes="serve-abc123 900\n")
table = _fake_table(monkeypatch, {900: (1, "bash"), 101: (900, tracked_cmd)})
# pid 101's slot reads differently every time it is asked, so whatever the
# capture recorded, the sweep's re-check cannot match it: the pid was
# recycled in between. Every other pid keeps a stable identity.
drift = {"n": 0}
def _drifting_token(pid):
if int(pid) == 101:
drift["n"] += 1
return f"token:101:{drift['n']}"
return f"token:{pid}" if int(pid or 0) in table else None
monkeypatch.setattr(process_ownership, "start_token", _drifting_token)
signalled = _install_effective_kill(monkeypatch, table)
result = await tools.do_stop_served_model(
json.dumps({"session_id": "serve-abc123"})
)
assert result["exit_code"] == 0
# The pane shell is genuinely ours and is signalled; 101 never is.
assert not any(pid == 101 for pid, _sig in signalled)
assert _stopped_statuses(posts, "serve-abc123") == ["stopped"]
+308
View File
@@ -0,0 +1,308 @@
"""The filesystem execution boundary: what the namespace binds, and what a
spawn says when there is no namespace to bind it with.
Two separate defects, both on the agent shell/Python path.
**The writable /home bind.** The workspace namespace bound ``/home`` and
``/mnt`` read-write. On the one platform where the namespace engages at all,
a command inside it reached outside the workspace and wrote to the user's home
directory. That was measured on a Linux host with working bubblewrap by
running this repo's own argv, so it is not a source read. Binding the user's
whole home directory into a workspace-confinement namespace gives back most of
what the namespace was for.
**The silent downgrade.** ``namespaced or _replace_workspace_alias(...)`` chose
between a mount namespace and a regex, with nothing in the tool result saying
which one ran. The fallback is a naming convenience — it rewrites the literal
token ``/workspace`` in the command string — so a command that never mentions
``/workspace`` is untouched by it and runs on the host unrestricted. That is
every agent shell command on macOS, which is a platform this project is
maintained and run on.
The argv tests assert the argv rather than running it: bubblewrap does not
exist on macOS, and it does not work in Docker either without
``--privileged`` (default and ``seccomp=unconfined`` both give
"Creating new namespace failed", ``--cap-add=SYS_ADMIN`` gives
"pivot_root: Operation not permitted"). An argv assertion is what can honestly
be checked on this host; the execution evidence for the defect itself came from
a Linux host.
"""
import os
import shlex
import pytest
from src import containment
from src.agent_tools import subprocess_tools
from src.constants import WORKSPACE_MOUNT
@pytest.fixture
def workspace(tmp_path):
(tmp_path / "artifact.txt").write_text("x", encoding="utf-8")
return str(tmp_path)
def _argv(workspace, **kwargs):
"""The namespace argv, with bubblewrap forced present.
`shutil.which` is patched rather than skipped so the argv is asserted on
every platform the suite runs on — the bind flags are the finding, and they
are wrong independently of whether this host can execute them.
"""
import shutil as _shutil
original = _shutil.which
original_available = containment._bwrap_available
try:
containment._bwrap_available = lambda: True
_shutil.which = lambda name, *a, **kw: (
"/usr/bin/bwrap" if name == "bwrap" else original(name, *a, **kw)
)
wrapped = subprocess_tools._wrap_workspace_namespace("true", workspace, **kwargs)
finally:
_shutil.which = original
containment._bwrap_available = original_available
assert wrapped is not None, "forced bwrap should produce a namespace argv"
return shlex.split(wrapped)
def _bind_mode(argv, dest):
"""The bind flag immediately preceding ``src dest`` in the argv, or None."""
for index in range(len(argv) - 2):
if argv[index + 2] == dest and argv[index].startswith("--"):
return argv[index]
return None
# ── what the namespace binds ────────────────────────────────────────────────
@pytest.mark.skipif(os.name == "nt", reason="bwrap argv is POSIX-only")
def test_home_and_mnt_are_read_only(workspace):
argv = _argv(workspace)
assert _bind_mode(argv, "/home") == "--ro-bind"
assert _bind_mode(argv, "/mnt") == "--ro-bind"
@pytest.mark.skipif(os.name == "nt", reason="bwrap argv is POSIX-only")
def test_the_workspace_is_the_writable_bind(workspace):
argv = _argv(workspace)
assert _bind_mode(argv, WORKSPACE_MOUNT) == "--bind"
assert argv[argv.index("--bind")] == "--bind"
@pytest.mark.skipif(os.name == "nt", reason="bwrap argv is POSIX-only")
def test_the_workspace_stays_writable_at_its_real_host_path_too(workspace):
"""A command can carry the absolute host path, not only /workspace.
BashTool's /tmp redirect rewrites `/tmp/` to `<agent_cwd()>/.tmp/` before
the namespace is built, so the command bwrap receives already names the real
path. Those writes used to land because the workspace happened to sit under
the writable `/home` bind. With /home read-only they need the workspace's
own bind, or making /home read-only silently breaks every command that uses
a real host path.
"""
real = os.path.realpath(workspace)
argv = _argv(workspace)
assert _bind_mode(argv, real) == "--bind", (
f"expected a writable bind of {real}; argv was {argv}"
)
@pytest.mark.skipif(os.name == "nt", reason="bwrap argv is POSIX-only")
def test_no_dir_chain_is_created_inside_a_read_only_bind(monkeypatch, tmp_path):
"""mkdir inside a read-only mount fails and takes the namespace with it.
A workspace under `/home` or `/mnt` already has its parents, because the
argv mounted those roots. Emitting `--dir /home/someone` for it would be an
error, not a no-op.
"""
assert subprocess_tools._namespace_dir_chain("/home/someone/ws") == []
assert subprocess_tools._namespace_dir_chain("/mnt/data/ws") == []
assert subprocess_tools._namespace_dir_chain("/usr/share/ws") == []
# Somewhere the argv does not mount: the parents have to be created in the
# private tmpfs root.
assert subprocess_tools._namespace_dir_chain("/srv/agents/ws") == [
"--dir", "/srv", "--dir", "/srv/agents",
]
@pytest.mark.skipif(os.name == "nt", reason="bwrap argv is POSIX-only")
def test_reserved_destinations_are_never_bound_over(workspace, monkeypatch):
"""Overlaying the private root, tmpfs or the workspace mount with a host
directory undoes the namespace from inside the argv that builds it."""
for reserved in ("/", "/tmp", WORKSPACE_MOUNT, "/proc", "/etc", "/usr"):
assert reserved in subprocess_tools._NAMESPACE_RESERVED_DESTS
@pytest.mark.skipif(os.name == "nt", reason="bwrap argv is POSIX-only")
def test_a_workspace_at_a_reserved_destination_gets_no_extra_bind(monkeypatch):
"""`/tmp` as the workspace must not produce `--bind /tmp /tmp` after the
argv has already put a private tmpfs there."""
monkeypatch.setattr(os.path, "realpath", lambda path: "/tmp")
argv = _argv("/tmp")
# `--tmpfs /tmp` takes a destination only, so it is a two-arg pair.
pairs = list(zip(argv, argv[1:]))
assert ("--tmpfs", "/tmp") in pairs
assert _bind_mode(argv, "/tmp") is None, (
f"a host bind of /tmp would undo the private tmpfs; argv was {argv}"
)
# ── the silent downgrade ────────────────────────────────────────────────────
def test_fallback_reports_that_filesystem_containment_did_not_hold(
workspace, monkeypatch,
):
"""The fallback still runs under report-only — but it is now recorded.
Before this, the only difference between a contained run and a host run was
whether a regex had rewritten a token, and nothing in the result said so.
"""
monkeypatch.setattr(
subprocess_tools, "_wrap_workspace_namespace",
lambda *args, **kwargs: None,
)
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_REPORT_ONLY)
command, block, confined = subprocess_tools._contained_command(
"echo hi", workspace,
)
assert confined is False
assert command == "echo hi"
assert block["contained"] is False
assert block["executed"] is True
assert block["unenforced_required"] == [containment.FILESYSTEM]
assert block["mechanism"] == subprocess_tools.ALIAS_REWRITE_MECHANISM
assert block["mode"] == containment.MODE_REPORT_ONLY
def test_the_fallback_mechanism_is_not_named_like_a_mechanism(workspace, monkeypatch):
"""A string rewrite reported as "bubblewrap" or "none" is the same silence
with extra steps. It gets its own name so a reader cannot mistake it."""
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_REPORT_ONLY)
monkeypatch.setattr(
subprocess_tools, "_wrap_workspace_namespace",
lambda *args, **kwargs: None,
)
_command, block, _confined = subprocess_tools._contained_command("echo hi", workspace)
assert block["mechanism"] == "workspace_alias_rewrite"
assert block["mechanism"] not in {name.name for name in containment.MECHANISMS}
def test_enforcing_mode_refuses_instead_of_falling_back(workspace, monkeypatch):
"""Fail closed. Containment required and unavailable means not executed."""
monkeypatch.setattr(
subprocess_tools, "_wrap_workspace_namespace",
lambda *args, **kwargs: None,
)
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_ENFORCING)
with pytest.raises(containment.ContainmentUnavailable) as caught:
subprocess_tools._contained_command("echo hi", workspace)
assert caught.value.missing == frozenset({containment.FILESYSTEM})
def test_a_namespaced_command_reports_the_mechanism_that_established_it(
workspace, monkeypatch,
):
monkeypatch.setattr(
subprocess_tools, "_wrap_workspace_namespace",
lambda *args, **kwargs: "bwrap --whatever true",
)
monkeypatch.setattr(
containment, "probe",
lambda spec: containment.ContainmentProbe(
mechanism="bubblewrap",
enforced=frozenset({containment.FILESYSTEM, containment.PROCESS_TREE}),
degraded=(),
unenforced_required=(),
mode=containment.MODE_REPORT_ONLY,
),
)
command, block, confined = subprocess_tools._contained_command("true", workspace)
assert confined is True
assert command == "bwrap --whatever true"
assert block["mechanism"] == "bubblewrap"
assert block["contained"] is True
assert block["enforced"] == [containment.FILESYSTEM]
def test_the_reported_block_claims_only_the_filesystem_dimension(workspace, monkeypatch):
"""These tools still build their own create_subprocess_* call and pass
neither start_new_session nor a group-wide kill, so listing process_tree or
wall_clock here would be a false claim. The block names its own scope."""
monkeypatch.setattr(
subprocess_tools, "_wrap_workspace_namespace",
lambda *args, **kwargs: "bwrap --whatever true",
)
_command, block, _confined = subprocess_tools._contained_command("true", workspace)
assert block["reported_dimensions"] == [containment.FILESYSTEM]
assert containment.PROCESS_TREE not in block["enforced"]
assert containment.WALL_CLOCK not in block["enforced"]
# ── containment.probe: the single answer both tools ask for ─────────────────
def test_probe_answers_without_writing_a_grant_record(workspace, monkeypatch, tmp_path):
"""A grant record whose pid is never filled in and whose release never runs
is an entry a restart reaper keeps finding, which is why the decision does
not go through acquire()."""
store = tmp_path / "grants.json"
monkeypatch.setattr(containment, "CONTAINMENT_STATE_FILE", str(store), raising=False)
monkeypatch.setattr(containment, "_store_path", lambda: store)
probe = containment.probe(
containment.agent_spec(workspace=workspace, env={}, wall_clock_s=5)
)
assert probe.mechanism
assert not store.exists()
assert containment.active_grants() == []
def test_probe_and_acquire_agree_on_what_this_host_enforces(workspace, monkeypatch, tmp_path):
"""One mechanism table, one answer. A second opinion about what this host
can enforce is the thing the probe exists to prevent."""
store = tmp_path / "grants.json"
monkeypatch.setattr(containment, "_store_path", lambda: store)
spec = containment.agent_spec(workspace=workspace, env={}, wall_clock_s=5)
probe = containment.probe(spec)
grant = containment.acquire(spec, owner="test")
assert probe.mechanism == grant.mechanism
assert probe.enforced == grant.enforced
assert probe.unenforced_required == grant.unenforced_required
assert probe.contained == grant.contained
def test_probe_refuses_only_under_enforcing_mode(workspace, monkeypatch):
spec = containment.agent_spec(workspace=workspace, env={}, wall_clock_s=5)
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_REPORT_ONLY)
assert containment.probe(spec).refuses is False
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_ENFORCING)
probe = containment.probe(spec)
# Only meaningful where something required is actually missing; on a host
# with bubblewrap nothing is.
assert probe.refuses == bool(probe.unenforced_required)
def test_probe_rejects_a_malformed_spec_in_either_mode(tmp_path):
missing = tmp_path / "not-a-directory"
with pytest.raises(ValueError, match="not a directory"):
containment.probe(
containment.agent_spec(workspace=str(missing), env={}, wall_clock_s=5)
)
# ── the isolated /tmp stand-in ──────────────────────────────────────────────
def test_isolated_tmp_is_created_inside_the_workspace(workspace):
path = subprocess_tools._isolated_tmp_dir(workspace)
assert os.path.isdir(path)
assert os.path.realpath(path).startswith(os.path.realpath(workspace))
def test_isolated_tmp_degrades_instead_of_raising_on_an_unwritable_workspace(
workspace, monkeypatch,
):
"""The source tree is read-only in Docker and a workspace can be mounted
read-only. A command that merely mentions `/tmp/` must not die with an
OSError traceback because a scratch directory could not be made."""
def _refuse(*args, **kwargs):
raise OSError(30, "Read-only file system")
monkeypatch.setattr(os, "makedirs", _refuse)
path = subprocess_tools._isolated_tmp_dir(workspace)
assert path == os.path.join(workspace, ".tmp")
+191
View File
@@ -0,0 +1,191 @@
"""Native execution must use the shared boundary and report actual teardown."""
import asyncio
import os
import sys
import pytest
from src import containment, tool_execution
from src.agent_tools import subprocess_tools
@pytest.fixture(autouse=True)
def native_boundary(tmp_path, monkeypatch):
monkeypatch.setattr(tool_execution, "agent_cwd", lambda: str(tmp_path))
monkeypatch.setattr(containment, "_store_path", lambda: tmp_path / "grants.json")
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_REPORT_ONLY)
monkeypatch.setattr(containment, "MECHANISMS", tuple(
m for m in containment.MECHANISMS if m.name == "process_group"
))
return tmp_path
@pytest.mark.skipif(os.name == "nt", reason="real POSIX group teardown")
async def test_native_bash_owns_and_releases_its_child(native_boundary):
result = await subprocess_tools.BashTool().execute(
"if read answer; then echo unexpected; else printf '%s' \"$ODY_TEST_ENV\"; fi",
{"subproc_env": {"PATH": "/usr/bin:/bin", "ODY_TEST_ENV": "captured"}},
)
assert result["output"] == "captured"
assert result["exit_code"] == 0
assert result["teardown"]["dead"] is True
assert result["containment"]["enforced"] == ["wall_clock"]
assert result["containment"]["unenforced_required"] == ["filesystem", "process_tree"]
assert result["containment"]["network"] == "inherit"
assert containment.active_grants() == []
async def test_native_bash_refuses_before_spawn_when_required_boundary_missing(monkeypatch):
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_ENFORCING)
async def forbidden(*args, **kwargs):
pytest.fail("refused command reached spawn")
monkeypatch.setattr(asyncio, "create_subprocess_exec", forbidden)
result = await subprocess_tools.BashTool().execute("echo hello", {})
assert result["containment"]["executed"] is False
assert result["containment"]["unenforced_required"] == ["filesystem", "process_tree"]
async def test_failed_spawn_releases_unstarted_grant(native_boundary, monkeypatch):
async def fail(*args, **kwargs):
raise OSError("spawn failed")
monkeypatch.setattr(asyncio, "create_subprocess_exec", fail)
result = await subprocess_tools.BashTool().execute("echo hello", {})
assert result["exit_code"] == 1
assert containment.active_grants() == []
@pytest.mark.skipif(os.name == "nt", reason="real POSIX process")
async def test_long_line_is_drained_and_truncation_reported(native_boundary):
spec = containment.ContainmentSpec(
workspace=str(native_boundary), env=dict(os.environ), wall_clock_s=5,
required=frozenset({containment.PROCESS_TREE, containment.WALL_CLOCK}),
max_output_bytes=100,
)
result = await containment.run(containment.acquire(spec, owner="long-line"),
[sys.executable, "-c", "print('x' * 200000)"], argv=True)
assert result.exit_code == 0
assert result.stdout == "x" * 100
assert result.output_truncated is True
assert result.release.dead is True
@pytest.mark.skipif(os.name == "nt", reason="real POSIX process")
async def test_output_exactly_at_cap_is_complete(native_boundary):
spec = containment.ContainmentSpec(
workspace=str(native_boundary), env=dict(os.environ), wall_clock_s=5,
required=frozenset({containment.PROCESS_TREE, containment.WALL_CLOCK}), max_output_bytes=100,
)
result = await containment.run(containment.acquire(spec, owner="exact-cap"),
[sys.executable, "-c", "import sys; sys.stdout.write('x' * 100)"], argv=True)
assert len(result.stdout) == 100
assert result.output_truncated is False
def test_permission_denied_is_not_verified_death(monkeypatch):
from core import platform_compat
def denied(*args):
raise PermissionError("EPERM")
monkeypatch.setattr(platform_compat, "IS_WINDOWS", False)
monkeypatch.setattr(os, "kill", denied)
monkeypatch.setattr(os, "killpg", denied)
monkeypatch.setattr(containment, "_own_pgid", lambda: 1)
assert platform_compat.pid_alive(987654) is True
assert containment._group_present(987654) is True
@pytest.mark.skipif(os.name == "nt", reason="real POSIX process")
async def test_blocked_stdin_is_inside_wall_clock(native_boundary):
spec = containment.ContainmentSpec(
workspace=str(native_boundary), env=dict(os.environ), wall_clock_s=1,
required=frozenset({containment.PROCESS_TREE, containment.WALL_CLOCK}),
)
result = await asyncio.wait_for(containment.run(
containment.acquire(spec, owner="blocked-stdin"), "sleep 60", stdin=b"x" * 2000000,
), timeout=8)
assert result.timed_out is True
assert result.release.dead is True
@pytest.mark.parametrize("source", ["print(1 + 1)", "import os; print(os.getcwd())",
"exec('print(2)')", "print('/workspace')"])
@pytest.mark.skipif(os.name == "nt", reason="POSIX namespace argv; Windows refusal tested separately")
async def test_python_namespace_is_independent_of_content(source, native_boundary, monkeypatch):
from tests.containment_helpers import capture_owned_spawn
captured = capture_owned_spawn(monkeypatch, native_boundary)
monkeypatch.setattr(containment, "MECHANISMS", (containment.Mechanism(
"bubblewrap", 30, lambda: True, lambda spec: containment.DEFAULT_REQUIRED,
),))
original_which = containment.shutil.which
monkeypatch.setattr(containment.shutil, "which", lambda name:
"/usr/bin/bwrap" if name == "bwrap" else original_which(name))
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_ENFORCING)
result = await subprocess_tools.PythonTool().execute(source, {})
assert os.path.basename(captured["argv"][0]) == "bwrap"
assert "--bind" in captured["argv"]
assert result["containment"]["enforced"] == sorted(containment.DEFAULT_REQUIRED)
assert "-I" in captured["argv"]
async def test_ordinary_python_cannot_bypass_unavailable_containment(monkeypatch):
monkeypatch.setattr(containment, "CONTAINMENT_MODE", containment.MODE_ENFORCING)
async def forbidden(*args, **kwargs):
pytest.fail("ordinary Python bypassed required containment")
monkeypatch.setattr(asyncio, "create_subprocess_exec", forbidden)
result = await subprocess_tools.PythonTool().execute("print(1 + 1)", {})
assert result["containment"]["executed"] is False
async def test_python_final_expression_and_opt_in_imports(native_boundary):
package = native_boundary / "packages"
package.mkdir()
(package / "demo.py").write_text("value = 42\n")
result = await subprocess_tools.PythonTool().execute("import demo; demo.value", {
"subproc_env": {**os.environ, "ODYSSEUS_PYTHON_TOOL_SITE_PACKAGES": str(package)},
})
assert result["output"] == "42"
assert result["teardown"]["dead"] is True
async def test_capture_preserves_multibyte_text_across_chunks(native_boundary):
spec = containment.ContainmentSpec(
workspace=str(native_boundary), env=dict(os.environ), wall_clock_s=5,
required=frozenset({containment.PROCESS_TREE, containment.WALL_CLOCK}), max_output_bytes=200000,
)
result = await containment.run(containment.acquire(spec, owner="unicode"),
[sys.executable, "-c", "import sys; sys.stdout.write('€' * 30000)"], argv=True)
assert result.stdout == "€" * 30000
assert result.output_truncated is False
async def test_chat_bash_captures_more_than_2000_lines(native_boundary):
result = await subprocess_tools.BashTool().execute(
"printf 'START\\n'; for i in $(seq 1 3000); do printf 'x\\n'; done; printf 'END\\n'",
{"session_id": "large-output-chat"},
)
assert result["output"].startswith("START\n")
assert result["output"].endswith("\nEND")
assert len(result["output"].splitlines()) == 3002
assert result["output_truncated"] is False
assert result["teardown"]["dead"] is True
async def test_chat_bash_reports_capture_limit_as_incomplete(native_boundary):
result = await subprocess_tools.BashTool().execute(
"for i in $(seq 1 12000); do printf 'x\\n'; done", {"session_id": "capped-chat"},
)
assert result["exit_code"] == 0
assert result["output_truncated"] is True
assert "truncated" in result["output"]
assert result["teardown"]["dead"] is True
async def test_repeated_chat_calls_refresh_environment(native_boundary):
first = await subprocess_tools.BashTool().execute('printf "%s" "$VALUE"', {
"session_id": "same-chat", "subproc_env": {"PATH": "/usr/bin:/bin", "VALUE": "first"},
})
second = await subprocess_tools.BashTool().execute('printf "%s" "$VALUE"', {
"session_id": "same-chat", "subproc_env": {"PATH": "/usr/bin:/bin", "VALUE": "second"},
})
assert first["output"] == "first"
assert second["output"] == "second"
assert first["containment"]["id"] != second["containment"]["id"]
+436
View File
@@ -0,0 +1,436 @@
"""Teardown across a restart: the pid in a store is a claim, not a handle.
Three stores here outlive the process that wrote them, deliberately — a restart
is supposed to keep a background job and its result. The consequence nobody had
closed is that the recorded pid is reassignable, so a teardown driven off an old
record can land on a process the kernel has since given to somebody else. That
was ODY-86's shape.
Covered:
* :func:`src.containment.release` gating a grant it recovered from the store,
and *not* gating one whose process the caller is holding.
* :func:`src.containment.reap_record`, the entry point for a reaper that has a
row and no grant object.
* :mod:`src.process_reaper`, which gives the two stores opposite treatment —
orphaned grants are torn down, detached jobs are only corrected.
* :func:`src.bg_jobs.disown_unverified`.
The verdicts come from :mod:`src.process_ownership`, substituted here so each
case is driven exactly; that module's own tests pin it against real processes.
"""
import json
import pytest
from src import bg_jobs, containment, process_ownership, process_reaper
@pytest.fixture
def grant_store(tmp_path, monkeypatch):
"""Redirect the containment grant store. Returns a reader for it."""
path = tmp_path / "containment_grants.json"
monkeypatch.setattr(containment, "_store_path", lambda: path)
def _read():
return json.loads(path.read_text(encoding="utf-8")) if path.exists() else {}
return _read
@pytest.fixture
def job_store(tmp_path, monkeypatch):
"""Redirect the background-job store and its spool directory."""
monkeypatch.setattr(bg_jobs, "_STORE", tmp_path / "bg_jobs.json")
monkeypatch.setattr(bg_jobs, "_JOBS_DIR", tmp_path / "bg_jobs")
(tmp_path / "bg_jobs").mkdir()
def verdicts(monkeypatch, mapping, default=process_ownership.OWNED):
"""Pin verify()'s answer per pid."""
monkeypatch.setattr(
process_ownership, "verify",
lambda pid, _token: mapping.get(int(pid or 0), default),
)
def seed_grant(pid=4242, pgid=4242, token="token:4242", **extra):
"""Write a grant record the way a previous run would have left it."""
record = {
"id": "grant-1", "owner": "session-7", "mechanism": "process_group",
"mode": containment.CONTAINMENT_MODE, "workspace": "/tmp",
"enforced": ["filesystem", "process_tree", "wall_clock"],
"degraded": [], "unenforced_required": [],
"required": ["filesystem", "process_tree", "wall_clock"],
"wall_clock_s": 60, "max_memory_bytes": None, "max_processes": None,
"network": "inherit", "external": False, "pid": pid, "pgid": pgid,
"start_token": token, "acquired_at": 0.0, "released_at": None, "release": None,
}
record.update(extra)
containment._save_records({record["id"]: record})
return record
def seed_job(pid=4242, token="token:4242", status="running", **extra):
record = {
"id": "job-1", "session_id": "chat-1", "command": "sleep 300",
"status": status, "pid": pid, "start_token": token, "started_at": 0.0,
"ended_at": None, "exit_code": None, "max_runtime_s": 3600,
"followed_up": False, "log_path": "", "exit_path": "",
}
record.update(extra)
bg_jobs._save({record["id"]: record})
return record
# ── The gate inside release() ───────────────────────────────────────────────
def test_a_recovered_grant_naming_a_recycled_pid_is_not_signalled(
grant_store, monkeypatch
):
"""The headline case. The pid is live, and it is not ours."""
seed_grant()
verdicts(monkeypatch, {4242: process_ownership.FOREIGN})
def _no_signals(*_args, **_kwargs):
raise AssertionError("a foreign pid must never be signalled")
monkeypatch.setattr(containment, "_signal_tree", _no_signals)
outcome = containment.reap_record(grant_store()["grant-1"])
assert outcome.ownership == process_ownership.FOREIGN
assert outcome.dead is False
# Not listed as a survivor of *our* grant either: naming a stranger's pid
# there invites the next reaper to kill it.
assert outcome.survivors == ()
def test_an_unverifiable_grant_is_not_signalled_and_stays_active(
grant_store, monkeypatch
):
"""An inspection mechanism this host does not have is a containment failure.
Reported as an undead tree and left in the store, so the orphan stays
visible in ``active_grants()`` rather than being written off as handled.
"""
seed_grant()
verdicts(monkeypatch, {4242: process_ownership.UNVERIFIABLE})
monkeypatch.setattr(
containment, "_signal_tree",
lambda *_a, **_k: pytest.fail("an unidentified pid must never be signalled"),
)
outcome = containment.reap_record(grant_store()["grant-1"])
assert outcome.ownership == process_ownership.UNVERIFIABLE
assert outcome.dead is False
assert containment.active_grants(), "the orphan must remain visible"
def test_a_recovered_grant_whose_process_is_gone_is_released_clean(
grant_store, monkeypatch
):
seed_grant()
verdicts(monkeypatch, {4242: process_ownership.GONE})
monkeypatch.setattr(containment, "_group_present", lambda _pgid: False)
outcome = containment.reap_record(grant_store()["grant-1"])
assert outcome.dead is True
assert outcome.ownership == process_ownership.GONE
assert containment.active_grants() == []
def test_a_gone_leader_with_a_live_group_is_reported_not_killed(
grant_store, monkeypatch
):
"""Children outlive the leader, but with the leader gone nothing proves the
group is still ours — and a recycled group id would mean killpg hits
strangers. The orphan is reported instead of guessed at."""
seed_grant()
verdicts(monkeypatch, {4242: process_ownership.GONE})
monkeypatch.setattr(containment, "_group_present", lambda _pgid: True)
monkeypatch.setattr(
containment, "_signal_tree",
lambda *_a, **_k: pytest.fail("an unprovable group must not be signalled"),
)
outcome = containment.reap_record(grant_store()["grant-1"])
assert outcome.dead is False
assert outcome.survivors == (4242,)
def test_a_verified_grant_is_torn_down_normally(grant_store, monkeypatch):
seed_grant()
verdicts(monkeypatch, {4242: process_ownership.OWNED})
monkeypatch.setattr(containment, "_pgid_of", lambda pid: 4242)
signals = []
monkeypatch.setattr(
containment, "_signal_tree",
lambda pid, pgid, sig: signals.append((pid, pgid, sig)),
)
# Dead on the first probe, so the teardown does not wait out its grace.
monkeypatch.setattr(containment, "_tree_gone", lambda *_a, **_k: True)
outcome = containment.reap_record(grant_store()["grant-1"])
assert outcome.dead is True
assert outcome.ownership == ""
def test_verified_leader_does_not_authorize_a_different_group(grant_store, monkeypatch):
seed_grant(pgid=9999)
verdicts(monkeypatch, {4242: process_ownership.OWNED})
monkeypatch.setattr(containment, "_pgid_of", lambda pid: 4242)
monkeypatch.setattr(containment, "_signal_tree", lambda *args: pytest.fail("foreign group signalled"))
outcome = containment.reap_record(grant_store()["grant-1"])
assert outcome.dead is False
assert outcome.ownership == process_ownership.UNVERIFIABLE
def test_reaper_retains_a_group_after_its_leader_dies(grant_store, monkeypatch):
from src import process_reaper
seed_grant()
verdicts(monkeypatch, {4242: process_ownership.GONE})
monkeypatch.setattr(containment, "_group_present", lambda pgid: True)
monkeypatch.setattr(containment, "_signal_tree", lambda *args: pytest.fail("unidentified group signalled"))
report = process_reaper.reap_containment_grants()
assert report["failed"] == 1
assert report["already_gone"] == 0
assert "grant-1" in grant_store()
def test_an_in_process_grant_is_not_subjected_to_the_gate(monkeypatch, tmp_path):
"""A grant carrying its own pid belongs to the caller holding it.
The caller launched the child, so there is no identity question — and
demanding a token here would refuse teardown of a perfectly ordinary tool
call on a host with no inspection mechanism.
"""
monkeypatch.setattr(containment, "_store_path", lambda: tmp_path / "grants.json")
monkeypatch.setattr(
process_ownership, "verify",
lambda *_a, **_k: pytest.fail("an in-process teardown must not consult ownership"),
)
spec = containment.ContainmentSpec(
workspace=str(tmp_path), env={}, wall_clock_s=5,
)
grant = containment.ContainmentGrant(
id="live-1", mechanism="process_group", workspace=str(tmp_path),
enforced=containment.DEFAULT_REQUIRED, degraded=(), unenforced_required=(),
owner="session-7", mode=containment.CONTAINMENT_MODE, spec=spec,
pid=4242, pgid=4242,
)
monkeypatch.setattr(containment, "_tree_gone", lambda *_a, **_k: True)
monkeypatch.setattr(containment, "_group_present", lambda _pgid: False)
outcome = containment.release(grant)
assert outcome.dead is True
assert outcome.ownership == ""
def test_the_release_block_names_the_ownership_verdict(grant_store, monkeypatch):
"""The verdict reaches the record, so "why is this still here" is answerable."""
seed_grant()
verdicts(monkeypatch, {4242: process_ownership.FOREIGN})
outcome = containment.reap_record(grant_store()["grant-1"])
assert outcome.to_dict()["ownership"] == process_ownership.FOREIGN
assert grant_store()["grant-1"]["release"]["ownership"] == process_ownership.FOREIGN
# ── The reaper ──────────────────────────────────────────────────────────────
def test_the_reaper_drops_a_foreign_grant_without_signalling_it(
grant_store, job_store, monkeypatch
):
seed_grant()
verdicts(monkeypatch, {4242: process_ownership.FOREIGN})
monkeypatch.setattr(
containment, "_signal_tree",
lambda *_a, **_k: pytest.fail("the reaper must not signal a foreign pid"),
)
report = process_reaper.reap_containment_grants()
assert report["foreign"] == 1
# Dropped rather than retried: the only thing left to do with a record
# about someone else's process is stop believing it.
assert containment.active_grants() == []
def test_the_reaper_keeps_an_unverifiable_grant_visible(
grant_store, job_store, monkeypatch
):
seed_grant()
verdicts(monkeypatch, {4242: process_ownership.UNVERIFIABLE})
report = process_reaper.reap_containment_grants()
assert report["unverifiable"] == 1
assert len(containment.active_grants()) == 1
def test_the_reaper_tears_down_a_verified_orphan(grant_store, job_store, monkeypatch):
"""A live process under an abandoned grant has no caller left. It goes."""
seed_grant()
verdicts(monkeypatch, {4242: process_ownership.OWNED})
torn_down = []
def _reap(record, **_kwargs):
torn_down.append(record["id"])
return containment.ReleaseOutcome(dead=True, escalated=True, mechanism="process_group")
monkeypatch.setattr(containment, "reap_record", _reap)
report = process_reaper.reap_containment_grants()
assert torn_down == ["grant-1"]
assert report["torn_down"] == 1
assert containment.active_grants() == []
def test_the_reaper_keeps_a_grant_that_survived_its_teardown(
grant_store, job_store, monkeypatch
):
seed_grant()
verdicts(monkeypatch, {4242: process_ownership.OWNED})
monkeypatch.setattr(
containment, "reap_record",
lambda record, **_k: containment.ReleaseOutcome(
dead=False, escalated=True, survivors=(4242,), mechanism="process_group",
),
)
report = process_reaper.reap_containment_grants()
assert report["failed"] == 1
assert len(containment.active_grants()) == 1
def test_the_reaper_forgets_an_external_grant_without_inspecting_anything(
grant_store, job_store, monkeypatch
):
"""Nothing local ever ran, so there is nothing local to reap."""
seed_grant(external=True)
monkeypatch.setattr(
process_ownership, "verify",
lambda *_a, **_k: pytest.fail("an external grant has no local pid to verify"),
)
report = process_reaper.reap_containment_grants()
assert report["already_gone"] == 1
assert containment.active_grants() == []
def test_the_reaper_survives_an_unreadable_store(monkeypatch, job_store):
monkeypatch.setattr(
containment, "active_grants",
lambda: (_ for _ in ()).throw(RuntimeError("store on fire")),
)
assert process_reaper.reap_containment_grants()["seen"] == 0
# ── Background jobs: corrected, never killed ────────────────────────────────
def test_a_job_whose_pid_was_reassigned_is_retired_unsignalled(
job_store, monkeypatch
):
"""Left alone, refresh() would SIGKILL this pid at max-runtime.
An hour after a restart, aimed at whatever now holds it.
"""
seed_job()
verdicts(monkeypatch, {4242: process_ownership.FOREIGN})
monkeypatch.setattr(
bg_jobs, "_kill",
lambda *_a, **_k: pytest.fail("disowning a job must not signal anything"),
)
report = bg_jobs.disown_unverified()
assert report == {"seen": 1, "retired": 1, "kept": 0}
record = bg_jobs._load()["job-1"]
assert record["status"] == "failed"
assert record["ownership_lost"] == process_ownership.FOREIGN
# The agent asked for this job and is still owed an answer.
assert record["followed_up"] is False
def test_an_unverifiable_job_is_also_retired(job_store, monkeypatch):
"""Fail closed. Retiring loses a result, which is visible; keeping it leaves
a pid this server will later signal without knowing what it points at."""
seed_job(token=None)
verdicts(monkeypatch, {4242: process_ownership.UNVERIFIABLE})
assert bg_jobs.disown_unverified()["retired"] == 1
def test_a_job_that_is_still_ours_keeps_running(job_store, monkeypatch):
"""A detached job is documented to survive a restart. Killing it here would
break the feature the store exists for."""
seed_job()
verdicts(monkeypatch, {4242: process_ownership.OWNED})
report = bg_jobs.disown_unverified()
assert report == {"seen": 1, "retired": 0, "kept": 1}
assert bg_jobs._load()["job-1"]["status"] == "running"
def test_a_job_whose_process_is_gone_is_left_for_refresh(job_store, monkeypatch):
"""refresh() may still find an exit-code file the job wrote before it went,
so retiring it here would discard a result that exists."""
seed_job()
verdicts(monkeypatch, {4242: process_ownership.GONE})
assert bg_jobs.disown_unverified()["kept"] == 1
assert bg_jobs._load()["job-1"]["status"] == "running"
def test_already_finished_jobs_are_not_reconsidered(job_store, monkeypatch):
seed_job(status="done")
verdicts(monkeypatch, {4242: process_ownership.FOREIGN})
assert bg_jobs.disown_unverified() == {"seen": 0, "retired": 0, "kept": 0}
def test_a_launched_job_records_an_identity_next_to_its_pid(job_store):
"""Without this the record is unverifiable forever and the reaper can only
refuse — the token has to be captured at launch or not at all."""
record = bg_jobs.launch("true", "chat-1")
assert "start_token" in record
assert process_ownership.verify(record["pid"], record["start_token"]) in (
process_ownership.OWNED, process_ownership.GONE,
)
def test_an_abandoned_job_says_so_in_its_follow_up(job_store):
"""The agent is told the job was lost, not that it failed for its own reasons."""
record = seed_job(status="failed", ownership_lost=process_ownership.FOREIGN)
text = bg_jobs.result_text(record)
assert "abandoned across a server restart" in text
assert "neither waited on nor signalled" in text
def test_reap_orphans_reports_both_stores_and_the_mechanism(
grant_store, job_store, monkeypatch
):
seed_grant()
seed_job()
verdicts(monkeypatch, {4242: process_ownership.GONE})
monkeypatch.setattr(process_reaper, "reap_legacy_agent_tmux", lambda: {"seen": 0})
monkeypatch.setattr(containment, "_group_present", lambda _pgid: False)
report = process_reaper.reap_orphans()
assert report["mechanism"] == process_ownership.inspection_mechanism()
assert report["grants"]["already_gone"] == 1
assert report["bg_jobs"]["kept"] == 1
+225
View File
@@ -0,0 +1,225 @@
"""The single filesystem confinement boundary.
Nine test files in this suite each prove one call site confines correctly, and
each call site had its own ``realpath``/``commonpath`` pair to prove it about.
This file covers the one implementation they now all go through, so a property
is asserted once instead of nine times and inconsistently.
Each test names the detail the scattered copies disagreed on. The macOS tests
are the ones with history: ``/tmp`` is a symlink to ``/private/tmp`` there, and
comparing a canonicalized candidate against a root that was not canonicalized
has already produced a false failure in this suite.
"""
import os
import sys
import pytest
from src.path_confinement import (
PathEscape,
canonical_root,
confine,
is_inside,
)
@pytest.fixture
def root(tmp_path):
"""A real directory, canonicalized the way a caller's root should be.
tmp_path is under ``/private/var/...`` on macOS via a ``/var`` symlink, so
this fixture is itself an instance of the aliasing the module exists to
handle — which is why it is used as-is rather than pre-resolved.
"""
(tmp_path / "inside.txt").write_text("x", encoding="utf-8")
(tmp_path / "sub").mkdir()
return str(tmp_path)
# ── the basic shape ─────────────────────────────────────────────────────────
def test_path_under_the_root_resolves(root):
assert confine(root, "inside.txt") == os.path.join(canonical_root(root), "inside.txt")
def test_relative_candidate_joins_the_root_not_the_process_cwd(root, tmp_path, monkeypatch):
"""abspath() of a relative path silently uses os.getcwd().
A confinement helper that does that resolves against whatever directory the
server happens to be running in, which is the wrong base before the
comparison even starts.
"""
elsewhere = tmp_path.parent / "elsewhere"
elsewhere.mkdir()
monkeypatch.chdir(elsewhere)
assert confine(root, "inside.txt") == os.path.join(canonical_root(root), "inside.txt")
def test_the_root_itself_is_inside_by_default(root):
assert confine(root, root) == canonical_root(root)
def test_allow_root_false_excludes_the_root(root):
"""For an operation only meaningful on something *under* the root."""
with pytest.raises(PathEscape):
confine(root, root, allow_root=False)
assert confine(root, "sub", allow_root=False)
def test_a_target_that_does_not_exist_yet_is_confined_not_refused(root):
"""A write target is a legitimate thing to confine.
realpath is the non-strict kind: it resolves what exists and normalizes the
rest, so a new file under the root passes while a new file above it does
not.
"""
assert confine(root, "not-created-yet.txt").startswith(canonical_root(root))
with pytest.raises(PathEscape):
confine(root, "../not-created-yet.txt")
# ── traversal ───────────────────────────────────────────────────────────────
def test_dotdot_escape_is_refused(root):
with pytest.raises(PathEscape):
confine(root, "../outside.txt")
with pytest.raises(PathEscape):
confine(root, "sub/../../outside.txt")
def test_absolute_candidate_outside_the_root_is_refused(root):
with pytest.raises(PathEscape):
confine(root, os.path.dirname(canonical_root(root)))
def test_sibling_with_a_shared_prefix_is_not_inside(tmp_path):
"""`/a/bc` begins with `/a/b` and is not inside it.
This is why the boundary uses commonpath and not startswith. A copy written
with startswith accepts the sibling.
"""
(tmp_path / "b").mkdir()
(tmp_path / "bc").mkdir()
(tmp_path / "bc" / "f.txt").write_text("x", encoding="utf-8")
assert not is_inside(tmp_path / "b", tmp_path / "bc" / "f.txt")
assert is_inside(tmp_path / "bc", tmp_path / "bc" / "f.txt")
# ── symlinks ────────────────────────────────────────────────────────────────
@pytest.mark.skipif(sys.platform.startswith("win"), reason="POSIX symlinks")
def test_symlink_as_the_final_component_is_followed_before_the_check(root, tmp_path):
outside = tmp_path.parent / "outside-secret.txt"
outside.write_text("secret", encoding="utf-8")
os.symlink(outside, os.path.join(root, "link.txt"))
with pytest.raises(PathEscape):
confine(root, "link.txt")
@pytest.mark.skipif(sys.platform.startswith("win"), reason="POSIX symlinks")
def test_symlinked_intermediate_directory_is_followed_before_the_check(root, tmp_path):
outside_dir = tmp_path.parent / "outside-dir"
outside_dir.mkdir()
(outside_dir / "f.txt").write_text("secret", encoding="utf-8")
os.symlink(outside_dir, os.path.join(root, "hop"))
with pytest.raises(PathEscape):
confine(root, "hop/f.txt")
@pytest.mark.skipif(sys.platform.startswith("win"), reason="POSIX symlinks")
def test_symlink_pointing_back_inside_the_root_is_allowed(root):
os.symlink(os.path.join(root, "inside.txt"), os.path.join(root, "loop.txt"))
assert confine(root, "loop.txt") == os.path.join(canonical_root(root), "inside.txt")
# ── the aliasing class that has already cost real time ──────────────────────
@pytest.mark.skipif(
not os.path.islink("/tmp"), reason="needs a platform where /tmp is a symlink",
)
def test_root_reached_through_a_symlink_still_contains_its_own_files(tmp_path):
"""macOS: /tmp is a symlink to /private/tmp.
A root given as `/tmp/x` and a candidate that canonicalizes to
`/private/tmp/x/f` describe the same file. Canonicalizing one side and not
the other reads as an escape and refuses a legitimate access — the false
false failure this suite has already recorded. Canonicalizing *neither*
side would agree, which is why the rule is both or nothing.
"""
unresolved_root = os.path.join("/tmp", os.path.basename(str(tmp_path)))
os.makedirs(unresolved_root, exist_ok=True)
try:
target = os.path.join(unresolved_root, "f.txt")
with open(target, "w", encoding="utf-8") as handle:
handle.write("x")
assert os.path.realpath(unresolved_root) != unresolved_root, (
"fixture assumption: /tmp should not canonicalize to itself here"
)
# Either spelling of the root, either spelling of the candidate.
assert is_inside(unresolved_root, target)
assert is_inside(unresolved_root, os.path.realpath(target))
assert is_inside(os.path.realpath(unresolved_root), target)
finally:
try:
os.remove(os.path.join(unresolved_root, "f.txt"))
os.rmdir(unresolved_root)
except OSError:
pass
def test_canonical_root_is_idempotent(root):
once = canonical_root(root)
assert canonical_root(once) == once
# ── malformed input is a reason, not an accidental "outside" ────────────────
@pytest.mark.parametrize("bad", ["", " ", None])
def test_empty_candidate_is_a_value_error_not_an_escape(root, bad):
with pytest.raises(ValueError) as caught:
confine(root, bad)
assert not isinstance(caught.value, PathEscape)
assert "required" in str(caught.value)
def test_nul_is_refused_with_a_reason(root):
with pytest.raises(ValueError, match="NUL"):
confine(root, "a\x00b")
def test_newline_is_refused_with_a_reason(root):
with pytest.raises(ValueError, match="newline"):
confine(root, "a\nb")
def test_is_inside_is_false_for_malformed_input_rather_than_raising(root):
"""The predicate form never raises; that is why callers wrapped the old
copies in `except Exception` and reached "outside" by accident."""
for bad in ("", None, "a\x00b", "a\nb", 17, object()):
assert is_inside(root, bad) is False
assert is_inside(None, "x") is False
def test_path_escape_is_a_value_error(root):
"""The call sites this replaces raised ValueError and their callers catch
it as such, so the subclass relationship is part of the contract."""
assert issubclass(PathEscape, ValueError)
with pytest.raises(ValueError):
confine(root, "../elsewhere")
def test_path_escape_names_the_root_and_the_candidate(root):
with pytest.raises(PathEscape) as caught:
confine(root, "../elsewhere")
message = str(caught.value)
assert "../elsewhere" in message
assert canonical_root(root) in message
# ── the deny list is somebody else's job ────────────────────────────────────
def test_confinement_does_not_decide_whether_a_path_is_sensitive(root):
"""`.ssh` inside the root is inside the root.
Confinement answers "inside"; the sensitive-file deny list answers
"allowed", and it stays with src.tool_execution, which owns that policy.
Folding the two together here is how a boundary acquires a second job and
then disagrees with itself.
"""
secret = os.path.join(root, ".ssh")
os.makedirs(secret, exist_ok=True)
assert is_inside(root, os.path.join(secret, "id_rsa"))
+47 -8
View File
@@ -6,14 +6,20 @@ so a symlink placed inside PERSONAL_DIR pointing outside it passes the
os.path.commonpath confinement check and lets index_personal_documents read
files outside the root. os.path.realpath resolves the symlink before the check.
_resolve_allowed_personal_dir is a closure inside setup_personal_routes, so the
source-level test pins the fix and the behavioural test proves the underlying
confinement principle.
_resolve_allowed_personal_dir is a closure inside setup_personal_routes, so it
cannot be imported and called directly. The resolution now happens in
src.path_confinement, so the behavioural test runs against that boundary and
the source-level test is reduced to the one thing still worth pinning here:
this closure must not grow its own abspath-based check again.
"""
import ast
import os
from pathlib import Path
import pytest
from src.path_confinement import PathEscape, confine
SRC = Path(__file__).resolve().parent.parent / "routes" / "personal_routes.py"
@@ -25,16 +31,49 @@ def _function_source(src_text, name):
raise AssertionError(f"{name} not found in {SRC}")
def test_confinement_uses_realpath_not_abspath():
def test_confinement_does_not_rely_on_abspath():
"""The resolver must not reach a confinement verdict through abspath.
Originally this asserted the presence of the literal ``os.path.realpath``.
The resolution now happens inside ``src.path_confinement.confine``, which
is the point — one boundary instead of a copy per call site — so the
literal is gone while the behaviour is unchanged. What is still worth
pinning at the source level is the negative: this closure must not grow its
own abspath-based check again.
"""
body = _function_source(SRC.read_text(), "_resolve_allowed_personal_dir")
assert "os.path.realpath" in body, (
"_resolve_allowed_personal_dir must use os.path.realpath so a symlink "
"inside PERSONAL_DIR cannot escape the confinement check"
)
assert "os.path.abspath" not in body, (
"os.path.abspath does not resolve symlinks; the confinement check must "
"not rely on it"
)
assert "confine(" in body, (
"the resolver must go through the shared confinement boundary rather "
"than reimplementing one"
)
def test_shared_boundary_refuses_a_symlink_out_of_the_base(tmp_path):
"""The behaviour the source assertion used to stand in for.
A symlink inside the base pointing outside it is refused, and the file it
points at is not reachable through it. This is asserted against the
boundary the resolver now calls, so it covers every call site that shares
it rather than this one closure.
"""
base = tmp_path / "personal"
base.mkdir()
outside = tmp_path / "outside"
outside.mkdir()
(outside / "secret.txt").write_text("nope", encoding="utf-8")
os.symlink(outside, base / "escape")
with pytest.raises(PathEscape):
confine(base, "escape")
with pytest.raises(PathEscape):
confine(base, "escape/secret.txt")
# A real directory inside the base is still reachable.
(base / "real").mkdir()
assert confine(base, "real") == os.path.join(os.path.realpath(base), "real")
def test_realpath_catches_symlink_escape(tmp_path):
+333
View File
@@ -0,0 +1,333 @@
"""Process identity: the four verdicts, and that the fourth is never permissive.
Split in two. The verdict tests use **real processes**, because the claim under
test is about the kernel's behaviour — a pid that has been reaped, a pid that was
never issued, a pid whose start time differs from the one recorded — and a fake
process table cannot be wrong about that in the same ways. The mechanism tests
substitute the inspection layer, so both the procfs branch and the ``ps`` branch
are exercised on whichever kind of host happens to be running them.
The invariant worth most here is negative: :data:`process_ownership.OWNED` is the
only verdict that permits a signal, and nothing — a missing token, an absent
mechanism, a probe that raised — may produce it by default. Process inspection
has broken off Linux four times in this tree (ODY-70, -86, -94, -99), every time
because an absent mechanism read as a successful answer.
"""
import os
import subprocess
import pytest
from core import platform_compat
from src import process_ownership as po
@pytest.fixture
def sleeper():
"""A real, short-lived child in its own session. Always reaped."""
procs = []
def _spawn(argv=("sleep", "30")):
proc = subprocess.Popen(list(argv), start_new_session=True)
procs.append(proc)
return proc
yield _spawn
for proc in procs:
try:
proc.kill()
proc.wait(timeout=5)
except Exception:
pass
# ── Verdicts, against real processes ────────────────────────────────────────
def test_a_live_process_with_its_own_token_is_owned(sleeper):
proc = sleeper()
token = po.start_token(proc.pid)
assert token
assert po.verify(proc.pid, token) == po.OWNED
def test_the_same_pid_with_a_different_token_is_foreign(sleeper):
"""The whole point: a pid is a slot, and the token says who is in it."""
proc = sleeper()
token = po.start_token(proc.pid)
assert po.verify(proc.pid, str(token) + "-not-this-one") == po.FOREIGN
def test_a_reaped_process_is_gone(sleeper):
proc = sleeper()
token = po.start_token(proc.pid)
proc.kill()
proc.wait(timeout=5)
assert po.verify(proc.pid, token) == po.GONE
def test_a_pid_that_was_never_issued_is_gone():
# Above any plausible pid_max, so this cannot collide with a real process.
assert po.verify(2 ** 30, "token:anything") == po.GONE
def test_a_missing_token_is_unverifiable_and_never_owned(sleeper):
"""A record that captured no identity cannot acquire one afterwards.
This is the pre-upgrade record, and the reason it must not be OWNED is that
treating "we did not write it down" as "it is ours" is what makes a recycled
pid lethal.
"""
proc = sleeper()
assert po.verify(proc.pid, None) == po.UNVERIFIABLE
assert po.verify(proc.pid, "") == po.UNVERIFIABLE
assert po.UNVERIFIABLE not in po.SIGNALLABLE
def test_only_owned_permits_a_signal():
assert po.SIGNALLABLE == frozenset({po.OWNED})
def test_a_falsy_pid_is_gone_rather_than_unverifiable():
"""Nothing to identify and nothing to signal; the record is just empty."""
assert po.verify(None, "token:x") == po.GONE
assert po.verify(0, "token:x") == po.GONE
def test_capture_always_returns_both_fields(sleeper):
proc = sleeper()
captured = po.capture(proc.pid)
assert set(captured) == {"pid", "start_token"}
assert captured["pid"] == proc.pid
assert po.verify(captured["pid"], captured["start_token"]) == po.OWNED
def test_this_process_verifies_as_itself():
assert po.verify(os.getpid(), po.start_token(os.getpid())) == po.OWNED
# ── An unavailable mechanism is a failure, not a default ────────────────────
def test_no_inspection_mechanism_yields_unverifiable(monkeypatch, sleeper):
proc = sleeper()
token = po.start_token(proc.pid)
monkeypatch.setattr(po, "inspection_mechanism", lambda: po.MECHANISM_NONE)
# Not GONE (which would abandon a live process) and not OWNED (which would
# license a signal at an unidentified one).
assert po.verify(proc.pid, token) == po.UNVERIFIABLE
def test_no_mechanism_makes_start_token_raise_rather_than_return_none(monkeypatch):
"""None means "no such process". A question we could not ask is not that."""
monkeypatch.setattr(po, "inspection_mechanism", lambda: po.MECHANISM_NONE)
with pytest.raises(po.InspectionUnavailable):
po.start_token(os.getpid())
def test_a_probe_that_raises_is_unverifiable_not_owned(monkeypatch, sleeper):
proc = sleeper()
def _broken(_pid):
raise po.InspectionUnavailable("deliberately broken probe")
monkeypatch.setattr(po, "start_token", _broken)
assert po.verify(proc.pid, "token:whatever") == po.UNVERIFIABLE
def test_capture_records_no_token_rather_than_failing(monkeypatch):
"""A host that cannot identify its children must still be able to launch.
The record then reads UNVERIFIABLE forever, which is the honest outcome:
the launch is allowed, and the later teardown refuses.
"""
monkeypatch.setattr(po, "inspection_mechanism", lambda: po.MECHANISM_NONE)
captured = po.capture(4242)
assert captured == {"pid": 4242, "start_token": None}
assert po.verify(4242, captured["start_token"]) == po.UNVERIFIABLE
def test_process_table_raises_without_any_mechanism(monkeypatch):
monkeypatch.setattr(po, "inspection_mechanism", lambda: po.MECHANISM_NONE)
with pytest.raises(po.InspectionUnavailable):
po.process_table()
def test_the_procfs_table_guards_its_own_scan(monkeypatch, tmp_path):
"""Guarded in the function that scans, not only in its caller.
tests/test_procfs_scan_guard.py pins this structurally; this pins the
behaviour, so calling the branch directly on a procfs-less host raises
instead of FileNotFoundError.
"""
monkeypatch.setattr(platform_compat, "PROC_ROOT", tmp_path / "absent")
with pytest.raises(po.InspectionUnavailable):
po._procfs_process_table()
# ── Mechanism selection ─────────────────────────────────────────────────────
def test_procfs_is_preferred_where_it_exists(monkeypatch, tmp_path):
procfs = tmp_path / "proc"
procfs.mkdir()
monkeypatch.setattr(platform_compat, "PROC_ROOT", procfs)
monkeypatch.setattr(po, "PROC_ROOT", procfs)
monkeypatch.setattr(po, "IS_WINDOWS", False)
assert po.inspection_mechanism() == po.MECHANISM_PROCFS
def test_ps_covers_hosts_with_no_procfs(monkeypatch, tmp_path):
"""macOS and the BSDs. The reason this module is not another /proc scan."""
monkeypatch.setattr(platform_compat, "PROC_ROOT", tmp_path / "absent")
monkeypatch.setattr(po, "IS_WINDOWS", False)
monkeypatch.setattr(po.shutil, "which", lambda name: "/bin/ps" if name == "ps" else None)
assert po.inspection_mechanism() == po.MECHANISM_PS
assert po.inspection_available()
def test_a_host_with_neither_reports_none(monkeypatch, tmp_path):
monkeypatch.setattr(platform_compat, "PROC_ROOT", tmp_path / "absent")
monkeypatch.setattr(po, "IS_WINDOWS", False)
monkeypatch.setattr(po.shutil, "which", lambda _name: None)
assert po.inspection_mechanism() == po.MECHANISM_NONE
assert not po.inspection_available()
def test_the_procfs_token_reads_a_comm_containing_spaces_and_parens(monkeypatch, tmp_path):
"""``/proc/<pid>/stat`` field 2 is attacker-adjacent: it is the executable name.
A process called ``my (weird) prog`` would shift every field after it if the
parser split on whitespace, which would silently read the wrong number as the
start time and make every verdict wrong.
"""
procfs = tmp_path / "proc"
(procfs / "77").mkdir(parents=True)
# "77 (comm) S" are fields 1-3, so the filler starts numbering at 4 and
# each value equals its own field number.
fields = " ".join(str(index) for index in range(4, 54))
(procfs / "77" / "stat").write_text(f"77 (my (weird) prog) S {fields}\n")
boot_path = procfs / "sys/kernel/random/boot_id"
boot_path.parent.mkdir(parents=True)
boot_path.write_text("first-boot\n")
monkeypatch.setattr(platform_compat, "PROC_ROOT", procfs)
monkeypatch.setattr(po, "PROC_ROOT", procfs)
monkeypatch.setattr(po, "IS_WINDOWS", False)
assert po.start_token(77) == "procfs:first-boot:22"
token = po.start_token(77)
boot_path.write_text("second-boot\n")
assert po.start_token(77) != token
# ── The process tree ────────────────────────────────────────────────────────
def test_descendants_walks_a_real_tree(sleeper):
"""The grandchild case: a shell that backgrounds work and the work itself."""
proc = sleeper(("bash", "-c", "sleep 30 & sleep 30"))
# Wait for the shell to have actually forked, without sleeping on a clock:
# poll the table until the children appear or the attempts run out.
found = []
for _attempt in range(100):
found = po.descendants([proc.pid])
if len(found) >= 3:
break
assert proc.pid in found
assert len(found) >= 3, f"expected the shell and its two children, got {found}"
def test_descendants_includes_the_root_even_with_no_children(sleeper):
proc = sleeper()
assert po.descendants([proc.pid]) == [proc.pid]
def test_descendants_takes_a_single_table_snapshot(monkeypatch):
"""One table in, one answer out — reparenting cannot hide a process.
Walking the tree with a fresh query per level lets a child be reparented
between queries and drop out of the result, which for a teardown means a
process nobody signals.
"""
rows = {
10: po.ProcessInfo(pid=10, ppid=1, command="root"),
11: po.ProcessInfo(pid=11, ppid=10, command="child"),
12: po.ProcessInfo(pid=12, ppid=11, command="grandchild"),
13: po.ProcessInfo(pid=13, ppid=1, command="unrelated"),
}
def _explode():
raise AssertionError("descendants must use the table it was given")
monkeypatch.setattr(po, "process_table", _explode)
assert po.descendants([10], table=rows) == [10, 11, 12]
def test_descendants_terminates_on_a_parent_cycle():
"""A table can report a cycle; the walk must not spin on it."""
rows = {
20: po.ProcessInfo(pid=20, ppid=21, command="a"),
21: po.ProcessInfo(pid=21, ppid=20, command="b"),
}
assert sorted(po.descendants([20], table=rows)) == [20, 21]
def test_the_procfs_table_reads_the_parent_pid(monkeypatch, tmp_path):
"""The procfs branch of the tree walk, exercised on a host without procfs.
macOS runs this suite and takes the ``ps`` branch, so without a substituted
``/proc`` the Linux parse — which is what the deployed image uses — would be
covered by nothing.
"""
procfs = tmp_path / "proc"
for pid, ppid in ((10, 1), (11, 10)):
(procfs / str(pid)).mkdir(parents=True)
(procfs / str(pid) / "cmdline").write_bytes(f"proc-{pid}\0--flag\0".encode())
filler = " ".join(str(index) for index in range(5, 54))
(procfs / str(pid) / "stat").write_text(f"{pid} (proc) S {ppid} {filler}\n")
monkeypatch.setattr(platform_compat, "PROC_ROOT", procfs)
monkeypatch.setattr(po, "PROC_ROOT", procfs)
monkeypatch.setattr(po, "IS_WINDOWS", False)
table = po.process_table()
assert table[11].ppid == 10
assert table[10].command == "proc-10 --flag"
assert po.descendants([10], table=table) == [10, 11]
def test_a_procfs_row_with_an_unreadable_stat_keeps_its_command_line(
monkeypatch, tmp_path
):
"""A kernel thread or a pid that exits mid-walk still matters to a
command-line match; dropping the row entirely would hide it."""
procfs = tmp_path / "proc"
(procfs / "12").mkdir(parents=True)
(procfs / "12" / "cmdline").write_bytes(b"orphan-cmd\0")
monkeypatch.setattr(platform_compat, "PROC_ROOT", procfs)
monkeypatch.setattr(po, "PROC_ROOT", procfs)
monkeypatch.setattr(po, "IS_WINDOWS", False)
table = po.process_table()
assert table[12].command == "orphan-cmd"
assert table[12].ppid == 0
def test_command_lines_sees_this_process():
table = po.command_lines()
assert os.getpid() in table
assert table[os.getpid()]
+197
View File
@@ -0,0 +1,197 @@
"""Tests verifying truthful representation of production external bridge execution.
P2-2 invariant: external bridge execution != local containment.
"""
import asyncio
from types import SimpleNamespace
from unittest.mock import patch
import pytest
from src import containment
from src.agent_tools import subprocess_tools
from src import tool_execution as _te
class _FakeResponse:
def __init__(self, data, status_code=200):
self._data = data
self.status_code = status_code
self.text = "error detail" if status_code >= 400 else ""
def json(self):
return self._data
class _FakeAsyncClient:
def __init__(self, *args, **kwargs):
pass
async def __aenter__(self):
return self
async def __aexit__(self, *exc):
return None
async def post(self, url, *args, **kwargs):
return _FakeResponse({"stdout": "remote stdout", "stderr": "", "exit_code": 0})
@pytest.fixture(autouse=True)
def _isolated_store(tmp_path, monkeypatch):
store = tmp_path / "containment_grants.json"
monkeypatch.setattr(containment, "_store_path", lambda: store)
return store
@pytest.mark.asyncio
async def test_host_shell_creates_uncontained_external_record(monkeypatch):
"""1. Production bridge execution creates an external/uncontained record.
2. It cannot be interpreted as contained."""
monkeypatch.setattr(subprocess_tools.httpx, "AsyncClient", _FakeAsyncClient)
tool = subprocess_tools.HostShellTool()
ctx = {
"client_runtime_context": {
"host_shell_bridge": {
"url": "http://127.0.0.1:17654/run",
"token": "secret-bridge-token",
}
},
"session_id": "test-session-host-shell",
}
result = await tool.execute('{"command": "echo host"}', ctx)
assert result["exit_code"] == 0
assert result["output"] == "remote stdout"
assert "containment" in result
c = result["containment"]
# Invariant: external bridge execution != local containment
assert c["external"] is True
assert c["contained"] is False
assert c["mechanism"] == "external_bridge"
assert c["enforced"] == []
assert c["executed"] is True
# Server-owned metadata identifies endpoint without secrets
assert c.get("endpoint") == "http://127.0.0.1:17654/run"
assert "secret-bridge-token" not in str(c)
@pytest.mark.asyncio
async def test_host_shell_failure_does_not_become_containment_or_effect_evidence(monkeypatch):
"""5. Bridge failure does not become successful containment/effect evidence."""
class _FailingClient(_FakeAsyncClient):
async def post(self, url, *args, **kwargs):
return _FakeResponse({"error": "bridge exploded"}, status_code=500)
monkeypatch.setattr(subprocess_tools.httpx, "AsyncClient", _FailingClient)
tool = subprocess_tools.HostShellTool()
ctx = {
"client_runtime_context": {
"host_shell_bridge": {
"url": "http://127.0.0.1:17654/run",
"token": "secret-bridge-token",
}
},
"session_id": "test-session-host-shell-fail",
}
result = await tool.execute('{"command": "echo fail"}', ctx)
assert result["exit_code"] == 1
assert "bridge returned HTTP 500" in result["error"]
assert "containment" in result
c = result["containment"]
assert c["external"] is True
assert c["contained"] is False
assert c["enforced"] == []
# Failure means not executed
assert c["executed"] is False
@pytest.mark.asyncio
async def test_routed_bash_and_python_via_bridge_creates_uncontained_external_record():
"""Prove _route_tool_via_bridge generates truthful external records for bash & python."""
bridge_ctx = {
"surface": "odysseus-tui",
"host_shell_bridge": {
"url": "http://127.0.0.1:17654/run",
"token": "bridge-token",
},
}
async def fake_bridge_post(bridge, path, payload, **kwargs):
return {"stdout": "bridge out", "stderr": "", "exit_code": 0}
from tests.runtime_evidence_helpers import server_authorized_executor
with patch.object(_te, "_bridge_post", fake_bridge_post), \
patch.object(_te, "_owner_is_admin", lambda owner: True):
# Routed bash
desc, result = await server_authorized_executor(_te.execute_tool_block)(
SimpleNamespace(tool_type="bash", content="ls -la"),
session_id="session-routed-bash",
client_runtime_context=bridge_ctx,
security_context=_te.NO_TOOL_SECURITY_CONTEXT,
)
assert result["exit_code"] == 0
assert "containment" in result
cb = result["containment"]
assert cb["external"] is True
assert cb["contained"] is False
assert cb["mechanism"] == "external_bridge"
assert cb["enforced"] == []
assert cb["executed"] is True
assert cb.get("endpoint") == "http://127.0.0.1:17654/run"
# Routed python
desc_py, result_py = await server_authorized_executor(_te.execute_tool_block)(
SimpleNamespace(tool_type="python", content="print('hi')"),
session_id="session-routed-py",
client_runtime_context=bridge_ctx,
security_context=_te.NO_TOOL_SECURITY_CONTEXT,
)
assert result_py["exit_code"] == 0
assert "containment" in result_py
cp = result_py["containment"]
assert cp["external"] is True
assert cp["contained"] is False
assert cp["mechanism"] == "external_bridge"
assert cp["enforced"] == []
assert cp["executed"] is True
@pytest.mark.asyncio
async def test_external_record_does_not_grant_authority(tmp_path):
"""3. The record does not grant execution authority."""
spec = containment.agent_spec(str(tmp_path), {}, 5)
grant = containment.declare_external_bridge(
spec, owner="auth-test-session", endpoint="http://127.0.0.1:17654/run"
)
assert grant.external is True
assert grant.contained is False
# Attempting to use this grant to run local command must be rejected
with pytest.raises(ValueError, match="backend does not own"):
await containment.run(grant, "id")
@pytest.mark.asyncio
async def test_native_local_bash_python_behavior_unchanged(tmp_path, monkeypatch):
"""4. Native local Bash/Python behavior is unchanged."""
tool_bash = subprocess_tools.BashTool()
ctx = {
"session_id": "native-session",
}
result = await tool_bash.execute("echo 'native run'", ctx)
assert result["exit_code"] == 0
assert "native run" in result["output"]
assert "containment" in result
c = result["containment"]
assert c["external"] is False
assert c["mechanism"] in ("bubblewrap", "process_group")
assert c["executed"] is True
+7 -1
View File
@@ -333,6 +333,8 @@ def test_approved_document_version_guard_rejects_changed_target():
@pytest.mark.asyncio
async def test_missing_sealed_document_does_not_fall_back_to_another(monkeypatch):
import sys
from types import ModuleType
import src.agent_tools.document_tools as document_tools
class FakeDb:
@@ -342,7 +344,11 @@ async def test_missing_sealed_document_does_not_fall_back_to_another(monkeypatch
def rollback(self):
pass
monkeypatch.setattr("src.database.SessionLocal", lambda: FakeDb())
database = ModuleType("src.database")
database.SessionLocal = lambda: FakeDb()
database.Document = object
database.DocumentVersion = object
monkeypatch.setitem(sys.modules, "src.database", database)
monkeypatch.setattr(
document_tools,
"_get_owned_document",
+17 -6
View File
@@ -2198,8 +2198,9 @@ def test_python_loaded_code_sees_virtual_workspace_alias(monkeypatch, tmp_path):
import venv
from types import SimpleNamespace
if not shutil.which("bwrap"):
return
from src import containment
if not containment._bwrap_available():
pytest.skip("functional bubblewrap namespaces unavailable")
from pathlib import Path
@@ -2210,12 +2211,13 @@ def test_python_loaded_code_sees_virtual_workspace_alias(monkeypatch, tmp_path):
venv.EnvBuilder(with_pip=False).create(environment)
monkeypatch.setattr(subprocess_tools, "sys", SimpleNamespace(
prefix=str(environment),
base_prefix=sys.base_prefix,
executable=str(environment / "bin" / "python"),
version_info=sys.version_info,
))
script = workspace / ".python-workspace-alias-test.py"
output = workspace / ".python-workspace-alias-test.txt"
outside = workspace / "host-sibling.txt"
outside = workspace.parent / "host-sibling.txt"
outside.write_text("must stay hidden from private /tmp")
script.write_text(
"from pathlib import Path; "
@@ -2237,6 +2239,7 @@ def test_workspace_namespace_mounts_only_a_nested_python_environment(monkeypatch
from src.agent_tools import subprocess_tools
monkeypatch.setattr(subprocess_tools.shutil, "which", lambda name: "/usr/bin/bwrap")
monkeypatch.setattr(subprocess_tools.containment, "_bwrap_available", lambda: True)
environment = tmp_path / "nested" / "venv"
environment.mkdir(parents=True)
(environment / "pyvenv.cfg").write_text("home = /usr/bin\n")
@@ -2258,16 +2261,23 @@ def test_workspace_namespace_rejects_broad_or_symlinked_python_prefixes(monkeypa
from src.agent_tools import subprocess_tools
monkeypatch.setattr(subprocess_tools.shutil, "which", lambda name: "/usr/bin/bwrap")
monkeypatch.setattr(subprocess_tools.containment, "_bwrap_available", lambda: True)
linked_root = tmp_path / "linked-root"
linked_root.symlink_to("/", target_is_directory=True)
# Compared against the argv with no interpreter prefix at all: an unsafe
# prefix must add *nothing*. Asserting the absence of a literal
# `--ro-bind <prefix> <prefix>` instead would also fire on a base mount the
# argv makes for its own reasons -- /home and /mnt are read-only binds
# there -- which says nothing about whether the prefix was rejected.
baseline = shlex.split(
subprocess_tools._wrap_workspace_namespace("echo ok", str(tmp_path))
)
for unsafe_prefix in ("/", "/tmp", "/var", "/home", str(linked_root)):
command = subprocess_tools._wrap_workspace_namespace(
"echo ok", str(tmp_path), interpreter_prefix=unsafe_prefix,
)
args = shlex.split(command)
assert ["--ro-bind", unsafe_prefix, unsafe_prefix] not in [
args[index:index + 3] for index in range(len(args) - 2)
]
assert args == baseline, f"prefix {unsafe_prefix} changed the namespace argv"
assert ["--tmpfs", "/tmp"] in [
args[index:index + 2] for index in range(len(args) - 1)
]
@@ -2283,6 +2293,7 @@ def test_workspace_namespace_preserves_the_64_bit_dynamic_loader(monkeypatch):
from src.agent_tools import subprocess_tools
monkeypatch.setattr(subprocess_tools.shutil, "which", lambda name: "/usr/bin/bwrap")
monkeypatch.setattr(subprocess_tools.containment, "_bwrap_available", lambda: True)
command = subprocess_tools._wrap_workspace_namespace("echo ok", "/tmp/workspace")
assert command is not None
args = shlex.split(command)
+2 -1
View File
@@ -287,7 +287,8 @@ async def test_subprocess_cwd_is_workspace_e2e(ws, admin):
"""python tool runs with cwd = workspace (OS-agnostic probe)."""
_, r = await execute_tool_block(_block("python", "import os; print(os.getcwd())"), owner="a", workspace=ws)
assert r["exit_code"] == 0
assert os.path.realpath(r["output"].strip()) == os.path.realpath(ws)
expected_cwd = "/workspace" if "filesystem" in r["containment"]["enforced"] else ws
assert os.path.realpath(r["output"].strip()) == os.path.realpath(expected_cwd)
@pytest.mark.asyncio
+7 -7
View File
@@ -52,7 +52,7 @@ The source tree reads **109** `ODYSSEUS_*` variables: 79 an operator may want to
| Variable | Default | Read in | What it does |
|---|---|---|---|
| `ODYSSEUS_DATA_DIR` | `get_default_data_dir()` | `src/constants.py:56` (+1 more) | Root directory for every persisted file. Prefer this over the per-path overrides; the rest of `src/constants.py` derives from it. |
| `ODYSSEUS_MAIL_ATTACHMENTS_DIR` | `os.path.join(DATA_DIR, 'mail-attachments')` | `src/constants.py:102` | Dedicated override for the mail attachment store, which otherwise lives under the data directory. |
| `ODYSSEUS_MAIL_ATTACHMENTS_DIR` | `os.path.join(DATA_DIR, 'mail-attachments')` | `src/constants.py:103` | Dedicated override for the mail attachment store, which otherwise lives under the data directory. |
### Model routing and providers
@@ -75,7 +75,7 @@ The source tree reads **109** `ODYSSEUS_*` variables: 79 an operator may want to
| `ODYSSEUS_MAX_VISUAL_EVIDENCE_FRAMES` | `'3'` | `src/agent_loop.py:15362` | How many video frames one tool result may contribute. Clamped to 1-8. |
| `ODYSSEUS_MAX_VISUAL_EVIDENCE_IMAGES` | `'1'` | `src/agent_loop.py:15330` | How many images one tool result may contribute to the model turn. Clamped to 1-8. |
| `ODYSSEUS_MCP_ALLOWED_COMMANDS` | `''` | `src/agent_tools/admin_tools.py:140` | Security-relevant. Comma-separated allowlist of MCP launcher basenames the agent may start. Empty by default, and the deny list still wins. |
| `ODYSSEUS_PYTHON_TOOL_SITE_PACKAGES` | `''` | `src/agent_tools/subprocess_tools.py:931` | Security-relevant. Absolute package roots, separated by the platform path separator, exposed to the sandboxed Python tool. Empty exposes none. |
| `ODYSSEUS_PYTHON_TOOL_SITE_PACKAGES` | `''` | `src/agent_tools/subprocess_tools.py:853` (+1 more) | Security-relevant. Absolute package roots, separated by the platform path separator, exposed to the sandboxed Python tool. Empty exposes none. |
| `ODYSSEUS_SCRIPT_HOST` | `'localhost'` | `src/builtin_actions.py:919` | Default host for the run-script action. `localhost`, `127.0.0.1`, `local` and empty run locally; any other value runs over SSH. |
| `ODYSSEUS_TOOL_APPROVAL_GATE` | `'0'` | `src/tool_capabilities.py:645` | Security-relevant. Truthy makes tool calls pass through the approval gate. Off by default. |
@@ -152,15 +152,15 @@ The source tree reads **109** `ODYSSEUS_*` variables: 79 an operator may want to
| Variable | Default | Read in | What it does |
|---|---|---|---|
| `ODYSSEUS_SKILL_SEMANTIC_RETRIEVAL` | `'1'` | `services/memory/skills.py:789` | On by default. Set 0, false, no or off to fall back to keyword-only skill retrieval when no vector store is reachable. |
| `ODYSSEUS_SKILL_SEMANTIC_THRESHOLD` | `'0.4'` | `services/memory/skills.py:800` | Minimum semantic score a skill needs to be retrieved. A non-numeric value falls back to the default. |
| `ODYSSEUS_SKILL_SEMANTIC_RETRIEVAL` | `'1'` | `services/memory/skills.py:796` | On by default. Set 0, false, no or off to fall back to keyword-only skill retrieval when no vector store is reachable. |
| `ODYSSEUS_SKILL_SEMANTIC_THRESHOLD` | `'0.4'` | `services/memory/skills.py:807` | Minimum semantic score a skill needs to be retrieved. A non-numeric value falls back to the default. |
### Speech and vision models
| Variable | Default | Read in | What it does |
|---|---|---|---|
| `ODYSSEUS_GROUNDING_MODEL` | `'google/owlvit-base-patch32'` | `routes/gallery/gallery_routes.py:95` | Object-grounding model id the gallery loads for text-driven selection. |
| `ODYSSEUS_SAM_MODEL` | `'facebook/sam-vit-base'` | `routes/gallery/gallery_routes.py:59` | Segmentation model id the gallery loads for subject selection. |
| `ODYSSEUS_GROUNDING_MODEL` | `'google/owlvit-base-patch32'` | `routes/gallery/gallery_routes.py:96` | Object-grounding model id the gallery loads for text-driven selection. |
| `ODYSSEUS_SAM_MODEL` | `'facebook/sam-vit-base'` | `routes/gallery/gallery_routes.py:60` | Segmentation model id the gallery loads for subject selection. |
| `ODYSSEUS_STT_MODEL` | *unset* | `src/agent_tools/media_tools.py:2184` | Default speech-to-text model for media transcription when the tool call does not name one. |
| `ODYSSEUS_TTS_CACHE_MAX_BYTES` | `500 * 1024 * 1024` | `services/tts/tts_service.py:47` | Cap on the synthesized-speech cache. A non-numeric value falls back to the default. |
@@ -168,7 +168,7 @@ The source tree reads **109** `ODYSSEUS_*` variables: 79 an operator may want to
| Variable | Default | Read in | What it does |
|---|---|---|---|
| `ODYSSEUS_INTERNAL_BASE` | *unset* | `src/constants.py:178` | Base URL the in-app tool layer uses for loopback HTTP calls. Set it when the app is not reachable at the port it thinks it is bound to. |
| `ODYSSEUS_INTERNAL_BASE` | *unset* | `src/constants.py:190` | Base URL the in-app tool layer uses for loopback HTTP calls. Set it when the app is not reachable at the port it thinks it is bound to. |
| `ODYSSEUS_INTERNAL_TOKEN` | *unset* | `core/middleware.py:20` | Security-relevant. Token that lets the in-app tool layer reach admin-gated routes over loopback. Unset generates a fresh per-process token, which is what you want unless something outside the process needs the same value. |
### Integrations (Claude, Codex)