odysseus/src/containment.py
Alexandre Teixeira f9aa2818c4 refactor(runtime): centralize verified process lifecycle
Extract the generic process lifecycle layer (src/process_lifecycle.py)
shared by runtime-owned subprocesses: process identity (pid + boot-bound
start token), identity-bound observation, group and pidfd probes, the
TERM -> verify -> KILL -> verify escalation with re-gating before
escalation, identity-scoped sweeps, and the termination receipt.

Containment, the PTY shell, the Cookbook survivor sweep, the browser
lifecycle, web_tools browser cleanup, kill_process_tree and the startup
reaper consume it while keeping their own ownership semantics.

Safety corrections:
- browser membership and identity are bound in one snapshot; no identity
  is recaptured after membership is decided
- web_tools legacy pid-file and profile-match kills signal only verified
  identities; browser CLI groups only while their spawn identity verifies
- Cookbook and legacy-tmux descendant capture bind membership to identity
- PTY teardown never signals the server's own process group
- unverifiable processes are reported, never signalled
2026-10-02 01:27:08 +01:00

1694 lines
69 KiB
Python

"""The runtime containment boundary.
One place decides *where* and *under what limits* an already-authorized process
may run. Nothing here decides *whether* it may run — that is request authority,
and it lives elsewhere. The chain is: request → authority decides whether →
containment decides how and where → effect inside the boundary.
Three properties this module exists to hold, in order of how badly the tree
needed them:
1. **No silent downgrade.** Today a missing sandbox binary turns into a regex
that rewrites ``/workspace`` to the real path, with no log line and no field
in the tool result — ``namespaced or _replace_workspace_alias(...)``. A
string rewrite is not a containment mechanism and :func:`acquire` cannot
return one, so that line becomes unwritable through this API.
2. **Truthful reporting.** A grant states which dimensions are actually
enforced, which were asked for best-effort and are missing, and which were
required and are missing. "Was that command contained?" gets one answer
instead of none.
3. **Authoritative teardown.** :func:`release` escalates SIGTERM → SIGKILL,
signals the whole process group, and verifies death before reporting it.
Nothing here marks a process killed that it did not observe die.
**Containment never reads the command.** :func:`acquire` is given a spec and an
owner; the command text only reaches :func:`run`, after the boundary is fixed.
That is structural, not a convention: no model output, tool argument or chain of
reasoning can widen a boundary it is never shown to. Limits come from
:data:`DEFAULT_REQUIRED` and the caller's configuration, never from the request.
Enforcement mode
----------------
:data:`CONTAINMENT_MODE` is a module-level constant, deliberately not a setting
and not an ``ODYSSEUS_*`` variable, so that changing the posture of every
agent-reachable spawn site is a one-line reviewable diff rather than a
deployment detail.
* :data:`MODE_ENFORCING` — a required dimension that cannot be established
raises :class:`ContainmentUnavailable` and the command does not run.
* :data:`MODE_REPORT_ONLY` — the same shortfall is recorded on the grant as
``unenforced_required``, logged once, and the command runs.
The shipped default is enforcing. Hosts without functional namespaces refuse
native agent execution requiring filesystem and process-tree containment.
Report-only remains an explicit internal diagnostic posture, never a tool or
deployment setting. Networking is inherited unless the spec requests isolation.
"""
from __future__ import annotations
import asyncio
import codecs
import json
import logging
import os
import shutil
import signal
import subprocess
import sys
import time
import uuid
from dataclasses import dataclass, replace
from pathlib import Path, PurePosixPath
from types import MappingProxyType
from typing import Any, Awaitable, Callable, Mapping, Optional
from core.atomic_io import atomic_write_json, store_transaction
from core.platform_compat import IS_WINDOWS, find_bash, pid_alive
from src import process_lifecycle, process_ownership
from src.constants import (
CONTAINMENT_STATE_FILE,
MAX_OUTPUT_CHARS,
WORKSPACE_MOUNT,
)
logger = logging.getLogger(__name__)
# ── Enforcement mode ────────────────────────────────────────────────────────
MODE_ENFORCING = "enforcing"
MODE_REPORT_ONLY = "report_only"
#: Required containment must be established before model-controlled code runs.
CONTAINMENT_MODE = MODE_ENFORCING
# ── Dimensions ──────────────────────────────────────────────────────────────
FILESYSTEM = "filesystem"
PROCESS_TREE = "process_tree"
WALL_CLOCK = "wall_clock"
NETWORK = "network"
MEMORY = "memory"
PROCESS_COUNT = "process_count"
DIMENSIONS = frozenset({
FILESYSTEM, PROCESS_TREE, WALL_CLOCK, NETWORK, MEMORY, PROCESS_COUNT,
})
#: What every model-reachable spawn must have. Filesystem scope and an
#: authoritative kill are the two the tree currently lacks; a wall clock it has
#: but does not enforce past the leader process. Network, memory and process
#: count stay best-effort until mechanisms for them exist on every platform
#: (Waves 5A/5B), because requiring a dimension no mechanism provides refuses
#: every command on every host.
DEFAULT_REQUIRED = frozenset({FILESYSTEM, PROCESS_TREE, WALL_CLOCK})
NETWORK_INHERIT = "inherit"
NETWORK_NONE = "none"
# A grant record is kept this long after release so a restart can tell a reaped
# job from one it never saw, then pruned so the store cannot grow without bound.
_RETENTION_S = 3600
# Teardown reads the group liveness probe this often while waiting out the
# grace period. Short enough that a cooperative child is not waited on for the
# full grace, long enough not to spin.
_DEATH_POLL_S = process_lifecycle.POLL_S
# Destinations a bind must never overlay: replacing the private root, the
# private /tmp or the workspace itself with a host directory would undo the
# namespace from inside the argv that builds it.
_RESERVED_BIND_DESTS = frozenset({
"/", "/tmp", "/home", "/proc", "/dev", "/sys", WORKSPACE_MOUNT,
})
# System hierarchies where a writable overlay would invalidate the boundary
# established by bubblewrap. Reject both exact roots and all descendants.
_PROTECTED_WRITABLE_HIERARCHIES = frozenset({
"/etc",
"/usr",
"/bin",
"/sbin",
"/lib",
"/lib64",
"/proc",
"/dev",
"/sys",
"/root",
WORKSPACE_MOUNT,
})
def _is_protected_writable_destination(path: str) -> bool:
normalized = os.path.abspath(path)
if normalized in _RESERVED_BIND_DESTS:
return True
for root in _PROTECTED_WRITABLE_HIERARCHIES:
if normalized == root or normalized.startswith(root.rstrip(os.sep) + os.sep):
return True
return False
# WORKSPACE_MOUNT is re-exported from src.constants: where the workspace is
# mounted inside a namespace is a property of the tool contract, not of this
# module, and two definitions of it would be two contracts.
class ContainmentUnavailable(RuntimeError):
"""A required dimension could not be established. Never downgraded.
Raised by :func:`acquire` under :data:`MODE_ENFORCING`, and by :func:`run`
whenever it is handed a grant whose postcondition does not hold — so a
hand-built grant claiming containment it does not have cannot reach a
spawn.
"""
def __init__(self, missing: frozenset[str], mechanism_tried: str) -> None:
self.missing = frozenset(missing)
self.mechanism_tried = str(mechanism_tried or "none")
listed = ", ".join(sorted(self.missing))
super().__init__(
f"containment unavailable ({listed}); strongest mechanism available "
f"was {self.mechanism_tried!r}"
)
# ── Records ─────────────────────────────────────────────────────────────────
@dataclass(frozen=True)
class ContainmentSpec:
"""What the caller needs. Declarative, and contains no policy decision.
``required`` is the whole contract: those dimensions hold or the command
does not run. Everything else is best-effort and is reported as fact rather
than assumed.
"""
workspace: str
env: Mapping[str, str]
wall_clock_s: int
required: frozenset[str] = DEFAULT_REQUIRED
network: str = NETWORK_INHERIT
writable_extra: tuple[str, ...] = ()
readonly_extra: tuple[str, ...] = ()
max_output_bytes: int = MAX_OUTPUT_CHARS
max_memory_bytes: Optional[int] = None
max_processes: Optional[int] = None
def __post_init__(self) -> None:
# Freeze env into a read-only view over a private copy. The child's
# environment is part of the boundary, so a caller holding the dict it
# passed in must not be able to edit it after acquire() validated it.
object.__setattr__(self, "env", MappingProxyType(dict(self.env or {})))
object.__setattr__(self, "required", frozenset(self.required or ()))
object.__setattr__(self, "writable_extra", tuple(self.writable_extra or ()))
object.__setattr__(self, "readonly_extra", tuple(self.readonly_extra or ()))
@property
def requested(self) -> frozenset[str]:
"""Dimensions this spec actually asks about.
A spec that leaves ``network`` inherited is not asking for network
containment, so a mechanism without it is not degraded — it gave the
spec everything the spec wanted.
"""
asked = {FILESYSTEM, PROCESS_TREE, WALL_CLOCK}
if self.network == NETWORK_NONE:
asked.add(NETWORK)
if self.max_memory_bytes is not None:
asked.add(MEMORY)
if self.max_processes is not None:
asked.add(PROCESS_COUNT)
return frozenset(asked)
@dataclass(frozen=True)
class ContainmentGrant:
"""What was actually established. Never a superset of the spec."""
id: str
mechanism: str
workspace: str
enforced: frozenset[str]
degraded: tuple[str, ...]
unenforced_required: tuple[str, ...]
owner: str
mode: str
spec: ContainmentSpec
external: bool = False
pid: Optional[int] = None
#: The child's process group, captured at spawn. Teardown needs it because
#: it outlives the leader's pid: the leader can exit while the processes it
#: backgrounded keep running in the same group.
pgid: Optional[int] = None
namespace_pid: Optional[int] = None
namespace_start_token: Optional[str] = None
endpoint: Optional[str] = None
@property
def contained(self) -> bool:
"""True when every required dimension is actually enforced."""
return not self.unenforced_required
def to_dict(self) -> dict[str, Any]:
"""The ``containment`` block a tool result carries.
Deliberately omits ``env``: it is part of the boundary but it is also
where credentials live, and a tool result is model-visible.
"""
data = {
"id": self.id,
"mechanism": self.mechanism,
"mode": self.mode,
"workspace": self.workspace,
"enforced": sorted(self.enforced),
"degraded": list(self.degraded),
"unenforced_required": list(self.unenforced_required),
"contained": self.contained,
"external": self.external,
"requested": sorted(self.spec.requested),
"network": self.spec.network,
}
if self.endpoint:
data["endpoint"] = self.endpoint
return data
@dataclass(frozen=True)
class ContainmentResult:
stdout: str
stderr: str
exit_code: Optional[int]
timed_out: bool
output_truncated: bool
grant: ContainmentGrant
release: Optional["ReleaseOutcome"] = None
#: The termination receipt is generic lifecycle evidence, not a containment
#: concept; the name is kept because tool results and records already carry it.
ReleaseOutcome = process_lifecycle.TerminationOutcome
# ── Mechanisms ──────────────────────────────────────────────────────────────
@dataclass(frozen=True)
class Mechanism:
"""A way to establish containment, and exactly what it is good for.
``provides`` is a function of the spec alone — never of the command — so
mechanism selection cannot be influenced by request text.
"""
name: str
rank: int
available: Callable[[], bool]
provides: Callable[[ContainmentSpec], frozenset[str]]
def _bwrap_available() -> bool:
if IS_WINDOWS:
return False
executable = shutil.which("bwrap")
if not executable:
return False
try:
# Binary installation says nothing about namespace permissions (notably
# under Docker's normal security profile). Only trusted probe code runs.
probe = subprocess.run(
[executable, "--die-with-parent", "--unshare-pid", "--ro-bind", "/", "/",
"--proc", "/proc", "--dev", "/dev", "/bin/true"],
stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE,
timeout=3, check=False,
)
return probe.returncode == 0
except (OSError, subprocess.SubprocessError):
return False
def _posix_group_available() -> bool:
return not IS_WINDOWS
def _windows_available() -> bool:
return IS_WINDOWS
#: macOS advertises an infinite ``RLIMIT_AS`` hard limit and then refuses every
#: attempt to lower it ("current limit exceeds maximum limit"), so an
#: address-space ceiling is a Linux-only mechanism. Claiming it anywhere else
#: would produce a grant saying `memory` is enforced and a spawn that dies in
#: ``preexec_fn`` — a false claim is worse than an honest absence.
_ADDRESS_SPACE_LIMIT_SUPPORTED = sys.platform.startswith("linux")
def _rlimit_fits(name: str, requested: int) -> bool:
"""True when ``requested`` is within the inherited hard limit for ``name``.
A soft limit above the hard limit is rejected by ``setrlimit``, so asking
for one would abort the spawn. Checked here, where it can be reported, not
in the child, where it can only crash.
"""
try:
import resource
except ImportError: # pragma: no cover - POSIX always has it
return False
which = getattr(resource, name, None)
if which is None:
return False
try:
_soft, hard = resource.getrlimit(which)
except (OSError, ValueError): # pragma: no cover - platform dependent
return False
return hard in (resource.RLIM_INFINITY, -1) or requested <= hard
def _rlimit_dimensions(spec: ContainmentSpec) -> set[str]:
"""Resource dimensions a POSIX ``setrlimit`` in the child can actually hold.
Probed rather than assumed: a dimension is only claimed when the limit
exists on this platform and the requested value is applicable.
"""
if IS_WINDOWS:
return set()
provided: set[str] = set()
if (
spec.max_memory_bytes is not None
and _ADDRESS_SPACE_LIMIT_SUPPORTED
and _rlimit_fits("RLIMIT_AS", spec.max_memory_bytes)
):
provided.add(MEMORY)
if (
spec.max_processes is not None
and os.geteuid() != 0 # RLIMIT_NPROC does not limit root.
and _rlimit_fits("RLIMIT_NPROC", spec.max_processes)
):
provided.add(PROCESS_COUNT)
return provided
def _bwrap_provides(spec: ContainmentSpec) -> frozenset[str]:
# bwrap gives the private root and the workspace bind (filesystem), a new
# PID namespace plus --die-with-parent (process_tree), and --unshare-net when the
# spec asked for no network. The wall clock and the resource limits are
# ours either way, applied to the bwrap process itself so its descendants
# inherit them.
provided = {FILESYSTEM, PROCESS_TREE, WALL_CLOCK} | _rlimit_dimensions(spec)
if spec.network == NETWORK_NONE:
provided.add(NETWORK)
return frozenset(provided)
def _posix_group_provides(spec: ContainmentSpec) -> frozenset[str]:
# Groups support escalating teardown, but a descendant can call setsid()
# and escape. They cannot truthfully establish process-tree containment.
return frozenset({WALL_CLOCK} | _rlimit_dimensions(spec))
def _windows_provides(spec: ContainmentSpec) -> frozenset[str]:
# taskkill supports teardown, but is not a Job Object preventing escaped
# descendants. No filesystem or process-tree containment is established.
return frozenset({WALL_CLOCK})
#: Strongest first. Selection walks this in order and stops at the first
#: mechanism that covers ``spec.required``; if none does, the strongest
#: available one is used and the shortfall is reported (or raised, under
#: MODE_ENFORCING). Tests substitute this list to drive selection
#: deterministically without needing a real sandbox.
MECHANISMS: tuple[Mechanism, ...] = (
Mechanism("bubblewrap", 30, _bwrap_available, _bwrap_provides),
Mechanism("process_group", 20, _posix_group_available, _posix_group_provides),
Mechanism("windows_tree", 10, _windows_available, _windows_provides),
)
# ── Durable grant records ───────────────────────────────────────────────────
def _store_path() -> Path:
return Path(CONTAINMENT_STATE_FILE)
def _load_records() -> dict[str, dict[str, Any]]:
try:
path = _store_path()
if path.exists():
data = json.loads(path.read_text(encoding="utf-8")) or {}
if isinstance(data, dict):
return {
str(key): value
for key, value in data.items()
if isinstance(value, dict)
}
except Exception:
# A corrupt or unreadable store must not take out execution. The grant
# itself is authoritative for this process; the file exists so a
# *restart* can reap rather than orphan.
logger.warning("containment: grant store unreadable; starting empty", exc_info=True)
return {}
def _save_records(records: Mapping[str, dict[str, Any]]) -> bool:
try:
atomic_write_json(str(_store_path()), dict(records), indent=2)
return True
except Exception:
logger.warning("containment: could not persist grant store", exc_info=True)
return False
def _prune(records: dict[str, dict[str, Any]]) -> dict[str, dict[str, Any]]:
now = time.time()
kept = {}
for grant_id, record in records.items():
released = record.get("released_at")
if released and (now - float(released)) > _RETENTION_S:
continue
kept[grant_id] = record
return kept
@store_transaction(lambda: _store_path())
def _write_record(grant: ContainmentGrant) -> None:
records = _prune(_load_records())
records[grant.id] = {
"id": grant.id,
"owner": grant.owner,
"manager_pid": os.getpid(),
"manager_token": process_ownership.capture(os.getpid())["start_token"],
"mechanism": grant.mechanism,
"mode": grant.mode,
"workspace": grant.workspace,
"enforced": sorted(grant.enforced),
"degraded": list(grant.degraded),
"unenforced_required": list(grant.unenforced_required),
"required": sorted(grant.spec.required),
"wall_clock_s": grant.spec.wall_clock_s,
"max_memory_bytes": grant.spec.max_memory_bytes,
"max_processes": grant.spec.max_processes,
"network": grant.spec.network,
"external": grant.external,
"pid": grant.pid,
"pgid": grant.pgid,
"namespace_pid": grant.namespace_pid,
"namespace_start_token": grant.namespace_start_token,
"endpoint": grant.endpoint,
"acquired_at": time.time(),
"released_at": None,
"release": None,
}
_save_records(records)
@store_transaction(lambda: _store_path())
def _update_record(grant_id: str, **fields: Any) -> None:
records = _load_records()
record = records.get(grant_id)
if record is None:
return
record.update(fields)
records[grant_id] = record
_save_records(records)
def active_grants() -> list[dict[str, Any]]:
"""Grant records that were never released — a restart's reaping input.
One owner, one record, one place to ask what is running on whose behalf.
"""
return [
record
for record in _prune(_load_records()).values()
if not record.get("released_at")
]
@store_transaction(lambda: _store_path())
def forget(grant_id: str) -> None:
"""Drop a record outright. For a reaper that has finished with it."""
records = _load_records()
if records.pop(str(grant_id), None) is not None:
_save_records(records)
# ── Spec validation ─────────────────────────────────────────────────────────
def _validate_abs_path(value: str, *, label: str) -> str:
text = str(value or "")
if not text or "\x00" in text:
raise ValueError(f"containment: {label} must be a non-empty path")
if not os.path.isabs(text):
raise ValueError(f"containment: {label} must be absolute, got {text!r}")
if ".." in PurePosixPath(text.replace(os.sep, "/")).parts:
raise ValueError(f"containment: {label} must not contain '..', got {text!r}")
return os.path.normpath(text)
def _validate_spec(spec: ContainmentSpec) -> ContainmentSpec:
"""Reject a malformed spec loudly, before any mechanism is considered.
These are caller bugs, not platform shortfalls, so they raise ValueError in
both modes: there is no report-only version of a workspace that is not a
directory.
"""
unknown = set(spec.required) - DIMENSIONS
if unknown:
raise ValueError(
f"containment: unknown required dimension(s) {sorted(unknown)}; "
f"known dimensions are {sorted(DIMENSIONS)}"
)
# Requiring a dimension the spec never asked for can never be satisfied,
# so it is a contradiction rather than an unavailable mechanism.
contradictory = set(spec.required) - set(spec.requested)
if contradictory:
raise ValueError(
f"containment: required {sorted(contradictory)} but the spec does not "
"request it (set network='none', max_memory_bytes or max_processes)"
)
if spec.network not in (NETWORK_INHERIT, NETWORK_NONE):
raise ValueError(f"containment: network must be 'inherit' or 'none', got {spec.network!r}")
if not isinstance(spec.wall_clock_s, int) or isinstance(spec.wall_clock_s, bool):
raise ValueError("containment: wall_clock_s must be an int")
if spec.wall_clock_s <= 0:
raise ValueError(f"containment: wall_clock_s must be positive, got {spec.wall_clock_s}")
if spec.max_output_bytes <= 0:
raise ValueError("containment: max_output_bytes must be positive")
for name, value in (("max_memory_bytes", spec.max_memory_bytes),
("max_processes", spec.max_processes)):
if value is not None and (not isinstance(value, int) or value <= 0):
raise ValueError(f"containment: {name} must be a positive int or None")
for key, value in spec.env.items():
if not isinstance(key, str) or not isinstance(value, str):
raise ValueError("containment: env keys and values must be str")
if "\x00" in key or "\x00" in value:
raise ValueError("containment: env must not contain NUL")
workspace = _validate_abs_path(spec.workspace, label="workspace")
if not os.path.isdir(workspace):
raise ValueError(f"containment: workspace is not a directory: {workspace}")
writable = tuple(
_validate_abs_path(path, label="writable_extra") for path in spec.writable_extra
)
readonly = tuple(
_validate_abs_path(path, label="readonly_extra") for path in spec.readonly_extra
)
for path in writable + readonly:
if path in _RESERVED_BIND_DESTS:
raise ValueError(f"containment: refusing to bind over reserved path {path}")
for path in writable:
if _is_protected_writable_destination(path):
raise ValueError(f"containment: refusing to bind over reserved path {path}")
return replace(spec, workspace=workspace, writable_extra=writable, readonly_extra=readonly)
def agent_spec(
workspace: str,
env: Mapping[str, str],
wall_clock_s: int,
**overrides: Any,
) -> ContainmentSpec:
"""Build the spec for a model-reachable spawn.
One factory so no call site can quietly pass a weaker ``required`` set:
``required`` is :data:`DEFAULT_REQUIRED` and is not overridable here.
Widening or narrowing it is a change to this module, reviewed as one.
"""
overrides.pop("required", None)
return ContainmentSpec(
workspace=workspace,
env=env,
wall_clock_s=wall_clock_s,
required=DEFAULT_REQUIRED,
**overrides,
)
# ── acquire ─────────────────────────────────────────────────────────────────
def _select(spec: ContainmentSpec) -> tuple[Optional[Mechanism], frozenset[str]]:
"""Strongest-first selection. Returns the mechanism and what it provides.
Deterministic: the only inputs are the spec and each mechanism's
availability probe. The command is not an input and is not in scope here.
"""
best: Optional[Mechanism] = None
best_provided: frozenset[str] = frozenset()
for mechanism in sorted(MECHANISMS, key=lambda item: item.rank, reverse=True):
try:
if not mechanism.available():
continue
except Exception:
logger.warning(
"containment: availability probe for %s failed; treating as unavailable",
mechanism.name, exc_info=True,
)
continue
provided = frozenset(mechanism.provides(spec)) & DIMENSIONS
if best is None:
best, best_provided = mechanism, provided
if spec.required <= provided:
return mechanism, provided
return best, best_provided
@dataclass(frozen=True)
class ContainmentProbe:
"""What a spec *would* get on this host. No grant, no record, no process.
For a spawn path that has not yet been rewritten to run through
:func:`run` and still builds its own ``create_subprocess_*`` call. Such a
caller still has to decide — refuse, or run and say so — and that decision
has to come from the same mechanism table :func:`acquire` consults, or the
tree grows a second opinion about what this host can enforce.
Calling :func:`acquire` for the answer is the wrong shape: it writes a
durable grant record, and a record whose pid is never filled in and whose
:func:`release` never runs is an entry a restart reaper will keep finding.
"""
mechanism: str
enforced: frozenset[str]
degraded: tuple[str, ...]
unenforced_required: tuple[str, ...]
mode: str
@property
def contained(self) -> bool:
return not self.unenforced_required
@property
def refuses(self) -> bool:
"""True when this spec cannot run at all under the current mode."""
return bool(self.unenforced_required) and self.mode == MODE_ENFORCING
def probe(spec: ContainmentSpec) -> ContainmentProbe:
"""Answer what this host can establish for ``spec``, without acquiring it.
Same selection, same mechanism table and same arithmetic as
:func:`acquire`; it just stops before the side effects. The command is not
an input here either.
:raises ValueError: the spec is malformed (a caller bug, in either mode).
"""
spec = _validate_spec(spec)
mechanism, provided = _select(spec)
enforced = provided & spec.requested
missing_required = frozenset(spec.required) - enforced
return ContainmentProbe(
mechanism=mechanism.name if mechanism else "none",
enforced=enforced,
degraded=tuple(sorted(spec.requested - enforced - spec.required)),
unenforced_required=tuple(sorted(missing_required)),
mode=CONTAINMENT_MODE,
)
def acquire(spec: ContainmentSpec, *, owner: str) -> ContainmentGrant:
"""Establish containment, or refuse.
Postcondition under :data:`MODE_ENFORCING`, asserted rather than assumed::
spec.required <= grant.enforced
Picks the strongest available mechanism and never substitutes a weaker one
for a required dimension. Under :data:`MODE_REPORT_ONLY` the same shortfall
lands in ``grant.unenforced_required`` and is logged, so the run is
distinguishable from a contained one after the fact.
:raises ValueError: the spec is malformed (a caller bug, in either mode).
:raises ContainmentUnavailable: a required dimension is unavailable, under
:data:`MODE_ENFORCING`.
"""
owner_id = str(owner or "").strip()
if not owner_id:
# A process with no owner is a process nothing will reap.
raise ValueError("containment: every grant needs an owner")
spec = _validate_spec(spec)
mechanism, provided = _select(spec)
enforced = provided & spec.requested
missing_required = frozenset(spec.required) - enforced
name = mechanism.name if mechanism else "none"
if missing_required and CONTAINMENT_MODE == MODE_ENFORCING:
# The command does not run. This is the whole point: "not executed" is
# the one outcome a model cannot mistake for success.
raise ContainmentUnavailable(missing_required, name)
degraded = tuple(sorted(spec.requested - enforced - spec.required))
grant = ContainmentGrant(
id=uuid.uuid4().hex[:12],
mechanism=name,
workspace=spec.workspace,
enforced=enforced,
degraded=degraded,
unenforced_required=tuple(sorted(missing_required)),
owner=owner_id,
mode=CONTAINMENT_MODE,
spec=spec,
)
if missing_required:
logger.warning(
"containment: grant %s for owner %s is NOT contained — required %s "
"not enforced by mechanism %s (report-only mode)",
grant.id, owner_id, sorted(missing_required), name,
)
elif degraded:
logger.info(
"containment: grant %s enforced %s; best-effort %s unavailable under %s",
grant.id, sorted(enforced), list(degraded), name,
)
_write_record(grant)
return grant
def declare_external_bridge(
spec: ContainmentSpec, *, owner: str, endpoint: str,
) -> ContainmentGrant:
"""Record that execution leaves this backend entirely.
A bridged tool runs in a process this backend does not own, so no local
mechanism can contain it. The honest record is ``enforced=frozenset()``
rather than a grant implying confinement; this exists so that path has a
record at all instead of looking like an absence of one.
"""
owner_id = str(owner or "").strip()
if not owner_id:
raise ValueError("containment: every grant needs an owner")
spec = _validate_spec(spec)
grant = ContainmentGrant(
id=uuid.uuid4().hex[:12],
mechanism="external_bridge",
workspace=spec.workspace,
enforced=frozenset(),
degraded=(),
unenforced_required=tuple(sorted(spec.required)),
owner=owner_id,
mode=CONTAINMENT_MODE,
spec=spec,
external=True,
endpoint=endpoint,
)
logger.info(
"containment: grant %s is external (%s); nothing local contains it",
grant.id, endpoint,
)
_write_record(grant)
return grant
def unavailable_tool_result(exc: ContainmentUnavailable, *, tool: str) -> dict[str, Any]:
"""The tool result for a request that could not be contained.
"not executed" is stated in the error text, not inferred from a missing
output field, so a run that could not be contained reads differently from a
contained run that failed.
"""
listed = ", ".join(sorted(exc.missing))
return {
"error": f"{tool}: containment unavailable ({listed}); command not executed",
"exit_code": 1,
"containment": {
"mechanism": exc.mechanism_tried,
"mode": CONTAINMENT_MODE,
"enforced": [],
"unenforced_required": sorted(exc.missing),
"contained": False,
"executed": False,
},
}
# ── Launch plumbing ─────────────────────────────────────────────────────────
def _dir_chain(path: str, mounted: tuple[str, ...] = ()) -> list[str]:
"""``--dir`` args for every ancestor of ``path`` inside the private root.
bwrap mounts into a tmpfs root, so the destination's parents have to exist
before the bind. Stops at the mount points the argv already creates.
"""
args: list[str] = []
roots = ("/usr", "/etc", *mounted)
if any(path == root or path.startswith(root + os.sep) for root in roots):
return args
parents: list[str] = []
parent = os.path.dirname(path)
while parent not in ("/", "", "/tmp", "/etc", "/usr", WORKSPACE_MOUNT):
parents.append(parent)
parent = os.path.dirname(parent)
for directory in reversed(parents):
args.extend(("--dir", directory))
return args
def _bwrap_prefix(spec: ContainmentSpec) -> list[str]:
"""The bubblewrap argv establishing the boundary this spec asked for.
Note what is *not* here, versus the namespace this replaces: ``/home`` and
``/mnt`` are not bound read-write. Binding the user's whole home directory
into a "workspace confinement" namespace gives back most of what the
namespace was for. Anything a command legitimately needs outside the
workspace is named by the spec, as ``readonly_extra`` or ``writable_extra``.
"""
executable = shutil.which("bwrap")
if not executable:
raise ContainmentUnavailable(spec.required, "bubblewrap")
args = [
os.path.abspath(executable), "--die-with-parent", "--new-session", "--unshare-pid",
"--tmpfs", "/",
"--dir", "/usr", "--ro-bind", "/usr", "/usr",
"--symlink", "usr/bin", "/bin",
"--symlink", "usr/lib", "/lib",
"--symlink", "usr/lib64", "/lib64",
"--symlink", "usr/bin", "/sbin",
"--dir", "/etc", "--ro-bind", "/etc", "/etc",
"--dir", "/tmp", "--tmpfs", "/tmp",
"--dev", "/dev", "--proc", "/proc",
]
mounted: tuple[str, ...] = ()
for flag, paths in (("--ro-bind", spec.readonly_extra), ("--bind", spec.writable_extra)):
for path in sorted(paths, key=lambda value: (value.count(os.sep), value)):
args.extend(_dir_chain(path, mounted))
args.extend((flag, path, path))
mounted += (path,)
# Mount workspace last so a read-only ancestor never hides its writable bind.
args.extend(("--dir", WORKSPACE_MOUNT, "--bind", spec.workspace, WORKSPACE_MOUNT))
# Preserve absolute workspace paths in generated scripts without exposing
# a writable parent directory.
workspace = os.path.realpath(spec.workspace)
if workspace not in _RESERVED_BIND_DESTS and workspace not in {"/usr", "/etc"}:
args.extend(_dir_chain(workspace, mounted))
args.extend(("--bind", workspace, workspace))
if spec.network == NETWORK_NONE:
args.append("--unshare-net")
args.extend(("--chdir", WORKSPACE_MOUNT))
return args
def _rlimit_preexec(grant: ContainmentGrant) -> Optional[Callable[[], None]]:
"""A child-side hook applying the limits the grant actually claimed, or None.
Only ever applies a dimension in ``grant.enforced``, so the child cannot
attempt a limit the probe already said this platform will refuse. If a limit
nevertheless fails to apply, the exception aborts the spawn: an unlimited
run under a grant that promised a ceiling is the one outcome worse than a
loud failure.
"""
if IS_WINDOWS:
return None
spec = grant.spec
memory = spec.max_memory_bytes if MEMORY in grant.enforced else None
processes = spec.max_processes if PROCESS_COUNT in grant.enforced else None
if memory is None and processes is None:
return None
try:
import resource
except ImportError: # pragma: no cover - POSIX always has it
return None
def _apply() -> None: # pragma: no cover - runs in the forked child
if memory is not None:
resource.setrlimit(resource.RLIMIT_AS, (memory, memory))
if processes is not None:
resource.setrlimit(resource.RLIMIT_NPROC, (processes, processes))
return _apply
def _launch_argv(grant: ContainmentGrant, command: Any, *, argv: bool,
ready_marker: Optional[str] = None, info_fd: Optional[int] = None) -> list[str]:
spec = grant.spec
if argv:
parts = [str(part) for part in command]
if not parts:
raise ValueError("containment: empty argv")
else:
text = str(command or "")
if not text.strip():
raise ValueError("containment: empty command")
if grant.mechanism == "bubblewrap":
# The namespace brings its own /bin/bash via the read-only /usr.
parts = ["/bin/bash", "-lc", text]
else:
shell = find_bash()
if not shell:
if IS_WINDOWS:
raise RuntimeError("Git Bash is required for the Bash tool on Windows; install Git for Windows.")
raise RuntimeError(
"containment: no POSIX shell available to run a shell command"
)
parts = [shell, "-c", text]
if grant.mechanism == "bubblewrap":
if ready_marker is not None:
# This trusted wrapper runs only after all bwrap setup succeeds.
# A launch-time bind/security failure must not claim containment.
parts = ["/bin/sh", "-c",
'printf "%s\\n" "$1"; IFS= read -r ody_ack || exit 125; '
'[ "$ody_ack" = "$1" ] || exit 125; shift; exec "$@"',
"ody-boundary", ready_marker, *parts]
info_args = ["--info-fd", str(info_fd)] if info_fd is not None else []
return _bwrap_prefix(spec) + info_args + parts
return parts
def _spawn_kwargs(grant: ContainmentGrant) -> dict[str, Any]:
kwargs: dict[str, Any] = {}
if IS_WINDOWS:
# No setsid; the child gets its own group so a console event cannot
# reach it, and teardown walks the tree with taskkill /T.
kwargs["creationflags"] = getattr(subprocess, "CREATE_NEW_PROCESS_GROUP", 0x00000200)
return kwargs
# A separate group makes ordinary tree teardown possible. The PID
# namespace, not setsid, prevents descendants from escaping containment.
kwargs["start_new_session"] = True
preexec = _rlimit_preexec(grant)
if preexec is not None:
kwargs["preexec_fn"] = preexec
return kwargs
async def _drain(stream, buffer: list[str], budget: list[int], output_cb=None) -> None:
"""Read a stream to EOF, keeping at most ``budget[0]`` bytes.
Reading past the cap and discarding is deliberate: stopping the read would
block the child on a full pipe, which turns an output cap into a hang.
``budget[0]`` is set to -1 once anything has actually been dropped, so the
caller reports truncation only when bytes were lost — output that exactly
fills the cap is not truncated.
Each stream gets its own budget so the split between stdout and stderr does
not depend on which reader happened to be scheduled first.
"""
if stream is None:
return
decoder = codecs.getincrementaldecoder("utf-8")(errors="replace")
def emit(text):
if not text:
return
buffer.append(text)
if output_cb:
try:
output_cb(text)
except OSError:
budget[0] = -1
logger.warning("containment: output sink failed", exc_info=True)
while True:
line = await stream.read(65536)
if not line:
emit(decoder.decode(b"", final=True))
break
if budget[0] < 0:
continue
chunk = line[:budget[0]]
if chunk:
emit(decoder.decode(chunk))
if budget[0] < 0:
continue
budget[0] = budget[0] - len(line) if len(line) <= budget[0] else -1
async def run(
grant: ContainmentGrant,
command: Any,
*,
argv: bool = False,
stdin: Optional[bytes] = None,
progress_cb: Optional[Callable[[dict], Awaitable[None]]] = None,
output_cb: Optional[Callable[[str], None]] = None,
) -> ContainmentResult:
"""Execute inside an existing grant.
Enforces the wall clock and the output cap, and on timeout tears the tree
down through :func:`release` so the reported outcome is the observed one.
:raises ContainmentUnavailable: the grant's postcondition does not hold
under :data:`MODE_ENFORCING`. Re-checked here, at the point of effect,
so a grant that was not produced by :func:`acquire` cannot buy a spawn
by claiming dimensions it does not have.
"""
if grant.external:
raise ValueError(
"containment: an external-bridge grant describes execution this "
"backend does not own; it cannot be run locally"
)
if (_load_records().get(grant.id) or {}).get("released_at"):
raise ValueError("containment: a released grant cannot execute again")
missing = frozenset(grant.spec.required) - frozenset(grant.enforced)
mechanism = next((item for item in MECHANISMS if item.name == grant.mechanism), None)
provided = mechanism.provides(grant.spec) if mechanism is not None else frozenset()
missing |= frozenset(grant.spec.required) - provided
overclaimed = frozenset(grant.enforced) - provided
if (missing or overclaimed) and grant.mode == MODE_ENFORCING:
raise ContainmentUnavailable(missing | overclaimed, grant.mechanism)
spec = grant.spec
marker = uuid.uuid4().hex if grant.mechanism == "bubblewrap" else None
info_read = info_write = None
try:
if marker is not None:
info_read, info_write = os.pipe()
os.set_blocking(info_read, False)
launch = _launch_argv(grant, command, argv=argv, ready_marker=marker, info_fd=info_write)
spawn_kwargs = _spawn_kwargs(grant)
if info_write is not None:
spawn_kwargs["pass_fds"] = (info_write,)
spawning = asyncio.create_task(asyncio.create_subprocess_exec(
*launch,
stdin=asyncio.subprocess.PIPE if stdin is not None or marker is not None else asyncio.subprocess.DEVNULL,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
cwd=spec.workspace,
env=dict(spec.env),
**spawn_kwargs,
))
try:
proc = await asyncio.shield(spawning)
except asyncio.CancelledError:
# Cancellation must not detach an OS spawn already in progress.
# Recover its handle before propagating cancellation to the caller.
try:
proc = await _complete_cleanup(spawning, propagate_cancel=False)
except Exception:
release(grant, grace_s=0)
else:
if info_write is not None:
os.close(info_write)
info_write = None
proc._ody_info_read = info_read
live = replace(grant, pid=proc.pid, pgid=None if IS_WINDOWS else proc.pid)
await _complete_cleanup(_release_awaited(live, proc), propagate_cancel=False)
raise
except BaseException:
if "proc" not in locals():
release(grant, grace_s=0)
raise
finally:
if info_write is not None:
os.close(info_write)
if "proc" not in locals() and info_read is not None:
os.close(info_read)
proc._ody_info_read = info_read
# start_new_session makes the child its own group leader, so the group id
# is the child's pid. Captured here rather than at teardown: once the leader
# exits, getpgid can no longer tell us which group its children are in.
pgid = None if IS_WINDOWS else proc.pid
live = replace(grant, pid=proc.pid, pgid=pgid)
# The start token is what makes this record signallable by a *later*
# process. Without it a restart reaper holds a pid and no way to tell
# whether the pid is still this child or something the kernel has since
# handed to a stranger; see src/process_ownership.py.
try:
_update_record(grant.id, pid=proc.pid, pgid=pgid, started_at=time.time(),
start_token=None if marker is not None else process_ownership.capture(proc.pid)["start_token"],
containment_ready=marker is None, execution_started=marker is None)
from src.agent_runtime.journal import mark_operation_started
mark_operation_started("subprocess", pid=proc.pid)
except BaseException:
await _complete_cleanup(_release_awaited(live, proc), propagate_cancel=False)
raise
out_buf: list[str] = []
err_buf: list[str] = []
out_budget = [int(spec.max_output_bytes)]
err_budget = [int(spec.max_output_bytes)]
started = time.time()
readers = [asyncio.create_task(_drain(proc.stderr, err_buf, err_budget, output_cb))]
ready = marker is None
execution_started = marker is None
async def _wait() -> None:
nonlocal ready, execution_started, live
if marker is not None:
expected = (marker + "\n").encode("ascii")
try:
receipt = await proc.stdout.readexactly(len(expected))
except (asyncio.IncompleteReadError, OSError):
receipt = b""
if receipt != expected:
raise ContainmentUnavailable(spec.required, grant.mechanism)
await _capture_namespace_identity(proc)
live = replace(live, namespace_pid=proc._ody_namespace_pid,
namespace_start_token=proc._ody_namespace_token)
# The trusted child is waiting for acknowledgment, so model code
# cannot exit/recycle the leader before we record its identity.
_update_record(grant.id, start_token=process_ownership.capture(proc.pid)["start_token"])
_update_record(grant.id, namespace_pid=live.namespace_pid,
namespace_start_token=live.namespace_start_token)
if hasattr(os, "pidfd_open") and hasattr(signal, "pidfd_send_signal"):
try:
proc._ody_pidfd = os.pidfd_open(proc.pid)
except OSError:
raise ContainmentUnavailable(spec.required, grant.mechanism) from None
ready = True
_update_record(grant.id, containment_ready=True, execution_started=True)
execution_started = True
proc.stdin.write(expected)
await proc.stdin.drain()
readers.append(asyncio.create_task(_drain(proc.stdout, out_buf, out_budget, output_cb)))
# Pipe backpressure is execution time too. Feeding a child that never
# reads stdin must remain inside the same timeout/cancellation scope.
if (stdin is not None or marker is not None) and proc.stdin is not None:
try:
if stdin is not None:
proc.stdin.write(stdin)
await proc.stdin.drain()
except (BrokenPipeError, ConnectionResetError):
pass
finally:
proc.stdin.close()
await proc.wait()
async def _progress() -> None:
while True:
await asyncio.sleep(2.0)
if progress_cb:
try:
tail = "\n".join(("".join(out_buf) + "".join(err_buf)).splitlines()[-12:])[-8192:]
await progress_cb({"elapsed_s": round(time.time() - started, 1), "tail": tail})
except Exception:
pass
progress_task = asyncio.create_task(_progress()) if progress_cb else None
timed_out = False
outcome: Optional[ReleaseOutcome] = None
async def _finish() -> ReleaseOutcome:
try:
return await _release_awaited(live, proc)
finally:
if progress_task is not None:
progress_task.cancel()
try:
await progress_task
except (asyncio.CancelledError, Exception):
pass
for task in readers:
try:
await asyncio.wait_for(asyncio.shield(task), timeout=1)
except (asyncio.TimeoutError, Exception):
out_budget[0] = -1
task.cancel()
await asyncio.gather(task, return_exceptions=True)
_close_process_handles(proc)
try:
try:
await asyncio.wait_for(_wait(), timeout=spec.wall_clock_s)
except asyncio.TimeoutError:
if not ready:
raise ContainmentUnavailable(spec.required, grant.mechanism)
timed_out = True
except BaseException as exc:
exc.containment_established = ready
exc.containment_executed = execution_started
raise
finally:
# Clean exit, partial initialization, timeout, and cancellation share
# the same teardown. Repeated cancellation cannot skip escalation.
outcome = await _complete_cleanup(_finish())
if timed_out:
_update_record(grant.id, timed_out=True)
return ContainmentResult(
stdout="".join(out_buf),
stderr="".join(err_buf),
exit_code=proc.returncode,
timed_out=timed_out,
output_truncated=out_budget[0] < 0 or err_budget[0] < 0,
grant=live,
release=outcome,
)
async def _complete_cleanup(awaitable, *, propagate_cancel: bool = True):
"""Finish ownership cleanup despite further cancellation, then propagate it."""
task = asyncio.ensure_future(awaitable)
cancelled = False
while not task.done():
try:
await asyncio.shield(task)
except asyncio.CancelledError:
if task.cancelled():
raise
cancelled = True
result = task.result()
if cancelled and propagate_cancel:
raise asyncio.CancelledError
return result
async def _capture_namespace_identity(proc) -> None:
"""Read bwrap's trusted init identity before acknowledging model execution."""
fd = getattr(proc, "_ody_info_read", None)
if fd is None:
return
data = bytearray()
try:
while True:
try:
chunk = os.read(fd, 4096)
except BlockingIOError:
await asyncio.sleep(.01)
continue
if not chunk:
break
data.extend(chunk)
if len(data) > 4096:
raise ValueError("oversized namespace identity")
info = json.loads(data)
pid = int(info["child-pid"])
if pid <= 0 or pid == os.getpid():
raise ValueError("invalid namespace init identity")
proc._ody_namespace_pid = pid
proc._ody_namespace_token = process_ownership.capture(pid)["start_token"]
if hasattr(os, "pidfd_open") and hasattr(signal, "pidfd_send_signal"):
proc._ody_namespace_pidfd = os.pidfd_open(pid)
elif not proc._ody_namespace_token:
raise ValueError("namespace init identity cannot be inspected")
except (OSError, ValueError, KeyError, TypeError) as exc:
raise ContainmentUnavailable(frozenset({PROCESS_TREE}), "bubblewrap") from exc
finally:
os.close(fd)
proc._ody_info_read = None
# ── release ─────────────────────────────────────────────────────────────────
# Process mechanics — group probes, signalling, escalation, verified death —
# live in src.process_lifecycle, shared with the PTY shell, the Cookbook sweep,
# the browser lifecycle and core.platform_compat.kill_process_tree. What stays
# here is what a grant means: its record, its namespace init and its gate.
# The thin wrappers below are this module's seams; teardown resolves them at
# call time so a test can substitute one probe without replacing the engine.
def _own_pgid() -> int:
return process_lifecycle.own_pgid()
def _pgid_of(pid: Optional[int]) -> Optional[int]:
if IS_WINDOWS:
return None
return process_lifecycle.pgid_of(pid)
def _group_present(pgid: Optional[int]) -> bool:
"""True while any process remains in ``pgid``; never true for our own group.
EPERM is a live group we cannot signal, not verified death.
"""
if IS_WINDOWS:
return False
return process_lifecycle.group_present(pgid, own=_own_pgid())
def _signal_tree(pid: Optional[int], pgid: Optional[int], sig: int) -> None:
"""Signal the whole group, falling back to the leader; never our own group."""
process_lifecycle.signal_group(pid, pgid, sig, own=_own_pgid())
def _reap_if_child(pid: Optional[int]) -> None:
"""Clear a zombie we parented, so "alive" means running (sync teardown only)."""
if IS_WINDOWS:
return
process_lifecycle.reap_if_child(pid)
def _tree_gone(pid: Optional[int], pgid: Optional[int], *, reap: bool = False) -> bool:
if reap:
_reap_if_child(pid)
return not _group_present(pgid) and not pid_alive(pid)
def _outcome_for(
grant: ContainmentGrant, *, dead: bool, escalated: bool,
) -> ReleaseOutcome:
survivors: tuple[int, ...] = ()
if not dead:
survivors = tuple(dict.fromkeys(
value for value in (grant.pid, grant.pgid) if value
))
logger.warning(
"containment: grant %s left survivors after escalation: %s",
grant.id, survivors,
)
return ReleaseOutcome(
dead=dead,
escalated=escalated,
survivors=survivors,
mechanism=grant.mechanism,
)
def _ownership_gate(
grant: ContainmentGrant,
pid: int,
pgid: Optional[int],
token: Optional[str],
) -> Optional[ReleaseOutcome]:
"""Decide whether a recovered grant may be signalled at all.
Returns None to let teardown proceed, or the outcome to report instead.
Reached only for a grant recovered from the durable store — the restart and
reaper path, where the recorded pid is a claim rather than a child this
process is holding.
The rule is fail-closed: **a signal requires a positive identity.** Anything
else is reported as an undead tree rather than silently killed, because the
alternative is sending SIGKILL to whatever the kernel has since given that
pid to. ODY-86 was this defect; the reason the record stays active on a
refusal is that an unreapable orphan has to remain visible instead of being
closed out as handled.
"""
# A valid leader identity does not establish ownership of an arbitrary
# recorded process group: a stale or inconsistent PGID is UNVERIFIABLE.
verdict = process_lifecycle.group_ownership_verdict(pid, pgid, token, pgid_of=_pgid_of)
if verdict == process_ownership.OWNED:
return None
if verdict == process_ownership.GONE:
# The leader is gone. Its group may still hold processes it
# backgrounded, but with the leader unverifiable there is nothing left
# to prove the group is still ours, and a recycled group id would mean
# killpg hits strangers. An empty group is the clean case.
if not _group_present(pgid):
return replace(
_outcome_for(grant, dead=True, escalated=False), ownership=verdict,
)
logger.warning(
"containment: grant %s leader pid %s is gone but group %s still has "
"members; not signalling a group whose ownership cannot be proven",
grant.id, pid, pgid,
)
return ReleaseOutcome(
dead=False,
escalated=False,
survivors=(pgid,) if pgid else (),
mechanism=grant.mechanism,
ownership=verdict,
)
if verdict == process_ownership.FOREIGN:
logger.warning(
"containment: grant %s records pid %s, which now belongs to a "
"different process; refusing to signal it",
grant.id, pid,
)
else:
logger.warning(
"containment: grant %s pid %s cannot be verified on this host (%s); "
"refusing to signal an unidentified process",
grant.id, pid, process_ownership.inspection_mechanism(),
)
return ReleaseOutcome(
dead=False,
escalated=False,
# Not ours to enumerate, and listing a foreign pid as a survivor of
# *our* grant would invite the next reaper to kill it.
survivors=(),
mechanism=grant.mechanism,
ownership=verdict,
)
def release(grant: ContainmentGrant, *, grace_s: float = 2.0,
start_token: Optional[str] = None, require_identity: bool = False) -> ReleaseOutcome:
"""Release owner and recorded namespace init; report death only for both."""
record = _load_records().get(grant.id, {})
previous = record.get("release") or {}
if record.get("released_at") and previous.get("dead"):
# Death belongs to the completed grant, not the current occupant of a
# reused PID slot. Repeated release must never signal it again.
return ReleaseOutcome(dead=True, escalated=bool(previous.get("escalated")),
mechanism=previous.get("mechanism", grant.mechanism),
ownership=previous.get("ownership"))
namespace_pid = grant.namespace_pid or record.get("namespace_pid")
namespace_token = grant.namespace_start_token or record.get("namespace_start_token")
owner = _release_owner(grant, grace_s=grace_s, start_token=start_token,
require_identity=require_identity, _record_release=False)
outcome = owner
if namespace_pid:
try:
namespace_pid = int(namespace_pid)
if namespace_pid <= 0 or namespace_pid == os.getpid():
raise ValueError("invalid namespace init")
except (TypeError, ValueError):
namespace = ReleaseOutcome(dead=False, escalated=False,
ownership=process_ownership.UNVERIFIABLE)
else:
verdict = process_ownership.verify(namespace_pid, namespace_token) if pid_alive(namespace_pid) else process_ownership.GONE
if verdict in (process_ownership.GONE, process_ownership.FOREIGN):
# Reusing PID 1's host slot proves its original namespace has
# completed death; never signal its new occupant.
namespace = ReleaseOutcome(dead=True, escalated=False, ownership=verdict)
else:
target = replace(grant, id=grant.id + ":namespace", mechanism="process_group",
pid=namespace_pid, pgid=None, namespace_pid=None,
namespace_start_token=None)
# Opened before the identity gate inside _release_owner runs:
# a pidfd that still verifies afterwards names that process.
namespace_fd = process_lifecycle.open_pidfd(namespace_pid)
try:
namespace = _release_owner(target, grace_s=grace_s, start_token=namespace_token,
require_identity=True, _record_release=False,
_pidfd=namespace_fd)
finally:
process_lifecycle.close_fd(namespace_fd)
outcome = replace(owner, dead=owner.dead and namespace.dead,
escalated=owner.escalated or namespace.escalated,
survivors=tuple(dict.fromkeys((*owner.survivors, *namespace.survivors))))
elif grant.mechanism == "bubblewrap" and record.get("execution_started"):
# A pre-upgrade receipt lacks proof of namespace completion. Keep it
# visible rather than declaring a potentially blocked tree dead.
outcome = replace(owner, dead=False, ownership=process_ownership.UNVERIFIABLE)
_finish_release(grant, outcome)
return outcome
def _release_owner(grant: ContainmentGrant, *, grace_s: float = 2.0,
start_token: Optional[str] = None, require_identity: bool = False,
_record_release: bool = True, _pidfd: Optional[int] = None) -> ReleaseOutcome:
"""Authoritative teardown: signal the group, escalate, then verify.
Returns whether the tree is **observed** gone. A caller must not record a
process as killed on anything weaker than ``dead=True`` — reporting an
outcome you did not achieve is how a surviving process becomes invisible.
This is the synchronous form, for a grant whose process this caller is not
awaiting: a restart reaper, or a detached job. For a child being awaited,
:func:`run` uses the async form, which reaps the leader before verifying —
a zombie still belongs to its process group, so the group probe would
otherwise report a tree that is already gone.
"""
def finish(target, outcome):
if _record_release:
_finish_release(target, outcome)
pid, pgid = grant.pid, grant.pgid
# A grant that carries its own pid belongs to the process holding it: this
# caller launched the child and no identity question arises. A grant whose
# pid had to be recovered from the durable store is the restart case, and
# there the pid is a *claim* about a process this run never started.
recovered = pid is None
token: Optional[str] = start_token
if pid is None or (pgid is None and not IS_WINDOWS):
record = _load_records().get(grant.id) or {}
pid = pid if pid is not None else record.get("pid")
pgid = pgid if pgid is not None else record.get("pgid")
token = token or record.get("start_token")
try:
pid = int(pid) if pid else 0
except (TypeError, ValueError):
pid = 0
try:
pgid = int(pgid) if pgid else None
except (TypeError, ValueError):
pgid = None
if pid <= 0:
pid = 0
if pgid is not None and pgid <= 0:
pgid = None
grant = replace(grant, pid=pid or None, pgid=pgid)
def gone():
if _pidfd is not None:
return process_lifecycle.pidfd_exited(_pidfd)
return _tree_gone(pid, pgid, reap=True)
def send(sig):
if _pidfd is not None:
process_lifecycle.pidfd_signal(_pidfd, sig)
else:
_signal_tree(pid, pgid, sig)
if not pid and not _group_present(pgid):
outcome = _outcome_for(grant, dead=True, escalated=False)
finish(grant, outcome)
return outcome
if recovered or require_identity:
refusal = _ownership_gate(grant, pid, pgid, token)
if refusal is not None:
finish(grant, refusal)
return refusal
if IS_WINDOWS:
process_lifecycle.taskkill_tree(pid)
deadline = time.monotonic() + max(grace_s, 0.0)
while time.monotonic() < deadline and pid_alive(pid):
time.sleep(_DEATH_POLL_S)
outcome = _outcome_for(grant, dead=not pid_alive(pid), escalated=True)
finish(grant, outcome)
return outcome
def regate(_sig):
# The grace period is long enough for the pid to be freed and reissued;
# a recovered claim must be re-proven before SIGKILL.
if recovered or require_identity:
return _ownership_gate(grant, pid, pgid, token)
return None
result = process_lifecycle.escalate(
gone, send, steps=process_lifecycle.term_kill_steps(grace_s),
poll_s=_DEATH_POLL_S, before_step=regate,
)
if result.refusal is not None:
finish(grant, result.refusal)
return result.refusal
outcome = _outcome_for(grant, dead=result.dead, escalated=result.escalated)
finish(grant, outcome)
return outcome
def reap_record(record: Mapping[str, Any], *, grace_s: float = 2.0) -> ReleaseOutcome:
"""Tear down a grant known only by its durable record.
The entry point for a reaper after a restart: the process that acquired the
grant is gone, so there is no :class:`ContainmentGrant` in memory, only the
row :func:`active_grants` returned. Reconstructs the minimum
:func:`release` needs and goes through the same ownership gate — a record is
a claim about a pid, and a reaper is exactly the caller that must not treat
it as more than that.
``env`` is not reconstructed because it is never persisted (it is where
credentials live) and teardown does not use it.
"""
record = dict(record or {})
spec = ContainmentSpec(
workspace=record.get("workspace") or os.getcwd(),
env={},
wall_clock_s=int(record.get("wall_clock_s") or 1),
required=frozenset(record.get("required") or ()),
)
grant = ContainmentGrant(
id=str(record.get("id") or ""),
mechanism=str(record.get("mechanism") or "none"),
workspace=spec.workspace,
enforced=frozenset(record.get("enforced") or ()),
degraded=tuple(record.get("degraded") or ()),
unenforced_required=tuple(record.get("unenforced_required") or ()),
owner=str(record.get("owner") or "reaper"),
mode=str(record.get("mode") or CONTAINMENT_MODE),
spec=spec,
external=bool(record.get("external")),
# Left as None on purpose: release() then recovers pid, pgid and the
# start token from the store itself and routes through the ownership
# gate. Passing them here would mark the grant as held in-process and
# skip the very check this path exists to apply.
pid=None,
pgid=None,
)
return release(grant, grace_s=grace_s)
async def _release_awaited(
grant: ContainmentGrant,
proc: "asyncio.subprocess.Process",
*, grace_s: float = 2.0,
) -> ReleaseOutcome:
try:
return await _release_awaited_impl(grant, proc, grace_s=grace_s)
finally:
_close_process_handles(proc)
def _close_process_handles(proc) -> None:
for name in ("_ody_info_read", "_ody_pidfd", "_ody_namespace_pidfd"):
fd = getattr(proc, name, None)
if fd is not None:
os.close(fd)
setattr(proc, name, None)
async def _release_awaited_impl(
grant: ContainmentGrant,
proc: "asyncio.subprocess.Process",
*,
grace_s: float = 2.0,
) -> ReleaseOutcome:
"""Teardown for a child this coroutine owns.
Identical contract to :func:`release`, with one necessary difference: the
leader is reaped through ``proc.wait()`` before the group is probed. An
unreaped child is a zombie, a zombie is still a member of its process
group, and so ``killpg(pgid, 0)`` would report survivors for a tree that
has entirely exited — turning every timeout into a false "survivors"
report.
"""
writer = getattr(proc, "stdin", None)
if writer is not None:
writer.close()
if hasattr(writer, "wait_closed"):
try:
await asyncio.wait_for(writer.wait_closed(), timeout=1)
except (OSError, asyncio.TimeoutError):
pass
if IS_WINDOWS:
return release(grant, grace_s=grace_s)
if getattr(proc, "_ody_info_read", None) is not None:
try:
await asyncio.wait_for(_capture_namespace_identity(proc), timeout=1)
except (ContainmentUnavailable, asyncio.TimeoutError):
pass # Setup never reached acknowledgment; no model code ran.
pid, pgid = grant.pid, grant.pgid
pidfd = getattr(proc, "_ody_pidfd", None)
namespace_pid = getattr(proc, "_ody_namespace_pid", None)
namespace_token = getattr(proc, "_ody_namespace_token", None)
namespace_fd = getattr(proc, "_ody_namespace_pidfd", None)
if namespace_pid:
_update_record(grant.id, namespace_pid=namespace_pid, namespace_start_token=namespace_token)
def namespace_gone():
if namespace_fd is not None:
return process_lifecycle.pidfd_exited(namespace_fd)
if namespace_pid:
if not pid_alive(namespace_pid):
return True
return process_ownership.verify(namespace_pid, namespace_token) in (
process_ownership.GONE, process_ownership.FOREIGN,
)
return True
def gone():
if pidfd is not None:
owner_gone = process_lifecycle.pidfd_exited(pidfd)
elif grant.mechanism == "bubblewrap":
owner_gone = proc.returncode is not None
else:
owner_gone = _tree_gone(pid, pgid)
return owner_gone and namespace_gone()
def send(sig):
if pidfd is not None:
process_lifecycle.pidfd_signal(pidfd, sig)
elif proc.returncode is None or grant.mechanism != "bubblewrap":
_signal_tree(pid, pgid, sig)
if namespace_fd is not None:
process_lifecycle.pidfd_signal(namespace_fd, sig)
elif namespace_pid and process_ownership.verify(namespace_pid, namespace_token) == process_ownership.OWNED:
_signal_tree(namespace_pid, None, sig)
# No precheck: SIGTERM goes out first and the leader is reaped through
# proc.wait() before any group probe, or its zombie reads as a survivor.
result = await process_lifecycle.escalate_async(
gone, send, steps=process_lifecycle.term_kill_steps(grace_s),
wait=proc.wait, poll_s=_DEATH_POLL_S, precheck=False,
)
outcome = _outcome_for(grant, dead=result.dead, escalated=result.escalated)
if not namespace_gone():
outcome = replace(outcome, survivors=tuple(dict.fromkeys((*outcome.survivors, namespace_pid))))
_finish_release(grant, outcome)
return outcome
def _finish_release(grant: ContainmentGrant, outcome: ReleaseOutcome) -> None:
if outcome.dead:
_update_record(grant.id, released_at=time.time(), release=outcome.to_dict())
else:
# Deliberately NOT released: the record stays active so a reaper sees it
# again. A record claiming teardown it did not achieve is the defect
# this reverses.
_update_record(grant.id, released_at=None, release=outcome.to_dict())