Skip to main content

supercode_harness/
sandbox.rs

1//! P5-10 (COMPOSABLE-HARNESS-DESIGN.md §2 module 12 `permissions.sandbox`,
2//! ~row 462): the OS-level enforcement BACKSTOP `permissions.rules`'
3//! rule-layer floor and the file-tool [`crate::tools::SandboxPolicy`] both
4//! defer to for full coverage (`crate::permissions` module doc: "complete
5//! OS-level write confinement of arbitrary bash… is `capabilities.
6//! permissions.sandbox`'s job (P5 module 10, a later unit), not this
7//! one's" — this IS that unit).
8//!
9//! # What this module adds
10//! - **Real Linux fs enforcement via Landlock** ([`landlock_available`],
11//!   `apply_linux_confinement`): the spawned `bash`/`shell` subprocess (and
12//!   its own children) is kernel-confined to the configured tier's writable
13//!   set — a genuine `EPERM` from the kernel on a disallowed write, not a
14//!   path string comparison. Applied via a `pre_exec` closure that runs in
15//!   the FORKED CHILD after `fork()`, before `exec()` — [`crate::agent::
16//!   Agent`]/supercode itself is never confined, only the subprocess tree
17//!   the tool spawns.
18//! - **Coarse network cut-off** ([`netns_available`],
19//!   `apply_linux_confinement`): when `network.enabled` is set with no
20//!   domain allow/deny lists, the subprocess is placed in a fresh, isolated
21//!   network namespace (`unshare(CLONE_NEWUSER|CLONE_NEWNET)`, self-mapped
22//!   so file-permission checks are unaffected) — a real kernel-level
23//!   all-network cutoff. Domain-level allow/deny is OUT OF REACH on this
24//!   kernel class (that needs the out-of-scope TLS-MITM proxy, or Landlock
25//!   ABI v4 network scoping, kernel ≥6.7) and is surfaced as an honest gap,
26//!   never silently dropped.
27//! - **Fail-closed, never silently-unsandboxed** ([`decide_fs`]): a
28//!   confining tier this platform/kernel genuinely cannot enforce refuses to
29//!   run the subprocess at all (`escalation = "deny"`, the default), unless
30//!   `escalation` explicitly says otherwise (`"ask"` routes through
31//!   `crate::permissions::PermissionsApprovalHandler`; `"allow"` runs
32//!   unconfined with a loud, one-time warning). The worst defect class named
33//!   for this unit — "a tier claiming enforcement but silently running
34//!   unconfined" — is structurally impossible here: [`decide_fs`] only ever
35//!   returns [`FsDecision::Confine`] when the caller already told it
36//!   enforcement IS available; every other input funnels through the
37//!   escalation gate.
38//! - **`env_policy`** ([`apply_env_policy`]): `inherit` (today's behavior,
39//!   byte-identical), `filtered` (strip a sensitive-var denylist),
40//!   `none` (bare `PATH` + a couple of universally-needed variables).
41//!
42//! # Pure decision, real effect
43//! [`decide_fs`]/[`decide_net`] are pure functions — every availability/
44//! approval input is a PARAMETER, never an internal `cfg!`/probe call — so
45//! the fail-closed/ask/allow/monotonic-tightening branches are all unit-
46//! testable without touching a real kernel or spawning a process. The real
47//! call sites (`crate::tools::builtins::BashTool::execute` et al.) supply
48//! real inputs via [`landlock_available`]/[`netns_available`] (cached,
49//! real-kernel probes) and the installed
50//! [`crate::permissions::PermissionsApprovalHandler`].
51
52#[cfg(target_os = "linux")]
53use std::path::{Path, PathBuf};
54use std::sync::{Arc, Mutex, OnceLock};
55
56use crate::permissions::{ApprovalOutcome, ApprovalRequest, PermissionsApprovalHandler};
57use crate::tools::SandboxPolicy;
58
59/// A thin, `Clone` + `Debug` wrapper around
60/// `Arc<dyn PermissionsApprovalHandler>` so
61/// [`crate::tools::ToolContext`] (which derives both) can carry one as an
62/// ambient field — mirroring the existing `write_observer: Option<Arc<dyn
63/// WriteObserver>>` precedent, except [`PermissionsApprovalHandler`] (a
64/// pre-existing P5-1 public trait) doesn't itself require `Debug` as a
65/// supertrait, so this newtype supplies a placeholder `Debug` impl instead
66/// of widening that trait's contract for every existing implementor.
67#[derive(Clone)]
68pub struct SandboxApprovalHandler(pub Arc<dyn PermissionsApprovalHandler>);
69
70impl std::fmt::Debug for SandboxApprovalHandler {
71    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
72        f.write_str("SandboxApprovalHandler(..)")
73    }
74}
75
76impl std::ops::Deref for SandboxApprovalHandler {
77    type Target = dyn PermissionsApprovalHandler;
78    fn deref(&self) -> &Self::Target {
79        &*self.0
80    }
81}
82
83// ---------------------------------------------------------------------------
84// Config-facing enums (§3.1 `capabilities.permissions.sandbox.escalation` /
85// `.env_policy`) — both carry a strictness RANK so the project-overlay
86// monotonic-tightening clamp (`crate::configfile::clamp_project_permissions`)
87// can compare a project's requested value against the trusted layer's,
88// exactly like `sandbox_rank`/`approval_rank` already do for `tier`/
89// `approval`.
90// ---------------------------------------------------------------------------
91
92/// `capabilities.permissions.sandbox.escalation` (§3.1): what happens when a
93/// confining fs tier is requested but this platform/kernel cannot actually
94/// enforce it. `Deny` (the default) refuses to run the subprocess at all —
95/// the cardinal "never silently unsandboxed" rule. `Ask` routes the decision
96/// through `crate::permissions::PermissionsApprovalHandler` (P5-1's
97/// `permissions.approvals` seam, wired here per this module's build brief).
98/// `Allow` auto-permits an unconfined run with a loud, one-time warning.
99#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
100pub enum SandboxEscalation {
101    /// Refuse to run the subprocess when confinement can't be established
102    /// (fail-closed; the default).
103    #[default]
104    Deny,
105    /// Consult the installed [`PermissionsApprovalHandler`] for a per-call
106    /// decision; no handler installed denies (fail-closed, same posture
107    /// `PermissionsApprovalHandler`'s own doc comment already documents).
108    Ask,
109    /// Auto-permit an unconfined run, with a loud one-time warning.
110    Allow,
111}
112
113impl SandboxEscalation {
114    /// Strictness rank — LOWER is stricter, same convention as
115    /// `configfile::sandbox_rank`/`approval_rank` (`Deny` prompts nothing
116    /// through, the strictest floor; `Allow` is the loosest, an unconfined
117    /// run with only a warning).
118    pub fn rank(self) -> u8 {
119        match self {
120            SandboxEscalation::Deny => 0,
121            SandboxEscalation::Ask => 1,
122            SandboxEscalation::Allow => 2,
123        }
124    }
125
126    /// Parse the config string (`"deny" | "ask" | "allow"`), `_`/`-`/case
127    /// normalized like every other sandbox-adjacent string parser in this
128    /// crate (`configfile::parse_sandbox_str`/`parse_approval_str`).
129    pub fn parse(s: &str) -> Option<Self> {
130        match s.replace('_', "-").to_ascii_lowercase().as_str() {
131            "deny" => Some(SandboxEscalation::Deny),
132            "ask" => Some(SandboxEscalation::Ask),
133            "allow" => Some(SandboxEscalation::Allow),
134            _ => None,
135        }
136    }
137}
138
139/// `capabilities.permissions.sandbox.env_policy` (§3.1): child-process
140/// environment sanitization for the spawned `bash`/`shell` subprocess.
141/// `Inherit` (the default) is today's behavior — the parent's environment
142/// (plus `core.shell_env_snapshot`, if configured) passes through
143/// unchanged. `Filtered` strips a sensitive-variable denylist (tokens,
144/// keys, cloud credentials). `None` keeps only `PATH` and a couple of
145/// universally-needed variables (`HOME`, `TERM`, `LANG`) — nearest to a
146/// bare-metal shell with nothing extra.
147#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
148pub enum SandboxEnvPolicy {
149    /// Full environment passthrough — today's behavior (byte-identical).
150    #[default]
151    Inherit,
152    /// Strip a sensitive-variable denylist; everything else passes through.
153    Filtered,
154    /// Bare minimum: `PATH`/`HOME`/`TERM`/`LANG` only.
155    None,
156}
157
158impl SandboxEnvPolicy {
159    /// Strictness rank — LOWER is stricter (`None` exposes the least,
160    /// `Inherit` the most), same convention as [`SandboxEscalation::rank`].
161    pub fn rank(self) -> u8 {
162        match self {
163            SandboxEnvPolicy::None => 0,
164            SandboxEnvPolicy::Filtered => 1,
165            SandboxEnvPolicy::Inherit => 2,
166        }
167    }
168
169    /// Parse the config string (`"inherit" | "filtered" | "none"`).
170    pub fn parse(s: &str) -> Option<Self> {
171        match s.replace('_', "-").to_ascii_lowercase().as_str() {
172            "inherit" => Some(SandboxEnvPolicy::Inherit),
173            "filtered" => Some(SandboxEnvPolicy::Filtered),
174            "none" => Some(SandboxEnvPolicy::None),
175            _ => None,
176        }
177    }
178}
179
180/// Environment variable NAMES (exact match, case-sensitive — POSIX env
181/// var convention) stripped under [`SandboxEnvPolicy::Filtered`]: API
182/// keys/tokens this crate itself resolves credentials from
183/// (`crate::config`'s `api_key_env`/provider-default env vars), common
184/// cloud-provider credential variables, and generic secret-shaped names.
185/// Deliberately a denylist rather than an allowlist (`Filtered` is the
186/// MIDDLE tier — "strip the obviously sensitive ones", not "start from
187/// nothing", which is what [`SandboxEnvPolicy::None`] is for).
188const FILTERED_ENV_DENYLIST_PREFIXES: &[&str] = &[
189    "OPENROUTER_API_KEY",
190    "OPENAI_API_KEY",
191    "ANTHROPIC_API_KEY",
192    "AWS_ACCESS_KEY_ID",
193    "AWS_SECRET_ACCESS_KEY",
194    "AWS_SESSION_TOKEN",
195    "GITHUB_TOKEN",
196    "GH_TOKEN",
197    "GITLAB_TOKEN",
198    "NPM_TOKEN",
199    "DOCKER_PASSWORD",
200    "GOOGLE_APPLICATION_CREDENTIALS",
201    "AZURE_CLIENT_SECRET",
202    "SSH_AUTH_SOCK",
203    "SUPERCODE_",
204];
205
206/// Whether `key` should be stripped under [`SandboxEnvPolicy::Filtered`]:
207/// an exact match against [`FILTERED_ENV_DENYLIST_PREFIXES`], OR a
208/// case-insensitive substring match on `TOKEN`/`SECRET`/`PASSWORD`/`_KEY`/
209/// `CREDENTIAL` — the generic "this looks like a secret" heuristic every
210/// credential-scanning tool uses, applied here as a denylist (a false
211/// positive just costs the child a var it didn't need; a false negative
212/// under `Filtered` is the actually dangerous direction, so the heuristic
213/// is deliberately broad).
214fn is_filtered_env_key(key: &str) -> bool {
215    let upper = key.to_ascii_uppercase();
216    if FILTERED_ENV_DENYLIST_PREFIXES
217        .iter()
218        .any(|p| upper == *p || upper.starts_with(p))
219    {
220        return true;
221    }
222    [
223        "TOKEN",
224        "SECRET",
225        "PASSWORD",
226        "_KEY",
227        "CREDENTIAL",
228        "APIKEY",
229    ]
230    .iter()
231    .any(|needle| upper.contains(needle))
232}
233
234/// Environment variables kept under [`SandboxEnvPolicy::None`] — the bare
235/// minimum a POSIX shell needs to do anything useful at all.
236const MINIMAL_ENV_KEEP: &[&str] = &["PATH", "HOME", "TERM", "LANG", "LC_ALL", "TMPDIR"];
237
238/// Build the environment the subprocess should see, starting from `base`
239/// (the process's own inherited environment, or `ctx.shell_env`'s snapshot
240/// when one is configured — the caller decides `base`, this function only
241/// applies the POLICY on top of it). `Inherit` returns `base` unchanged
242/// (byte-identical to pre-P5-10 behavior — the common case, since
243/// `env_policy` defaults to `Inherit`).
244pub fn apply_env_policy<I, K, V>(policy: SandboxEnvPolicy, base: I) -> Vec<(String, String)>
245where
246    I: IntoIterator<Item = (K, V)>,
247    K: Into<String>,
248    V: Into<String>,
249{
250    let base: Vec<(String, String)> = base
251        .into_iter()
252        .map(|(k, v)| (k.into(), v.into()))
253        .collect();
254    match policy {
255        SandboxEnvPolicy::Inherit => base,
256        SandboxEnvPolicy::Filtered => base
257            .into_iter()
258            .filter(|(k, _)| !is_filtered_env_key(k))
259            .collect(),
260        SandboxEnvPolicy::None => base
261            .into_iter()
262            .filter(|(k, _)| MINIMAL_ENV_KEEP.contains(&k.as_str()))
263            .collect(),
264    }
265}
266
267// ---------------------------------------------------------------------------
268// The pure decision layer — no I/O, fully unit-testable.
269// ---------------------------------------------------------------------------
270
271/// What to do about filesystem confinement for one subprocess spawn.
272#[derive(Debug, Clone, PartialEq, Eq)]
273pub enum FsDecision {
274    /// The tier is `DangerFullAccess`, or the OS backstop isn't active for
275    /// this call (`enabled` resolves to `false`) — no confinement is even
276    /// attempted. Byte-identical to pre-P5-10 behavior.
277    NotRequested,
278    /// Confinement IS available on this platform/kernel — apply it for
279    /// real (the caller installs the `pre_exec` closure).
280    Confine,
281    /// Confinement was requested but this platform/kernel cannot provide
282    /// it, and `escalation` says to proceed anyway (`allow`, or `ask` with
283    /// an approving handler) — run UNCONFINED, with `reason` surfaced as a
284    /// loud, one-time warning.
285    RunUnconfinedWithWarning {
286        /// Human-readable reason, fed to the one-time warning + the tool
287        /// error message (on the `Refuse` sibling) so the honest gap is
288        /// always named, never silent.
289        reason: String,
290    },
291    /// Confinement was requested, this platform/kernel cannot provide it,
292    /// and `escalation` says to refuse (`deny`, the default; or `ask` with
293    /// no handler installed / a denying handler). The subprocess is NOT
294    /// spawned at all.
295    Refuse {
296        /// Human-readable reason, returned to the model as the tool error.
297        reason: String,
298    },
299}
300
301/// What to do about network confinement for one subprocess spawn. Unlike
302/// [`FsDecision`], this never refuses the whole call — network.enabled is
303/// an independent, best-effort axis (§build brief item 3): "surface the
304/// gap, never claim enforcement you lack", not a hard fs-style gate. A
305/// caller that ALSO has an [`FsDecision::Refuse`] for the same call still
306/// refuses (that decision wins), but a network-only gap never blocks a
307/// call that has no fs confinement problem.
308#[derive(Debug, Clone, PartialEq, Eq)]
309pub enum NetDecision {
310    /// `network.enabled` is `false` (the default) — nothing to do,
311    /// byte-identical to pre-P5-10 behavior.
312    NotRequested,
313    /// A coarse (no domain granularity) network cut-off is available and
314    /// requested — apply it for real.
315    Confine,
316    /// `network.enabled` is `true` but this platform/kernel can't deliver
317    /// what was asked (domain allow/deny lists — needs the out-of-scope
318    /// TLS-MITM proxy — or coarse cut-off itself is unavailable). Never
319    /// silently dropped: surfaced as a one-time warning, the subprocess
320    /// still runs (network-UNCONFINED, everything else about the call is
321    /// unaffected).
322    GapWarn {
323        /// Human-readable reason, fed to the one-time warning.
324        reason: String,
325    },
326}
327
328/// Whether the OS-level backstop is ACTIVE for `tier`/`os_enabled` at all —
329/// the "turn the sandbox ENUM into an enabled/disabled OS engagement" half
330/// of §3.1's table. `DangerFullAccess` is an absolute opt-out (never
331/// confine, regardless of `os_enabled` — §3.1 item 1: "no confinement
332/// (opt-out)"). Otherwise: an EXPLICIT `os_enabled` value wins; `None`
333/// (never set — the bare `sandbox = "<tier>"` shorthand, or a CLI
334/// `--sandbox` flag, neither of which touch the table's `enabled` key at
335/// all) preserves the PRE-P5-10 trigger this crate already shipped
336/// (`tools/builtins.rs`'s macOS seatbelt firing off `ctx.sandbox` alone,
337/// no separate gate) — so an existing CLI user or the `cx-parity` preset
338/// (bare `sandbox = "workspace_write"`, no `enabled` key) keeps its
339/// current confining behavior byte-for-byte, while `cc-parity`'s explicit
340/// table-form `enabled = false` (§3.1's own "OS sandbox OFF… opt-in"
341/// comment) is honored as a real, independent off-switch.
342pub fn os_sandbox_active(tier: SandboxPolicy, os_enabled: Option<bool>) -> bool {
343    match tier {
344        SandboxPolicy::DangerFullAccess => false,
345        _ => os_enabled.unwrap_or(true),
346    }
347}
348
349/// Decide what to do about FILESYSTEM confinement for one subprocess spawn.
350/// Pure — `fs_available` is the caller's REAL probe result
351/// ([`landlock_available`] on Linux, `true` on macOS via the existing
352/// seatbelt path which this function is not consulted for — see
353/// `tools::builtins::build_sandboxed_sh`'s doc comment), never computed
354/// internally, so every branch (including the platform-can't-enforce ones)
355/// is directly testable without touching a kernel.
356#[allow(clippy::too_many_arguments)]
357pub fn decide_fs(
358    tier: SandboxPolicy,
359    os_enabled: Option<bool>,
360    fs_available: bool,
361    escalation: SandboxEscalation,
362    approval: Option<&dyn PermissionsApprovalHandler>,
363    subject: &str,
364) -> FsDecision {
365    if !os_sandbox_active(tier, os_enabled) {
366        return FsDecision::NotRequested;
367    }
368    if fs_available {
369        return FsDecision::Confine;
370    }
371    let reason = format!(
372        "sandbox: filesystem confinement ({tier:?}) was requested but is unavailable on this \
373         platform/kernel (no Landlock support) for `{subject}`"
374    );
375    resolve_escalation(escalation, approval, "bash", subject, reason)
376}
377
378/// Decide what to do about NETWORK confinement for one subprocess spawn.
379/// Pure — `net_available` is the caller's real probe result
380/// ([`netns_available`] on Linux). Never gates on `escalation` (see
381/// [`NetDecision`]'s doc comment) — a network gap is always a warn, never a
382/// refuse, so this needs no approval handler at all.
383pub fn decide_net(
384    network_enabled: bool,
385    has_domain_rules: bool,
386    net_available: bool,
387) -> NetDecision {
388    if !network_enabled {
389        return NetDecision::NotRequested;
390    }
391    if has_domain_rules {
392        return NetDecision::GapWarn {
393            reason: "sandbox: capabilities.permissions.sandbox.network.allow_domains/\
394                     deny_domains was set, but domain-level network filtering has no OS \
395                     primitive on this platform — that needs an out-of-scope TLS-MITM proxy \
396                     (COMPOSABLE-HARNESS-DESIGN.md gap honesty note). Network was NOT \
397                     confined for this call."
398                .to_string(),
399        };
400    }
401    if net_available {
402        return NetDecision::Confine;
403    }
404    NetDecision::GapWarn {
405        reason: "sandbox: capabilities.permissions.sandbox.network.enabled was set, but a \
406                 coarse network cut-off is unavailable on this platform/kernel (no \
407                 unprivileged network-namespace support). Network was NOT confined for this \
408                 call."
409            .to_string(),
410    }
411}
412
413/// Shared `deny`/`ask`/`allow` resolution for an unenforceable FS request —
414/// factored out of [`decide_fs`] so a future confining axis (were one ever
415/// added) reuses the exact same escalation semantics rather than a second,
416/// possibly-drifting copy.
417fn resolve_escalation(
418    escalation: SandboxEscalation,
419    approval: Option<&dyn PermissionsApprovalHandler>,
420    tool: &str,
421    subject: &str,
422    reason: String,
423) -> FsDecision {
424    match escalation {
425        SandboxEscalation::Deny => FsDecision::Refuse { reason },
426        SandboxEscalation::Allow => FsDecision::RunUnconfinedWithWarning { reason },
427        SandboxEscalation::Ask => match approval {
428            Some(handler) => {
429                let raw_args = serde_json::Value::Null;
430                let req = ApprovalRequest {
431                    tool,
432                    subject: Some(subject),
433                    raw_args: &raw_args,
434                };
435                match handler.ask(&req) {
436                    ApprovalOutcome::Deny => FsDecision::Refuse { reason },
437                    ApprovalOutcome::Allow | ApprovalOutcome::AllowForSession => {
438                        FsDecision::RunUnconfinedWithWarning { reason }
439                    }
440                }
441            }
442            // No handler installed: fail-closed, same posture
443            // `PermissionsApprovalHandler`'s own doc comment documents for
444            // the P5-1 rule engine's `Ask` tier.
445            None => FsDecision::Refuse { reason },
446        },
447    }
448}
449
450/// Print `reason` to stderr ONCE per distinct reason string, for the
451/// lifetime of this process — the "loud, one-time persistent warning" the
452/// build brief calls for on an `escalation = "allow"`/approved-`ask` run,
453/// and on a network gap. Deduped by exact text (not a blanket
454/// once-per-process `Once`) so a DIFFERENT gap later in the same run still
455/// gets its own warning — only an EXACT repeat is suppressed.
456pub fn warn_once(reason: &str) {
457    static WARNED: OnceLock<Mutex<std::collections::HashSet<String>>> = OnceLock::new();
458    let set = WARNED.get_or_init(|| Mutex::new(std::collections::HashSet::new()));
459    if let Ok(mut set) = set.lock() {
460        if set.insert(reason.to_string()) {
461            eprintln!("\x1b[33mwarning: {reason}\x1b[0m");
462        }
463    }
464}
465
466// ---------------------------------------------------------------------------
467// Linux: real Landlock fs enforcement + coarse network-namespace cut-off.
468// ---------------------------------------------------------------------------
469
470/// Whether real Landlock filesystem confinement is available on THIS
471/// process's kernel — a genuine, side-effect-free (beyond dropping one
472/// ruleset file descriptor) PARENT-PROCESS probe: it builds a
473/// `CompatLevel::HardRequirement` ruleset requiring exactly the write-access
474/// rights `apply_linux_confinement` would later request and checks
475/// whether `Ruleset::create()` succeeds — it deliberately never calls
476/// `restrict_self()` (that confines the CALLING process/thread permanently
477/// and every future child of it — calling it here would confine supercode
478/// ITSELF, exactly the "confinement targets the child, not supercode"
479/// invariant this module must never violate). Cached for the process
480/// lifetime (the kernel's Landlock support can't change at runtime).
481#[cfg(target_os = "linux")]
482pub fn landlock_available() -> bool {
483    static AVAILABLE: OnceLock<bool> = OnceLock::new();
484    *AVAILABLE.get_or_init(|| {
485        use landlock::{AccessFs, CompatLevel, Compatible, Ruleset, RulesetAttr, ABI};
486        Ruleset::default()
487            .set_compatibility(CompatLevel::HardRequirement)
488            .handle_access(AccessFs::from_write(ABI::V1))
489            .and_then(|r| r.create())
490            .is_ok()
491    })
492}
493
494/// Non-Linux: Landlock never exists — always unavailable. Kept as a real
495/// function (not `cfg!`-inlined at call sites) so callers stay identical
496/// across platforms.
497#[cfg(not(target_os = "linux"))]
498pub fn landlock_available() -> bool {
499    false
500}
501
502/// Whether an unprivileged, self-contained network namespace cut-off is
503/// available on THIS process's kernel — a real probe, but one that (unlike
504/// [`landlock_available`]) genuinely can't be done risk-free in the calling
505/// process itself (`unshare(2)` acts on the CALLING process/thread, so
506/// probing it directly would isolate supercode's own network, not just
507/// check availability). Instead this forks a disposable, single-purpose
508/// child that does nothing but attempt the unshare and immediately
509/// `_exit()` with the result — never touches the allocator, locks, or any
510/// other state the parent might hold mid-fork (the standard safe shape for
511/// a post-fork child that never execs), so it carries none of `pre_exec`'s
512/// usual multi-threaded-fork hazards. Cached for the process lifetime.
513#[cfg(target_os = "linux")]
514pub fn netns_available() -> bool {
515    static AVAILABLE: OnceLock<bool> = OnceLock::new();
516    *AVAILABLE.get_or_init(probe_netns_fork)
517}
518
519#[cfg(not(target_os = "linux"))]
520/// Non-Linux platforms cannot create the Linux network namespace used for
521/// coarse network isolation, so the capability is always unavailable.
522pub fn netns_available() -> bool {
523    false
524}
525
526#[cfg(target_os = "linux")]
527fn probe_netns_fork() -> bool {
528    // SAFETY: the child touches nothing but raw syscalls (`unshare`,
529    // `_exit`) between `fork()` and exit — no allocation, no locks, no
530    // library calls that could be mid-acquired in another thread at fork
531    // time. This is the textbook safe post-fork-no-exec shape.
532    unsafe {
533        let pid = libc::fork();
534        if pid == 0 {
535            let rc = libc::unshare(libc::CLONE_NEWUSER | libc::CLONE_NEWNET);
536            libc::_exit(i32::from(rc != 0));
537        } else if pid > 0 {
538            let mut status: libc::c_int = 0;
539            if libc::waitpid(pid, &mut status, 0) != pid {
540                return false;
541            }
542            libc::WIFEXITED(status) && libc::WEXITSTATUS(status) == 0
543        } else {
544            false
545        }
546    }
547}
548
549/// Real fs/net confinement, installed on `cmd` as a `pre_exec` closure that
550/// runs in the FORKED CHILD (after `fork()`, before `exec()` —
551/// [`crate::agent::Agent`]/supercode itself is never touched; only the
552/// spawned subprocess tree is). `cwd`/`extra_write_dirs` MUST already be
553/// resolved via [`crate::safe_path::resolve_real`] (real, symlink-resolved
554/// paths) — Landlock rules operate on directory file descriptors opened
555/// from these exact paths, so the same dual lexical+resolved discipline
556/// every other containment check in this crate uses applies here too (a
557/// symlink'd `cwd` must grant the REAL target directory, not the symlink's
558/// lexical location).
559#[cfg(target_os = "linux")]
560pub fn apply_linux_confinement(
561    cmd: &mut tokio::process::Command,
562    confine_fs: bool,
563    fs_allow_writes: bool,
564    cwd: PathBuf,
565    extra_write_dirs: Vec<PathBuf>,
566    confine_net: bool,
567) {
568    if !confine_fs && !confine_net {
569        return;
570    }
571    // Captured by value into the closure — no shared/borrowed state crosses
572    // the fork boundary.
573    let uid = unsafe { libc::getuid() };
574    let gid = unsafe { libc::getgid() };
575    // SAFETY: see the closure body's own comments — every operation is a
576    // raw syscall (or a `/proc/self/*` write via raw fd ops), no
577    // allocation-heavy std IO beyond what the `landlock` crate itself does
578    // (small, bounded `Vec`s over a handful of paths), matching this
579    // module's build brief ("a pre_exec closure calling the landlock
580    // crate's restrict_self() in the child after fork, before exec").
581    unsafe {
582        cmd.pre_exec(move || {
583            if confine_net {
584                netns_isolate_self(uid, gid)
585                    .map_err(|e| std::io::Error::other(format!("sandbox netns: {e}")))?;
586            }
587            if confine_fs {
588                landlock_restrict_self(&cwd, &extra_write_dirs, fs_allow_writes)
589                    .map_err(|e| std::io::Error::other(format!("sandbox landlock: {e}")))?;
590            }
591            Ok(())
592        });
593    }
594}
595
596/// Isolate the CALLING process (the forked child, pre-exec) into a fresh,
597/// unprivileged user+network namespace with NO network interfaces beyond
598/// loopback — a real kernel-level all-network cutoff (coarse: no domain
599/// granularity, see [`decide_net`]'s doc comment for why that's out of
600/// reach here). `uid`/`gid` (captured in the PARENT before `fork()`) are
601/// mapped identity-onto-self inside the new user namespace
602/// (`/proc/self/uid_map`/`gid_map`, the same `unshare(1) --map-root-user`
603/// technique) so file-permission checks against the workspace are
604/// UNAFFECTED — without this mapping the process would run as the
605/// namespace's unmapped "overflow" uid and lose access to its own files.
606#[cfg(target_os = "linux")]
607fn netns_isolate_self(uid: libc::uid_t, gid: libc::gid_t) -> Result<(), String> {
608    unsafe {
609        if libc::unshare(libc::CLONE_NEWUSER | libc::CLONE_NEWNET) != 0 {
610            return Err(format!(
611                "unshare(CLONE_NEWUSER|CLONE_NEWNET): errno {}",
612                *libc::__errno_location()
613            ));
614        }
615    }
616    write_proc_self_raw("setgroups", b"deny")?;
617    write_proc_self_raw("uid_map", format!("0 {uid} 1\n").as_bytes())?;
618    write_proc_self_raw("gid_map", format!("0 {gid} 1\n").as_bytes())?;
619    Ok(())
620}
621
622/// Write `contents` to `/proc/self/<name>` using raw `open`/`write`/`close`
623/// syscalls (not `std::fs`) — deliberately minimal post-fork-pre-exec code,
624/// consistent with `apply_linux_confinement`'s safety comment.
625#[cfg(target_os = "linux")]
626fn write_proc_self_raw(name: &str, contents: &[u8]) -> Result<(), String> {
627    let path = format!("/proc/self/{name}\0");
628    unsafe {
629        let fd = libc::open(path.as_ptr() as *const libc::c_char, libc::O_WRONLY);
630        if fd < 0 {
631            return Err(format!(
632                "open(/proc/self/{name}): errno {}",
633                *libc::__errno_location()
634            ));
635        }
636        let n = libc::write(fd, contents.as_ptr() as *const libc::c_void, contents.len());
637        let write_errno = *libc::__errno_location();
638        libc::close(fd);
639        if n != contents.len() as isize {
640            return Err(format!("write(/proc/self/{name}): errno {write_errno}"));
641        }
642    }
643    Ok(())
644}
645
646/// Real Landlock ruleset construction + `restrict_self()` — runs ONLY
647/// inside the forked child's `pre_exec` closure (see
648/// `apply_linux_confinement`). Restricts WRITE-family access rights
649/// crate-wide (`AccessFs::from_write`, ABI V1 — the conservative baseline
650/// every Landlock-supporting kernel honors; read/execute are never
651/// "handled" by this ruleset at all, so they stay exactly as unrestricted
652/// as [`crate::tools::SandboxPolicy`]'s own doc comment already promises:
653/// "reads broad" for `workspace_write`, "reads allowed" for `read_only` —
654/// this module only ever tightens WRITES). `fs_allow_writes` (true for
655/// `WorkspaceWrite`, false for `ReadOnly`) gates whether ANY path gets a
656/// write-grant rule at all; when true, `cwd` + `extra_write_dirs` (system
657/// temp) are the only writable roots. Fails (never silently degrades) if
658/// the resulting status isn't `RulesetStatus::FullyEnforced` — a
659/// `PartiallyEnforced`/`NotEnforced` status would mean this function
660/// claimed confinement it didn't actually get.
661#[cfg(target_os = "linux")]
662fn landlock_restrict_self(
663    cwd: &Path,
664    extra_write_dirs: &[PathBuf],
665    fs_allow_writes: bool,
666) -> Result<(), String> {
667    use landlock::{
668        path_beneath_rules, AccessFs, CompatLevel, Compatible, Ruleset, RulesetAttr,
669        RulesetCreatedAttr, RulesetStatus, ABI,
670    };
671    let write_access = AccessFs::from_write(ABI::V1);
672    let created = Ruleset::default()
673        .set_compatibility(CompatLevel::HardRequirement)
674        .handle_access(write_access)
675        .map_err(|e| e.to_string())?
676        .create()
677        .map_err(|e| e.to_string())?
678        .set_compatibility(CompatLevel::HardRequirement);
679    let created = if fs_allow_writes {
680        let mut dirs = Vec::with_capacity(1 + extra_write_dirs.len());
681        dirs.push(cwd.to_path_buf());
682        dirs.extend(extra_write_dirs.iter().cloned());
683        created
684            .add_rules(path_beneath_rules(&dirs, write_access))
685            .map_err(|e| e.to_string())?
686    } else {
687        created
688    };
689    let status = created.restrict_self().map_err(|e| e.to_string())?;
690    if status.ruleset != RulesetStatus::FullyEnforced {
691        return Err(format!(
692            "ruleset not fully enforced ({:?}) — refusing to claim confinement it doesn't have",
693            status.ruleset
694        ));
695    }
696    Ok(())
697}