Skip to main content

systemprompt_models/subprocess/
mod.rs

1//! Spawning, identifying, and reaping the detached agent and MCP children the
2//! supervisor owns.
3//!
4//! # Spawning
5//!
6//! [`spawn_supervised`] is the only sanctioned way to start a child. It runs
7//! every spawn on one dedicated thread and, where the platform offers it, asks
8//! the kernel to `SIGTERM` the child if this process dies, so a crash, panic,
9//! or `SIGKILL` of the supervisor cannot strand an agent holding a port.
10//!
11//! # Identity
12//!
13//! The supervisor stamps environment markers at spawn time; shutdown,
14//! reconciliation, and port reclamation read them back off the live process to
15//! confirm a registry PID still names *this* installation's child before
16//! signalling it. PIDs are recycled, and group-signalling a stale PID
17//! (`kill(-pid)`) could reach an unrelated session leader — so a row is only
18//! ever signalled once both the subprocess marker and the exact
19//! `name_key=service_name` pairing are found.
20//!
21//! # Platform support
22//!
23//! The two halves of supervision have different reach, and conflating them is
24//! what stranded ports on macOS:
25//!
26//! - **Identity and reap checks** ([`live_pid_is_subprocess`], [`is_zombie`])
27//!   work on Linux, via `/proc`, and on macOS, via `sysctl(KERN_PROCARGS2)` and
28//!   `proc_pidinfo`. Report the platform's coverage with
29//!   [`identity_verification_supported`]; where it is absent the checks are
30//!   fail-closed stubs that never confirm an identity, so no process is ever
31//!   signalled on a guess.
32//! - **Parent-death prevention** is `prctl(PR_SET_PDEATHSIG)` and therefore
33//!   Linux-only. macOS has no equivalent that survives `execve`, and the kqueue
34//!   and pipe-EOF alternatives all require cooperation from the child binary —
35//!   which is an arbitrary MCP server or agent executable here. A `SIGKILL`ed
36//!   supervisor on macOS therefore leaves its children reparented to `launchd`
37//!   and still holding their ports; the identity check above is what lets the
38//!   next start reclaim them instead of erroring out.
39//!
40//! Copyright (c) systemprompt.io — Business Source License 1.1.
41//! See <https://systemprompt.io> for licensing details.
42
43use std::process::Command;
44use std::sync::OnceLock;
45use std::sync::mpsc::{Sender, channel};
46
47#[cfg(target_os = "linux")]
48mod linux;
49#[cfg(target_os = "linux")]
50pub use linux::{is_zombie, live_pid_is_subprocess};
51
52#[cfg(target_os = "macos")]
53mod darwin;
54#[cfg(target_os = "macos")]
55pub use darwin::{is_zombie, live_pid_is_subprocess};
56
57#[cfg(not(any(target_os = "linux", target_os = "macos")))]
58mod unsupported;
59#[cfg(not(any(target_os = "linux", target_os = "macos")))]
60pub use unsupported::{is_zombie, live_pid_is_subprocess};
61
62pub const SUBPROCESS_MARKER_ENV: &str = "SYSTEMPROMPT_SUBPROCESS";
63pub const AGENT_NAME_ENV: &str = "AGENT_NAME";
64pub const MCP_SERVICE_ID_ENV: &str = "MCP_SERVICE_ID";
65
66// Why: an env var, not a profile field. The same profile YAML is used on the
67// operator's machine (must route to the remote tenant) and inside the container
68// (must not — it is the tenant); the document is byte-identical in both, so
69// only the environment can tell.
70pub const DEPLOYMENT_HOST_ENV: &str = "SYSTEMPROMPT_DEPLOYMENT_HOST";
71
72// Why: fallback so tenants deployed before `DEPLOYMENT_HOST_ENV` existed keep
73// working without a redeploy. Fly injects it; nothing we generate does.
74const FLY_HOST_ENV: &str = "FLY_APP_NAME";
75
76// Why: `DEPLOYMENT_HOST_ENV` wins over the Fly marker because the one we
77// generate is the one an operator can steer; takes a lookup so tests never
78// touch process-global state.
79pub fn deployment_host(lookup: impl Fn(&str) -> Option<String>) -> Option<String> {
80    [DEPLOYMENT_HOST_ENV, FLY_HOST_ENV].iter().find_map(|name| {
81        lookup(name)
82            .map(|value| value.trim().to_owned())
83            .filter(|value| !value.is_empty())
84    })
85}
86
87// Why: a process that answers `false` here will try to reach its deployment
88// host over the network. Answering `false` while actually on that host is what
89// makes a server attempt to route a command to itself, so any one marker is
90// proof and only their absence means "elsewhere".
91pub fn is_deployment_host(lookup: impl Fn(&str) -> Option<String>) -> bool {
92    deployment_host(lookup).is_some()
93}
94
95// Why: both spawners clear the child environment and rebuild it from an
96// allowlist, and both need exactly this set. It lives here so the two cannot
97// diverge.
98pub fn inherited_parent_env(lookup: impl Fn(&str) -> Option<String>) -> Vec<(String, String)> {
99    let mut env: Vec<(String, String)> = [DEPLOYMENT_HOST_ENV, FLY_HOST_ENV, "PATH", "HOME"]
100        .iter()
101        .filter_map(|name| lookup(name).map(|value| ((*name).to_owned(), value)))
102        .collect();
103
104    if let Some(entry) = crate::net::trusted_hosts_env_entry(&lookup) {
105        env.push(entry);
106    }
107
108    env
109}
110
111type SpawnReply = Sender<std::io::Result<u32>>;
112
113pub fn spawn_supervised(cmd: Command) -> std::io::Result<u32> {
114    let sender = spawner()
115        .as_ref()
116        .map_err(|e| std::io::Error::other(e.clone()))?;
117
118    let (reply_tx, reply_rx) = channel();
119    sender
120        .send((cmd, reply_tx))
121        .map_err(|disconnected| std::io::Error::other(disconnected.to_string()))?;
122    reply_rx
123        .recv()
124        .map_err(|disconnected| std::io::Error::other(disconnected.to_string()))?
125}
126
127fn spawner() -> &'static Result<Sender<(Command, SpawnReply)>, String> {
128    static SPAWNER: OnceLock<Result<Sender<(Command, SpawnReply)>, String>> = OnceLock::new();
129    SPAWNER.get_or_init(|| {
130        let (tx, rx) = channel::<(Command, SpawnReply)>();
131        std::thread::Builder::new()
132            .name("subprocess-spawner".to_owned())
133            .spawn(move || {
134                while let Ok((mut cmd, reply)) = rx.recv() {
135                    let outcome = spawn_on_this_thread(&mut cmd);
136                    if reply.send(outcome).is_err() {
137                        tracing::warn!(
138                            "Spawn requester vanished before collecting the child pid; the child \
139                             is unregistered and will only be cleaned up by its parent-death signal"
140                        );
141                    }
142                }
143            })
144            .map(|_handle| tx)
145            .map_err(|e| format!("could not start the subprocess spawner thread: {e}"))
146    })
147}
148
149fn spawn_on_this_thread(cmd: &mut Command) -> std::io::Result<u32> {
150    #[cfg(target_os = "linux")]
151    linux::arm_parent_death_signal(cmd);
152
153    let child = cmd.spawn()?;
154    let pid = child.id();
155    #[expect(
156        clippy::mem_forget,
157        reason = "detached child: skip Child's drop-time wait so it keeps running after this \
158                  returns; reaping is the caller's business via is_zombie"
159    )]
160    std::mem::forget(child);
161    Ok(pid)
162}
163
164// Why: pgid 0 makes the child its own group leader (pgid == pid), so the
165// supervisor can signal the whole group on shutdown and reach any helper
166// processes the child spawns, not just the child itself.
167#[cfg(unix)]
168pub fn place_in_own_process_group(command: &mut Command) {
169    use std::os::unix::process::CommandExt;
170    command.process_group(0);
171}
172
173#[cfg(windows)]
174pub fn place_in_own_process_group(command: &mut Command) {
175    use std::os::windows::process::CommandExt;
176    const CREATE_NEW_PROCESS_GROUP: u32 = 0x0000_0200;
177    command.creation_flags(CREATE_NEW_PROCESS_GROUP);
178}
179
180#[must_use]
181pub const fn identity_verification_supported() -> bool {
182    cfg!(any(target_os = "linux", target_os = "macos"))
183}
184
185#[must_use]
186pub fn signalable_pid(pid: u32) -> Option<i32> {
187    if pid == 0 {
188        return None;
189    }
190    i32::try_from(pid).ok()
191}
192
193#[must_use]
194pub fn environ_identifies_child(environ: &[u8], name_key: &str, service_name: &str) -> bool {
195    let marker = format!("{SUBPROCESS_MARKER_ENV}=1");
196    let expected_name = format!("{name_key}={service_name}");
197
198    let mut has_marker = false;
199    let mut has_name = false;
200    for entry in environ.split(|&b| b == 0) {
201        if entry == marker.as_bytes() {
202            has_marker = true;
203        } else if entry == expected_name.as_bytes() {
204            has_name = true;
205        }
206    }
207
208    has_marker && has_name
209}
210
211// Why: a `KERN_PROCARGS2` blob is `argc`, the exec path, NUL padding, `argc`
212// argv entries, then the environment — all NUL-delimited in one buffer. The
213// argv entries have to be skipped by count rather than searched past: an entry
214// is matched whole by `environ_identifies_child`, so a command line such as
215// `env MCP_SERVICE_ID=files …` would otherwise read as a marked environment and
216// get an unrelated process signalled. Kept here rather than in `darwin` so the
217// parse is unit-testable on every platform.
218#[must_use]
219pub fn environ_from_procargs2(blob: &[u8]) -> Option<&[u8]> {
220    const ARGC_LEN: usize = size_of::<i32>();
221
222    let argc_bytes: [u8; ARGC_LEN] = blob.get(..ARGC_LEN)?.try_into().ok()?;
223    let argc = usize::try_from(i32::from_ne_bytes(argc_bytes)).ok()?;
224
225    let mut rest = blob.get(ARGC_LEN..)?;
226    let exec_path_end = rest.iter().position(|&b| b == 0)?;
227    rest = rest.get(exec_path_end + 1..)?;
228
229    let argv_start = rest.iter().position(|&b| b != 0)?;
230    rest = rest.get(argv_start..)?;
231
232    for _ in 0..argc {
233        let entry_end = rest.iter().position(|&b| b == 0)?;
234        rest = rest.get(entry_end + 1..)?;
235    }
236
237    Some(rest)
238}