Skip to main content

ic_host_process/child/
mod.rs

1//! Explicit child ownership and cleanup, without application lifecycle policy.
2//!
3//! Callers own executable admission, IO, readiness, cancellation and deadlines.
4//! Group cleanup signals members of a newly created group; it cannot contain
5//! processes that escape that group or prove completion of external effects.
6
7#[cfg(target_os = "macos")]
8mod macos;
9#[cfg(test)]
10mod tests;
11
12use rustix::{
13    io::Errno,
14    process::{Pid, Signal, WaitId, WaitIdOptions, kill_process_group, waitid},
15};
16use std::{
17    fmt, io,
18    os::unix::process::CommandExt,
19    process::{Child, ChildStderr, ChildStdin, ChildStdout, Command, ExitStatus},
20};
21
22/// One exclusively owned child, normally spawned as a new process-group leader.
23///
24/// [`Self::spawn`] preserves the command's IO, environment and other settings,
25/// replacing its process-group selection with a new group. It performs no
26/// executable admission. The child must not change groups, and callers must
27/// not independently reap it (including through a global SIGCHLD handler).
28///
29/// Polling an exited leader signals remaining group members before reaping it.
30/// Drop makes a best-effort kill/reap attempt, including during unwinding. Use
31/// [`Self::terminate`] to observe cleanup failures. Cleanup is synchronous and
32/// has no wall-clock bound; successful signalling is not proof that descendants
33/// have exited or completed external effects. Only the direct child is reaped.
34/// Group signalling can succeed for only some members when credentials differ.
35pub struct OwnedChild {
36    child: Child,
37    group: bool,
38    status: Option<ExitStatus>,
39    owned: bool,
40}
41
42/// Failures observed during one explicit termination attempt.
43///
44/// Keep this separately from the caller's original cancellation/operation error.
45/// If group signalling fails, direct-child kill and reaping are still attempted.
46#[derive(Debug)]
47pub struct CleanupError {
48    /// Status retained if the direct child was reaped despite another failure.
49    pub status: Option<ExitStatus>,
50    /// Failure signalling the owned process group.
51    pub group_error: Option<io::Error>,
52    /// Failure killing the direct child (including fallback after group failure).
53    pub kill_error: Option<io::Error>,
54    /// Failure reaping the direct child.
55    pub wait_error: Option<io::Error>,
56}
57
58impl fmt::Display for CleanupError {
59    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
60        f.write_str("child cleanup failed")?;
61        for (operation, error) in [
62            ("group signal", &self.group_error),
63            ("child kill", &self.kill_error),
64            ("child wait", &self.wait_error),
65        ] {
66            if let Some(error) = error {
67                write!(f, "; {operation}: {error}")?;
68            }
69        }
70        Ok(())
71    }
72}
73
74impl std::error::Error for CleanupError {
75    fn source(&self) -> Option<&(dyn std::error::Error + 'static)> {
76        self.group_error
77            .as_ref()
78            .or(self.kill_error.as_ref())
79            .or(self.wait_error.as_ref())
80            .map(|error| error as &dyn std::error::Error)
81    }
82}
83
84impl OwnedChild {
85    /// Spawn once in a new owned process group, preserving caller-configured IO.
86    ///
87    /// # Errors
88    /// Returns the native spawn/setup failure. No retries are performed.
89    pub fn spawn(command: &mut Command) -> io::Result<Self> {
90        command.process_group(0);
91        Self::spawn_inner(command, true)
92    }
93
94    pub(crate) fn spawn_direct(command: &mut Command) -> io::Result<Self> {
95        Self::spawn_inner(command, false)
96    }
97
98    fn spawn_inner(command: &mut Command, group: bool) -> io::Result<Self> {
99        command.spawn().map(|child| Self {
100            child,
101            group,
102            status: None,
103            owned: true,
104        })
105    }
106
107    /// Direct-child PID, for observation only; it can be reused after reaping.
108    #[must_use]
109    pub fn id(&self) -> u32 {
110        self.child.id()
111    }
112
113    /// Take caller-configured piped stdin. Close it before waiting for EOF-driven children.
114    pub const fn take_stdin(&mut self) -> Option<ChildStdin> {
115        self.child.stdin.take()
116    }
117
118    /// Take caller-configured piped stdout; the caller owns draining and bounds.
119    pub const fn take_stdout(&mut self) -> Option<ChildStdout> {
120        self.child.stdout.take()
121    }
122
123    /// Take caller-configured piped stderr; the caller owns draining and bounds.
124    pub const fn take_stderr(&mut self) -> Option<ChildStderr> {
125        self.child.stderr.take()
126    }
127
128    /// Inspect exit without blocking on a running child; clean its group before reaping.
129    ///
130    /// Repeated successful calls return the cached status without signalling again.
131    /// # Errors
132    /// Returns native inspection, group-signal or reap errors. The leader remains
133    /// reserved on a group-signal failure, so explicit cleanup can still be attempted.
134    pub fn try_wait(&mut self) -> io::Result<Option<ExitStatus>> {
135        if let Some(status) = self.status {
136            return Ok(Some(status));
137        }
138        if !self.owned {
139            return Err(Errno::CHILD.into());
140        }
141        if self.group {
142            if !self.observe_exit(true)? {
143                return Ok(None);
144            }
145            self.signal_group()?;
146            self.reap().map(Some)
147        } else {
148            let result = retry_interrupted(|| self.child.try_wait());
149            if let Ok(Some(status)) = result {
150                self.status = Some(status);
151                self.owned = false;
152            }
153            self.check_wait_ownership(&result);
154            result
155        }
156    }
157
158    /// Wait for natural leader exit, then clean its group and reap the leader.
159    ///
160    /// Close/drain caller-owned pipes as needed before waiting. No deadline or
161    /// cancellation policy is installed; callers may use polling instead.
162    /// # Errors
163    /// Returns native inspection, group-signal or reap errors.
164    pub fn wait(&mut self) -> io::Result<ExitStatus> {
165        if let Some(status) = self.status {
166            return Ok(status);
167        }
168        if self.group {
169            self.observe_exit(false)?;
170            self.signal_group()?;
171        }
172        self.reap()
173    }
174
175    /// Kill the owned group (or internal direct child), then reap the leader.
176    ///
177    /// Repeated calls after reaping return the cached status and never signal a
178    /// reused PID. A prior group failure still matters even if reaping succeeded;
179    /// later calls cannot recover group ownership and do not erase that evidence.
180    /// # Errors
181    /// Retains each failed cleanup step separately. Group failure triggers a
182    /// direct-child kill fallback. Drop cannot report errors; call this explicitly
183    /// when cleanup evidence matters.
184    pub fn terminate(&mut self) -> Result<ExitStatus, CleanupError> {
185        if let Some(status) = self.status {
186            return Ok(status);
187        }
188        let group_error = if self.group {
189            self.signal_group().err()
190        } else {
191            None
192        };
193        let kill_error = if self.owned && (!self.group || group_error.is_some()) {
194            retry_interrupted(|| self.child.kill()).err()
195        } else {
196            None
197        };
198        let waited = self.reap();
199        match waited {
200            Ok(status) if group_error.is_none() && kill_error.is_none() => Ok(status),
201            other => Err(CleanupError {
202                status: self.status,
203                group_error,
204                kill_error,
205                wait_error: other.err(),
206            }),
207        }
208    }
209
210    fn pid(&self) -> io::Result<Pid> {
211        if !self.owned {
212            return Err(Errno::CHILD.into());
213        }
214        Pid::from_raw(i32::try_from(self.id()).map_err(io::Error::other)?)
215            .ok_or_else(|| io::Error::other("child PID is zero"))
216    }
217
218    fn observe_exit(&mut self, nonblocking: bool) -> io::Result<bool> {
219        let pid = self.pid()?;
220        // NOWAIT reserves the leader PID until the final group signal, avoiding
221        // signalling an unrelated group if the leader exited before cleanup.
222        let mut options = WaitIdOptions::EXITED | WaitIdOptions::NOWAIT;
223        if nonblocking {
224            options |= WaitIdOptions::NOHANG;
225        }
226        let result = retry_interrupted(|| waitid(WaitId::Pid(pid), options).map_err(Into::into));
227        self.check_wait_ownership(&result);
228        result.map(|status| status.is_some())
229    }
230
231    #[cfg_attr(
232        not(target_os = "macos"),
233        allow(
234            clippy::needless_pass_by_ref_mut,
235            reason = "Darwin inspects and may invalidate child ownership"
236        )
237    )]
238    fn signal_group(&mut self) -> io::Result<()> {
239        let pid = self.pid()?;
240        match retry_interrupted(|| kill_process_group(pid, Signal::KILL).map_err(Into::into)) {
241            Ok(()) => Ok(()),
242            Err(error) if error.raw_os_error() == Some(Errno::SRCH.raw_os_error()) => Ok(()),
243            #[cfg(target_os = "macos")]
244            Err(error)
245                if error.raw_os_error() == Some(Errno::PERM.raw_os_error())
246                    && self.observe_exit(true)?
247                    && macos::sole_group_member(pid) =>
248            {
249                Ok(())
250            }
251            Err(error) => Err(error),
252        }
253    }
254
255    fn reap(&mut self) -> io::Result<ExitStatus> {
256        if !self.owned {
257            return Err(Errno::CHILD.into());
258        }
259        let result = retry_interrupted(|| self.child.wait());
260        if let Ok(status) = result {
261            self.status = Some(status);
262            self.owned = false;
263        }
264        self.check_wait_ownership(&result);
265        result
266    }
267
268    fn check_wait_ownership<T>(&mut self, result: &io::Result<T>) {
269        if result
270            .as_ref()
271            .is_err_and(|error| error.raw_os_error() == Some(Errno::CHILD.raw_os_error()))
272        {
273            // An external reaper violates exclusive ownership; never signal a
274            // potentially reused PID after observing that ownership was lost.
275            self.owned = false;
276        }
277    }
278}
279
280impl Drop for OwnedChild {
281    fn drop(&mut self) {
282        if self.owned {
283            let _ = self.terminate();
284        }
285    }
286}
287
288fn retry_interrupted<T>(mut operation: impl FnMut() -> io::Result<T>) -> io::Result<T> {
289    loop {
290        match operation() {
291            Err(error) if error.kind() == io::ErrorKind::Interrupted => {}
292            result => return result,
293        }
294    }
295}