Skip to main content

ic_host_process/child/
mod.rs

1//! Explicit child ownership and cleanup, without application lifecycle policy.
2//!
3//! Callers own executable admission, IO, readiness, cancellation and deadlines.
4//! Group cleanup signals members of a newly created group; it cannot contain
5//! processes that escape that group or prove completion of external effects.
6
7#[cfg(target_os = "macos")]
8mod macos;
9#[cfg(test)]
10mod tests;
11
12use rustix::{
13    io::Errno,
14    process::{Pid, Signal, WaitId, WaitIdOptions, WaitIdStatus, kill_process_group, waitid},
15};
16use std::{
17    fmt, io,
18    os::unix::process::{CommandExt, ExitStatusExt},
19    process::{Child, ChildStderr, ChildStdin, ChildStdout, Command, ExitStatus},
20};
21
22/// One exclusively owned child, normally spawned as a new process-group leader.
23///
24/// [`Self::spawn`] preserves the command's IO, environment and other settings,
25/// replacing its process-group selection with a new group. It performs no
26/// executable admission. The child must not change groups, and callers must
27/// not independently reap it (including through a global SIGCHLD handler).
28///
29/// Ordinary waiting signals remaining group members before reaping the leader.
30/// For a deliberate background handoff, [`Self::poll_exit`] observes without
31/// releasing cleanup ownership, then [`Self::handoff`] reaps a successful leader
32/// without signalling its group. The caller then owns the background lifetime.
33/// Drop makes a best-effort kill/reap attempt, including during unwinding. Use
34/// [`Self::terminate`] to observe cleanup failures. Cleanup is synchronous and
35/// has no wall-clock bound; successful signalling is not proof that descendants
36/// have exited or completed external effects. Only the direct child is reaped.
37/// Group signalling can succeed for only some members when credentials differ.
38pub struct OwnedChild {
39    child: Child,
40    group: bool,
41    status: Option<ExitStatus>,
42    owned: bool,
43}
44
45/// Failures observed during one explicit termination attempt.
46///
47/// Keep this separately from the caller's original cancellation/operation error.
48/// If group signalling fails, direct-child kill and reaping are still attempted.
49#[derive(Debug)]
50pub struct CleanupError {
51    /// Status retained if the direct child was reaped despite another failure.
52    pub status: Option<ExitStatus>,
53    /// Failure signalling the owned process group.
54    pub group_error: Option<io::Error>,
55    /// Failure killing the direct child (including fallback after group failure).
56    pub kill_error: Option<io::Error>,
57    /// Failure reaping the direct child.
58    pub wait_error: Option<io::Error>,
59}
60
61impl fmt::Display for CleanupError {
62    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
63        f.write_str("child cleanup failed")?;
64        for (operation, error) in [
65            ("group signal", &self.group_error),
66            ("child kill", &self.kill_error),
67            ("child wait", &self.wait_error),
68        ] {
69            if let Some(error) = error {
70                write!(f, "; {operation}: {error}")?;
71            }
72        }
73        Ok(())
74    }
75}
76
77impl std::error::Error for CleanupError {
78    fn source(&self) -> Option<&(dyn std::error::Error + 'static)> {
79        self.group_error
80            .as_ref()
81            .or(self.kill_error.as_ref())
82            .or(self.wait_error.as_ref())
83            .map(|error| error as &dyn std::error::Error)
84    }
85}
86
87impl OwnedChild {
88    /// Spawn once in a new owned process group, preserving caller-configured IO.
89    ///
90    /// # Errors
91    /// Returns the native spawn/setup failure. No retries are performed.
92    pub fn spawn(command: &mut Command) -> io::Result<Self> {
93        command.process_group(0);
94        Self::spawn_inner(command, true)
95    }
96
97    pub(crate) fn spawn_direct(command: &mut Command) -> io::Result<Self> {
98        Self::spawn_inner(command, false)
99    }
100
101    fn spawn_inner(command: &mut Command, group: bool) -> io::Result<Self> {
102        command.spawn().map(|child| Self {
103            child,
104            group,
105            status: None,
106            owned: true,
107        })
108    }
109
110    /// Direct-child PID, for observation only; it can be reused after reaping.
111    #[must_use]
112    pub fn id(&self) -> u32 {
113        self.child.id()
114    }
115
116    /// Take caller-configured piped stdin. Close it before waiting for EOF-driven children.
117    pub const fn take_stdin(&mut self) -> Option<ChildStdin> {
118        self.child.stdin.take()
119    }
120
121    /// Take caller-configured piped stdout; the caller owns draining and bounds.
122    pub const fn take_stdout(&mut self) -> Option<ChildStdout> {
123        self.child.stdout.take()
124    }
125
126    /// Take caller-configured piped stderr; the caller owns draining and bounds.
127    pub const fn take_stderr(&mut self) -> Option<ChildStderr> {
128        self.child.stderr.take()
129    }
130
131    /// Observe leader exit without signalling or reaping it.
132    ///
133    /// An exited leader stays reserved with `WNOWAIT`, so cancellation, failed IO
134    /// admission and unwinding still clean its group. Repeated observations do
135    /// not release ownership. After a completed wait/termination/handoff, returns
136    /// the cached status. This status alone does not establish a handoff.
137    ///
138    /// Call [`Self::handoff`] only after admitting a successful background start.
139    /// Otherwise use [`Self::wait`] or [`Self::terminate`] to clean and reap.
140    /// # Errors
141    /// Returns native inspection errors; external reaping invalidates ownership.
142    pub fn poll_exit(&mut self) -> io::Result<Option<ExitStatus>> {
143        if let Some(status) = self.status {
144            return Ok(Some(status));
145        }
146        self.observe_exit(true)?
147            .map(|status| {
148                // Unix wait status encoding used by Linux and Darwin. waitid's
149                // siginfo status is an exit code/signal, not an encoded wait status.
150                let raw = if let Some(code) = status.exit_status() {
151                    code << 8
152                } else if let Some(signal) = status.terminating_signal() {
153                    signal | if status.dumped() { 0x80 } else { 0 }
154                } else {
155                    return Err(io::Error::other("waitid returned a non-exit observation"));
156                };
157                Ok(ExitStatus::from_raw(raw))
158            })
159            .transpose()
160    }
161
162    /// Reap an already successful leader without signalling its remaining group.
163    ///
164    /// This is the explicit transfer point for a background lifetime. The caller
165    /// must first admit its IO/result and arrange application-owned readiness,
166    /// cancellation and stop/recovery. Zero exit does not prove those obligations.
167    /// No PID/group handle is transferred: it could be reused after reaping.
168    /// Subsequent wait/termination/Drop never signal the handed-off group.
169    ///
170    /// Use [`Self::poll_exit`] while draining IO and checking cancellation. A
171    /// running or unsuccessful leader is refused without releasing ownership;
172    /// use ordinary wait/termination for failed startup, keeping its original
173    /// failure separate from any cleanup error. Drop still attempts cleanup.
174    /// # Errors
175    /// Returns `InvalidInput` for a running, unsuccessful or already-reaped
176    /// leader, or native inspection/reap errors. Reap failures retain cleanup
177    /// ownership unless it was lost externally.
178    pub fn handoff(&mut self) -> io::Result<ExitStatus> {
179        if !self.owned || !self.poll_exit()?.is_some_and(|status| status.success()) {
180            return Err(io::Error::new(
181                io::ErrorKind::InvalidInput,
182                "handoff requires an owned, successfully exited leader",
183            ));
184        }
185        self.reap()
186    }
187
188    /// Inspect exit without blocking on a running child; clean its group before reaping.
189    ///
190    /// Repeated successful calls return the cached status without signalling again.
191    /// # Errors
192    /// Returns native inspection, group-signal or reap errors. The leader remains
193    /// reserved on a group-signal failure, so explicit cleanup can still be attempted.
194    pub fn try_wait(&mut self) -> io::Result<Option<ExitStatus>> {
195        if let Some(status) = self.status {
196            return Ok(Some(status));
197        }
198        if !self.owned {
199            return Err(Errno::CHILD.into());
200        }
201        if self.group {
202            if self.observe_exit(true)?.is_none() {
203                return Ok(None);
204            }
205            self.signal_group()?;
206            self.reap().map(Some)
207        } else {
208            let result = retry_interrupted(|| self.child.try_wait());
209            if let Ok(Some(status)) = result {
210                self.status = Some(status);
211                self.owned = false;
212            }
213            self.check_wait_ownership(&result);
214            result
215        }
216    }
217
218    /// Wait for natural leader exit, then clean its group and reap the leader.
219    ///
220    /// Close/drain caller-owned pipes as needed before waiting. No deadline or
221    /// cancellation policy is installed; callers may use polling instead.
222    /// # Errors
223    /// Returns native inspection, group-signal or reap errors.
224    pub fn wait(&mut self) -> io::Result<ExitStatus> {
225        if let Some(status) = self.status {
226            return Ok(status);
227        }
228        if self.group {
229            self.observe_exit(false)?;
230            self.signal_group()?;
231        }
232        self.reap()
233    }
234
235    /// Kill the owned group (or internal direct child), then reap the leader.
236    ///
237    /// Repeated calls after reaping return the cached status and never signal a
238    /// reused PID. A prior group failure still matters even if reaping succeeded;
239    /// later calls cannot recover group ownership and do not erase that evidence.
240    /// # Errors
241    /// Retains each failed cleanup step separately. Group failure triggers a
242    /// direct-child kill fallback. Drop cannot report errors; call this explicitly
243    /// when cleanup evidence matters.
244    pub fn terminate(&mut self) -> Result<ExitStatus, CleanupError> {
245        if let Some(status) = self.status {
246            return Ok(status);
247        }
248        let group_error = if self.group {
249            self.signal_group().err()
250        } else {
251            None
252        };
253        let kill_error = if self.owned && (!self.group || group_error.is_some()) {
254            retry_interrupted(|| self.child.kill()).err()
255        } else {
256            None
257        };
258        let waited = self.reap();
259        match waited {
260            Ok(status) if group_error.is_none() && kill_error.is_none() => Ok(status),
261            other => Err(CleanupError {
262                status: self.status,
263                group_error,
264                kill_error,
265                wait_error: other.err(),
266            }),
267        }
268    }
269
270    fn pid(&self) -> io::Result<Pid> {
271        if !self.owned {
272            return Err(Errno::CHILD.into());
273        }
274        Pid::from_raw(i32::try_from(self.id()).map_err(io::Error::other)?)
275            .ok_or_else(|| io::Error::other("child PID is zero"))
276    }
277
278    fn observe_exit(&mut self, nonblocking: bool) -> io::Result<Option<WaitIdStatus>> {
279        let pid = self.pid()?;
280        // NOWAIT reserves the leader PID until cleanup or explicit handoff,
281        // avoiding signals to an unrelated group after an early leader exit.
282        let mut options = WaitIdOptions::EXITED | WaitIdOptions::NOWAIT;
283        if nonblocking {
284            options |= WaitIdOptions::NOHANG;
285        }
286        let result = retry_interrupted(|| waitid(WaitId::Pid(pid), options).map_err(Into::into));
287        self.check_wait_ownership(&result);
288        result
289    }
290
291    #[cfg_attr(
292        not(target_os = "macos"),
293        allow(
294            clippy::needless_pass_by_ref_mut,
295            reason = "Darwin inspects and may invalidate child ownership"
296        )
297    )]
298    fn signal_group(&mut self) -> io::Result<()> {
299        let pid = self.pid()?;
300        match retry_interrupted(|| kill_process_group(pid, Signal::KILL).map_err(Into::into)) {
301            Ok(()) => Ok(()),
302            Err(error) if error.raw_os_error() == Some(Errno::SRCH.raw_os_error()) => Ok(()),
303            #[cfg(target_os = "macos")]
304            Err(error)
305                if error.raw_os_error() == Some(Errno::PERM.raw_os_error())
306                    && self.observe_exit(true)?.is_some()
307                    && macos::sole_group_member(pid) =>
308            {
309                Ok(())
310            }
311            Err(error) => Err(error),
312        }
313    }
314
315    fn reap(&mut self) -> io::Result<ExitStatus> {
316        if !self.owned {
317            return Err(Errno::CHILD.into());
318        }
319        let result = retry_interrupted(|| self.child.wait());
320        if let Ok(status) = result {
321            self.status = Some(status);
322            self.owned = false;
323        }
324        self.check_wait_ownership(&result);
325        result
326    }
327
328    fn check_wait_ownership<T>(&mut self, result: &io::Result<T>) {
329        if result
330            .as_ref()
331            .is_err_and(|error| error.raw_os_error() == Some(Errno::CHILD.raw_os_error()))
332        {
333            // An external reaper violates exclusive ownership; never signal a
334            // potentially reused PID after observing that ownership was lost.
335            self.owned = false;
336        }
337    }
338}
339
340impl Drop for OwnedChild {
341    fn drop(&mut self) {
342        if self.owned {
343            let _ = self.terminate();
344        }
345    }
346}
347
348fn retry_interrupted<T>(mut operation: impl FnMut() -> io::Result<T>) -> io::Result<T> {
349    loop {
350        match operation() {
351            Err(error) if error.kind() == io::ErrorKind::Interrupted => {}
352            result => return result,
353        }
354    }
355}