ic_host_process/child/mod.rs
1//! Explicit child ownership and cleanup, without application lifecycle policy.
2//!
3//! Callers own executable admission, IO, readiness, cancellation and deadlines.
4//! Group cleanup signals members of a newly created group; it cannot contain
5//! processes that escape that group or prove completion of external effects.
6
7#[cfg(target_os = "macos")]
8mod macos;
9#[cfg(test)]
10mod tests;
11
12use rustix::{
13 io::Errno,
14 process::{Pid, Signal, WaitId, WaitIdOptions, WaitIdStatus, kill_process_group, waitid},
15};
16use std::{
17 fmt, io,
18 os::unix::process::{CommandExt, ExitStatusExt},
19 process::{Child, ChildStderr, ChildStdin, ChildStdout, Command, ExitStatus},
20};
21
22/// One exclusively owned child, normally spawned as a new process-group leader.
23///
24/// [`Self::spawn`] preserves the command's IO, environment and other settings,
25/// replacing its process-group selection with a new group. It performs no
26/// executable admission. The child must not change groups, and callers must
27/// not independently reap it (including through a global SIGCHLD handler).
28///
29/// Ordinary waiting signals remaining group members before reaping the leader.
30/// For a deliberate background handoff, [`Self::poll_exit`] observes without
31/// releasing cleanup ownership, then [`Self::handoff`] reaps a successful leader
32/// without signalling its group. The caller then owns the background lifetime.
33/// Drop makes a best-effort kill/reap attempt, including during unwinding. Use
34/// [`Self::terminate`] to observe cleanup failures. Cleanup is synchronous and
35/// has no wall-clock bound; successful signalling is not proof that descendants
36/// have exited or completed external effects. Only the direct child is reaped.
37/// Group signalling can succeed for only some members when credentials differ.
38pub struct OwnedChild {
39 child: Child,
40 group: bool,
41 status: Option<ExitStatus>,
42 owned: bool,
43}
44
45/// Failures observed during one explicit termination attempt.
46///
47/// Keep this separately from the caller's original cancellation/operation error.
48/// If group signalling fails, direct-child kill and reaping are still attempted.
49#[derive(Debug)]
50pub struct CleanupError {
51 /// Status retained if the direct child was reaped despite another failure.
52 pub status: Option<ExitStatus>,
53 /// Failure signalling the owned process group.
54 pub group_error: Option<io::Error>,
55 /// Failure killing the direct child (including fallback after group failure).
56 pub kill_error: Option<io::Error>,
57 /// Failure reaping the direct child.
58 pub wait_error: Option<io::Error>,
59}
60
61impl fmt::Display for CleanupError {
62 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
63 f.write_str("child cleanup failed")?;
64 for (operation, error) in [
65 ("group signal", &self.group_error),
66 ("child kill", &self.kill_error),
67 ("child wait", &self.wait_error),
68 ] {
69 if let Some(error) = error {
70 write!(f, "; {operation}: {error}")?;
71 }
72 }
73 Ok(())
74 }
75}
76
77impl std::error::Error for CleanupError {
78 fn source(&self) -> Option<&(dyn std::error::Error + 'static)> {
79 self.group_error
80 .as_ref()
81 .or(self.kill_error.as_ref())
82 .or(self.wait_error.as_ref())
83 .map(|error| error as &dyn std::error::Error)
84 }
85}
86
87impl OwnedChild {
88 /// Spawn once in a new owned process group, preserving caller-configured IO.
89 ///
90 /// # Errors
91 /// Returns the native spawn/setup failure. No retries are performed.
92 pub fn spawn(command: &mut Command) -> io::Result<Self> {
93 command.process_group(0);
94 Self::spawn_inner(command, true)
95 }
96
97 pub(crate) fn spawn_direct(command: &mut Command) -> io::Result<Self> {
98 Self::spawn_inner(command, false)
99 }
100
101 pub(crate) const fn is_owned(&self) -> bool {
102 self.owned
103 }
104
105 fn spawn_inner(command: &mut Command, group: bool) -> io::Result<Self> {
106 command.spawn().map(|child| Self {
107 child,
108 group,
109 status: None,
110 owned: true,
111 })
112 }
113
114 /// Direct-child PID, for observation only; it can be reused after reaping.
115 #[must_use]
116 pub fn id(&self) -> u32 {
117 self.child.id()
118 }
119
120 /// Take caller-configured piped stdin. Close it before waiting for EOF-driven children.
121 pub const fn take_stdin(&mut self) -> Option<ChildStdin> {
122 self.child.stdin.take()
123 }
124
125 /// Take caller-configured piped stdout; the caller owns draining and bounds.
126 pub const fn take_stdout(&mut self) -> Option<ChildStdout> {
127 self.child.stdout.take()
128 }
129
130 /// Take caller-configured piped stderr; the caller owns draining and bounds.
131 pub const fn take_stderr(&mut self) -> Option<ChildStderr> {
132 self.child.stderr.take()
133 }
134
135 /// Observe leader exit without signalling or reaping it.
136 ///
137 /// An exited leader stays reserved with `WNOWAIT`, so cancellation, failed IO
138 /// admission and unwinding still clean its group. Repeated observations do
139 /// not release ownership. After a completed wait/termination/handoff, returns
140 /// the cached status. This status alone does not establish a handoff.
141 ///
142 /// Call [`Self::handoff`] only after admitting a successful background start.
143 /// Otherwise use [`Self::wait`] or [`Self::terminate`] to clean and reap.
144 /// # Errors
145 /// Returns native inspection errors; external reaping invalidates ownership.
146 pub fn poll_exit(&mut self) -> io::Result<Option<ExitStatus>> {
147 if let Some(status) = self.status {
148 return Ok(Some(status));
149 }
150 self.observe_exit(true)?
151 .map(|status| {
152 // Unix wait status encoding used by Linux and Darwin. waitid's
153 // siginfo status is an exit code/signal, not an encoded wait status.
154 let raw = if let Some(code) = status.exit_status() {
155 code << 8
156 } else if let Some(signal) = status.terminating_signal() {
157 signal | if status.dumped() { 0x80 } else { 0 }
158 } else {
159 return Err(io::Error::other("waitid returned a non-exit observation"));
160 };
161 Ok(ExitStatus::from_raw(raw))
162 })
163 .transpose()
164 }
165
166 /// Reap an already successful leader without signalling its remaining group.
167 ///
168 /// This is the explicit transfer point for a background lifetime. The caller
169 /// must first admit its IO/result and arrange application-owned readiness,
170 /// cancellation and stop/recovery. Zero exit does not prove those obligations.
171 /// No PID/group handle is transferred: it could be reused after reaping.
172 /// Subsequent wait/termination/Drop never signal the handed-off group.
173 ///
174 /// Use [`Self::poll_exit`] while draining IO and checking cancellation. A
175 /// running or unsuccessful leader is refused without releasing ownership;
176 /// use ordinary wait/termination for failed startup, keeping its original
177 /// failure separate from any cleanup error. Drop still attempts cleanup.
178 /// # Errors
179 /// Returns `InvalidInput` for a running, unsuccessful or already-reaped
180 /// leader, or native inspection/reap errors. Reap failures retain cleanup
181 /// ownership unless it was lost externally.
182 pub fn handoff(&mut self) -> io::Result<ExitStatus> {
183 if !self.owned || !self.poll_exit()?.is_some_and(|status| status.success()) {
184 return Err(io::Error::new(
185 io::ErrorKind::InvalidInput,
186 "handoff requires an owned, successfully exited leader",
187 ));
188 }
189 self.reap()
190 }
191
192 /// Inspect exit without blocking on a running child; clean its group before reaping.
193 ///
194 /// Repeated successful calls return the cached status without signalling again.
195 /// # Errors
196 /// Returns native inspection, group-signal or reap errors. The leader remains
197 /// reserved on a group-signal failure, so explicit cleanup can still be attempted.
198 pub fn try_wait(&mut self) -> io::Result<Option<ExitStatus>> {
199 if let Some(status) = self.status {
200 return Ok(Some(status));
201 }
202 if !self.owned {
203 return Err(Errno::CHILD.into());
204 }
205 if self.group {
206 if self.observe_exit(true)?.is_none() {
207 return Ok(None);
208 }
209 self.signal_group()?;
210 self.reap().map(Some)
211 } else {
212 let result = retry_interrupted(|| self.child.try_wait());
213 if let Ok(Some(status)) = result {
214 self.status = Some(status);
215 self.owned = false;
216 }
217 self.check_wait_ownership(&result);
218 result
219 }
220 }
221
222 /// Wait for natural leader exit, then clean its group and reap the leader.
223 ///
224 /// Close/drain caller-owned pipes as needed before waiting. No deadline or
225 /// cancellation policy is installed; callers may use polling instead.
226 /// # Errors
227 /// Returns native inspection, group-signal or reap errors.
228 pub fn wait(&mut self) -> io::Result<ExitStatus> {
229 if let Some(status) = self.status {
230 return Ok(status);
231 }
232 if self.group {
233 self.observe_exit(false)?;
234 self.signal_group()?;
235 }
236 self.reap()
237 }
238
239 /// Kill the owned group (or internal direct child), then reap the leader.
240 ///
241 /// Repeated calls after reaping return the cached status and never signal a
242 /// reused PID. A prior group failure still matters even if reaping succeeded;
243 /// later calls cannot recover group ownership and do not erase that evidence.
244 /// # Errors
245 /// Retains each failed cleanup step separately. Group failure triggers a
246 /// direct-child kill fallback. Drop cannot report errors; call this explicitly
247 /// when cleanup evidence matters.
248 pub fn terminate(&mut self) -> Result<ExitStatus, CleanupError> {
249 if let Some(status) = self.status {
250 return Ok(status);
251 }
252 let group_error = if self.group {
253 self.signal_group().err()
254 } else {
255 None
256 };
257 let kill_error = if self.owned && (!self.group || group_error.is_some()) {
258 retry_interrupted(|| self.child.kill()).err()
259 } else {
260 None
261 };
262 let waited = self.reap();
263 match waited {
264 Ok(status) if group_error.is_none() && kill_error.is_none() => Ok(status),
265 other => Err(CleanupError {
266 status: self.status,
267 group_error,
268 kill_error,
269 wait_error: other.err(),
270 }),
271 }
272 }
273
274 fn pid(&self) -> io::Result<Pid> {
275 if !self.owned {
276 return Err(Errno::CHILD.into());
277 }
278 Pid::from_raw(i32::try_from(self.id()).map_err(io::Error::other)?)
279 .ok_or_else(|| io::Error::other("child PID is zero"))
280 }
281
282 fn observe_exit(&mut self, nonblocking: bool) -> io::Result<Option<WaitIdStatus>> {
283 let pid = self.pid()?;
284 // NOWAIT reserves the leader PID until cleanup or explicit handoff,
285 // avoiding signals to an unrelated group after an early leader exit.
286 let mut options = WaitIdOptions::EXITED | WaitIdOptions::NOWAIT;
287 if nonblocking {
288 options |= WaitIdOptions::NOHANG;
289 }
290 let result = retry_interrupted(|| waitid(WaitId::Pid(pid), options).map_err(Into::into));
291 self.check_wait_ownership(&result);
292 result
293 }
294
295 #[cfg_attr(
296 not(target_os = "macos"),
297 allow(
298 clippy::needless_pass_by_ref_mut,
299 reason = "Darwin inspects and may invalidate child ownership"
300 )
301 )]
302 fn signal_group(&mut self) -> io::Result<()> {
303 let pid = self.pid()?;
304 match retry_interrupted(|| kill_process_group(pid, Signal::KILL).map_err(Into::into)) {
305 Ok(()) => Ok(()),
306 Err(error) if error.raw_os_error() == Some(Errno::SRCH.raw_os_error()) => Ok(()),
307 #[cfg(target_os = "macos")]
308 Err(error)
309 if error.raw_os_error() == Some(Errno::PERM.raw_os_error())
310 && self.observe_exit(true)?.is_some()
311 && macos::sole_group_member(pid) =>
312 {
313 Ok(())
314 }
315 Err(error) => Err(error),
316 }
317 }
318
319 fn reap(&mut self) -> io::Result<ExitStatus> {
320 if !self.owned {
321 return Err(Errno::CHILD.into());
322 }
323 let result = retry_interrupted(|| self.child.wait());
324 if let Ok(status) = result {
325 self.status = Some(status);
326 self.owned = false;
327 }
328 self.check_wait_ownership(&result);
329 result
330 }
331
332 fn check_wait_ownership<T>(&mut self, result: &io::Result<T>) {
333 if result
334 .as_ref()
335 .is_err_and(|error| error.raw_os_error() == Some(Errno::CHILD.raw_os_error()))
336 {
337 // An external reaper violates exclusive ownership; never signal a
338 // potentially reused PID after observing that ownership was lost.
339 self.owned = false;
340 }
341 }
342}
343
344impl Drop for OwnedChild {
345 fn drop(&mut self) {
346 if self.owned {
347 let _ = self.terminate();
348 }
349 }
350}
351
352fn retry_interrupted<T>(mut operation: impl FnMut() -> io::Result<T>) -> io::Result<T> {
353 loop {
354 match operation() {
355 Err(error) if error.kind() == io::ErrorKind::Interrupted => {}
356 result => return result,
357 }
358 }
359}