Skip to main content

reverie_process/
container.rs

1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 * All rights reserved.
4 *
5 * This source code is licensed under the BSD-style license found in the
6 * LICENSE file in the root directory of this source tree.
7 */
8
9use std::borrow::Cow;
10use std::collections::BTreeMap;
11use std::ffi::CString;
12use std::ffi::OsStr;
13use std::ffi::OsString;
14use std::io::Read;
15use std::io::Write;
16#[cfg(test)]
17use std::os::fd::RawFd;
18use std::os::unix::ffi::OsStrExt;
19use std::os::unix::io::AsRawFd;
20use std::path::Path;
21#[cfg(test)]
22use std::sync::atomic::Ordering;
23
24use nix::sched::CpuSet;
25use nix::sched::sched_setaffinity;
26use serde::Serialize;
27use serde::de::DeserializeOwned;
28use syscalls::Errno;
29
30use super::clone::child_stack;
31use super::clone::clone_with_stack;
32use super::env::Env;
33use super::error::AddContext;
34use super::error::Context;
35use super::error::Error;
36use super::exit_status::ExitStatus;
37use super::fd::Fd;
38use super::fd::pipe;
39use super::fd::write_bytes;
40use super::id_map::make_id_map;
41use super::mount::Mount;
42use super::namespace::Namespace;
43use super::net::IfName;
44use super::pid::Pid;
45use super::pty::PtyChild;
46use super::seccomp;
47use super::stdio::Stdio;
48use super::util::reset_signal_handling;
49use super::util::to_cstring;
50
51/// A `Container` is a configuration of how a process shall be spawned. It can,
52/// but doesn't have to, include Linux namespace configuration.
53///
54/// NOTE: Configuring resource limits via cgroups is not yet supported.
55pub struct Container {
56    pub(super) env: Env,
57    current_dir: Option<CString>,
58    chroot: Option<CString>,
59    pub(super) namespace: Namespace,
60    pub(super) stdin: Stdio,
61    pub(super) stdout: Stdio,
62    pub(super) stderr: Stdio,
63    pub(super) uid_map: Vec<(libc::uid_t, libc::uid_t, u32)>,
64    pub(super) gid_map: Vec<(libc::uid_t, libc::uid_t, u32)>,
65    mounts: Vec<Mount>,
66    local_networking_only: bool,
67    hostname: Option<OsString>,
68    domainname: Option<OsString>,
69    pub(super) seccomp: Option<seccomp::Filter>,
70    pub(super) seccomp_notify: bool,
71    pub(super) pty: Option<PtyChild>,
72    /// The core number to which the new process, and descendents, will be
73    /// pinned.
74    affinity: Option<usize>,
75}
76
77impl Default for Container {
78    fn default() -> Self {
79        Self {
80            env: Default::default(),
81            current_dir: None,
82            chroot: None,
83            namespace: Default::default(),
84            stdin: Stdio::inherit(),
85            stdout: Stdio::inherit(),
86            stderr: Stdio::inherit(),
87            uid_map: Vec::new(),
88            gid_map: Vec::new(),
89            mounts: Vec::new(),
90            local_networking_only: false,
91            hostname: None,
92            domainname: None,
93            seccomp: None,
94            seccomp_notify: false,
95            pty: None,
96            affinity: None,
97        }
98    }
99}
100
101impl Container {
102    /// Returns the configured features that cannot be represented by
103    /// `std::process::Command`.
104    pub(super) fn std_conversion_blockers(&self) -> Vec<&'static str> {
105        // Keep this exhaustive: adding Container state must fail to compile
106        // until the standard-command conversion explicitly classifies it.
107        let Self {
108            env: _,
109            current_dir: _,
110            chroot,
111            namespace,
112            stdin: _,
113            stdout: _,
114            stderr: _,
115            uid_map,
116            gid_map,
117            mounts,
118            local_networking_only,
119            hostname,
120            domainname,
121            seccomp,
122            seccomp_notify,
123            pty,
124            affinity,
125        } = self;
126
127        let mut blockers = Vec::new();
128
129        if chroot.is_some() {
130            blockers.push("chroot");
131        }
132        if !namespace.is_empty() {
133            blockers.push("Linux namespaces");
134        }
135        if !uid_map.is_empty() {
136            blockers.push("user ID mappings");
137        }
138        if !gid_map.is_empty() {
139            blockers.push("group ID mappings");
140        }
141        if !mounts.is_empty() {
142            blockers.push("mounts");
143        }
144        if *local_networking_only {
145            blockers.push("local-only networking");
146        }
147        if hostname.is_some() {
148            blockers.push("hostname");
149        }
150        if domainname.is_some() {
151            blockers.push("domain name");
152        }
153        if seccomp.is_some() {
154            blockers.push("seccomp filter");
155        }
156        if *seccomp_notify {
157            blockers.push("seccomp notification");
158        }
159        if pty.is_some() {
160            blockers.push("pseudoterminal");
161        }
162        if affinity.is_some() {
163            blockers.push("CPU affinity");
164        }
165
166        blockers
167    }
168
169    /// Creates a new `Container` that inherits everything from the parent
170    /// process.
171    pub fn new() -> Self {
172        Self::default()
173    }
174
175    /// Inserts or updates an environment variable mapping.
176    ///
177    /// Note that environment variable names are case-insensitive (but
178    /// case-preserving) on Windows, and case-sensitive on all other platforms.
179    ///
180    /// # Examples
181    ///
182    /// Basic usage:
183    ///
184    /// ```no_run
185    /// use reverie_process::Container;
186    ///
187    /// let container = Container::new().env("PATH", "/bin");
188    /// ```
189    pub fn env<K, V>(&mut self, key: K, val: V) -> &mut Self
190    where
191        K: AsRef<OsStr>,
192        V: AsRef<OsStr>,
193    {
194        self.env.set(key.as_ref(), val.as_ref());
195        self
196    }
197
198    /// Adds or updates multiple environment variable mappings.
199    ///
200    /// # Examples
201    ///
202    /// Basic usage:
203    ///
204    /// ```no_run
205    /// use std::collections::HashMap;
206    /// use std::env;
207    ///
208    /// use reverie_process::Container;
209    /// use reverie_process::Stdio;
210    ///
211    /// let filtered_env: HashMap<String, String> = env::vars()
212    ///     .filter(|&(ref k, _)| k == "TERM" || k == "TZ" || k == "LANG" || k == "PATH")
213    ///     .collect();
214    ///
215    /// let container = Container::new()
216    ///     .stdin(Stdio::null())
217    ///     .stdout(Stdio::inherit())
218    ///     .env_clear()
219    ///     .envs(&filtered_env);
220    /// ```
221    pub fn envs<I, K, V>(&mut self, vars: I) -> &mut Self
222    where
223        I: IntoIterator<Item = (K, V)>,
224        K: AsRef<OsStr>,
225        V: AsRef<OsStr>,
226    {
227        for (k, v) in vars.into_iter() {
228            self.env(k, v);
229        }
230        self
231    }
232
233    /// Removes an environment variable mapping.
234    ///
235    /// # Examples
236    ///
237    /// Basic usage:
238    ///
239    /// ```no_run
240    /// use reverie_process::Container;
241    ///
242    /// let container = Container::new().env_remove("PATH");
243    /// ```
244    pub fn env_remove<K: AsRef<OsStr>>(&mut self, key: K) -> &mut Self {
245        self.env.remove(key.as_ref());
246        self
247    }
248
249    /// Clears the entire environment map for the child process.
250    ///
251    /// # Examples
252    ///
253    /// Basic usage:
254    ///
255    /// ```no_run
256    /// use reverie_process::Container;
257    ///
258    /// let container = Container::new().env_clear();
259    /// ```
260    pub fn env_clear(&mut self) -> &mut Self {
261        self.env.clear();
262        self
263    }
264
265    /// Sets the working directory for the child process.
266    ///
267    /// # Interaction with `chroot`
268    ///
269    /// The working directory is set *after* the chroot is performed (if a chroot
270    /// directory is specified). Thus, the path given is relative to the chroot
271    /// directory. Otherwise, if no chroot directory is specified, the working
272    /// directory is relative to the current working directory of the parent
273    /// process at the time the child process is spawned.
274    ///
275    /// # Platform-specific behavior
276    ///
277    /// If the program path is relative (e.g., `"./script.sh"`), it's ambiguous
278    /// whether it should be interpreted relative to the parent's working
279    /// directory or relative to `current_dir`. The behavior in this case is
280    /// platform specific and unstable, and it's recommended to use
281    /// [`canonicalize`] to get an absolute program path instead.
282    ///
283    /// [`canonicalize`]: std::fs::canonicalize()
284    ///
285    /// # Examples
286    ///
287    /// Basic usage:
288    ///
289    /// ```no_run
290    /// use reverie_process::Container;
291    ///
292    /// let container = Container::new().current_dir("/bin");
293    /// ```
294    pub fn current_dir<P: AsRef<Path>>(&mut self, dir: P) -> &mut Self {
295        self.current_dir = Some(to_cstring(dir.as_ref()));
296        self
297    }
298
299    /// Sets configuration for the child process's standard input (stdin) handle.
300    ///
301    /// Defaults to [`Stdio::inherit`] when used with `spawn` or `status`, and
302    /// defaults to [`Stdio::piped`] when used with `output`.
303    ///
304    /// # Examples
305    ///
306    /// Basic usage:
307    ///
308    /// ```no_run
309    /// use reverie_process::Container;
310    /// use reverie_process::Stdio;
311    ///
312    /// let container = Container::new().stdin(Stdio::null());
313    /// ```
314    pub fn stdin<T: Into<Stdio>>(&mut self, cfg: T) -> &mut Self {
315        self.stdin = cfg.into();
316        self
317    }
318
319    /// Sets configuration for the child process's standard output (stdout)
320    /// handle.
321    ///
322    /// Defaults to [`Stdio::inherit`] when used with `spawn` or `status`, and
323    /// defaults to [`Stdio::piped`] when used with `output`.
324    ///
325    /// # Examples
326    ///
327    /// Basic usage:
328    ///
329    /// ```no_run
330    /// use reverie_process::Container;
331    /// use reverie_process::Stdio;
332    ///
333    /// let container = Container::new().stdout(Stdio::null());
334    /// ```
335    pub fn stdout<T: Into<Stdio>>(&mut self, cfg: T) -> &mut Self {
336        self.stdout = cfg.into();
337        self
338    }
339
340    /// Sets configuration for the child process's standard error (stderr)
341    /// handle.
342    ///
343    /// Defaults to [`Stdio::inherit`] when used with `spawn` or `status`, and
344    /// defaults to [`Stdio::piped`] when used with `output`.
345    ///
346    /// # Examples
347    ///
348    /// Basic usage:
349    ///
350    /// ```no_run
351    /// use reverie_process::Container;
352    /// use reverie_process::Stdio;
353    ///
354    /// let container = Container::new().stderr(Stdio::null());
355    /// ```
356    pub fn stderr<T: Into<Stdio>>(&mut self, cfg: T) -> &mut Self {
357        self.stderr = cfg.into();
358        self
359    }
360
361    /// Changes the root directory of the calling process to the specified path.
362    /// This directory will be inherited by all child processes of the calling
363    /// process.
364    ///
365    /// Note that changing the root directory may cause the program to not be
366    /// found. As such, the program path should be relative to this directory.
367    pub fn chroot<P: AsRef<Path>>(&mut self, chroot: P) -> &mut Self {
368        self.chroot = Some(to_cstring(chroot.as_ref()));
369        self
370    }
371
372    /// Unshares parts of the process execution context that are normally shared
373    /// with the parent process. This is useful for executing the child process
374    /// in a new namespace.
375    pub fn unshare(&mut self, namespace: Namespace) -> &mut Self {
376        self.namespace |= namespace;
377        self
378    }
379
380    /// Returns the working directory for the child process.
381    ///
382    /// This returns None if the working directory will not be changed.
383    pub fn get_current_dir(&self) -> Option<&Path> {
384        if let Some(dir) = &self.current_dir {
385            Some(Path::new(OsStr::from_bytes(dir.to_bytes())))
386        } else {
387            None
388        }
389    }
390
391    /// Returns an iterator of the environment variables that will be set when
392    /// the process is spawned. Note that this does not include any environment
393    /// variables inherited from the parent process.
394    pub fn get_envs(&self) -> impl Iterator<Item = (&OsStr, Option<&OsStr>)> {
395        self.env.iter()
396    }
397
398    /// Returns a mapping of all environment variables that the new child process
399    /// will inherit.
400    pub fn get_captured_envs(&self) -> BTreeMap<OsString, OsString> {
401        self.env.capture()
402    }
403
404    /// Gets an environment variable. If the child process is to inherit this
405    /// environment variable from the current process, then this returns the
406    /// current process's environment variable unless it is to be overridden.
407    pub fn get_env<K: AsRef<OsStr>>(&self, env: K) -> Option<Cow<'_, OsStr>> {
408        self.env.get_captured(env)
409    }
410
411    /// Maps one user ID to another.
412    ///
413    /// Implies `Namespace::USER`.
414    ///
415    /// # Example
416    ///
417    /// This is can be used to gain `CAP_SYS_ADMIN` privileges in the user
418    /// namespace by mapping the root user inside the container to the current
419    /// user outside of the container.
420    ///
421    /// ```no_run
422    /// use reverie_process::Container;
423    ///
424    /// let container = Container::new().map_uid(1, unsafe { libc::getuid() });
425    /// ```
426    ///
427    /// # Implementation
428    ///
429    /// This modifies `/proc/{pid}/uid_map` where `{pid}` is the PID of the child
430    /// process. See [`user_namespaces(7)`] for more details.
431    ///
432    /// [`user_namespaces(7)`]: https://man7.org/linux/man-pages/man7/user_namespaces.7.html
433    pub fn map_uid(&mut self, inside_uid: libc::uid_t, outside_uid: libc::uid_t) -> &mut Self {
434        self.map_uid_range(inside_uid, outside_uid, 1)
435    }
436
437    /// Maps potentially many user IDs inside the new user namespace to user IDs
438    /// outside of the user namespace.
439    ///
440    /// Implies `Namespace::USER`.
441    ///
442    /// # Implementation
443    ///
444    /// This modifies `/proc/{pid}/uid_map` where `{pid}` is the PID of the child
445    /// process. See [`user_namespaces(7)`] for more details.
446    ///
447    /// [`user_namespaces(7)`]: https://man7.org/linux/man-pages/man7/user_namespaces.7.html
448    pub fn map_uid_range(
449        &mut self,
450        starting_inside_uid: libc::uid_t,
451        starting_outside_uid: libc::uid_t,
452        count: u32,
453    ) -> &mut Self {
454        self.uid_map
455            .push((starting_inside_uid, starting_outside_uid, count));
456        self.namespace |= Namespace::USER;
457        self
458    }
459
460    /// Convience function for mapping root (inside the container) to the current
461    /// user ID (outside the container). This is useful for gaining new
462    /// capabilities inside the container, such as being able to mount file
463    /// systems.
464    ///
465    /// Implies `Namespace::USER`.
466    ///
467    /// This is the same as:
468    /// ```no_run
469    /// use reverie_process::Container;
470    ///
471    /// let container = Container::new()
472    ///     .map_uid(0, unsafe { libc::geteuid() })
473    ///     .map_gid(0, unsafe { libc::getegid() });
474    /// ```
475    pub fn map_root(&mut self) -> &mut Self {
476        self.map_uid(0, unsafe { libc::geteuid() });
477        self.map_gid(0, unsafe { libc::getegid() })
478    }
479
480    /// Maps one group ID to another.
481    ///
482    /// Implies `Namespace::USER`.
483    ///
484    /// # Implementation
485    ///
486    /// This modifies `/proc/{pid}/gid_map` where `{pid}` is the PID of the child
487    /// process. See [`user_namespaces(7)`] for more details.
488    ///
489    /// [`user_namespaces(7)`]: https://man7.org/linux/man-pages/man7/user_namespaces.7.html
490    pub fn map_gid(&mut self, inside_gid: libc::gid_t, outside_gid: libc::gid_t) -> &mut Self {
491        self.map_gid_range(inside_gid, outside_gid, 1)
492    }
493
494    /// Maps potentially many group IDs inside the new user namespace to group
495    /// IDs outside of the user namespace.
496    ///
497    /// Implies `Namespace::USER`.
498    ///
499    /// # Implementation
500    ///
501    /// This modifies `/proc/{pid}/gid_map` where `{pid}` is the PID of the child
502    /// process. See [`user_namespaces(7)`] for more details.
503    ///
504    /// [`user_namespaces(7)`]: https://man7.org/linux/man-pages/man7/user_namespaces.7.html
505    pub fn map_gid_range(
506        &mut self,
507        starting_inside_gid: libc::gid_t,
508        starting_outside_gid: libc::gid_t,
509        count: u32,
510    ) -> &mut Self {
511        self.namespace |= Namespace::USER;
512        self.gid_map
513            .push((starting_inside_gid, starting_outside_gid, count));
514        self
515    }
516
517    /// Sets the hostname of the container.
518    ///
519    /// Implies `Namespace::UTS`, which requires `CAP_SYS_ADMIN`.
520    ///
521    /// ```no_run
522    /// use reverie_process::Container;
523    ///
524    /// let container = Container::new().map_root().hostname("foobar.local");
525    /// ```
526    pub fn hostname<S: Into<OsString>>(&mut self, hostname: S) -> &mut Self {
527        self.namespace |= Namespace::UTS;
528        self.hostname = Some(hostname.into());
529        self
530    }
531
532    /// Sets the domain name of the container.
533    ///
534    /// Implies `Namespace::UTS`, which requires `CAP_SYS_ADMIN`.
535    ///
536    /// # Example
537    ///
538    /// ```no_run
539    /// use reverie_process::Container;
540    ///
541    /// let container = Container::new().map_root().domainname("foobar");
542    /// ```
543    pub fn domainname<S: Into<OsString>>(&mut self, domainname: S) -> &mut Self {
544        self.namespace |= Namespace::UTS;
545        self.domainname = Some(domainname.into());
546        self
547    }
548
549    /// Gets the hostname of the container.
550    pub fn get_hostname(&self) -> Option<&OsStr> {
551        self.hostname.as_ref().map(AsRef::as_ref)
552    }
553
554    /// Gets the domainname of the container.
555    pub fn get_domainname(&self) -> Option<&OsStr> {
556        self.domainname.as_ref().map(AsRef::as_ref)
557    }
558
559    /// Adds a file system to be mounted. Note that these are mounted in the same
560    /// order as given.
561    ///
562    /// Implies `Namespace::MOUNT`. Note that `Namespace::USER` should also have
563    /// been set and `map_uid` should have been called in order to gain the
564    /// privileges required to mount.
565    pub fn mount(&mut self, mount: Mount) -> &mut Self {
566        self.namespace |= Namespace::MOUNT;
567        self.mounts.push(mount);
568        self
569    }
570
571    /// Adds multiple mounts.
572    pub fn mounts<I>(&mut self, mounts: I) -> &mut Self
573    where
574        I: IntoIterator<Item = Mount>,
575    {
576        self.namespace |= Namespace::MOUNT;
577        self.mounts.extend(mounts);
578        self
579    }
580
581    /// Sets up the container to have local networking only. This will prevent
582    /// any network communication to the outside world.
583    ///
584    /// Implies `Namespace::NETWORK` and `Namespace::MOUNT`.
585    ///
586    /// This also causes a fresh `/sys` to be mounted to avoid seeing the host
587    /// network interfaces in `/sys/class/net`.
588    pub fn local_networking_only(&mut self) -> &mut Self {
589        if !self.local_networking_only {
590            self.local_networking_only = true;
591            self.namespace |= Namespace::NETWORK;
592            self.mount(Mount::sysfs("/sys"));
593        }
594        self
595    }
596
597    /// Sets the seccomp filter. The filter is loaded immediately before `execve`
598    /// and *after* all `pre_exec` callbacks have been executed. Thus, you will
599    /// still be able to call filtered syscalls from `pre_exec` callbacks.
600    pub fn seccomp(&mut self, filter: seccomp::Filter) -> &mut Self {
601        self.seccomp = Some(filter);
602        self
603    }
604
605    /// Indicates that we want to listen for seccomp events using
606    /// [seccomp_unotify(2)](https://man7.org/linux/man-pages/man2/seccomp_unotify.2.html).
607    ///
608    /// If this is set, the seccomp listener file descriptor will be accessible
609    /// via the `Child`.
610    pub fn seccomp_notify(&mut self) -> &mut Self {
611        self.seccomp_notify = true;
612        self
613    }
614
615    /// Sets the controlling pseudoterminal for the child process).
616    ///
617    /// In the child process, this has the effect of:
618    ///  1. Creating a new session (with `setsid()`).
619    ///  2. Using an `ioctl` to set the controlling terminal.
620    ///  3. Setting this file descriptor as the stdio streams.
621    ///
622    /// NOTE: Since this modifies the stdio streams, calling this will reset
623    /// [`Self::stdin`], [`Self::stdout`], and [`Self::stderr`] back to
624    /// [`Stdio::inherit()`].
625    pub fn pty(&mut self, child: PtyChild) -> &mut Self {
626        self.pty = Some(child);
627        self.stdin = Stdio::inherit();
628        self.stdout = Stdio::inherit();
629        self.stderr = Stdio::inherit();
630        self
631    }
632
633    /// Sets the CPU to which the child threads/processes will be pinned.
634    pub fn affinity(&mut self, affinity: usize) -> &mut Self {
635        self.affinity = Some(affinity);
636        self
637    }
638
639    /// Called by the child process after `clone` to get itself set up for either
640    /// `execve` or running an arbitrary function.
641    ///
642    /// NOTE: Although this function takes `&mut self`, it is only called in the
643    /// context of the child process (which has a copy-on-write view of the
644    /// parent's virtual memory). Thus, the parent's version isn't actually
645    /// modified.
646    pub(super) fn setup(
647        &mut self,
648        context: &ChildContext,
649        pre_exec: &mut [Box<dyn FnMut() -> Result<(), Errno> + Send + Sync>],
650    ) -> Result<(), Error> {
651        self.setup_before_filter(context, pre_exec)?;
652        self.setup_filter(context)
653    }
654
655    fn setup_before_filter(
656        &mut self,
657        context: &ChildContext,
658        pre_exec: &mut [Box<dyn FnMut() -> Result<(), Errno> + Send + Sync>],
659    ) -> Result<(), Error> {
660        // NOTE: This function MUST NOT allocate or deallocate any memory! Doing
661        // so can cause random, difficult to diagnose deadlocks.
662
663        if let Some(pty) = self.pty.take() {
664            // NOTE: This is done *before* setting the stdio streams so that the
665            // user can still override individual streams if they only want them
666            // to be partially attached to the tty.
667            pty.login().context(Context::Tty)?;
668        }
669
670        if let Some(fd) = context.stdin {
671            fd.dup2(libc::STDIN_FILENO)
672                .context(Context::Stdio)?
673                .leave_open();
674        }
675        if let Some(fd) = context.stdout {
676            fd.dup2(libc::STDOUT_FILENO)
677                .context(Context::Stdio)?
678                .leave_open();
679        }
680        if let Some(fd) = context.stderr {
681            fd.dup2(libc::STDERR_FILENO)
682                .context(Context::Stdio)?
683                .leave_open();
684        }
685
686        unsafe { reset_signal_handling() }.context(Context::ResetSignals)?;
687
688        // Set up UID and GID maps.
689        if !context.uid_map.is_empty() {
690            context.map_uid().context(Context::MapUid)?;
691        }
692
693        if !context.gid_map.is_empty() {
694            context.setgroups(false).context(Context::MapGid)?;
695            context.map_gid().context(Context::MapGid)?;
696        }
697
698        // Set host name, if any.
699        if let Some(name) = &self.hostname {
700            Error::result(
701                unsafe { libc::sethostname(name.as_bytes().as_ptr() as *const _, name.len()) },
702                Context::Hostname,
703            )?;
704        }
705
706        // Set domain name, if any.
707        if let Some(name) = &self.domainname {
708            Error::result(
709                unsafe { libc::setdomainname(name.as_bytes().as_ptr() as *const _, name.len()) },
710                Context::Domainname,
711            )?;
712        }
713
714        // Mount all the things.
715        for mount in &mut self.mounts {
716            mount.mount().context(Context::Mount)?;
717        }
718
719        // Change root directory. Note that we do this *after* mounting anything
720        // so that bind mounts sources that live outside of the chroot directory
721        // can work.
722        if let Some(chroot) = &self.chroot {
723            Error::result(unsafe { libc::chroot(chroot.as_ptr()) }, Context::Chroot)?;
724        }
725
726        // Set working directory, if any.
727        if let Some(current_dir) = &self.current_dir {
728            Error::result(unsafe { libc::chdir(current_dir.as_ptr()) }, Context::Chdir)?;
729        }
730
731        // Configure networking.
732        // TODO: Generalize this a bit to allow more complex configuration.
733        if self.local_networking_only {
734            // Need a socket to access the network interface.
735            let sock = Fd::socket(libc::AF_INET, libc::SOCK_DGRAM, libc::IPPROTO_IP)
736                .context(Context::Network)?;
737
738            let loopback = IfName::LOOPBACK;
739
740            // Bring up the loopback interface in the newly mounted sysfs.
741            let flags = loopback.get_flags(&sock).context(Context::Network)?;
742            let flags = flags | libc::IFF_UP as i16;
743            loopback.set_flags(&sock, flags).context(Context::Network)?;
744        }
745
746        if let Some(cpu) = self.affinity {
747            let mut cpu_set = CpuSet::new();
748            cpu_set.set(cpu).context(Context::Affinity)?;
749            sched_setaffinity(nix::unistd::Pid::from_raw(0), &cpu_set)
750                .context(Context::Affinity)?;
751        }
752
753        // NOTE: We must call our pre_exec callbacks BEFORE installing the
754        // seccomp filter because our callbacks could be calling syscalls that
755        // our seccomp filter may be intending to block.
756        for f in pre_exec {
757            f().context(Context::PreExec)?;
758        }
759
760        Ok(())
761    }
762
763    fn setup_filter(&self, context: &ChildContext) -> Result<(), Error> {
764        // Set up the seccomp filter, if any.
765        if let Some(filter) = &self.seccomp {
766            use core::sync::atomic::Ordering;
767
768            // no_new_privs must be set or seccomp will not work.
769            Error::result(
770                unsafe { libc::prctl(libc::PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0) },
771                Context::Seccomp,
772            )?;
773
774            // NOTE: If the supervisor (parent process) wants to listen for
775            // seccomp notifications, we need to be able to pass the file
776            // descriptor to the parent. The most common way to do this is to
777            // set up a socket connection and send the file descriptor. However,
778            // since we just set up a seccomp filter, the filter could apply to
779            // any syscalls we make from here on out. This is especially
780            // troublesome if we're also ptracing this child because our syscall
781            // could result in a premature seccomp stop and cause a deadlock.
782            // Thus, instead, we should pass the file descriptor to the parent
783            // process without making any syscalls. The only way to do that is
784            // to create some shared memory and atomically set an integer.
785            if let Some(shared_fd) = context.seccomp_fd {
786                use std::os::unix::io::IntoRawFd;
787
788                let fd = filter
789                    .load_and_listen()
790                    .context(Context::Seccomp)?
791                    .into_raw_fd();
792
793                shared_fd.store(fd, Ordering::Relaxed);
794
795                // Wait until the parent changes the value back. The parent only
796                // does this after it calls pidfd_getfd to copy the file
797                // descriptor into its own file descriptor table. After this,
798                // the file descriptor can be safely closed, but we won't do
799                // that in order to avoid doing a syscall. The fd will be closed
800                // automatically when execve happens anyway.
801                //
802                // NOTE: Again, we must not perform any syscalls after the
803                // seccomp filter has been installed (except for execve of
804                // course).
805                while shared_fd.load(Ordering::Relaxed) == fd {
806                    // Spin spin spin
807                }
808            } else {
809                filter.load().context(Context::Seccomp)?;
810            }
811        }
812
813        Ok(())
814    }
815
816    /// Runs a function in a new process with the specified namespaces unshared. This
817    /// blocks until the function itself returns and the process has exited.
818    ///
819    /// # Safety
820    ///
821    ///  - This should be called early on in the life of a process, before any
822    ///    other threads are created. This reduces the chance that any global
823    ///    resources (like the Tokio runtime) have been created yet.
824    ///
825    ///  - Memory allocated in the parent must not be freed in the child,
826    ///    especially if using jemalloc where a separate thread does deallocations.
827    pub fn run<F, T>(&mut self, mut f: F) -> Result<T, RunError>
828    where
829        F: FnMut() -> T,
830        T: Serialize + DeserializeOwned,
831    {
832        let clone_flags = self.namespace.bits() | libc::SIGCHLD;
833
834        let uid_map = &make_id_map(&self.uid_map);
835        let gid_map = &make_id_map(&self.gid_map);
836
837        let context = ChildContext {
838            // TODO: Honor stdio options. For now, always inherit from the
839            // parent process.
840            stdin: None,
841            stdout: None,
842            stderr: None,
843            uid_map,
844            gid_map,
845            seccomp_fd: None,
846        };
847
848        // Use a pipe for getting the result of the function out of the child
849        // process.
850        let (mut reader, writer) = pipe()?;
851
852        let writer_fd = writer.as_raw_fd();
853
854        // NOTE: Must use a dynamically allocated stack here. Programs expect to
855        // have at least 2 MB of stack space and if we've already used up some
856        // stack space before this is called we could overflow the stack. See
857        // CHILD_STACK_SIZE for the size child_stack() provides (8 MiB).
858        let mut stack = child_stack()?;
859
860        // Disable io redirection just before forking. We want the child process to
861        // be able to call `println!()` and have that output go to stdout.
862        //
863        // See: https://github.com/rust-lang/rust/issues/35136
864        //
865        // Another way around this weirdness is to not use the default
866        // `print!()` and `println!()` macros so that we can completely bypass
867        // this output capturing.
868        #[cfg(feature = "nightly")]
869        let output_capture = std::io::set_output_capture(None);
870
871        let result = clone_with_stack(
872            || {
873                let value = self.setup(&context, &mut []).map(|()| f());
874
875                let mut writer = std::io::BufWriter::new(Fd::new(writer_fd));
876
877                // Serialize this result with bincode and send it to the parent
878                // process via a pipe.
879                //
880                // TODO: Handle serialization errors(?)
881                bincode::serde::encode_into_std_write(
882                    &value,
883                    &mut writer,
884                    bincode::config::legacy(),
885                )
886                .expect("Failed to serialize return value");
887
888                0
889            },
890            clone_flags,
891            &mut stack,
892        );
893
894        #[cfg(feature = "nightly")]
895        std::io::set_output_capture(output_capture);
896
897        let child = WaitGuard::new(result?);
898
899        // The writer end must be dropped first so that our reader doesn't block
900        // forever.
901        drop(writer);
902
903        // Read the return value. Note that we do this *before* waiting on the
904        // process to exit. Otherwise, for return values that exceed the pipe
905        // capacity, we would deadlock.
906        let mut buf = Vec::new();
907        match reader.read_to_end(&mut buf) {
908            Ok(0) => {
909                // The writer end was closed before anything could be written.
910                // This indicates that the process exited before the return
911                // value could be serialized. The only thing we can do in this
912                // case is collect the exit status of the process.
913                //
914                // NOTE: Since we always send `Result<T, _>` through the pipe,
915                // we can guarantee that a successful serialization will never
916                // be 0 bytes (since it always takes more than 0 bytes to encode
917                // that type).
918                //
919                // NOTE: Since `WaitGuard` is used, we guarantee that the
920                // process will be waited on in the other cases.
921                Err(RunError::ExitStatus(child.wait()?))
922            }
923            Ok(n) => {
924                let value: Result<T, Error> =
925                    bincode::serde::decode_from_slice(&buf[0..n], bincode::config::legacy())
926                        .unwrap()
927                        .0;
928                value.map_err(RunError::Spawn)
929            }
930            Err(err) => {
931                // FIXME: Handle this error
932                panic!("Got unexpected error: {}", err)
933            }
934        }
935    }
936
937    /// Runs child setup and a parent readiness callback before installing the
938    /// unchanged seccomp filter and entering the child workload.
939    ///
940    /// `child_start` runs after namespace/filesystem setup, without creating a
941    /// helper task. It returns child-local state and may transfer up to
942    /// [`MAX_STARTUP_FDS`] owned descriptors through its context. `parent_start`
943    /// runs in the original process, with the actual owned child and received
944    /// descriptors, and must return only when its external resources are ready.
945    /// Its returned owner stays in the parent. Only then may `run` consume the
946    /// child state. Startup endpoint aliases close before seccomp is installed.
947    ///
948    /// The positive, representable `timeout` gives the entire protocol one
949    /// monotonic I/O deadline, including time spent in callbacks. It does not
950    /// preempt arbitrary callback code or destructors. Failure cancels and
951    /// reaps the owned child; actual cleanup errors remain errors. Kernel waits
952    /// for an uninterruptible child still require outer process supervision.
953    ///
954    /// Like [`Self::run`], call this before starting other threads. No signal
955    /// handler or other thread may reap this child; SIGCHLD auto-reaping is
956    /// rejected before clone. Callbacks must obey the existing fork-safety
957    /// rules, close unrelated inherited descriptors, and not fork workers in
958    /// the child. This API does not prove capture completion or guest teardown.
959    ///
960    /// Results are drained before wait, including large values. `run` returns
961    /// a deferred cleanup value just as [`Self::run_with_deferred_drop`] does;
962    /// the returned handle's [`DeferredContainerRun::finalize_with_status`]
963    /// checks the real terminal status before yielding the value.
964    pub fn run_with_startup<P, C, F, O, S, T, D>(
965        &mut self,
966        timeout: std::time::Duration,
967        parent_start: P,
968        mut child_start: C,
969        mut run: F,
970    ) -> Result<(O, DeferredContainerRun<T>), StartupRunError>
971    where
972        P: FnOnce(ParentStartContext<'_>) -> Result<O, StartupError>,
973        C: FnMut(&mut ChildStartContext) -> Result<S, StartupError>,
974        F: FnMut(S) -> (T, D),
975        T: Serialize + DeserializeOwned,
976    {
977        let deadline = std::time::Instant::now()
978            .checked_add(timeout)
979            .filter(|_| !timeout.is_zero())
980            .ok_or(StartupRunError::BeforeClone(StartupError::InvalidTimeout))?;
981        let mut disposition: libc::sigaction = unsafe { std::mem::zeroed() };
982        Errno::result(unsafe {
983            libc::sigaction(libc::SIGCHLD, std::ptr::null(), &mut disposition)
984        })
985        .map_err(|error| StartupRunError::BeforeClone(error.into()))?;
986        if disposition.sa_sigaction == libc::SIG_IGN
987            || disposition.sa_flags & libc::SA_NOCLDWAIT != 0
988        {
989            return Err(StartupRunError::BeforeClone(StartupError::Io(
990                Errno::ECHILD,
991            )));
992        }
993        let (parent_socket, child_socket) =
994            StartupSocket::pair(deadline).map_err(StartupRunError::BeforeClone)?;
995        let uid_map = &make_id_map(&self.uid_map);
996        let gid_map = &make_id_map(&self.gid_map);
997        let context = ChildContext {
998            stdin: None,
999            stdout: None,
1000            stderr: None,
1001            uid_map,
1002            gid_map,
1003            seccomp_fd: None,
1004        };
1005        let (mut reader, writer) =
1006            pipe().map_err(|error| StartupRunError::BeforeClone(error.into()))?;
1007        let writer_fd = writer.as_raw_fd();
1008        let reader_fd = reader.as_raw_fd();
1009        let parent_fd = parent_socket.fd.as_raw_fd();
1010        let child_fd = child_socket.fd.as_raw_fd();
1011        let mut stack =
1012            child_stack().map_err(|error| StartupRunError::BeforeClone(error.into()))?;
1013        let clone_flags = self.namespace.bits() | libc::SIGCHLD;
1014        #[cfg(feature = "nightly")]
1015        let output_capture = std::io::set_output_capture(None);
1016        let result = clone_with_stack(
1017            || {
1018                self.startup_child_run(
1019                    &context,
1020                    StartupChildIo {
1021                        parent_fd,
1022                        child_fd,
1023                        reader_fd,
1024                        writer_fd,
1025                        deadline,
1026                    },
1027                    &mut child_start,
1028                    &mut run,
1029                )
1030            },
1031            clone_flags,
1032            &mut stack,
1033        );
1034        #[cfg(feature = "nightly")]
1035        std::io::set_output_capture(output_capture);
1036        let pid = result.map_err(|error| StartupRunError::BeforeClone(error.into()))?;
1037        let mut child = StartupChild {
1038            wait: Some(WaitGuard::new(pid)),
1039            pidfd: None,
1040        };
1041        drop(child_socket);
1042        drop(writer);
1043        child.pidfd = match Fd::pidfd_open(pid.as_raw(), 0) {
1044            Ok(fd) => Some(fd),
1045            Err(error) => return Err(child.fail(error.into())),
1046        };
1047        let descriptors =
1048            match parent_socket
1049                .receive(Some(STARTUP_REQUEST))
1050                .and_then(|descriptors| {
1051                    // Authorize nothing until the one request, including all ancillary
1052                    // rights and its true stream EOF, has been validated.
1053                    parent_socket.receive(None)?;
1054                    Ok(descriptors)
1055                }) {
1056                Ok(descriptors) => descriptors,
1057                Err(error) => return Err(child.fail(error)),
1058            };
1059        let owner = match parent_start(ParentStartContext {
1060            child_pid: child.wait.as_ref().unwrap().0.unwrap(),
1061            // SAFETY: the context borrows this still-owned descriptor for the call.
1062            child_pidfd: unsafe {
1063                std::os::fd::BorrowedFd::borrow_raw(child.pidfd.as_ref().unwrap().as_raw_fd())
1064            },
1065            deadline,
1066            descriptors,
1067        }) {
1068            Ok(owner) => owner,
1069            Err(error) => return Err(child.fail(error)),
1070        };
1071        let ready = (|| {
1072            parent_socket.send(STARTUP_READY, &StartupFds::default(), None)?;
1073            parent_socket.close_write()?;
1074            // No fallible startup validation remains after final permission.
1075            #[cfg(test)]
1076            if STARTUP_TEST_FAULT.with(|fault| fault.get())
1077                == StartupTestFault::ObservePermissionLate
1078            {
1079                std::thread::sleep(std::time::Duration::from_millis(200));
1080            }
1081            Ok::<_, StartupError>(())
1082        })();
1083        if let Err(error) = ready {
1084            // Reap before dropping the parent's resource owner.
1085            return Err(child.fail(error));
1086        }
1087        drop(parent_socket);
1088        let mut bytes = Vec::new();
1089        match reader.read_to_end(&mut bytes) {
1090            Ok(0) => return Err(child.fail(StartupError::MissingResult)),
1091            Ok(_) => (),
1092            Err(error) => {
1093                return Err(child.fail(StartupError::Io(Errno::new(
1094                    error.raw_os_error().unwrap_or(libc::EIO),
1095                ))));
1096            }
1097        }
1098        let value = match bincode::serde::decode_from_slice::<Result<T, StartupError>, _>(
1099            &bytes,
1100            bincode::config::legacy(),
1101        ) {
1102            Ok((Ok(value), used)) if used == bytes.len() => value,
1103            Ok((Err(error), used)) if used == bytes.len() => return Err(child.fail(error)),
1104            _ => return Err(child.fail(StartupError::Protocol)),
1105        };
1106        Ok((
1107            owner,
1108            DeferredContainerRun {
1109                value: Some(value),
1110                child: child.into_wait(),
1111            },
1112        ))
1113    }
1114
1115    // Shared verbatim child exchange/setup/result path for both ownership APIs.
1116    fn startup_child_run<C, F, S, T, U>(
1117        &mut self,
1118        context: &ChildContext<'_>,
1119        io: StartupChildIo,
1120        child_start: &mut C,
1121        run: &mut F,
1122    ) -> i32
1123    where
1124        C: FnMut(&mut ChildStartContext) -> Result<S, StartupError>,
1125        F: FnMut(S) -> (T, U),
1126        T: Serialize,
1127    {
1128        let StartupChildIo {
1129            parent_fd,
1130            child_fd,
1131            reader_fd,
1132            writer_fd,
1133            deadline,
1134        } = io;
1135        // The outer Rust owners live only in the parent. This branch
1136        // owns its inherited child endpoint and result writer only.
1137        unsafe {
1138            libc::close(parent_fd);
1139            libc::close(reader_fd);
1140        }
1141        let socket = StartupSocket {
1142            fd: Fd::new(child_fd),
1143            deadline,
1144        };
1145        let startup = (|| {
1146            self.setup_before_filter(context, &mut [])
1147                .map_err(StartupError::Setup)?;
1148            let mut child_context = ChildStartContext {
1149                deadline,
1150                descriptors: StartupFds::default(),
1151                failure: None,
1152            };
1153            let state = child_start(&mut child_context)?;
1154            if let Some(error) = child_context.failure {
1155                return Err(error);
1156            }
1157            socket.send(STARTUP_REQUEST, &child_context.descriptors, None)?;
1158            drop(child_context);
1159            socket.close_write()?;
1160            socket.receive(Some(STARTUP_READY))?;
1161            socket.receive(None)?; // Require completed final permission.
1162
1163            Ok(state)
1164        })();
1165        let state = match startup {
1166            Ok(state) => state,
1167            Err(error) => {
1168                let _ = socket.send(STARTUP_FAILURE, &StartupFds::default(), Some(error));
1169                drop(socket);
1170                let mut writer = std::io::BufWriter::new(Fd::new(writer_fd));
1171                bincode::serde::encode_into_std_write(
1172                    Err::<T, StartupError>(error),
1173                    &mut writer,
1174                    bincode::config::legacy(),
1175                )
1176                .expect("Failed to serialize startup refusal");
1177                writer.flush().expect("Failed to flush startup refusal");
1178                drop(writer);
1179                return 1;
1180            }
1181        };
1182        drop(socket);
1183        let (value, deferred) = match self.setup_filter(context) {
1184            Ok(()) => {
1185                let (value, deferred) = run(state);
1186                (Ok(value), Some(deferred))
1187            }
1188            Err(error) => (Err(StartupError::Setup(error)), None),
1189        };
1190        let mut writer = std::io::BufWriter::new(Fd::new(writer_fd));
1191        bincode::serde::encode_into_std_write(&value, &mut writer, bincode::config::legacy())
1192            .expect("Failed to serialize return value");
1193        writer.flush().expect("Failed to flush return value");
1194        drop(writer);
1195        drop(deferred);
1196        0
1197    }
1198
1199    /// Runs the existing startup protocol with retained child/result ownership.
1200    ///
1201    /// Call before starting threads; no handler or other thread may reap this
1202    /// child. The callbacks are borrowed. Parent setup returns unit: retain
1203    /// initialized parent resources outside this call and pair them with every
1204    /// returned owner, including errors. This API cannot clean external state.
1205    /// Child callbacks obey the same fork-safety and no-child-worker contract
1206    /// as [`Self::run_with_startup`]. Namespace/filter/signal policy is unchanged.
1207    ///
1208    /// Linux pidfd support is required and probed before clone. Startup uses
1209    /// one finite timeout, without preempting callbacks. Result acquisition then
1210    /// blocks draining the pipe before wait, with no workload deadline or value
1211    /// size cap. On read failure the same FD and partial bytes remain owned.
1212    /// Generic deserialization happens only after actual successful child wait.
1213    /// Explicit cleanup observation is bounded; implicit Drop can block and
1214    /// requires outer process supervision for uninterruptible/unknown cleanup.
1215    pub fn run_with_startup_owned<P, C, F, S, T, U>(
1216        &mut self,
1217        timeout: std::time::Duration,
1218        parent_start: &mut P,
1219        child_start: &mut C,
1220        run: &mut F,
1221    ) -> Result<OwnedDeferredContainerRun<T>, StartupOwnedFailure<T>>
1222    where
1223        P: FnMut(ParentStartContext<'_>) -> Result<(), StartupError>,
1224        C: FnMut(&mut ChildStartContext) -> Result<S, StartupError>,
1225        F: FnMut(S) -> (T, U),
1226        T: Serialize,
1227    {
1228        use std::os::fd::AsFd;
1229        let before = |cause| StartupOwnedFailure::BeforeClone { cause };
1230        let deadline = std::time::Instant::now()
1231            .checked_add(timeout)
1232            .filter(|_| !timeout.is_zero())
1233            .ok_or_else(|| before(StartupError::InvalidTimeout))?;
1234        let mut disposition: libc::sigaction = unsafe { std::mem::zeroed() };
1235        Errno::result(unsafe {
1236            libc::sigaction(libc::SIGCHLD, std::ptr::null(), &mut disposition)
1237        })
1238        .map_err(|error| before(error.into()))?;
1239        if disposition.sa_sigaction == libc::SIG_IGN
1240            || disposition.sa_flags & libc::SA_NOCLDWAIT != 0
1241        {
1242            return Err(before(StartupError::Io(Errno::ECHILD)));
1243        }
1244        let (parent_socket, child_socket) = StartupSocket::pair(deadline).map_err(before)?;
1245        let uid_map = &make_id_map(&self.uid_map);
1246        let gid_map = &make_id_map(&self.gid_map);
1247        let context = ChildContext {
1248            stdin: None,
1249            stdout: None,
1250            stderr: None,
1251            uid_map,
1252            gid_map,
1253            seccomp_fd: None,
1254        };
1255        let (reader, writer) = pipe().map_err(|error| before(error.into()))?;
1256        #[cfg(test)]
1257        OWNED_RESULT_PIPE_CAPACITY.with(|capacity| {
1258            capacity.set(
1259                Errno::result(unsafe { libc::fcntl(reader.as_raw_fd(), libc::F_GETPIPE_SZ) }).ok(),
1260            );
1261        });
1262        let io = StartupChildIo {
1263            parent_fd: parent_socket.fd.as_raw_fd(),
1264            child_fd: child_socket.fd.as_raw_fd(),
1265            reader_fd: reader.as_raw_fd(),
1266            writer_fd: writer.as_raw_fd(),
1267            deadline,
1268        };
1269        let mut stack = child_stack().map_err(|error| before(error.into()))?;
1270        #[cfg(feature = "nightly")]
1271        let output_capture = std::io::set_output_capture(None);
1272        let namespace = self.namespace;
1273        let result = super::clone::clone_with_stack_owned(
1274            || self.startup_child_run(&context, io, child_start, run),
1275            namespace,
1276            &mut stack,
1277        );
1278        #[cfg(feature = "nightly")]
1279        std::io::set_output_capture(output_capture);
1280        let child = OwnedContainerCleanup::new(result.map_err(|error| before(error.into()))?);
1281        // Install the guard before any fallible parent step or user callback.
1282        let mut owned = OwnedFinalization::new(child, reader);
1283        drop(child_socket);
1284        drop(writer);
1285        if owned.cleanup().pidfd.is_none() {
1286            return Err(owned.fail(OwnedRunFailure::Startup(StartupError::Protocol), deadline));
1287        }
1288        let ready = (|| {
1289            let descriptors = parent_socket.receive(Some(STARTUP_REQUEST))?;
1290            parent_socket.receive(None)?;
1291            parent_start(ParentStartContext {
1292                child_pid: owned.cleanup().pid,
1293                child_pidfd: owned.cleanup().pidfd.as_ref().unwrap().as_fd(),
1294                deadline,
1295                descriptors,
1296            })?;
1297            parent_socket.send(STARTUP_READY, &StartupFds::default(), None)?;
1298            parent_socket.close_write()?;
1299            // As in the original path, no fallible startup check follows final
1300            // permission. Work may already have started when O is rescheduled.
1301            Ok::<_, StartupError>(())
1302        })();
1303        if let Err(error) = ready {
1304            return Err(owned.fail(OwnedRunFailure::Startup(error), deadline));
1305        }
1306        drop(parent_socket);
1307        if let Err(error) = owned.drain() {
1308            return Err(owned.fail(error, deadline));
1309        }
1310        Ok(OwnedDeferredContainerRun { inner: owned })
1311    }
1312
1313    /// Runs a deferred workload while retaining the original child/result owner.
1314    ///
1315    /// This has the fork-safety requirements of [`Self::run`]: call before
1316    /// starting threads, with no handler or other thread reaping this child.
1317    /// The workload may create its own workers; it is not subject to the
1318    /// startup callbacks' no-child-worker contract. The borrowed factory and
1319    /// any external parent resources must remain alive through every returned
1320    /// owner, including errors. Ownership here covers the direct child, not
1321    /// arbitrary descendants.
1322    /// A normal raw-clone callback return exits only its calling thread. Join
1323    /// worker threads before returning, or arrange an explicit group exit
1324    /// (for example `_exit` in `D`) if those workers must end with the callback.
1325    ///
1326    /// The child publishes and closes its encoded result before dropping `D`.
1327    /// Linux pidfd support is required and probed before clone. After clone,
1328    /// failures retain the original wait, reader and exact partial bytes, with
1329    /// **no implicit cleanup attempt**. The workload may already have started
1330    /// before such a failure is detected. Call the retained owner's explicit
1331    /// cancellation/observation methods with an absolute deadline; failed
1332    /// results stay failed even when the child later exits successfully.
1333    /// Atomic pidfd availability is validated after the initial drain: a real
1334    /// read failure is reported first; otherwise missing identity is a protocol
1335    /// refusal with complete encoded bytes and EOF retained.
1336    ///
1337    /// Initial result acquisition blocks draining the pipe before wait, without
1338    /// a workload deadline or size cap. Explicit finalization bounds do not
1339    /// bound that acquisition or child serialization. Implicit owner Drop can
1340    /// block; callers requiring a hard bound need outer process supervision.
1341    /// Generic result decoding is available only after an actual successful
1342    /// child wait, including for an encoded container-setup refusal.
1343    pub fn run_with_deferred_drop_owned<F, T, D>(
1344        &mut self,
1345        run: &mut F,
1346    ) -> Result<OwnedDeferredContainerRun<T>, StartupOwnedFailure<T>>
1347    where
1348        F: FnMut() -> (T, D),
1349        T: Serialize,
1350    {
1351        let before = |cause| StartupOwnedFailure::BeforeClone { cause };
1352        let mut disposition: libc::sigaction = unsafe { std::mem::zeroed() };
1353        Errno::result(unsafe {
1354            libc::sigaction(libc::SIGCHLD, std::ptr::null(), &mut disposition)
1355        })
1356        .map_err(|error| before(error.into()))?;
1357        if disposition.sa_sigaction == libc::SIG_IGN
1358            || disposition.sa_flags & libc::SA_NOCLDWAIT != 0
1359        {
1360            return Err(before(StartupError::Io(Errno::ECHILD)));
1361        }
1362        let uid_map = &make_id_map(&self.uid_map);
1363        let gid_map = &make_id_map(&self.gid_map);
1364        let context = ChildContext {
1365            stdin: None,
1366            stdout: None,
1367            stderr: None,
1368            uid_map,
1369            gid_map,
1370            seccomp_fd: None,
1371        };
1372        let (reader, writer) = pipe().map_err(|error| before(error.into()))?;
1373        let reader_fd = reader.as_raw_fd();
1374        let writer_fd = writer.as_raw_fd();
1375        let mut stack = child_stack().map_err(|error| before(error.into()))?;
1376        #[cfg(feature = "nightly")]
1377        let output_capture = std::io::set_output_capture(None);
1378        let namespace = self.namespace;
1379        let result = super::clone::clone_with_stack_owned(
1380            || {
1381                // Only the child writer is transferred into an owning wrapper.
1382                // Close the inherited reader before setup or workload effects.
1383                unsafe { libc::close(reader_fd) };
1384                let (value, deferred) = match self.setup(&context, &mut []) {
1385                    Ok(()) => {
1386                        let (value, deferred) = run();
1387                        (Ok(value), Some(deferred))
1388                    }
1389                    Err(error) => (Err(StartupError::Setup(error)), None),
1390                };
1391                let mut writer = std::io::BufWriter::new(Fd::new(writer_fd));
1392                bincode::serde::encode_into_std_write(
1393                    &value,
1394                    &mut writer,
1395                    bincode::config::legacy(),
1396                )
1397                .expect("Failed to serialize return value");
1398                writer.flush().expect("Failed to flush return value");
1399                drop(writer);
1400                drop(deferred);
1401                0
1402            },
1403            namespace,
1404            &mut stack,
1405        );
1406        #[cfg(feature = "nightly")]
1407        std::io::set_output_capture(output_capture);
1408        let child = OwnedContainerCleanup::new(result.map_err(|error| before(error.into()))?);
1409        // Nothing fallible or user-controlled precedes installation of this
1410        // original wait owner after clone has succeeded.
1411        let mut owned = OwnedFinalization::new(child, reader);
1412        drop(writer);
1413        #[cfg(test)]
1414        owned_deferred_before_drain(&mut owned);
1415        if let Err(error) = owned.drain() {
1416            return Err(owned.refuse(error));
1417        }
1418        if owned.cleanup().pidfd.is_none() {
1419            return Err(owned.refuse(OwnedRunFailure::Startup(StartupError::Protocol)));
1420        }
1421        Ok(OwnedDeferredContainerRun { inner: owned })
1422    }
1423
1424    /// Runs a function in a new process, publishes its result, and only then
1425    /// drops a child-owned cleanup value.
1426    ///
1427    /// The returned handle owns the mandatory wait for the child. Callers may
1428    /// inspect the provisional value while doing independent work, but can
1429    /// only take ownership of it through
1430    /// [`DeferredContainerRun::finalize`], which rejects an unsuccessful child
1431    /// exit. Dropping the handle still reaps the child, but yields no value.
1432    ///
1433    /// A caller therefore cannot accidentally destructure the value away from
1434    /// the mandatory cleanup check:
1435    ///
1436    /// ```compile_fail
1437    /// use reverie_process::Container;
1438    /// let (value, cleanup) = Container::new()
1439    ///     .run_with_deferred_drop(|| (42, ()))
1440    ///     .unwrap();
1441    /// ```
1442    ///
1443    /// This has the same fork-safety requirements as [`Container::run`].
1444    pub fn run_with_deferred_drop<F, T, D>(
1445        &mut self,
1446        mut f: F,
1447    ) -> Result<DeferredContainerRun<T>, RunError>
1448    where
1449        F: FnMut() -> (T, D),
1450        T: Serialize + DeserializeOwned,
1451    {
1452        let clone_flags = self.namespace.bits() | libc::SIGCHLD;
1453        let uid_map = &make_id_map(&self.uid_map);
1454        let gid_map = &make_id_map(&self.gid_map);
1455        let context = ChildContext {
1456            stdin: None,
1457            stdout: None,
1458            stderr: None,
1459            uid_map,
1460            gid_map,
1461            seccomp_fd: None,
1462        };
1463        let (mut reader, writer) = pipe()?;
1464        let writer_fd = writer.as_raw_fd();
1465        let mut stack = child_stack()?;
1466
1467        #[cfg(feature = "nightly")]
1468        let output_capture = std::io::set_output_capture(None);
1469
1470        let result = clone_with_stack(
1471            || {
1472                let (value, deferred) = match self.setup(&context, &mut []) {
1473                    Ok(()) => {
1474                        let (value, deferred) = f();
1475                        (Ok(value), Some(deferred))
1476                    }
1477                    Err(error) => (Err(error), None),
1478                };
1479                let mut writer = std::io::BufWriter::new(Fd::new(writer_fd));
1480                bincode::serde::encode_into_std_write(
1481                    &value,
1482                    &mut writer,
1483                    bincode::config::legacy(),
1484                )
1485                .expect("Failed to serialize return value");
1486                writer.flush().expect("Failed to flush return value");
1487                drop(writer);
1488                drop(deferred);
1489                0
1490            },
1491            clone_flags,
1492            &mut stack,
1493        );
1494
1495        #[cfg(feature = "nightly")]
1496        std::io::set_output_capture(output_capture);
1497
1498        let child = WaitGuard::new(result?);
1499        drop(writer);
1500
1501        let mut buf = Vec::new();
1502        match reader.read_to_end(&mut buf) {
1503            Ok(0) => Err(RunError::ExitStatus(child.wait()?)),
1504            Ok(n) => {
1505                let value: Result<T, Error> =
1506                    bincode::serde::decode_from_slice(&buf[0..n], bincode::config::legacy())
1507                        .unwrap()
1508                        .0;
1509                Ok(DeferredContainerRun {
1510                    value: Some(value.map_err(RunError::Spawn)?),
1511                    child,
1512                })
1513            }
1514            Err(error) => panic!("Got unexpected error: {error}"),
1515        }
1516    }
1517}
1518
1519/// Maximum number of owned descriptors transferred by one container startup.
1520pub const MAX_STARTUP_FDS: usize = 8;
1521
1522/// Failure of the finite container startup exchange.
1523#[derive(
1524    thiserror::Error,
1525    Debug,
1526    Copy,
1527    Clone,
1528    Eq,
1529    PartialEq,
1530    Serialize,
1531    serde::Deserialize
1532)]
1533pub enum StartupError {
1534    /// Container namespace/filesystem/filter setup failed.
1535    #[error("container setup failed: {0}")]
1536    Setup(Error),
1537    /// A startup syscall failed.
1538    #[error("startup syscall failed: {0}")]
1539    Io(Errno),
1540    /// The caller supplied a zero or unrepresentable startup timeout.
1541    #[error("startup timeout must be positive and representable")]
1542    InvalidTimeout,
1543    /// The single monotonic startup deadline elapsed.
1544    #[error("startup deadline elapsed")]
1545    TimedOut,
1546    /// A callback refused startup.
1547    #[error("startup callback refused")]
1548    Refused,
1549    /// The peer closed its endpoint before completing the exchange.
1550    #[error("startup peer closed prematurely")]
1551    PeerClosed,
1552    /// A frame, phase, descriptor count or result encoding was invalid.
1553    #[error("invalid startup or result protocol")]
1554    Protocol,
1555    /// The child exited before publishing its ordinary result.
1556    #[error("child exited before publishing its result")]
1557    MissingResult,
1558}
1559
1560impl From<Errno> for StartupError {
1561    fn from(error: Errno) -> Self {
1562        Self::Io(error)
1563    }
1564}
1565
1566/// A startup failure together with the actual owned-child cleanup outcome.
1567#[derive(thiserror::Error, Debug, Eq, PartialEq)]
1568pub enum StartupRunError {
1569    /// Failure before a child was successfully cloned.
1570    #[error("before clone: {0}")]
1571    BeforeClone(StartupError),
1572    /// Failure after clone; the child was reaped with this actual status.
1573    #[error("{cause}; child terminal status: {status:?}")]
1574    Child {
1575        /// Original startup or result failure.
1576        cause: StartupError,
1577        /// Actual status returned by waitpid, not an inferred success.
1578        status: ExitStatus,
1579    },
1580    /// Cleanup itself failed; no terminal status is claimed.
1581    #[error("{cause}; child cleanup failed: {errno}")]
1582    Cleanup {
1583        /// Original startup or result failure.
1584        cause: StartupError,
1585        /// Actual cancellation/wait error.
1586        errno: Errno,
1587    },
1588}
1589
1590#[derive(Default)]
1591struct StartupFds {
1592    values: [Option<std::os::fd::OwnedFd>; MAX_STARTUP_FDS],
1593    len: usize,
1594}
1595
1596impl StartupFds {
1597    fn push(&mut self, fd: std::os::fd::OwnedFd) -> Result<(), StartupError> {
1598        if self.len == MAX_STARTUP_FDS {
1599            return Err(StartupError::Protocol);
1600        }
1601        self.values[self.len] = Some(fd);
1602        self.len += 1;
1603        Ok(())
1604    }
1605}
1606
1607/// Child-only setup context, constructed after namespace/filesystem setup.
1608///
1609/// It is neither clonable nor serializable. Transferred descriptors are sent
1610/// once with SCM_RIGHTS, then the child's originals are closed before seccomp.
1611/// This context creates no worker and carries no authority to emit capture data.
1612pub struct ChildStartContext {
1613    deadline: std::time::Instant,
1614    descriptors: StartupFds,
1615    failure: Option<StartupError>,
1616}
1617
1618impl ChildStartContext {
1619    /// The same finite monotonic deadline used by both endpoints.
1620    pub fn deadline(&self) -> std::time::Instant {
1621        self.deadline
1622    }
1623
1624    /// Transfers ownership of a descriptor to the parent startup callback.
1625    /// Exceeding [`MAX_STARTUP_FDS`] refuses startup and closes the supplied FD.
1626    pub fn transfer_fd(&mut self, fd: std::os::fd::OwnedFd) -> Result<(), StartupError> {
1627        let result = self.descriptors.push(fd);
1628        if let Err(error) = result {
1629            self.failure = Some(error);
1630        }
1631        result
1632    }
1633}
1634
1635/// Parent-only startup context bound to this invocation's unreaped child.
1636///
1637/// The PID is a locator; the borrowed pidfd and privately held wait guard bind
1638/// the actual child generation. Constructors are private. No raw PID or FD
1639/// supplied by a caller can manufacture this context.
1640pub struct ParentStartContext<'a> {
1641    child_pid: Pid,
1642    child_pidfd: std::os::fd::BorrowedFd<'a>,
1643    deadline: std::time::Instant,
1644    descriptors: StartupFds,
1645}
1646
1647impl ParentStartContext<'_> {
1648    /// The owned child PID in the parent's namespace; do not reap it separately.
1649    pub fn child_pid(&self) -> Pid {
1650        self.child_pid
1651    }
1652
1653    /// Borrows the owned child generation's pidfd for identity-sensitive setup.
1654    pub fn child_pidfd(&self) -> std::os::fd::BorrowedFd<'_> {
1655        self.child_pidfd
1656    }
1657
1658    /// The same finite monotonic deadline used by both endpoints.
1659    pub fn deadline(&self) -> std::time::Instant {
1660        self.deadline
1661    }
1662
1663    /// Number of descriptors supplied by the child (at most [`MAX_STARTUP_FDS`]).
1664    pub fn descriptor_count(&self) -> usize {
1665        self.descriptors.len
1666    }
1667
1668    /// Takes one received descriptor exactly once. Untaken descriptors close
1669    /// when this context is dropped. An out-of-range index returns `None`.
1670    pub fn take_fd(&mut self, index: usize) -> Option<std::os::fd::OwnedFd> {
1671        self.descriptors.values.get_mut(index)?.take()
1672    }
1673}
1674
1675#[derive(Clone, Copy)]
1676struct StartupChildIo {
1677    parent_fd: i32,
1678    child_fd: i32,
1679    reader_fd: i32,
1680    writer_fd: i32,
1681    deadline: std::time::Instant,
1682}
1683
1684// A fixed, private readiness exchange, not an evidence/event transport. The
1685// sole pair is created before clone; each branch closes the other endpoint.
1686const STARTUP_REQUEST: u8 = 1;
1687const STARTUP_READY: u8 = 2;
1688const STARTUP_FAILURE: u8 = 4;
1689const STARTUP_FRAME_SIZE: usize = 64;
1690
1691struct StartupSocket {
1692    fd: Fd,
1693    deadline: std::time::Instant,
1694}
1695
1696impl StartupSocket {
1697    fn pair(deadline: std::time::Instant) -> Result<(Self, Self), StartupError> {
1698        let mut pair = [-1; 2];
1699        Errno::result(unsafe {
1700            libc::socketpair(
1701                libc::AF_UNIX,
1702                libc::SOCK_STREAM | libc::SOCK_CLOEXEC | libc::SOCK_NONBLOCK,
1703                0,
1704                pair.as_mut_ptr(),
1705            )
1706        })?;
1707        Ok((
1708            Self {
1709                fd: Fd::new(pair[0]),
1710                deadline,
1711            },
1712            Self {
1713                fd: Fd::new(pair[1]),
1714                deadline,
1715            },
1716        ))
1717    }
1718
1719    fn poll(&self, events: libc::c_short) -> Result<(), StartupError> {
1720        loop {
1721            let remaining = self
1722                .deadline
1723                .checked_duration_since(std::time::Instant::now())
1724                .filter(|value| !value.is_zero())
1725                .ok_or(StartupError::TimedOut)?;
1726            let millis = remaining
1727                .as_millis()
1728                .saturating_add(1)
1729                .min(i32::MAX as u128) as i32;
1730            let mut fd = libc::pollfd {
1731                fd: self.fd.as_raw_fd(),
1732                events,
1733                revents: 0,
1734            };
1735            match Errno::result(unsafe { libc::poll(&mut fd, 1, millis) }) {
1736                Ok(0) | Err(Errno::EINTR) => continue,
1737                Ok(_) if fd.revents & libc::POLLNVAL != 0 => {
1738                    return Err(StartupError::Io(Errno::EBADF));
1739                }
1740                Ok(_) => return Ok(()),
1741                Err(error) => return Err(error.into()),
1742            }
1743        }
1744    }
1745
1746    fn send_bytes(&self, bytes: &[u8], fds: &StartupFds) -> Result<(), StartupError> {
1747        // The first successful send carries rights exactly once, even when its
1748        // data is partial. Subsequent sends carry only remaining frame bytes.
1749        let mut ancillary = [0usize; 16];
1750        let mut message: libc::msghdr = unsafe { std::mem::zeroed() };
1751        if fds.len != 0 {
1752            message.msg_control = ancillary.as_mut_ptr().cast();
1753            message.msg_controllen =
1754                unsafe { libc::CMSG_SPACE((fds.len * std::mem::size_of::<i32>()) as u32) as usize };
1755            assert!(message.msg_controllen <= std::mem::size_of_val(&ancillary));
1756            unsafe {
1757                let header = libc::CMSG_FIRSTHDR(&message);
1758                (*header).cmsg_level = libc::SOL_SOCKET;
1759                (*header).cmsg_type = libc::SCM_RIGHTS;
1760                (*header).cmsg_len =
1761                    libc::CMSG_LEN((fds.len * std::mem::size_of::<i32>()) as u32) as usize;
1762                let data = libc::CMSG_DATA(header).cast::<i32>();
1763                for index in 0..fds.len {
1764                    data.add(index)
1765                        .write(fds.values[index].as_ref().unwrap().as_raw_fd());
1766                }
1767            }
1768        }
1769        let mut offset = 0;
1770        while offset < bytes.len() {
1771            self.poll(libc::POLLOUT)?;
1772            let mut iov = libc::iovec {
1773                iov_base: bytes[offset..].as_ptr().cast_mut().cast(),
1774                iov_len: bytes.len() - offset,
1775            };
1776            #[cfg(test)]
1777            if STARTUP_TEST_FAULT.with(|fault| fault.get()) == StartupTestFault::Fragmented {
1778                iov.iov_len = 1;
1779            }
1780            message.msg_iov = &mut iov;
1781            message.msg_iovlen = 1;
1782            match Errno::result(unsafe {
1783                libc::sendmsg(self.fd.as_raw_fd(), &message, libc::MSG_NOSIGNAL)
1784            }) {
1785                Ok(0) => return Err(StartupError::Protocol),
1786                Ok(size) => {
1787                    offset += size as usize;
1788                    message.msg_control = std::ptr::null_mut();
1789                    message.msg_controllen = 0;
1790                }
1791                Err(Errno::EINTR | Errno::EAGAIN) => continue,
1792                Err(error) => return Err(error.into()),
1793            }
1794        }
1795        Ok(())
1796    }
1797
1798    fn send(
1799        &self,
1800        phase: u8,
1801        fds: &StartupFds,
1802        failure: Option<StartupError>,
1803    ) -> Result<(), StartupError> {
1804        let mut frame = [0u8; STARTUP_FRAME_SIZE];
1805        frame[..4].copy_from_slice(b"RVS1");
1806        frame[4] = phase;
1807        frame[5] = fds.len as u8;
1808        if let Some(error) = failure {
1809            frame[6] = bincode::serde::encode_into_slice(
1810                error,
1811                &mut frame[8..],
1812                bincode::config::legacy(),
1813            )
1814            .map_err(|_| StartupError::Protocol)? as u8;
1815        }
1816        #[cfg(test)]
1817        if let Some(result) = self.inject_test_fault(phase, &frame, fds) {
1818            return result;
1819        }
1820        self.send_bytes(&frame, fds)
1821    }
1822
1823    fn receive(&self, phase: Option<u8>) -> Result<StartupFds, StartupError> {
1824        use std::os::fd::FromRawFd;
1825        let mut frame = [0u8; STARTUP_FRAME_SIZE];
1826        let mut offset = 0;
1827        let mut fds = StartupFds::default();
1828        loop {
1829            self.poll(libc::POLLIN)?;
1830            let mut ancillary = [0usize; 16];
1831            let capacity = if phase.is_none() {
1832                1
1833            } else {
1834                frame.len() - offset
1835            };
1836            let mut iov = libc::iovec {
1837                iov_base: frame[offset..].as_mut_ptr().cast(),
1838                iov_len: capacity,
1839            };
1840            let mut message: libc::msghdr = unsafe { std::mem::zeroed() };
1841            message.msg_iov = &mut iov;
1842            message.msg_iovlen = 1;
1843            message.msg_control = ancillary.as_mut_ptr().cast();
1844            message.msg_controllen = unsafe {
1845                libc::CMSG_SPACE((MAX_STARTUP_FDS * std::mem::size_of::<i32>()) as u32) as usize
1846            };
1847            assert!(message.msg_controllen <= std::mem::size_of_val(&ancillary));
1848            let size = match Errno::result(unsafe {
1849                libc::recvmsg(self.fd.as_raw_fd(), &mut message, libc::MSG_CMSG_CLOEXEC)
1850            }) {
1851                Ok(size) => size as usize,
1852                Err(Errno::EINTR | Errno::EAGAIN) => continue,
1853                Err(error) => return Err(error.into()),
1854            };
1855            let mut malformed = message.msg_flags & (libc::MSG_TRUNC | libc::MSG_CTRUNC) != 0;
1856            let previous_fds = fds.len;
1857            // Own every received FD before checking bytes/phase, including
1858            // trailing data and EOF. Linux closes rights lost to MSG_CTRUNC.
1859            unsafe {
1860                let mut header = libc::CMSG_FIRSTHDR(&message);
1861                while !header.is_null() {
1862                    if (*header).cmsg_level == libc::SOL_SOCKET
1863                        && (*header).cmsg_type == libc::SCM_RIGHTS
1864                        && (*header).cmsg_len >= libc::CMSG_LEN(0) as usize
1865                    {
1866                        let bytes = (*header).cmsg_len - libc::CMSG_LEN(0) as usize;
1867                        malformed |= !bytes.is_multiple_of(std::mem::size_of::<i32>());
1868                        let data = libc::CMSG_DATA(header).cast::<i32>();
1869                        for index in 0..bytes / std::mem::size_of::<i32>() {
1870                            let fd = std::os::fd::OwnedFd::from_raw_fd(data.add(index).read());
1871                            if fds.push(fd).is_err() {
1872                                malformed = true;
1873                            }
1874                        }
1875                    } else {
1876                        malformed = true;
1877                    }
1878                    header = libc::CMSG_NXTHDR(&message, header);
1879                }
1880            }
1881            if malformed
1882                || (fds.len != previous_fds && (offset != 0 || phase != Some(STARTUP_REQUEST)))
1883            {
1884                return Err(StartupError::Protocol);
1885            }
1886            if size == 0 {
1887                // SOCK_STREAM has no zero-length data messages. Unlike
1888                // SEQPACKET, this is genuine EOF, never an empty packet.
1889                return if phase.is_none() && fds.len == 0 {
1890                    Ok(fds)
1891                } else if offset == 0 && fds.len == 0 {
1892                    Err(StartupError::PeerClosed)
1893                } else {
1894                    Err(StartupError::Protocol)
1895                };
1896            }
1897            if phase.is_none() {
1898                return Err(StartupError::Protocol);
1899            }
1900            offset += size;
1901            if offset < frame.len() {
1902                continue;
1903            }
1904            if &frame[..4] != b"RVS1" || frame[5] as usize != fds.len || frame[7] != 0 {
1905                return Err(StartupError::Protocol);
1906            }
1907            if frame[4] == STARTUP_FAILURE
1908                && fds.len == 0
1909                && frame[6] != 0
1910                && frame[6] as usize <= frame.len() - 8
1911            {
1912                let end = 8 + frame[6] as usize;
1913                let (error, used) = bincode::serde::decode_from_slice::<StartupError, _>(
1914                    &frame[8..end],
1915                    bincode::config::legacy(),
1916                )
1917                .map_err(|_| StartupError::Protocol)?;
1918                if used != end - 8 || frame[end..].iter().any(|byte| *byte != 0) {
1919                    return Err(StartupError::Protocol);
1920                }
1921                return Err(error);
1922            }
1923            if phase != Some(frame[4])
1924                || frame[6..].iter().any(|byte| *byte != 0)
1925                || (frame[4] != STARTUP_REQUEST && fds.len != 0)
1926            {
1927                return Err(StartupError::Protocol);
1928            }
1929            return Ok(fds);
1930        }
1931    }
1932
1933    fn close_write(&self) -> Result<(), StartupError> {
1934        Errno::result(unsafe { libc::shutdown(self.fd.as_raw_fd(), libc::SHUT_WR) })?;
1935        Ok(())
1936    }
1937}
1938
1939// Test-only wire corruption. It substitutes bytes at the actual private send
1940// boundary; it never bypasses the production decoder, readiness or workload
1941// gate. Thread-local state keeps unrelated container tests independent.
1942#[cfg(test)]
1943#[derive(Copy, Clone, Eq, PartialEq)]
1944enum StartupTestFault {
1945    None,
1946    Fragmented,
1947    RequestEmptyTrailing,
1948    RequestDuplicate,
1949    RequestMalformed,
1950    RequestTrailingRights,
1951    PermissionEmptyTrailing,
1952    PermissionDuplicate,
1953    PermissionMalformed,
1954    PermissionTrailingRights,
1955    ObservePermissionLate,
1956}
1957
1958#[cfg(test)]
1959std::thread_local! {
1960    static STARTUP_TEST_FAULT: std::cell::Cell<StartupTestFault> = const { std::cell::Cell::new(StartupTestFault::None) };
1961}
1962
1963#[cfg(test)]
1964impl StartupSocket {
1965    fn inject_test_fault(
1966        &self,
1967        phase: u8,
1968        frame: &[u8],
1969        fds: &StartupFds,
1970    ) -> Option<Result<(), StartupError>> {
1971        use StartupTestFault::*;
1972        let fault = STARTUP_TEST_FAULT.with(|value| value.get());
1973        let request = phase == STARTUP_REQUEST;
1974        let permission = phase == STARTUP_READY;
1975        if (request && fault == RequestEmptyTrailing)
1976            || (permission && fault == PermissionEmptyTrailing)
1977        {
1978            return Some({
1979                assert_eq!(
1980                    unsafe {
1981                        libc::send(self.fd.as_raw_fd(), std::ptr::null(), 0, libc::MSG_NOSIGNAL)
1982                    },
1983                    0
1984                );
1985                self.send_bytes(b"X", &StartupFds::default())
1986            });
1987        }
1988        if (request && fault == RequestMalformed) || (permission && fault == PermissionMalformed) {
1989            let mut bad = frame.to_vec();
1990            bad[0] ^= 1;
1991            return Some(self.send_bytes(&bad, fds));
1992        }
1993        if (request && fault == RequestDuplicate) || (permission && fault == PermissionDuplicate) {
1994            return Some(
1995                self.send_bytes(frame, fds)
1996                    .and_then(|()| self.send_bytes(frame, fds)),
1997            );
1998        }
1999        if (request && fault == RequestTrailingRights)
2000            || (permission && fault == PermissionTrailingRights)
2001        {
2002            return Some((|| {
2003                self.send_bytes(frame, fds)?;
2004                let mut trailing = StartupFds::default();
2005                trailing.push(std::fs::File::open("/dev/null").unwrap().into())?;
2006                self.send_bytes(b"X", &trailing)
2007            })());
2008        }
2009        Option::None
2010    }
2011}
2012
2013/// An observation of the original child, never an inferred successful exit.
2014#[derive(Clone, Copy, Debug, Eq, PartialEq)]
2015pub enum ChildCleanupObservation {
2016    /// The original child has not yielded a terminal observation yet.
2017    Pending,
2018    /// The exclusive wait obtained this actual terminal status.
2019    Reaped(ExitStatus),
2020    /// The pidfd proves exit, but another reaper consumed the wait status.
2021    ExitedWithoutWaitStatus,
2022    /// Cleanup failed without proving physical termination.
2023    Unknown,
2024}
2025
2026/// Retains the atomically acquired child identity and exclusive wait obligation.
2027///
2028/// No other thread/handler may reap this child or close its private pidfd.
2029/// Explicit waits are bounded; Drop cancels and waits and can block indefinitely
2030/// on an uninterruptible child or persistent inability to observe its exit.
2031/// Such failures require outer process supervision, never a detached reaper.
2032#[derive(Debug)]
2033pub struct OwnedContainerCleanup {
2034    pid: Pid,
2035    wait_owned: bool,
2036    pidfd: Option<std::os::fd::OwnedFd>,
2037    observation: ChildCleanupObservation,
2038    last_error: Option<Errno>,
2039    #[cfg(test)]
2040    signal_error_once: Option<Errno>,
2041    #[cfg(test)]
2042    wait_error_once: Option<Errno>,
2043    #[cfg(test)]
2044    cancellation_observed: Option<std::sync::Arc<std::sync::atomic::AtomicBool>>,
2045}
2046
2047impl OwnedContainerCleanup {
2048    fn new(child: super::clone::OwnedClone) -> Self {
2049        Self {
2050            pid: child.pid,
2051            wait_owned: true,
2052            pidfd: child.pidfd,
2053            observation: ChildCleanupObservation::Pending,
2054            last_error: None,
2055            #[cfg(test)]
2056            signal_error_once: OWNED_STARTUP_CANCEL_ERROR.with(|error| error.take()),
2057            #[cfg(test)]
2058            wait_error_once: None,
2059            #[cfg(test)]
2060            cancellation_observed: None,
2061        }
2062    }
2063
2064    /// The original PID, for diagnostics only; do not signal or reap it separately.
2065    pub fn child_pid(&self) -> Pid {
2066        self.pid
2067    }
2068    /// The last actual observation; a timeout does not fabricate a wait status.
2069    pub fn observation(&self) -> ChildCleanupObservation {
2070        self.observation
2071    }
2072    /// The most recent cleanup error, retained even after later physical exit.
2073    pub fn last_error(&self) -> Option<Errno> {
2074        self.last_error
2075    }
2076
2077    fn settled(&self) -> bool {
2078        matches!(
2079            self.observation,
2080            ChildCleanupObservation::Reaped(_) | ChildCleanupObservation::ExitedWithoutWaitStatus
2081        )
2082    }
2083
2084    fn observe_wait(&mut self) -> Result<(), Errno> {
2085        if !self.wait_owned {
2086            return Ok(());
2087        }
2088        #[cfg(test)]
2089        if let Some(error) = self.wait_error_once.take() {
2090            return Err(error);
2091        }
2092        let mut status = 0;
2093        match Errno::result(unsafe { libc::waitpid(self.pid.as_raw(), &mut status, libc::WNOHANG) })
2094        {
2095            Ok(0) => (),
2096            Ok(pid) => {
2097                assert_eq!(pid, self.pid.as_raw());
2098                self.wait_owned = false;
2099                self.observation = ChildCleanupObservation::Reaped(ExitStatus::from_raw(status));
2100            }
2101            Err(Errno::ECHILD) => {
2102                // Never issue another numeric-PID wait after losing ownership.
2103                self.wait_owned = false;
2104                self.last_error = Some(Errno::ECHILD);
2105                self.observation = ChildCleanupObservation::Unknown;
2106            }
2107            Err(error) => return Err(error),
2108        }
2109        Ok(())
2110    }
2111
2112    /// Observes the actual child until the absolute deadline, retaining ownership.
2113    pub fn wait_until(&mut self, deadline: std::time::Instant) -> ChildCleanupObservation {
2114        loop {
2115            if self.settled() {
2116                return self.observation;
2117            }
2118            match self.observe_wait() {
2119                Ok(()) => (),
2120                Err(Errno::EINTR) => {
2121                    self.last_error = Some(Errno::EINTR);
2122                    if std::time::Instant::now() < deadline {
2123                        continue;
2124                    }
2125                }
2126                Err(error) => {
2127                    self.last_error = Some(error);
2128                    self.observation = ChildCleanupObservation::Unknown;
2129                    return self.observation;
2130                }
2131            }
2132            if self.settled() {
2133                return self.observation;
2134            }
2135            let remaining = deadline.saturating_duration_since(std::time::Instant::now());
2136            let milliseconds = remaining
2137                .as_millis()
2138                .saturating_add(u128::from(!remaining.is_zero()))
2139                .min(10) as i32;
2140            let mut poll = libc::pollfd {
2141                fd: self.pidfd.as_ref().map_or(-1, AsRawFd::as_raw_fd),
2142                events: libc::POLLIN,
2143                revents: 0,
2144            };
2145            match Errno::result(unsafe { libc::poll(&mut poll, 1, milliseconds) }) {
2146                Ok(_) => {
2147                    if poll.revents & (libc::POLLNVAL | libc::POLLERR) != 0 {
2148                        self.last_error = Some(Errno::EBADF);
2149                        self.observation = ChildCleanupObservation::Unknown;
2150                        return self.observation;
2151                    }
2152                    if !self.wait_owned && poll.revents & (libc::POLLIN | libc::POLLHUP) != 0 {
2153                        self.observation = ChildCleanupObservation::ExitedWithoutWaitStatus;
2154                        return self.observation;
2155                    }
2156                }
2157                Err(Errno::EINTR) => {
2158                    self.last_error = Some(Errno::EINTR);
2159                    #[cfg(test)]
2160                    OWNED_POLL_INTERRUPTED.fetch_add(1, std::sync::atomic::Ordering::Release);
2161                }
2162                Err(error) => {
2163                    self.last_error = Some(error);
2164                    self.observation = ChildCleanupObservation::Unknown;
2165                    return self.observation;
2166                }
2167            }
2168            if std::time::Instant::now() >= deadline {
2169                // One nonblocking wait after readiness keeps its real status
2170                // when exit raced with the deadline. It never extends the wait.
2171                if let Err(error) = self.observe_wait() {
2172                    self.last_error = Some(error);
2173                    self.observation = ChildCleanupObservation::Unknown;
2174                }
2175                return self.observation;
2176            }
2177        }
2178    }
2179
2180    fn signal_cancel(&mut self) -> Result<(), Errno> {
2181        #[cfg(test)]
2182        if let Some(error) = self.signal_error_once.take() {
2183            return Err(error);
2184        }
2185        let fd = self.pidfd.as_ref().ok_or(Errno::EBADF)?;
2186        Errno::result(unsafe {
2187            libc::syscall(
2188                libc::SYS_pidfd_send_signal,
2189                fd.as_raw_fd(),
2190                libc::SIGKILL,
2191                std::ptr::null::<libc::siginfo_t>(),
2192                0,
2193            )
2194        })
2195        .map(|_| ())
2196    }
2197
2198    /// Sends cancellation through the held pidfd, then observes until the deadline.
2199    /// ESRCH alone is not evidence of termination. Errors retain this owner.
2200    pub fn cancel_and_wait_until(
2201        &mut self,
2202        deadline: std::time::Instant,
2203    ) -> ChildCleanupObservation {
2204        if self.settled() {
2205            return self.observation;
2206        }
2207        match self.signal_cancel() {
2208            Ok(()) => (),
2209            Err(error) => {
2210                self.last_error = Some(error);
2211                // A refused signal does not prevent independent child exit.
2212                // Always observe the original wait/pidfd, retaining the errno.
2213                // Missing atomic output never permits numeric signalling.
2214            }
2215        }
2216        #[cfg(test)]
2217        if let Some(observed) = &self.cancellation_observed {
2218            observed.store(true, std::sync::atomic::Ordering::Release);
2219        }
2220        self.wait_until(deadline)
2221    }
2222}
2223
2224impl Drop for OwnedContainerCleanup {
2225    fn drop(&mut self) {
2226        while !self.settled() {
2227            self.cancel_and_wait_until(
2228                std::time::Instant::now() + std::time::Duration::from_millis(100),
2229            );
2230            if !self.settled() {
2231                std::thread::sleep(std::time::Duration::from_millis(10));
2232            }
2233        }
2234    }
2235}
2236
2237#[cfg(test)]
2238std::thread_local! {
2239    static OWNED_STARTUP_CANCEL_ERROR: std::cell::Cell<Option<Errno>> = const { std::cell::Cell::new(None) };
2240    static OWNED_DEFERRED_DRAIN_HOOK: std::cell::Cell<Option<fn(RawFd)>> = const { std::cell::Cell::new(None) };
2241    static OWNED_RESULT_PIPE_CAPACITY: std::cell::Cell<Option<i32>> = const { std::cell::Cell::new(None) };
2242}
2243#[cfg(test)]
2244fn owned_deferred_before_drain<T>(owned: &mut OwnedFinalization<T>) {
2245    if let Some(hook) = OWNED_DEFERRED_DRAIN_HOOK.with(|hook| hook.take()) {
2246        hook(owned.reader.as_ref().unwrap().as_raw_fd());
2247    }
2248}
2249
2250#[cfg(test)]
2251static OWNED_POLL_INTERRUPTED: std::sync::atomic::AtomicUsize =
2252    std::sync::atomic::AtomicUsize::new(0);
2253
2254/// The original failure, separate from subsequent cleanup observations.
2255#[derive(Clone, Copy, Debug, Eq, PartialEq)]
2256pub enum OwnedRunFailure {
2257    /// The caller explicitly cancelled this run. External error details remain
2258    /// with that caller; later successful child exit cannot erase cancellation.
2259    Cancelled,
2260    /// Startup or the child's encoded startup/setup refusal.
2261    Startup(StartupError),
2262    /// The actual result-reader syscall failed; partial bytes remain retained.
2263    ResultRead(Errno),
2264    /// The real child returned an unsuccessful terminal status.
2265    ChildStatus(ExitStatus),
2266    /// Cleanup failed before an actual terminal status could be obtained.
2267    Cleanup(Errno),
2268    /// Physical exit is established, but a competing reaper lost the status.
2269    WaitStatusUnavailable,
2270}
2271
2272/// An owned startup or result-acquisition refusal. Every post-clone failure
2273/// retains the real child, including failures from an owned deferred workload.
2274#[derive(Debug)]
2275pub enum StartupOwnedFailure<T> {
2276    /// The clone did not create a child.
2277    BeforeClone {
2278        /// The original refusal.
2279        cause: StartupError,
2280    },
2281    /// A child exists; inspect/retry/dispose its retained owner explicitly.
2282    AfterClone {
2283        /// The original refusal, independent of cleanup success or failure.
2284        cause: OwnedRunFailure,
2285        /// Original child, open result FD if any, and exact partial bytes.
2286        run: OwnedFinalization<T>,
2287    },
2288}
2289
2290/// Encoded bytes and the child, retained through pending or failed cleanup.
2291/// No generic result value has been deserialized in the parent.
2292#[must_use = "retain or settle the actual child before releasing external owners"]
2293pub struct OwnedFinalization<T> {
2294    child: Option<OwnedContainerCleanup>,
2295    reader: Option<Fd>,
2296    bytes: Vec<u8>,
2297    eof: bool,
2298    failure: Option<OwnedRunFailure>,
2299    marker: std::marker::PhantomData<fn() -> T>,
2300}
2301
2302impl<T> std::fmt::Debug for OwnedFinalization<T> {
2303    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
2304        f.debug_struct("OwnedFinalization")
2305            .field("child", &self.child)
2306            .field("bytes", &self.bytes.len())
2307            .field("eof", &self.eof)
2308            .field("failure", &self.failure)
2309            .finish()
2310    }
2311}
2312
2313impl<T> OwnedFinalization<T> {
2314    fn new(child: OwnedContainerCleanup, reader: Fd) -> Self {
2315        Self {
2316            child: Some(child),
2317            reader: Some(reader),
2318            bytes: Vec::new(),
2319            eof: false,
2320            failure: None,
2321            marker: std::marker::PhantomData,
2322        }
2323    }
2324    /// Exact bytes received so far, which are not a decoded result or authority.
2325    pub fn provisional_bytes(&self) -> &[u8] {
2326        &self.bytes
2327    }
2328    /// Whether an actual EOF completed result acquisition.
2329    pub fn result_eof(&self) -> bool {
2330        self.eof
2331    }
2332    /// Whether cleanup abandoned an unread result reader because a failed
2333    /// owner lacked a pidfd. Collected bytes and the first failure stay intact;
2334    /// abandonment is not EOF, cancellation, or proof of child termination.
2335    pub fn result_reader_abandoned(&self) -> bool {
2336        !self.eof && self.reader.is_none()
2337    }
2338    /// Borrows diagnostic identity and cleanup observations without disarming ownership.
2339    pub fn cleanup(&self) -> &OwnedContainerCleanup {
2340        self.child.as_ref().unwrap()
2341    }
2342    /// The frozen original failure, if one was observed.
2343    pub fn failure(&self) -> Option<OwnedRunFailure> {
2344        self.failure
2345    }
2346
2347    fn fail(
2348        mut self,
2349        cause: OwnedRunFailure,
2350        deadline: std::time::Instant,
2351    ) -> StartupOwnedFailure<T> {
2352        self.failure = Some(cause);
2353        self.child.as_mut().unwrap().cancel_and_wait_until(deadline);
2354        StartupOwnedFailure::AfterClone { cause, run: self }
2355    }
2356    fn refuse(mut self, cause: OwnedRunFailure) -> StartupOwnedFailure<T> {
2357        self.failure = Some(cause);
2358        StartupOwnedFailure::AfterClone { cause, run: self }
2359    }
2360
2361    fn drain(&mut self) -> Result<(), OwnedRunFailure> {
2362        match self.reader.as_mut().unwrap().read_to_end(&mut self.bytes) {
2363            Ok(_) => {
2364                self.eof = true;
2365                self.reader.take();
2366                if self.bytes.is_empty() {
2367                    Err(OwnedRunFailure::Startup(StartupError::MissingResult))
2368                } else {
2369                    Ok(())
2370                }
2371            }
2372            Err(error) => Err(OwnedRunFailure::ResultRead(Errno::new(
2373                error.raw_os_error().unwrap_or(libc::EIO),
2374            ))),
2375        }
2376    }
2377    /// Retries observation/cancellation while preserving bytes and the original
2378    /// child identity, with unread-reader abandonment only as described below.
2379    /// A previously failed result stays failed even if cleanup later succeeds.
2380    /// If that failed owner has no pidfd, close its unread result reader before
2381    /// waiting: retaining it could block the child serializer forever while
2382    /// waiting for that same child. This preserves the original wait and bytes,
2383    /// but permits the child to observe a broken pipe. It does not guarantee
2384    /// arbitrary workers or a child ignoring write failures will terminate.
2385    pub fn retry_until(mut self, deadline: std::time::Instant) -> OwnedFinalize<T> {
2386        self.abandon_unread_result_without_pidfd();
2387        let child = self.child.as_mut().unwrap();
2388        if let Some(cause) = self.failure {
2389            child.cancel_and_wait_until(deadline);
2390            return OwnedFinalize::Failed {
2391                cause,
2392                cleanup: self,
2393            };
2394        }
2395        match child.wait_until(deadline) {
2396            ChildCleanupObservation::Reaped(status) if status.success() => {
2397                assert!(self.eof);
2398                self.child.take();
2399                OwnedFinalize::Complete(OwnedReapedResult {
2400                    bytes: std::mem::take(&mut self.bytes),
2401                    status,
2402                    marker: std::marker::PhantomData,
2403                })
2404            }
2405            ChildCleanupObservation::Reaped(status) => {
2406                let cause = OwnedRunFailure::ChildStatus(status);
2407                self.failure = Some(cause);
2408                OwnedFinalize::Failed {
2409                    cause,
2410                    cleanup: self,
2411                }
2412            }
2413            ChildCleanupObservation::ExitedWithoutWaitStatus => {
2414                let cause = OwnedRunFailure::WaitStatusUnavailable;
2415                self.failure = Some(cause);
2416                OwnedFinalize::Failed {
2417                    cause,
2418                    cleanup: self,
2419                }
2420            }
2421            ChildCleanupObservation::Unknown => {
2422                if let Some(error) = child.last_error() {
2423                    let cause = OwnedRunFailure::Cleanup(error);
2424                    self.failure = Some(cause);
2425                    OwnedFinalize::Failed {
2426                        cause,
2427                        cleanup: self,
2428                    }
2429                } else {
2430                    OwnedFinalize::Pending(self)
2431                }
2432            }
2433            ChildCleanupObservation::Pending => OwnedFinalize::Pending(self),
2434        }
2435    }
2436
2437    /// Cancels through the original owner up to the absolute deadline.
2438    /// An earlier failure takes precedence; otherwise cancellation becomes the
2439    /// sticky failure even if C subsequently exits successfully. A returned
2440    /// failed owner can still be pending cleanup and must be retained or retried.
2441    pub fn cancel_until(mut self, deadline: std::time::Instant) -> OwnedFinalize<T> {
2442        self.failure.get_or_insert(OwnedRunFailure::Cancelled);
2443        self.retry_until(deadline)
2444    }
2445
2446    fn abandon_unread_result_without_pidfd(&mut self) {
2447        if self.failure.is_some()
2448            && self
2449                .child
2450                .as_ref()
2451                .is_some_and(|child| child.pidfd.is_none())
2452        {
2453            self.reader.take();
2454        }
2455    }
2456}
2457
2458impl<T> Drop for OwnedFinalization<T> {
2459    fn drop(&mut self) {
2460        // An unread serializer cannot make progress while this owner both
2461        // keeps its reader open and waits without a cancellation capability.
2462        // Keep the original child wait; only abandon that pipe endpoint.
2463        self.abandon_unread_result_without_pidfd();
2464        drop(self.child.take());
2465        self.reader.take();
2466    }
2467}
2468
2469/// A complete encoded result whose original child has not yet been settled.
2470#[derive(Debug)]
2471#[must_use = "the actual child must be finalized before decoding"]
2472pub struct OwnedDeferredContainerRun<T> {
2473    inner: OwnedFinalization<T>,
2474}
2475impl<T> OwnedDeferredContainerRun<T> {
2476    /// Bytes are provisional until actual child settlement and decoding succeed.
2477    pub fn provisional_bytes(&self) -> &[u8] {
2478        self.inner.provisional_bytes()
2479    }
2480    /// Borrows the actual retained child's diagnostic identity.
2481    pub fn cleanup(&self) -> &OwnedContainerCleanup {
2482        self.inner.cleanup()
2483    }
2484    /// Observes the real child up to the deadline, retaining ownership on failure.
2485    pub fn finalize_until(self, deadline: std::time::Instant) -> OwnedFinalize<T> {
2486        self.inner.retry_until(deadline)
2487    }
2488
2489    /// Records explicit cancellation and makes a bounded attempt through the
2490    /// same child capability, retaining the complete result bytes on failure.
2491    pub fn cancel_until(self, deadline: std::time::Instant) -> OwnedFinalize<T> {
2492        self.inner.cancel_until(deadline)
2493    }
2494}
2495
2496/// The distinction between real completion, persistent failure and an owned pending wait.
2497#[derive(Debug)]
2498pub enum OwnedFinalize<T> {
2499    /// Actual successful child status; result bytes are still encoded.
2500    Complete(OwnedReapedResult<T>),
2501    /// Frozen failure and the still-owned resources, including actual cleanup state.
2502    Failed {
2503        /// Original cause.
2504        cause: OwnedRunFailure,
2505        /// Owned child and result resources.
2506        cleanup: OwnedFinalization<T>,
2507    },
2508    /// The observation deadline elapsed without relinquishing the original child.
2509    Pending(OwnedFinalization<T>),
2510}
2511
2512/// A successful child wait plus original encoded bytes. Decode is deliberately separate.
2513#[derive(Debug)]
2514pub struct OwnedReapedResult<T> {
2515    bytes: Vec<u8>,
2516    status: ExitStatus,
2517    marker: std::marker::PhantomData<fn() -> T>,
2518}
2519impl<T> OwnedReapedResult<T> {
2520    /// The actual successful child status.
2521    pub fn status(&self) -> ExitStatus {
2522        self.status
2523    }
2524    /// The original complete wire bytes.
2525    pub fn encoded_bytes(&self) -> &[u8] {
2526        &self.bytes
2527    }
2528}
2529impl<T: DeserializeOwned> OwnedReapedResult<T> {
2530    /// Decodes the entire original value only after actual child termination.
2531    /// Callers owning external workers/factories must settle those before this call.
2532    pub fn decode(self) -> Result<T, OwnedDecodeFailure> {
2533        let (cause, detail) = match bincode::serde::decode_from_slice::<Result<T, StartupError>, _>(
2534            &self.bytes,
2535            bincode::config::legacy(),
2536        ) {
2537            Ok((Ok(value), used)) if used == self.bytes.len() => return Ok(value),
2538            Ok((Err(error), used)) if used == self.bytes.len() => {
2539                (OwnedRunFailure::Startup(error), None)
2540            }
2541            Ok(_) => (
2542                OwnedRunFailure::Startup(StartupError::Protocol),
2543                Some("trailing result bytes".to_owned()),
2544            ),
2545            Err(error) => (
2546                OwnedRunFailure::Startup(StartupError::Protocol),
2547                Some(error.to_string()),
2548            ),
2549        };
2550        Err(OwnedDecodeFailure {
2551            cause,
2552            detail,
2553            bytes: self.bytes,
2554            status: self.status,
2555        })
2556    }
2557}
2558
2559/// Decode refusal retaining the complete bytes and genuine child status.
2560#[derive(Debug)]
2561pub struct OwnedDecodeFailure {
2562    cause: OwnedRunFailure,
2563    detail: Option<String>,
2564    bytes: Vec<u8>,
2565    status: ExitStatus,
2566}
2567impl OwnedDecodeFailure {
2568    /// The original typed refusal.
2569    pub fn cause(&self) -> OwnedRunFailure {
2570        self.cause
2571    }
2572    /// An actual decoding diagnostic when available.
2573    pub fn detail(&self) -> Option<&str> {
2574        self.detail.as_deref()
2575    }
2576    /// Unchanged bytes which caused the refusal.
2577    pub fn encoded_bytes(&self) -> &[u8] {
2578        &self.bytes
2579    }
2580    /// Actual successful child status, which does not erase a decoding failure.
2581    pub fn status(&self) -> ExitStatus {
2582        self.status
2583    }
2584}
2585
2586// Owns cancellation only during startup/result acquisition. Old WaitGuard and
2587// deferred-result drop semantics remain unchanged after successful acquisition.
2588struct StartupChild {
2589    wait: Option<WaitGuard>,
2590    pidfd: Option<Fd>,
2591}
2592
2593impl StartupChild {
2594    fn cancel(&mut self) -> Result<ExitStatus, Errno> {
2595        let wait = self.wait.as_ref().unwrap();
2596        let result = match &self.pidfd {
2597            Some(fd) => Errno::result(unsafe {
2598                libc::syscall(
2599                    libc::SYS_pidfd_send_signal,
2600                    fd.as_raw_fd(),
2601                    libc::SIGKILL,
2602                    std::ptr::null::<libc::siginfo_t>(),
2603                    0,
2604                )
2605            })
2606            .map(|_| ()),
2607            // Before pidfd_open completes, this is still the unreaped child
2608            // returned by clone. Callers must not install an auto-reaper or
2609            // wait for this invocation's child from another thread/handler.
2610            None => Errno::result(unsafe { libc::kill(wait.0.unwrap().as_raw(), libc::SIGKILL) })
2611                .map(|_| ()),
2612        };
2613        match result {
2614            Ok(()) | Err(Errno::ESRCH) => (),
2615            Err(error) => return Err(error),
2616        }
2617        self.wait.take().unwrap().wait()
2618    }
2619
2620    fn fail(&mut self, cause: StartupError) -> StartupRunError {
2621        match self.cancel() {
2622            Ok(status) => StartupRunError::Child { cause, status },
2623            Err(errno) => StartupRunError::Cleanup { cause, errno },
2624        }
2625    }
2626
2627    fn into_wait(mut self) -> WaitGuard {
2628        self.wait.take().unwrap()
2629    }
2630}
2631
2632impl Drop for StartupChild {
2633    fn drop(&mut self) {
2634        if self.wait.is_some() {
2635            let _ = self.cancel();
2636        }
2637    }
2638}
2639
2640pub(super) struct ChildContext<'a> {
2641    pub stdin: Option<&'a Fd>,
2642    pub stdout: Option<&'a Fd>,
2643    pub stderr: Option<&'a Fd>,
2644    pub uid_map: &'a [u8],
2645    pub gid_map: &'a [u8],
2646    pub seccomp_fd: Option<&'a core::sync::atomic::AtomicI32>,
2647}
2648
2649impl<'a> ChildContext<'a> {
2650    fn map_uid(&self) -> Result<(), Errno> {
2651        write_bytes(b"/proc/self/uid_map\0", self.uid_map)
2652    }
2653
2654    fn map_gid(&self) -> Result<(), Errno> {
2655        write_bytes(b"/proc/self/gid_map\0", self.gid_map)
2656    }
2657
2658    fn setgroups(&self, allow: bool) -> Result<(), Errno> {
2659        write_bytes(
2660            b"/proc/self/setgroups\0",
2661            if allow { b"allow\0" } else { b"deny\0" },
2662        )
2663    }
2664}
2665
2666/// An error that ocurred while running a containerized function.
2667#[derive(thiserror::Error, Debug, Eq, PartialEq)]
2668pub enum RunError {
2669    /// An error that occurred while spawning the container.
2670    #[error("Process failed to spawn: {0}")]
2671    Spawn(#[from] Error),
2672
2673    /// The function exited prematurely. This can happen if the function called
2674    /// `std::process::exit(0)`, preventing the return value from being sent to
2675    /// the parent. It can also happen if the process panics.
2676    #[error("Process exited with code: {0:?}")]
2677    ExitStatus(ExitStatus),
2678}
2679
2680impl From<Errno> for RunError {
2681    fn from(errno: Errno) -> Self {
2682        Self::Spawn(Error::from(errno))
2683    }
2684}
2685
2686// Helper guard for making sure that the process gets waited on even if an error
2687// is encountered.
2688struct WaitGuard(Option<Pid>);
2689
2690impl WaitGuard {
2691    pub fn new(pid: Pid) -> Self {
2692        Self(Some(pid))
2693    }
2694
2695    /// Eagerly waits for the pid. Otherwise, it'll get waited on upon drop.
2696    pub fn wait(mut self) -> Result<ExitStatus, Errno> {
2697        self.wait_inner()
2698    }
2699
2700    fn wait_inner(&mut self) -> Result<ExitStatus, Errno> {
2701        let pid = self.0.expect("child wait guard has already been consumed");
2702        #[cfg(test)]
2703        let instrumented_wait = WAITPID_TEST_PID.load(Ordering::Acquire) == pid.as_raw();
2704        let mut status = 0;
2705        loop {
2706            #[cfg(test)]
2707            if instrumented_wait {
2708                WAITPID_ENTERED.store(true, Ordering::Release);
2709            }
2710            match Errno::result(unsafe { libc::waitpid(pid.as_raw(), &mut status, 0) }) {
2711                Ok(ret) => {
2712                    assert_eq!(ret, pid.as_raw());
2713                    self.0 = None;
2714                    return Ok(ExitStatus::from_raw(status));
2715                }
2716                Err(Errno::EINTR) => {
2717                    #[cfg(test)]
2718                    {
2719                        if instrumented_wait {
2720                            WAITPID_INTERRUPTED.fetch_add(1, Ordering::Relaxed);
2721                        }
2722                    }
2723                }
2724                Err(Errno::ECHILD) => {
2725                    self.0 = None;
2726                    return Err(Errno::ECHILD);
2727                }
2728                Err(error) => return Err(error),
2729            }
2730        }
2731    }
2732}
2733
2734/// A provisional container result whose successful value remains owned by its
2735/// mandatory cleanup check.
2736#[must_use = "a deferred container result must be finalized before its value can be returned"]
2737pub struct DeferredContainerRun<T> {
2738    value: Option<T>,
2739    child: WaitGuard,
2740}
2741
2742impl<T> DeferredContainerRun<T> {
2743    /// Borrows the value while the child finishes cleanup.
2744    pub fn provisional(&self) -> &T {
2745        self.value.as_ref().expect("provisional value is present")
2746    }
2747
2748    /// Waits for cleanup and returns the value only after a successful exit.
2749    pub fn finalize(self) -> Result<T, RunError> {
2750        self.finalize_with_status().map(|(value, _status)| value)
2751    }
2752
2753    /// Waits for cleanup and returns the value and actual successful child
2754    /// status. A nonzero/signal status remains [`RunError::ExitStatus`]. This
2755    /// observes this container child only, not any guest's separate teardown.
2756    pub fn finalize_with_status(mut self) -> Result<(T, ExitStatus), RunError> {
2757        let status = self.child.wait()?;
2758        if !status.success() {
2759            return Err(RunError::ExitStatus(status));
2760        }
2761        Ok((
2762            self.value.take().expect("provisional value is present"),
2763            status,
2764        ))
2765    }
2766}
2767
2768impl Drop for WaitGuard {
2769    fn drop(&mut self) {
2770        if self.0.is_some() {
2771            let _ = self.wait_inner();
2772        }
2773    }
2774}
2775
2776#[cfg(test)]
2777static WAITPID_TEST_PID: std::sync::atomic::AtomicI32 = std::sync::atomic::AtomicI32::new(0);
2778#[cfg(test)]
2779static WAITPID_ENTERED: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
2780#[cfg(test)]
2781static WAITPID_INTERRUPTED: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0);
2782
2783#[cfg(test)]
2784mod tests {
2785    use std::sync::Mutex;
2786    use std::sync::atomic::AtomicBool;
2787    use std::sync::atomic::Ordering;
2788    use std::time::Duration;
2789    use std::time::Instant;
2790
2791    use nix::sys::signal::SaFlags;
2792    use nix::sys::signal::SigAction;
2793    use nix::sys::signal::SigHandler;
2794    use nix::sys::signal::SigSet;
2795    use nix::sys::signal::Signal;
2796    use nix::sys::signal::sigaction;
2797
2798    use super::*;
2799
2800    include!("owned_deferred_tests.rs");
2801
2802    fn owned_complete<T>(handle: OwnedDeferredContainerRun<T>) -> OwnedReapedResult<T> {
2803        match handle.finalize_until(Instant::now() + Duration::from_secs(2)) {
2804            OwnedFinalize::Complete(result) => result,
2805            other => panic!(
2806                "expected actual successful child settlement: {:?}",
2807                outcome_kind(&other)
2808            ),
2809        }
2810    }
2811
2812    fn outcome_kind<T>(outcome: &OwnedFinalize<T>) -> &'static str {
2813        match outcome {
2814            OwnedFinalize::Complete(_) => "complete",
2815            OwnedFinalize::Failed { .. } => "failed",
2816            OwnedFinalize::Pending(_) => "pending",
2817        }
2818    }
2819
2820    struct OwnedCloneFaultGuard;
2821    impl OwnedCloneFaultGuard {
2822        fn install(fault: super::super::clone::OwnedCloneTestFault) -> Self {
2823            super::super::clone::OWNED_CLONE_FAULT.with(|f| {
2824                assert_eq!(f.get(), super::super::clone::OwnedCloneTestFault::None);
2825                f.set(fault);
2826            });
2827            Self
2828        }
2829    }
2830    impl Drop for OwnedCloneFaultGuard {
2831        fn drop(&mut self) {
2832            super::super::clone::OWNED_CLONE_FAULT
2833                .with(|f| f.set(super::super::clone::OwnedCloneTestFault::None));
2834        }
2835    }
2836
2837    fn owned_test_pidfd_link(path: &Path) -> bool {
2838        let Some(link) = path.to_str() else {
2839            return false;
2840        };
2841        link == "anon_inode:[pidfd]"
2842            || link
2843                .strip_prefix("pidfd:[")
2844                .and_then(|suffix| suffix.strip_suffix(']'))
2845                .is_some_and(|inode| !inode.is_empty() && inode.bytes().all(|b| b.is_ascii_digit()))
2846    }
2847
2848    #[test]
2849    fn owned_startup_atomic_identity_and_complete_decode() {
2850        if crate::test_runs_in_own_process() {
2851            return;
2852        }
2853        use std::os::fd::AsFd;
2854        let parent = Pid::this();
2855        let mut actual_child = None;
2856        let handle = Container::new()
2857            .run_with_startup_owned(
2858                Duration::from_secs(2),
2859                &mut |mut context| {
2860                    actual_child = Some(context.child_pid());
2861                    assert_ne!(context.child_pid(), parent);
2862                    let fd = context.child_pidfd().as_raw_fd();
2863                    let pidfd_link = std::fs::read_link(format!("/proc/self/fd/{fd}")).unwrap();
2864                    assert!(
2865                        owned_test_pidfd_link(&pidfd_link),
2866                        "live pidfd: {pidfd_link:?}"
2867                    );
2868                    eprintln!("owned pidfd anchor: {pidfd_link:?}");
2869                    assert_ne!(
2870                        unsafe { libc::fcntl(fd, libc::F_GETFD) } & libc::FD_CLOEXEC,
2871                        0
2872                    );
2873                    let mut transferred = std::fs::File::from(context.take_fd(0).unwrap());
2874                    assert!(context.take_fd(0).is_none());
2875                    let mut text = String::new();
2876                    transferred.read_to_string(&mut text).unwrap();
2877                    assert!(text.starts_with(&format!("{} ", context.child_pid())));
2878                    Ok(())
2879                },
2880                &mut |context| {
2881                    // The freshly created parent pidfd is not an inherited C alias.
2882                    let aliases = std::fs::read_dir("/proc/self/fd")
2883                        .unwrap()
2884                        .filter_map(Result::ok)
2885                        .filter_map(|e| std::fs::read_link(e.path()).ok())
2886                        .filter(|p| owned_test_pidfd_link(p))
2887                        .count();
2888                    eprintln!("owned child inherited pidfd aliases: {aliases}");
2889                    let file = std::fs::File::open("/proc/self/stat").unwrap();
2890                    context.transfer_fd(file.as_fd().try_clone_to_owned().unwrap())?;
2891                    Ok((Pid::this(), aliases))
2892                },
2893                &mut |state| {
2894                    assert_eq!(Pid::parent(), parent);
2895                    (state, ())
2896                },
2897            )
2898            .unwrap();
2899        let pid = actual_child.unwrap();
2900        assert_eq!(handle.cleanup().child_pid(), pid);
2901        let result = owned_complete(handle);
2902        assert_eq!(result.status(), ExitStatus::Exited(0));
2903        assert_eq!(result.decode().unwrap(), (pid, 0));
2904        assert_reaped(pid);
2905    }
2906
2907    #[test]
2908    fn owned_parent_callback_unwind_reaps_child_and_retains_factory() {
2909        if crate::test_runs_in_own_process() {
2910            return;
2911        }
2912        struct FactoryCapture {
2913            shared: *mut SharedDropState,
2914            parent: Pid,
2915        }
2916        impl Drop for FactoryCapture {
2917            fn drop(&mut self) {
2918                assert!(
2919                    !unsafe { &*self.shared }
2920                        .finished
2921                        .swap(true, Ordering::SeqCst)
2922                );
2923                assert_eq!(Pid::this(), self.parent, "factory capture belongs to O");
2924            }
2925        }
2926
2927        let (mapping, shared) = new_shared_drop_state();
2928        let capture = FactoryCapture {
2929            shared,
2930            parent: Pid::this(),
2931        };
2932        let actual_child = std::cell::Cell::new(None);
2933        let original_pidfd = std::cell::Cell::new(None);
2934        let observer_pidfd = std::cell::RefCell::new(None);
2935        let child_ref = &actual_child;
2936        let original_ref = &original_pidfd;
2937        let observer_ref = &observer_pidfd;
2938        let mut parent_start = move |context: ParentStartContext<'_>| -> Result<(), StartupError> {
2939            let _keep = &capture;
2940            child_ref.set(Some(context.child_pid()));
2941            original_ref.set(Some(context.child_pidfd().as_raw_fd()));
2942            // This duplicate observes actual exit; it never reaps or cancels C.
2943            *observer_ref.borrow_mut() = Some(context.child_pidfd().try_clone_to_owned().unwrap());
2944            panic!("owned startup parent panic");
2945        };
2946        let caught = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
2947            Container::new().run_with_startup_owned(
2948                Duration::from_secs(2),
2949                &mut parent_start,
2950                &mut |_| Ok(()),
2951                &mut |()| {
2952                    unsafe { &*shared }.started.store(true, Ordering::Release);
2953                    ((), ())
2954                },
2955            )
2956        }));
2957        let panic = match caught {
2958            Err(panic) => panic,
2959            Ok(_) => panic!("the borrowed parent callback must unwind through the API"),
2960        };
2961        assert_eq!(
2962            panic.downcast_ref::<&str>(),
2963            Some(&"owned startup parent panic")
2964        );
2965        let pid = actual_child
2966            .get()
2967            .expect("real child recorded before panic");
2968        let fd = original_pidfd.get().unwrap();
2969        assert_eq!(unsafe { libc::fcntl(fd, libc::F_GETFD) }, -1);
2970        assert_eq!(Errno::last(), Errno::EBADF, "library pidfd must be closed");
2971        let observer = observer_pidfd.borrow_mut().take().unwrap();
2972        let mut poll = libc::pollfd {
2973            fd: observer.as_raw_fd(),
2974            events: libc::POLLIN,
2975            revents: 0,
2976        };
2977        assert_eq!(unsafe { libc::poll(&mut poll, 1, 0) }, 1);
2978        assert_ne!(poll.revents & (libc::POLLIN | libc::POLLHUP), 0);
2979        assert_eq!(poll.revents & libc::POLLNVAL, 0);
2980        assert_reaped(pid);
2981        assert!(!unsafe { &*shared }.started.load(Ordering::Acquire));
2982        assert!(!unsafe { &*shared }.finished.load(Ordering::Acquire));
2983        eprintln!(
2984            "owned parent unwind: child={pid} exit_events={} reaped=true library_pidfd_closed=true factory_retained=true workload_started=false",
2985            poll.revents
2986        );
2987        drop(parent_start);
2988        assert!(unsafe { &*shared }.finished.load(Ordering::Acquire));
2989        drop(observer);
2990        unsafe { unmap_shared_drop_state(mapping, shared) };
2991    }
2992
2993    #[test]
2994    fn owned_startup_namespace_and_filter_are_unchanged() {
2995        if crate::test_runs_in_own_process() {
2996            return;
2997        }
2998        let result = Container::new()
2999            .unshare(Namespace::USER | Namespace::PID)
3000            .run_with_startup_owned(
3001                Duration::from_secs(2),
3002                &mut |_| Ok(()),
3003                &mut |_| Ok(()),
3004                &mut |()| (namespace_population_probe(), ()),
3005            )
3006            .unwrap();
3007        assert_eq!(owned_complete(result).decode().unwrap(), (1, 2, 3));
3008        let filter = seccomp::FilterBuilder::new()
3009            .default_action(seccomp::Action::Allow)
3010            .syscalls([
3011                (
3012                    syscalls::Sysno::sendmsg,
3013                    seccomp::Action::Errno(Errno::EPERM),
3014                ),
3015                (
3016                    syscalls::Sysno::recvmsg,
3017                    seccomp::Action::Errno(Errno::EPERM),
3018                ),
3019                (
3020                    syscalls::Sysno::getppid,
3021                    seccomp::Action::Errno(Errno::EPERM),
3022                ),
3023            ])
3024            .build();
3025        let result = Container::new()
3026            .seccomp(filter)
3027            .run_with_startup_owned(
3028                Duration::from_secs(2),
3029                &mut |_| Ok(()),
3030                &mut |_| Ok(()),
3031                &mut |()| {
3032                    (
3033                        Errno::result(unsafe { libc::syscall(libc::SYS_getppid) }),
3034                        (),
3035                    )
3036                },
3037            )
3038            .unwrap();
3039        assert_eq!(owned_complete(result).decode().unwrap(), Err(Errno::EPERM));
3040    }
3041
3042    #[test]
3043    fn owned_startup_large_result_drains_before_pending_teardown() {
3044        if crate::test_runs_in_own_process() {
3045            return;
3046        }
3047        let (mapping, shared) = new_shared_drop_state();
3048        let handle = Container::new()
3049            .run_with_startup_owned(
3050                Duration::from_secs(2),
3051                &mut |_| Ok(()),
3052                &mut |_| Ok(()),
3053                &mut |()| (vec![37_u8; 10 * 1024 * 1024], BlockingDrop { shared }),
3054            )
3055            .unwrap();
3056        let bytes = handle.provisional_bytes().to_vec();
3057        let capacity = OWNED_RESULT_PIPE_CAPACITY
3058            .with(|capacity| capacity.get())
3059            .unwrap();
3060        assert!(capacity > 0);
3061        assert!(bytes.len() > capacity as usize);
3062        eprintln!(
3063            "owned result pipe capacity={capacity} encoded_bytes={}",
3064            bytes.len()
3065        );
3066        let pid = handle.cleanup().child_pid();
3067        let pending = match handle.finalize_until(Instant::now() + Duration::from_millis(20)) {
3068            OwnedFinalize::Pending(run) => run,
3069            other => panic!("blocked teardown unexpectedly {}", outcome_kind(&other)),
3070        };
3071        assert_eq!(pending.cleanup().child_pid(), pid);
3072        assert!(pending.result_eof());
3073        assert_eq!(pending.provisional_bytes(), bytes);
3074        unsafe { &*shared }.release.store(true, Ordering::Release);
3075        let result = match pending.retry_until(Instant::now() + Duration::from_secs(2)) {
3076            OwnedFinalize::Complete(result) => result,
3077            other => panic!("released child unexpectedly {}", outcome_kind(&other)),
3078        };
3079        assert_eq!(result.encoded_bytes(), bytes);
3080        assert_eq!(result.decode().unwrap(), vec![37_u8; 10 * 1024 * 1024]);
3081        assert_reaped(pid);
3082        unsafe { unmap_shared_drop_state(mapping, shared) };
3083    }
3084
3085    static OWNED_DECODE_COUNT: std::sync::atomic::AtomicUsize =
3086        std::sync::atomic::AtomicUsize::new(0);
3087    #[derive(Debug, serde::Serialize)]
3088    struct OwnedDecodeProbe(u32);
3089    impl<'de> serde::Deserialize<'de> for OwnedDecodeProbe {
3090        fn deserialize<D: serde::Deserializer<'de>>(deserializer: D) -> Result<Self, D::Error> {
3091            OWNED_DECODE_COUNT.fetch_add(1, Ordering::SeqCst);
3092            Ok(Self(<u32 as serde::Deserialize>::deserialize(
3093                deserializer,
3094            )?))
3095        }
3096    }
3097
3098    #[test]
3099    fn owned_pending_result_never_constructs_generic_value() {
3100        if crate::test_runs_in_own_process() {
3101            return;
3102        }
3103        OWNED_DECODE_COUNT.store(0, Ordering::SeqCst);
3104        let (mapping, shared) = new_shared_drop_state();
3105        let handle = Container::new()
3106            .run_with_startup_owned(
3107                Duration::from_secs(2),
3108                &mut |_| Ok(()),
3109                &mut |_| Ok(()),
3110                &mut |()| (OwnedDecodeProbe(91), BlockingDrop { shared }),
3111            )
3112            .unwrap();
3113        assert_eq!(OWNED_DECODE_COUNT.load(Ordering::SeqCst), 0);
3114        let pending = match handle.finalize_until(Instant::now()) {
3115            OwnedFinalize::Pending(run) => run,
3116            _ => panic!("child should still own deferred teardown"),
3117        };
3118        assert_eq!(OWNED_DECODE_COUNT.load(Ordering::SeqCst), 0);
3119        unsafe { &*shared }.release.store(true, Ordering::Release);
3120        let result = match pending.retry_until(Instant::now() + Duration::from_secs(2)) {
3121            OwnedFinalize::Complete(result) => result,
3122            _ => panic!("actual child should complete"),
3123        };
3124        assert_eq!(OWNED_DECODE_COUNT.load(Ordering::SeqCst), 0);
3125        assert_eq!(result.decode().unwrap().0, 91);
3126        assert_eq!(OWNED_DECODE_COUNT.load(Ordering::SeqCst), 1);
3127        unsafe { unmap_shared_drop_state(mapping, shared) };
3128    }
3129
3130    #[test]
3131    fn owned_lost_wait_status_is_physical_exit_not_success() {
3132        if crate::test_runs_in_own_process() {
3133            return;
3134        }
3135        let handle = Container::new()
3136            .run_with_startup_owned(
3137                Duration::from_secs(2),
3138                &mut |_| Ok(()),
3139                &mut |_| Ok(()),
3140                &mut |()| (42_u32, ()),
3141            )
3142            .unwrap();
3143        let pid = handle.cleanup().child_pid();
3144        let bytes = handle.provisional_bytes().to_vec();
3145        // Deliberate opposing violation of the exclusive-reaper contract.
3146        let mut status = 0;
3147        assert_eq!(
3148            unsafe { libc::waitpid(pid.as_raw(), &mut status, 0) },
3149            pid.as_raw()
3150        );
3151        assert_eq!(ExitStatus::from_raw(status), ExitStatus::Exited(0));
3152        match handle.finalize_until(Instant::now() + Duration::from_secs(2)) {
3153            OwnedFinalize::Failed { cause, cleanup } => {
3154                assert_eq!(cause, OwnedRunFailure::WaitStatusUnavailable);
3155                assert_eq!(
3156                    cleanup.cleanup().observation(),
3157                    ChildCleanupObservation::ExitedWithoutWaitStatus
3158                );
3159                assert_eq!(cleanup.cleanup().last_error(), Some(Errno::ECHILD));
3160                assert_eq!(cleanup.provisional_bytes(), bytes);
3161                assert_reaped(pid);
3162            }
3163            _ => panic!("lost wait status must never qualify"),
3164        }
3165    }
3166
3167    #[test]
3168    fn owned_startup_before_parent_failure_retains_cleanup_authority() {
3169        if crate::test_runs_in_own_process() {
3170            return;
3171        }
3172        let (mapping, shared) = new_shared_drop_state();
3173        OWNED_STARTUP_CANCEL_ERROR.with(|v| v.set(Some(Errno::EPERM)));
3174        let mut called = false;
3175        let failed = Container::new()
3176            .run_with_startup_owned(
3177                Duration::from_millis(50),
3178                &mut |_| {
3179                    called = true;
3180                    Ok(())
3181                },
3182                &mut |_| {
3183                    while !unsafe { &*shared }.release.load(Ordering::Acquire) {
3184                        unsafe { libc::sched_yield() };
3185                    }
3186                    Ok(())
3187                },
3188                &mut |()| ((), ()),
3189            )
3190            .unwrap_err();
3191        assert!(!called);
3192        match failed {
3193            StartupOwnedFailure::AfterClone { cause, run } => {
3194                assert_eq!(cause, OwnedRunFailure::Startup(StartupError::TimedOut));
3195                assert_eq!(run.cleanup().last_error(), Some(Errno::EPERM));
3196                assert_eq!(
3197                    run.cleanup().observation(),
3198                    ChildCleanupObservation::Pending
3199                );
3200                let pid = run.cleanup().child_pid();
3201                assert_eq!(unsafe { libc::kill(pid.as_raw(), 0) }, 0);
3202                let fd = run.cleanup().pidfd.as_ref().unwrap().as_raw_fd();
3203                match run.retry_until(Instant::now() + Duration::from_secs(2)) {
3204                    OwnedFinalize::Failed {
3205                        cause: again,
3206                        cleanup,
3207                    } => {
3208                        assert_eq!(cause, again);
3209                        assert_eq!(cleanup.cleanup().pidfd.as_ref().unwrap().as_raw_fd(), fd);
3210                        assert_eq!(
3211                            cleanup.cleanup().observation(),
3212                            ChildCleanupObservation::Reaped(ExitStatus::Signaled(
3213                                Signal::SIGKILL,
3214                                false
3215                            ))
3216                        );
3217                        assert_reaped(pid);
3218                    }
3219                    _ => panic!("cleanup must not erase first refusal"),
3220                }
3221            }
3222            _ => panic!("a real child was created"),
3223        }
3224        unsafe { unmap_shared_drop_state(mapping, shared) };
3225    }
3226
3227    #[test]
3228    fn owned_signal_esrch_is_not_a_terminal_status() {
3229        if crate::test_runs_in_own_process() {
3230            return;
3231        }
3232        let (mapping, shared) = new_shared_drop_state();
3233        let mut handle = Container::new()
3234            .run_with_startup_owned(
3235                Duration::from_secs(2),
3236                &mut |_| Ok(()),
3237                &mut |_| Ok(()),
3238                &mut |()| (7_u32, BlockingDrop { shared }),
3239            )
3240            .unwrap();
3241        let child = handle.inner.child.as_mut().unwrap();
3242        child.signal_error_once = Some(Errno::ESRCH);
3243        assert_eq!(
3244            child.cancel_and_wait_until(Instant::now() + Duration::from_millis(20)),
3245            ChildCleanupObservation::Pending
3246        );
3247        assert_eq!(child.last_error(), Some(Errno::ESRCH));
3248        unsafe { &*shared }.release.store(true, Ordering::Release);
3249        assert_eq!(owned_complete(handle).decode().unwrap(), 7);
3250        unsafe { unmap_shared_drop_state(mapping, shared) };
3251    }
3252
3253    #[test]
3254    fn owned_result_read_error_retains_same_fd_partial_bytes_and_child() {
3255        if crate::test_runs_in_own_process() {
3256            return;
3257        }
3258        let (mapping, shared) = new_shared_drop_state();
3259        let (reader, writer) = pipe().unwrap();
3260        let reader_fd = reader.as_raw_fd();
3261        let writer_fd = writer.as_raw_fd();
3262        let mut stack = child_stack().unwrap();
3263        let child = super::super::clone::clone_with_stack_owned(
3264            || {
3265                unsafe { libc::close(reader_fd) };
3266                let mut out = Fd::new(writer_fd);
3267                out.write_all(b"abc").unwrap();
3268                unsafe { &*shared }.started.store(true, Ordering::Release);
3269                while !unsafe { &*shared }.release.load(Ordering::Acquire) {
3270                    unsafe { libc::sched_yield() };
3271                }
3272                0
3273            },
3274            Namespace::empty(),
3275            &mut stack,
3276        )
3277        .unwrap();
3278        drop(writer);
3279        let mut owned = OwnedFinalization::<u32>::new(OwnedContainerCleanup::new(child), reader);
3280        let deadline = Instant::now() + Duration::from_secs(2);
3281        while !unsafe { &*shared }.started.load(Ordering::Acquire) && Instant::now() < deadline {
3282            std::thread::yield_now();
3283        }
3284        assert!(unsafe { &*shared }.started.load(Ordering::Acquire));
3285        owned.reader.as_ref().unwrap().set_nonblocking().unwrap();
3286        let identity = std::fs::read_link(format!("/proc/self/fd/{reader_fd}")).unwrap();
3287        let cause = owned.drain().unwrap_err();
3288        assert_eq!(cause, OwnedRunFailure::ResultRead(Errno::EAGAIN));
3289        assert_eq!(owned.bytes, b"abc");
3290        owned.child.as_mut().unwrap().signal_error_once = Some(Errno::EPERM);
3291        let run = match owned.fail(cause, Instant::now()) {
3292            StartupOwnedFailure::AfterClone { run, .. } => run,
3293            _ => unreachable!(),
3294        };
3295        assert_eq!(run.reader.as_ref().unwrap().as_raw_fd(), reader_fd);
3296        assert_eq!(
3297            std::fs::read_link(format!("/proc/self/fd/{reader_fd}")).unwrap(),
3298            identity
3299        );
3300        assert_eq!(run.provisional_bytes(), b"abc");
3301        assert!(!run.result_eof());
3302        match run.retry_until(Instant::now() + Duration::from_secs(2)) {
3303            OwnedFinalize::Failed {
3304                cause: again,
3305                cleanup,
3306            } => {
3307                assert_eq!(again, cause);
3308                assert_eq!(cleanup.provisional_bytes(), b"abc");
3309                assert!(matches!(
3310                    cleanup.cleanup().observation(),
3311                    ChildCleanupObservation::Reaped(_)
3312                ));
3313                assert_eq!(cleanup.reader.as_ref().unwrap().as_raw_fd(), reader_fd);
3314                drop(cleanup);
3315            }
3316            _ => panic!("partial failed result cannot become successful"),
3317        }
3318        assert_eq!(unsafe { libc::fcntl(reader_fd, libc::F_GETFD) }, -1);
3319        assert_eq!(Errno::last(), Errno::EBADF);
3320        unsafe { unmap_shared_drop_state(mapping, shared) };
3321    }
3322
3323    #[test]
3324    fn owned_atomic_clone_refusals_do_not_start_work() {
3325        if crate::test_runs_in_own_process() {
3326            return;
3327        }
3328        use super::super::clone::OwnedCloneTestFault;
3329        use super::super::clone::clone_with_stack_owned;
3330        let _fault = OwnedCloneFaultGuard::install(OwnedCloneTestFault::Probe(Errno::ENOSYS));
3331        let result = Container::new().run_with_startup_owned(
3332            Duration::from_secs(2),
3333            &mut |_| panic!("no parent callback"),
3334            &mut |_| -> Result<(), StartupError> { panic!("no child callback") },
3335            &mut |()| ((), ()),
3336        );
3337        assert!(matches!(
3338            result,
3339            Err(StartupOwnedFailure::BeforeClone {
3340                cause: StartupError::Io(Errno::ENOSYS)
3341            })
3342        ));
3343        drop(_fault);
3344        for flag in [
3345            libc::CLONE_VM,
3346            libc::CLONE_FILES,
3347            libc::CLONE_FS,
3348            libc::CLONE_SIGHAND,
3349            libc::CLONE_PARENT_SETTID,
3350            libc::CLONE_THREAD,
3351            libc::CLONE_VFORK,
3352            libc::CLONE_PARENT,
3353            libc::CLONE_DETACHED,
3354        ] {
3355            let mut stack = child_stack().unwrap();
3356            let result = clone_with_stack_owned(
3357                || panic!("invalid flags must not clone"),
3358                Namespace::from_bits_retain(flag),
3359                &mut stack,
3360            );
3361            assert!(matches!(result, Err(Errno::EINVAL)));
3362        }
3363        // RLIMIT is changed ONLY inside this actual child subprocess. The
3364        // shared libtest process and its threads retain their original limits.
3365        for atomic in [false, true] {
3366            let errno = Container::new()
3367                .run(|| {
3368                    if atomic {
3369                        let _fault =
3370                            OwnedCloneFaultGuard::install(OwnedCloneTestFault::ExhaustAtClone);
3371                        let mut stack = child_stack().unwrap();
3372                        clone_with_stack_owned(|| 0, Namespace::empty(), &mut stack)
3373                            .err()
3374                            .unwrap()
3375                    } else {
3376                        let limit = libc::rlimit {
3377                            rlim_cur: 0,
3378                            rlim_max: 0,
3379                        };
3380                        assert_eq!(unsafe { libc::setrlimit(libc::RLIMIT_NOFILE, &limit) }, 0);
3381                        let mut stack = child_stack().unwrap();
3382                        clone_with_stack_owned(|| 0, Namespace::empty(), &mut stack)
3383                            .err()
3384                            .unwrap()
3385                    }
3386                })
3387                .unwrap();
3388            assert_eq!(errno, Errno::EMFILE);
3389        }
3390    }
3391
3392    #[test]
3393    fn owned_missing_atomic_pidfd_refuses_permission_without_fake_owner() {
3394        if crate::test_runs_in_own_process() {
3395            return;
3396        }
3397        let _fault =
3398            OwnedCloneFaultGuard::install(super::super::clone::OwnedCloneTestFault::MissingPidfd);
3399        let (mapping, shared) = new_shared_drop_state();
3400        let result = Container::new().run_with_startup_owned(
3401            Duration::from_millis(50),
3402            &mut |_| panic!("invalid atomic owner must not authorize parent setup"),
3403            &mut |_| Ok(()),
3404            &mut |()| {
3405                unsafe { &*shared }.started.store(true, Ordering::Release);
3406                ((), ())
3407            },
3408        );
3409        match result {
3410            Err(StartupOwnedFailure::AfterClone { cause, run }) => {
3411                assert_eq!(cause, OwnedRunFailure::Startup(StartupError::Protocol));
3412                assert!(run.cleanup().pidfd.is_none());
3413                let pid = run.cleanup().child_pid();
3414                drop(run);
3415                assert_reaped(pid);
3416            }
3417            _ => panic!("missing atomic output must refuse"),
3418        }
3419        assert!(!unsafe { &*shared }.started.load(Ordering::Acquire));
3420        unsafe { unmap_shared_drop_state(mapping, shared) };
3421    }
3422
3423    struct OwnedExitDrop(i32);
3424    impl Drop for OwnedExitDrop {
3425        fn drop(&mut self) {
3426            unsafe { libc::_exit(self.0) }
3427        }
3428    }
3429    struct OwnedSignalDrop;
3430    impl Drop for OwnedSignalDrop {
3431        fn drop(&mut self) {
3432            unsafe { libc::raise(libc::SIGPIPE) };
3433        }
3434    }
3435
3436    #[test]
3437    fn owned_nonzero_and_signal_after_result_remain_failures() {
3438        if crate::test_runs_in_own_process() {
3439            return;
3440        }
3441        let run = Container::new()
3442            .run_with_startup_owned(
3443                Duration::from_secs(2),
3444                &mut |_| Ok(()),
3445                &mut |_| Ok(()),
3446                &mut |()| (123_u32, OwnedExitDrop(73)),
3447            )
3448            .unwrap();
3449        match run.finalize_until(Instant::now() + Duration::from_secs(2)) {
3450            OwnedFinalize::Failed { cause, cleanup } => {
3451                assert_eq!(cause, OwnedRunFailure::ChildStatus(ExitStatus::Exited(73)));
3452                assert!(!cleanup.provisional_bytes().is_empty());
3453            }
3454            _ => panic!("nonzero cleanup cannot yield the provisional result"),
3455        }
3456        let run = Container::new()
3457            .run_with_startup_owned(
3458                Duration::from_secs(2),
3459                &mut |_| Ok(()),
3460                &mut |_| Ok(()),
3461                &mut |()| (123_u32, OwnedSignalDrop),
3462            )
3463            .unwrap();
3464        match run.finalize_until(Instant::now() + Duration::from_secs(2)) {
3465            OwnedFinalize::Failed { cause, .. } => assert_eq!(
3466                cause,
3467                OwnedRunFailure::ChildStatus(ExitStatus::Signaled(Signal::SIGPIPE, false))
3468            ),
3469            _ => panic!("SIGPIPE cannot become serde error or exit zero"),
3470        }
3471    }
3472
3473    #[test]
3474    fn owned_closed_result_reader_observes_actual_sigpipe() {
3475        if crate::test_runs_in_own_process() {
3476            return;
3477        }
3478        let (reader, writer) = pipe().unwrap();
3479        let rfd = reader.as_raw_fd();
3480        let wfd = writer.as_raw_fd();
3481        let (mapping, shared) = new_shared_drop_state();
3482        let mut stack = child_stack().unwrap();
3483        let child = super::super::clone::clone_with_stack_owned(
3484            || {
3485                unsafe { libc::close(rfd) };
3486                unsafe { reset_signal_handling() }.unwrap();
3487                while !unsafe { &*shared }.release.load(Ordering::Acquire) {
3488                    unsafe { libc::sched_yield() };
3489                }
3490                Fd::new(wfd)
3491                    .write_all(b"ordinary serialized output")
3492                    .unwrap();
3493                0
3494            },
3495            Namespace::empty(),
3496            &mut stack,
3497        )
3498        .unwrap();
3499        let mut owner = OwnedContainerCleanup::new(child);
3500        drop(reader);
3501        drop(writer);
3502        unsafe { &*shared }.release.store(true, Ordering::Release);
3503        assert_eq!(
3504            owner.wait_until(Instant::now() + Duration::from_secs(2)),
3505            ChildCleanupObservation::Reaped(ExitStatus::Signaled(Signal::SIGPIPE, false))
3506        );
3507        unsafe { unmap_shared_drop_state(mapping, shared) };
3508    }
3509
3510    #[test]
3511    fn owned_invalid_request_and_permission_never_enter_work() {
3512        if crate::test_runs_in_own_process() {
3513            return;
3514        }
3515        use StartupTestFault::*;
3516        for fault in [
3517            RequestEmptyTrailing,
3518            RequestDuplicate,
3519            RequestMalformed,
3520            RequestTrailingRights,
3521            PermissionEmptyTrailing,
3522            PermissionDuplicate,
3523            PermissionMalformed,
3524            PermissionTrailingRights,
3525        ] {
3526            let _fault = StartupFaultGuard::install(fault);
3527            let (mapping, shared) = new_shared_drop_state();
3528            let result = Container::new().run_with_startup_owned(
3529                Duration::from_secs(2),
3530                &mut |_| Ok(()),
3531                &mut |_| Ok(()),
3532                &mut |()| {
3533                    unsafe { &*shared }.started.store(true, Ordering::Release);
3534                    ((), ())
3535                },
3536            );
3537            match result {
3538                Err(StartupOwnedFailure::AfterClone { run, .. }) => drop(run),
3539                Ok(run) => match run.finalize_until(Instant::now() + Duration::from_secs(2)) {
3540                    OwnedFinalize::Complete(_) => panic!("corrupt permission cannot complete"),
3541                    OwnedFinalize::Failed { .. } => (),
3542                    OwnedFinalize::Pending(_) => panic!("corrupt child did not settle"),
3543                },
3544                _ => panic!("real child must exist"),
3545            }
3546            assert!(!unsafe { &*shared }.started.load(Ordering::Acquire));
3547            unsafe { unmap_shared_drop_state(mapping, shared) };
3548        }
3549    }
3550
3551    #[test]
3552    fn owned_persistent_signal_refusal_still_observes_independent_exit() {
3553        if crate::test_runs_in_own_process() {
3554            return;
3555        }
3556        // Install the real policy ONLY in this isolated process. Both inner
3557        // cancellation attempts must see EPERM; actual wait still obtains 17.
3558        let filter = seccomp::FilterBuilder::new()
3559            .default_action(seccomp::Action::Allow)
3560            .syscalls([(
3561                syscalls::Sysno::pidfd_send_signal,
3562                seccomp::Action::Errno(Errno::EPERM),
3563            )])
3564            .build();
3565        let observed = Container::new()
3566            .seccomp(filter)
3567            .run(|| {
3568                let (mapping, shared) = new_shared_drop_state();
3569                let mut stack = child_stack().unwrap();
3570                let child = super::super::clone::clone_with_stack_owned(
3571                    || {
3572                        while !unsafe { &*shared }.release.load(Ordering::Acquire) {
3573                            unsafe { libc::sched_yield() };
3574                        }
3575                        17
3576                    },
3577                    Namespace::empty(),
3578                    &mut stack,
3579                )
3580                .unwrap();
3581                let pid = child.pid;
3582                let mut owner = OwnedContainerCleanup::new(child);
3583                let fd = owner.pidfd.as_ref().unwrap().as_raw_fd();
3584                assert_ne!(
3585                    unsafe { libc::fcntl(fd, libc::F_GETFD) } & libc::FD_CLOEXEC,
3586                    0
3587                );
3588                assert_eq!(
3589                    owner.cancel_and_wait_until(Instant::now() + Duration::from_millis(20)),
3590                    ChildCleanupObservation::Pending
3591                );
3592                assert_eq!(owner.last_error(), Some(Errno::EPERM));
3593                unsafe { &*shared }.release.store(true, Ordering::Release);
3594                assert_eq!(
3595                    owner.cancel_and_wait_until(Instant::now() + Duration::from_secs(2)),
3596                    ChildCleanupObservation::Reaped(ExitStatus::Exited(17))
3597                );
3598                assert_eq!(owner.last_error(), Some(Errno::EPERM));
3599                assert_eq!(owner.child_pid(), pid);
3600                assert_reaped(pid);
3601                drop(owner);
3602                assert_eq!(unsafe { libc::fcntl(fd, libc::F_GETFD) }, -1);
3603                assert_eq!(Errno::last(), Errno::EBADF);
3604                unsafe { unmap_shared_drop_state(mapping, shared) };
3605                (17, Errno::EPERM)
3606            })
3607            .unwrap();
3608        assert_eq!(observed, (17, Errno::EPERM));
3609    }
3610
3611    #[test]
3612    fn owned_atomic_immediate_exit_has_actual_status_and_reclaims_fd() {
3613        if crate::test_runs_in_own_process() {
3614            return;
3615        }
3616        let mut stack = child_stack().unwrap();
3617        let child =
3618            super::super::clone::clone_with_stack_owned(|| 17, Namespace::empty(), &mut stack)
3619                .unwrap();
3620        let pid = child.pid;
3621        let mut owner = OwnedContainerCleanup::new(child);
3622        let fd = owner.pidfd.as_ref().unwrap().as_raw_fd();
3623        assert_ne!(
3624            unsafe { libc::fcntl(fd, libc::F_GETFD) } & libc::FD_CLOEXEC,
3625            0
3626        );
3627        assert_eq!(
3628            owner.wait_until(Instant::now() + Duration::from_secs(2)),
3629            ChildCleanupObservation::Reaped(ExitStatus::Exited(17))
3630        );
3631        assert_reaped(pid);
3632        drop(owner);
3633        assert_eq!(unsafe { libc::fcntl(fd, libc::F_GETFD) }, -1);
3634        assert_eq!(Errno::last(), Errno::EBADF);
3635    }
3636
3637    #[test]
3638    fn owned_wait_retries_actual_interrupted_pidfd_poll() {
3639        if crate::test_runs_in_own_process() {
3640            return;
3641        }
3642        let _serial = WAITPID_SIGNAL_TEST.lock().unwrap();
3643        OWNED_POLL_INTERRUPTED.store(0, Ordering::Release);
3644        let previous = unsafe {
3645            sigaction(
3646                Signal::SIGUSR2,
3647                &SigAction::new(
3648                    SigHandler::Handler(ignore_test_signal),
3649                    SaFlags::empty(),
3650                    SigSet::empty(),
3651                ),
3652            )
3653        }
3654        .unwrap();
3655        let (mapping, shared) = new_shared_drop_state();
3656        let run = Container::new()
3657            .run_with_startup_owned(
3658                Duration::from_secs(2),
3659                &mut |_| Ok(()),
3660                &mut |_| Ok(()),
3661                &mut |()| (52_u32, BlockingDrop { shared }),
3662            )
3663            .unwrap();
3664        let pid = run.cleanup().child_pid();
3665        let waiting_thread = unsafe { libc::pthread_self() };
3666        let shared_address = shared as usize;
3667        // The O helper is created only after clone; no worker is inherited.
3668        let interrupter = std::thread::spawn(move || {
3669            let until = Instant::now() + Duration::from_secs(2);
3670            while OWNED_POLL_INTERRUPTED.load(Ordering::Acquire) == 0 && Instant::now() < until {
3671                assert_eq!(
3672                    unsafe { libc::pthread_kill(waiting_thread, libc::SIGUSR2) },
3673                    0
3674                );
3675                std::thread::sleep(Duration::from_millis(1));
3676            }
3677            unsafe { &*(shared_address as *mut SharedDropState) }
3678                .release
3679                .store(true, Ordering::Release);
3680        });
3681        let result = owned_complete(run);
3682        interrupter.join().unwrap();
3683        unsafe { sigaction(Signal::SIGUSR2, &previous) }.unwrap();
3684        assert!(OWNED_POLL_INTERRUPTED.load(Ordering::Acquire) > 0);
3685        assert_eq!(result.decode().unwrap(), 52);
3686        assert_reaped(pid);
3687        unsafe { unmap_shared_drop_state(mapping, shared) };
3688    }
3689
3690    struct OwnedFactoryDrop {
3691        index: usize,
3692        counters: *mut [std::sync::atomic::AtomicUsize; 8],
3693        parent: Pid,
3694    }
3695    impl Drop for OwnedFactoryDrop {
3696        fn drop(&mut self) {
3697            assert_eq!(
3698                Pid::this(),
3699                self.parent,
3700                "borrowed O factories must not be dropped in C"
3701            );
3702            let counters = unsafe { &*self.counters };
3703            assert_eq!(
3704                counters[3].load(Ordering::Acquire),
3705                1,
3706                "O worker must already be joined"
3707            );
3708            counters[self.index].fetch_add(1, Ordering::SeqCst);
3709        }
3710    }
3711    #[derive(Debug)]
3712    struct OwnedLifecycleValue(*mut [std::sync::atomic::AtomicUsize; 8]);
3713    impl serde::Serialize for OwnedLifecycleValue {
3714        fn serialize<S: serde::Serializer>(&self, serializer: S) -> Result<S::Ok, S::Error> {
3715            (unsafe { &*self.0 })[0].fetch_add(1, Ordering::SeqCst);
3716            serializer.serialize_u32(29)
3717        }
3718    }
3719    impl Drop for OwnedLifecycleValue {
3720        fn drop(&mut self) {
3721            (unsafe { &*self.0 })[2].fetch_add(1, Ordering::SeqCst);
3722        }
3723    }
3724    struct OwnedLifecycleDeferred {
3725        shared: *mut SharedDropState,
3726        counters: *mut [std::sync::atomic::AtomicUsize; 8],
3727    }
3728    impl Drop for OwnedLifecycleDeferred {
3729        fn drop(&mut self) {
3730            drop(BlockingDrop {
3731                shared: self.shared,
3732            });
3733            (unsafe { &*self.counters })[1].fetch_add(1, Ordering::SeqCst);
3734        }
3735    }
3736
3737    #[test]
3738    fn owned_borrowed_factories_remain_in_parent_until_child_and_worker_settle() {
3739        if crate::test_runs_in_own_process() {
3740            return;
3741        }
3742        use std::sync::atomic::AtomicUsize;
3743        let mapping = unsafe {
3744            libc::mmap(
3745                std::ptr::null_mut(),
3746                std::mem::size_of::<[AtomicUsize; 8]>(),
3747                libc::PROT_READ | libc::PROT_WRITE,
3748                libc::MAP_SHARED | libc::MAP_ANONYMOUS,
3749                -1,
3750                0,
3751            )
3752        };
3753        assert_ne!(mapping, libc::MAP_FAILED);
3754        let counters = mapping.cast::<[AtomicUsize; 8]>();
3755        unsafe { counters.write(std::array::from_fn(|_| AtomicUsize::new(0))) };
3756        let state = unsafe { &*counters };
3757        let (child_mapping, shared) = new_shared_drop_state();
3758        let worker_release = std::sync::Arc::new(AtomicBool::new(false));
3759        let worker = std::cell::RefCell::new(None);
3760        let parent_guard = OwnedFactoryDrop {
3761            index: 4,
3762            counters,
3763            parent: Pid::this(),
3764        };
3765        let child_guard = OwnedFactoryDrop {
3766            index: 5,
3767            counters,
3768            parent: Pid::this(),
3769        };
3770        let run_guard = OwnedFactoryDrop {
3771            index: 6,
3772            counters,
3773            parent: Pid::this(),
3774        };
3775        let worker_ref = &worker;
3776        let release_ref = &worker_release;
3777        let mut parent_start = move |_: ParentStartContext<'_>| {
3778            let _keep = &parent_guard;
3779            let release = std::sync::Arc::clone(release_ref);
3780            *worker_ref.borrow_mut() = Some(std::thread::spawn(move || {
3781                while !release.load(Ordering::Acquire) {
3782                    std::thread::yield_now();
3783                }
3784            }));
3785            Ok(())
3786        };
3787        let mut child_start = move |_: &mut ChildStartContext| {
3788            let _keep = &child_guard;
3789            Ok(())
3790        };
3791        let mut run = move |()| {
3792            let _keep = &run_guard;
3793            (
3794                OwnedLifecycleValue(counters),
3795                OwnedLifecycleDeferred { shared, counters },
3796            )
3797        };
3798        let handle = Container::new()
3799            .run_with_startup_owned(
3800                Duration::from_secs(2),
3801                &mut parent_start,
3802                &mut child_start,
3803                &mut run,
3804            )
3805            .unwrap();
3806        let pending = match handle.finalize_until(Instant::now()) {
3807            OwnedFinalize::Pending(pending) => pending,
3808            _ => panic!("C is still in user teardown"),
3809        };
3810        assert_eq!(state[0].load(Ordering::Acquire), 1);
3811        for counter in &state[1..] {
3812            assert_eq!(counter.load(Ordering::Acquire), 0);
3813        }
3814        unsafe { &*shared }.release.store(true, Ordering::Release);
3815        let result = match pending.retry_until(Instant::now() + Duration::from_secs(2)) {
3816            OwnedFinalize::Complete(result) => result,
3817            _ => panic!("original child should settle"),
3818        };
3819        assert_eq!(state[1].load(Ordering::Acquire), 1);
3820        assert_eq!(state[2].load(Ordering::Acquire), 1);
3821        for counter in &state[3..] {
3822            assert_eq!(counter.load(Ordering::Acquire), 0);
3823        }
3824        worker_release.store(true, Ordering::Release);
3825        worker.borrow_mut().take().unwrap().join().unwrap();
3826        state[3].store(1, Ordering::Release);
3827        drop(parent_start);
3828        drop(child_start);
3829        drop(run);
3830        for counter in &state[..7] {
3831            assert_eq!(counter.load(Ordering::Acquire), 1);
3832        }
3833        assert_eq!(state[7].load(Ordering::Acquire), 0);
3834        // The result remains encoded throughout C/O settlement; a different
3835        // decoding type here verifies the stable u32 wire, not user construction.
3836        assert_eq!(
3837            bincode::serde::decode_from_slice::<Result<u32, StartupError>, _>(
3838                result.encoded_bytes(),
3839                bincode::config::legacy()
3840            )
3841            .unwrap(),
3842            (Ok(29), result.encoded_bytes().len())
3843        );
3844        unsafe { unmap_shared_drop_state(child_mapping, shared) };
3845        unsafe {
3846            std::ptr::drop_in_place(counters);
3847            assert_eq!(
3848                libc::munmap(mapping, std::mem::size_of::<[AtomicUsize; 8]>()),
3849                0
3850            );
3851        }
3852    }
3853
3854    #[test]
3855    fn owned_decode_refusal_retains_exact_bytes_and_actual_status() {
3856        if crate::test_runs_in_own_process() {
3857            return;
3858        }
3859        for bytes in [
3860            vec![255],
3861            {
3862                let mut bytes = bincode::serde::encode_to_vec(
3863                    Ok::<u32, StartupError>(19),
3864                    bincode::config::legacy(),
3865                )
3866                .unwrap();
3867                bytes.push(88);
3868                bytes
3869            },
3870            bincode::serde::encode_to_vec(
3871                Err::<u32, StartupError>(StartupError::Protocol),
3872                bincode::config::legacy(),
3873            )
3874            .unwrap(),
3875        ] {
3876            let run = Container::new()
3877                .run_with_startup_owned(
3878                    Duration::from_secs(2),
3879                    &mut |_| Ok(()),
3880                    &mut |_| Ok(()),
3881                    &mut |()| (19_u32, ()),
3882                )
3883                .unwrap();
3884            let mut result = owned_complete(run);
3885            // Deliberately corrupt only the held wire after genuine C settlement.
3886            result.bytes = bytes.clone();
3887            let failure = result.decode().unwrap_err();
3888            assert_eq!(failure.encoded_bytes(), bytes);
3889            assert_eq!(failure.status(), ExitStatus::Exited(0));
3890            assert_eq!(
3891                failure.cause(),
3892                OwnedRunFailure::Startup(StartupError::Protocol)
3893            );
3894        }
3895    }
3896
3897    #[test]
3898    fn owned_unknown_wait_retains_payload_and_same_child_on_retry() {
3899        if crate::test_runs_in_own_process() {
3900            return;
3901        }
3902        let (mapping, shared) = new_shared_drop_state();
3903        let mut run = Container::new()
3904            .run_with_startup_owned(
3905                Duration::from_secs(2),
3906                &mut |_| Ok(()),
3907                &mut |_| Ok(()),
3908                &mut |()| (93_u32, BlockingDrop { shared }),
3909            )
3910            .unwrap();
3911        let bytes = run.provisional_bytes().to_vec();
3912        let pid = run.cleanup().child_pid();
3913        let fd = run.cleanup().pidfd.as_ref().unwrap().as_raw_fd();
3914        // Inject only the wait syscall error: the real child remains alive,
3915        // its real pidfd remains owned, and the retry performs real cancellation.
3916        run.inner.child.as_mut().unwrap().wait_error_once = Some(Errno::EIO);
3917        let cleanup = match run.finalize_until(Instant::now()) {
3918            OwnedFinalize::Failed { cause, cleanup } => {
3919                assert_eq!(cause, OwnedRunFailure::Cleanup(Errno::EIO));
3920                assert_eq!(
3921                    cleanup.cleanup().observation(),
3922                    ChildCleanupObservation::Unknown
3923                );
3924                assert_eq!(cleanup.provisional_bytes(), bytes);
3925                assert_eq!(cleanup.cleanup().pidfd.as_ref().unwrap().as_raw_fd(), fd);
3926                assert_eq!(cleanup.cleanup().child_pid(), pid);
3927                cleanup
3928            }
3929            _ => panic!("unknown wait must retain a failed owner"),
3930        };
3931        match cleanup.retry_until(Instant::now() + Duration::from_secs(2)) {
3932            OwnedFinalize::Failed { cause, cleanup } => {
3933                assert_eq!(cause, OwnedRunFailure::Cleanup(Errno::EIO));
3934                assert_eq!(cleanup.provisional_bytes(), bytes);
3935                assert_eq!(cleanup.cleanup().child_pid(), pid);
3936                assert_eq!(cleanup.cleanup().pidfd.as_ref().unwrap().as_raw_fd(), fd);
3937                assert_eq!(
3938                    cleanup.cleanup().observation(),
3939                    ChildCleanupObservation::Reaped(ExitStatus::Signaled(Signal::SIGKILL, false))
3940                );
3941                assert_reaped(pid);
3942            }
3943            _ => panic!("settlement must not erase original wait failure"),
3944        }
3945        unsafe { unmap_shared_drop_state(mapping, shared) };
3946    }
3947
3948    struct OwnedExternalFactory {
3949        parent: Pid,
3950        joined: std::sync::Arc<AtomicBool>,
3951        drops: std::sync::Arc<std::sync::atomic::AtomicUsize>,
3952    }
3953    impl Drop for OwnedExternalFactory {
3954        fn drop(&mut self) {
3955            assert_eq!(Pid::this(), self.parent);
3956            assert!(
3957                self.joined.load(Ordering::Acquire),
3958                "factory dropped before O worker joined"
3959            );
3960            self.drops.fetch_add(1, Ordering::SeqCst);
3961        }
3962    }
3963    #[derive(Debug)]
3964    struct OwnedSerializeExit;
3965    impl serde::Serialize for OwnedSerializeExit {
3966        fn serialize<S: serde::Serializer>(&self, _serializer: S) -> Result<S::Ok, S::Error> {
3967            // Die during serialization, before the buffered result can flush.
3968            unsafe { libc::_exit(73) }
3969        }
3970    }
3971
3972    #[test]
3973    fn owned_post_ready_serialization_death_keeps_external_worker_and_factory() {
3974        if crate::test_runs_in_own_process() {
3975            return;
3976        }
3977        use std::sync::Arc;
3978        let release = Arc::new(AtomicBool::new(false));
3979        let joined = Arc::new(AtomicBool::new(false));
3980        let drops = Arc::new(std::sync::atomic::AtomicUsize::new(0));
3981        let factory = OwnedExternalFactory {
3982            parent: Pid::this(),
3983            joined: joined.clone(),
3984            drops: drops.clone(),
3985        };
3986        let worker = std::cell::RefCell::new(None);
3987        let worker_ref = &worker;
3988        let release_ref = &release;
3989        let mut parent_called = false;
3990        let called_ref = &mut parent_called;
3991        let mut parent = move |_: ParentStartContext<'_>| {
3992            let _keep = &factory;
3993            *called_ref = true;
3994            let release = Arc::clone(release_ref);
3995            *worker_ref.borrow_mut() = Some(std::thread::spawn(move || {
3996                while !release.load(Ordering::Acquire) {
3997                    std::thread::yield_now();
3998                }
3999            }));
4000            Ok(())
4001        };
4002        let failure = Container::new()
4003            .run_with_startup_owned(
4004                Duration::from_secs(2),
4005                &mut parent,
4006                &mut |_| Ok(()),
4007                &mut |()| (OwnedSerializeExit, ()),
4008            )
4009            .unwrap_err();
4010        match failure {
4011            StartupOwnedFailure::AfterClone { cause, run } => {
4012                assert_eq!(cause, OwnedRunFailure::Startup(StartupError::MissingResult));
4013                assert_eq!(
4014                    run.cleanup().observation(),
4015                    ChildCleanupObservation::Reaped(ExitStatus::Exited(73))
4016                );
4017                assert!(run.provisional_bytes().is_empty());
4018                assert_eq!(drops.load(Ordering::Acquire), 0);
4019                assert!(!joined.load(Ordering::Acquire));
4020                assert!(!worker.borrow().as_ref().unwrap().is_finished());
4021                let pid = run.cleanup().child_pid();
4022                drop(run);
4023                assert_reaped(pid);
4024            }
4025            _ => panic!("post-ready child death must retain actual cleanup"),
4026        }
4027        assert_eq!(drops.load(Ordering::Acquire), 0);
4028        release.store(true, Ordering::Release);
4029        worker.borrow_mut().take().unwrap().join().unwrap();
4030        joined.store(true, Ordering::Release);
4031        drop(parent);
4032        assert!(parent_called);
4033        assert_eq!(drops.load(Ordering::Acquire), 1);
4034    }
4035
4036    #[test]
4037    fn owned_implicit_disposal_waits_before_worker_factory_and_second_clone() {
4038        if crate::test_runs_in_own_process() {
4039            return;
4040        }
4041        // Persistent cancellation refusal forces Drop to observe natural exit.
4042        // The policy and all helper threads live only in this isolated process.
4043        let filter = seccomp::FilterBuilder::new()
4044            .default_action(seccomp::Action::Allow)
4045            .syscalls([(
4046                syscalls::Sysno::pidfd_send_signal,
4047                seccomp::Action::Errno(Errno::EPERM),
4048            )])
4049            .build();
4050        let result = Container::new()
4051            .seccomp(filter)
4052            .run(|| {
4053                use std::sync::Arc;
4054                let (mapping, shared) = new_shared_drop_state();
4055                let release = Arc::new(AtomicBool::new(false));
4056                let joined = Arc::new(AtomicBool::new(false));
4057                let drops = Arc::new(std::sync::atomic::AtomicUsize::new(0));
4058                let cancel_observed = Arc::new(AtomicBool::new(false));
4059                let second_clone = Arc::new(AtomicBool::new(false));
4060                let worker = std::cell::RefCell::new(None);
4061                let worker_ref = &worker;
4062                let release_ref = &release;
4063                let factory = OwnedExternalFactory {
4064                    parent: Pid::this(),
4065                    joined: joined.clone(),
4066                    drops: drops.clone(),
4067                };
4068                let mut parent = move |_: ParentStartContext<'_>| {
4069                    let _keep = &factory;
4070                    let release = Arc::clone(release_ref);
4071                    *worker_ref.borrow_mut() = Some(std::thread::spawn(move || {
4072                        while !release.load(Ordering::Acquire) {
4073                            std::thread::yield_now();
4074                        }
4075                    }));
4076                    Ok(())
4077                };
4078                let run = Container::new()
4079                    .run_with_startup_owned(
4080                        Duration::from_secs(2),
4081                        &mut parent,
4082                        &mut |_| Ok(()),
4083                        &mut |()| (41_u32, BlockingDrop { shared }),
4084                    )
4085                    .unwrap();
4086                let pid = run.cleanup().child_pid();
4087                let mut pending = match run.finalize_until(Instant::now()) {
4088                    OwnedFinalize::Pending(pending) => pending,
4089                    _ => panic!("child must still hold its teardown gate"),
4090                };
4091                pending.child.as_mut().unwrap().cancellation_observed =
4092                    Some(cancel_observed.clone());
4093                let shared_address = shared as usize;
4094                let observed = cancel_observed.clone();
4095                let observer_joined = joined.clone();
4096                let observer_drops = drops.clone();
4097                let observer_second = second_clone.clone();
4098                let controller = std::thread::spawn(move || {
4099                    let deadline = Instant::now() + Duration::from_secs(2);
4100                    while !observed.load(Ordering::Acquire) && Instant::now() < deadline {
4101                        std::thread::yield_now();
4102                    }
4103                    assert!(
4104                        observed.load(Ordering::Acquire),
4105                        "actual cancellation was not attempted"
4106                    );
4107                    assert!(!observer_joined.load(Ordering::Acquire));
4108                    assert_eq!(observer_drops.load(Ordering::Acquire), 0);
4109                    assert!(!observer_second.load(Ordering::Acquire));
4110                    unsafe { &*(shared_address as *mut SharedDropState) }
4111                        .release
4112                        .store(true, Ordering::Release);
4113                });
4114                drop(pending); // implicit disposal retains ownership across EPERM
4115                assert_reaped(pid);
4116                assert!(unsafe { &*shared }.finished.load(Ordering::Acquire));
4117                controller.join().unwrap();
4118                assert_eq!(drops.load(Ordering::Acquire), 0);
4119                assert!(!joined.load(Ordering::Acquire));
4120                release.store(true, Ordering::Release);
4121                worker.borrow_mut().take().unwrap().join().unwrap();
4122                joined.store(true, Ordering::Release);
4123                drop(parent);
4124                assert_eq!(drops.load(Ordering::Acquire), 1);
4125                // This native caller's ordering is explicit; the API does not
4126                // certify arbitrary outside threads or supply an A3 clone token.
4127                second_clone.store(true, Ordering::Release);
4128                assert_eq!(Container::new().run(|| 23), Ok(23));
4129                unsafe { unmap_shared_drop_state(mapping, shared) };
4130                23
4131            })
4132            .unwrap();
4133        assert_eq!(result, 23);
4134    }
4135
4136    #[derive(Debug)]
4137    struct OwnedStalledSerializer(*mut SharedDropState);
4138    impl serde::Serialize for OwnedStalledSerializer {
4139        fn serialize<S: serde::Serializer>(&self, _serializer: S) -> Result<S::Ok, S::Error> {
4140            unsafe { &*self.0 }.started.store(true, Ordering::Release);
4141            loop {
4142                std::thread::sleep(Duration::from_millis(1));
4143            }
4144        }
4145    }
4146
4147    #[test]
4148    fn owned_stalled_serializer_requires_outer_process_containment() {
4149        if crate::test_runs_in_own_process() {
4150            return;
4151        }
4152        let (mapping, shared) = new_shared_drop_state();
4153        let mut stack = child_stack().unwrap();
4154        // The supervised outer process is PID-namespace init. Its forced death
4155        // also terminates the intentionally stuck inner serializer; no orphan
4156        // helper is left behind. This is a native control, not a guest run.
4157        let outer = super::super::clone::clone_with_stack_owned(
4158            || {
4159                assert_eq!(Pid::this().as_raw(), 1);
4160                let _never_returns = Container::new().run_with_startup_owned(
4161                    Duration::from_millis(50),
4162                    &mut |_| Ok(()),
4163                    &mut |_| Ok(()),
4164                    &mut |()| (OwnedStalledSerializer(shared), ()),
4165                );
4166                unsafe { &*shared }.finished.store(true, Ordering::Release);
4167                99
4168            },
4169            Namespace::USER | Namespace::PID,
4170            &mut stack,
4171        )
4172        .unwrap();
4173        let mut outer = OwnedContainerCleanup::new(outer);
4174        let pid = outer.child_pid();
4175        let ready_deadline = Instant::now() + Duration::from_secs(2);
4176        while !unsafe { &*shared }.started.load(Ordering::Acquire)
4177            && Instant::now() < ready_deadline
4178        {
4179            std::thread::yield_now();
4180        }
4181        assert!(unsafe { &*shared }.started.load(Ordering::Acquire));
4182        // Exceeds the expired startup deadline, which is NOT a result budget.
4183        assert_eq!(
4184            outer.wait_until(Instant::now() + Duration::from_millis(100)),
4185            ChildCleanupObservation::Pending
4186        );
4187        assert!(!unsafe { &*shared }.finished.load(Ordering::Acquire));
4188        assert_eq!(
4189            outer.cancel_and_wait_until(Instant::now() + Duration::from_secs(2)),
4190            ChildCleanupObservation::Reaped(ExitStatus::Signaled(Signal::SIGKILL, false))
4191        );
4192        assert_reaped(pid);
4193        eprintln!(
4194            "stalled serializer: outer_pid={pid} actual_exit=SIGKILL after actual Pending; library result acquisition did not return"
4195        );
4196        unsafe { unmap_shared_drop_state(mapping, shared) };
4197    }
4198
4199    #[test]
4200    fn owned_provisional_cancel_reaps_actual_child_and_preserves_bytes() {
4201        if crate::test_runs_in_own_process() {
4202            return;
4203        }
4204        let (mapping, shared) = new_shared_drop_state();
4205        let run = Container::new()
4206            .run_with_startup_owned(
4207                Duration::from_secs(2),
4208                &mut |_| Ok(()),
4209                &mut |_| Ok(()),
4210                &mut |()| (77_u32, BlockingDrop { shared }),
4211            )
4212            .unwrap();
4213        let pid = run.cleanup().child_pid();
4214        let fd = run.cleanup().pidfd.as_ref().unwrap().as_raw_fd();
4215        let bytes = run.provisional_bytes().to_vec();
4216        let failed = match run.cancel_until(Instant::now() + Duration::from_secs(2)) {
4217            OwnedFinalize::Failed { cause, cleanup } => {
4218                assert_eq!(cause, OwnedRunFailure::Cancelled);
4219                assert_eq!(
4220                    cleanup.cleanup().observation(),
4221                    ChildCleanupObservation::Reaped(ExitStatus::Signaled(Signal::SIGKILL, false))
4222                );
4223                assert_eq!(cleanup.cleanup().child_pid(), pid);
4224                assert_eq!(cleanup.cleanup().pidfd.as_ref().unwrap().as_raw_fd(), fd);
4225                assert_eq!(cleanup.provisional_bytes(), bytes);
4226                assert!(cleanup.result_eof());
4227                cleanup
4228            }
4229            _ => panic!("explicit cancellation must remain failed"),
4230        };
4231        assert_reaped(pid);
4232        match failed.retry_until(Instant::now()) {
4233            OwnedFinalize::Failed { cause, cleanup } => {
4234                assert_eq!(cause, OwnedRunFailure::Cancelled);
4235                assert_eq!(cleanup.provisional_bytes(), bytes);
4236            }
4237            _ => panic!("cleanup must not erase cancellation"),
4238        }
4239        unsafe { unmap_shared_drop_state(mapping, shared) };
4240    }
4241
4242    #[test]
4243    fn owned_pending_cancel_stays_failed_after_refused_signal_and_exit_zero() {
4244        if crate::test_runs_in_own_process() {
4245            return;
4246        }
4247        let filter = seccomp::FilterBuilder::new()
4248            .default_action(seccomp::Action::Allow)
4249            .syscalls([(
4250                syscalls::Sysno::pidfd_send_signal,
4251                seccomp::Action::Errno(Errno::EPERM),
4252            )])
4253            .build();
4254        let result = Container::new()
4255            .seccomp(filter)
4256            .run(|| {
4257                let (mapping, shared) = new_shared_drop_state();
4258                let run = Container::new()
4259                    .run_with_startup_owned(
4260                        Duration::from_secs(2),
4261                        &mut |_| Ok(()),
4262                        &mut |_| Ok(()),
4263                        &mut |()| (78_u32, BlockingDrop { shared }),
4264                    )
4265                    .unwrap();
4266                let pid = run.cleanup().child_pid();
4267                let fd = run.cleanup().pidfd.as_ref().unwrap().as_raw_fd();
4268                let bytes = run.provisional_bytes().to_vec();
4269                let pending = match run.finalize_until(Instant::now()) {
4270                    OwnedFinalize::Pending(pending) => pending,
4271                    _ => panic!("real child must remain held after result EOF"),
4272                };
4273                let failed = match pending.cancel_until(Instant::now() + Duration::from_millis(20))
4274                {
4275                    OwnedFinalize::Failed { cause, cleanup } => {
4276                        assert_eq!(cause, OwnedRunFailure::Cancelled);
4277                        assert_eq!(
4278                            cleanup.cleanup().observation(),
4279                            ChildCleanupObservation::Pending
4280                        );
4281                        assert_eq!(cleanup.cleanup().last_error(), Some(Errno::EPERM));
4282                        assert_eq!(cleanup.cleanup().child_pid(), pid);
4283                        assert_eq!(cleanup.cleanup().pidfd.as_ref().unwrap().as_raw_fd(), fd);
4284                        assert_eq!(cleanup.provisional_bytes(), bytes);
4285                        cleanup
4286                    }
4287                    _ => panic!("failed bounded cancellation must retain the pending owner"),
4288                };
4289                unsafe { &*shared }.release.store(true, Ordering::Release);
4290                match failed.retry_until(Instant::now() + Duration::from_secs(2)) {
4291                    OwnedFinalize::Failed { cause, cleanup } => {
4292                        assert_eq!(cause, OwnedRunFailure::Cancelled);
4293                        assert_eq!(
4294                            cleanup.cleanup().observation(),
4295                            ChildCleanupObservation::Reaped(ExitStatus::Exited(0))
4296                        );
4297                        assert_eq!(cleanup.cleanup().last_error(), Some(Errno::EPERM));
4298                        assert_eq!(cleanup.cleanup().child_pid(), pid);
4299                        assert_eq!(cleanup.cleanup().pidfd.as_ref().unwrap().as_raw_fd(), fd);
4300                        assert_eq!(cleanup.provisional_bytes(), bytes);
4301                    }
4302                    _ => panic!("real exit zero cannot promote a cancelled run"),
4303                }
4304                assert_reaped(pid);
4305                unsafe { unmap_shared_drop_state(mapping, shared) };
4306                true
4307            })
4308            .unwrap();
4309        assert!(result);
4310    }
4311
4312    struct StartupFaultGuard;
4313
4314    impl StartupFaultGuard {
4315        fn install(fault: StartupTestFault) -> Self {
4316            STARTUP_TEST_FAULT.with(|value| {
4317                assert!(value.get() == StartupTestFault::None);
4318                value.set(fault);
4319            });
4320            Self
4321        }
4322    }
4323
4324    impl Drop for StartupFaultGuard {
4325        fn drop(&mut self) {
4326            STARTUP_TEST_FAULT.with(|value| value.set(StartupTestFault::None));
4327        }
4328    }
4329
4330    #[test]
4331    fn startup_corrupt_request_or_permission_never_runs_workload() {
4332        if crate::test_runs_in_own_process() {
4333            return;
4334        }
4335        use StartupTestFault::*;
4336        for fault in [
4337            RequestEmptyTrailing,
4338            RequestDuplicate,
4339            RequestMalformed,
4340            RequestTrailingRights,
4341            PermissionEmptyTrailing,
4342            PermissionDuplicate,
4343            PermissionMalformed,
4344            PermissionTrailingRights,
4345        ] {
4346            let _fault = StartupFaultGuard::install(fault);
4347            let (mapping, shared) = new_shared_drop_state();
4348            let mut parent_called = false;
4349            let result = Container::new().run_with_startup(
4350                Duration::from_secs(2),
4351                |_| {
4352                    parent_called = true;
4353                    Ok(())
4354                },
4355                |_| Ok(()),
4356                |()| {
4357                    unsafe { &*shared }.started.store(true, Ordering::Release);
4358                    ((), ())
4359                },
4360            );
4361            assert!(matches!(
4362                result,
4363                Err(StartupRunError::Child {
4364                    cause: StartupError::Protocol,
4365                    ..
4366                })
4367            ));
4368            assert_eq!(
4369                parent_called,
4370                matches!(
4371                    fault,
4372                    PermissionEmptyTrailing
4373                        | PermissionDuplicate
4374                        | PermissionMalformed
4375                        | PermissionTrailingRights
4376                )
4377            );
4378            assert!(!unsafe { &*shared }.started.load(Ordering::Acquire));
4379            unsafe { unmap_shared_drop_state(mapping, shared) };
4380        }
4381    }
4382
4383    #[test]
4384    fn startup_fragmented_request_transfers_each_right_exactly_once() {
4385        if crate::test_runs_in_own_process() {
4386            return;
4387        }
4388        let _fault = StartupFaultGuard::install(StartupTestFault::Fragmented);
4389        let (pid, handle) = Container::new()
4390            .run_with_startup(
4391                Duration::from_secs(2),
4392                |mut context| {
4393                    assert_eq!(context.descriptor_count(), MAX_STARTUP_FDS);
4394                    for index in 0..MAX_STARTUP_FDS {
4395                        assert!(context.take_fd(index).is_some());
4396                    }
4397                    Ok(context.child_pid())
4398                },
4399                |context| {
4400                    for _ in 0..MAX_STARTUP_FDS {
4401                        context.transfer_fd(std::fs::File::open("/dev/null").unwrap().into())?;
4402                    }
4403                    Ok(())
4404                },
4405                |()| (42, ()),
4406            )
4407            .unwrap();
4408        assert_eq!(
4409            handle.finalize_with_status(),
4410            Ok((42, ExitStatus::Exited(0)))
4411        );
4412        assert_reaped(pid);
4413    }
4414
4415    #[test]
4416    fn startup_final_permission_has_no_later_parent_deadline_validation() {
4417        if crate::test_runs_in_own_process() {
4418            return;
4419        }
4420        // Deliberately delay the parent after final permission is sent and its
4421        // write side closed. This models scheduling after release, without
4422        // bypassing any protocol checks. The child's genuine result still must
4423        // be drained and its actual terminal status checked.
4424        let _fault = StartupFaultGuard::install(StartupTestFault::ObservePermissionLate);
4425        let (pid, handle) = Container::new()
4426            .run_with_startup(
4427                Duration::from_millis(100),
4428                |context| Ok(context.child_pid()),
4429                |_| Ok(()),
4430                |()| (42, ()),
4431            )
4432            .unwrap();
4433        assert_eq!(
4434            handle.finalize_with_status(),
4435            Ok((42, ExitStatus::Exited(0)))
4436        );
4437        assert_reaped(pid);
4438    }
4439
4440    #[test]
4441    fn startup_ignored_descriptor_overflow_still_refuses_workload() {
4442        if crate::test_runs_in_own_process() {
4443            return;
4444        }
4445        let (mapping, shared) = new_shared_drop_state();
4446        let result = Container::new().run_with_startup(
4447            Duration::from_secs(2),
4448            |_| Ok(()),
4449            |context| {
4450                for _ in 0..=MAX_STARTUP_FDS {
4451                    let _ = context.transfer_fd(std::fs::File::open("/dev/null").unwrap().into());
4452                }
4453                Ok(())
4454            },
4455            |()| {
4456                unsafe { &*shared }.started.store(true, Ordering::Release);
4457                ((), ())
4458            },
4459        );
4460        assert!(matches!(
4461            result,
4462            Err(StartupRunError::Child {
4463                cause: StartupError::Protocol,
4464                ..
4465            })
4466        ));
4467        assert!(!unsafe { &*shared }.started.load(Ordering::Acquire));
4468        unsafe { unmap_shared_drop_state(mapping, shared) };
4469    }
4470
4471    fn assert_reaped(pid: Pid) {
4472        let mut status = 0;
4473        assert_eq!(
4474            unsafe { libc::waitpid(pid.as_raw(), &mut status, libc::WNOHANG) },
4475            -1
4476        );
4477        assert_eq!(Errno::last(), Errno::ECHILD);
4478    }
4479
4480    #[test]
4481    fn startup_binds_parent_child_and_transfers_owned_descriptor_once() {
4482        if crate::test_runs_in_own_process() {
4483            return;
4484        }
4485        use std::os::fd::AsFd;
4486        let parent = Pid::this();
4487        let (mapping, shared) = new_shared_drop_state();
4488        let (owner, handle) = Container::new()
4489            .run_with_startup(
4490                Duration::from_secs(2),
4491                |mut context| {
4492                    assert_eq!(Pid::this(), parent);
4493                    assert_ne!(context.child_pid(), parent);
4494                    assert!(context.child_pidfd().as_raw_fd() >= 0);
4495                    assert_eq!(context.descriptor_count(), 1);
4496                    let fd = context.take_fd(0).unwrap();
4497                    assert!(context.take_fd(0).is_none());
4498                    assert!(context.take_fd(MAX_STARTUP_FDS).is_none());
4499                    assert_ne!(
4500                        unsafe { libc::fcntl(fd.as_raw_fd(), libc::F_GETFD) } & libc::FD_CLOEXEC,
4501                        0
4502                    );
4503                    let mut file = std::fs::File::from(fd);
4504                    let mut contents = String::new();
4505                    file.read_to_string(&mut contents).unwrap();
4506                    assert!(contents.starts_with(&format!("{} ", context.child_pid())));
4507                    unsafe { &*shared }.release.store(true, Ordering::Release);
4508                    Ok(context.child_pid())
4509                },
4510                |context| {
4511                    let file = std::fs::File::open("/proc/self/stat").unwrap();
4512                    context.transfer_fd(file.as_fd().try_clone_to_owned().unwrap())?;
4513                    Ok(Pid::this())
4514                },
4515                |pid| {
4516                    assert!(unsafe { &*shared }.release.load(Ordering::Acquire));
4517                    assert_eq!(pid, Pid::this());
4518                    assert_eq!(Pid::parent(), parent);
4519                    (pid, ())
4520                },
4521            )
4522            .unwrap();
4523        assert_eq!(
4524            handle.finalize_with_status(),
4525            Ok((owner, ExitStatus::Exited(0)))
4526        );
4527        assert_reaped(owner);
4528        unsafe { unmap_shared_drop_state(mapping, shared) };
4529    }
4530
4531    fn namespace_population_probe() -> (i32, i32, i32) {
4532        let root = Pid::this().as_raw();
4533        let child = unsafe { libc::fork() };
4534        assert!(child >= 0);
4535        if child == 0 {
4536            unsafe { libc::_exit(0) }
4537        }
4538        let mut status = 0;
4539        assert_eq!(unsafe { libc::waitpid(child, &mut status, 0) }, child);
4540        assert_eq!(ExitStatus::from_raw(status), ExitStatus::Exited(0));
4541        let thread = std::thread::spawn(|| unsafe { libc::syscall(libc::SYS_gettid) as i32 })
4542            .join()
4543            .unwrap();
4544        (root, child, thread)
4545    }
4546
4547    #[test]
4548    fn startup_does_not_allocate_a_child_namespace_helper_pid() {
4549        if crate::test_runs_in_own_process() {
4550            return;
4551        }
4552        let baseline = Container::new()
4553            .unshare(Namespace::USER | Namespace::PID)
4554            .run(namespace_population_probe)
4555            .unwrap();
4556        assert_eq!(baseline, (1, 2, 3));
4557        let parent = Pid::this();
4558        let ((), handle) = Container::new()
4559            .unshare(Namespace::USER | Namespace::PID)
4560            .run_with_startup(
4561                Duration::from_secs(2),
4562                |context| {
4563                    assert_eq!(Pid::this(), parent);
4564                    assert_ne!(context.child_pid(), Pid::from_raw(1));
4565                    Ok(())
4566                },
4567                |_| {
4568                    assert_eq!(Pid::this().as_raw(), 1);
4569                    Ok(())
4570                },
4571                |()| (namespace_population_probe(), ()),
4572            )
4573            .unwrap();
4574        assert_eq!(
4575            handle.finalize_with_status(),
4576            Ok((baseline, ExitStatus::Exited(0)))
4577        );
4578        // Opposing control: a genuine in-child helper consumes a PID and must
4579        // change this exact observation. Do not normalize namespace identities.
4580        let wrong = Container::new()
4581            .unshare(Namespace::USER | Namespace::PID)
4582            .run(|| {
4583                std::thread::spawn(|| ()).join().unwrap();
4584                namespace_population_probe()
4585            })
4586            .unwrap();
4587        assert_eq!(wrong, (1, 3, 4));
4588        assert_ne!(wrong, baseline);
4589    }
4590
4591    #[test]
4592    fn startup_precedes_seccomp_without_widening_the_filter() {
4593        if crate::test_runs_in_own_process() {
4594            return;
4595        }
4596        use syscalls::Sysno;
4597
4598        use super::seccomp::Action;
4599        use super::seccomp::FilterBuilder;
4600        let filter = || {
4601            FilterBuilder::new()
4602                .default_action(Action::Allow)
4603                .syscalls([
4604                    (Sysno::sendmsg, Action::Errno(Errno::EPERM)),
4605                    (Sysno::recvmsg, Action::Errno(Errno::EPERM)),
4606                    #[cfg(target_arch = "x86_64")]
4607                    (Sysno::poll, Action::Errno(Errno::EPERM)),
4608                    // Without a poll syscall, libc::poll uses ppoll.
4609                    #[cfg(not(target_arch = "x86_64"))]
4610                    (Sysno::ppoll, Action::Errno(Errno::EPERM)),
4611                    (Sysno::getppid, Action::Errno(Errno::EPERM)),
4612                ])
4613                .build()
4614        };
4615        let denied = || Errno::result(unsafe { libc::syscall(libc::SYS_getppid) });
4616        assert_eq!(
4617            Container::new().seccomp(filter()).run(denied),
4618            Ok(Err(Errno::EPERM))
4619        );
4620        let ((), handle) = Container::new()
4621            .seccomp(filter())
4622            .run_with_startup(
4623                Duration::from_secs(2),
4624                |_| Ok(()),
4625                |_| Ok(()),
4626                |()| (denied(), ()),
4627            )
4628            .unwrap();
4629        assert_eq!(
4630            handle.finalize_with_status(),
4631            Ok((Err(Errno::EPERM), ExitStatus::Exited(0)))
4632        );
4633    }
4634
4635    #[test]
4636    fn startup_drains_large_result_before_deferred_cleanup_and_actual_wait() {
4637        if crate::test_runs_in_own_process() {
4638            return;
4639        }
4640        let (mapping, shared) = new_shared_drop_state();
4641        let (pid, handle) = Container::new()
4642            .run_with_startup(
4643                Duration::from_secs(2),
4644                |context| Ok(context.child_pid()),
4645                |_| Ok(()),
4646                |()| (vec![42u8; 10 * 1024 * 1024], BlockingDrop { shared }),
4647            )
4648            .unwrap();
4649        assert_eq!(handle.provisional(), &vec![42u8; 10 * 1024 * 1024]);
4650        let shared_ref = unsafe { &*shared };
4651        while !shared_ref.started.load(Ordering::Acquire) {
4652            unsafe { libc::sched_yield() };
4653        }
4654        assert!(!shared_ref.finished.load(Ordering::Acquire));
4655        shared_ref.release.store(true, Ordering::Release);
4656        let (value, status) = handle.finalize_with_status().unwrap();
4657        assert_eq!(value, vec![42u8; 10 * 1024 * 1024]);
4658        assert_eq!(status, ExitStatus::Exited(0));
4659        assert!(shared_ref.finished.load(Ordering::Acquire));
4660        assert_reaped(pid);
4661        unsafe { unmap_shared_drop_state(mapping, shared) };
4662    }
4663
4664    #[test]
4665    fn startup_keeps_cleanup_failure_and_drop_reap_semantics() {
4666        if crate::test_runs_in_own_process() {
4667            return;
4668        }
4669        let (pid, handle) = Container::new()
4670            .run_with_startup(
4671                Duration::from_secs(2),
4672                |context| Ok(context.child_pid()),
4673                |_| Ok(()),
4674                |()| (42, ExitDuringDrop(71)),
4675            )
4676            .unwrap();
4677        assert_eq!(handle.provisional(), &42);
4678        assert_eq!(
4679            handle.finalize_with_status(),
4680            Err(RunError::ExitStatus(ExitStatus::Exited(71)))
4681        );
4682        assert_reaped(pid);
4683        let (pid, handle) = Container::new()
4684            .run_with_startup(
4685                Duration::from_secs(2),
4686                |context| Ok(context.child_pid()),
4687                |_| Ok(()),
4688                |()| (42, ()),
4689            )
4690            .unwrap();
4691        drop(handle);
4692        assert_reaped(pid);
4693    }
4694
4695    #[test]
4696    fn startup_parent_refusal_cancels_owned_child_without_running_workload() {
4697        if crate::test_runs_in_own_process() {
4698            return;
4699        }
4700        let (mapping, shared) = new_shared_drop_state();
4701        let mut pid = None;
4702        let result = Container::new().run_with_startup(
4703            Duration::from_secs(2),
4704            |context| {
4705                pid = Some(context.child_pid());
4706                Err::<(), _>(StartupError::Refused)
4707            },
4708            |_| Ok(()),
4709            |()| {
4710                unsafe { &*shared }.started.store(true, Ordering::Release);
4711                ((), ())
4712            },
4713        );
4714        assert!(matches!(
4715            result,
4716            Err(StartupRunError::Child {
4717                cause: StartupError::Refused,
4718                status: ExitStatus::Signaled(Signal::SIGKILL, false)
4719            })
4720        ));
4721        assert!(!unsafe { &*shared }.started.load(Ordering::Acquire));
4722        assert_reaped(pid.unwrap());
4723        unsafe { unmap_shared_drop_state(mapping, shared) };
4724    }
4725
4726    #[test]
4727    fn startup_child_setup_and_callback_refusals_are_not_readiness() {
4728        if crate::test_runs_in_own_process() {
4729            return;
4730        }
4731        let temp = tempfile::tempdir().unwrap();
4732        let missing = temp.path().join("absent");
4733        let result = Container::new().current_dir(missing).run_with_startup(
4734            Duration::from_secs(2),
4735            |_| -> Result<(), StartupError> {
4736                panic!("parent hook must not see failed child setup")
4737            },
4738            |_| -> Result<(), StartupError> { panic!("child hook must not see failed setup") },
4739            |()| -> ((), ()) { panic!("workload must not run") },
4740        );
4741        assert!(matches!(
4742            result,
4743            Err(StartupRunError::Child {
4744                cause: StartupError::Setup(Error { .. }),
4745                ..
4746            })
4747        ));
4748        if let Err(StartupRunError::Child {
4749            cause: StartupError::Setup(error),
4750            ..
4751        }) = result
4752        {
4753            assert_eq!(error, Error::new(Errno::ENOENT, Context::Chdir));
4754        }
4755        let result = Container::new().run_with_startup(
4756            Duration::from_secs(2),
4757            |_| -> Result<(), StartupError> {
4758                panic!("parent hook must not see refused child setup")
4759            },
4760            |_| Err::<(), _>(StartupError::Refused),
4761            |()| ((), ()),
4762        );
4763        assert!(matches!(
4764            result,
4765            Err(StartupRunError::Child {
4766                cause: StartupError::Refused,
4767                ..
4768            })
4769        ));
4770    }
4771
4772    #[test]
4773    fn startup_premature_child_exit_retains_actual_status() {
4774        if crate::test_runs_in_own_process() {
4775            return;
4776        }
4777        let result = Container::new().run_with_startup(
4778            Duration::from_secs(2),
4779            |_| Ok(()),
4780            |_| -> Result<(), StartupError> { unsafe { libc::_exit(73) } },
4781            |()| ((), ()),
4782        );
4783        assert!(matches!(
4784            result,
4785            Err(StartupRunError::Child {
4786                cause: StartupError::PeerClosed,
4787                status: ExitStatus::Exited(73)
4788            })
4789        ));
4790    }
4791
4792    #[test]
4793    fn startup_deadline_kills_a_child_stuck_before_readiness() {
4794        if crate::test_runs_in_own_process() {
4795            return;
4796        }
4797        let result = Container::new().run_with_startup(
4798            Duration::from_millis(100),
4799            |_| Ok(()),
4800            |_| -> Result<(), StartupError> {
4801                loop {
4802                    unsafe { libc::pause() };
4803                }
4804            },
4805            |()| ((), ()),
4806        );
4807        assert!(matches!(
4808            result,
4809            Err(StartupRunError::Child {
4810                cause: StartupError::TimedOut,
4811                status: ExitStatus::Signaled(Signal::SIGKILL, false)
4812            })
4813        ));
4814    }
4815
4816    #[test]
4817    fn startup_late_parent_callback_cannot_release_workload() {
4818        if crate::test_runs_in_own_process() {
4819            return;
4820        }
4821        let (mapping, shared) = new_shared_drop_state();
4822        let mut pid = None;
4823        let result = Container::new().run_with_startup(
4824            Duration::from_millis(100),
4825            |context| {
4826                pid = Some(context.child_pid());
4827                std::thread::sleep(Duration::from_millis(200));
4828                Ok(())
4829            },
4830            |_| Ok(()),
4831            |()| {
4832                unsafe { &*shared }.started.store(true, Ordering::Release);
4833                ((), ())
4834            },
4835        );
4836        assert!(matches!(
4837            result,
4838            Err(StartupRunError::Child {
4839                cause: StartupError::TimedOut,
4840                ..
4841            })
4842        ));
4843        assert!(!unsafe { &*shared }.started.load(Ordering::Acquire));
4844        assert_reaped(pid.unwrap());
4845        unsafe { unmap_shared_drop_state(mapping, shared) };
4846    }
4847
4848    #[test]
4849    fn startup_invalid_timeout_refuses_before_clone_or_callbacks() {
4850        if crate::test_runs_in_own_process() {
4851            return;
4852        }
4853        for timeout in [Duration::ZERO, Duration::MAX] {
4854            let result = Container::new().run_with_startup(
4855                timeout,
4856                |_| -> Result<(), StartupError> { panic!("no parent callback") },
4857                |_| -> Result<(), StartupError> { panic!("no child callback") },
4858                |()| ((), ()),
4859            );
4860            assert!(matches!(
4861                result,
4862                Err(StartupRunError::BeforeClone(StartupError::InvalidTimeout))
4863            ));
4864        }
4865    }
4866
4867    #[test]
4868    fn startup_protocol_rejects_malformed_wrong_phase_and_trailing_frames() {
4869        if crate::test_runs_in_own_process() {
4870            return;
4871        }
4872        let good = {
4873            let mut frame = [0u8; STARTUP_FRAME_SIZE];
4874            frame[..4].copy_from_slice(b"RVS1");
4875            frame[4] = STARTUP_READY;
4876            frame
4877        };
4878        let mut cases = vec![
4879            vec![0],
4880            good[..STARTUP_FRAME_SIZE - 1].to_vec(),
4881            [good.as_slice(), &[0]].concat(),
4882        ];
4883        let mut wrong = good;
4884        wrong[4] = 3;
4885        cases.push(wrong.to_vec());
4886        let mut wrong = good;
4887        wrong[5] = 1;
4888        cases.push(wrong.to_vec());
4889        let mut wrong = good;
4890        wrong[7] = 1;
4891        cases.push(wrong.to_vec());
4892        for frame in cases {
4893            let (a, b) = StartupSocket::pair(Instant::now() + Duration::from_secs(2)).unwrap();
4894            assert_eq!(
4895                unsafe {
4896                    libc::send(
4897                        a.fd.as_raw_fd(),
4898                        frame.as_ptr().cast(),
4899                        frame.len(),
4900                        libc::MSG_NOSIGNAL,
4901                    )
4902                },
4903                frame.len() as isize
4904            );
4905            a.close_write().unwrap();
4906            let result = b.receive(Some(STARTUP_READY)).and_then(|_| b.receive(None));
4907            assert!(matches!(result, Err(StartupError::Protocol)));
4908        }
4909        let (a, b) = StartupSocket::pair(Instant::now() + Duration::from_secs(2)).unwrap();
4910        a.send(STARTUP_READY, &StartupFds::default(), None).unwrap();
4911        a.send(STARTUP_READY, &StartupFds::default(), None).unwrap();
4912        a.close_write().unwrap();
4913        b.receive(Some(STARTUP_READY)).unwrap();
4914        assert!(matches!(b.receive(None), Err(StartupError::Protocol)));
4915        let (a, b) = StartupSocket::pair(Instant::now() + Duration::from_secs(2)).unwrap();
4916        drop(a);
4917        assert!(matches!(
4918            b.receive(Some(STARTUP_READY)),
4919            Err(StartupError::PeerClosed)
4920        ));
4921    }
4922
4923    #[test]
4924    fn startup_descriptor_cardinality_is_finite_and_refusal_closes_rights() {
4925        if crate::test_runs_in_own_process() {
4926            return;
4927        }
4928        let mut context = ChildStartContext {
4929            deadline: Instant::now() + Duration::from_secs(2),
4930            descriptors: StartupFds::default(),
4931            failure: None,
4932        };
4933        for _ in 0..MAX_STARTUP_FDS {
4934            context
4935                .transfer_fd(std::fs::File::open("/dev/null").unwrap().into())
4936                .unwrap();
4937        }
4938        let extra: std::os::fd::OwnedFd = std::fs::File::open("/dev/null").unwrap().into();
4939        let extra_number = extra.as_raw_fd();
4940        assert_eq!(context.transfer_fd(extra), Err(StartupError::Protocol));
4941        assert_eq!(unsafe { libc::fcntl(extra_number, libc::F_GETFD) }, -1);
4942        assert_eq!(Errno::last(), Errno::EBADF);
4943        let (a, b) = StartupSocket::pair(context.deadline).unwrap();
4944        a.send(STARTUP_REQUEST, &context.descriptors, None).unwrap();
4945        let received = b.receive(Some(STARTUP_REQUEST)).unwrap();
4946        assert_eq!(received.len, MAX_STARTUP_FDS);
4947        let numbers: Vec<_> = received
4948            .values
4949            .iter()
4950            .map(|fd| fd.as_ref().unwrap().as_raw_fd())
4951            .collect();
4952        drop(received);
4953        for fd in numbers {
4954            assert_eq!(unsafe { libc::fcntl(fd, libc::F_GETFD) }, -1);
4955            assert_eq!(Errno::last(), Errno::EBADF);
4956        }
4957    }
4958
4959    #[test]
4960    fn can_panic() {
4961        if crate::test_runs_in_own_process() {
4962            return;
4963        }
4964        let result = Container::new().run::<_, ()>(|| panic!());
4965        assert!(
4966            matches!(
4967                result,
4968                Err(RunError::ExitStatus(ExitStatus::Signaled(
4969                    Signal::SIGABRT,
4970                    _
4971                )))
4972            ),
4973            "Expected Err(ExitStatus(Signaled(SIGABRT, _))), got {:?}",
4974            result
4975        );
4976    }
4977
4978    /// A `Container` child copies the descriptor table of the process that
4979    /// runs the test, and without an `execve` its `O_CLOEXEC` descriptors
4980    /// survive. In the libtest harness process that table holds the pipes and
4981    /// pidfds of tests running on other threads. The memfd opened here, only
4982    /// in the harness process, stands in for such a descriptor. The child must
4983    /// not hold it, which is true only when the test body runs in a process of
4984    /// its own.
4985    #[test]
4986    fn run_child_holds_no_descriptor_of_another_test() {
4987        use std::os::fd::FromRawFd;
4988        const NAME: &std::ffi::CStr = c"reverie-process-another-tests-descriptor";
4989        let _another_tests_descriptor = std::env::var_os(crate::ISOLATED_TEST_MARKER)
4990            .is_none()
4991            .then(|| {
4992                // SAFETY: NAME is NUL-terminated and the flags are valid.
4993                let fd = unsafe { libc::memfd_create(NAME.as_ptr(), libc::MFD_CLOEXEC) };
4994                assert!(fd >= 0, "memfd_create: {}", std::io::Error::last_os_error());
4995                // SAFETY: memfd_create just returned this descriptor to us alone.
4996                unsafe { std::os::fd::OwnedFd::from_raw_fd(fd) }
4997            });
4998        if crate::test_runs_in_own_process() {
4999            return;
5000        }
5001        let held = Container::new()
5002            .run(|| {
5003                let name = NAME.to_str().unwrap();
5004                std::fs::read_dir("/proc/self/fd")
5005                    .unwrap()
5006                    .filter_map(Result::ok)
5007                    .filter_map(|e| std::fs::read_link(e.path()).ok())
5008                    .map(|p| p.to_string_lossy().into_owned())
5009                    .filter(|p| p.contains(name))
5010                    .collect::<Vec<String>>()
5011            })
5012            .unwrap();
5013        assert_eq!(
5014            held,
5015            Vec::<String>::new(),
5016            "the child holds a descriptor that another test opened"
5017        );
5018    }
5019
5020    #[test]
5021    fn is_new_process() {
5022        if crate::test_runs_in_own_process() {
5023            return;
5024        }
5025        let my_pid = unsafe { libc::getpid() };
5026
5027        assert_eq!(
5028            Container::new().run(|| {
5029                assert_ne!(unsafe { libc::getpid() }, 1);
5030                assert_ne!(unsafe { libc::getpid() }, my_pid);
5031                assert_eq!(unsafe { libc::getppid() }, my_pid);
5032            }),
5033            Ok(())
5034        );
5035    }
5036
5037    #[test]
5038    fn pid_namespace() {
5039        if crate::test_runs_in_own_process() {
5040            return;
5041        }
5042        assert_eq!(
5043            Container::new()
5044                .unshare(Namespace::USER | Namespace::PID)
5045                .run(|| {
5046                    // New PID namespace, so this should be the init process.
5047                    assert_eq!(unsafe { libc::getpid() }, 1);
5048                }),
5049            Ok(())
5050        );
5051    }
5052
5053    #[test]
5054    fn return_value() {
5055        if crate::test_runs_in_own_process() {
5056            return;
5057        }
5058        assert_eq!(Container::new().run(|| 42), Ok(42));
5059
5060        assert_eq!(
5061            Container::new().run(|| String::from("foobar")),
5062            Ok("foobar".into())
5063        );
5064    }
5065
5066    struct BlockingDrop {
5067        shared: *mut SharedDropState,
5068    }
5069
5070    impl Drop for BlockingDrop {
5071        fn drop(&mut self) {
5072            let shared = unsafe { &*self.shared };
5073            shared.started.store(true, Ordering::Release);
5074            while !shared.release.load(Ordering::Acquire) {
5075                unsafe { libc::sched_yield() };
5076            }
5077            shared.finished.store(true, Ordering::Release);
5078        }
5079    }
5080
5081    struct SharedDropState {
5082        started: AtomicBool,
5083        release: AtomicBool,
5084        finished: AtomicBool,
5085    }
5086
5087    fn new_shared_drop_state() -> (*mut libc::c_void, *mut SharedDropState) {
5088        let mapping = unsafe {
5089            libc::mmap(
5090                std::ptr::null_mut(),
5091                std::mem::size_of::<SharedDropState>(),
5092                libc::PROT_READ | libc::PROT_WRITE,
5093                libc::MAP_SHARED | libc::MAP_ANONYMOUS,
5094                -1,
5095                0,
5096            )
5097        };
5098        assert_ne!(mapping, libc::MAP_FAILED);
5099        let shared = mapping.cast::<SharedDropState>();
5100        unsafe {
5101            shared.write(SharedDropState {
5102                started: AtomicBool::new(false),
5103                release: AtomicBool::new(false),
5104                finished: AtomicBool::new(false),
5105            });
5106        }
5107        (mapping, shared)
5108    }
5109
5110    unsafe fn unmap_shared_drop_state(mapping: *mut libc::c_void, shared: *mut SharedDropState) {
5111        unsafe {
5112            std::ptr::drop_in_place(shared);
5113            assert_eq!(
5114                libc::munmap(mapping, std::mem::size_of::<SharedDropState>()),
5115                0
5116            );
5117        }
5118    }
5119
5120    #[test]
5121    fn deferred_drop_publishes_result_before_cleanup_completes() {
5122        if crate::test_runs_in_own_process() {
5123            return;
5124        }
5125        let (mapping, shared) = new_shared_drop_state();
5126
5127        let run = Container::new()
5128            .run_with_deferred_drop(|| (42, BlockingDrop { shared }))
5129            .unwrap();
5130        assert_eq!(run.provisional(), &42);
5131
5132        let shared_ref = unsafe { &*shared };
5133        while !shared_ref.started.load(Ordering::Acquire) {
5134            unsafe { libc::sched_yield() };
5135        }
5136        shared_ref.release.store(true, Ordering::Release);
5137        assert_eq!(run.finalize(), Ok(42));
5138        assert!(shared_ref.finished.load(Ordering::Acquire));
5139
5140        unsafe { unmap_shared_drop_state(mapping, shared) };
5141    }
5142
5143    #[test]
5144    fn dropping_cleanup_handle_still_reaps_the_child() {
5145        if crate::test_runs_in_own_process() {
5146            return;
5147        }
5148        let (mapping, shared) = new_shared_drop_state();
5149        let run = Container::new()
5150            .run_with_deferred_drop(|| ((), BlockingDrop { shared }))
5151            .unwrap();
5152        let shared_ref = unsafe { &*shared };
5153        while !shared_ref.started.load(Ordering::Acquire) {
5154            unsafe { libc::sched_yield() };
5155        }
5156        shared_ref.release.store(true, Ordering::Release);
5157        drop(run);
5158        assert!(shared_ref.finished.load(Ordering::Acquire));
5159
5160        unsafe { unmap_shared_drop_state(mapping, shared) };
5161    }
5162
5163    static WAITPID_SIGNAL_TEST: Mutex<()> = Mutex::new(());
5164
5165    extern "C" fn ignore_test_signal(_signal: libc::c_int) {}
5166
5167    #[test]
5168    fn deferred_finalize_retries_an_interrupted_wait_and_reaps() {
5169        if crate::test_runs_in_own_process() {
5170            return;
5171        }
5172        let _serial = WAITPID_SIGNAL_TEST.lock().unwrap();
5173        WAITPID_ENTERED.store(false, Ordering::Release);
5174        WAITPID_INTERRUPTED.store(0, Ordering::Release);
5175
5176        let action = SigAction::new(
5177            SigHandler::Handler(ignore_test_signal),
5178            SaFlags::empty(),
5179            SigSet::empty(),
5180        );
5181        let previous = unsafe { sigaction(Signal::SIGUSR2, &action) }.unwrap();
5182
5183        let (mapping, shared) = new_shared_drop_state();
5184        let run = Container::new()
5185            .run_with_deferred_drop(|| (42, BlockingDrop { shared }))
5186            .unwrap();
5187        let child_pid = run.child.0.expect("deferred child pid");
5188        WAITPID_TEST_PID.store(child_pid.as_raw(), Ordering::Release);
5189        let waiting_thread = unsafe { libc::pthread_self() };
5190        let shared_address = shared as usize;
5191        let interrupter = std::thread::spawn(move || {
5192            while !WAITPID_ENTERED.load(Ordering::Acquire) {
5193                std::thread::yield_now();
5194            }
5195            let deadline = Instant::now() + Duration::from_secs(2);
5196            while WAITPID_INTERRUPTED.load(Ordering::Acquire) == 0 && Instant::now() < deadline {
5197                assert_eq!(
5198                    unsafe { libc::pthread_kill(waiting_thread, libc::SIGUSR2) },
5199                    0
5200                );
5201                std::thread::sleep(Duration::from_millis(1));
5202            }
5203            let shared = unsafe { &*(shared_address as *mut SharedDropState) };
5204            shared.release.store(true, Ordering::Release);
5205        });
5206
5207        assert_eq!(run.finalize(), Ok(42));
5208        interrupter.join().unwrap();
5209        unsafe { sigaction(Signal::SIGUSR2, &previous) }.unwrap();
5210        WAITPID_TEST_PID.store(0, Ordering::Release);
5211        assert!(
5212            WAITPID_INTERRUPTED.load(Ordering::Acquire) > 0,
5213            "the signal did not interrupt waitpid"
5214        );
5215        let mut status = 0;
5216        assert_eq!(
5217            unsafe { libc::waitpid(child_pid.as_raw(), &mut status, libc::WNOHANG) },
5218            -1
5219        );
5220        assert_eq!(Errno::last(), Errno::ECHILD);
5221
5222        unsafe { unmap_shared_drop_state(mapping, shared) };
5223    }
5224
5225    struct ExitDuringDrop(i32);
5226
5227    impl Drop for ExitDuringDrop {
5228        fn drop(&mut self) {
5229            unsafe { libc::_exit(self.0) }
5230        }
5231    }
5232
5233    #[test]
5234    fn deferred_drop_exposes_cleanup_failure() {
5235        if crate::test_runs_in_own_process() {
5236            return;
5237        }
5238        let run = Container::new()
5239            .run_with_deferred_drop(|| (42, ExitDuringDrop(71)))
5240            .unwrap();
5241
5242        assert_eq!(run.provisional(), &42);
5243        assert_eq!(
5244            run.finalize(),
5245            Err(RunError::ExitStatus(ExitStatus::Exited(71)))
5246        );
5247    }
5248
5249    #[test]
5250    fn mount_error_from_child_is_returned() {
5251        if crate::test_runs_in_own_process() {
5252            return;
5253        }
5254        let source_dir = tempfile::tempdir().unwrap();
5255        let missing_source = source_dir.path().join("missing");
5256
5257        let result = Container::new()
5258            .unshare(Namespace::USER | Namespace::MOUNT)
5259            .map_root()
5260            .mount(Mount::bind(missing_source, "/test"))
5261            .run(|| 42);
5262
5263        assert_eq!(
5264            result,
5265            Err(RunError::Spawn(Error::new(Errno::ENOENT, Context::Mount)))
5266        );
5267    }
5268
5269    #[test]
5270    fn test_directory_is_available_after_mount() {
5271        if crate::test_runs_in_own_process() {
5272            return;
5273        }
5274        let result = Container::new()
5275            .unshare(Namespace::USER | Namespace::MOUNT)
5276            .map_root()
5277            .mount(Mount::tmpfs("/test").touch_target())
5278            .run(|| Path::new("/test").is_dir());
5279
5280        assert_eq!(result, Ok(true));
5281    }
5282
5283    /// A read-only bind of a source on a `nosuid`/`nodev` filesystem must work.
5284    ///
5285    /// ⚠️ REGRESSION TEST FOR A WHOLE LOST VALIDATE ARM, not a corner case.
5286    /// Inside a user namespace the kernel LOCKS the flags of mounts inherited
5287    /// from the parent namespace and refuses any remount that would clear one.
5288    /// A read-only bind is bind-then-remount, and the remount used to pass
5289    /// `MS_RDONLY` alone -- which asks to drop every other flag the source had.
5290    /// The mount returned EPERM and the container never spawned, so the guest
5291    /// did not fail, it never existed.
5292    ///
5293    /// Measured 2026-08-27: Hermit puts its frozen `/etc/group` and empty nscd
5294    /// directory in TMPDIR and binds each read-only, so a TMPDIR on
5295    /// `/run/user/<uid>` -- `nosuid,nodev` on any systemd host -- killed every
5296    /// container spawn. 610 of one arm's 612 e2e rows came from this one mount.
5297    ///
5298    /// ⚠️ THE SOURCE MUST BE MOUNTED BY THE HOST, NOT BY THIS CONTAINER. A tmpfs
5299    /// this test mounts itself lives in the container's own namespace, so its
5300    /// flags are NOT locked and the remount succeeds even unfixed -- a test that
5301    /// cannot fail. `/dev/shm` is host-mounted and carries both flags, so it
5302    /// reproduces the inheritance that makes them locked.
5303    #[test]
5304    fn a_readonly_bind_survives_a_nosuid_nodev_source() {
5305        if crate::test_runs_in_own_process() {
5306            return;
5307        }
5308        let shm = Path::new("/dev/shm");
5309        // Fail closed rather than silently stop exercising the condition.
5310        let flags = nix::sys::statvfs::statvfs(shm).expect("statvfs /dev/shm");
5311        assert!(
5312            flags
5313                .flags()
5314                .contains(nix::sys::statvfs::FsFlags::ST_NOSUID)
5315                || flags.flags().contains(nix::sys::statvfs::FsFlags::ST_NODEV),
5316            "/dev/shm carries neither nosuid nor nodev on this host, so this test \
5317             would pass without exercising the locked-flag remount at all"
5318        );
5319
5320        let source = tempfile::tempdir_in(shm).unwrap();
5321        let target = tempfile::tempdir().unwrap();
5322
5323        let result = Container::new()
5324            .unshare(Namespace::USER | Namespace::MOUNT)
5325            .map_root()
5326            .mount(Mount::bind(source.path(), target.path()).readonly())
5327            .run(|| Path::new("/proc/self/mounts").is_file());
5328
5329        assert_eq!(
5330            result,
5331            Ok(true),
5332            "a read-only bind whose source is nosuid/nodev must not fail; \
5333             EPERM here means the remount is dropping the source's locked flags"
5334        );
5335    }
5336
5337    #[test]
5338    fn huge_return_value() {
5339        if crate::test_runs_in_own_process() {
5340            return;
5341        }
5342        assert_eq!(
5343            Container::new().run(|| {
5344                // Need something larger than /proc/sys/fs/pipe-max-size, which
5345                // is typically 1MB.
5346                vec![42; 10 * 1024 * 1024 /* 10 MB */]
5347            }),
5348            Ok(vec![42; 10 * 1024 * 1024])
5349        );
5350    }
5351
5352    #[test]
5353    pub fn bind_to_low_port() {
5354        if crate::test_runs_in_own_process() {
5355            return;
5356        }
5357        use std::net::Ipv4Addr;
5358        use std::net::SocketAddrV4;
5359        use std::net::TcpListener;
5360
5361        let addr = Container::new()
5362            .map_root()
5363            .local_networking_only()
5364            .run(|| {
5365                let listener = TcpListener::bind("127.0.0.1:80").unwrap();
5366                listener.local_addr().unwrap()
5367            })
5368            .unwrap();
5369
5370        assert_eq!(
5371            addr,
5372            SocketAddrV4::new(Ipv4Addr::new(127, 0, 0, 1), 80).into()
5373        );
5374    }
5375
5376    /// Pinning each guest to a different CPU must put it on a different CPU.
5377    ///
5378    /// ⚠️ THE IDENTIFIER THIS READS IS 32-BIT ON PURPOSE, AND AN 8-BIT ONE MADE
5379    /// THIS TEST UNPASSABLE ON THIS HOST. It used to read
5380    /// `get_feature_info().initial_local_apic_id()`, the LEGACY CPUID leaf 1
5381    /// `EBX[31:24]` field, which is a `u8` and so has 256 possible values.
5382    /// Measured at reverie main `b181b1bba20c846d277f500c25182a00e18add9a` on a
5383    /// 316-CPU host:
5384    ///
5385    /// ```text
5386    /// 316 observations, min id 0, max id 255, 256 distinct ids,
5387    /// exactly 60 ids seen twice -- and 316 - 256 = 60.
5388    /// ```
5389    ///
5390    /// So `max(count) == 1` could not hold for ANY amount of correct pinning,
5391    /// and the failure was deterministic rather than flaky. It had been
5392    /// recorded several times as "pre-existing and host-specific" and routinely
5393    /// skipped, which was true and left `validate.sh` step 2 red on main.
5394    ///
5395    /// ⚠️ BUT THE 256 CEILING IS NOT WHERE IT BREAKS, AND ASSUMING SO WOULD
5396    /// LEAVE THE NEXT READER ON A 64-CORE BOX BELIEVING THEY ARE SAFE. The
5397    /// legacy field drops the HIGH TOPOLOGY BITS, so ids repeat across sockets
5398    /// long before the count reaches 256: on this host the first collision is
5399    /// core 32 against core 0, both reporting id 0. Measured by enumerating
5400    /// only the first N cores with each identifier:
5401    ///
5402    /// ```text
5403    /// cores    16   32   33   64  128  256  316
5404    /// 8-bit    ok   ok  FAIL FAIL FAIL FAIL FAIL
5405    /// 32-bit   ok   ok   ok   ok   ok   ok   ok
5406    /// ```
5407    ///
5408    /// So this is not a test that was always broken. It is a test whose hidden
5409    /// assumption -- that the enumerated CPUs have distinct LEGACY apic ids --
5410    /// held on the smaller machines it was written against and stopped holding
5411    /// here, at 33 cores rather than at 257.
5412    ///
5413    /// CPUID leaf `0x0B` reports the 32-bit x2APIC id, which distinguishes as
5414    /// many CPUs as the machine has. Reading it does not weaken the assertion --
5415    /// the assertion is unchanged, and it is now able to fail for the reason it
5416    /// was written to catch instead of failing for arithmetic. Verified in that
5417    /// direction too: with affinity broken outright the repaired test reports
5418    /// `left: 316, right: 1`, and with affinity broken for half the cores it
5419    /// also fails.
5420    #[cfg(target_arch = "x86_64")]
5421    #[test]
5422    pub fn pin_affinity_to_all_cores() -> Result<(), Error> {
5423        if crate::test_runs_in_own_process() {
5424            return Ok(());
5425        }
5426        use std::collections::HashMap;
5427
5428        use raw_cpuid::CpuId;
5429
5430        let cpus = num_cpus::get();
5431        println!("Total cpus {}", cpus);
5432
5433        // Map the x2APIC id to the number of times we observed it:
5434        let mut results: HashMap<u32, usize> = HashMap::new();
5435        for core in 0..cpus {
5436            println!("  Launching guest with affinity set to {}", core);
5437            let mut container = Container::new();
5438            container.affinity(core);
5439            let which_core = container
5440                .run(|| {
5441                    let cpuid = CpuId::new();
5442                    // Every level of leaf 0x0B reports the same x2APIC id for
5443                    // the executing logical processor, so the first is enough.
5444                    cpuid
5445                        .get_extended_topology_info()
5446                        .and_then(|mut levels| levels.next())
5447                        .map(|level| level.x2apic_id())
5448                })
5449                .unwrap();
5450            // ⚠️ REFUSE RATHER THAN FALL BACK TO THE 8-BIT FIELD. On a host
5451            // this large the legacy id provably cannot answer the question, so
5452            // silently using it would restore exactly the defect above.
5453            let which_core = which_core.unwrap_or_else(|| {
5454                panic!(
5455                    "CPUID leaf 0x0B (extended topology) is unavailable, so no \
5456                     32-bit x2APIC id can be read; with {cpus} CPUs the legacy \
5457                     8-bit APIC id cannot distinguish them and this test cannot \
5458                     decide anything"
5459                )
5460            });
5461            println!("    Guest sees its on x2APIC id {}", which_core);
5462            *results.entry(which_core).or_default() += 1;
5463        }
5464
5465        println!("Final table size {:?}", results.len());
5466        assert_eq!(
5467            results.values().fold(0, |n, v| std::cmp::max(n, *v)),
5468            1,
5469            "two guests pinned to different CPUs reported the same x2APIC id, \
5470             so affinity did not place them on distinct CPUs"
5471        );
5472        Ok(())
5473    }
5474}