1use std::borrow::Cow;
10use std::collections::BTreeMap;
11use std::ffi::CString;
12use std::ffi::OsStr;
13use std::ffi::OsString;
14use std::io::Read;
15use std::io::Write;
16#[cfg(test)]
17use std::os::fd::RawFd;
18use std::os::unix::ffi::OsStrExt;
19use std::os::unix::io::AsRawFd;
20use std::path::Path;
21#[cfg(test)]
22use std::sync::atomic::Ordering;
23
24use nix::sched::CpuSet;
25use nix::sched::sched_setaffinity;
26use serde::Serialize;
27use serde::de::DeserializeOwned;
28use syscalls::Errno;
29
30use super::clone::child_stack;
31use super::clone::clone_with_stack;
32use super::env::Env;
33use super::error::AddContext;
34use super::error::Context;
35use super::error::Error;
36use super::exit_status::ExitStatus;
37use super::fd::Fd;
38use super::fd::pipe;
39use super::fd::write_bytes;
40use super::id_map::make_id_map;
41use super::mount::Mount;
42use super::namespace::Namespace;
43use super::net::IfName;
44use super::pid::Pid;
45use super::pty::PtyChild;
46use super::seccomp;
47use super::stdio::Stdio;
48use super::util::reset_signal_handling;
49use super::util::to_cstring;
50
51pub struct Container {
56 pub(super) env: Env,
57 current_dir: Option<CString>,
58 chroot: Option<CString>,
59 pub(super) namespace: Namespace,
60 pub(super) stdin: Stdio,
61 pub(super) stdout: Stdio,
62 pub(super) stderr: Stdio,
63 pub(super) uid_map: Vec<(libc::uid_t, libc::uid_t, u32)>,
64 pub(super) gid_map: Vec<(libc::uid_t, libc::uid_t, u32)>,
65 mounts: Vec<Mount>,
66 local_networking_only: bool,
67 hostname: Option<OsString>,
68 domainname: Option<OsString>,
69 pub(super) seccomp: Option<seccomp::Filter>,
70 pub(super) seccomp_notify: bool,
71 pub(super) pty: Option<PtyChild>,
72 affinity: Option<usize>,
75}
76
77impl Default for Container {
78 fn default() -> Self {
79 Self {
80 env: Default::default(),
81 current_dir: None,
82 chroot: None,
83 namespace: Default::default(),
84 stdin: Stdio::inherit(),
85 stdout: Stdio::inherit(),
86 stderr: Stdio::inherit(),
87 uid_map: Vec::new(),
88 gid_map: Vec::new(),
89 mounts: Vec::new(),
90 local_networking_only: false,
91 hostname: None,
92 domainname: None,
93 seccomp: None,
94 seccomp_notify: false,
95 pty: None,
96 affinity: None,
97 }
98 }
99}
100
101impl Container {
102 pub(super) fn std_conversion_blockers(&self) -> Vec<&'static str> {
105 let Self {
108 env: _,
109 current_dir: _,
110 chroot,
111 namespace,
112 stdin: _,
113 stdout: _,
114 stderr: _,
115 uid_map,
116 gid_map,
117 mounts,
118 local_networking_only,
119 hostname,
120 domainname,
121 seccomp,
122 seccomp_notify,
123 pty,
124 affinity,
125 } = self;
126
127 let mut blockers = Vec::new();
128
129 if chroot.is_some() {
130 blockers.push("chroot");
131 }
132 if !namespace.is_empty() {
133 blockers.push("Linux namespaces");
134 }
135 if !uid_map.is_empty() {
136 blockers.push("user ID mappings");
137 }
138 if !gid_map.is_empty() {
139 blockers.push("group ID mappings");
140 }
141 if !mounts.is_empty() {
142 blockers.push("mounts");
143 }
144 if *local_networking_only {
145 blockers.push("local-only networking");
146 }
147 if hostname.is_some() {
148 blockers.push("hostname");
149 }
150 if domainname.is_some() {
151 blockers.push("domain name");
152 }
153 if seccomp.is_some() {
154 blockers.push("seccomp filter");
155 }
156 if *seccomp_notify {
157 blockers.push("seccomp notification");
158 }
159 if pty.is_some() {
160 blockers.push("pseudoterminal");
161 }
162 if affinity.is_some() {
163 blockers.push("CPU affinity");
164 }
165
166 blockers
167 }
168
169 pub fn new() -> Self {
172 Self::default()
173 }
174
175 pub fn env<K, V>(&mut self, key: K, val: V) -> &mut Self
190 where
191 K: AsRef<OsStr>,
192 V: AsRef<OsStr>,
193 {
194 self.env.set(key.as_ref(), val.as_ref());
195 self
196 }
197
198 pub fn envs<I, K, V>(&mut self, vars: I) -> &mut Self
222 where
223 I: IntoIterator<Item = (K, V)>,
224 K: AsRef<OsStr>,
225 V: AsRef<OsStr>,
226 {
227 for (k, v) in vars.into_iter() {
228 self.env(k, v);
229 }
230 self
231 }
232
233 pub fn env_remove<K: AsRef<OsStr>>(&mut self, key: K) -> &mut Self {
245 self.env.remove(key.as_ref());
246 self
247 }
248
249 pub fn env_clear(&mut self) -> &mut Self {
261 self.env.clear();
262 self
263 }
264
265 pub fn current_dir<P: AsRef<Path>>(&mut self, dir: P) -> &mut Self {
295 self.current_dir = Some(to_cstring(dir.as_ref()));
296 self
297 }
298
299 pub fn stdin<T: Into<Stdio>>(&mut self, cfg: T) -> &mut Self {
315 self.stdin = cfg.into();
316 self
317 }
318
319 pub fn stdout<T: Into<Stdio>>(&mut self, cfg: T) -> &mut Self {
336 self.stdout = cfg.into();
337 self
338 }
339
340 pub fn stderr<T: Into<Stdio>>(&mut self, cfg: T) -> &mut Self {
357 self.stderr = cfg.into();
358 self
359 }
360
361 pub fn chroot<P: AsRef<Path>>(&mut self, chroot: P) -> &mut Self {
368 self.chroot = Some(to_cstring(chroot.as_ref()));
369 self
370 }
371
372 pub fn unshare(&mut self, namespace: Namespace) -> &mut Self {
376 self.namespace |= namespace;
377 self
378 }
379
380 pub fn get_current_dir(&self) -> Option<&Path> {
384 if let Some(dir) = &self.current_dir {
385 Some(Path::new(OsStr::from_bytes(dir.to_bytes())))
386 } else {
387 None
388 }
389 }
390
391 pub fn get_envs(&self) -> impl Iterator<Item = (&OsStr, Option<&OsStr>)> {
395 self.env.iter()
396 }
397
398 pub fn get_captured_envs(&self) -> BTreeMap<OsString, OsString> {
401 self.env.capture()
402 }
403
404 pub fn get_env<K: AsRef<OsStr>>(&self, env: K) -> Option<Cow<'_, OsStr>> {
408 self.env.get_captured(env)
409 }
410
411 pub fn map_uid(&mut self, inside_uid: libc::uid_t, outside_uid: libc::uid_t) -> &mut Self {
434 self.map_uid_range(inside_uid, outside_uid, 1)
435 }
436
437 pub fn map_uid_range(
449 &mut self,
450 starting_inside_uid: libc::uid_t,
451 starting_outside_uid: libc::uid_t,
452 count: u32,
453 ) -> &mut Self {
454 self.uid_map
455 .push((starting_inside_uid, starting_outside_uid, count));
456 self.namespace |= Namespace::USER;
457 self
458 }
459
460 pub fn map_root(&mut self) -> &mut Self {
476 self.map_uid(0, unsafe { libc::geteuid() });
477 self.map_gid(0, unsafe { libc::getegid() })
478 }
479
480 pub fn map_gid(&mut self, inside_gid: libc::gid_t, outside_gid: libc::gid_t) -> &mut Self {
491 self.map_gid_range(inside_gid, outside_gid, 1)
492 }
493
494 pub fn map_gid_range(
506 &mut self,
507 starting_inside_gid: libc::gid_t,
508 starting_outside_gid: libc::gid_t,
509 count: u32,
510 ) -> &mut Self {
511 self.namespace |= Namespace::USER;
512 self.gid_map
513 .push((starting_inside_gid, starting_outside_gid, count));
514 self
515 }
516
517 pub fn hostname<S: Into<OsString>>(&mut self, hostname: S) -> &mut Self {
527 self.namespace |= Namespace::UTS;
528 self.hostname = Some(hostname.into());
529 self
530 }
531
532 pub fn domainname<S: Into<OsString>>(&mut self, domainname: S) -> &mut Self {
544 self.namespace |= Namespace::UTS;
545 self.domainname = Some(domainname.into());
546 self
547 }
548
549 pub fn get_hostname(&self) -> Option<&OsStr> {
551 self.hostname.as_ref().map(AsRef::as_ref)
552 }
553
554 pub fn get_domainname(&self) -> Option<&OsStr> {
556 self.domainname.as_ref().map(AsRef::as_ref)
557 }
558
559 pub fn mount(&mut self, mount: Mount) -> &mut Self {
566 self.namespace |= Namespace::MOUNT;
567 self.mounts.push(mount);
568 self
569 }
570
571 pub fn mounts<I>(&mut self, mounts: I) -> &mut Self
573 where
574 I: IntoIterator<Item = Mount>,
575 {
576 self.namespace |= Namespace::MOUNT;
577 self.mounts.extend(mounts);
578 self
579 }
580
581 pub fn local_networking_only(&mut self) -> &mut Self {
589 if !self.local_networking_only {
590 self.local_networking_only = true;
591 self.namespace |= Namespace::NETWORK;
592 self.mount(Mount::sysfs("/sys"));
593 }
594 self
595 }
596
597 pub fn seccomp(&mut self, filter: seccomp::Filter) -> &mut Self {
601 self.seccomp = Some(filter);
602 self
603 }
604
605 pub fn seccomp_notify(&mut self) -> &mut Self {
611 self.seccomp_notify = true;
612 self
613 }
614
615 pub fn pty(&mut self, child: PtyChild) -> &mut Self {
626 self.pty = Some(child);
627 self.stdin = Stdio::inherit();
628 self.stdout = Stdio::inherit();
629 self.stderr = Stdio::inherit();
630 self
631 }
632
633 pub fn affinity(&mut self, affinity: usize) -> &mut Self {
635 self.affinity = Some(affinity);
636 self
637 }
638
639 pub(super) fn setup(
647 &mut self,
648 context: &ChildContext,
649 pre_exec: &mut [Box<dyn FnMut() -> Result<(), Errno> + Send + Sync>],
650 ) -> Result<(), Error> {
651 self.setup_before_filter(context, pre_exec)?;
652 self.setup_filter(context)
653 }
654
655 fn setup_before_filter(
656 &mut self,
657 context: &ChildContext,
658 pre_exec: &mut [Box<dyn FnMut() -> Result<(), Errno> + Send + Sync>],
659 ) -> Result<(), Error> {
660 if let Some(pty) = self.pty.take() {
664 pty.login().context(Context::Tty)?;
668 }
669
670 if let Some(fd) = context.stdin {
671 fd.dup2(libc::STDIN_FILENO)
672 .context(Context::Stdio)?
673 .leave_open();
674 }
675 if let Some(fd) = context.stdout {
676 fd.dup2(libc::STDOUT_FILENO)
677 .context(Context::Stdio)?
678 .leave_open();
679 }
680 if let Some(fd) = context.stderr {
681 fd.dup2(libc::STDERR_FILENO)
682 .context(Context::Stdio)?
683 .leave_open();
684 }
685
686 unsafe { reset_signal_handling() }.context(Context::ResetSignals)?;
687
688 if !context.uid_map.is_empty() {
690 context.map_uid().context(Context::MapUid)?;
691 }
692
693 if !context.gid_map.is_empty() {
694 context.setgroups(false).context(Context::MapGid)?;
695 context.map_gid().context(Context::MapGid)?;
696 }
697
698 if let Some(name) = &self.hostname {
700 Error::result(
701 unsafe { libc::sethostname(name.as_bytes().as_ptr() as *const _, name.len()) },
702 Context::Hostname,
703 )?;
704 }
705
706 if let Some(name) = &self.domainname {
708 Error::result(
709 unsafe { libc::setdomainname(name.as_bytes().as_ptr() as *const _, name.len()) },
710 Context::Domainname,
711 )?;
712 }
713
714 for mount in &mut self.mounts {
716 mount.mount().context(Context::Mount)?;
717 }
718
719 if let Some(chroot) = &self.chroot {
723 Error::result(unsafe { libc::chroot(chroot.as_ptr()) }, Context::Chroot)?;
724 }
725
726 if let Some(current_dir) = &self.current_dir {
728 Error::result(unsafe { libc::chdir(current_dir.as_ptr()) }, Context::Chdir)?;
729 }
730
731 if self.local_networking_only {
734 let sock = Fd::socket(libc::AF_INET, libc::SOCK_DGRAM, libc::IPPROTO_IP)
736 .context(Context::Network)?;
737
738 let loopback = IfName::LOOPBACK;
739
740 let flags = loopback.get_flags(&sock).context(Context::Network)?;
742 let flags = flags | libc::IFF_UP as i16;
743 loopback.set_flags(&sock, flags).context(Context::Network)?;
744 }
745
746 if let Some(cpu) = self.affinity {
747 let mut cpu_set = CpuSet::new();
748 cpu_set.set(cpu).context(Context::Affinity)?;
749 sched_setaffinity(nix::unistd::Pid::from_raw(0), &cpu_set)
750 .context(Context::Affinity)?;
751 }
752
753 for f in pre_exec {
757 f().context(Context::PreExec)?;
758 }
759
760 Ok(())
761 }
762
763 fn setup_filter(&self, context: &ChildContext) -> Result<(), Error> {
764 if let Some(filter) = &self.seccomp {
766 use core::sync::atomic::Ordering;
767
768 Error::result(
770 unsafe { libc::prctl(libc::PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0) },
771 Context::Seccomp,
772 )?;
773
774 if let Some(shared_fd) = context.seccomp_fd {
786 use std::os::unix::io::IntoRawFd;
787
788 let fd = filter
789 .load_and_listen()
790 .context(Context::Seccomp)?
791 .into_raw_fd();
792
793 shared_fd.store(fd, Ordering::Relaxed);
794
795 while shared_fd.load(Ordering::Relaxed) == fd {
806 }
808 } else {
809 filter.load().context(Context::Seccomp)?;
810 }
811 }
812
813 Ok(())
814 }
815
816 pub fn run<F, T>(&mut self, mut f: F) -> Result<T, RunError>
828 where
829 F: FnMut() -> T,
830 T: Serialize + DeserializeOwned,
831 {
832 let clone_flags = self.namespace.bits() | libc::SIGCHLD;
833
834 let uid_map = &make_id_map(&self.uid_map);
835 let gid_map = &make_id_map(&self.gid_map);
836
837 let context = ChildContext {
838 stdin: None,
841 stdout: None,
842 stderr: None,
843 uid_map,
844 gid_map,
845 seccomp_fd: None,
846 };
847
848 let (mut reader, writer) = pipe()?;
851
852 let writer_fd = writer.as_raw_fd();
853
854 let mut stack = child_stack()?;
859
860 #[cfg(feature = "nightly")]
869 let output_capture = std::io::set_output_capture(None);
870
871 let result = clone_with_stack(
872 || {
873 let value = self.setup(&context, &mut []).map(|()| f());
874
875 let mut writer = std::io::BufWriter::new(Fd::new(writer_fd));
876
877 bincode::serde::encode_into_std_write(
882 &value,
883 &mut writer,
884 bincode::config::legacy(),
885 )
886 .expect("Failed to serialize return value");
887
888 0
889 },
890 clone_flags,
891 &mut stack,
892 );
893
894 #[cfg(feature = "nightly")]
895 std::io::set_output_capture(output_capture);
896
897 let child = WaitGuard::new(result?);
898
899 drop(writer);
902
903 let mut buf = Vec::new();
907 match reader.read_to_end(&mut buf) {
908 Ok(0) => {
909 Err(RunError::ExitStatus(child.wait()?))
922 }
923 Ok(n) => {
924 let value: Result<T, Error> =
925 bincode::serde::decode_from_slice(&buf[0..n], bincode::config::legacy())
926 .unwrap()
927 .0;
928 value.map_err(RunError::Spawn)
929 }
930 Err(err) => {
931 panic!("Got unexpected error: {}", err)
933 }
934 }
935 }
936
937 pub fn run_with_startup<P, C, F, O, S, T, D>(
965 &mut self,
966 timeout: std::time::Duration,
967 parent_start: P,
968 mut child_start: C,
969 mut run: F,
970 ) -> Result<(O, DeferredContainerRun<T>), StartupRunError>
971 where
972 P: FnOnce(ParentStartContext<'_>) -> Result<O, StartupError>,
973 C: FnMut(&mut ChildStartContext) -> Result<S, StartupError>,
974 F: FnMut(S) -> (T, D),
975 T: Serialize + DeserializeOwned,
976 {
977 let deadline = std::time::Instant::now()
978 .checked_add(timeout)
979 .filter(|_| !timeout.is_zero())
980 .ok_or(StartupRunError::BeforeClone(StartupError::InvalidTimeout))?;
981 let mut disposition: libc::sigaction = unsafe { std::mem::zeroed() };
982 Errno::result(unsafe {
983 libc::sigaction(libc::SIGCHLD, std::ptr::null(), &mut disposition)
984 })
985 .map_err(|error| StartupRunError::BeforeClone(error.into()))?;
986 if disposition.sa_sigaction == libc::SIG_IGN
987 || disposition.sa_flags & libc::SA_NOCLDWAIT != 0
988 {
989 return Err(StartupRunError::BeforeClone(StartupError::Io(
990 Errno::ECHILD,
991 )));
992 }
993 let (parent_socket, child_socket) =
994 StartupSocket::pair(deadline).map_err(StartupRunError::BeforeClone)?;
995 let uid_map = &make_id_map(&self.uid_map);
996 let gid_map = &make_id_map(&self.gid_map);
997 let context = ChildContext {
998 stdin: None,
999 stdout: None,
1000 stderr: None,
1001 uid_map,
1002 gid_map,
1003 seccomp_fd: None,
1004 };
1005 let (mut reader, writer) =
1006 pipe().map_err(|error| StartupRunError::BeforeClone(error.into()))?;
1007 let writer_fd = writer.as_raw_fd();
1008 let reader_fd = reader.as_raw_fd();
1009 let parent_fd = parent_socket.fd.as_raw_fd();
1010 let child_fd = child_socket.fd.as_raw_fd();
1011 let mut stack =
1012 child_stack().map_err(|error| StartupRunError::BeforeClone(error.into()))?;
1013 let clone_flags = self.namespace.bits() | libc::SIGCHLD;
1014 #[cfg(feature = "nightly")]
1015 let output_capture = std::io::set_output_capture(None);
1016 let result = clone_with_stack(
1017 || {
1018 self.startup_child_run(
1019 &context,
1020 StartupChildIo {
1021 parent_fd,
1022 child_fd,
1023 reader_fd,
1024 writer_fd,
1025 deadline,
1026 },
1027 &mut child_start,
1028 &mut run,
1029 )
1030 },
1031 clone_flags,
1032 &mut stack,
1033 );
1034 #[cfg(feature = "nightly")]
1035 std::io::set_output_capture(output_capture);
1036 let pid = result.map_err(|error| StartupRunError::BeforeClone(error.into()))?;
1037 let mut child = StartupChild {
1038 wait: Some(WaitGuard::new(pid)),
1039 pidfd: None,
1040 };
1041 drop(child_socket);
1042 drop(writer);
1043 child.pidfd = match Fd::pidfd_open(pid.as_raw(), 0) {
1044 Ok(fd) => Some(fd),
1045 Err(error) => return Err(child.fail(error.into())),
1046 };
1047 let descriptors =
1048 match parent_socket
1049 .receive(Some(STARTUP_REQUEST))
1050 .and_then(|descriptors| {
1051 parent_socket.receive(None)?;
1054 Ok(descriptors)
1055 }) {
1056 Ok(descriptors) => descriptors,
1057 Err(error) => return Err(child.fail(error)),
1058 };
1059 let owner = match parent_start(ParentStartContext {
1060 child_pid: child.wait.as_ref().unwrap().0.unwrap(),
1061 child_pidfd: unsafe {
1063 std::os::fd::BorrowedFd::borrow_raw(child.pidfd.as_ref().unwrap().as_raw_fd())
1064 },
1065 deadline,
1066 descriptors,
1067 }) {
1068 Ok(owner) => owner,
1069 Err(error) => return Err(child.fail(error)),
1070 };
1071 let ready = (|| {
1072 parent_socket.send(STARTUP_READY, &StartupFds::default(), None)?;
1073 parent_socket.close_write()?;
1074 #[cfg(test)]
1076 if STARTUP_TEST_FAULT.with(|fault| fault.get())
1077 == StartupTestFault::ObservePermissionLate
1078 {
1079 std::thread::sleep(std::time::Duration::from_millis(200));
1080 }
1081 Ok::<_, StartupError>(())
1082 })();
1083 if let Err(error) = ready {
1084 return Err(child.fail(error));
1086 }
1087 drop(parent_socket);
1088 let mut bytes = Vec::new();
1089 match reader.read_to_end(&mut bytes) {
1090 Ok(0) => return Err(child.fail(StartupError::MissingResult)),
1091 Ok(_) => (),
1092 Err(error) => {
1093 return Err(child.fail(StartupError::Io(Errno::new(
1094 error.raw_os_error().unwrap_or(libc::EIO),
1095 ))));
1096 }
1097 }
1098 let value = match bincode::serde::decode_from_slice::<Result<T, StartupError>, _>(
1099 &bytes,
1100 bincode::config::legacy(),
1101 ) {
1102 Ok((Ok(value), used)) if used == bytes.len() => value,
1103 Ok((Err(error), used)) if used == bytes.len() => return Err(child.fail(error)),
1104 _ => return Err(child.fail(StartupError::Protocol)),
1105 };
1106 Ok((
1107 owner,
1108 DeferredContainerRun {
1109 value: Some(value),
1110 child: child.into_wait(),
1111 },
1112 ))
1113 }
1114
1115 fn startup_child_run<C, F, S, T, U>(
1117 &mut self,
1118 context: &ChildContext<'_>,
1119 io: StartupChildIo,
1120 child_start: &mut C,
1121 run: &mut F,
1122 ) -> i32
1123 where
1124 C: FnMut(&mut ChildStartContext) -> Result<S, StartupError>,
1125 F: FnMut(S) -> (T, U),
1126 T: Serialize,
1127 {
1128 let StartupChildIo {
1129 parent_fd,
1130 child_fd,
1131 reader_fd,
1132 writer_fd,
1133 deadline,
1134 } = io;
1135 unsafe {
1138 libc::close(parent_fd);
1139 libc::close(reader_fd);
1140 }
1141 let socket = StartupSocket {
1142 fd: Fd::new(child_fd),
1143 deadline,
1144 };
1145 let startup = (|| {
1146 self.setup_before_filter(context, &mut [])
1147 .map_err(StartupError::Setup)?;
1148 let mut child_context = ChildStartContext {
1149 deadline,
1150 descriptors: StartupFds::default(),
1151 failure: None,
1152 };
1153 let state = child_start(&mut child_context)?;
1154 if let Some(error) = child_context.failure {
1155 return Err(error);
1156 }
1157 socket.send(STARTUP_REQUEST, &child_context.descriptors, None)?;
1158 drop(child_context);
1159 socket.close_write()?;
1160 socket.receive(Some(STARTUP_READY))?;
1161 socket.receive(None)?; Ok(state)
1164 })();
1165 let state = match startup {
1166 Ok(state) => state,
1167 Err(error) => {
1168 let _ = socket.send(STARTUP_FAILURE, &StartupFds::default(), Some(error));
1169 drop(socket);
1170 let mut writer = std::io::BufWriter::new(Fd::new(writer_fd));
1171 bincode::serde::encode_into_std_write(
1172 Err::<T, StartupError>(error),
1173 &mut writer,
1174 bincode::config::legacy(),
1175 )
1176 .expect("Failed to serialize startup refusal");
1177 writer.flush().expect("Failed to flush startup refusal");
1178 drop(writer);
1179 return 1;
1180 }
1181 };
1182 drop(socket);
1183 let (value, deferred) = match self.setup_filter(context) {
1184 Ok(()) => {
1185 let (value, deferred) = run(state);
1186 (Ok(value), Some(deferred))
1187 }
1188 Err(error) => (Err(StartupError::Setup(error)), None),
1189 };
1190 let mut writer = std::io::BufWriter::new(Fd::new(writer_fd));
1191 bincode::serde::encode_into_std_write(&value, &mut writer, bincode::config::legacy())
1192 .expect("Failed to serialize return value");
1193 writer.flush().expect("Failed to flush return value");
1194 drop(writer);
1195 drop(deferred);
1196 0
1197 }
1198
1199 pub fn run_with_startup_owned<P, C, F, S, T, U>(
1216 &mut self,
1217 timeout: std::time::Duration,
1218 parent_start: &mut P,
1219 child_start: &mut C,
1220 run: &mut F,
1221 ) -> Result<OwnedDeferredContainerRun<T>, StartupOwnedFailure<T>>
1222 where
1223 P: FnMut(ParentStartContext<'_>) -> Result<(), StartupError>,
1224 C: FnMut(&mut ChildStartContext) -> Result<S, StartupError>,
1225 F: FnMut(S) -> (T, U),
1226 T: Serialize,
1227 {
1228 use std::os::fd::AsFd;
1229 let before = |cause| StartupOwnedFailure::BeforeClone { cause };
1230 let deadline = std::time::Instant::now()
1231 .checked_add(timeout)
1232 .filter(|_| !timeout.is_zero())
1233 .ok_or_else(|| before(StartupError::InvalidTimeout))?;
1234 let mut disposition: libc::sigaction = unsafe { std::mem::zeroed() };
1235 Errno::result(unsafe {
1236 libc::sigaction(libc::SIGCHLD, std::ptr::null(), &mut disposition)
1237 })
1238 .map_err(|error| before(error.into()))?;
1239 if disposition.sa_sigaction == libc::SIG_IGN
1240 || disposition.sa_flags & libc::SA_NOCLDWAIT != 0
1241 {
1242 return Err(before(StartupError::Io(Errno::ECHILD)));
1243 }
1244 let (parent_socket, child_socket) = StartupSocket::pair(deadline).map_err(before)?;
1245 let uid_map = &make_id_map(&self.uid_map);
1246 let gid_map = &make_id_map(&self.gid_map);
1247 let context = ChildContext {
1248 stdin: None,
1249 stdout: None,
1250 stderr: None,
1251 uid_map,
1252 gid_map,
1253 seccomp_fd: None,
1254 };
1255 let (reader, writer) = pipe().map_err(|error| before(error.into()))?;
1256 #[cfg(test)]
1257 OWNED_RESULT_PIPE_CAPACITY.with(|capacity| {
1258 capacity.set(
1259 Errno::result(unsafe { libc::fcntl(reader.as_raw_fd(), libc::F_GETPIPE_SZ) }).ok(),
1260 );
1261 });
1262 let io = StartupChildIo {
1263 parent_fd: parent_socket.fd.as_raw_fd(),
1264 child_fd: child_socket.fd.as_raw_fd(),
1265 reader_fd: reader.as_raw_fd(),
1266 writer_fd: writer.as_raw_fd(),
1267 deadline,
1268 };
1269 let mut stack = child_stack().map_err(|error| before(error.into()))?;
1270 #[cfg(feature = "nightly")]
1271 let output_capture = std::io::set_output_capture(None);
1272 let namespace = self.namespace;
1273 let result = super::clone::clone_with_stack_owned(
1274 || self.startup_child_run(&context, io, child_start, run),
1275 namespace,
1276 &mut stack,
1277 );
1278 #[cfg(feature = "nightly")]
1279 std::io::set_output_capture(output_capture);
1280 let child = OwnedContainerCleanup::new(result.map_err(|error| before(error.into()))?);
1281 let mut owned = OwnedFinalization::new(child, reader);
1283 drop(child_socket);
1284 drop(writer);
1285 if owned.cleanup().pidfd.is_none() {
1286 return Err(owned.fail(OwnedRunFailure::Startup(StartupError::Protocol), deadline));
1287 }
1288 let ready = (|| {
1289 let descriptors = parent_socket.receive(Some(STARTUP_REQUEST))?;
1290 parent_socket.receive(None)?;
1291 parent_start(ParentStartContext {
1292 child_pid: owned.cleanup().pid,
1293 child_pidfd: owned.cleanup().pidfd.as_ref().unwrap().as_fd(),
1294 deadline,
1295 descriptors,
1296 })?;
1297 parent_socket.send(STARTUP_READY, &StartupFds::default(), None)?;
1298 parent_socket.close_write()?;
1299 Ok::<_, StartupError>(())
1302 })();
1303 if let Err(error) = ready {
1304 return Err(owned.fail(OwnedRunFailure::Startup(error), deadline));
1305 }
1306 drop(parent_socket);
1307 if let Err(error) = owned.drain() {
1308 return Err(owned.fail(error, deadline));
1309 }
1310 Ok(OwnedDeferredContainerRun { inner: owned })
1311 }
1312
1313 pub fn run_with_deferred_drop_owned<F, T, D>(
1344 &mut self,
1345 run: &mut F,
1346 ) -> Result<OwnedDeferredContainerRun<T>, StartupOwnedFailure<T>>
1347 where
1348 F: FnMut() -> (T, D),
1349 T: Serialize,
1350 {
1351 let before = |cause| StartupOwnedFailure::BeforeClone { cause };
1352 let mut disposition: libc::sigaction = unsafe { std::mem::zeroed() };
1353 Errno::result(unsafe {
1354 libc::sigaction(libc::SIGCHLD, std::ptr::null(), &mut disposition)
1355 })
1356 .map_err(|error| before(error.into()))?;
1357 if disposition.sa_sigaction == libc::SIG_IGN
1358 || disposition.sa_flags & libc::SA_NOCLDWAIT != 0
1359 {
1360 return Err(before(StartupError::Io(Errno::ECHILD)));
1361 }
1362 let uid_map = &make_id_map(&self.uid_map);
1363 let gid_map = &make_id_map(&self.gid_map);
1364 let context = ChildContext {
1365 stdin: None,
1366 stdout: None,
1367 stderr: None,
1368 uid_map,
1369 gid_map,
1370 seccomp_fd: None,
1371 };
1372 let (reader, writer) = pipe().map_err(|error| before(error.into()))?;
1373 let reader_fd = reader.as_raw_fd();
1374 let writer_fd = writer.as_raw_fd();
1375 let mut stack = child_stack().map_err(|error| before(error.into()))?;
1376 #[cfg(feature = "nightly")]
1377 let output_capture = std::io::set_output_capture(None);
1378 let namespace = self.namespace;
1379 let result = super::clone::clone_with_stack_owned(
1380 || {
1381 unsafe { libc::close(reader_fd) };
1384 let (value, deferred) = match self.setup(&context, &mut []) {
1385 Ok(()) => {
1386 let (value, deferred) = run();
1387 (Ok(value), Some(deferred))
1388 }
1389 Err(error) => (Err(StartupError::Setup(error)), None),
1390 };
1391 let mut writer = std::io::BufWriter::new(Fd::new(writer_fd));
1392 bincode::serde::encode_into_std_write(
1393 &value,
1394 &mut writer,
1395 bincode::config::legacy(),
1396 )
1397 .expect("Failed to serialize return value");
1398 writer.flush().expect("Failed to flush return value");
1399 drop(writer);
1400 drop(deferred);
1401 0
1402 },
1403 namespace,
1404 &mut stack,
1405 );
1406 #[cfg(feature = "nightly")]
1407 std::io::set_output_capture(output_capture);
1408 let child = OwnedContainerCleanup::new(result.map_err(|error| before(error.into()))?);
1409 let mut owned = OwnedFinalization::new(child, reader);
1412 drop(writer);
1413 #[cfg(test)]
1414 owned_deferred_before_drain(&mut owned);
1415 if let Err(error) = owned.drain() {
1416 return Err(owned.refuse(error));
1417 }
1418 if owned.cleanup().pidfd.is_none() {
1419 return Err(owned.refuse(OwnedRunFailure::Startup(StartupError::Protocol)));
1420 }
1421 Ok(OwnedDeferredContainerRun { inner: owned })
1422 }
1423
1424 pub fn run_with_deferred_drop<F, T, D>(
1445 &mut self,
1446 mut f: F,
1447 ) -> Result<DeferredContainerRun<T>, RunError>
1448 where
1449 F: FnMut() -> (T, D),
1450 T: Serialize + DeserializeOwned,
1451 {
1452 let clone_flags = self.namespace.bits() | libc::SIGCHLD;
1453 let uid_map = &make_id_map(&self.uid_map);
1454 let gid_map = &make_id_map(&self.gid_map);
1455 let context = ChildContext {
1456 stdin: None,
1457 stdout: None,
1458 stderr: None,
1459 uid_map,
1460 gid_map,
1461 seccomp_fd: None,
1462 };
1463 let (mut reader, writer) = pipe()?;
1464 let writer_fd = writer.as_raw_fd();
1465 let mut stack = child_stack()?;
1466
1467 #[cfg(feature = "nightly")]
1468 let output_capture = std::io::set_output_capture(None);
1469
1470 let result = clone_with_stack(
1471 || {
1472 let (value, deferred) = match self.setup(&context, &mut []) {
1473 Ok(()) => {
1474 let (value, deferred) = f();
1475 (Ok(value), Some(deferred))
1476 }
1477 Err(error) => (Err(error), None),
1478 };
1479 let mut writer = std::io::BufWriter::new(Fd::new(writer_fd));
1480 bincode::serde::encode_into_std_write(
1481 &value,
1482 &mut writer,
1483 bincode::config::legacy(),
1484 )
1485 .expect("Failed to serialize return value");
1486 writer.flush().expect("Failed to flush return value");
1487 drop(writer);
1488 drop(deferred);
1489 0
1490 },
1491 clone_flags,
1492 &mut stack,
1493 );
1494
1495 #[cfg(feature = "nightly")]
1496 std::io::set_output_capture(output_capture);
1497
1498 let child = WaitGuard::new(result?);
1499 drop(writer);
1500
1501 let mut buf = Vec::new();
1502 match reader.read_to_end(&mut buf) {
1503 Ok(0) => Err(RunError::ExitStatus(child.wait()?)),
1504 Ok(n) => {
1505 let value: Result<T, Error> =
1506 bincode::serde::decode_from_slice(&buf[0..n], bincode::config::legacy())
1507 .unwrap()
1508 .0;
1509 Ok(DeferredContainerRun {
1510 value: Some(value.map_err(RunError::Spawn)?),
1511 child,
1512 })
1513 }
1514 Err(error) => panic!("Got unexpected error: {error}"),
1515 }
1516 }
1517}
1518
1519pub const MAX_STARTUP_FDS: usize = 8;
1521
1522#[derive(
1524 thiserror::Error,
1525 Debug,
1526 Copy,
1527 Clone,
1528 Eq,
1529 PartialEq,
1530 Serialize,
1531 serde::Deserialize
1532)]
1533pub enum StartupError {
1534 #[error("container setup failed: {0}")]
1536 Setup(Error),
1537 #[error("startup syscall failed: {0}")]
1539 Io(Errno),
1540 #[error("startup timeout must be positive and representable")]
1542 InvalidTimeout,
1543 #[error("startup deadline elapsed")]
1545 TimedOut,
1546 #[error("startup callback refused")]
1548 Refused,
1549 #[error("startup peer closed prematurely")]
1551 PeerClosed,
1552 #[error("invalid startup or result protocol")]
1554 Protocol,
1555 #[error("child exited before publishing its result")]
1557 MissingResult,
1558}
1559
1560impl From<Errno> for StartupError {
1561 fn from(error: Errno) -> Self {
1562 Self::Io(error)
1563 }
1564}
1565
1566#[derive(thiserror::Error, Debug, Eq, PartialEq)]
1568pub enum StartupRunError {
1569 #[error("before clone: {0}")]
1571 BeforeClone(StartupError),
1572 #[error("{cause}; child terminal status: {status:?}")]
1574 Child {
1575 cause: StartupError,
1577 status: ExitStatus,
1579 },
1580 #[error("{cause}; child cleanup failed: {errno}")]
1582 Cleanup {
1583 cause: StartupError,
1585 errno: Errno,
1587 },
1588}
1589
1590#[derive(Default)]
1591struct StartupFds {
1592 values: [Option<std::os::fd::OwnedFd>; MAX_STARTUP_FDS],
1593 len: usize,
1594}
1595
1596impl StartupFds {
1597 fn push(&mut self, fd: std::os::fd::OwnedFd) -> Result<(), StartupError> {
1598 if self.len == MAX_STARTUP_FDS {
1599 return Err(StartupError::Protocol);
1600 }
1601 self.values[self.len] = Some(fd);
1602 self.len += 1;
1603 Ok(())
1604 }
1605}
1606
1607pub struct ChildStartContext {
1613 deadline: std::time::Instant,
1614 descriptors: StartupFds,
1615 failure: Option<StartupError>,
1616}
1617
1618impl ChildStartContext {
1619 pub fn deadline(&self) -> std::time::Instant {
1621 self.deadline
1622 }
1623
1624 pub fn transfer_fd(&mut self, fd: std::os::fd::OwnedFd) -> Result<(), StartupError> {
1627 let result = self.descriptors.push(fd);
1628 if let Err(error) = result {
1629 self.failure = Some(error);
1630 }
1631 result
1632 }
1633}
1634
1635pub struct ParentStartContext<'a> {
1641 child_pid: Pid,
1642 child_pidfd: std::os::fd::BorrowedFd<'a>,
1643 deadline: std::time::Instant,
1644 descriptors: StartupFds,
1645}
1646
1647impl ParentStartContext<'_> {
1648 pub fn child_pid(&self) -> Pid {
1650 self.child_pid
1651 }
1652
1653 pub fn child_pidfd(&self) -> std::os::fd::BorrowedFd<'_> {
1655 self.child_pidfd
1656 }
1657
1658 pub fn deadline(&self) -> std::time::Instant {
1660 self.deadline
1661 }
1662
1663 pub fn descriptor_count(&self) -> usize {
1665 self.descriptors.len
1666 }
1667
1668 pub fn take_fd(&mut self, index: usize) -> Option<std::os::fd::OwnedFd> {
1671 self.descriptors.values.get_mut(index)?.take()
1672 }
1673}
1674
1675#[derive(Clone, Copy)]
1676struct StartupChildIo {
1677 parent_fd: i32,
1678 child_fd: i32,
1679 reader_fd: i32,
1680 writer_fd: i32,
1681 deadline: std::time::Instant,
1682}
1683
1684const STARTUP_REQUEST: u8 = 1;
1687const STARTUP_READY: u8 = 2;
1688const STARTUP_FAILURE: u8 = 4;
1689const STARTUP_FRAME_SIZE: usize = 64;
1690
1691struct StartupSocket {
1692 fd: Fd,
1693 deadline: std::time::Instant,
1694}
1695
1696impl StartupSocket {
1697 fn pair(deadline: std::time::Instant) -> Result<(Self, Self), StartupError> {
1698 let mut pair = [-1; 2];
1699 Errno::result(unsafe {
1700 libc::socketpair(
1701 libc::AF_UNIX,
1702 libc::SOCK_STREAM | libc::SOCK_CLOEXEC | libc::SOCK_NONBLOCK,
1703 0,
1704 pair.as_mut_ptr(),
1705 )
1706 })?;
1707 Ok((
1708 Self {
1709 fd: Fd::new(pair[0]),
1710 deadline,
1711 },
1712 Self {
1713 fd: Fd::new(pair[1]),
1714 deadline,
1715 },
1716 ))
1717 }
1718
1719 fn poll(&self, events: libc::c_short) -> Result<(), StartupError> {
1720 loop {
1721 let remaining = self
1722 .deadline
1723 .checked_duration_since(std::time::Instant::now())
1724 .filter(|value| !value.is_zero())
1725 .ok_or(StartupError::TimedOut)?;
1726 let millis = remaining
1727 .as_millis()
1728 .saturating_add(1)
1729 .min(i32::MAX as u128) as i32;
1730 let mut fd = libc::pollfd {
1731 fd: self.fd.as_raw_fd(),
1732 events,
1733 revents: 0,
1734 };
1735 match Errno::result(unsafe { libc::poll(&mut fd, 1, millis) }) {
1736 Ok(0) | Err(Errno::EINTR) => continue,
1737 Ok(_) if fd.revents & libc::POLLNVAL != 0 => {
1738 return Err(StartupError::Io(Errno::EBADF));
1739 }
1740 Ok(_) => return Ok(()),
1741 Err(error) => return Err(error.into()),
1742 }
1743 }
1744 }
1745
1746 fn send_bytes(&self, bytes: &[u8], fds: &StartupFds) -> Result<(), StartupError> {
1747 let mut ancillary = [0usize; 16];
1750 let mut message: libc::msghdr = unsafe { std::mem::zeroed() };
1751 if fds.len != 0 {
1752 message.msg_control = ancillary.as_mut_ptr().cast();
1753 message.msg_controllen =
1754 unsafe { libc::CMSG_SPACE((fds.len * std::mem::size_of::<i32>()) as u32) as usize };
1755 assert!(message.msg_controllen <= std::mem::size_of_val(&ancillary));
1756 unsafe {
1757 let header = libc::CMSG_FIRSTHDR(&message);
1758 (*header).cmsg_level = libc::SOL_SOCKET;
1759 (*header).cmsg_type = libc::SCM_RIGHTS;
1760 (*header).cmsg_len =
1761 libc::CMSG_LEN((fds.len * std::mem::size_of::<i32>()) as u32) as usize;
1762 let data = libc::CMSG_DATA(header).cast::<i32>();
1763 for index in 0..fds.len {
1764 data.add(index)
1765 .write(fds.values[index].as_ref().unwrap().as_raw_fd());
1766 }
1767 }
1768 }
1769 let mut offset = 0;
1770 while offset < bytes.len() {
1771 self.poll(libc::POLLOUT)?;
1772 let mut iov = libc::iovec {
1773 iov_base: bytes[offset..].as_ptr().cast_mut().cast(),
1774 iov_len: bytes.len() - offset,
1775 };
1776 #[cfg(test)]
1777 if STARTUP_TEST_FAULT.with(|fault| fault.get()) == StartupTestFault::Fragmented {
1778 iov.iov_len = 1;
1779 }
1780 message.msg_iov = &mut iov;
1781 message.msg_iovlen = 1;
1782 match Errno::result(unsafe {
1783 libc::sendmsg(self.fd.as_raw_fd(), &message, libc::MSG_NOSIGNAL)
1784 }) {
1785 Ok(0) => return Err(StartupError::Protocol),
1786 Ok(size) => {
1787 offset += size as usize;
1788 message.msg_control = std::ptr::null_mut();
1789 message.msg_controllen = 0;
1790 }
1791 Err(Errno::EINTR | Errno::EAGAIN) => continue,
1792 Err(error) => return Err(error.into()),
1793 }
1794 }
1795 Ok(())
1796 }
1797
1798 fn send(
1799 &self,
1800 phase: u8,
1801 fds: &StartupFds,
1802 failure: Option<StartupError>,
1803 ) -> Result<(), StartupError> {
1804 let mut frame = [0u8; STARTUP_FRAME_SIZE];
1805 frame[..4].copy_from_slice(b"RVS1");
1806 frame[4] = phase;
1807 frame[5] = fds.len as u8;
1808 if let Some(error) = failure {
1809 frame[6] = bincode::serde::encode_into_slice(
1810 error,
1811 &mut frame[8..],
1812 bincode::config::legacy(),
1813 )
1814 .map_err(|_| StartupError::Protocol)? as u8;
1815 }
1816 #[cfg(test)]
1817 if let Some(result) = self.inject_test_fault(phase, &frame, fds) {
1818 return result;
1819 }
1820 self.send_bytes(&frame, fds)
1821 }
1822
1823 fn receive(&self, phase: Option<u8>) -> Result<StartupFds, StartupError> {
1824 use std::os::fd::FromRawFd;
1825 let mut frame = [0u8; STARTUP_FRAME_SIZE];
1826 let mut offset = 0;
1827 let mut fds = StartupFds::default();
1828 loop {
1829 self.poll(libc::POLLIN)?;
1830 let mut ancillary = [0usize; 16];
1831 let capacity = if phase.is_none() {
1832 1
1833 } else {
1834 frame.len() - offset
1835 };
1836 let mut iov = libc::iovec {
1837 iov_base: frame[offset..].as_mut_ptr().cast(),
1838 iov_len: capacity,
1839 };
1840 let mut message: libc::msghdr = unsafe { std::mem::zeroed() };
1841 message.msg_iov = &mut iov;
1842 message.msg_iovlen = 1;
1843 message.msg_control = ancillary.as_mut_ptr().cast();
1844 message.msg_controllen = unsafe {
1845 libc::CMSG_SPACE((MAX_STARTUP_FDS * std::mem::size_of::<i32>()) as u32) as usize
1846 };
1847 assert!(message.msg_controllen <= std::mem::size_of_val(&ancillary));
1848 let size = match Errno::result(unsafe {
1849 libc::recvmsg(self.fd.as_raw_fd(), &mut message, libc::MSG_CMSG_CLOEXEC)
1850 }) {
1851 Ok(size) => size as usize,
1852 Err(Errno::EINTR | Errno::EAGAIN) => continue,
1853 Err(error) => return Err(error.into()),
1854 };
1855 let mut malformed = message.msg_flags & (libc::MSG_TRUNC | libc::MSG_CTRUNC) != 0;
1856 let previous_fds = fds.len;
1857 unsafe {
1860 let mut header = libc::CMSG_FIRSTHDR(&message);
1861 while !header.is_null() {
1862 if (*header).cmsg_level == libc::SOL_SOCKET
1863 && (*header).cmsg_type == libc::SCM_RIGHTS
1864 && (*header).cmsg_len >= libc::CMSG_LEN(0) as usize
1865 {
1866 let bytes = (*header).cmsg_len - libc::CMSG_LEN(0) as usize;
1867 malformed |= !bytes.is_multiple_of(std::mem::size_of::<i32>());
1868 let data = libc::CMSG_DATA(header).cast::<i32>();
1869 for index in 0..bytes / std::mem::size_of::<i32>() {
1870 let fd = std::os::fd::OwnedFd::from_raw_fd(data.add(index).read());
1871 if fds.push(fd).is_err() {
1872 malformed = true;
1873 }
1874 }
1875 } else {
1876 malformed = true;
1877 }
1878 header = libc::CMSG_NXTHDR(&message, header);
1879 }
1880 }
1881 if malformed
1882 || (fds.len != previous_fds && (offset != 0 || phase != Some(STARTUP_REQUEST)))
1883 {
1884 return Err(StartupError::Protocol);
1885 }
1886 if size == 0 {
1887 return if phase.is_none() && fds.len == 0 {
1890 Ok(fds)
1891 } else if offset == 0 && fds.len == 0 {
1892 Err(StartupError::PeerClosed)
1893 } else {
1894 Err(StartupError::Protocol)
1895 };
1896 }
1897 if phase.is_none() {
1898 return Err(StartupError::Protocol);
1899 }
1900 offset += size;
1901 if offset < frame.len() {
1902 continue;
1903 }
1904 if &frame[..4] != b"RVS1" || frame[5] as usize != fds.len || frame[7] != 0 {
1905 return Err(StartupError::Protocol);
1906 }
1907 if frame[4] == STARTUP_FAILURE
1908 && fds.len == 0
1909 && frame[6] != 0
1910 && frame[6] as usize <= frame.len() - 8
1911 {
1912 let end = 8 + frame[6] as usize;
1913 let (error, used) = bincode::serde::decode_from_slice::<StartupError, _>(
1914 &frame[8..end],
1915 bincode::config::legacy(),
1916 )
1917 .map_err(|_| StartupError::Protocol)?;
1918 if used != end - 8 || frame[end..].iter().any(|byte| *byte != 0) {
1919 return Err(StartupError::Protocol);
1920 }
1921 return Err(error);
1922 }
1923 if phase != Some(frame[4])
1924 || frame[6..].iter().any(|byte| *byte != 0)
1925 || (frame[4] != STARTUP_REQUEST && fds.len != 0)
1926 {
1927 return Err(StartupError::Protocol);
1928 }
1929 return Ok(fds);
1930 }
1931 }
1932
1933 fn close_write(&self) -> Result<(), StartupError> {
1934 Errno::result(unsafe { libc::shutdown(self.fd.as_raw_fd(), libc::SHUT_WR) })?;
1935 Ok(())
1936 }
1937}
1938
1939#[cfg(test)]
1943#[derive(Copy, Clone, Eq, PartialEq)]
1944enum StartupTestFault {
1945 None,
1946 Fragmented,
1947 RequestEmptyTrailing,
1948 RequestDuplicate,
1949 RequestMalformed,
1950 RequestTrailingRights,
1951 PermissionEmptyTrailing,
1952 PermissionDuplicate,
1953 PermissionMalformed,
1954 PermissionTrailingRights,
1955 ObservePermissionLate,
1956}
1957
1958#[cfg(test)]
1959std::thread_local! {
1960 static STARTUP_TEST_FAULT: std::cell::Cell<StartupTestFault> = const { std::cell::Cell::new(StartupTestFault::None) };
1961}
1962
1963#[cfg(test)]
1964impl StartupSocket {
1965 fn inject_test_fault(
1966 &self,
1967 phase: u8,
1968 frame: &[u8],
1969 fds: &StartupFds,
1970 ) -> Option<Result<(), StartupError>> {
1971 use StartupTestFault::*;
1972 let fault = STARTUP_TEST_FAULT.with(|value| value.get());
1973 let request = phase == STARTUP_REQUEST;
1974 let permission = phase == STARTUP_READY;
1975 if (request && fault == RequestEmptyTrailing)
1976 || (permission && fault == PermissionEmptyTrailing)
1977 {
1978 return Some({
1979 assert_eq!(
1980 unsafe {
1981 libc::send(self.fd.as_raw_fd(), std::ptr::null(), 0, libc::MSG_NOSIGNAL)
1982 },
1983 0
1984 );
1985 self.send_bytes(b"X", &StartupFds::default())
1986 });
1987 }
1988 if (request && fault == RequestMalformed) || (permission && fault == PermissionMalformed) {
1989 let mut bad = frame.to_vec();
1990 bad[0] ^= 1;
1991 return Some(self.send_bytes(&bad, fds));
1992 }
1993 if (request && fault == RequestDuplicate) || (permission && fault == PermissionDuplicate) {
1994 return Some(
1995 self.send_bytes(frame, fds)
1996 .and_then(|()| self.send_bytes(frame, fds)),
1997 );
1998 }
1999 if (request && fault == RequestTrailingRights)
2000 || (permission && fault == PermissionTrailingRights)
2001 {
2002 return Some((|| {
2003 self.send_bytes(frame, fds)?;
2004 let mut trailing = StartupFds::default();
2005 trailing.push(std::fs::File::open("/dev/null").unwrap().into())?;
2006 self.send_bytes(b"X", &trailing)
2007 })());
2008 }
2009 Option::None
2010 }
2011}
2012
2013#[derive(Clone, Copy, Debug, Eq, PartialEq)]
2015pub enum ChildCleanupObservation {
2016 Pending,
2018 Reaped(ExitStatus),
2020 ExitedWithoutWaitStatus,
2022 Unknown,
2024}
2025
2026#[derive(Debug)]
2033pub struct OwnedContainerCleanup {
2034 pid: Pid,
2035 wait_owned: bool,
2036 pidfd: Option<std::os::fd::OwnedFd>,
2037 observation: ChildCleanupObservation,
2038 last_error: Option<Errno>,
2039 #[cfg(test)]
2040 signal_error_once: Option<Errno>,
2041 #[cfg(test)]
2042 wait_error_once: Option<Errno>,
2043 #[cfg(test)]
2044 cancellation_observed: Option<std::sync::Arc<std::sync::atomic::AtomicBool>>,
2045}
2046
2047impl OwnedContainerCleanup {
2048 fn new(child: super::clone::OwnedClone) -> Self {
2049 Self {
2050 pid: child.pid,
2051 wait_owned: true,
2052 pidfd: child.pidfd,
2053 observation: ChildCleanupObservation::Pending,
2054 last_error: None,
2055 #[cfg(test)]
2056 signal_error_once: OWNED_STARTUP_CANCEL_ERROR.with(|error| error.take()),
2057 #[cfg(test)]
2058 wait_error_once: None,
2059 #[cfg(test)]
2060 cancellation_observed: None,
2061 }
2062 }
2063
2064 pub fn child_pid(&self) -> Pid {
2066 self.pid
2067 }
2068 pub fn observation(&self) -> ChildCleanupObservation {
2070 self.observation
2071 }
2072 pub fn last_error(&self) -> Option<Errno> {
2074 self.last_error
2075 }
2076
2077 fn settled(&self) -> bool {
2078 matches!(
2079 self.observation,
2080 ChildCleanupObservation::Reaped(_) | ChildCleanupObservation::ExitedWithoutWaitStatus
2081 )
2082 }
2083
2084 fn observe_wait(&mut self) -> Result<(), Errno> {
2085 if !self.wait_owned {
2086 return Ok(());
2087 }
2088 #[cfg(test)]
2089 if let Some(error) = self.wait_error_once.take() {
2090 return Err(error);
2091 }
2092 let mut status = 0;
2093 match Errno::result(unsafe { libc::waitpid(self.pid.as_raw(), &mut status, libc::WNOHANG) })
2094 {
2095 Ok(0) => (),
2096 Ok(pid) => {
2097 assert_eq!(pid, self.pid.as_raw());
2098 self.wait_owned = false;
2099 self.observation = ChildCleanupObservation::Reaped(ExitStatus::from_raw(status));
2100 }
2101 Err(Errno::ECHILD) => {
2102 self.wait_owned = false;
2104 self.last_error = Some(Errno::ECHILD);
2105 self.observation = ChildCleanupObservation::Unknown;
2106 }
2107 Err(error) => return Err(error),
2108 }
2109 Ok(())
2110 }
2111
2112 pub fn wait_until(&mut self, deadline: std::time::Instant) -> ChildCleanupObservation {
2114 loop {
2115 if self.settled() {
2116 return self.observation;
2117 }
2118 match self.observe_wait() {
2119 Ok(()) => (),
2120 Err(Errno::EINTR) => {
2121 self.last_error = Some(Errno::EINTR);
2122 if std::time::Instant::now() < deadline {
2123 continue;
2124 }
2125 }
2126 Err(error) => {
2127 self.last_error = Some(error);
2128 self.observation = ChildCleanupObservation::Unknown;
2129 return self.observation;
2130 }
2131 }
2132 if self.settled() {
2133 return self.observation;
2134 }
2135 let remaining = deadline.saturating_duration_since(std::time::Instant::now());
2136 let milliseconds = remaining
2137 .as_millis()
2138 .saturating_add(u128::from(!remaining.is_zero()))
2139 .min(10) as i32;
2140 let mut poll = libc::pollfd {
2141 fd: self.pidfd.as_ref().map_or(-1, AsRawFd::as_raw_fd),
2142 events: libc::POLLIN,
2143 revents: 0,
2144 };
2145 match Errno::result(unsafe { libc::poll(&mut poll, 1, milliseconds) }) {
2146 Ok(_) => {
2147 if poll.revents & (libc::POLLNVAL | libc::POLLERR) != 0 {
2148 self.last_error = Some(Errno::EBADF);
2149 self.observation = ChildCleanupObservation::Unknown;
2150 return self.observation;
2151 }
2152 if !self.wait_owned && poll.revents & (libc::POLLIN | libc::POLLHUP) != 0 {
2153 self.observation = ChildCleanupObservation::ExitedWithoutWaitStatus;
2154 return self.observation;
2155 }
2156 }
2157 Err(Errno::EINTR) => {
2158 self.last_error = Some(Errno::EINTR);
2159 #[cfg(test)]
2160 OWNED_POLL_INTERRUPTED.fetch_add(1, std::sync::atomic::Ordering::Release);
2161 }
2162 Err(error) => {
2163 self.last_error = Some(error);
2164 self.observation = ChildCleanupObservation::Unknown;
2165 return self.observation;
2166 }
2167 }
2168 if std::time::Instant::now() >= deadline {
2169 if let Err(error) = self.observe_wait() {
2172 self.last_error = Some(error);
2173 self.observation = ChildCleanupObservation::Unknown;
2174 }
2175 return self.observation;
2176 }
2177 }
2178 }
2179
2180 fn signal_cancel(&mut self) -> Result<(), Errno> {
2181 #[cfg(test)]
2182 if let Some(error) = self.signal_error_once.take() {
2183 return Err(error);
2184 }
2185 let fd = self.pidfd.as_ref().ok_or(Errno::EBADF)?;
2186 Errno::result(unsafe {
2187 libc::syscall(
2188 libc::SYS_pidfd_send_signal,
2189 fd.as_raw_fd(),
2190 libc::SIGKILL,
2191 std::ptr::null::<libc::siginfo_t>(),
2192 0,
2193 )
2194 })
2195 .map(|_| ())
2196 }
2197
2198 pub fn cancel_and_wait_until(
2201 &mut self,
2202 deadline: std::time::Instant,
2203 ) -> ChildCleanupObservation {
2204 if self.settled() {
2205 return self.observation;
2206 }
2207 match self.signal_cancel() {
2208 Ok(()) => (),
2209 Err(error) => {
2210 self.last_error = Some(error);
2211 }
2215 }
2216 #[cfg(test)]
2217 if let Some(observed) = &self.cancellation_observed {
2218 observed.store(true, std::sync::atomic::Ordering::Release);
2219 }
2220 self.wait_until(deadline)
2221 }
2222}
2223
2224impl Drop for OwnedContainerCleanup {
2225 fn drop(&mut self) {
2226 while !self.settled() {
2227 self.cancel_and_wait_until(
2228 std::time::Instant::now() + std::time::Duration::from_millis(100),
2229 );
2230 if !self.settled() {
2231 std::thread::sleep(std::time::Duration::from_millis(10));
2232 }
2233 }
2234 }
2235}
2236
2237#[cfg(test)]
2238std::thread_local! {
2239 static OWNED_STARTUP_CANCEL_ERROR: std::cell::Cell<Option<Errno>> = const { std::cell::Cell::new(None) };
2240 static OWNED_DEFERRED_DRAIN_HOOK: std::cell::Cell<Option<fn(RawFd)>> = const { std::cell::Cell::new(None) };
2241 static OWNED_RESULT_PIPE_CAPACITY: std::cell::Cell<Option<i32>> = const { std::cell::Cell::new(None) };
2242}
2243#[cfg(test)]
2244fn owned_deferred_before_drain<T>(owned: &mut OwnedFinalization<T>) {
2245 if let Some(hook) = OWNED_DEFERRED_DRAIN_HOOK.with(|hook| hook.take()) {
2246 hook(owned.reader.as_ref().unwrap().as_raw_fd());
2247 }
2248}
2249
2250#[cfg(test)]
2251static OWNED_POLL_INTERRUPTED: std::sync::atomic::AtomicUsize =
2252 std::sync::atomic::AtomicUsize::new(0);
2253
2254#[derive(Clone, Copy, Debug, Eq, PartialEq)]
2256pub enum OwnedRunFailure {
2257 Cancelled,
2260 Startup(StartupError),
2262 ResultRead(Errno),
2264 ChildStatus(ExitStatus),
2266 Cleanup(Errno),
2268 WaitStatusUnavailable,
2270}
2271
2272#[derive(Debug)]
2275pub enum StartupOwnedFailure<T> {
2276 BeforeClone {
2278 cause: StartupError,
2280 },
2281 AfterClone {
2283 cause: OwnedRunFailure,
2285 run: OwnedFinalization<T>,
2287 },
2288}
2289
2290#[must_use = "retain or settle the actual child before releasing external owners"]
2293pub struct OwnedFinalization<T> {
2294 child: Option<OwnedContainerCleanup>,
2295 reader: Option<Fd>,
2296 bytes: Vec<u8>,
2297 eof: bool,
2298 failure: Option<OwnedRunFailure>,
2299 marker: std::marker::PhantomData<fn() -> T>,
2300}
2301
2302impl<T> std::fmt::Debug for OwnedFinalization<T> {
2303 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
2304 f.debug_struct("OwnedFinalization")
2305 .field("child", &self.child)
2306 .field("bytes", &self.bytes.len())
2307 .field("eof", &self.eof)
2308 .field("failure", &self.failure)
2309 .finish()
2310 }
2311}
2312
2313impl<T> OwnedFinalization<T> {
2314 fn new(child: OwnedContainerCleanup, reader: Fd) -> Self {
2315 Self {
2316 child: Some(child),
2317 reader: Some(reader),
2318 bytes: Vec::new(),
2319 eof: false,
2320 failure: None,
2321 marker: std::marker::PhantomData,
2322 }
2323 }
2324 pub fn provisional_bytes(&self) -> &[u8] {
2326 &self.bytes
2327 }
2328 pub fn result_eof(&self) -> bool {
2330 self.eof
2331 }
2332 pub fn result_reader_abandoned(&self) -> bool {
2336 !self.eof && self.reader.is_none()
2337 }
2338 pub fn cleanup(&self) -> &OwnedContainerCleanup {
2340 self.child.as_ref().unwrap()
2341 }
2342 pub fn failure(&self) -> Option<OwnedRunFailure> {
2344 self.failure
2345 }
2346
2347 fn fail(
2348 mut self,
2349 cause: OwnedRunFailure,
2350 deadline: std::time::Instant,
2351 ) -> StartupOwnedFailure<T> {
2352 self.failure = Some(cause);
2353 self.child.as_mut().unwrap().cancel_and_wait_until(deadline);
2354 StartupOwnedFailure::AfterClone { cause, run: self }
2355 }
2356 fn refuse(mut self, cause: OwnedRunFailure) -> StartupOwnedFailure<T> {
2357 self.failure = Some(cause);
2358 StartupOwnedFailure::AfterClone { cause, run: self }
2359 }
2360
2361 fn drain(&mut self) -> Result<(), OwnedRunFailure> {
2362 match self.reader.as_mut().unwrap().read_to_end(&mut self.bytes) {
2363 Ok(_) => {
2364 self.eof = true;
2365 self.reader.take();
2366 if self.bytes.is_empty() {
2367 Err(OwnedRunFailure::Startup(StartupError::MissingResult))
2368 } else {
2369 Ok(())
2370 }
2371 }
2372 Err(error) => Err(OwnedRunFailure::ResultRead(Errno::new(
2373 error.raw_os_error().unwrap_or(libc::EIO),
2374 ))),
2375 }
2376 }
2377 pub fn retry_until(mut self, deadline: std::time::Instant) -> OwnedFinalize<T> {
2386 self.abandon_unread_result_without_pidfd();
2387 let child = self.child.as_mut().unwrap();
2388 if let Some(cause) = self.failure {
2389 child.cancel_and_wait_until(deadline);
2390 return OwnedFinalize::Failed {
2391 cause,
2392 cleanup: self,
2393 };
2394 }
2395 match child.wait_until(deadline) {
2396 ChildCleanupObservation::Reaped(status) if status.success() => {
2397 assert!(self.eof);
2398 self.child.take();
2399 OwnedFinalize::Complete(OwnedReapedResult {
2400 bytes: std::mem::take(&mut self.bytes),
2401 status,
2402 marker: std::marker::PhantomData,
2403 })
2404 }
2405 ChildCleanupObservation::Reaped(status) => {
2406 let cause = OwnedRunFailure::ChildStatus(status);
2407 self.failure = Some(cause);
2408 OwnedFinalize::Failed {
2409 cause,
2410 cleanup: self,
2411 }
2412 }
2413 ChildCleanupObservation::ExitedWithoutWaitStatus => {
2414 let cause = OwnedRunFailure::WaitStatusUnavailable;
2415 self.failure = Some(cause);
2416 OwnedFinalize::Failed {
2417 cause,
2418 cleanup: self,
2419 }
2420 }
2421 ChildCleanupObservation::Unknown => {
2422 if let Some(error) = child.last_error() {
2423 let cause = OwnedRunFailure::Cleanup(error);
2424 self.failure = Some(cause);
2425 OwnedFinalize::Failed {
2426 cause,
2427 cleanup: self,
2428 }
2429 } else {
2430 OwnedFinalize::Pending(self)
2431 }
2432 }
2433 ChildCleanupObservation::Pending => OwnedFinalize::Pending(self),
2434 }
2435 }
2436
2437 pub fn cancel_until(mut self, deadline: std::time::Instant) -> OwnedFinalize<T> {
2442 self.failure.get_or_insert(OwnedRunFailure::Cancelled);
2443 self.retry_until(deadline)
2444 }
2445
2446 fn abandon_unread_result_without_pidfd(&mut self) {
2447 if self.failure.is_some()
2448 && self
2449 .child
2450 .as_ref()
2451 .is_some_and(|child| child.pidfd.is_none())
2452 {
2453 self.reader.take();
2454 }
2455 }
2456}
2457
2458impl<T> Drop for OwnedFinalization<T> {
2459 fn drop(&mut self) {
2460 self.abandon_unread_result_without_pidfd();
2464 drop(self.child.take());
2465 self.reader.take();
2466 }
2467}
2468
2469#[derive(Debug)]
2471#[must_use = "the actual child must be finalized before decoding"]
2472pub struct OwnedDeferredContainerRun<T> {
2473 inner: OwnedFinalization<T>,
2474}
2475impl<T> OwnedDeferredContainerRun<T> {
2476 pub fn provisional_bytes(&self) -> &[u8] {
2478 self.inner.provisional_bytes()
2479 }
2480 pub fn cleanup(&self) -> &OwnedContainerCleanup {
2482 self.inner.cleanup()
2483 }
2484 pub fn finalize_until(self, deadline: std::time::Instant) -> OwnedFinalize<T> {
2486 self.inner.retry_until(deadline)
2487 }
2488
2489 pub fn cancel_until(self, deadline: std::time::Instant) -> OwnedFinalize<T> {
2492 self.inner.cancel_until(deadline)
2493 }
2494}
2495
2496#[derive(Debug)]
2498pub enum OwnedFinalize<T> {
2499 Complete(OwnedReapedResult<T>),
2501 Failed {
2503 cause: OwnedRunFailure,
2505 cleanup: OwnedFinalization<T>,
2507 },
2508 Pending(OwnedFinalization<T>),
2510}
2511
2512#[derive(Debug)]
2514pub struct OwnedReapedResult<T> {
2515 bytes: Vec<u8>,
2516 status: ExitStatus,
2517 marker: std::marker::PhantomData<fn() -> T>,
2518}
2519impl<T> OwnedReapedResult<T> {
2520 pub fn status(&self) -> ExitStatus {
2522 self.status
2523 }
2524 pub fn encoded_bytes(&self) -> &[u8] {
2526 &self.bytes
2527 }
2528}
2529impl<T: DeserializeOwned> OwnedReapedResult<T> {
2530 pub fn decode(self) -> Result<T, OwnedDecodeFailure> {
2533 let (cause, detail) = match bincode::serde::decode_from_slice::<Result<T, StartupError>, _>(
2534 &self.bytes,
2535 bincode::config::legacy(),
2536 ) {
2537 Ok((Ok(value), used)) if used == self.bytes.len() => return Ok(value),
2538 Ok((Err(error), used)) if used == self.bytes.len() => {
2539 (OwnedRunFailure::Startup(error), None)
2540 }
2541 Ok(_) => (
2542 OwnedRunFailure::Startup(StartupError::Protocol),
2543 Some("trailing result bytes".to_owned()),
2544 ),
2545 Err(error) => (
2546 OwnedRunFailure::Startup(StartupError::Protocol),
2547 Some(error.to_string()),
2548 ),
2549 };
2550 Err(OwnedDecodeFailure {
2551 cause,
2552 detail,
2553 bytes: self.bytes,
2554 status: self.status,
2555 })
2556 }
2557}
2558
2559#[derive(Debug)]
2561pub struct OwnedDecodeFailure {
2562 cause: OwnedRunFailure,
2563 detail: Option<String>,
2564 bytes: Vec<u8>,
2565 status: ExitStatus,
2566}
2567impl OwnedDecodeFailure {
2568 pub fn cause(&self) -> OwnedRunFailure {
2570 self.cause
2571 }
2572 pub fn detail(&self) -> Option<&str> {
2574 self.detail.as_deref()
2575 }
2576 pub fn encoded_bytes(&self) -> &[u8] {
2578 &self.bytes
2579 }
2580 pub fn status(&self) -> ExitStatus {
2582 self.status
2583 }
2584}
2585
2586struct StartupChild {
2589 wait: Option<WaitGuard>,
2590 pidfd: Option<Fd>,
2591}
2592
2593impl StartupChild {
2594 fn cancel(&mut self) -> Result<ExitStatus, Errno> {
2595 let wait = self.wait.as_ref().unwrap();
2596 let result = match &self.pidfd {
2597 Some(fd) => Errno::result(unsafe {
2598 libc::syscall(
2599 libc::SYS_pidfd_send_signal,
2600 fd.as_raw_fd(),
2601 libc::SIGKILL,
2602 std::ptr::null::<libc::siginfo_t>(),
2603 0,
2604 )
2605 })
2606 .map(|_| ()),
2607 None => Errno::result(unsafe { libc::kill(wait.0.unwrap().as_raw(), libc::SIGKILL) })
2611 .map(|_| ()),
2612 };
2613 match result {
2614 Ok(()) | Err(Errno::ESRCH) => (),
2615 Err(error) => return Err(error),
2616 }
2617 self.wait.take().unwrap().wait()
2618 }
2619
2620 fn fail(&mut self, cause: StartupError) -> StartupRunError {
2621 match self.cancel() {
2622 Ok(status) => StartupRunError::Child { cause, status },
2623 Err(errno) => StartupRunError::Cleanup { cause, errno },
2624 }
2625 }
2626
2627 fn into_wait(mut self) -> WaitGuard {
2628 self.wait.take().unwrap()
2629 }
2630}
2631
2632impl Drop for StartupChild {
2633 fn drop(&mut self) {
2634 if self.wait.is_some() {
2635 let _ = self.cancel();
2636 }
2637 }
2638}
2639
2640pub(super) struct ChildContext<'a> {
2641 pub stdin: Option<&'a Fd>,
2642 pub stdout: Option<&'a Fd>,
2643 pub stderr: Option<&'a Fd>,
2644 pub uid_map: &'a [u8],
2645 pub gid_map: &'a [u8],
2646 pub seccomp_fd: Option<&'a core::sync::atomic::AtomicI32>,
2647}
2648
2649impl<'a> ChildContext<'a> {
2650 fn map_uid(&self) -> Result<(), Errno> {
2651 write_bytes(b"/proc/self/uid_map\0", self.uid_map)
2652 }
2653
2654 fn map_gid(&self) -> Result<(), Errno> {
2655 write_bytes(b"/proc/self/gid_map\0", self.gid_map)
2656 }
2657
2658 fn setgroups(&self, allow: bool) -> Result<(), Errno> {
2659 write_bytes(
2660 b"/proc/self/setgroups\0",
2661 if allow { b"allow\0" } else { b"deny\0" },
2662 )
2663 }
2664}
2665
2666#[derive(thiserror::Error, Debug, Eq, PartialEq)]
2668pub enum RunError {
2669 #[error("Process failed to spawn: {0}")]
2671 Spawn(#[from] Error),
2672
2673 #[error("Process exited with code: {0:?}")]
2677 ExitStatus(ExitStatus),
2678}
2679
2680impl From<Errno> for RunError {
2681 fn from(errno: Errno) -> Self {
2682 Self::Spawn(Error::from(errno))
2683 }
2684}
2685
2686struct WaitGuard(Option<Pid>);
2689
2690impl WaitGuard {
2691 pub fn new(pid: Pid) -> Self {
2692 Self(Some(pid))
2693 }
2694
2695 pub fn wait(mut self) -> Result<ExitStatus, Errno> {
2697 self.wait_inner()
2698 }
2699
2700 fn wait_inner(&mut self) -> Result<ExitStatus, Errno> {
2701 let pid = self.0.expect("child wait guard has already been consumed");
2702 #[cfg(test)]
2703 let instrumented_wait = WAITPID_TEST_PID.load(Ordering::Acquire) == pid.as_raw();
2704 let mut status = 0;
2705 loop {
2706 #[cfg(test)]
2707 if instrumented_wait {
2708 WAITPID_ENTERED.store(true, Ordering::Release);
2709 }
2710 match Errno::result(unsafe { libc::waitpid(pid.as_raw(), &mut status, 0) }) {
2711 Ok(ret) => {
2712 assert_eq!(ret, pid.as_raw());
2713 self.0 = None;
2714 return Ok(ExitStatus::from_raw(status));
2715 }
2716 Err(Errno::EINTR) => {
2717 #[cfg(test)]
2718 {
2719 if instrumented_wait {
2720 WAITPID_INTERRUPTED.fetch_add(1, Ordering::Relaxed);
2721 }
2722 }
2723 }
2724 Err(Errno::ECHILD) => {
2725 self.0 = None;
2726 return Err(Errno::ECHILD);
2727 }
2728 Err(error) => return Err(error),
2729 }
2730 }
2731 }
2732}
2733
2734#[must_use = "a deferred container result must be finalized before its value can be returned"]
2737pub struct DeferredContainerRun<T> {
2738 value: Option<T>,
2739 child: WaitGuard,
2740}
2741
2742impl<T> DeferredContainerRun<T> {
2743 pub fn provisional(&self) -> &T {
2745 self.value.as_ref().expect("provisional value is present")
2746 }
2747
2748 pub fn finalize(self) -> Result<T, RunError> {
2750 self.finalize_with_status().map(|(value, _status)| value)
2751 }
2752
2753 pub fn finalize_with_status(mut self) -> Result<(T, ExitStatus), RunError> {
2757 let status = self.child.wait()?;
2758 if !status.success() {
2759 return Err(RunError::ExitStatus(status));
2760 }
2761 Ok((
2762 self.value.take().expect("provisional value is present"),
2763 status,
2764 ))
2765 }
2766}
2767
2768impl Drop for WaitGuard {
2769 fn drop(&mut self) {
2770 if self.0.is_some() {
2771 let _ = self.wait_inner();
2772 }
2773 }
2774}
2775
2776#[cfg(test)]
2777static WAITPID_TEST_PID: std::sync::atomic::AtomicI32 = std::sync::atomic::AtomicI32::new(0);
2778#[cfg(test)]
2779static WAITPID_ENTERED: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
2780#[cfg(test)]
2781static WAITPID_INTERRUPTED: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0);
2782
2783#[cfg(test)]
2784mod tests {
2785 use std::sync::Mutex;
2786 use std::sync::atomic::AtomicBool;
2787 use std::sync::atomic::Ordering;
2788 use std::time::Duration;
2789 use std::time::Instant;
2790
2791 use nix::sys::signal::SaFlags;
2792 use nix::sys::signal::SigAction;
2793 use nix::sys::signal::SigHandler;
2794 use nix::sys::signal::SigSet;
2795 use nix::sys::signal::Signal;
2796 use nix::sys::signal::sigaction;
2797
2798 use super::*;
2799
2800 include!("owned_deferred_tests.rs");
2801
2802 fn owned_complete<T>(handle: OwnedDeferredContainerRun<T>) -> OwnedReapedResult<T> {
2803 match handle.finalize_until(Instant::now() + Duration::from_secs(2)) {
2804 OwnedFinalize::Complete(result) => result,
2805 other => panic!(
2806 "expected actual successful child settlement: {:?}",
2807 outcome_kind(&other)
2808 ),
2809 }
2810 }
2811
2812 fn outcome_kind<T>(outcome: &OwnedFinalize<T>) -> &'static str {
2813 match outcome {
2814 OwnedFinalize::Complete(_) => "complete",
2815 OwnedFinalize::Failed { .. } => "failed",
2816 OwnedFinalize::Pending(_) => "pending",
2817 }
2818 }
2819
2820 struct OwnedCloneFaultGuard;
2821 impl OwnedCloneFaultGuard {
2822 fn install(fault: super::super::clone::OwnedCloneTestFault) -> Self {
2823 super::super::clone::OWNED_CLONE_FAULT.with(|f| {
2824 assert_eq!(f.get(), super::super::clone::OwnedCloneTestFault::None);
2825 f.set(fault);
2826 });
2827 Self
2828 }
2829 }
2830 impl Drop for OwnedCloneFaultGuard {
2831 fn drop(&mut self) {
2832 super::super::clone::OWNED_CLONE_FAULT
2833 .with(|f| f.set(super::super::clone::OwnedCloneTestFault::None));
2834 }
2835 }
2836
2837 fn owned_test_pidfd_link(path: &Path) -> bool {
2838 let Some(link) = path.to_str() else {
2839 return false;
2840 };
2841 link == "anon_inode:[pidfd]"
2842 || link
2843 .strip_prefix("pidfd:[")
2844 .and_then(|suffix| suffix.strip_suffix(']'))
2845 .is_some_and(|inode| !inode.is_empty() && inode.bytes().all(|b| b.is_ascii_digit()))
2846 }
2847
2848 #[test]
2849 fn owned_startup_atomic_identity_and_complete_decode() {
2850 if crate::test_runs_in_own_process() {
2851 return;
2852 }
2853 use std::os::fd::AsFd;
2854 let parent = Pid::this();
2855 let mut actual_child = None;
2856 let handle = Container::new()
2857 .run_with_startup_owned(
2858 Duration::from_secs(2),
2859 &mut |mut context| {
2860 actual_child = Some(context.child_pid());
2861 assert_ne!(context.child_pid(), parent);
2862 let fd = context.child_pidfd().as_raw_fd();
2863 let pidfd_link = std::fs::read_link(format!("/proc/self/fd/{fd}")).unwrap();
2864 assert!(
2865 owned_test_pidfd_link(&pidfd_link),
2866 "live pidfd: {pidfd_link:?}"
2867 );
2868 eprintln!("owned pidfd anchor: {pidfd_link:?}");
2869 assert_ne!(
2870 unsafe { libc::fcntl(fd, libc::F_GETFD) } & libc::FD_CLOEXEC,
2871 0
2872 );
2873 let mut transferred = std::fs::File::from(context.take_fd(0).unwrap());
2874 assert!(context.take_fd(0).is_none());
2875 let mut text = String::new();
2876 transferred.read_to_string(&mut text).unwrap();
2877 assert!(text.starts_with(&format!("{} ", context.child_pid())));
2878 Ok(())
2879 },
2880 &mut |context| {
2881 let aliases = std::fs::read_dir("/proc/self/fd")
2883 .unwrap()
2884 .filter_map(Result::ok)
2885 .filter_map(|e| std::fs::read_link(e.path()).ok())
2886 .filter(|p| owned_test_pidfd_link(p))
2887 .count();
2888 eprintln!("owned child inherited pidfd aliases: {aliases}");
2889 let file = std::fs::File::open("/proc/self/stat").unwrap();
2890 context.transfer_fd(file.as_fd().try_clone_to_owned().unwrap())?;
2891 Ok((Pid::this(), aliases))
2892 },
2893 &mut |state| {
2894 assert_eq!(Pid::parent(), parent);
2895 (state, ())
2896 },
2897 )
2898 .unwrap();
2899 let pid = actual_child.unwrap();
2900 assert_eq!(handle.cleanup().child_pid(), pid);
2901 let result = owned_complete(handle);
2902 assert_eq!(result.status(), ExitStatus::Exited(0));
2903 assert_eq!(result.decode().unwrap(), (pid, 0));
2904 assert_reaped(pid);
2905 }
2906
2907 #[test]
2908 fn owned_parent_callback_unwind_reaps_child_and_retains_factory() {
2909 if crate::test_runs_in_own_process() {
2910 return;
2911 }
2912 struct FactoryCapture {
2913 shared: *mut SharedDropState,
2914 parent: Pid,
2915 }
2916 impl Drop for FactoryCapture {
2917 fn drop(&mut self) {
2918 assert!(
2919 !unsafe { &*self.shared }
2920 .finished
2921 .swap(true, Ordering::SeqCst)
2922 );
2923 assert_eq!(Pid::this(), self.parent, "factory capture belongs to O");
2924 }
2925 }
2926
2927 let (mapping, shared) = new_shared_drop_state();
2928 let capture = FactoryCapture {
2929 shared,
2930 parent: Pid::this(),
2931 };
2932 let actual_child = std::cell::Cell::new(None);
2933 let original_pidfd = std::cell::Cell::new(None);
2934 let observer_pidfd = std::cell::RefCell::new(None);
2935 let child_ref = &actual_child;
2936 let original_ref = &original_pidfd;
2937 let observer_ref = &observer_pidfd;
2938 let mut parent_start = move |context: ParentStartContext<'_>| -> Result<(), StartupError> {
2939 let _keep = &capture;
2940 child_ref.set(Some(context.child_pid()));
2941 original_ref.set(Some(context.child_pidfd().as_raw_fd()));
2942 *observer_ref.borrow_mut() = Some(context.child_pidfd().try_clone_to_owned().unwrap());
2944 panic!("owned startup parent panic");
2945 };
2946 let caught = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
2947 Container::new().run_with_startup_owned(
2948 Duration::from_secs(2),
2949 &mut parent_start,
2950 &mut |_| Ok(()),
2951 &mut |()| {
2952 unsafe { &*shared }.started.store(true, Ordering::Release);
2953 ((), ())
2954 },
2955 )
2956 }));
2957 let panic = match caught {
2958 Err(panic) => panic,
2959 Ok(_) => panic!("the borrowed parent callback must unwind through the API"),
2960 };
2961 assert_eq!(
2962 panic.downcast_ref::<&str>(),
2963 Some(&"owned startup parent panic")
2964 );
2965 let pid = actual_child
2966 .get()
2967 .expect("real child recorded before panic");
2968 let fd = original_pidfd.get().unwrap();
2969 assert_eq!(unsafe { libc::fcntl(fd, libc::F_GETFD) }, -1);
2970 assert_eq!(Errno::last(), Errno::EBADF, "library pidfd must be closed");
2971 let observer = observer_pidfd.borrow_mut().take().unwrap();
2972 let mut poll = libc::pollfd {
2973 fd: observer.as_raw_fd(),
2974 events: libc::POLLIN,
2975 revents: 0,
2976 };
2977 assert_eq!(unsafe { libc::poll(&mut poll, 1, 0) }, 1);
2978 assert_ne!(poll.revents & (libc::POLLIN | libc::POLLHUP), 0);
2979 assert_eq!(poll.revents & libc::POLLNVAL, 0);
2980 assert_reaped(pid);
2981 assert!(!unsafe { &*shared }.started.load(Ordering::Acquire));
2982 assert!(!unsafe { &*shared }.finished.load(Ordering::Acquire));
2983 eprintln!(
2984 "owned parent unwind: child={pid} exit_events={} reaped=true library_pidfd_closed=true factory_retained=true workload_started=false",
2985 poll.revents
2986 );
2987 drop(parent_start);
2988 assert!(unsafe { &*shared }.finished.load(Ordering::Acquire));
2989 drop(observer);
2990 unsafe { unmap_shared_drop_state(mapping, shared) };
2991 }
2992
2993 #[test]
2994 fn owned_startup_namespace_and_filter_are_unchanged() {
2995 if crate::test_runs_in_own_process() {
2996 return;
2997 }
2998 let result = Container::new()
2999 .unshare(Namespace::USER | Namespace::PID)
3000 .run_with_startup_owned(
3001 Duration::from_secs(2),
3002 &mut |_| Ok(()),
3003 &mut |_| Ok(()),
3004 &mut |()| (namespace_population_probe(), ()),
3005 )
3006 .unwrap();
3007 assert_eq!(owned_complete(result).decode().unwrap(), (1, 2, 3));
3008 let filter = seccomp::FilterBuilder::new()
3009 .default_action(seccomp::Action::Allow)
3010 .syscalls([
3011 (
3012 syscalls::Sysno::sendmsg,
3013 seccomp::Action::Errno(Errno::EPERM),
3014 ),
3015 (
3016 syscalls::Sysno::recvmsg,
3017 seccomp::Action::Errno(Errno::EPERM),
3018 ),
3019 (
3020 syscalls::Sysno::getppid,
3021 seccomp::Action::Errno(Errno::EPERM),
3022 ),
3023 ])
3024 .build();
3025 let result = Container::new()
3026 .seccomp(filter)
3027 .run_with_startup_owned(
3028 Duration::from_secs(2),
3029 &mut |_| Ok(()),
3030 &mut |_| Ok(()),
3031 &mut |()| {
3032 (
3033 Errno::result(unsafe { libc::syscall(libc::SYS_getppid) }),
3034 (),
3035 )
3036 },
3037 )
3038 .unwrap();
3039 assert_eq!(owned_complete(result).decode().unwrap(), Err(Errno::EPERM));
3040 }
3041
3042 #[test]
3043 fn owned_startup_large_result_drains_before_pending_teardown() {
3044 if crate::test_runs_in_own_process() {
3045 return;
3046 }
3047 let (mapping, shared) = new_shared_drop_state();
3048 let handle = Container::new()
3049 .run_with_startup_owned(
3050 Duration::from_secs(2),
3051 &mut |_| Ok(()),
3052 &mut |_| Ok(()),
3053 &mut |()| (vec![37_u8; 10 * 1024 * 1024], BlockingDrop { shared }),
3054 )
3055 .unwrap();
3056 let bytes = handle.provisional_bytes().to_vec();
3057 let capacity = OWNED_RESULT_PIPE_CAPACITY
3058 .with(|capacity| capacity.get())
3059 .unwrap();
3060 assert!(capacity > 0);
3061 assert!(bytes.len() > capacity as usize);
3062 eprintln!(
3063 "owned result pipe capacity={capacity} encoded_bytes={}",
3064 bytes.len()
3065 );
3066 let pid = handle.cleanup().child_pid();
3067 let pending = match handle.finalize_until(Instant::now() + Duration::from_millis(20)) {
3068 OwnedFinalize::Pending(run) => run,
3069 other => panic!("blocked teardown unexpectedly {}", outcome_kind(&other)),
3070 };
3071 assert_eq!(pending.cleanup().child_pid(), pid);
3072 assert!(pending.result_eof());
3073 assert_eq!(pending.provisional_bytes(), bytes);
3074 unsafe { &*shared }.release.store(true, Ordering::Release);
3075 let result = match pending.retry_until(Instant::now() + Duration::from_secs(2)) {
3076 OwnedFinalize::Complete(result) => result,
3077 other => panic!("released child unexpectedly {}", outcome_kind(&other)),
3078 };
3079 assert_eq!(result.encoded_bytes(), bytes);
3080 assert_eq!(result.decode().unwrap(), vec![37_u8; 10 * 1024 * 1024]);
3081 assert_reaped(pid);
3082 unsafe { unmap_shared_drop_state(mapping, shared) };
3083 }
3084
3085 static OWNED_DECODE_COUNT: std::sync::atomic::AtomicUsize =
3086 std::sync::atomic::AtomicUsize::new(0);
3087 #[derive(Debug, serde::Serialize)]
3088 struct OwnedDecodeProbe(u32);
3089 impl<'de> serde::Deserialize<'de> for OwnedDecodeProbe {
3090 fn deserialize<D: serde::Deserializer<'de>>(deserializer: D) -> Result<Self, D::Error> {
3091 OWNED_DECODE_COUNT.fetch_add(1, Ordering::SeqCst);
3092 Ok(Self(<u32 as serde::Deserialize>::deserialize(
3093 deserializer,
3094 )?))
3095 }
3096 }
3097
3098 #[test]
3099 fn owned_pending_result_never_constructs_generic_value() {
3100 if crate::test_runs_in_own_process() {
3101 return;
3102 }
3103 OWNED_DECODE_COUNT.store(0, Ordering::SeqCst);
3104 let (mapping, shared) = new_shared_drop_state();
3105 let handle = Container::new()
3106 .run_with_startup_owned(
3107 Duration::from_secs(2),
3108 &mut |_| Ok(()),
3109 &mut |_| Ok(()),
3110 &mut |()| (OwnedDecodeProbe(91), BlockingDrop { shared }),
3111 )
3112 .unwrap();
3113 assert_eq!(OWNED_DECODE_COUNT.load(Ordering::SeqCst), 0);
3114 let pending = match handle.finalize_until(Instant::now()) {
3115 OwnedFinalize::Pending(run) => run,
3116 _ => panic!("child should still own deferred teardown"),
3117 };
3118 assert_eq!(OWNED_DECODE_COUNT.load(Ordering::SeqCst), 0);
3119 unsafe { &*shared }.release.store(true, Ordering::Release);
3120 let result = match pending.retry_until(Instant::now() + Duration::from_secs(2)) {
3121 OwnedFinalize::Complete(result) => result,
3122 _ => panic!("actual child should complete"),
3123 };
3124 assert_eq!(OWNED_DECODE_COUNT.load(Ordering::SeqCst), 0);
3125 assert_eq!(result.decode().unwrap().0, 91);
3126 assert_eq!(OWNED_DECODE_COUNT.load(Ordering::SeqCst), 1);
3127 unsafe { unmap_shared_drop_state(mapping, shared) };
3128 }
3129
3130 #[test]
3131 fn owned_lost_wait_status_is_physical_exit_not_success() {
3132 if crate::test_runs_in_own_process() {
3133 return;
3134 }
3135 let handle = Container::new()
3136 .run_with_startup_owned(
3137 Duration::from_secs(2),
3138 &mut |_| Ok(()),
3139 &mut |_| Ok(()),
3140 &mut |()| (42_u32, ()),
3141 )
3142 .unwrap();
3143 let pid = handle.cleanup().child_pid();
3144 let bytes = handle.provisional_bytes().to_vec();
3145 let mut status = 0;
3147 assert_eq!(
3148 unsafe { libc::waitpid(pid.as_raw(), &mut status, 0) },
3149 pid.as_raw()
3150 );
3151 assert_eq!(ExitStatus::from_raw(status), ExitStatus::Exited(0));
3152 match handle.finalize_until(Instant::now() + Duration::from_secs(2)) {
3153 OwnedFinalize::Failed { cause, cleanup } => {
3154 assert_eq!(cause, OwnedRunFailure::WaitStatusUnavailable);
3155 assert_eq!(
3156 cleanup.cleanup().observation(),
3157 ChildCleanupObservation::ExitedWithoutWaitStatus
3158 );
3159 assert_eq!(cleanup.cleanup().last_error(), Some(Errno::ECHILD));
3160 assert_eq!(cleanup.provisional_bytes(), bytes);
3161 assert_reaped(pid);
3162 }
3163 _ => panic!("lost wait status must never qualify"),
3164 }
3165 }
3166
3167 #[test]
3168 fn owned_startup_before_parent_failure_retains_cleanup_authority() {
3169 if crate::test_runs_in_own_process() {
3170 return;
3171 }
3172 let (mapping, shared) = new_shared_drop_state();
3173 OWNED_STARTUP_CANCEL_ERROR.with(|v| v.set(Some(Errno::EPERM)));
3174 let mut called = false;
3175 let failed = Container::new()
3176 .run_with_startup_owned(
3177 Duration::from_millis(50),
3178 &mut |_| {
3179 called = true;
3180 Ok(())
3181 },
3182 &mut |_| {
3183 while !unsafe { &*shared }.release.load(Ordering::Acquire) {
3184 unsafe { libc::sched_yield() };
3185 }
3186 Ok(())
3187 },
3188 &mut |()| ((), ()),
3189 )
3190 .unwrap_err();
3191 assert!(!called);
3192 match failed {
3193 StartupOwnedFailure::AfterClone { cause, run } => {
3194 assert_eq!(cause, OwnedRunFailure::Startup(StartupError::TimedOut));
3195 assert_eq!(run.cleanup().last_error(), Some(Errno::EPERM));
3196 assert_eq!(
3197 run.cleanup().observation(),
3198 ChildCleanupObservation::Pending
3199 );
3200 let pid = run.cleanup().child_pid();
3201 assert_eq!(unsafe { libc::kill(pid.as_raw(), 0) }, 0);
3202 let fd = run.cleanup().pidfd.as_ref().unwrap().as_raw_fd();
3203 match run.retry_until(Instant::now() + Duration::from_secs(2)) {
3204 OwnedFinalize::Failed {
3205 cause: again,
3206 cleanup,
3207 } => {
3208 assert_eq!(cause, again);
3209 assert_eq!(cleanup.cleanup().pidfd.as_ref().unwrap().as_raw_fd(), fd);
3210 assert_eq!(
3211 cleanup.cleanup().observation(),
3212 ChildCleanupObservation::Reaped(ExitStatus::Signaled(
3213 Signal::SIGKILL,
3214 false
3215 ))
3216 );
3217 assert_reaped(pid);
3218 }
3219 _ => panic!("cleanup must not erase first refusal"),
3220 }
3221 }
3222 _ => panic!("a real child was created"),
3223 }
3224 unsafe { unmap_shared_drop_state(mapping, shared) };
3225 }
3226
3227 #[test]
3228 fn owned_signal_esrch_is_not_a_terminal_status() {
3229 if crate::test_runs_in_own_process() {
3230 return;
3231 }
3232 let (mapping, shared) = new_shared_drop_state();
3233 let mut handle = Container::new()
3234 .run_with_startup_owned(
3235 Duration::from_secs(2),
3236 &mut |_| Ok(()),
3237 &mut |_| Ok(()),
3238 &mut |()| (7_u32, BlockingDrop { shared }),
3239 )
3240 .unwrap();
3241 let child = handle.inner.child.as_mut().unwrap();
3242 child.signal_error_once = Some(Errno::ESRCH);
3243 assert_eq!(
3244 child.cancel_and_wait_until(Instant::now() + Duration::from_millis(20)),
3245 ChildCleanupObservation::Pending
3246 );
3247 assert_eq!(child.last_error(), Some(Errno::ESRCH));
3248 unsafe { &*shared }.release.store(true, Ordering::Release);
3249 assert_eq!(owned_complete(handle).decode().unwrap(), 7);
3250 unsafe { unmap_shared_drop_state(mapping, shared) };
3251 }
3252
3253 #[test]
3254 fn owned_result_read_error_retains_same_fd_partial_bytes_and_child() {
3255 if crate::test_runs_in_own_process() {
3256 return;
3257 }
3258 let (mapping, shared) = new_shared_drop_state();
3259 let (reader, writer) = pipe().unwrap();
3260 let reader_fd = reader.as_raw_fd();
3261 let writer_fd = writer.as_raw_fd();
3262 let mut stack = child_stack().unwrap();
3263 let child = super::super::clone::clone_with_stack_owned(
3264 || {
3265 unsafe { libc::close(reader_fd) };
3266 let mut out = Fd::new(writer_fd);
3267 out.write_all(b"abc").unwrap();
3268 unsafe { &*shared }.started.store(true, Ordering::Release);
3269 while !unsafe { &*shared }.release.load(Ordering::Acquire) {
3270 unsafe { libc::sched_yield() };
3271 }
3272 0
3273 },
3274 Namespace::empty(),
3275 &mut stack,
3276 )
3277 .unwrap();
3278 drop(writer);
3279 let mut owned = OwnedFinalization::<u32>::new(OwnedContainerCleanup::new(child), reader);
3280 let deadline = Instant::now() + Duration::from_secs(2);
3281 while !unsafe { &*shared }.started.load(Ordering::Acquire) && Instant::now() < deadline {
3282 std::thread::yield_now();
3283 }
3284 assert!(unsafe { &*shared }.started.load(Ordering::Acquire));
3285 owned.reader.as_ref().unwrap().set_nonblocking().unwrap();
3286 let identity = std::fs::read_link(format!("/proc/self/fd/{reader_fd}")).unwrap();
3287 let cause = owned.drain().unwrap_err();
3288 assert_eq!(cause, OwnedRunFailure::ResultRead(Errno::EAGAIN));
3289 assert_eq!(owned.bytes, b"abc");
3290 owned.child.as_mut().unwrap().signal_error_once = Some(Errno::EPERM);
3291 let run = match owned.fail(cause, Instant::now()) {
3292 StartupOwnedFailure::AfterClone { run, .. } => run,
3293 _ => unreachable!(),
3294 };
3295 assert_eq!(run.reader.as_ref().unwrap().as_raw_fd(), reader_fd);
3296 assert_eq!(
3297 std::fs::read_link(format!("/proc/self/fd/{reader_fd}")).unwrap(),
3298 identity
3299 );
3300 assert_eq!(run.provisional_bytes(), b"abc");
3301 assert!(!run.result_eof());
3302 match run.retry_until(Instant::now() + Duration::from_secs(2)) {
3303 OwnedFinalize::Failed {
3304 cause: again,
3305 cleanup,
3306 } => {
3307 assert_eq!(again, cause);
3308 assert_eq!(cleanup.provisional_bytes(), b"abc");
3309 assert!(matches!(
3310 cleanup.cleanup().observation(),
3311 ChildCleanupObservation::Reaped(_)
3312 ));
3313 assert_eq!(cleanup.reader.as_ref().unwrap().as_raw_fd(), reader_fd);
3314 drop(cleanup);
3315 }
3316 _ => panic!("partial failed result cannot become successful"),
3317 }
3318 assert_eq!(unsafe { libc::fcntl(reader_fd, libc::F_GETFD) }, -1);
3319 assert_eq!(Errno::last(), Errno::EBADF);
3320 unsafe { unmap_shared_drop_state(mapping, shared) };
3321 }
3322
3323 #[test]
3324 fn owned_atomic_clone_refusals_do_not_start_work() {
3325 if crate::test_runs_in_own_process() {
3326 return;
3327 }
3328 use super::super::clone::OwnedCloneTestFault;
3329 use super::super::clone::clone_with_stack_owned;
3330 let _fault = OwnedCloneFaultGuard::install(OwnedCloneTestFault::Probe(Errno::ENOSYS));
3331 let result = Container::new().run_with_startup_owned(
3332 Duration::from_secs(2),
3333 &mut |_| panic!("no parent callback"),
3334 &mut |_| -> Result<(), StartupError> { panic!("no child callback") },
3335 &mut |()| ((), ()),
3336 );
3337 assert!(matches!(
3338 result,
3339 Err(StartupOwnedFailure::BeforeClone {
3340 cause: StartupError::Io(Errno::ENOSYS)
3341 })
3342 ));
3343 drop(_fault);
3344 for flag in [
3345 libc::CLONE_VM,
3346 libc::CLONE_FILES,
3347 libc::CLONE_FS,
3348 libc::CLONE_SIGHAND,
3349 libc::CLONE_PARENT_SETTID,
3350 libc::CLONE_THREAD,
3351 libc::CLONE_VFORK,
3352 libc::CLONE_PARENT,
3353 libc::CLONE_DETACHED,
3354 ] {
3355 let mut stack = child_stack().unwrap();
3356 let result = clone_with_stack_owned(
3357 || panic!("invalid flags must not clone"),
3358 Namespace::from_bits_retain(flag),
3359 &mut stack,
3360 );
3361 assert!(matches!(result, Err(Errno::EINVAL)));
3362 }
3363 for atomic in [false, true] {
3366 let errno = Container::new()
3367 .run(|| {
3368 if atomic {
3369 let _fault =
3370 OwnedCloneFaultGuard::install(OwnedCloneTestFault::ExhaustAtClone);
3371 let mut stack = child_stack().unwrap();
3372 clone_with_stack_owned(|| 0, Namespace::empty(), &mut stack)
3373 .err()
3374 .unwrap()
3375 } else {
3376 let limit = libc::rlimit {
3377 rlim_cur: 0,
3378 rlim_max: 0,
3379 };
3380 assert_eq!(unsafe { libc::setrlimit(libc::RLIMIT_NOFILE, &limit) }, 0);
3381 let mut stack = child_stack().unwrap();
3382 clone_with_stack_owned(|| 0, Namespace::empty(), &mut stack)
3383 .err()
3384 .unwrap()
3385 }
3386 })
3387 .unwrap();
3388 assert_eq!(errno, Errno::EMFILE);
3389 }
3390 }
3391
3392 #[test]
3393 fn owned_missing_atomic_pidfd_refuses_permission_without_fake_owner() {
3394 if crate::test_runs_in_own_process() {
3395 return;
3396 }
3397 let _fault =
3398 OwnedCloneFaultGuard::install(super::super::clone::OwnedCloneTestFault::MissingPidfd);
3399 let (mapping, shared) = new_shared_drop_state();
3400 let result = Container::new().run_with_startup_owned(
3401 Duration::from_millis(50),
3402 &mut |_| panic!("invalid atomic owner must not authorize parent setup"),
3403 &mut |_| Ok(()),
3404 &mut |()| {
3405 unsafe { &*shared }.started.store(true, Ordering::Release);
3406 ((), ())
3407 },
3408 );
3409 match result {
3410 Err(StartupOwnedFailure::AfterClone { cause, run }) => {
3411 assert_eq!(cause, OwnedRunFailure::Startup(StartupError::Protocol));
3412 assert!(run.cleanup().pidfd.is_none());
3413 let pid = run.cleanup().child_pid();
3414 drop(run);
3415 assert_reaped(pid);
3416 }
3417 _ => panic!("missing atomic output must refuse"),
3418 }
3419 assert!(!unsafe { &*shared }.started.load(Ordering::Acquire));
3420 unsafe { unmap_shared_drop_state(mapping, shared) };
3421 }
3422
3423 struct OwnedExitDrop(i32);
3424 impl Drop for OwnedExitDrop {
3425 fn drop(&mut self) {
3426 unsafe { libc::_exit(self.0) }
3427 }
3428 }
3429 struct OwnedSignalDrop;
3430 impl Drop for OwnedSignalDrop {
3431 fn drop(&mut self) {
3432 unsafe { libc::raise(libc::SIGPIPE) };
3433 }
3434 }
3435
3436 #[test]
3437 fn owned_nonzero_and_signal_after_result_remain_failures() {
3438 if crate::test_runs_in_own_process() {
3439 return;
3440 }
3441 let run = Container::new()
3442 .run_with_startup_owned(
3443 Duration::from_secs(2),
3444 &mut |_| Ok(()),
3445 &mut |_| Ok(()),
3446 &mut |()| (123_u32, OwnedExitDrop(73)),
3447 )
3448 .unwrap();
3449 match run.finalize_until(Instant::now() + Duration::from_secs(2)) {
3450 OwnedFinalize::Failed { cause, cleanup } => {
3451 assert_eq!(cause, OwnedRunFailure::ChildStatus(ExitStatus::Exited(73)));
3452 assert!(!cleanup.provisional_bytes().is_empty());
3453 }
3454 _ => panic!("nonzero cleanup cannot yield the provisional result"),
3455 }
3456 let run = Container::new()
3457 .run_with_startup_owned(
3458 Duration::from_secs(2),
3459 &mut |_| Ok(()),
3460 &mut |_| Ok(()),
3461 &mut |()| (123_u32, OwnedSignalDrop),
3462 )
3463 .unwrap();
3464 match run.finalize_until(Instant::now() + Duration::from_secs(2)) {
3465 OwnedFinalize::Failed { cause, .. } => assert_eq!(
3466 cause,
3467 OwnedRunFailure::ChildStatus(ExitStatus::Signaled(Signal::SIGPIPE, false))
3468 ),
3469 _ => panic!("SIGPIPE cannot become serde error or exit zero"),
3470 }
3471 }
3472
3473 #[test]
3474 fn owned_closed_result_reader_observes_actual_sigpipe() {
3475 if crate::test_runs_in_own_process() {
3476 return;
3477 }
3478 let (reader, writer) = pipe().unwrap();
3479 let rfd = reader.as_raw_fd();
3480 let wfd = writer.as_raw_fd();
3481 let (mapping, shared) = new_shared_drop_state();
3482 let mut stack = child_stack().unwrap();
3483 let child = super::super::clone::clone_with_stack_owned(
3484 || {
3485 unsafe { libc::close(rfd) };
3486 unsafe { reset_signal_handling() }.unwrap();
3487 while !unsafe { &*shared }.release.load(Ordering::Acquire) {
3488 unsafe { libc::sched_yield() };
3489 }
3490 Fd::new(wfd)
3491 .write_all(b"ordinary serialized output")
3492 .unwrap();
3493 0
3494 },
3495 Namespace::empty(),
3496 &mut stack,
3497 )
3498 .unwrap();
3499 let mut owner = OwnedContainerCleanup::new(child);
3500 drop(reader);
3501 drop(writer);
3502 unsafe { &*shared }.release.store(true, Ordering::Release);
3503 assert_eq!(
3504 owner.wait_until(Instant::now() + Duration::from_secs(2)),
3505 ChildCleanupObservation::Reaped(ExitStatus::Signaled(Signal::SIGPIPE, false))
3506 );
3507 unsafe { unmap_shared_drop_state(mapping, shared) };
3508 }
3509
3510 #[test]
3511 fn owned_invalid_request_and_permission_never_enter_work() {
3512 if crate::test_runs_in_own_process() {
3513 return;
3514 }
3515 use StartupTestFault::*;
3516 for fault in [
3517 RequestEmptyTrailing,
3518 RequestDuplicate,
3519 RequestMalformed,
3520 RequestTrailingRights,
3521 PermissionEmptyTrailing,
3522 PermissionDuplicate,
3523 PermissionMalformed,
3524 PermissionTrailingRights,
3525 ] {
3526 let _fault = StartupFaultGuard::install(fault);
3527 let (mapping, shared) = new_shared_drop_state();
3528 let result = Container::new().run_with_startup_owned(
3529 Duration::from_secs(2),
3530 &mut |_| Ok(()),
3531 &mut |_| Ok(()),
3532 &mut |()| {
3533 unsafe { &*shared }.started.store(true, Ordering::Release);
3534 ((), ())
3535 },
3536 );
3537 match result {
3538 Err(StartupOwnedFailure::AfterClone { run, .. }) => drop(run),
3539 Ok(run) => match run.finalize_until(Instant::now() + Duration::from_secs(2)) {
3540 OwnedFinalize::Complete(_) => panic!("corrupt permission cannot complete"),
3541 OwnedFinalize::Failed { .. } => (),
3542 OwnedFinalize::Pending(_) => panic!("corrupt child did not settle"),
3543 },
3544 _ => panic!("real child must exist"),
3545 }
3546 assert!(!unsafe { &*shared }.started.load(Ordering::Acquire));
3547 unsafe { unmap_shared_drop_state(mapping, shared) };
3548 }
3549 }
3550
3551 #[test]
3552 fn owned_persistent_signal_refusal_still_observes_independent_exit() {
3553 if crate::test_runs_in_own_process() {
3554 return;
3555 }
3556 let filter = seccomp::FilterBuilder::new()
3559 .default_action(seccomp::Action::Allow)
3560 .syscalls([(
3561 syscalls::Sysno::pidfd_send_signal,
3562 seccomp::Action::Errno(Errno::EPERM),
3563 )])
3564 .build();
3565 let observed = Container::new()
3566 .seccomp(filter)
3567 .run(|| {
3568 let (mapping, shared) = new_shared_drop_state();
3569 let mut stack = child_stack().unwrap();
3570 let child = super::super::clone::clone_with_stack_owned(
3571 || {
3572 while !unsafe { &*shared }.release.load(Ordering::Acquire) {
3573 unsafe { libc::sched_yield() };
3574 }
3575 17
3576 },
3577 Namespace::empty(),
3578 &mut stack,
3579 )
3580 .unwrap();
3581 let pid = child.pid;
3582 let mut owner = OwnedContainerCleanup::new(child);
3583 let fd = owner.pidfd.as_ref().unwrap().as_raw_fd();
3584 assert_ne!(
3585 unsafe { libc::fcntl(fd, libc::F_GETFD) } & libc::FD_CLOEXEC,
3586 0
3587 );
3588 assert_eq!(
3589 owner.cancel_and_wait_until(Instant::now() + Duration::from_millis(20)),
3590 ChildCleanupObservation::Pending
3591 );
3592 assert_eq!(owner.last_error(), Some(Errno::EPERM));
3593 unsafe { &*shared }.release.store(true, Ordering::Release);
3594 assert_eq!(
3595 owner.cancel_and_wait_until(Instant::now() + Duration::from_secs(2)),
3596 ChildCleanupObservation::Reaped(ExitStatus::Exited(17))
3597 );
3598 assert_eq!(owner.last_error(), Some(Errno::EPERM));
3599 assert_eq!(owner.child_pid(), pid);
3600 assert_reaped(pid);
3601 drop(owner);
3602 assert_eq!(unsafe { libc::fcntl(fd, libc::F_GETFD) }, -1);
3603 assert_eq!(Errno::last(), Errno::EBADF);
3604 unsafe { unmap_shared_drop_state(mapping, shared) };
3605 (17, Errno::EPERM)
3606 })
3607 .unwrap();
3608 assert_eq!(observed, (17, Errno::EPERM));
3609 }
3610
3611 #[test]
3612 fn owned_atomic_immediate_exit_has_actual_status_and_reclaims_fd() {
3613 if crate::test_runs_in_own_process() {
3614 return;
3615 }
3616 let mut stack = child_stack().unwrap();
3617 let child =
3618 super::super::clone::clone_with_stack_owned(|| 17, Namespace::empty(), &mut stack)
3619 .unwrap();
3620 let pid = child.pid;
3621 let mut owner = OwnedContainerCleanup::new(child);
3622 let fd = owner.pidfd.as_ref().unwrap().as_raw_fd();
3623 assert_ne!(
3624 unsafe { libc::fcntl(fd, libc::F_GETFD) } & libc::FD_CLOEXEC,
3625 0
3626 );
3627 assert_eq!(
3628 owner.wait_until(Instant::now() + Duration::from_secs(2)),
3629 ChildCleanupObservation::Reaped(ExitStatus::Exited(17))
3630 );
3631 assert_reaped(pid);
3632 drop(owner);
3633 assert_eq!(unsafe { libc::fcntl(fd, libc::F_GETFD) }, -1);
3634 assert_eq!(Errno::last(), Errno::EBADF);
3635 }
3636
3637 #[test]
3638 fn owned_wait_retries_actual_interrupted_pidfd_poll() {
3639 if crate::test_runs_in_own_process() {
3640 return;
3641 }
3642 let _serial = WAITPID_SIGNAL_TEST.lock().unwrap();
3643 OWNED_POLL_INTERRUPTED.store(0, Ordering::Release);
3644 let previous = unsafe {
3645 sigaction(
3646 Signal::SIGUSR2,
3647 &SigAction::new(
3648 SigHandler::Handler(ignore_test_signal),
3649 SaFlags::empty(),
3650 SigSet::empty(),
3651 ),
3652 )
3653 }
3654 .unwrap();
3655 let (mapping, shared) = new_shared_drop_state();
3656 let run = Container::new()
3657 .run_with_startup_owned(
3658 Duration::from_secs(2),
3659 &mut |_| Ok(()),
3660 &mut |_| Ok(()),
3661 &mut |()| (52_u32, BlockingDrop { shared }),
3662 )
3663 .unwrap();
3664 let pid = run.cleanup().child_pid();
3665 let waiting_thread = unsafe { libc::pthread_self() };
3666 let shared_address = shared as usize;
3667 let interrupter = std::thread::spawn(move || {
3669 let until = Instant::now() + Duration::from_secs(2);
3670 while OWNED_POLL_INTERRUPTED.load(Ordering::Acquire) == 0 && Instant::now() < until {
3671 assert_eq!(
3672 unsafe { libc::pthread_kill(waiting_thread, libc::SIGUSR2) },
3673 0
3674 );
3675 std::thread::sleep(Duration::from_millis(1));
3676 }
3677 unsafe { &*(shared_address as *mut SharedDropState) }
3678 .release
3679 .store(true, Ordering::Release);
3680 });
3681 let result = owned_complete(run);
3682 interrupter.join().unwrap();
3683 unsafe { sigaction(Signal::SIGUSR2, &previous) }.unwrap();
3684 assert!(OWNED_POLL_INTERRUPTED.load(Ordering::Acquire) > 0);
3685 assert_eq!(result.decode().unwrap(), 52);
3686 assert_reaped(pid);
3687 unsafe { unmap_shared_drop_state(mapping, shared) };
3688 }
3689
3690 struct OwnedFactoryDrop {
3691 index: usize,
3692 counters: *mut [std::sync::atomic::AtomicUsize; 8],
3693 parent: Pid,
3694 }
3695 impl Drop for OwnedFactoryDrop {
3696 fn drop(&mut self) {
3697 assert_eq!(
3698 Pid::this(),
3699 self.parent,
3700 "borrowed O factories must not be dropped in C"
3701 );
3702 let counters = unsafe { &*self.counters };
3703 assert_eq!(
3704 counters[3].load(Ordering::Acquire),
3705 1,
3706 "O worker must already be joined"
3707 );
3708 counters[self.index].fetch_add(1, Ordering::SeqCst);
3709 }
3710 }
3711 #[derive(Debug)]
3712 struct OwnedLifecycleValue(*mut [std::sync::atomic::AtomicUsize; 8]);
3713 impl serde::Serialize for OwnedLifecycleValue {
3714 fn serialize<S: serde::Serializer>(&self, serializer: S) -> Result<S::Ok, S::Error> {
3715 (unsafe { &*self.0 })[0].fetch_add(1, Ordering::SeqCst);
3716 serializer.serialize_u32(29)
3717 }
3718 }
3719 impl Drop for OwnedLifecycleValue {
3720 fn drop(&mut self) {
3721 (unsafe { &*self.0 })[2].fetch_add(1, Ordering::SeqCst);
3722 }
3723 }
3724 struct OwnedLifecycleDeferred {
3725 shared: *mut SharedDropState,
3726 counters: *mut [std::sync::atomic::AtomicUsize; 8],
3727 }
3728 impl Drop for OwnedLifecycleDeferred {
3729 fn drop(&mut self) {
3730 drop(BlockingDrop {
3731 shared: self.shared,
3732 });
3733 (unsafe { &*self.counters })[1].fetch_add(1, Ordering::SeqCst);
3734 }
3735 }
3736
3737 #[test]
3738 fn owned_borrowed_factories_remain_in_parent_until_child_and_worker_settle() {
3739 if crate::test_runs_in_own_process() {
3740 return;
3741 }
3742 use std::sync::atomic::AtomicUsize;
3743 let mapping = unsafe {
3744 libc::mmap(
3745 std::ptr::null_mut(),
3746 std::mem::size_of::<[AtomicUsize; 8]>(),
3747 libc::PROT_READ | libc::PROT_WRITE,
3748 libc::MAP_SHARED | libc::MAP_ANONYMOUS,
3749 -1,
3750 0,
3751 )
3752 };
3753 assert_ne!(mapping, libc::MAP_FAILED);
3754 let counters = mapping.cast::<[AtomicUsize; 8]>();
3755 unsafe { counters.write(std::array::from_fn(|_| AtomicUsize::new(0))) };
3756 let state = unsafe { &*counters };
3757 let (child_mapping, shared) = new_shared_drop_state();
3758 let worker_release = std::sync::Arc::new(AtomicBool::new(false));
3759 let worker = std::cell::RefCell::new(None);
3760 let parent_guard = OwnedFactoryDrop {
3761 index: 4,
3762 counters,
3763 parent: Pid::this(),
3764 };
3765 let child_guard = OwnedFactoryDrop {
3766 index: 5,
3767 counters,
3768 parent: Pid::this(),
3769 };
3770 let run_guard = OwnedFactoryDrop {
3771 index: 6,
3772 counters,
3773 parent: Pid::this(),
3774 };
3775 let worker_ref = &worker;
3776 let release_ref = &worker_release;
3777 let mut parent_start = move |_: ParentStartContext<'_>| {
3778 let _keep = &parent_guard;
3779 let release = std::sync::Arc::clone(release_ref);
3780 *worker_ref.borrow_mut() = Some(std::thread::spawn(move || {
3781 while !release.load(Ordering::Acquire) {
3782 std::thread::yield_now();
3783 }
3784 }));
3785 Ok(())
3786 };
3787 let mut child_start = move |_: &mut ChildStartContext| {
3788 let _keep = &child_guard;
3789 Ok(())
3790 };
3791 let mut run = move |()| {
3792 let _keep = &run_guard;
3793 (
3794 OwnedLifecycleValue(counters),
3795 OwnedLifecycleDeferred { shared, counters },
3796 )
3797 };
3798 let handle = Container::new()
3799 .run_with_startup_owned(
3800 Duration::from_secs(2),
3801 &mut parent_start,
3802 &mut child_start,
3803 &mut run,
3804 )
3805 .unwrap();
3806 let pending = match handle.finalize_until(Instant::now()) {
3807 OwnedFinalize::Pending(pending) => pending,
3808 _ => panic!("C is still in user teardown"),
3809 };
3810 assert_eq!(state[0].load(Ordering::Acquire), 1);
3811 for counter in &state[1..] {
3812 assert_eq!(counter.load(Ordering::Acquire), 0);
3813 }
3814 unsafe { &*shared }.release.store(true, Ordering::Release);
3815 let result = match pending.retry_until(Instant::now() + Duration::from_secs(2)) {
3816 OwnedFinalize::Complete(result) => result,
3817 _ => panic!("original child should settle"),
3818 };
3819 assert_eq!(state[1].load(Ordering::Acquire), 1);
3820 assert_eq!(state[2].load(Ordering::Acquire), 1);
3821 for counter in &state[3..] {
3822 assert_eq!(counter.load(Ordering::Acquire), 0);
3823 }
3824 worker_release.store(true, Ordering::Release);
3825 worker.borrow_mut().take().unwrap().join().unwrap();
3826 state[3].store(1, Ordering::Release);
3827 drop(parent_start);
3828 drop(child_start);
3829 drop(run);
3830 for counter in &state[..7] {
3831 assert_eq!(counter.load(Ordering::Acquire), 1);
3832 }
3833 assert_eq!(state[7].load(Ordering::Acquire), 0);
3834 assert_eq!(
3837 bincode::serde::decode_from_slice::<Result<u32, StartupError>, _>(
3838 result.encoded_bytes(),
3839 bincode::config::legacy()
3840 )
3841 .unwrap(),
3842 (Ok(29), result.encoded_bytes().len())
3843 );
3844 unsafe { unmap_shared_drop_state(child_mapping, shared) };
3845 unsafe {
3846 std::ptr::drop_in_place(counters);
3847 assert_eq!(
3848 libc::munmap(mapping, std::mem::size_of::<[AtomicUsize; 8]>()),
3849 0
3850 );
3851 }
3852 }
3853
3854 #[test]
3855 fn owned_decode_refusal_retains_exact_bytes_and_actual_status() {
3856 if crate::test_runs_in_own_process() {
3857 return;
3858 }
3859 for bytes in [
3860 vec![255],
3861 {
3862 let mut bytes = bincode::serde::encode_to_vec(
3863 Ok::<u32, StartupError>(19),
3864 bincode::config::legacy(),
3865 )
3866 .unwrap();
3867 bytes.push(88);
3868 bytes
3869 },
3870 bincode::serde::encode_to_vec(
3871 Err::<u32, StartupError>(StartupError::Protocol),
3872 bincode::config::legacy(),
3873 )
3874 .unwrap(),
3875 ] {
3876 let run = Container::new()
3877 .run_with_startup_owned(
3878 Duration::from_secs(2),
3879 &mut |_| Ok(()),
3880 &mut |_| Ok(()),
3881 &mut |()| (19_u32, ()),
3882 )
3883 .unwrap();
3884 let mut result = owned_complete(run);
3885 result.bytes = bytes.clone();
3887 let failure = result.decode().unwrap_err();
3888 assert_eq!(failure.encoded_bytes(), bytes);
3889 assert_eq!(failure.status(), ExitStatus::Exited(0));
3890 assert_eq!(
3891 failure.cause(),
3892 OwnedRunFailure::Startup(StartupError::Protocol)
3893 );
3894 }
3895 }
3896
3897 #[test]
3898 fn owned_unknown_wait_retains_payload_and_same_child_on_retry() {
3899 if crate::test_runs_in_own_process() {
3900 return;
3901 }
3902 let (mapping, shared) = new_shared_drop_state();
3903 let mut run = Container::new()
3904 .run_with_startup_owned(
3905 Duration::from_secs(2),
3906 &mut |_| Ok(()),
3907 &mut |_| Ok(()),
3908 &mut |()| (93_u32, BlockingDrop { shared }),
3909 )
3910 .unwrap();
3911 let bytes = run.provisional_bytes().to_vec();
3912 let pid = run.cleanup().child_pid();
3913 let fd = run.cleanup().pidfd.as_ref().unwrap().as_raw_fd();
3914 run.inner.child.as_mut().unwrap().wait_error_once = Some(Errno::EIO);
3917 let cleanup = match run.finalize_until(Instant::now()) {
3918 OwnedFinalize::Failed { cause, cleanup } => {
3919 assert_eq!(cause, OwnedRunFailure::Cleanup(Errno::EIO));
3920 assert_eq!(
3921 cleanup.cleanup().observation(),
3922 ChildCleanupObservation::Unknown
3923 );
3924 assert_eq!(cleanup.provisional_bytes(), bytes);
3925 assert_eq!(cleanup.cleanup().pidfd.as_ref().unwrap().as_raw_fd(), fd);
3926 assert_eq!(cleanup.cleanup().child_pid(), pid);
3927 cleanup
3928 }
3929 _ => panic!("unknown wait must retain a failed owner"),
3930 };
3931 match cleanup.retry_until(Instant::now() + Duration::from_secs(2)) {
3932 OwnedFinalize::Failed { cause, cleanup } => {
3933 assert_eq!(cause, OwnedRunFailure::Cleanup(Errno::EIO));
3934 assert_eq!(cleanup.provisional_bytes(), bytes);
3935 assert_eq!(cleanup.cleanup().child_pid(), pid);
3936 assert_eq!(cleanup.cleanup().pidfd.as_ref().unwrap().as_raw_fd(), fd);
3937 assert_eq!(
3938 cleanup.cleanup().observation(),
3939 ChildCleanupObservation::Reaped(ExitStatus::Signaled(Signal::SIGKILL, false))
3940 );
3941 assert_reaped(pid);
3942 }
3943 _ => panic!("settlement must not erase original wait failure"),
3944 }
3945 unsafe { unmap_shared_drop_state(mapping, shared) };
3946 }
3947
3948 struct OwnedExternalFactory {
3949 parent: Pid,
3950 joined: std::sync::Arc<AtomicBool>,
3951 drops: std::sync::Arc<std::sync::atomic::AtomicUsize>,
3952 }
3953 impl Drop for OwnedExternalFactory {
3954 fn drop(&mut self) {
3955 assert_eq!(Pid::this(), self.parent);
3956 assert!(
3957 self.joined.load(Ordering::Acquire),
3958 "factory dropped before O worker joined"
3959 );
3960 self.drops.fetch_add(1, Ordering::SeqCst);
3961 }
3962 }
3963 #[derive(Debug)]
3964 struct OwnedSerializeExit;
3965 impl serde::Serialize for OwnedSerializeExit {
3966 fn serialize<S: serde::Serializer>(&self, _serializer: S) -> Result<S::Ok, S::Error> {
3967 unsafe { libc::_exit(73) }
3969 }
3970 }
3971
3972 #[test]
3973 fn owned_post_ready_serialization_death_keeps_external_worker_and_factory() {
3974 if crate::test_runs_in_own_process() {
3975 return;
3976 }
3977 use std::sync::Arc;
3978 let release = Arc::new(AtomicBool::new(false));
3979 let joined = Arc::new(AtomicBool::new(false));
3980 let drops = Arc::new(std::sync::atomic::AtomicUsize::new(0));
3981 let factory = OwnedExternalFactory {
3982 parent: Pid::this(),
3983 joined: joined.clone(),
3984 drops: drops.clone(),
3985 };
3986 let worker = std::cell::RefCell::new(None);
3987 let worker_ref = &worker;
3988 let release_ref = &release;
3989 let mut parent_called = false;
3990 let called_ref = &mut parent_called;
3991 let mut parent = move |_: ParentStartContext<'_>| {
3992 let _keep = &factory;
3993 *called_ref = true;
3994 let release = Arc::clone(release_ref);
3995 *worker_ref.borrow_mut() = Some(std::thread::spawn(move || {
3996 while !release.load(Ordering::Acquire) {
3997 std::thread::yield_now();
3998 }
3999 }));
4000 Ok(())
4001 };
4002 let failure = Container::new()
4003 .run_with_startup_owned(
4004 Duration::from_secs(2),
4005 &mut parent,
4006 &mut |_| Ok(()),
4007 &mut |()| (OwnedSerializeExit, ()),
4008 )
4009 .unwrap_err();
4010 match failure {
4011 StartupOwnedFailure::AfterClone { cause, run } => {
4012 assert_eq!(cause, OwnedRunFailure::Startup(StartupError::MissingResult));
4013 assert_eq!(
4014 run.cleanup().observation(),
4015 ChildCleanupObservation::Reaped(ExitStatus::Exited(73))
4016 );
4017 assert!(run.provisional_bytes().is_empty());
4018 assert_eq!(drops.load(Ordering::Acquire), 0);
4019 assert!(!joined.load(Ordering::Acquire));
4020 assert!(!worker.borrow().as_ref().unwrap().is_finished());
4021 let pid = run.cleanup().child_pid();
4022 drop(run);
4023 assert_reaped(pid);
4024 }
4025 _ => panic!("post-ready child death must retain actual cleanup"),
4026 }
4027 assert_eq!(drops.load(Ordering::Acquire), 0);
4028 release.store(true, Ordering::Release);
4029 worker.borrow_mut().take().unwrap().join().unwrap();
4030 joined.store(true, Ordering::Release);
4031 drop(parent);
4032 assert!(parent_called);
4033 assert_eq!(drops.load(Ordering::Acquire), 1);
4034 }
4035
4036 #[test]
4037 fn owned_implicit_disposal_waits_before_worker_factory_and_second_clone() {
4038 if crate::test_runs_in_own_process() {
4039 return;
4040 }
4041 let filter = seccomp::FilterBuilder::new()
4044 .default_action(seccomp::Action::Allow)
4045 .syscalls([(
4046 syscalls::Sysno::pidfd_send_signal,
4047 seccomp::Action::Errno(Errno::EPERM),
4048 )])
4049 .build();
4050 let result = Container::new()
4051 .seccomp(filter)
4052 .run(|| {
4053 use std::sync::Arc;
4054 let (mapping, shared) = new_shared_drop_state();
4055 let release = Arc::new(AtomicBool::new(false));
4056 let joined = Arc::new(AtomicBool::new(false));
4057 let drops = Arc::new(std::sync::atomic::AtomicUsize::new(0));
4058 let cancel_observed = Arc::new(AtomicBool::new(false));
4059 let second_clone = Arc::new(AtomicBool::new(false));
4060 let worker = std::cell::RefCell::new(None);
4061 let worker_ref = &worker;
4062 let release_ref = &release;
4063 let factory = OwnedExternalFactory {
4064 parent: Pid::this(),
4065 joined: joined.clone(),
4066 drops: drops.clone(),
4067 };
4068 let mut parent = move |_: ParentStartContext<'_>| {
4069 let _keep = &factory;
4070 let release = Arc::clone(release_ref);
4071 *worker_ref.borrow_mut() = Some(std::thread::spawn(move || {
4072 while !release.load(Ordering::Acquire) {
4073 std::thread::yield_now();
4074 }
4075 }));
4076 Ok(())
4077 };
4078 let run = Container::new()
4079 .run_with_startup_owned(
4080 Duration::from_secs(2),
4081 &mut parent,
4082 &mut |_| Ok(()),
4083 &mut |()| (41_u32, BlockingDrop { shared }),
4084 )
4085 .unwrap();
4086 let pid = run.cleanup().child_pid();
4087 let mut pending = match run.finalize_until(Instant::now()) {
4088 OwnedFinalize::Pending(pending) => pending,
4089 _ => panic!("child must still hold its teardown gate"),
4090 };
4091 pending.child.as_mut().unwrap().cancellation_observed =
4092 Some(cancel_observed.clone());
4093 let shared_address = shared as usize;
4094 let observed = cancel_observed.clone();
4095 let observer_joined = joined.clone();
4096 let observer_drops = drops.clone();
4097 let observer_second = second_clone.clone();
4098 let controller = std::thread::spawn(move || {
4099 let deadline = Instant::now() + Duration::from_secs(2);
4100 while !observed.load(Ordering::Acquire) && Instant::now() < deadline {
4101 std::thread::yield_now();
4102 }
4103 assert!(
4104 observed.load(Ordering::Acquire),
4105 "actual cancellation was not attempted"
4106 );
4107 assert!(!observer_joined.load(Ordering::Acquire));
4108 assert_eq!(observer_drops.load(Ordering::Acquire), 0);
4109 assert!(!observer_second.load(Ordering::Acquire));
4110 unsafe { &*(shared_address as *mut SharedDropState) }
4111 .release
4112 .store(true, Ordering::Release);
4113 });
4114 drop(pending); assert_reaped(pid);
4116 assert!(unsafe { &*shared }.finished.load(Ordering::Acquire));
4117 controller.join().unwrap();
4118 assert_eq!(drops.load(Ordering::Acquire), 0);
4119 assert!(!joined.load(Ordering::Acquire));
4120 release.store(true, Ordering::Release);
4121 worker.borrow_mut().take().unwrap().join().unwrap();
4122 joined.store(true, Ordering::Release);
4123 drop(parent);
4124 assert_eq!(drops.load(Ordering::Acquire), 1);
4125 second_clone.store(true, Ordering::Release);
4128 assert_eq!(Container::new().run(|| 23), Ok(23));
4129 unsafe { unmap_shared_drop_state(mapping, shared) };
4130 23
4131 })
4132 .unwrap();
4133 assert_eq!(result, 23);
4134 }
4135
4136 #[derive(Debug)]
4137 struct OwnedStalledSerializer(*mut SharedDropState);
4138 impl serde::Serialize for OwnedStalledSerializer {
4139 fn serialize<S: serde::Serializer>(&self, _serializer: S) -> Result<S::Ok, S::Error> {
4140 unsafe { &*self.0 }.started.store(true, Ordering::Release);
4141 loop {
4142 std::thread::sleep(Duration::from_millis(1));
4143 }
4144 }
4145 }
4146
4147 #[test]
4148 fn owned_stalled_serializer_requires_outer_process_containment() {
4149 if crate::test_runs_in_own_process() {
4150 return;
4151 }
4152 let (mapping, shared) = new_shared_drop_state();
4153 let mut stack = child_stack().unwrap();
4154 let outer = super::super::clone::clone_with_stack_owned(
4158 || {
4159 assert_eq!(Pid::this().as_raw(), 1);
4160 let _never_returns = Container::new().run_with_startup_owned(
4161 Duration::from_millis(50),
4162 &mut |_| Ok(()),
4163 &mut |_| Ok(()),
4164 &mut |()| (OwnedStalledSerializer(shared), ()),
4165 );
4166 unsafe { &*shared }.finished.store(true, Ordering::Release);
4167 99
4168 },
4169 Namespace::USER | Namespace::PID,
4170 &mut stack,
4171 )
4172 .unwrap();
4173 let mut outer = OwnedContainerCleanup::new(outer);
4174 let pid = outer.child_pid();
4175 let ready_deadline = Instant::now() + Duration::from_secs(2);
4176 while !unsafe { &*shared }.started.load(Ordering::Acquire)
4177 && Instant::now() < ready_deadline
4178 {
4179 std::thread::yield_now();
4180 }
4181 assert!(unsafe { &*shared }.started.load(Ordering::Acquire));
4182 assert_eq!(
4184 outer.wait_until(Instant::now() + Duration::from_millis(100)),
4185 ChildCleanupObservation::Pending
4186 );
4187 assert!(!unsafe { &*shared }.finished.load(Ordering::Acquire));
4188 assert_eq!(
4189 outer.cancel_and_wait_until(Instant::now() + Duration::from_secs(2)),
4190 ChildCleanupObservation::Reaped(ExitStatus::Signaled(Signal::SIGKILL, false))
4191 );
4192 assert_reaped(pid);
4193 eprintln!(
4194 "stalled serializer: outer_pid={pid} actual_exit=SIGKILL after actual Pending; library result acquisition did not return"
4195 );
4196 unsafe { unmap_shared_drop_state(mapping, shared) };
4197 }
4198
4199 #[test]
4200 fn owned_provisional_cancel_reaps_actual_child_and_preserves_bytes() {
4201 if crate::test_runs_in_own_process() {
4202 return;
4203 }
4204 let (mapping, shared) = new_shared_drop_state();
4205 let run = Container::new()
4206 .run_with_startup_owned(
4207 Duration::from_secs(2),
4208 &mut |_| Ok(()),
4209 &mut |_| Ok(()),
4210 &mut |()| (77_u32, BlockingDrop { shared }),
4211 )
4212 .unwrap();
4213 let pid = run.cleanup().child_pid();
4214 let fd = run.cleanup().pidfd.as_ref().unwrap().as_raw_fd();
4215 let bytes = run.provisional_bytes().to_vec();
4216 let failed = match run.cancel_until(Instant::now() + Duration::from_secs(2)) {
4217 OwnedFinalize::Failed { cause, cleanup } => {
4218 assert_eq!(cause, OwnedRunFailure::Cancelled);
4219 assert_eq!(
4220 cleanup.cleanup().observation(),
4221 ChildCleanupObservation::Reaped(ExitStatus::Signaled(Signal::SIGKILL, false))
4222 );
4223 assert_eq!(cleanup.cleanup().child_pid(), pid);
4224 assert_eq!(cleanup.cleanup().pidfd.as_ref().unwrap().as_raw_fd(), fd);
4225 assert_eq!(cleanup.provisional_bytes(), bytes);
4226 assert!(cleanup.result_eof());
4227 cleanup
4228 }
4229 _ => panic!("explicit cancellation must remain failed"),
4230 };
4231 assert_reaped(pid);
4232 match failed.retry_until(Instant::now()) {
4233 OwnedFinalize::Failed { cause, cleanup } => {
4234 assert_eq!(cause, OwnedRunFailure::Cancelled);
4235 assert_eq!(cleanup.provisional_bytes(), bytes);
4236 }
4237 _ => panic!("cleanup must not erase cancellation"),
4238 }
4239 unsafe { unmap_shared_drop_state(mapping, shared) };
4240 }
4241
4242 #[test]
4243 fn owned_pending_cancel_stays_failed_after_refused_signal_and_exit_zero() {
4244 if crate::test_runs_in_own_process() {
4245 return;
4246 }
4247 let filter = seccomp::FilterBuilder::new()
4248 .default_action(seccomp::Action::Allow)
4249 .syscalls([(
4250 syscalls::Sysno::pidfd_send_signal,
4251 seccomp::Action::Errno(Errno::EPERM),
4252 )])
4253 .build();
4254 let result = Container::new()
4255 .seccomp(filter)
4256 .run(|| {
4257 let (mapping, shared) = new_shared_drop_state();
4258 let run = Container::new()
4259 .run_with_startup_owned(
4260 Duration::from_secs(2),
4261 &mut |_| Ok(()),
4262 &mut |_| Ok(()),
4263 &mut |()| (78_u32, BlockingDrop { shared }),
4264 )
4265 .unwrap();
4266 let pid = run.cleanup().child_pid();
4267 let fd = run.cleanup().pidfd.as_ref().unwrap().as_raw_fd();
4268 let bytes = run.provisional_bytes().to_vec();
4269 let pending = match run.finalize_until(Instant::now()) {
4270 OwnedFinalize::Pending(pending) => pending,
4271 _ => panic!("real child must remain held after result EOF"),
4272 };
4273 let failed = match pending.cancel_until(Instant::now() + Duration::from_millis(20))
4274 {
4275 OwnedFinalize::Failed { cause, cleanup } => {
4276 assert_eq!(cause, OwnedRunFailure::Cancelled);
4277 assert_eq!(
4278 cleanup.cleanup().observation(),
4279 ChildCleanupObservation::Pending
4280 );
4281 assert_eq!(cleanup.cleanup().last_error(), Some(Errno::EPERM));
4282 assert_eq!(cleanup.cleanup().child_pid(), pid);
4283 assert_eq!(cleanup.cleanup().pidfd.as_ref().unwrap().as_raw_fd(), fd);
4284 assert_eq!(cleanup.provisional_bytes(), bytes);
4285 cleanup
4286 }
4287 _ => panic!("failed bounded cancellation must retain the pending owner"),
4288 };
4289 unsafe { &*shared }.release.store(true, Ordering::Release);
4290 match failed.retry_until(Instant::now() + Duration::from_secs(2)) {
4291 OwnedFinalize::Failed { cause, cleanup } => {
4292 assert_eq!(cause, OwnedRunFailure::Cancelled);
4293 assert_eq!(
4294 cleanup.cleanup().observation(),
4295 ChildCleanupObservation::Reaped(ExitStatus::Exited(0))
4296 );
4297 assert_eq!(cleanup.cleanup().last_error(), Some(Errno::EPERM));
4298 assert_eq!(cleanup.cleanup().child_pid(), pid);
4299 assert_eq!(cleanup.cleanup().pidfd.as_ref().unwrap().as_raw_fd(), fd);
4300 assert_eq!(cleanup.provisional_bytes(), bytes);
4301 }
4302 _ => panic!("real exit zero cannot promote a cancelled run"),
4303 }
4304 assert_reaped(pid);
4305 unsafe { unmap_shared_drop_state(mapping, shared) };
4306 true
4307 })
4308 .unwrap();
4309 assert!(result);
4310 }
4311
4312 struct StartupFaultGuard;
4313
4314 impl StartupFaultGuard {
4315 fn install(fault: StartupTestFault) -> Self {
4316 STARTUP_TEST_FAULT.with(|value| {
4317 assert!(value.get() == StartupTestFault::None);
4318 value.set(fault);
4319 });
4320 Self
4321 }
4322 }
4323
4324 impl Drop for StartupFaultGuard {
4325 fn drop(&mut self) {
4326 STARTUP_TEST_FAULT.with(|value| value.set(StartupTestFault::None));
4327 }
4328 }
4329
4330 #[test]
4331 fn startup_corrupt_request_or_permission_never_runs_workload() {
4332 if crate::test_runs_in_own_process() {
4333 return;
4334 }
4335 use StartupTestFault::*;
4336 for fault in [
4337 RequestEmptyTrailing,
4338 RequestDuplicate,
4339 RequestMalformed,
4340 RequestTrailingRights,
4341 PermissionEmptyTrailing,
4342 PermissionDuplicate,
4343 PermissionMalformed,
4344 PermissionTrailingRights,
4345 ] {
4346 let _fault = StartupFaultGuard::install(fault);
4347 let (mapping, shared) = new_shared_drop_state();
4348 let mut parent_called = false;
4349 let result = Container::new().run_with_startup(
4350 Duration::from_secs(2),
4351 |_| {
4352 parent_called = true;
4353 Ok(())
4354 },
4355 |_| Ok(()),
4356 |()| {
4357 unsafe { &*shared }.started.store(true, Ordering::Release);
4358 ((), ())
4359 },
4360 );
4361 assert!(matches!(
4362 result,
4363 Err(StartupRunError::Child {
4364 cause: StartupError::Protocol,
4365 ..
4366 })
4367 ));
4368 assert_eq!(
4369 parent_called,
4370 matches!(
4371 fault,
4372 PermissionEmptyTrailing
4373 | PermissionDuplicate
4374 | PermissionMalformed
4375 | PermissionTrailingRights
4376 )
4377 );
4378 assert!(!unsafe { &*shared }.started.load(Ordering::Acquire));
4379 unsafe { unmap_shared_drop_state(mapping, shared) };
4380 }
4381 }
4382
4383 #[test]
4384 fn startup_fragmented_request_transfers_each_right_exactly_once() {
4385 if crate::test_runs_in_own_process() {
4386 return;
4387 }
4388 let _fault = StartupFaultGuard::install(StartupTestFault::Fragmented);
4389 let (pid, handle) = Container::new()
4390 .run_with_startup(
4391 Duration::from_secs(2),
4392 |mut context| {
4393 assert_eq!(context.descriptor_count(), MAX_STARTUP_FDS);
4394 for index in 0..MAX_STARTUP_FDS {
4395 assert!(context.take_fd(index).is_some());
4396 }
4397 Ok(context.child_pid())
4398 },
4399 |context| {
4400 for _ in 0..MAX_STARTUP_FDS {
4401 context.transfer_fd(std::fs::File::open("/dev/null").unwrap().into())?;
4402 }
4403 Ok(())
4404 },
4405 |()| (42, ()),
4406 )
4407 .unwrap();
4408 assert_eq!(
4409 handle.finalize_with_status(),
4410 Ok((42, ExitStatus::Exited(0)))
4411 );
4412 assert_reaped(pid);
4413 }
4414
4415 #[test]
4416 fn startup_final_permission_has_no_later_parent_deadline_validation() {
4417 if crate::test_runs_in_own_process() {
4418 return;
4419 }
4420 let _fault = StartupFaultGuard::install(StartupTestFault::ObservePermissionLate);
4425 let (pid, handle) = Container::new()
4426 .run_with_startup(
4427 Duration::from_millis(100),
4428 |context| Ok(context.child_pid()),
4429 |_| Ok(()),
4430 |()| (42, ()),
4431 )
4432 .unwrap();
4433 assert_eq!(
4434 handle.finalize_with_status(),
4435 Ok((42, ExitStatus::Exited(0)))
4436 );
4437 assert_reaped(pid);
4438 }
4439
4440 #[test]
4441 fn startup_ignored_descriptor_overflow_still_refuses_workload() {
4442 if crate::test_runs_in_own_process() {
4443 return;
4444 }
4445 let (mapping, shared) = new_shared_drop_state();
4446 let result = Container::new().run_with_startup(
4447 Duration::from_secs(2),
4448 |_| Ok(()),
4449 |context| {
4450 for _ in 0..=MAX_STARTUP_FDS {
4451 let _ = context.transfer_fd(std::fs::File::open("/dev/null").unwrap().into());
4452 }
4453 Ok(())
4454 },
4455 |()| {
4456 unsafe { &*shared }.started.store(true, Ordering::Release);
4457 ((), ())
4458 },
4459 );
4460 assert!(matches!(
4461 result,
4462 Err(StartupRunError::Child {
4463 cause: StartupError::Protocol,
4464 ..
4465 })
4466 ));
4467 assert!(!unsafe { &*shared }.started.load(Ordering::Acquire));
4468 unsafe { unmap_shared_drop_state(mapping, shared) };
4469 }
4470
4471 fn assert_reaped(pid: Pid) {
4472 let mut status = 0;
4473 assert_eq!(
4474 unsafe { libc::waitpid(pid.as_raw(), &mut status, libc::WNOHANG) },
4475 -1
4476 );
4477 assert_eq!(Errno::last(), Errno::ECHILD);
4478 }
4479
4480 #[test]
4481 fn startup_binds_parent_child_and_transfers_owned_descriptor_once() {
4482 if crate::test_runs_in_own_process() {
4483 return;
4484 }
4485 use std::os::fd::AsFd;
4486 let parent = Pid::this();
4487 let (mapping, shared) = new_shared_drop_state();
4488 let (owner, handle) = Container::new()
4489 .run_with_startup(
4490 Duration::from_secs(2),
4491 |mut context| {
4492 assert_eq!(Pid::this(), parent);
4493 assert_ne!(context.child_pid(), parent);
4494 assert!(context.child_pidfd().as_raw_fd() >= 0);
4495 assert_eq!(context.descriptor_count(), 1);
4496 let fd = context.take_fd(0).unwrap();
4497 assert!(context.take_fd(0).is_none());
4498 assert!(context.take_fd(MAX_STARTUP_FDS).is_none());
4499 assert_ne!(
4500 unsafe { libc::fcntl(fd.as_raw_fd(), libc::F_GETFD) } & libc::FD_CLOEXEC,
4501 0
4502 );
4503 let mut file = std::fs::File::from(fd);
4504 let mut contents = String::new();
4505 file.read_to_string(&mut contents).unwrap();
4506 assert!(contents.starts_with(&format!("{} ", context.child_pid())));
4507 unsafe { &*shared }.release.store(true, Ordering::Release);
4508 Ok(context.child_pid())
4509 },
4510 |context| {
4511 let file = std::fs::File::open("/proc/self/stat").unwrap();
4512 context.transfer_fd(file.as_fd().try_clone_to_owned().unwrap())?;
4513 Ok(Pid::this())
4514 },
4515 |pid| {
4516 assert!(unsafe { &*shared }.release.load(Ordering::Acquire));
4517 assert_eq!(pid, Pid::this());
4518 assert_eq!(Pid::parent(), parent);
4519 (pid, ())
4520 },
4521 )
4522 .unwrap();
4523 assert_eq!(
4524 handle.finalize_with_status(),
4525 Ok((owner, ExitStatus::Exited(0)))
4526 );
4527 assert_reaped(owner);
4528 unsafe { unmap_shared_drop_state(mapping, shared) };
4529 }
4530
4531 fn namespace_population_probe() -> (i32, i32, i32) {
4532 let root = Pid::this().as_raw();
4533 let child = unsafe { libc::fork() };
4534 assert!(child >= 0);
4535 if child == 0 {
4536 unsafe { libc::_exit(0) }
4537 }
4538 let mut status = 0;
4539 assert_eq!(unsafe { libc::waitpid(child, &mut status, 0) }, child);
4540 assert_eq!(ExitStatus::from_raw(status), ExitStatus::Exited(0));
4541 let thread = std::thread::spawn(|| unsafe { libc::syscall(libc::SYS_gettid) as i32 })
4542 .join()
4543 .unwrap();
4544 (root, child, thread)
4545 }
4546
4547 #[test]
4548 fn startup_does_not_allocate_a_child_namespace_helper_pid() {
4549 if crate::test_runs_in_own_process() {
4550 return;
4551 }
4552 let baseline = Container::new()
4553 .unshare(Namespace::USER | Namespace::PID)
4554 .run(namespace_population_probe)
4555 .unwrap();
4556 assert_eq!(baseline, (1, 2, 3));
4557 let parent = Pid::this();
4558 let ((), handle) = Container::new()
4559 .unshare(Namespace::USER | Namespace::PID)
4560 .run_with_startup(
4561 Duration::from_secs(2),
4562 |context| {
4563 assert_eq!(Pid::this(), parent);
4564 assert_ne!(context.child_pid(), Pid::from_raw(1));
4565 Ok(())
4566 },
4567 |_| {
4568 assert_eq!(Pid::this().as_raw(), 1);
4569 Ok(())
4570 },
4571 |()| (namespace_population_probe(), ()),
4572 )
4573 .unwrap();
4574 assert_eq!(
4575 handle.finalize_with_status(),
4576 Ok((baseline, ExitStatus::Exited(0)))
4577 );
4578 let wrong = Container::new()
4581 .unshare(Namespace::USER | Namespace::PID)
4582 .run(|| {
4583 std::thread::spawn(|| ()).join().unwrap();
4584 namespace_population_probe()
4585 })
4586 .unwrap();
4587 assert_eq!(wrong, (1, 3, 4));
4588 assert_ne!(wrong, baseline);
4589 }
4590
4591 #[test]
4592 fn startup_precedes_seccomp_without_widening_the_filter() {
4593 if crate::test_runs_in_own_process() {
4594 return;
4595 }
4596 use syscalls::Sysno;
4597
4598 use super::seccomp::Action;
4599 use super::seccomp::FilterBuilder;
4600 let filter = || {
4601 FilterBuilder::new()
4602 .default_action(Action::Allow)
4603 .syscalls([
4604 (Sysno::sendmsg, Action::Errno(Errno::EPERM)),
4605 (Sysno::recvmsg, Action::Errno(Errno::EPERM)),
4606 #[cfg(target_arch = "x86_64")]
4607 (Sysno::poll, Action::Errno(Errno::EPERM)),
4608 #[cfg(not(target_arch = "x86_64"))]
4610 (Sysno::ppoll, Action::Errno(Errno::EPERM)),
4611 (Sysno::getppid, Action::Errno(Errno::EPERM)),
4612 ])
4613 .build()
4614 };
4615 let denied = || Errno::result(unsafe { libc::syscall(libc::SYS_getppid) });
4616 assert_eq!(
4617 Container::new().seccomp(filter()).run(denied),
4618 Ok(Err(Errno::EPERM))
4619 );
4620 let ((), handle) = Container::new()
4621 .seccomp(filter())
4622 .run_with_startup(
4623 Duration::from_secs(2),
4624 |_| Ok(()),
4625 |_| Ok(()),
4626 |()| (denied(), ()),
4627 )
4628 .unwrap();
4629 assert_eq!(
4630 handle.finalize_with_status(),
4631 Ok((Err(Errno::EPERM), ExitStatus::Exited(0)))
4632 );
4633 }
4634
4635 #[test]
4636 fn startup_drains_large_result_before_deferred_cleanup_and_actual_wait() {
4637 if crate::test_runs_in_own_process() {
4638 return;
4639 }
4640 let (mapping, shared) = new_shared_drop_state();
4641 let (pid, handle) = Container::new()
4642 .run_with_startup(
4643 Duration::from_secs(2),
4644 |context| Ok(context.child_pid()),
4645 |_| Ok(()),
4646 |()| (vec![42u8; 10 * 1024 * 1024], BlockingDrop { shared }),
4647 )
4648 .unwrap();
4649 assert_eq!(handle.provisional(), &vec![42u8; 10 * 1024 * 1024]);
4650 let shared_ref = unsafe { &*shared };
4651 while !shared_ref.started.load(Ordering::Acquire) {
4652 unsafe { libc::sched_yield() };
4653 }
4654 assert!(!shared_ref.finished.load(Ordering::Acquire));
4655 shared_ref.release.store(true, Ordering::Release);
4656 let (value, status) = handle.finalize_with_status().unwrap();
4657 assert_eq!(value, vec![42u8; 10 * 1024 * 1024]);
4658 assert_eq!(status, ExitStatus::Exited(0));
4659 assert!(shared_ref.finished.load(Ordering::Acquire));
4660 assert_reaped(pid);
4661 unsafe { unmap_shared_drop_state(mapping, shared) };
4662 }
4663
4664 #[test]
4665 fn startup_keeps_cleanup_failure_and_drop_reap_semantics() {
4666 if crate::test_runs_in_own_process() {
4667 return;
4668 }
4669 let (pid, handle) = Container::new()
4670 .run_with_startup(
4671 Duration::from_secs(2),
4672 |context| Ok(context.child_pid()),
4673 |_| Ok(()),
4674 |()| (42, ExitDuringDrop(71)),
4675 )
4676 .unwrap();
4677 assert_eq!(handle.provisional(), &42);
4678 assert_eq!(
4679 handle.finalize_with_status(),
4680 Err(RunError::ExitStatus(ExitStatus::Exited(71)))
4681 );
4682 assert_reaped(pid);
4683 let (pid, handle) = Container::new()
4684 .run_with_startup(
4685 Duration::from_secs(2),
4686 |context| Ok(context.child_pid()),
4687 |_| Ok(()),
4688 |()| (42, ()),
4689 )
4690 .unwrap();
4691 drop(handle);
4692 assert_reaped(pid);
4693 }
4694
4695 #[test]
4696 fn startup_parent_refusal_cancels_owned_child_without_running_workload() {
4697 if crate::test_runs_in_own_process() {
4698 return;
4699 }
4700 let (mapping, shared) = new_shared_drop_state();
4701 let mut pid = None;
4702 let result = Container::new().run_with_startup(
4703 Duration::from_secs(2),
4704 |context| {
4705 pid = Some(context.child_pid());
4706 Err::<(), _>(StartupError::Refused)
4707 },
4708 |_| Ok(()),
4709 |()| {
4710 unsafe { &*shared }.started.store(true, Ordering::Release);
4711 ((), ())
4712 },
4713 );
4714 assert!(matches!(
4715 result,
4716 Err(StartupRunError::Child {
4717 cause: StartupError::Refused,
4718 status: ExitStatus::Signaled(Signal::SIGKILL, false)
4719 })
4720 ));
4721 assert!(!unsafe { &*shared }.started.load(Ordering::Acquire));
4722 assert_reaped(pid.unwrap());
4723 unsafe { unmap_shared_drop_state(mapping, shared) };
4724 }
4725
4726 #[test]
4727 fn startup_child_setup_and_callback_refusals_are_not_readiness() {
4728 if crate::test_runs_in_own_process() {
4729 return;
4730 }
4731 let temp = tempfile::tempdir().unwrap();
4732 let missing = temp.path().join("absent");
4733 let result = Container::new().current_dir(missing).run_with_startup(
4734 Duration::from_secs(2),
4735 |_| -> Result<(), StartupError> {
4736 panic!("parent hook must not see failed child setup")
4737 },
4738 |_| -> Result<(), StartupError> { panic!("child hook must not see failed setup") },
4739 |()| -> ((), ()) { panic!("workload must not run") },
4740 );
4741 assert!(matches!(
4742 result,
4743 Err(StartupRunError::Child {
4744 cause: StartupError::Setup(Error { .. }),
4745 ..
4746 })
4747 ));
4748 if let Err(StartupRunError::Child {
4749 cause: StartupError::Setup(error),
4750 ..
4751 }) = result
4752 {
4753 assert_eq!(error, Error::new(Errno::ENOENT, Context::Chdir));
4754 }
4755 let result = Container::new().run_with_startup(
4756 Duration::from_secs(2),
4757 |_| -> Result<(), StartupError> {
4758 panic!("parent hook must not see refused child setup")
4759 },
4760 |_| Err::<(), _>(StartupError::Refused),
4761 |()| ((), ()),
4762 );
4763 assert!(matches!(
4764 result,
4765 Err(StartupRunError::Child {
4766 cause: StartupError::Refused,
4767 ..
4768 })
4769 ));
4770 }
4771
4772 #[test]
4773 fn startup_premature_child_exit_retains_actual_status() {
4774 if crate::test_runs_in_own_process() {
4775 return;
4776 }
4777 let result = Container::new().run_with_startup(
4778 Duration::from_secs(2),
4779 |_| Ok(()),
4780 |_| -> Result<(), StartupError> { unsafe { libc::_exit(73) } },
4781 |()| ((), ()),
4782 );
4783 assert!(matches!(
4784 result,
4785 Err(StartupRunError::Child {
4786 cause: StartupError::PeerClosed,
4787 status: ExitStatus::Exited(73)
4788 })
4789 ));
4790 }
4791
4792 #[test]
4793 fn startup_deadline_kills_a_child_stuck_before_readiness() {
4794 if crate::test_runs_in_own_process() {
4795 return;
4796 }
4797 let result = Container::new().run_with_startup(
4798 Duration::from_millis(100),
4799 |_| Ok(()),
4800 |_| -> Result<(), StartupError> {
4801 loop {
4802 unsafe { libc::pause() };
4803 }
4804 },
4805 |()| ((), ()),
4806 );
4807 assert!(matches!(
4808 result,
4809 Err(StartupRunError::Child {
4810 cause: StartupError::TimedOut,
4811 status: ExitStatus::Signaled(Signal::SIGKILL, false)
4812 })
4813 ));
4814 }
4815
4816 #[test]
4817 fn startup_late_parent_callback_cannot_release_workload() {
4818 if crate::test_runs_in_own_process() {
4819 return;
4820 }
4821 let (mapping, shared) = new_shared_drop_state();
4822 let mut pid = None;
4823 let result = Container::new().run_with_startup(
4824 Duration::from_millis(100),
4825 |context| {
4826 pid = Some(context.child_pid());
4827 std::thread::sleep(Duration::from_millis(200));
4828 Ok(())
4829 },
4830 |_| Ok(()),
4831 |()| {
4832 unsafe { &*shared }.started.store(true, Ordering::Release);
4833 ((), ())
4834 },
4835 );
4836 assert!(matches!(
4837 result,
4838 Err(StartupRunError::Child {
4839 cause: StartupError::TimedOut,
4840 ..
4841 })
4842 ));
4843 assert!(!unsafe { &*shared }.started.load(Ordering::Acquire));
4844 assert_reaped(pid.unwrap());
4845 unsafe { unmap_shared_drop_state(mapping, shared) };
4846 }
4847
4848 #[test]
4849 fn startup_invalid_timeout_refuses_before_clone_or_callbacks() {
4850 if crate::test_runs_in_own_process() {
4851 return;
4852 }
4853 for timeout in [Duration::ZERO, Duration::MAX] {
4854 let result = Container::new().run_with_startup(
4855 timeout,
4856 |_| -> Result<(), StartupError> { panic!("no parent callback") },
4857 |_| -> Result<(), StartupError> { panic!("no child callback") },
4858 |()| ((), ()),
4859 );
4860 assert!(matches!(
4861 result,
4862 Err(StartupRunError::BeforeClone(StartupError::InvalidTimeout))
4863 ));
4864 }
4865 }
4866
4867 #[test]
4868 fn startup_protocol_rejects_malformed_wrong_phase_and_trailing_frames() {
4869 if crate::test_runs_in_own_process() {
4870 return;
4871 }
4872 let good = {
4873 let mut frame = [0u8; STARTUP_FRAME_SIZE];
4874 frame[..4].copy_from_slice(b"RVS1");
4875 frame[4] = STARTUP_READY;
4876 frame
4877 };
4878 let mut cases = vec![
4879 vec![0],
4880 good[..STARTUP_FRAME_SIZE - 1].to_vec(),
4881 [good.as_slice(), &[0]].concat(),
4882 ];
4883 let mut wrong = good;
4884 wrong[4] = 3;
4885 cases.push(wrong.to_vec());
4886 let mut wrong = good;
4887 wrong[5] = 1;
4888 cases.push(wrong.to_vec());
4889 let mut wrong = good;
4890 wrong[7] = 1;
4891 cases.push(wrong.to_vec());
4892 for frame in cases {
4893 let (a, b) = StartupSocket::pair(Instant::now() + Duration::from_secs(2)).unwrap();
4894 assert_eq!(
4895 unsafe {
4896 libc::send(
4897 a.fd.as_raw_fd(),
4898 frame.as_ptr().cast(),
4899 frame.len(),
4900 libc::MSG_NOSIGNAL,
4901 )
4902 },
4903 frame.len() as isize
4904 );
4905 a.close_write().unwrap();
4906 let result = b.receive(Some(STARTUP_READY)).and_then(|_| b.receive(None));
4907 assert!(matches!(result, Err(StartupError::Protocol)));
4908 }
4909 let (a, b) = StartupSocket::pair(Instant::now() + Duration::from_secs(2)).unwrap();
4910 a.send(STARTUP_READY, &StartupFds::default(), None).unwrap();
4911 a.send(STARTUP_READY, &StartupFds::default(), None).unwrap();
4912 a.close_write().unwrap();
4913 b.receive(Some(STARTUP_READY)).unwrap();
4914 assert!(matches!(b.receive(None), Err(StartupError::Protocol)));
4915 let (a, b) = StartupSocket::pair(Instant::now() + Duration::from_secs(2)).unwrap();
4916 drop(a);
4917 assert!(matches!(
4918 b.receive(Some(STARTUP_READY)),
4919 Err(StartupError::PeerClosed)
4920 ));
4921 }
4922
4923 #[test]
4924 fn startup_descriptor_cardinality_is_finite_and_refusal_closes_rights() {
4925 if crate::test_runs_in_own_process() {
4926 return;
4927 }
4928 let mut context = ChildStartContext {
4929 deadline: Instant::now() + Duration::from_secs(2),
4930 descriptors: StartupFds::default(),
4931 failure: None,
4932 };
4933 for _ in 0..MAX_STARTUP_FDS {
4934 context
4935 .transfer_fd(std::fs::File::open("/dev/null").unwrap().into())
4936 .unwrap();
4937 }
4938 let extra: std::os::fd::OwnedFd = std::fs::File::open("/dev/null").unwrap().into();
4939 let extra_number = extra.as_raw_fd();
4940 assert_eq!(context.transfer_fd(extra), Err(StartupError::Protocol));
4941 assert_eq!(unsafe { libc::fcntl(extra_number, libc::F_GETFD) }, -1);
4942 assert_eq!(Errno::last(), Errno::EBADF);
4943 let (a, b) = StartupSocket::pair(context.deadline).unwrap();
4944 a.send(STARTUP_REQUEST, &context.descriptors, None).unwrap();
4945 let received = b.receive(Some(STARTUP_REQUEST)).unwrap();
4946 assert_eq!(received.len, MAX_STARTUP_FDS);
4947 let numbers: Vec<_> = received
4948 .values
4949 .iter()
4950 .map(|fd| fd.as_ref().unwrap().as_raw_fd())
4951 .collect();
4952 drop(received);
4953 for fd in numbers {
4954 assert_eq!(unsafe { libc::fcntl(fd, libc::F_GETFD) }, -1);
4955 assert_eq!(Errno::last(), Errno::EBADF);
4956 }
4957 }
4958
4959 #[test]
4960 fn can_panic() {
4961 if crate::test_runs_in_own_process() {
4962 return;
4963 }
4964 let result = Container::new().run::<_, ()>(|| panic!());
4965 assert!(
4966 matches!(
4967 result,
4968 Err(RunError::ExitStatus(ExitStatus::Signaled(
4969 Signal::SIGABRT,
4970 _
4971 )))
4972 ),
4973 "Expected Err(ExitStatus(Signaled(SIGABRT, _))), got {:?}",
4974 result
4975 );
4976 }
4977
4978 #[test]
4986 fn run_child_holds_no_descriptor_of_another_test() {
4987 use std::os::fd::FromRawFd;
4988 const NAME: &std::ffi::CStr = c"reverie-process-another-tests-descriptor";
4989 let _another_tests_descriptor = std::env::var_os(crate::ISOLATED_TEST_MARKER)
4990 .is_none()
4991 .then(|| {
4992 let fd = unsafe { libc::memfd_create(NAME.as_ptr(), libc::MFD_CLOEXEC) };
4994 assert!(fd >= 0, "memfd_create: {}", std::io::Error::last_os_error());
4995 unsafe { std::os::fd::OwnedFd::from_raw_fd(fd) }
4997 });
4998 if crate::test_runs_in_own_process() {
4999 return;
5000 }
5001 let held = Container::new()
5002 .run(|| {
5003 let name = NAME.to_str().unwrap();
5004 std::fs::read_dir("/proc/self/fd")
5005 .unwrap()
5006 .filter_map(Result::ok)
5007 .filter_map(|e| std::fs::read_link(e.path()).ok())
5008 .map(|p| p.to_string_lossy().into_owned())
5009 .filter(|p| p.contains(name))
5010 .collect::<Vec<String>>()
5011 })
5012 .unwrap();
5013 assert_eq!(
5014 held,
5015 Vec::<String>::new(),
5016 "the child holds a descriptor that another test opened"
5017 );
5018 }
5019
5020 #[test]
5021 fn is_new_process() {
5022 if crate::test_runs_in_own_process() {
5023 return;
5024 }
5025 let my_pid = unsafe { libc::getpid() };
5026
5027 assert_eq!(
5028 Container::new().run(|| {
5029 assert_ne!(unsafe { libc::getpid() }, 1);
5030 assert_ne!(unsafe { libc::getpid() }, my_pid);
5031 assert_eq!(unsafe { libc::getppid() }, my_pid);
5032 }),
5033 Ok(())
5034 );
5035 }
5036
5037 #[test]
5038 fn pid_namespace() {
5039 if crate::test_runs_in_own_process() {
5040 return;
5041 }
5042 assert_eq!(
5043 Container::new()
5044 .unshare(Namespace::USER | Namespace::PID)
5045 .run(|| {
5046 assert_eq!(unsafe { libc::getpid() }, 1);
5048 }),
5049 Ok(())
5050 );
5051 }
5052
5053 #[test]
5054 fn return_value() {
5055 if crate::test_runs_in_own_process() {
5056 return;
5057 }
5058 assert_eq!(Container::new().run(|| 42), Ok(42));
5059
5060 assert_eq!(
5061 Container::new().run(|| String::from("foobar")),
5062 Ok("foobar".into())
5063 );
5064 }
5065
5066 struct BlockingDrop {
5067 shared: *mut SharedDropState,
5068 }
5069
5070 impl Drop for BlockingDrop {
5071 fn drop(&mut self) {
5072 let shared = unsafe { &*self.shared };
5073 shared.started.store(true, Ordering::Release);
5074 while !shared.release.load(Ordering::Acquire) {
5075 unsafe { libc::sched_yield() };
5076 }
5077 shared.finished.store(true, Ordering::Release);
5078 }
5079 }
5080
5081 struct SharedDropState {
5082 started: AtomicBool,
5083 release: AtomicBool,
5084 finished: AtomicBool,
5085 }
5086
5087 fn new_shared_drop_state() -> (*mut libc::c_void, *mut SharedDropState) {
5088 let mapping = unsafe {
5089 libc::mmap(
5090 std::ptr::null_mut(),
5091 std::mem::size_of::<SharedDropState>(),
5092 libc::PROT_READ | libc::PROT_WRITE,
5093 libc::MAP_SHARED | libc::MAP_ANONYMOUS,
5094 -1,
5095 0,
5096 )
5097 };
5098 assert_ne!(mapping, libc::MAP_FAILED);
5099 let shared = mapping.cast::<SharedDropState>();
5100 unsafe {
5101 shared.write(SharedDropState {
5102 started: AtomicBool::new(false),
5103 release: AtomicBool::new(false),
5104 finished: AtomicBool::new(false),
5105 });
5106 }
5107 (mapping, shared)
5108 }
5109
5110 unsafe fn unmap_shared_drop_state(mapping: *mut libc::c_void, shared: *mut SharedDropState) {
5111 unsafe {
5112 std::ptr::drop_in_place(shared);
5113 assert_eq!(
5114 libc::munmap(mapping, std::mem::size_of::<SharedDropState>()),
5115 0
5116 );
5117 }
5118 }
5119
5120 #[test]
5121 fn deferred_drop_publishes_result_before_cleanup_completes() {
5122 if crate::test_runs_in_own_process() {
5123 return;
5124 }
5125 let (mapping, shared) = new_shared_drop_state();
5126
5127 let run = Container::new()
5128 .run_with_deferred_drop(|| (42, BlockingDrop { shared }))
5129 .unwrap();
5130 assert_eq!(run.provisional(), &42);
5131
5132 let shared_ref = unsafe { &*shared };
5133 while !shared_ref.started.load(Ordering::Acquire) {
5134 unsafe { libc::sched_yield() };
5135 }
5136 shared_ref.release.store(true, Ordering::Release);
5137 assert_eq!(run.finalize(), Ok(42));
5138 assert!(shared_ref.finished.load(Ordering::Acquire));
5139
5140 unsafe { unmap_shared_drop_state(mapping, shared) };
5141 }
5142
5143 #[test]
5144 fn dropping_cleanup_handle_still_reaps_the_child() {
5145 if crate::test_runs_in_own_process() {
5146 return;
5147 }
5148 let (mapping, shared) = new_shared_drop_state();
5149 let run = Container::new()
5150 .run_with_deferred_drop(|| ((), BlockingDrop { shared }))
5151 .unwrap();
5152 let shared_ref = unsafe { &*shared };
5153 while !shared_ref.started.load(Ordering::Acquire) {
5154 unsafe { libc::sched_yield() };
5155 }
5156 shared_ref.release.store(true, Ordering::Release);
5157 drop(run);
5158 assert!(shared_ref.finished.load(Ordering::Acquire));
5159
5160 unsafe { unmap_shared_drop_state(mapping, shared) };
5161 }
5162
5163 static WAITPID_SIGNAL_TEST: Mutex<()> = Mutex::new(());
5164
5165 extern "C" fn ignore_test_signal(_signal: libc::c_int) {}
5166
5167 #[test]
5168 fn deferred_finalize_retries_an_interrupted_wait_and_reaps() {
5169 if crate::test_runs_in_own_process() {
5170 return;
5171 }
5172 let _serial = WAITPID_SIGNAL_TEST.lock().unwrap();
5173 WAITPID_ENTERED.store(false, Ordering::Release);
5174 WAITPID_INTERRUPTED.store(0, Ordering::Release);
5175
5176 let action = SigAction::new(
5177 SigHandler::Handler(ignore_test_signal),
5178 SaFlags::empty(),
5179 SigSet::empty(),
5180 );
5181 let previous = unsafe { sigaction(Signal::SIGUSR2, &action) }.unwrap();
5182
5183 let (mapping, shared) = new_shared_drop_state();
5184 let run = Container::new()
5185 .run_with_deferred_drop(|| (42, BlockingDrop { shared }))
5186 .unwrap();
5187 let child_pid = run.child.0.expect("deferred child pid");
5188 WAITPID_TEST_PID.store(child_pid.as_raw(), Ordering::Release);
5189 let waiting_thread = unsafe { libc::pthread_self() };
5190 let shared_address = shared as usize;
5191 let interrupter = std::thread::spawn(move || {
5192 while !WAITPID_ENTERED.load(Ordering::Acquire) {
5193 std::thread::yield_now();
5194 }
5195 let deadline = Instant::now() + Duration::from_secs(2);
5196 while WAITPID_INTERRUPTED.load(Ordering::Acquire) == 0 && Instant::now() < deadline {
5197 assert_eq!(
5198 unsafe { libc::pthread_kill(waiting_thread, libc::SIGUSR2) },
5199 0
5200 );
5201 std::thread::sleep(Duration::from_millis(1));
5202 }
5203 let shared = unsafe { &*(shared_address as *mut SharedDropState) };
5204 shared.release.store(true, Ordering::Release);
5205 });
5206
5207 assert_eq!(run.finalize(), Ok(42));
5208 interrupter.join().unwrap();
5209 unsafe { sigaction(Signal::SIGUSR2, &previous) }.unwrap();
5210 WAITPID_TEST_PID.store(0, Ordering::Release);
5211 assert!(
5212 WAITPID_INTERRUPTED.load(Ordering::Acquire) > 0,
5213 "the signal did not interrupt waitpid"
5214 );
5215 let mut status = 0;
5216 assert_eq!(
5217 unsafe { libc::waitpid(child_pid.as_raw(), &mut status, libc::WNOHANG) },
5218 -1
5219 );
5220 assert_eq!(Errno::last(), Errno::ECHILD);
5221
5222 unsafe { unmap_shared_drop_state(mapping, shared) };
5223 }
5224
5225 struct ExitDuringDrop(i32);
5226
5227 impl Drop for ExitDuringDrop {
5228 fn drop(&mut self) {
5229 unsafe { libc::_exit(self.0) }
5230 }
5231 }
5232
5233 #[test]
5234 fn deferred_drop_exposes_cleanup_failure() {
5235 if crate::test_runs_in_own_process() {
5236 return;
5237 }
5238 let run = Container::new()
5239 .run_with_deferred_drop(|| (42, ExitDuringDrop(71)))
5240 .unwrap();
5241
5242 assert_eq!(run.provisional(), &42);
5243 assert_eq!(
5244 run.finalize(),
5245 Err(RunError::ExitStatus(ExitStatus::Exited(71)))
5246 );
5247 }
5248
5249 #[test]
5250 fn mount_error_from_child_is_returned() {
5251 if crate::test_runs_in_own_process() {
5252 return;
5253 }
5254 let source_dir = tempfile::tempdir().unwrap();
5255 let missing_source = source_dir.path().join("missing");
5256
5257 let result = Container::new()
5258 .unshare(Namespace::USER | Namespace::MOUNT)
5259 .map_root()
5260 .mount(Mount::bind(missing_source, "/test"))
5261 .run(|| 42);
5262
5263 assert_eq!(
5264 result,
5265 Err(RunError::Spawn(Error::new(Errno::ENOENT, Context::Mount)))
5266 );
5267 }
5268
5269 #[test]
5270 fn test_directory_is_available_after_mount() {
5271 if crate::test_runs_in_own_process() {
5272 return;
5273 }
5274 let result = Container::new()
5275 .unshare(Namespace::USER | Namespace::MOUNT)
5276 .map_root()
5277 .mount(Mount::tmpfs("/test").touch_target())
5278 .run(|| Path::new("/test").is_dir());
5279
5280 assert_eq!(result, Ok(true));
5281 }
5282
5283 #[test]
5304 fn a_readonly_bind_survives_a_nosuid_nodev_source() {
5305 if crate::test_runs_in_own_process() {
5306 return;
5307 }
5308 let shm = Path::new("/dev/shm");
5309 let flags = nix::sys::statvfs::statvfs(shm).expect("statvfs /dev/shm");
5311 assert!(
5312 flags
5313 .flags()
5314 .contains(nix::sys::statvfs::FsFlags::ST_NOSUID)
5315 || flags.flags().contains(nix::sys::statvfs::FsFlags::ST_NODEV),
5316 "/dev/shm carries neither nosuid nor nodev on this host, so this test \
5317 would pass without exercising the locked-flag remount at all"
5318 );
5319
5320 let source = tempfile::tempdir_in(shm).unwrap();
5321 let target = tempfile::tempdir().unwrap();
5322
5323 let result = Container::new()
5324 .unshare(Namespace::USER | Namespace::MOUNT)
5325 .map_root()
5326 .mount(Mount::bind(source.path(), target.path()).readonly())
5327 .run(|| Path::new("/proc/self/mounts").is_file());
5328
5329 assert_eq!(
5330 result,
5331 Ok(true),
5332 "a read-only bind whose source is nosuid/nodev must not fail; \
5333 EPERM here means the remount is dropping the source's locked flags"
5334 );
5335 }
5336
5337 #[test]
5338 fn huge_return_value() {
5339 if crate::test_runs_in_own_process() {
5340 return;
5341 }
5342 assert_eq!(
5343 Container::new().run(|| {
5344 vec![42; 10 * 1024 * 1024 ]
5347 }),
5348 Ok(vec![42; 10 * 1024 * 1024])
5349 );
5350 }
5351
5352 #[test]
5353 pub fn bind_to_low_port() {
5354 if crate::test_runs_in_own_process() {
5355 return;
5356 }
5357 use std::net::Ipv4Addr;
5358 use std::net::SocketAddrV4;
5359 use std::net::TcpListener;
5360
5361 let addr = Container::new()
5362 .map_root()
5363 .local_networking_only()
5364 .run(|| {
5365 let listener = TcpListener::bind("127.0.0.1:80").unwrap();
5366 listener.local_addr().unwrap()
5367 })
5368 .unwrap();
5369
5370 assert_eq!(
5371 addr,
5372 SocketAddrV4::new(Ipv4Addr::new(127, 0, 0, 1), 80).into()
5373 );
5374 }
5375
5376 #[cfg(target_arch = "x86_64")]
5421 #[test]
5422 pub fn pin_affinity_to_all_cores() -> Result<(), Error> {
5423 if crate::test_runs_in_own_process() {
5424 return Ok(());
5425 }
5426 use std::collections::HashMap;
5427
5428 use raw_cpuid::CpuId;
5429
5430 let cpus = num_cpus::get();
5431 println!("Total cpus {}", cpus);
5432
5433 let mut results: HashMap<u32, usize> = HashMap::new();
5435 for core in 0..cpus {
5436 println!(" Launching guest with affinity set to {}", core);
5437 let mut container = Container::new();
5438 container.affinity(core);
5439 let which_core = container
5440 .run(|| {
5441 let cpuid = CpuId::new();
5442 cpuid
5445 .get_extended_topology_info()
5446 .and_then(|mut levels| levels.next())
5447 .map(|level| level.x2apic_id())
5448 })
5449 .unwrap();
5450 let which_core = which_core.unwrap_or_else(|| {
5454 panic!(
5455 "CPUID leaf 0x0B (extended topology) is unavailable, so no \
5456 32-bit x2APIC id can be read; with {cpus} CPUs the legacy \
5457 8-bit APIC id cannot distinguish them and this test cannot \
5458 decide anything"
5459 )
5460 });
5461 println!(" Guest sees its on x2APIC id {}", which_core);
5462 *results.entry(which_core).or_default() += 1;
5463 }
5464
5465 println!("Final table size {:?}", results.len());
5466 assert_eq!(
5467 results.values().fold(0, |n, v| std::cmp::max(n, *v)),
5468 1,
5469 "two guests pinned to different CPUs reported the same x2APIC id, \
5470 so affinity did not place them on distinct CPUs"
5471 );
5472 Ok(())
5473 }
5474}