Skip to main content

reverie_process/
lib.rs

1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 * All rights reserved.
4 *
5 * This source code is licensed under the BSD-style license found in the
6 * LICENSE file in the root directory of this source tree.
7 */
8
9//! A drop-in replacement for `std::process::Command` that provides the ability
10//! to set up namespaces, a seccomp filter, and more.
11
12#![deny(missing_docs)]
13#![deny(rustdoc::broken_intra_doc_links)]
14#![cfg(target_os = "linux")]
15#![cfg_attr(feature = "nightly", feature(internal_output_capture))]
16
17mod builder;
18mod child;
19mod clone;
20mod container;
21mod env;
22mod error;
23mod exit_status;
24mod fd;
25mod id_map;
26#[doc(hidden)]
27pub mod launch_window;
28mod mount;
29mod namespace;
30mod net;
31mod pid;
32mod pty;
33pub mod seccomp;
34mod spawn;
35mod stdio;
36mod util;
37
38use std::ffi::CString;
39use std::io;
40
41pub use child::Child;
42pub use child::Output;
43pub use container::ChildCleanupObservation;
44pub use container::ChildStartContext;
45pub use container::Container;
46pub use container::DeferredContainerRun;
47pub use container::MAX_STARTUP_FDS;
48pub use container::OwnedContainerCleanup;
49pub use container::OwnedDecodeFailure;
50pub use container::OwnedDeferredContainerRun;
51pub use container::OwnedFinalization;
52pub use container::OwnedFinalize;
53pub use container::OwnedReapedResult;
54pub use container::OwnedRunFailure;
55pub use container::ParentStartContext;
56pub use container::RunError;
57pub use container::StartupError;
58pub use container::StartupOwnedFailure;
59pub use container::StartupRunError;
60pub use error::Context;
61pub use error::Error;
62pub use exit_status::ExitStatus;
63pub use mount::Bind;
64pub use mount::Mount;
65pub use mount::MountFlags;
66pub use mount::MountParseError;
67pub use namespace::Namespace;
68// Re-export Signal since it is used by `Child::signal`.
69pub use nix::sys::signal::Signal;
70pub use pid::Pid;
71pub use pty::Pty;
72pub use pty::PtyChild;
73pub use stdio::ChildStderr;
74pub use stdio::ChildStdin;
75pub use stdio::ChildStdout;
76pub use stdio::Stdio;
77use syscalls::Errno;
78
79/// A builder for spawning a process.
80// See the builder.rs for documentation of each field.
81pub struct Command {
82    program: CString,
83    args: util::CStringArray,
84    pre_exec: Vec<Box<dyn FnMut() -> Result<(), Errno> + Send + Sync>>,
85    container: Container,
86}
87
88impl Command {
89    /// Converts [`std::process::Command`] into [`Command`]. Note that this is a
90    /// very basic and *lossy* conversion.
91    ///
92    /// This only preserves the
93    ///  - program path,
94    ///  - arguments,
95    ///  - environment variables,
96    ///  - and working directory.
97    ///
98    /// # Caveats
99    ///
100    /// Since [`std::process::Command`] is rather opaque and doesn't provide
101    /// access to all fields, this will *not* preserve:
102    ///  - stdio handles,
103    ///  - `env_clear`,
104    ///  - any `pre_exec` callbacks,
105    ///  - `arg0` (if not the same as `program`),
106    ///  - `uid`, `gid`, or `groups`.
107    pub fn from_std_lossy(cmd: &std::process::Command) -> Command {
108        let mut result = Command::new(cmd.get_program());
109        result.args(cmd.get_args());
110
111        for (key, value) in cmd.get_envs() {
112            match value {
113                Some(value) => result.env(key, value),
114                None => result.env_remove(key),
115            };
116        }
117
118        if let Some(dir) = cmd.get_current_dir() {
119            result.current_dir(dir);
120        }
121
122        result
123    }
124
125    /// Converts this command to [`std::process::Command`].
126    ///
127    /// This fails if the command contains container configuration that cannot
128    /// be represented by [`std::process::Command`], rather than silently
129    /// discarding that configuration. This includes namespaces, mounts,
130    /// seccomp filters, pseudoterminals, and CPU affinity.
131    pub fn try_into_std(self) -> io::Result<std::process::Command> {
132        let blockers = self.container.std_conversion_blockers();
133        if !blockers.is_empty() {
134            return Err(io::Error::new(
135                io::ErrorKind::InvalidInput,
136                format!(
137                    "cannot convert to std::process::Command without losing: {}",
138                    blockers.join(", ")
139                ),
140            ));
141        }
142
143        Ok(self.into_std())
144    }
145
146    /// Converts this command to [`std::process::Command`], refusing to discard
147    /// any container configuration.
148    ///
149    /// This compatibility shim preserves the former return type for callers
150    /// whose commands are representable. It panics instead of silently losing
151    /// unsupported configuration. New callers should use [`Self::try_into_std`]
152    /// to handle that refusal explicitly.
153    pub fn into_std_lossy(self) -> std::process::Command {
154        self.try_into_std().unwrap_or_else(|error| {
155            panic!("Command::into_std_lossy refused unsupported configuration: {error}")
156        })
157    }
158
159    fn into_std(self) -> std::process::Command {
160        use std::ffi::OsStr;
161        use std::os::unix::ffi::OsStrExt;
162
163        // Keep this exhaustive: adding Command state must fail to compile until
164        // the standard-command conversion explicitly preserves or refuses it.
165        let Self {
166            program,
167            args,
168            pre_exec,
169            container,
170        } = self;
171
172        let mut result = std::process::Command::new(OsStr::from_bytes(program.to_bytes()));
173        result.args(
174            args.iter()
175                .skip(1)
176                .map(|arg| OsStr::from_bytes(arg.to_bytes())),
177        );
178
179        if container.env.is_cleared() {
180            result.env_clear();
181        }
182
183        for (key, value) in container.get_envs() {
184            match value {
185                Some(value) => result.env(key, value),
186                None => result.env_remove(key),
187            };
188        }
189
190        if let Some(dir) = container.get_current_dir() {
191            result.current_dir(dir);
192        }
193
194        #[cfg(unix)]
195        {
196            use std::os::unix::process::CommandExt;
197
198            result.arg0(OsStr::from_bytes(args.get(0).to_bytes()));
199
200            for mut f in pre_exec {
201                unsafe {
202                    result.pre_exec(move || f().map_err(Into::into));
203                }
204            }
205        }
206
207        result.stdin(container.stdin);
208        result.stdout(container.stdout);
209        result.stderr(container.stderr);
210
211        result
212    }
213}
214
215/// Names the unit test that a fresh single-test process was started to run; it
216/// is set only in that process, never in the libtest harness process.
217#[cfg(test)]
218pub(crate) const ISOLATED_TEST_MARKER: &str = "REVERIE_PROCESS_ISOLATED_TEST";
219
220/// Runs the calling unit test in a fresh single-test process.
221///
222/// Returns `true` in the libtest harness process, after the fresh process has
223/// run the test body and passed; the caller must then return without running
224/// the body itself. Returns `false` inside the fresh process, where the caller
225/// runs the body. Use it as the first statement of every test that creates a
226/// child process:
227///
228/// ```ignore
229/// if crate::test_runs_in_own_process() {
230///     return;
231/// }
232/// ```
233///
234/// The children these tests create are made with a raw `clone` that shares
235/// neither the file-descriptor table nor glibc's `atfork` handlers, and most of
236/// them never `execve`. Such a child therefore receives a copy of every
237/// descriptor open anywhere in the process that runs the test, including the
238/// pipe ends and pidfds of tests running on other libtest threads, and a copy
239/// of every userspace lock another thread held at the instant of the clone.
240/// That is why the `Container` entry points require a single-threaded caller.
241/// The multi-threaded libtest harness breaks that requirement: another test's
242/// child can hold this test's pipe ends open (so a closed reader raises no
243/// `SIGPIPE` and a drain to end-of-file waits forever), a startup child counts
244/// another test's pidfd, a closed descriptor number is reused by another
245/// thread, and a panicking child can block forever on an allocator lock that
246/// one of libtest's own threads held at the clone. A mutex around the tests
247/// cannot stop libtest's own threads from allocating. The fresh process runs
248/// with `--test-threads=1`, where the only other thread is libtest's main
249/// thread blocked in a channel receive, which is the documented
250/// single-threaded setting.
251#[cfg(test)]
252#[must_use]
253pub(crate) fn test_runs_in_own_process() -> bool {
254    let thread = std::thread::current();
255    let name = thread
256        .name()
257        .expect("libtest names each test thread after its test")
258        .to_owned();
259    if let Some(marker) = std::env::var_os(ISOLATED_TEST_MARKER) {
260        assert_eq!(
261            marker.to_str(),
262            Some(name.as_str()),
263            "the isolated test process ran a test other than the one it was started for"
264        );
265        return false;
266    }
267    let output = std::process::Command::new(std::env::current_exe().unwrap())
268        .args([name.as_str(), "--exact", "--test-threads=1", "--nocapture"])
269        .env(ISOLATED_TEST_MARKER, &name)
270        .stdin(std::process::Stdio::null())
271        .output()
272        .unwrap();
273    let stdout = String::from_utf8_lossy(&output.stdout);
274    print!("{stdout}");
275    eprint!("{}", String::from_utf8_lossy(&output.stderr));
276    assert!(
277        output.status.success(),
278        "the isolated process for {name} failed: {:?}",
279        output.status
280    );
281    assert!(
282        stdout.contains("test result: ok. 1 passed; 0 failed;"),
283        "the isolated process for {name} did not run exactly that one test"
284    );
285    true
286}
287
288#[cfg(test)]
289mod tests {
290    use std::collections::BTreeMap;
291    use std::fs;
292    use std::path::Path;
293    use std::str::from_utf8;
294
295    use super::*;
296    use crate::ExitStatus;
297
298    #[tokio::test]
299    async fn spawn() {
300        if crate::test_runs_in_own_process() {
301            return;
302        }
303        assert_eq!(
304            Command::new("true").spawn().unwrap().wait().await.unwrap(),
305            ExitStatus::Exited(0)
306        );
307
308        assert_eq!(
309            Command::new("false").spawn().unwrap().wait().await.unwrap(),
310            ExitStatus::Exited(1)
311        );
312    }
313
314    #[test]
315    fn wait_blocking() {
316        if crate::test_runs_in_own_process() {
317            return;
318        }
319        assert_eq!(
320            Command::new("true")
321                .spawn()
322                .unwrap()
323                .wait_blocking()
324                .unwrap(),
325            ExitStatus::Exited(0)
326        );
327
328        assert_eq!(
329            Command::new("false")
330                .spawn()
331                .unwrap()
332                .wait_blocking()
333                .unwrap(),
334            ExitStatus::Exited(1)
335        );
336    }
337
338    #[tokio::test]
339    async fn spawn_fail() {
340        if crate::test_runs_in_own_process() {
341            return;
342        }
343        assert_eq!(
344            Command::new("/iprobablydonotexist").spawn().unwrap_err(),
345            Error::new(Errno::ENOENT, Context::Exec)
346        );
347    }
348
349    #[tokio::test]
350    async fn double_wait() {
351        if crate::test_runs_in_own_process() {
352            return;
353        }
354        let mut child = Command::new("true").spawn().unwrap();
355        assert_eq!(child.wait().await.unwrap(), ExitStatus::Exited(0));
356        assert_eq!(child.wait().await.unwrap(), ExitStatus::Exited(0));
357    }
358
359    #[tokio::test]
360    async fn output() {
361        if crate::test_runs_in_own_process() {
362            return;
363        }
364        let output = Command::new("echo")
365            .arg("foo")
366            .arg("bar")
367            .output()
368            .await
369            .unwrap();
370        assert_eq!(output.stdout, b"foo bar\n");
371        assert_eq!(output.stderr, b"");
372        assert_eq!(output.status, ExitStatus::Exited(0));
373    }
374
375    fn parse_proc_status(stdout: &[u8]) -> BTreeMap<&str, &str> {
376        from_utf8(stdout)
377            .unwrap()
378            .trim_end()
379            .split('\n')
380            .map(|line| {
381                let (first, second) = line.split_once(':').unwrap();
382                (first, second.trim())
383            })
384            .collect()
385    }
386
387    #[tokio::test]
388    async fn uid_namespace() {
389        if crate::test_runs_in_own_process() {
390            return;
391        }
392        let output = Command::new("cat")
393            .arg("/proc/self/status")
394            .map_root()
395            .output()
396            .await
397            .unwrap();
398        assert_eq!(output.status, ExitStatus::Exited(0));
399
400        let proc_status = parse_proc_status(&output.stdout);
401
402        // We should be root user inside of the container.
403        assert_eq!(proc_status["Uid"], "0\t0\t0\t0");
404    }
405
406    #[tokio::test]
407    async fn pid_namespace() {
408        if crate::test_runs_in_own_process() {
409            return;
410        }
411        let output = Command::new("cat")
412            .arg("/proc/self/status")
413            .map_root()
414            .unshare(Namespace::PID)
415            .output()
416            .await
417            .unwrap();
418        assert_eq!(output.status, ExitStatus::Exited(0));
419
420        let proc_status = parse_proc_status(&output.stdout);
421
422        assert_eq!(proc_status["NSpid"].split('\t').nth(1), Some("1"),);
423
424        // Note that, since we haven't mounted a fresh /proc into the container,
425        // the child still sees what the parent sees and so the PID will *not*
426        // be 1.
427        assert_ne!(proc_status["Pid"], "1");
428    }
429
430    #[tokio::test]
431    async fn mount_proc() {
432        if crate::test_runs_in_own_process() {
433            return;
434        }
435        let output = Command::new("cat")
436            .arg("/proc/self/status")
437            .map_root()
438            .unshare(Namespace::PID)
439            .mount(Mount::proc())
440            .output()
441            .await
442            .unwrap();
443        assert_eq!(output.status, ExitStatus::Exited(0));
444
445        let proc_status = parse_proc_status(&output.stdout);
446
447        // With /proc mounted, the child really believes it is the root process.
448        assert_eq!(proc_status["NSpid"], "1");
449        assert_eq!(proc_status["Pid"], "1");
450    }
451
452    #[tokio::test]
453    async fn mount_proc_with_readonly_fallback() {
454        if crate::test_runs_in_own_process() {
455            return;
456        }
457        let proc = tempfile::tempdir().unwrap();
458        let proc_path = proc.path().to_str().unwrap();
459        let output = Command::new("sh")
460            .arg("-c")
461            .arg(
462                "cat \"$REVERIE_PROC_PATH/mounts\"; echo __STATUS__; \
463                 exec cat \"$REVERIE_PROC_PATH/self/status\"",
464            )
465            .env("REVERIE_PROC_PATH", proc_path)
466            .map_root()
467            .unshare(Namespace::PID)
468            .mount(
469                Mount::new(proc.path())
470                    .fstype("proc")
471                    .allow_readonly_fallback(),
472            )
473            .output()
474            .await
475            .unwrap();
476        assert_eq!(output.status, ExitStatus::Exited(0));
477
478        let stdout = from_utf8(&output.stdout).unwrap();
479        let (mounts, status) = stdout.split_once("__STATUS__\n").unwrap();
480        let proc_options = mounts
481            .lines()
482            .find_map(|line| {
483                let mut fields = line.split_whitespace();
484                let _source = fields.next()?;
485                let target = fields.next()?;
486                let fstype = fields.next()?;
487                let options = fields.next()?;
488                (target == proc_path && fstype == "proc").then_some(options)
489            })
490            .unwrap();
491        assert!(
492            proc_options
493                .split(',')
494                .any(|option| option == "ro" || option == "rw")
495        );
496        println!("nested_proc_options={proc_options}");
497
498        let proc_status = parse_proc_status(status.as_bytes());
499        assert_eq!(proc_status["NSpid"], "1");
500        assert_eq!(proc_status["Pid"], "1");
501    }
502
503    #[tokio::test]
504    async fn hostname() {
505        if crate::test_runs_in_own_process() {
506            return;
507        }
508        let output = Command::new("cat")
509            .arg("/proc/sys/kernel/hostname")
510            .map_root()
511            .hostname("foobar.local")
512            .output()
513            .await
514            .unwrap();
515        assert_eq!(output.status, ExitStatus::Exited(0));
516
517        let hostname = from_utf8(&output.stdout).unwrap().trim();
518
519        assert_eq!(hostname, "foobar.local");
520    }
521
522    #[tokio::test]
523    async fn domainname() {
524        if crate::test_runs_in_own_process() {
525            return;
526        }
527        let output = Command::new("cat")
528            .arg("/proc/sys/kernel/domainname")
529            .map_root()
530            .domainname("foobar")
531            .output()
532            .await
533            .unwrap();
534
535        assert_eq!(output.status, ExitStatus::Exited(0));
536
537        let domainname = from_utf8(&output.stdout).unwrap().trim();
538
539        assert_eq!(domainname, "foobar");
540    }
541
542    #[tokio::test]
543    async fn pty() {
544        if crate::test_runs_in_own_process() {
545            return;
546        }
547        use tokio::io::AsyncReadExt;
548
549        let mut pty = Pty::open().unwrap();
550        let pty_child = pty.child().unwrap();
551
552        let mut tty = pty_child.terminal_params().unwrap();
553        // Prevent post-processing of output so `\n` isn't translated to `\r\n`.
554        tty.c_oflag &= !libc::OPOST;
555        pty_child.set_terminal_params(&tty).unwrap();
556
557        pty_child.set_window_size(40, 80).unwrap();
558
559        // stty is in coreutils and should be available on most systems.
560        let mut child = Command::new("stty")
561            .arg("size")
562            .pty(pty_child)
563            .spawn()
564            .unwrap();
565
566        // NOTE: read_to_end returns an EIO error once the child has exited.
567        let mut buf = Vec::new();
568        assert!(pty.read_to_end(&mut buf).await.is_err());
569
570        assert_eq!(from_utf8(&buf).unwrap(), "40 80\n");
571
572        assert_eq!(child.wait().await.unwrap(), ExitStatus::SUCCESS);
573    }
574
575    #[tokio::test]
576    async fn mount_devpts_basic() {
577        if crate::test_runs_in_own_process() {
578            return;
579        }
580        let output = Command::new("ls")
581            .arg("/dev/pts")
582            .map_root()
583            .mount(Mount::devpts("/dev/pts"))
584            .output()
585            .await
586            .unwrap();
587
588        assert_eq!(output.status, ExitStatus::Exited(0));
589
590        // Should be totally empty except for `/dev/pts/ptmx` since we mounted a
591        // new devpts.
592        assert_eq!(output.stderr, b"");
593        assert_eq!(output.stdout, b"ptmx\n");
594    }
595
596    #[tokio::test]
597    async fn mount_devpts_isolated() {
598        if crate::test_runs_in_own_process() {
599            return;
600        }
601        let output = Command::new("ls")
602            .arg("/dev/pts")
603            .map_root()
604            .mount(Mount::devpts("/dev/pts").data("newinstance,ptmxmode=0666"))
605            .mount(Mount::bind("/dev/pts/ptmx", "/dev/ptmx"))
606            .output()
607            .await
608            .unwrap();
609
610        assert_eq!(output.status, ExitStatus::Exited(0));
611
612        // Should be totally empty except for `/dev/pts/ptmx` since we mounted a
613        // new devpts.
614        assert_eq!(output.stderr, b"");
615        assert_eq!(output.stdout, b"ptmx\n");
616    }
617
618    #[tokio::test]
619    async fn mount_tmpfs() {
620        if crate::test_runs_in_own_process() {
621            return;
622        }
623        let mount = "type=tmpfs,target=/tmp"
624            .parse::<Mount>()
625            .expect("tmpfs mount syntax should parse");
626        let output = Command::new("ls")
627            .arg("/tmp")
628            .map_root()
629            .mount(mount)
630            .output()
631            .await
632            .unwrap();
633
634        assert_eq!(output.status, ExitStatus::Exited(0));
635
636        // Should be totally empty since we mounted a new tmpfs.
637        assert_eq!(output.stderr, b"");
638        assert_eq!(output.stdout, b"");
639    }
640
641    #[tokio::test]
642    async fn mount_and_move_tmpfs() {
643        if crate::test_runs_in_own_process() {
644            return;
645        }
646        let tmpfs = tempfile::tempdir().unwrap();
647
648        // Create a temporary directory that will be the only thing to remain in
649        // the `/tmp` mount.
650        let persistent = tempfile::tempdir().unwrap();
651        fs::write(persistent.path().join("foobar"), b"").unwrap();
652
653        let output = Command::new("ls")
654            .arg("/tmp")
655            .map_root()
656            .mount(Mount::tmpfs(tmpfs.path()))
657            // Bind-mount a directory from our upper /tmp to our new /tmp.
658            .mount(Mount::bind(persistent.path(), tmpfs.path().join("my-dir")).touch_target())
659            // Move our newly-created tmpfs to hide the upper /tmp folder.
660            .mount(Mount::rename(tmpfs.path(), Path::new("/tmp")))
661            .output()
662            .await
663            .unwrap();
664
665        assert_eq!(output.status, ExitStatus::Exited(0));
666
667        // The only thing there should be our bind-mounted directory.
668        assert_eq!(output.stderr, b"");
669        assert_eq!(output.stdout, b"my-dir\n");
670    }
671
672    #[tokio::test]
673    async fn mount_bind() {
674        if crate::test_runs_in_own_process() {
675            return;
676        }
677        let temp = tempfile::tempdir().unwrap();
678        let a = temp.path().join("a");
679        let b = temp.path().join("b");
680
681        fs::create_dir(&a).unwrap();
682        fs::create_dir(&b).unwrap();
683
684        fs::write(a.join("foobar"), "im a test").unwrap();
685
686        let output = Command::new("ls")
687            .arg(&b)
688            .map_root()
689            .mount(Mount::bind(&a, &b))
690            .output()
691            .await
692            .unwrap();
693
694        assert_eq!(output.status, ExitStatus::Exited(0));
695        assert_eq!(output.stdout, b"foobar\n");
696        assert_eq!(output.stderr, b"");
697    }
698
699    #[tokio::test]
700    async fn mount_bind_readonly_rejects_writes() {
701        if crate::test_runs_in_own_process() {
702            return;
703        }
704        let temp = tempfile::tempdir().unwrap();
705        let source = temp.path().join("source");
706        let target = temp.path().join("target");
707        fs::create_dir(&source).unwrap();
708        fs::create_dir(&target).unwrap();
709        fs::write(source.join("data"), "original").unwrap();
710
711        let output = Command::new("sh")
712            .args(["-c", "printf changed > \"$TARGET\""])
713            .env("TARGET", target.join("data"))
714            .map_root()
715            .mount(Mount::bind(&source, &target).readonly())
716            .output()
717            .await
718            .unwrap();
719
720        assert_ne!(output.status, ExitStatus::Exited(0));
721        assert_eq!(fs::read(source.join("data")).unwrap(), b"original");
722    }
723
724    #[tokio::test]
725    async fn local_networking_ping() {
726        if crate::test_runs_in_own_process() {
727            return;
728        }
729        const CHILD_ENV: &str = "REVERIE_PROCESS_LOOPBACK_TEST_CHILD";
730
731        if std::env::var_os(CHILD_ENV).is_some() {
732            let socket = std::net::UdpSocket::bind("[::1]:0").unwrap();
733            let address = socket.local_addr().unwrap();
734            assert_eq!(socket.send_to(b"ping", address).unwrap(), 4);
735
736            let mut buffer = [0; 4];
737            let (length, source) = socket.recv_from(&mut buffer).unwrap();
738            assert_eq!(source, address);
739            assert_eq!(&buffer[..length], b"ping");
740            return;
741        }
742
743        let output = Command::new(std::env::current_exe().unwrap())
744            .arg("--exact")
745            .arg("tests::local_networking_ping")
746            .env(CHILD_ENV, "1")
747            .map_root()
748            .local_networking_only()
749            .output()
750            .await
751            .unwrap();
752
753        assert_eq!(output.status, ExitStatus::Exited(0), "{:?}", output);
754    }
755
756    #[tokio::test]
757    async fn local_networking_loopback_flags() {
758        if crate::test_runs_in_own_process() {
759            return;
760        }
761        let output = Command::new("cat")
762            .arg("/sys/class/net/lo/flags")
763            .map_root()
764            .local_networking_only()
765            .output()
766            .await
767            .unwrap();
768
769        assert_eq!(output.status, ExitStatus::Exited(0), "{:?}", output);
770        assert_eq!(output.stdout, b"0x9\n", "{:?}", output);
771    }
772
773    /// Show that processes in two separate network namespaces can bind to the
774    /// same port.
775    #[tokio::test]
776    async fn port_isolation() {
777        if crate::test_runs_in_own_process() {
778            return;
779        }
780        use std::thread::sleep;
781        use std::time::Duration;
782
783        let mut command = Command::new("nc");
784        command
785            .arg("-l")
786            .arg("127.0.0.1")
787            // Can bind to a low port without real root inside the namespace.
788            .arg("80")
789            .stdin(Stdio::null())
790            .stdout(Stdio::piped())
791            .stderr(Stdio::piped())
792            .map_root()
793            .local_networking_only();
794
795        let server1 = match command.spawn() {
796            // If netcat is not installed just exit successfully.
797            Err(error) if error.errno() == Errno::ENOENT => return,
798            other => other,
799        }
800        .unwrap();
801
802        let server2 = command.spawn().unwrap();
803
804        // Give them both time to start up.
805        sleep(Duration::from_millis(100));
806
807        // Stop them with a signal that cannot be ignored. A test binary launched
808        // as a background shell job inherits SIGINT as ignored, and that
809        // disposition survives exec into nc.
810        server1.signal(Signal::SIGKILL).unwrap();
811        server2.signal(Signal::SIGKILL).unwrap();
812
813        let (output1, output2) = tokio::join!(
814            tokio::time::timeout(Duration::from_secs(1), server1.wait_with_output()),
815            tokio::time::timeout(Duration::from_secs(1), server2.wait_with_output()),
816        );
817        let output1 = output1
818            .expect("port_isolation: server 1 did not exit within 1 second after SIGKILL")
819            .unwrap();
820        let output2 = output2
821            .expect("port_isolation: server 2 did not exit within 1 second after SIGKILL")
822            .unwrap();
823
824        // Without network isolation, one of the servers would exit with an
825        // "Address already in use" (exit status 2) error.
826        assert_eq!(
827            output1.status,
828            ExitStatus::Signaled(Signal::SIGKILL, false),
829            "{:?}",
830            output1
831        );
832        assert_eq!(
833            output2.status,
834            ExitStatus::Signaled(Signal::SIGKILL, false),
835            "{:?}",
836            output2
837        );
838    }
839
840    /// Make sure we can call `.local_networking_only` more than once.
841    #[tokio::test]
842    async fn local_networking_there_can_be_only_one() {
843        if crate::test_runs_in_own_process() {
844            return;
845        }
846        let output = Command::new("true")
847            .map_root()
848            .local_networking_only()
849            // If calling this twice mounted /sys twice, then we'd get a "Device
850            // or resource busy" error.
851            .local_networking_only()
852            .output()
853            .await
854            .unwrap();
855        assert_eq!(output.status, ExitStatus::Exited(0), "{:?}", output);
856        assert_eq!(output.stdout, b"", "{:?}", output);
857        assert_eq!(output.stderr, b"", "{:?}", output);
858    }
859
860    #[test]
861    fn from_std_lossy() {
862        if crate::test_runs_in_own_process() {
863            return;
864        }
865        let mut stdcmd = std::process::Command::new("echo");
866        stdcmd.args(["arg1", "arg2"]);
867        stdcmd.current_dir("/foo/bar");
868        stdcmd.env_clear();
869        stdcmd.env("FOO", "1");
870        stdcmd.env("BAR", "2");
871
872        let cmd = Command::from_std_lossy(&stdcmd);
873
874        assert_eq!(cmd.get_program(), "echo");
875        assert_eq!(cmd.get_arg0(), "echo");
876        assert_eq!(cmd.get_args().collect::<Vec<_>>(), ["arg1", "arg2"]);
877
878        let envs = cmd
879            .get_envs()
880            .filter_map(|(k, v)| Some((k.to_str()?, v.and_then(|v| v.to_str()))))
881            .collect::<Vec<_>>();
882        assert_eq!(envs, [("BAR", Some("2")), ("FOO", Some("1"))]);
883    }
884
885    #[test]
886    fn into_std_lossy_compatibility() {
887        if crate::test_runs_in_own_process() {
888            return;
889        }
890        let mut cmd = Command::new("env");
891        cmd.args(["-0"]);
892        cmd.current_dir("/foo/bar");
893        cmd.env_clear();
894        cmd.env("FOO", "1");
895        cmd.env("BAR", "2");
896
897        let stdcmd = cmd.into_std_lossy();
898
899        assert_eq!(stdcmd.get_program(), "env");
900        assert_eq!(stdcmd.get_args().collect::<Vec<_>>(), ["-0"]);
901
902        let envs = stdcmd
903            .get_envs()
904            .filter_map(|(k, v)| Some((k.to_str()?, v.and_then(|v| v.to_str()))))
905            .collect::<Vec<_>>();
906
907        assert_eq!(envs, [("BAR", Some("2")), ("FOO", Some("1"))]);
908    }
909
910    #[test]
911    fn try_into_std_refuses_container_configuration() {
912        if crate::test_runs_in_own_process() {
913            return;
914        }
915        use syscalls::Sysno;
916
917        use super::seccomp::Action;
918        use super::seccomp::FilterBuilder;
919
920        let filter = FilterBuilder::new()
921            .default_action(Action::Allow)
922            .syscalls([(Sysno::brk, Action::KillProcess)])
923            .build();
924        let mut command = Command::new("true");
925        command.seccomp(filter);
926        let error = command.try_into_std().unwrap_err();
927        assert_eq!(error.kind(), io::ErrorKind::InvalidInput);
928        assert_eq!(
929            error.to_string(),
930            "cannot convert to std::process::Command without losing: seccomp filter"
931        );
932
933        let filter = FilterBuilder::new()
934            .default_action(Action::Allow)
935            .syscalls([(Sysno::brk, Action::KillProcess)])
936            .build();
937        let mut command = Command::new("true");
938        command.seccomp(filter);
939        let panic =
940            std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| command.into_std_lossy()))
941                .unwrap_err();
942        let message = panic
943            .downcast_ref::<String>()
944            .map(String::as_str)
945            .or_else(|| panic.downcast_ref::<&str>().copied())
946            .expect("legacy conversion panic must carry a string diagnostic");
947        assert_eq!(
948            message,
949            "Command::into_std_lossy refused unsupported configuration: cannot convert to std::process::Command without losing: seccomp filter"
950        );
951
952        let mut command = Command::new("true");
953        command.unshare(Namespace::MOUNT);
954        let error = command.try_into_std().unwrap_err();
955        assert_eq!(error.kind(), io::ErrorKind::InvalidInput);
956        assert_eq!(
957            error.to_string(),
958            "cannot convert to std::process::Command without losing: Linux namespaces"
959        );
960
961        let mut command = Command::new("true");
962        command.container.affinity(0);
963        let error = command.try_into_std().unwrap_err();
964        assert_eq!(error.kind(), io::ErrorKind::InvalidInput);
965        assert_eq!(
966            error.to_string(),
967            "cannot convert to std::process::Command without losing: CPU affinity"
968        );
969    }
970
971    #[tokio::test]
972    async fn seccomp() {
973        if crate::test_runs_in_own_process() {
974            return;
975        }
976        use syscalls::Sysno;
977
978        use super::seccomp::*;
979
980        let filter = FilterBuilder::new()
981            .default_action(Action::Allow)
982            .syscalls([(Sysno::brk, Action::KillProcess)])
983            .build();
984
985        let output = Command::new("cat")
986            .arg("/proc/self/status")
987            .seccomp(filter)
988            .output()
989            .await
990            .unwrap();
991        assert!(
992            matches!(output.status, ExitStatus::Signaled(Signal::SIGSYS, _)),
993            "Expected Signaled(SIGSYS, _), got {:?}",
994            output.status
995        );
996    }
997
998    /// The launcher's `execve` must reach the kernel with the argument
999    /// registers that `execve` ignores (arg3..arg5: r10, r8 and r9) set to
1000    /// zero. A ptrace tracer records all six argument registers at the seccomp
1001    /// stop, so whatever the launcher's own code last left in them would
1002    /// otherwise enter the recorded launch as host-dependent state.
1003    ///
1004    /// A SIGSYS handler observes the registers at the `execve` instruction:
1005    /// the filter traps `execve`, the handler reports the three registers
1006    /// through a pipe, and the child exits before any exec happens. The
1007    /// `pre_exec` callback leaves a recognizable value in those registers, the
1008    /// way arbitrary launcher code can.
1009    #[cfg(target_arch = "x86_64")]
1010    #[tokio::test]
1011    async fn exec_zeroes_unused_argument_registers() {
1012        if crate::test_runs_in_own_process() {
1013            return;
1014        }
1015        use std::sync::atomic::AtomicI32;
1016        use std::sync::atomic::Ordering;
1017
1018        use syscalls::Sysno;
1019
1020        use super::seccomp::*;
1021
1022        const POISON: u64 = 0x5a5a_0000_0000_0030;
1023        static REPORT_FD: AtomicI32 = AtomicI32::new(-1);
1024
1025        extern "C" fn on_sigsys(
1026            _sig: libc::c_int,
1027            _info: *mut libc::siginfo_t,
1028            ctx: *mut libc::c_void,
1029        ) {
1030            // SAFETY: SA_SIGINFO handlers receive a valid ucontext_t.
1031            let gregs = unsafe { &(*(ctx as *const libc::ucontext_t)).uc_mcontext.gregs };
1032            let words = [
1033                gregs[libc::REG_R10 as usize] as u64,
1034                gregs[libc::REG_R8 as usize] as u64,
1035                gregs[libc::REG_R9 as usize] as u64,
1036            ];
1037            let fd = REPORT_FD.load(Ordering::Relaxed);
1038            // SAFETY: write and _exit are async-signal-safe.
1039            unsafe {
1040                libc::write(fd, words.as_ptr() as *const libc::c_void, 24);
1041                libc::_exit(0);
1042            }
1043        }
1044
1045        let mut fds = [0; 2];
1046        assert_eq!(unsafe { libc::pipe(fds.as_mut_ptr()) }, 0);
1047        let (reader, writer) = (fds[0], fds[1]);
1048        REPORT_FD.store(writer, Ordering::Relaxed);
1049
1050        let filter = FilterBuilder::new()
1051            .default_action(Action::Allow)
1052            .syscalls([(Sysno::execve, Action::Trap)])
1053            .build();
1054
1055        let mut command = Command::new("/bin/true");
1056        command.seccomp(filter);
1057        // SAFETY: the callback only calls async-signal-safe sigaction and
1058        // writes registers; it does not allocate.
1059        unsafe {
1060            command.pre_exec(|| {
1061                let mut action: libc::sigaction = std::mem::zeroed();
1062                action.sa_sigaction = on_sigsys as *const () as usize;
1063                action.sa_flags = libc::SA_SIGINFO;
1064                if libc::sigaction(libc::SIGSYS, &action, std::ptr::null_mut()) != 0 {
1065                    return Err(Errno::last());
1066                }
1067                std::arch::asm!(
1068                    "mov r8, {poison}",
1069                    "mov r9, {poison}",
1070                    "mov r10, {poison}",
1071                    poison = in(reg) POISON,
1072                    out("r8") _,
1073                    out("r9") _,
1074                    out("r10") _,
1075                );
1076                Ok(())
1077            });
1078        }
1079
1080        let status = command.spawn().unwrap().wait().await.unwrap();
1081        unsafe { libc::close(writer) };
1082        let mut words = [0u64; 3];
1083        let n = unsafe { libc::read(reader, words.as_mut_ptr() as *mut libc::c_void, 24) };
1084        unsafe { libc::close(reader) };
1085
1086        assert_eq!(status, ExitStatus::SUCCESS);
1087        assert_eq!(
1088            n, 24,
1089            "the SIGSYS handler did not report the execve registers"
1090        );
1091        assert_eq!(
1092            words,
1093            [0, 0, 0],
1094            "execve reached the kernel with leftover launcher registers \
1095             [r10, r8, r9] = [{:#x}, {:#x}, {:#x}]",
1096            words[0],
1097            words[1],
1098            words[2],
1099        );
1100    }
1101
1102    /// `Command` resolves its program the way glibc's `execvpe(3)` does: the
1103    /// `PATH` of the spawning process (default `/bin:/usr/bin`), an empty entry
1104    /// meaning the current directory, `EACCES` remembered while the search
1105    /// continues past `ENOENT` and `ENOTDIR`, any other error ending the
1106    /// search, and a script without a shebang (`ENOEXEC`) run through
1107    /// `/bin/sh`. The expected values are glibc's results, so this passes
1108    /// whether the launcher calls glibc's `execvpe` or reimplements its search.
1109    #[tokio::test]
1110    async fn exec_path_search_matches_glibc_execvpe() {
1111        if crate::test_runs_in_own_process() {
1112            return;
1113        }
1114        use std::os::unix::fs::PermissionsExt;
1115
1116        const NAME: &str = "reverie-exec-probe";
1117
1118        // The directory holding this test binary is on a mount that allows
1119        // exec, which the script cases need; a temporary directory may not be.
1120        let exe = std::env::current_exe().unwrap();
1121        let scratch = tempfile::tempdir_in(exe.parent().unwrap()).unwrap();
1122        let make_dir = |name: &str| {
1123            let dir = scratch.path().join(name);
1124            fs::create_dir(&dir).unwrap();
1125            dir
1126        };
1127        let empty = make_dir("empty");
1128        let not_executable = make_dir("not-executable");
1129        let script = make_dir("script");
1130        let symlink_loop = make_dir("symlink-loop");
1131        let regular_file = scratch.path().join("regular-file");
1132        fs::write(&regular_file, "").unwrap();
1133
1134        let write_probe = |dir: &Path, mode: u32| {
1135            let path = dir.join(NAME);
1136            fs::write(&path, "exit 7\n").unwrap();
1137            fs::set_permissions(&path, fs::Permissions::from_mode(mode)).unwrap();
1138        };
1139        write_probe(&not_executable, 0o644);
1140        // No shebang: `execve` fails with ENOEXEC and the shell runs it.
1141        write_probe(&script, 0o755);
1142        std::os::unix::fs::symlink(NAME, symlink_loop.join(NAME)).unwrap();
1143
1144        // The search reads PATH from this process, not from the child's
1145        // environment. This test runs alone in its own process.
1146        let set_path = |entries: &[&Path]| {
1147            let value = entries
1148                .iter()
1149                .map(|entry| entry.to_str().unwrap())
1150                .collect::<Vec<_>>()
1151                .join(":");
1152            unsafe { std::env::set_var("PATH", value) };
1153        };
1154        let exec_error = |errno| Error::new(errno, Context::Exec);
1155
1156        assert_eq!(
1157            Command::new("").spawn().unwrap_err(),
1158            exec_error(Errno::ENOENT),
1159            "an empty program name"
1160        );
1161        assert_eq!(
1162            Command::new("x".repeat(libc::NAME_MAX as usize + 1))
1163                .spawn()
1164                .unwrap_err(),
1165            exec_error(Errno::ENAMETOOLONG),
1166            "a program name longer than NAME_MAX"
1167        );
1168
1169        set_path(&[&empty]);
1170        assert_eq!(
1171            Command::new(NAME).spawn().unwrap_err(),
1172            exec_error(Errno::ENOENT),
1173            "a program that is on no PATH entry"
1174        );
1175
1176        set_path(&[&not_executable, &empty]);
1177        assert_eq!(
1178            Command::new(NAME).spawn().unwrap_err(),
1179            exec_error(Errno::EACCES),
1180            "EACCES from an earlier entry outranks a later ENOENT"
1181        );
1182
1183        set_path(&[&symlink_loop, &script]);
1184        assert_eq!(
1185            Command::new(NAME).spawn().unwrap_err(),
1186            exec_error(Errno::ELOOP),
1187            "an error other than EACCES, ENOENT or ENOTDIR ends the search"
1188        );
1189
1190        set_path(&[&not_executable, &regular_file, &script]);
1191        assert_eq!(
1192            Command::new(NAME).spawn().unwrap().wait().await.unwrap(),
1193            ExitStatus::Exited(7),
1194            "the search continues past EACCES and ENOTDIR to a script without a shebang"
1195        );
1196
1197        set_path(&[&empty, Path::new("")]);
1198        assert_eq!(
1199            Command::new(NAME)
1200                .current_dir(&script)
1201                .spawn()
1202                .unwrap()
1203                .wait()
1204                .await
1205                .unwrap(),
1206            ExitStatus::Exited(7),
1207            "an empty PATH entry means the current directory"
1208        );
1209
1210        assert_eq!(
1211            Command::new(script.join(NAME))
1212                .spawn()
1213                .unwrap()
1214                .wait()
1215                .await
1216                .unwrap(),
1217            ExitStatus::Exited(7),
1218            "a name containing a slash is executed without a search"
1219        );
1220
1221        unsafe { std::env::remove_var("PATH") };
1222        assert_eq!(
1223            Command::new("true").spawn().unwrap().wait().await.unwrap(),
1224            ExitStatus::Exited(0),
1225            "an unset PATH defaults to /bin:/usr/bin"
1226        );
1227    }
1228
1229    #[tokio::test]
1230    async fn seccomp_notify() {
1231        if crate::test_runs_in_own_process() {
1232            return;
1233        }
1234        use std::collections::HashMap;
1235
1236        use futures::future::Either;
1237        use futures::future::select;
1238        use futures::stream::TryStreamExt;
1239        use syscalls::Sysno;
1240
1241        use super::seccomp::*;
1242
1243        let filter = FilterBuilder::new()
1244            .default_action(Action::Notify)
1245            .syscalls([
1246                // FIXME: Because the first execve happens when the child is
1247                // spawned, we must allow this through. Otherwise, the
1248                // `.spawn()` below will deadlock because we can't process
1249                // seccomp notifications until after it returns.
1250                (Sysno::execve, Action::Allow),
1251            ])
1252            .build();
1253
1254        let mut child = Command::new("cat")
1255            .arg("/proc/self/status")
1256            .seccomp(filter)
1257            .seccomp_notify()
1258            .stdout(Stdio::null())
1259            .spawn()
1260            .unwrap();
1261
1262        let mut summary = HashMap::new();
1263
1264        let exit_status = {
1265            let seccomp_notif = child.seccomp_notif.take();
1266
1267            let notifier = async {
1268                if let Some(mut notifier) = seccomp_notif {
1269                    while let Some(notif) = notifier.try_next().await.unwrap() {
1270                        *summary.entry(Sysno::from(notif.data.nr)).or_insert(0u64) += 1;
1271
1272                        // Simply let the syscall through.
1273                        let resp = seccomp_notif_resp {
1274                            id: notif.id,
1275                            val: 0,
1276                            error: 0,
1277                            flags: SECCOMP_USER_NOTIF_FLAG_CONTINUE,
1278                        };
1279                        notifier.send(&resp).unwrap();
1280                    }
1281                }
1282            };
1283
1284            let exit_status = child.wait();
1285
1286            futures::pin_mut!(notifier);
1287            futures::pin_mut!(exit_status);
1288
1289            match select(notifier, exit_status).await {
1290                Either::Left((_, _)) => unreachable!(),
1291                Either::Right((exit_status, _)) => exit_status.unwrap(),
1292            }
1293        };
1294
1295        assert_eq!(exit_status, ExitStatus::SUCCESS);
1296
1297        assert!(summary[&Sysno::read] > 0);
1298        assert!(summary[&Sysno::write] > 0);
1299        assert!(summary[&Sysno::close] > 0);
1300    }
1301}