Skip to main content

reverie_process/
spawn.rs

1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 * All rights reserved.
4 *
5 * This source code is licensed under the BSD-style license found in the
6 * LICENSE file in the root directory of this source tree.
7 */
8
9use std::io;
10use std::io::Write;
11
12use super::Child;
13use super::Command;
14use super::clone::clone;
15use super::container::ChildContext;
16use super::error::Context;
17use super::error::Error;
18use super::fd::Fd;
19use super::fd::pipe;
20use super::id_map::make_id_map;
21use super::launch_window::Launch;
22use super::launch_window::TransientOpen;
23use super::seccomp::SeccompNotif;
24use super::stdio::ChildStderr;
25use super::stdio::ChildStdin;
26use super::stdio::ChildStdout;
27use super::util::CStringArray;
28use super::util::SharedValue;
29
30/// The allocated parts of a launch, made before its lock (`Command::prepare`).
31struct Prepared {
32    env: CStringArray,
33    uid_map: Vec<u8>,
34    gid_map: Vec<u8>,
35}
36
37impl Command {
38    /// Executes the command as a child process, returning a handle to it.
39    ///
40    /// By default, stdin, stdout and stderr are inherited from the parent.
41    pub fn spawn(&mut self) -> Result<Child, Error> {
42        let prepared = self.prepare();
43
44        // Every descriptor the child may inherit is made under the launch
45        // lock, which ends when clone returns (see `launch_window`). On an
46        // early return the lock, declared later, ends before `prepared` is
47        // freed.
48        let launch = Launch::begin();
49
50        // Create a pipe to send back errors to the parent process if `execve`
51        // fails.
52        let (reader, mut writer) = pipe()?;
53
54        let child = self.spawn_launched(prepared, launch, |err| {
55            send_error(&mut writer, err);
56            1
57        })?;
58
59        // Close the writer end. Otherwise, the following read will hang
60        // forever.
61        drop(writer);
62
63        recv_error(reader)?;
64
65        Ok(child)
66    }
67
68    /// Spawn the child with helper functions. The `onfail` callback runs in the
69    /// child process if an error occurs during execution of the process. The
70    /// `wait` function can be used to wait for the child to fully start up and
71    /// to transform it into another type.
72    pub fn spawn_with<F>(&mut self, onfail: F) -> Result<Child, Error>
73    where
74        F: FnMut(Error) -> i32,
75    {
76        let prepared = self.prepare();
77        self.spawn_launched(prepared, Launch::begin(), onfail)
78    }
79
80    /// Builds what a launch needs that allocates, before the launch lock is
81    /// taken. Nothing may allocate or free under that lock: an allocator
82    /// can hold its own lock, or call code that needs a transient open, while
83    /// it waits for the launch to end.
84    fn prepare(&self) -> Prepared {
85        Prepared {
86            env: self.container.env.array(),
87            uid_map: make_id_map(&self.container.uid_map),
88            gid_map: make_id_map(&self.container.gid_map),
89        }
90    }
91
92    fn spawn_launched<F>(
93        &mut self,
94        prepared: Prepared,
95        launch: Launch,
96        mut onfail: F,
97    ) -> Result<Child, Error>
98    where
99        F: FnMut(Error) -> i32,
100    {
101        let Prepared {
102            env,
103            uid_map,
104            gid_map,
105        } = prepared;
106
107        let clone_flags = self.container.namespace.bits() | libc::SIGCHLD;
108
109        // Under the lock: the child's descriptors, its shared page and the
110        // clone, all system calls. On failure only descriptors and that page
111        // are released, which frees no memory, and the error is plain data.
112        let launched = (|| {
113            // Set up IO pipes
114            let (stdin, child_stdin) = self.container.stdin.pipes(true)?;
115            let (stdout, child_stdout) = self.container.stdout.pipes(false)?;
116            let (stderr, child_stderr) = self.container.stdout.pipes(false)?;
117
118            let seccomp_fd = if self.container.seccomp_notify {
119                Some(SharedValue::new(core::sync::atomic::AtomicI32::new(0))?)
120            } else {
121                None
122            };
123
124            let context = ChildContext {
125                stdin: child_stdin.as_ref(),
126                stdout: child_stdout.as_ref(),
127                stderr: child_stderr.as_ref(),
128                uid_map: &uid_map,
129                gid_map: &gid_map,
130                seccomp_fd: seccomp_fd.as_ref().map(|x| x.as_ref()),
131            };
132
133            let pid = clone(
134                || {
135                    let code = onfail(self.do_exec(&context, &env));
136                    unsafe { libc::_exit(code) }
137                },
138                clone_flags,
139            )?;
140            Ok::<_, Error>((
141                pid,
142                [stdin, stdout, stderr],
143                [child_stdin, child_stdout, child_stderr],
144                seccomp_fd,
145            ))
146        })();
147        // The child exists, so it cannot inherit what the parent opens from
148        // here on. The waits below for the child's startup must not hold the
149        // lock: the child's setup may wait on a thread that needs it.
150        drop(launch);
151        let (pid, [stdin, stdout, stderr], [child_stdin, child_stdout, child_stderr], seccomp_fd) =
152            launched?;
153
154        drop(child_stdin);
155        drop(child_stdout);
156        drop(child_stderr);
157        drop(self.container.pty.take());
158
159        let seccomp_notif = match seccomp_fd {
160            Some(shared_fd) => {
161                use core::sync::atomic::Ordering;
162
163                // Spin until the value changes in the child.
164                let mut targetfd = 0;
165                while targetfd == 0 {
166                    targetfd = shared_fd.as_ref().load(Ordering::Relaxed);
167                    std::thread::yield_now();
168                }
169
170                // Use pidfd_getfd to copy the file descriptor
171                let fd = {
172                    let _open = TransientOpen::begin();
173                    let pidfd = Fd::pidfd_open(pid.into(), 0)?;
174                    pidfd.pidfd_getfd(targetfd, 0)?
175                };
176
177                // We've successfully duplicated the file descriptor. Let the
178                // child continue on to execve.
179                shared_fd.as_ref().store(0, Ordering::Relaxed);
180
181                Some(SeccompNotif::new(fd)?)
182            }
183            None => None,
184        };
185
186        let stdin = stdin.map(ChildStdin::new).transpose()?;
187        let stdout = stdout.map(ChildStdout::new).transpose()?;
188        let stderr = stderr.map(ChildStderr::new).transpose()?;
189
190        Ok(Child {
191            pid,
192            exit_status: None,
193            seccomp_notif,
194            stdin,
195            stdout,
196            stderr,
197        })
198    }
199
200    /// Note: This function MUST NOT allocate or deallocate any memory. Doing so
201    /// can cause deadlocks.
202    ///
203    /// Only returns if an error occurs, thus it is only possible for it to
204    /// return an error.
205    fn do_exec(&mut self, context: &ChildContext, env: &CStringArray) -> Error {
206        if let Err(err) = self.container.setup(context, &mut self.pre_exec) {
207            return err;
208        }
209
210        Error::result(
211            unsafe { execvpe_zeroed_tail(&self.program, self.args.as_ptr(), env.as_ptr()) },
212            Context::Exec,
213        )
214        .unwrap_err()
215    }
216}
217
218/// Issues `execve` with the three argument registers `execve` ignores
219/// (arg3..arg5) set to zero.
220///
221/// `libc::execvpe` reaches the `execve` instruction with whatever the launcher
222/// last left in those registers. The kernel ignores them and clears them in the
223/// new image, but a ptrace tracer records all six argument registers at the
224/// seccomp stop, so leftover launcher state would enter the recorded launch.
225/// glibc's `syscall(3)` moves its fifth, sixth and seventh arguments into r10,
226/// r8 and r9, so passing explicit zeros defines them.
227///
228/// Returns -1 with `errno` set, like `execve(2)`.
229unsafe fn execve_zeroed_tail(
230    path: *const libc::c_char,
231    argv: *const *const libc::c_char,
232    envp: *const *const libc::c_char,
233) -> libc::c_int {
234    let zero: libc::c_long = 0;
235    unsafe { libc::syscall(libc::SYS_execve, path, argv, envp, zero, zero, zero) as libc::c_int }
236}
237
238/// Behaves like glibc `execvpe(3)`, but issues every `execve` through
239/// [`execve_zeroed_tail`]. The `PATH` search and its error precedence follow
240/// glibc: `EACCES` is remembered, and `ENOENT`, `ESTALE`, `ENOTDIR`, `ENODEV`
241/// and `ETIMEDOUT` move on to the next entry. A candidate that fails with
242/// `ENOEXEC` is handed to `libc::execvpe`, which runs it through `/bin/sh`.
243///
244/// MUST NOT allocate: this runs in the child between `clone` and `execve`.
245unsafe fn execvpe_zeroed_tail(
246    program: &std::ffi::CStr,
247    argv: *const *const libc::c_char,
248    envp: *const *const libc::c_char,
249) -> libc::c_int {
250    const NAME_MAX: usize = libc::NAME_MAX as usize;
251    const PATH_MAX: usize = libc::PATH_MAX as usize;
252    let errno = || io::Error::last_os_error().raw_os_error().unwrap_or(0);
253    let set_errno = |value| unsafe { *libc::__errno_location() = value };
254
255    let file = program.to_bytes();
256    if file.is_empty() {
257        set_errno(libc::ENOENT);
258        return -1;
259    }
260    if file.contains(&b'/') {
261        unsafe { execve_zeroed_tail(program.as_ptr(), argv, envp) };
262        if errno() == libc::ENOEXEC {
263            return unsafe { libc::execvpe(program.as_ptr(), argv, envp) };
264        }
265        return -1;
266    }
267    if file.len() > NAME_MAX {
268        set_errno(libc::ENAMETOOLONG);
269        return -1;
270    }
271
272    let path = unsafe { libc::getenv(c"PATH".as_ptr()) };
273    let path = if path.is_null() {
274        &b"/bin:/usr/bin"[..]
275    } else {
276        unsafe { std::ffi::CStr::from_ptr(path) }.to_bytes()
277    };
278
279    let mut buffer = [0u8; PATH_MAX + NAME_MAX + 2];
280    let mut got_eacces = false;
281    for dir in path.split(|byte| *byte == b':') {
282        if dir.len() >= PATH_MAX {
283            continue;
284        }
285        // An empty entry means the current directory, as in glibc.
286        let mut len = dir.len();
287        buffer[..len].copy_from_slice(dir);
288        if !dir.is_empty() {
289            buffer[len] = b'/';
290            len += 1;
291        }
292        buffer[len..len + file.len()].copy_from_slice(file);
293        buffer[len + file.len()] = 0;
294        let candidate = buffer.as_ptr() as *const libc::c_char;
295
296        unsafe { execve_zeroed_tail(candidate, argv, envp) };
297        match errno() {
298            libc::ENOEXEC => return unsafe { libc::execvpe(candidate, argv, envp) },
299            libc::EACCES => got_eacces = true,
300            libc::ENOENT | libc::ESTALE | libc::ENOTDIR | libc::ENODEV | libc::ETIMEDOUT => {}
301            _ => return -1,
302        }
303    }
304    if got_eacces {
305        set_errno(libc::EACCES);
306    }
307    -1
308}
309
310/// Sends an error and closes the pipe. Ignore any errors if this fails.
311pub fn send_error(fd: &mut Fd, err: Error) {
312    // Writes up to PIPE_BUF (4096) should be atomic. There's also nothing we
313    // can do with an error if this fails.
314    let bytes: [u8; 8] = err.into();
315    let _ = fd.write(&bytes);
316}
317
318/// Tries to receive an error code from the pipe. If the other end of the
319/// pipe is closed before sending an error, then `Ok(())` is returned.
320pub fn recv_error(mut fd: Fd) -> Result<(), Error> {
321    use std::io::Read;
322    let mut err = [0u8; 8];
323    loop {
324        match fd.read(&mut err) {
325            Ok(0) => return Ok(()),
326            Ok(8) => return Err(Error::from(err)),
327            Ok(n) => {
328                // Sends up to PIPE_BUF (4096) should be atomic.
329                panic!("execve pipe: got unexpected number of bytes {}", n);
330            }
331            Err(err) if err.kind() == io::ErrorKind::Interrupted => {}
332            Err(err) => {
333                panic!("execve pipe: read returned unexpected error {}", err);
334            }
335        }
336    }
337}