Skip to main content

subc_os/
lib.rs

1//! Operating-system primitives the subc daemon needs and cannot reach without
2//! unsafe code, each behind a small safe API.
3//!
4//! The daemon crates forbid unsafe code. This crate is the one deliberate
5//! exception (like `subc-uptime` and `subc-cgroup`): every `unsafe` block here
6//! is a single foreign call with its preconditions stated beside it, and nothing
7//! unsafe is exported.
8//!
9//! Today it answers one question: is the process now holding pid N the same
10//! process the daemon spawned earlier? A pid alone cannot say, because the
11//! kernel reuses pids once a process has been reaped. [`Process`] reads the two
12//! facts that tell processes apart, the kernel's start time for the pid and the
13//! file identity (device and inode) of the executable image it runs, and sends
14//! signals to it.
15//!
16//! Sources, per platform:
17//!
18//! - Linux: the start time is field 22 of `/proc/<pid>/stat` (clock ticks since
19//!   boot), and the executable is `stat` through `/proc/<pid>/exe`, which
20//!   resolves to the running image even if its file has since been replaced or
21//!   deleted. A pidfd is opened before either is read and signals go through it
22//!   (`pidfd_send_signal`), so the process that was checked is the process that
23//!   is signalled. No unsafe code is needed: rustix wraps both calls.
24//! - macOS: the start time is `kp_proc.p_starttime` from `sysctl`
25//!   `KERN_PROC_PID` (microseconds since the epoch), and the executable is the
26//!   path `proc_pidpath` reports, then `stat` on that path. These two calls are
27//!   unsafe. macOS has no pidfd, so a signal is a plain
28//!   `kill` sent right after the checks; see [`Process::signal`].
29//! - Anywhere else: [`Process::open`] reports [`std::io::ErrorKind::Unsupported`].
30//!
31//! For persisted PID owners, [`process_identity`] reads versioned kernel start
32//! identities and distinguishes alive, dead and unknown without spawning a
33//! process. Its foreign calls are signal-zero `kill` on Unix and `proc_pidinfo`
34//! on macOS. Only dead owners may be reclaimed; unknown owners stay protected.
35//!
36//! It also reads how much memory and CPU time one process is using, for
37//! reporting only; see [`resource_usage`]. On Linux that is procfs again; on
38//! macOS it is `proc_pid_rusage`, plus `mach_timebase_info` to convert its CPU
39//! times to nanoseconds, the other two unsafe calls in the crate.
40//!
41//! And it carries the launch nonce from the daemon to each module it spawns
42//! over an inherited pipe instead of the environment: [`launch_nonce`] is the
43//! one reader every module uses, and [`LaunchNonceHandoff`] the daemon's half.
44//! That module's unsafe code is `dup2`, `fcntl`, `fstat` and `ioctl` on
45//! descriptors, each with its preconditions stated beside it.
46
47#![deny(unsafe_code)]
48
49#[cfg(all(unix, feature = "test-support"))]
50pub mod fork_exec_test;
51pub mod launch_nonce;
52pub mod privacy_identity;
53pub mod process_identity;
54#[cfg(unix)]
55pub use launch_nonce::LaunchNonceHandoff;
56pub use launch_nonce::{
57    launch_nonce, LaunchNonce, LaunchNonceError, LaunchNonceSource, LAUNCH_NONCE_ENV,
58    LAUNCH_NONCE_FD, LAUNCH_NONCE_FD_ENV,
59};
60
61#[cfg(target_os = "linux")]
62mod linux;
63#[cfg(target_os = "macos")]
64mod macos;
65
66#[cfg(target_os = "linux")]
67use linux as platform;
68#[cfg(target_os = "macos")]
69use macos as platform;
70
71use std::{io, path::Path};
72
73/// True where [`Process`] can identify and signal a process by pid.
74pub const PROCESS_IDENTITY_SUPPORTED: bool = cfg!(any(target_os = "linux", target_os = "macos"));
75
76/// Device and inode of a file: which file, independent of the name used to
77/// reach it.
78#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
79pub struct FileIdentity {
80    pub device: u64,
81    pub inode: u64,
82}
83
84/// The device and inode of the file at `path`, following symlinks. `None` if it
85/// cannot be read or the platform has no inode numbers.
86pub fn file_identity(path: &Path) -> Option<FileIdentity> {
87    #[cfg(unix)]
88    {
89        use std::os::unix::fs::MetadataExt;
90
91        std::fs::metadata(path).ok().map(|metadata| FileIdentity {
92            device: metadata.dev(),
93            inode: metadata.ino(),
94        })
95    }
96    #[cfg(not(unix))]
97    {
98        let _ = path;
99        None
100    }
101}
102
103/// What a live process looks like right now.
104#[derive(Debug, Clone, Copy, PartialEq, Eq)]
105pub struct Observation {
106    /// The kernel's start time for the process. Opaque: compare it only with a
107    /// value read on the same host by this crate. Linux counts clock ticks since
108    /// boot; macOS counts microseconds since the epoch.
109    pub start_time: u64,
110    /// The file the process is executing, or `None` if it could not be read
111    /// (for example, a process owned by another user).
112    pub executable: Option<FileIdentity>,
113}
114
115/// A signal [`Process::signal`] can send.
116#[derive(Debug, Clone, Copy, PartialEq, Eq)]
117pub enum Signal {
118    /// SIGTERM: a request to exit, which the process may handle or ignore.
119    Terminate,
120    /// SIGKILL: ends the process; it cannot be handled or ignored.
121    Kill,
122}
123
124/// The kernel start time of the process holding `pid`, or `None` if there is
125/// none, it has already exited (a zombie waiting to be reaped counts as exited),
126/// or the platform has no source.
127pub fn start_time(pid: u32) -> Option<u64> {
128    #[cfg(any(target_os = "linux", target_os = "macos"))]
129    {
130        platform::start_time(pid)
131    }
132    #[cfg(not(any(target_os = "linux", target_os = "macos")))]
133    {
134        let _ = pid;
135        None
136    }
137}
138
139/// True where [`resource_usage`] can read a live process. Elsewhere it always
140/// answers `None`, and a caller can use this to say "not supported here"
141/// rather than "could not read".
142pub const RESOURCE_USAGE_SUPPORTED: bool = cfg!(any(target_os = "linux", target_os = "macos"));
143
144/// What [`ResourceUsage::memory_bytes`] measures. The platforms offer
145/// different figures, and they are not interchangeable.
146#[derive(Debug, Clone, Copy, PartialEq, Eq)]
147pub enum MemoryKind {
148    /// macOS `phys_footprint`: the memory the kernel charges to the process
149    /// (dirty and compressed pages, among others), which is also what jetsam
150    /// acts on. Pages an allocator has released with `MADV_FREE` do not count.
151    PhysFootprint,
152    /// Linux `VmRSS`: pages of the process resident in RAM, including shared
153    /// file-backed pages. Swapped-out pages are not included; see
154    /// [`ResourceUsage::swap_bytes`].
155    ResidentSet,
156}
157
158/// One reading of a process's memory and cumulative CPU time.
159///
160/// It covers the process named by the pid alone: its threads are included,
161/// processes it has started are not.
162#[derive(Debug, Clone, Copy, PartialEq, Eq)]
163pub struct ResourceUsage {
164    /// Memory in bytes, measured as [`Self::memory_kind`] says.
165    pub memory_bytes: u64,
166    pub memory_kind: MemoryKind,
167    /// Bytes swapped out (Linux `VmSwap`). `None` where the platform does not
168    /// report it for a single process, which is not the same as zero.
169    pub swap_bytes: Option<u64>,
170    /// CPU time spent in user mode since the process started.
171    pub cpu_user: std::time::Duration,
172    /// CPU time spent in the kernel on the process's behalf since it started.
173    pub cpu_system: std::time::Duration,
174}
175
176/// Memory and cumulative CPU time of the process holding `pid`, read now.
177///
178/// `None` when there is no such process, it has exited (a zombie awaiting its
179/// reap counts as exited), it cannot be read (for example, another user's
180/// process on macOS), or the platform has no source
181/// (see [`RESOURCE_USAGE_SUPPORTED`]). Never a reading of zeros in place of
182/// one of those.
183///
184/// Like any pid-based read, this describes whatever process holds `pid` now;
185/// a caller that needs it to be a particular process should confirm that
186/// process's [`start_time`] around the call.
187pub fn resource_usage(pid: u32) -> Option<ResourceUsage> {
188    #[cfg(any(target_os = "linux", target_os = "macos"))]
189    {
190        platform::resource_usage(pid)
191    }
192    #[cfg(not(any(target_os = "linux", target_os = "macos")))]
193    {
194        let _ = pid;
195        None
196    }
197}
198
199/// A handle on the process holding one pid at the moment it was opened.
200#[derive(Debug)]
201pub struct Process {
202    pid: u32,
203    #[cfg(target_os = "linux")]
204    pidfd: Option<std::os::fd::OwnedFd>,
205}
206
207impl Process {
208    /// Open a handle on the process now holding `pid`.
209    ///
210    /// `Ok(None)` means no process holds that pid (on macOS, also a zombie
211    /// awaiting its reap; on Linux a zombie opens, and [`Self::observe`] then
212    /// reports it as exited). On Linux this opens a pidfd,
213    /// which from then on refers to this exact process even if it exits and the
214    /// pid is reused; when the kernel cannot open one (older than 5.3, or a
215    /// seccomp policy refusing the call) the handle falls back to the pid, as
216    /// on macOS.
217    pub fn open(pid: u32) -> io::Result<Option<Self>> {
218        #[cfg(target_os = "linux")]
219        {
220            linux::open(pid).map(|opened| opened.map(|pidfd| Self { pid, pidfd }))
221        }
222        #[cfg(target_os = "macos")]
223        {
224            Ok(platform::exists(pid).then_some(Self { pid }))
225        }
226        #[cfg(not(any(target_os = "linux", target_os = "macos")))]
227        {
228            let _ = pid;
229            Err(io::Error::new(
230                io::ErrorKind::Unsupported,
231                "process identity is not available on this platform",
232            ))
233        }
234    }
235
236    pub fn pid(&self) -> u32 {
237        self.pid
238    }
239
240    /// True when signals go through a pidfd, so they cannot reach a different
241    /// process that has since reused this pid.
242    pub fn signals_through_pidfd(&self) -> bool {
243        #[cfg(target_os = "linux")]
244        {
245            self.pidfd.is_some()
246        }
247        #[cfg(not(target_os = "linux"))]
248        {
249            false
250        }
251    }
252
253    /// The process's start time and executable, or `None` once it has exited
254    /// (including as a zombie not yet reaped by its parent).
255    pub fn observe(&self) -> Option<Observation> {
256        #[cfg(any(target_os = "linux", target_os = "macos"))]
257        {
258            #[cfg(target_os = "linux")]
259            if !linux::pidfd_alive(self.pidfd.as_ref()) {
260                return None;
261            }
262            let start_time = platform::start_time(self.pid)?;
263            Some(Observation {
264                start_time,
265                executable: platform::executable_identity(self.pid),
266            })
267        }
268        #[cfg(not(any(target_os = "linux", target_os = "macos")))]
269        {
270            None
271        }
272    }
273
274    /// Send `signal` to the process.
275    ///
276    /// With a pidfd the signal can only reach the process this handle was
277    /// opened on: if that process has exited, the call fails with `ESRCH` even
278    /// if the pid has been reused. Without one (macOS, or a Linux kernel with no
279    /// pidfd) the signal goes to whatever holds the pid now, so callers should
280    /// [`Self::observe`] immediately before signalling. What remains is the
281    /// time between that check and this call; for a different process to be
282    /// hit, the checked one must exit, be reaped, and have its pid handed to a
283    /// new process inside that window, and both kernels hand out pids in
284    /// increasing order, so a reuse needs the whole pid space to wrap first.
285    ///
286    /// `Ok(false)` means the process had already exited (`ESRCH`).
287    pub fn signal(&self, signal: Signal) -> io::Result<bool> {
288        #[cfg(any(target_os = "linux", target_os = "macos"))]
289        {
290            #[cfg(target_os = "linux")]
291            let result = linux::signal(self.pid, self.pidfd.as_ref(), signal);
292            #[cfg(target_os = "macos")]
293            let result = macos::signal(self.pid, signal);
294            match result {
295                Ok(()) => Ok(true),
296                Err(rustix::io::Errno::SRCH) => Ok(false),
297                Err(error) => Err(error.into()),
298            }
299        }
300        #[cfg(not(any(target_os = "linux", target_os = "macos")))]
301        {
302            let _ = signal;
303            Err(io::Error::new(
304                io::ErrorKind::Unsupported,
305                "process signalling is not available on this platform",
306            ))
307        }
308    }
309}
310
311#[cfg(all(test, any(target_os = "linux", target_os = "macos")))]
312mod tests {
313    use std::{
314        process::{Child, Command},
315        time::{Duration, Instant},
316    };
317
318    use super::*;
319
320    fn spawn_sleep() -> Child {
321        Command::new("sleep")
322            .arg("60")
323            .spawn()
324            .expect("spawn sleep")
325    }
326
327    /// The executable a spawned `sleep` runs, resolved the way `Command` found it.
328    fn sleep_identity() -> FileIdentity {
329        let path = ["/bin/sleep", "/usr/bin/sleep"]
330            .into_iter()
331            .find(|path| Path::new(path).exists())
332            .expect("sleep is installed");
333        file_identity(Path::new(path)).expect("stat sleep")
334    }
335
336    /// Right after `spawn` returns the child may not have finished exec yet,
337    /// and until then it still runs the test binary's image.
338    fn wait_for_executable(process: &Process, expected: FileIdentity) -> Observation {
339        let deadline = Instant::now() + Duration::from_secs(5);
340        loop {
341            let observation = process.observe().expect("child is alive");
342            if observation.executable == Some(expected) || Instant::now() > deadline {
343                return observation;
344            }
345            std::thread::sleep(Duration::from_millis(10));
346        }
347    }
348
349    #[test]
350    fn own_process_is_observable_with_its_own_image() {
351        let process = Process::open(std::process::id())
352            .expect("open own process")
353            .expect("own process exists");
354        let observation = process.observe().expect("own process is alive");
355        let own_image = file_identity(&std::env::current_exe().unwrap()).unwrap();
356        assert_eq!(observation.executable, Some(own_image));
357        assert_eq!(start_time(std::process::id()), Some(observation.start_time));
358    }
359
360    #[test]
361    fn child_start_time_is_stable_and_differs_from_ours() {
362        let mut child = spawn_sleep();
363        let pid = child.id();
364        let process = Process::open(pid).unwrap().unwrap();
365        let observation = wait_for_executable(&process, sleep_identity());
366        assert_eq!(observation.executable, Some(sleep_identity()));
367        assert_eq!(start_time(pid), Some(observation.start_time));
368        child.kill().unwrap();
369        child.wait().unwrap();
370    }
371
372    #[test]
373    fn a_signalled_and_unreaped_child_reads_as_exited() {
374        let mut child = spawn_sleep();
375        let process = Process::open(child.id()).unwrap().unwrap();
376        assert!(process.signal(Signal::Terminate).unwrap());
377        let deadline = Instant::now() + Duration::from_secs(5);
378        while process.observe().is_some() {
379            assert!(Instant::now() < deadline, "child still observed as alive");
380            std::thread::sleep(Duration::from_millis(10));
381        }
382        // Not yet reaped: the pid is still a zombie here, and still reads as exited.
383        assert_eq!(start_time(child.id()), None);
384        child.wait().unwrap();
385    }
386
387    #[test]
388    fn a_reaped_child_cannot_be_opened_or_observed() {
389        let mut child = spawn_sleep();
390        let pid = child.id();
391        child.kill().unwrap();
392        child.wait().unwrap();
393        // The pid could in principle be reused by now; either way it is not the child.
394        if let Some(process) = Process::open(pid).unwrap() {
395            if let Some(observation) = process.observe() {
396                assert_ne!(observation.executable, Some(sleep_identity()));
397            }
398        }
399    }
400
401    /// The macOS fields are read at fixed offsets, so check the value is a
402    /// plausible start time and not some other field: our own process started
403    /// in the past, and not long ago.
404    #[cfg(target_os = "macos")]
405    #[test]
406    fn macos_start_time_is_microseconds_since_the_epoch() {
407        let now = std::time::SystemTime::now()
408            .duration_since(std::time::UNIX_EPOCH)
409            .unwrap()
410            .as_micros() as u64;
411        let started = start_time(std::process::id()).unwrap();
412        assert!(started <= now, "start time {started} is after now {now}");
413        assert!(
414            now - started < 3_600 * 1_000_000,
415            "start time {started} is more than an hour before now {now}"
416        );
417    }
418
419    /// Keeps one core busy for at least `wall` of wall-clock time.
420    /// This thread's CPU time, from the thread CPU clock rather than the
421    /// process-usage API under test.
422    fn thread_cpu_time() -> Duration {
423        let now = rustix::time::clock_gettime(rustix::time::ClockId::ThreadCPUTime);
424        Duration::new(now.tv_sec as u64, now.tv_nsec as u32)
425    }
426
427    /// Spend `cpu` of this thread's CPU time. Measured on CPU time, not wall
428    /// time: on a loaded machine the thread is descheduled for part of any
429    /// wall interval, so a wall-timed loop can do far less work than its
430    /// duration suggests. A generous wall cap keeps a stalled clock from
431    /// hanging the test.
432    fn burn_cpu(cpu: Duration) {
433        let start = thread_cpu_time();
434        let give_up = Instant::now() + Duration::from_secs(60);
435        let mut value = 0u64;
436        while thread_cpu_time().saturating_sub(start) < cpu {
437            assert!(
438                Instant::now() < give_up,
439                "thread CPU clock stopped advancing"
440            );
441            for step in 0..10_000u64 {
442                value = std::hint::black_box(value.wrapping_mul(31).wrapping_add(step));
443            }
444        }
445        std::hint::black_box(value);
446    }
447
448    #[test]
449    fn own_resource_usage_is_present_and_plausible() {
450        // Clean executable pages need not count toward physical footprint, and
451        // nextest runs this case in a fresh process with little private memory.
452        // Touch and retain private pages so the byte/unit check has a known
453        // lower bound instead of assuming a minimum footprint for the binary.
454        let pages = vec![0xa5u8; 8 * 1024 * 1024];
455        std::hint::black_box(&pages);
456        let usage = resource_usage(std::process::id()).expect("own process is readable");
457        assert!(
458            usage.memory_bytes >= pages.len() as u64,
459            "memory {} bytes cannot account for {} touched private bytes",
460            usage.memory_bytes,
461            pages.len()
462        );
463        std::hint::black_box(&pages);
464        assert!(
465            usage.memory_bytes < 64 * 1024 * 1024 * 1024,
466            "memory {} bytes is implausibly large",
467            usage.memory_bytes
468        );
469        #[cfg(target_os = "macos")]
470        assert_eq!(usage.memory_kind, MemoryKind::PhysFootprint);
471        #[cfg(target_os = "linux")]
472        {
473            assert_eq!(usage.memory_kind, MemoryKind::ResidentSet);
474            assert!(usage.swap_bytes.is_some(), "Linux reports VmSwap");
475        }
476    }
477
478    /// CPU time must grow with busy work, and by roughly the amount of work
479    /// done: a reading in the wrong unit (for example Mach ticks taken as
480    /// nanoseconds on Apple silicon, about 24 times too small) grows too, but
481    /// not by enough.
482    #[test]
483    fn own_cpu_time_grows_by_about_the_busy_work_done() {
484        let pid = std::process::id();
485        let total = |usage: ResourceUsage| usage.cpu_user + usage.cpu_system;
486        let before = total(resource_usage(pid).unwrap());
487        let busy = Duration::from_millis(400);
488        burn_cpu(busy);
489        let after = total(resource_usage(pid).unwrap());
490        let grown = after.saturating_sub(before);
491        // This thread alone spent `busy` of CPU time, so the process total
492        // grew by at least that much; other tests' threads only add to it.
493        // The 10% allowance covers tick rounding in the reading, and is far
494        // tighter than the ~24x a unit error would cause.
495        assert!(
496            grown >= busy * 9 / 10,
497            "cpu time grew by {grown:?} over {busy:?} of busy work"
498        );
499    }
500
501    #[test]
502    fn a_child_reads_its_own_usage_not_ours() {
503        let mut child = spawn_sleep();
504        let process = Process::open(child.id()).unwrap().unwrap();
505        wait_for_executable(&process, sleep_identity());
506        let ours = resource_usage(std::process::id()).unwrap();
507        let usage = resource_usage(child.id()).expect("live child is readable");
508        assert!(usage.memory_bytes > 0);
509        assert!(
510            usage.memory_bytes < ours.memory_bytes,
511            "a sleeping child ({} bytes) should be smaller than the test binary ({} bytes)",
512            usage.memory_bytes,
513            ours.memory_bytes
514        );
515        child.kill().unwrap();
516        child.wait().unwrap();
517    }
518
519    #[test]
520    fn an_exited_child_reads_as_unavailable_not_zero() {
521        let mut child = spawn_sleep();
522        let pid = child.id();
523        child.kill().unwrap();
524        // Killed but not reaped: a zombie, which still has a pid.
525        let deadline = Instant::now() + Duration::from_secs(5);
526        while start_time(pid).is_some() {
527            assert!(Instant::now() < deadline, "child still observed as alive");
528            std::thread::sleep(Duration::from_millis(10));
529        }
530        assert_eq!(resource_usage(pid), None, "a zombie reads as unavailable");
531        child.wait().unwrap();
532        // Reaped: the pid names nothing (barring reuse, which would be some
533        // other live process and so still not a reading of zeros).
534        if let Some(usage) = resource_usage(pid) {
535            assert!(usage.memory_bytes > 0, "a reused pid is some live process");
536        }
537    }
538
539    #[test]
540    fn a_pid_with_no_process_reads_as_unavailable() {
541        // Above both kernels' pid limits (Linux caps pid_max at 2^22, macOS at
542        // 99998), so nothing can hold it.
543        assert_eq!(resource_usage(i32::MAX as u32), None);
544        // Not a representable pid at all.
545        assert_eq!(resource_usage(u32::MAX), None);
546    }
547
548    #[cfg(target_os = "linux")]
549    #[test]
550    fn linux_signals_go_through_a_pidfd() {
551        let mut child = spawn_sleep();
552        let process = Process::open(child.id()).unwrap().unwrap();
553        assert!(process.signals_through_pidfd());
554        child.kill().unwrap();
555        child.wait().unwrap();
556        // The pidfd still names the reaped child, so a signal cannot reach anything else.
557        assert!(!process.signal(Signal::Kill).unwrap());
558    }
559}