subc_os/lib.rs
1//! Operating-system primitives the subc daemon needs and cannot reach without
2//! unsafe code, each behind a small safe API.
3//!
4//! The daemon crates forbid unsafe code. This crate is the one deliberate
5//! exception (like `subc-uptime` and `subc-cgroup`): every `unsafe` block here
6//! is a single foreign call with its preconditions stated beside it, and nothing
7//! unsafe is exported.
8//!
9//! Today it answers one question: is the process now holding pid N the same
10//! process the daemon spawned earlier? A pid alone cannot say, because the
11//! kernel reuses pids once a process has been reaped. [`Process`] reads the two
12//! facts that tell processes apart, the kernel's start time for the pid and the
13//! file identity (device and inode) of the executable image it runs, and sends
14//! signals to it.
15//!
16//! Sources, per platform:
17//!
18//! - Linux: the start time is field 22 of `/proc/<pid>/stat` (clock ticks since
19//! boot), and the executable is `stat` through `/proc/<pid>/exe`, which
20//! resolves to the running image even if its file has since been replaced or
21//! deleted. A pidfd is opened before either is read and signals go through it
22//! (`pidfd_send_signal`), so the process that was checked is the process that
23//! is signalled. No unsafe code is needed: rustix wraps both calls.
24//! - macOS: the start time is `kp_proc.p_starttime` from `sysctl`
25//! `KERN_PROC_PID` (microseconds since the epoch), and the executable is the
26//! path `proc_pidpath` reports, then `stat` on that path. These two calls are
27//! unsafe. macOS has no pidfd, so a signal is a plain
28//! `kill` sent right after the checks; see [`Process::signal`].
29//! - Anywhere else: [`Process::open`] reports [`std::io::ErrorKind::Unsupported`].
30//!
31//! For persisted PID owners, [`process_identity`] reads versioned kernel start
32//! identities and distinguishes alive, dead and unknown without spawning a
33//! process. Its foreign calls are signal-zero `kill` on Unix and `proc_pidinfo`
34//! on macOS. Only dead owners may be reclaimed; unknown owners stay protected.
35//!
36//! It also reads how much memory and CPU time one process is using, for
37//! reporting only; see [`resource_usage`]. On Linux that is procfs again; on
38//! macOS it is `proc_pid_rusage`, plus `mach_timebase_info` to convert its CPU
39//! times to nanoseconds, the other two unsafe calls in the crate.
40//!
41//! And it carries the launch nonce from the daemon to each module it spawns
42//! over an inherited pipe instead of the environment: [`launch_nonce`] is the
43//! one reader every module uses, and [`LaunchNonceHandoff`] the daemon's half.
44//! That module's unsafe code is `dup2`, `fcntl`, `fstat` and `ioctl` on
45//! descriptors, each with its preconditions stated beside it.
46
47#![deny(unsafe_code)]
48
49#[cfg(all(unix, feature = "test-support"))]
50pub mod fork_exec_test;
51pub mod launch_nonce;
52pub mod privacy_identity;
53pub mod process_identity;
54#[cfg(unix)]
55pub use launch_nonce::LaunchNonceHandoff;
56pub use launch_nonce::{
57 launch_nonce, LaunchNonce, LaunchNonceError, LaunchNonceSource, LAUNCH_NONCE_ENV,
58 LAUNCH_NONCE_FD, LAUNCH_NONCE_FD_ENV,
59};
60
61#[cfg(target_os = "linux")]
62mod linux;
63#[cfg(target_os = "macos")]
64mod macos;
65
66#[cfg(target_os = "linux")]
67use linux as platform;
68#[cfg(target_os = "macos")]
69use macos as platform;
70
71use std::{io, path::Path};
72
73/// True where [`Process`] can identify and signal a process by pid.
74pub const PROCESS_IDENTITY_SUPPORTED: bool = cfg!(any(target_os = "linux", target_os = "macos"));
75
76/// Device and inode of a file: which file, independent of the name used to
77/// reach it.
78#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
79pub struct FileIdentity {
80 pub device: u64,
81 pub inode: u64,
82}
83
84/// The device and inode of the file at `path`, following symlinks. `None` if it
85/// cannot be read or the platform has no inode numbers.
86pub fn file_identity(path: &Path) -> Option<FileIdentity> {
87 #[cfg(unix)]
88 {
89 use std::os::unix::fs::MetadataExt;
90
91 std::fs::metadata(path).ok().map(|metadata| FileIdentity {
92 device: metadata.dev(),
93 inode: metadata.ino(),
94 })
95 }
96 #[cfg(not(unix))]
97 {
98 let _ = path;
99 None
100 }
101}
102
103/// What a live process looks like right now.
104#[derive(Debug, Clone, Copy, PartialEq, Eq)]
105pub struct Observation {
106 /// The kernel's start time for the process. Opaque: compare it only with a
107 /// value read on the same host by this crate. Linux counts clock ticks since
108 /// boot; macOS counts microseconds since the epoch.
109 pub start_time: u64,
110 /// The file the process is executing, or `None` if it could not be read
111 /// (for example, a process owned by another user).
112 pub executable: Option<FileIdentity>,
113}
114
115/// A signal [`Process::signal`] can send.
116#[derive(Debug, Clone, Copy, PartialEq, Eq)]
117pub enum Signal {
118 /// SIGTERM: a request to exit, which the process may handle or ignore.
119 Terminate,
120 /// SIGKILL: ends the process; it cannot be handled or ignored.
121 Kill,
122}
123
124/// The kernel start time of the process holding `pid`, or `None` if there is
125/// none, it has already exited (a zombie waiting to be reaped counts as exited),
126/// or the platform has no source.
127pub fn start_time(pid: u32) -> Option<u64> {
128 #[cfg(any(target_os = "linux", target_os = "macos"))]
129 {
130 platform::start_time(pid)
131 }
132 #[cfg(not(any(target_os = "linux", target_os = "macos")))]
133 {
134 let _ = pid;
135 None
136 }
137}
138
139/// True where [`resource_usage`] can read a live process. Elsewhere it always
140/// answers `None`, and a caller can use this to say "not supported here"
141/// rather than "could not read".
142pub const RESOURCE_USAGE_SUPPORTED: bool = cfg!(any(target_os = "linux", target_os = "macos"));
143
144/// What [`ResourceUsage::memory_bytes`] measures. The platforms offer
145/// different figures, and they are not interchangeable.
146#[derive(Debug, Clone, Copy, PartialEq, Eq)]
147pub enum MemoryKind {
148 /// macOS `phys_footprint`: the memory the kernel charges to the process
149 /// (dirty and compressed pages, among others), which is also what jetsam
150 /// acts on. Pages an allocator has released with `MADV_FREE` do not count.
151 PhysFootprint,
152 /// Linux `VmRSS`: pages of the process resident in RAM, including shared
153 /// file-backed pages. Swapped-out pages are not included; see
154 /// [`ResourceUsage::swap_bytes`].
155 ResidentSet,
156}
157
158/// One reading of a process's memory and cumulative CPU time.
159///
160/// It covers the process named by the pid alone: its threads are included,
161/// processes it has started are not.
162#[derive(Debug, Clone, Copy, PartialEq, Eq)]
163pub struct ResourceUsage {
164 /// Memory in bytes, measured as [`Self::memory_kind`] says.
165 pub memory_bytes: u64,
166 pub memory_kind: MemoryKind,
167 /// Bytes swapped out (Linux `VmSwap`). `None` where the platform does not
168 /// report it for a single process, which is not the same as zero.
169 pub swap_bytes: Option<u64>,
170 /// CPU time spent in user mode since the process started.
171 pub cpu_user: std::time::Duration,
172 /// CPU time spent in the kernel on the process's behalf since it started.
173 pub cpu_system: std::time::Duration,
174}
175
176/// Memory and cumulative CPU time of the process holding `pid`, read now.
177///
178/// `None` when there is no such process, it has exited (a zombie awaiting its
179/// reap counts as exited), it cannot be read (for example, another user's
180/// process on macOS), or the platform has no source
181/// (see [`RESOURCE_USAGE_SUPPORTED`]). Never a reading of zeros in place of
182/// one of those.
183///
184/// Like any pid-based read, this describes whatever process holds `pid` now;
185/// a caller that needs it to be a particular process should confirm that
186/// process's [`start_time`] around the call.
187pub fn resource_usage(pid: u32) -> Option<ResourceUsage> {
188 #[cfg(any(target_os = "linux", target_os = "macos"))]
189 {
190 platform::resource_usage(pid)
191 }
192 #[cfg(not(any(target_os = "linux", target_os = "macos")))]
193 {
194 let _ = pid;
195 None
196 }
197}
198
199/// A handle on the process holding one pid at the moment it was opened.
200#[derive(Debug)]
201pub struct Process {
202 pid: u32,
203 #[cfg(target_os = "linux")]
204 pidfd: Option<std::os::fd::OwnedFd>,
205}
206
207impl Process {
208 /// Open a handle on the process now holding `pid`.
209 ///
210 /// `Ok(None)` means no process holds that pid (on macOS, also a zombie
211 /// awaiting its reap; on Linux a zombie opens, and [`Self::observe`] then
212 /// reports it as exited). On Linux this opens a pidfd,
213 /// which from then on refers to this exact process even if it exits and the
214 /// pid is reused; when the kernel cannot open one (older than 5.3, or a
215 /// seccomp policy refusing the call) the handle falls back to the pid, as
216 /// on macOS.
217 pub fn open(pid: u32) -> io::Result<Option<Self>> {
218 #[cfg(target_os = "linux")]
219 {
220 linux::open(pid).map(|opened| opened.map(|pidfd| Self { pid, pidfd }))
221 }
222 #[cfg(target_os = "macos")]
223 {
224 Ok(platform::exists(pid).then_some(Self { pid }))
225 }
226 #[cfg(not(any(target_os = "linux", target_os = "macos")))]
227 {
228 let _ = pid;
229 Err(io::Error::new(
230 io::ErrorKind::Unsupported,
231 "process identity is not available on this platform",
232 ))
233 }
234 }
235
236 pub fn pid(&self) -> u32 {
237 self.pid
238 }
239
240 /// True when signals go through a pidfd, so they cannot reach a different
241 /// process that has since reused this pid.
242 pub fn signals_through_pidfd(&self) -> bool {
243 #[cfg(target_os = "linux")]
244 {
245 self.pidfd.is_some()
246 }
247 #[cfg(not(target_os = "linux"))]
248 {
249 false
250 }
251 }
252
253 /// The process's start time and executable, or `None` once it has exited
254 /// (including as a zombie not yet reaped by its parent).
255 pub fn observe(&self) -> Option<Observation> {
256 #[cfg(any(target_os = "linux", target_os = "macos"))]
257 {
258 #[cfg(target_os = "linux")]
259 if !linux::pidfd_alive(self.pidfd.as_ref()) {
260 return None;
261 }
262 let start_time = platform::start_time(self.pid)?;
263 Some(Observation {
264 start_time,
265 executable: platform::executable_identity(self.pid),
266 })
267 }
268 #[cfg(not(any(target_os = "linux", target_os = "macos")))]
269 {
270 None
271 }
272 }
273
274 /// Send `signal` to the process.
275 ///
276 /// With a pidfd the signal can only reach the process this handle was
277 /// opened on: if that process has exited, the call fails with `ESRCH` even
278 /// if the pid has been reused. Without one (macOS, or a Linux kernel with no
279 /// pidfd) the signal goes to whatever holds the pid now, so callers should
280 /// [`Self::observe`] immediately before signalling. What remains is the
281 /// time between that check and this call; for a different process to be
282 /// hit, the checked one must exit, be reaped, and have its pid handed to a
283 /// new process inside that window, and both kernels hand out pids in
284 /// increasing order, so a reuse needs the whole pid space to wrap first.
285 ///
286 /// `Ok(false)` means the process had already exited (`ESRCH`).
287 pub fn signal(&self, signal: Signal) -> io::Result<bool> {
288 #[cfg(any(target_os = "linux", target_os = "macos"))]
289 {
290 #[cfg(target_os = "linux")]
291 let result = linux::signal(self.pid, self.pidfd.as_ref(), signal);
292 #[cfg(target_os = "macos")]
293 let result = macos::signal(self.pid, signal);
294 match result {
295 Ok(()) => Ok(true),
296 Err(rustix::io::Errno::SRCH) => Ok(false),
297 Err(error) => Err(error.into()),
298 }
299 }
300 #[cfg(not(any(target_os = "linux", target_os = "macos")))]
301 {
302 let _ = signal;
303 Err(io::Error::new(
304 io::ErrorKind::Unsupported,
305 "process signalling is not available on this platform",
306 ))
307 }
308 }
309}
310
311#[cfg(all(test, any(target_os = "linux", target_os = "macos")))]
312mod tests {
313 use std::{
314 process::{Child, Command},
315 time::{Duration, Instant},
316 };
317
318 use super::*;
319
320 fn spawn_sleep() -> Child {
321 Command::new("sleep")
322 .arg("60")
323 .spawn()
324 .expect("spawn sleep")
325 }
326
327 /// The executable a spawned `sleep` runs, resolved the way `Command` found it.
328 fn sleep_identity() -> FileIdentity {
329 let path = ["/bin/sleep", "/usr/bin/sleep"]
330 .into_iter()
331 .find(|path| Path::new(path).exists())
332 .expect("sleep is installed");
333 file_identity(Path::new(path)).expect("stat sleep")
334 }
335
336 /// Right after `spawn` returns the child may not have finished exec yet,
337 /// and until then it still runs the test binary's image.
338 fn wait_for_executable(process: &Process, expected: FileIdentity) -> Observation {
339 let deadline = Instant::now() + Duration::from_secs(5);
340 loop {
341 let observation = process.observe().expect("child is alive");
342 if observation.executable == Some(expected) || Instant::now() > deadline {
343 return observation;
344 }
345 std::thread::sleep(Duration::from_millis(10));
346 }
347 }
348
349 #[test]
350 fn own_process_is_observable_with_its_own_image() {
351 let process = Process::open(std::process::id())
352 .expect("open own process")
353 .expect("own process exists");
354 let observation = process.observe().expect("own process is alive");
355 let own_image = file_identity(&std::env::current_exe().unwrap()).unwrap();
356 assert_eq!(observation.executable, Some(own_image));
357 assert_eq!(start_time(std::process::id()), Some(observation.start_time));
358 }
359
360 #[test]
361 fn child_start_time_is_stable_and_differs_from_ours() {
362 let mut child = spawn_sleep();
363 let pid = child.id();
364 let process = Process::open(pid).unwrap().unwrap();
365 let observation = wait_for_executable(&process, sleep_identity());
366 assert_eq!(observation.executable, Some(sleep_identity()));
367 assert_eq!(start_time(pid), Some(observation.start_time));
368 child.kill().unwrap();
369 child.wait().unwrap();
370 }
371
372 #[test]
373 fn a_signalled_and_unreaped_child_reads_as_exited() {
374 let mut child = spawn_sleep();
375 let process = Process::open(child.id()).unwrap().unwrap();
376 assert!(process.signal(Signal::Terminate).unwrap());
377 let deadline = Instant::now() + Duration::from_secs(5);
378 while process.observe().is_some() {
379 assert!(Instant::now() < deadline, "child still observed as alive");
380 std::thread::sleep(Duration::from_millis(10));
381 }
382 // Not yet reaped: the pid is still a zombie here, and still reads as exited.
383 assert_eq!(start_time(child.id()), None);
384 child.wait().unwrap();
385 }
386
387 #[test]
388 fn a_reaped_child_cannot_be_opened_or_observed() {
389 let mut child = spawn_sleep();
390 let pid = child.id();
391 child.kill().unwrap();
392 child.wait().unwrap();
393 // The pid could in principle be reused by now; either way it is not the child.
394 if let Some(process) = Process::open(pid).unwrap() {
395 if let Some(observation) = process.observe() {
396 assert_ne!(observation.executable, Some(sleep_identity()));
397 }
398 }
399 }
400
401 /// The macOS fields are read at fixed offsets, so check the value is a
402 /// plausible start time and not some other field: our own process started
403 /// in the past, and not long ago.
404 #[cfg(target_os = "macos")]
405 #[test]
406 fn macos_start_time_is_microseconds_since_the_epoch() {
407 let now = std::time::SystemTime::now()
408 .duration_since(std::time::UNIX_EPOCH)
409 .unwrap()
410 .as_micros() as u64;
411 let started = start_time(std::process::id()).unwrap();
412 assert!(started <= now, "start time {started} is after now {now}");
413 assert!(
414 now - started < 3_600 * 1_000_000,
415 "start time {started} is more than an hour before now {now}"
416 );
417 }
418
419 /// Keeps one core busy for at least `wall` of wall-clock time.
420 /// This thread's CPU time, from the thread CPU clock rather than the
421 /// process-usage API under test.
422 fn thread_cpu_time() -> Duration {
423 let now = rustix::time::clock_gettime(rustix::time::ClockId::ThreadCPUTime);
424 Duration::new(now.tv_sec as u64, now.tv_nsec as u32)
425 }
426
427 /// Spend `cpu` of this thread's CPU time. Measured on CPU time, not wall
428 /// time: on a loaded machine the thread is descheduled for part of any
429 /// wall interval, so a wall-timed loop can do far less work than its
430 /// duration suggests. A generous wall cap keeps a stalled clock from
431 /// hanging the test.
432 fn burn_cpu(cpu: Duration) {
433 let start = thread_cpu_time();
434 let give_up = Instant::now() + Duration::from_secs(60);
435 let mut value = 0u64;
436 while thread_cpu_time().saturating_sub(start) < cpu {
437 assert!(
438 Instant::now() < give_up,
439 "thread CPU clock stopped advancing"
440 );
441 for step in 0..10_000u64 {
442 value = std::hint::black_box(value.wrapping_mul(31).wrapping_add(step));
443 }
444 }
445 std::hint::black_box(value);
446 }
447
448 #[test]
449 fn own_resource_usage_is_present_and_plausible() {
450 // Clean executable pages need not count toward physical footprint, and
451 // nextest runs this case in a fresh process with little private memory.
452 // Touch and retain private pages so the byte/unit check has a known
453 // lower bound instead of assuming a minimum footprint for the binary.
454 let pages = vec![0xa5u8; 8 * 1024 * 1024];
455 std::hint::black_box(&pages);
456 let usage = resource_usage(std::process::id()).expect("own process is readable");
457 assert!(
458 usage.memory_bytes >= pages.len() as u64,
459 "memory {} bytes cannot account for {} touched private bytes",
460 usage.memory_bytes,
461 pages.len()
462 );
463 std::hint::black_box(&pages);
464 assert!(
465 usage.memory_bytes < 64 * 1024 * 1024 * 1024,
466 "memory {} bytes is implausibly large",
467 usage.memory_bytes
468 );
469 #[cfg(target_os = "macos")]
470 assert_eq!(usage.memory_kind, MemoryKind::PhysFootprint);
471 #[cfg(target_os = "linux")]
472 {
473 assert_eq!(usage.memory_kind, MemoryKind::ResidentSet);
474 assert!(usage.swap_bytes.is_some(), "Linux reports VmSwap");
475 }
476 }
477
478 /// CPU time must grow with busy work, and by roughly the amount of work
479 /// done: a reading in the wrong unit (for example Mach ticks taken as
480 /// nanoseconds on Apple silicon, about 24 times too small) grows too, but
481 /// not by enough.
482 #[test]
483 fn own_cpu_time_grows_by_about_the_busy_work_done() {
484 let pid = std::process::id();
485 let total = |usage: ResourceUsage| usage.cpu_user + usage.cpu_system;
486 let before = total(resource_usage(pid).unwrap());
487 let busy = Duration::from_millis(400);
488 burn_cpu(busy);
489 let after = total(resource_usage(pid).unwrap());
490 let grown = after.saturating_sub(before);
491 // This thread alone spent `busy` of CPU time, so the process total
492 // grew by at least that much; other tests' threads only add to it.
493 // The 10% allowance covers tick rounding in the reading, and is far
494 // tighter than the ~24x a unit error would cause.
495 assert!(
496 grown >= busy * 9 / 10,
497 "cpu time grew by {grown:?} over {busy:?} of busy work"
498 );
499 }
500
501 #[test]
502 fn a_child_reads_its_own_usage_not_ours() {
503 let mut child = spawn_sleep();
504 let process = Process::open(child.id()).unwrap().unwrap();
505 wait_for_executable(&process, sleep_identity());
506 let ours = resource_usage(std::process::id()).unwrap();
507 let usage = resource_usage(child.id()).expect("live child is readable");
508 assert!(usage.memory_bytes > 0);
509 assert!(
510 usage.memory_bytes < ours.memory_bytes,
511 "a sleeping child ({} bytes) should be smaller than the test binary ({} bytes)",
512 usage.memory_bytes,
513 ours.memory_bytes
514 );
515 child.kill().unwrap();
516 child.wait().unwrap();
517 }
518
519 #[test]
520 fn an_exited_child_reads_as_unavailable_not_zero() {
521 let mut child = spawn_sleep();
522 let pid = child.id();
523 child.kill().unwrap();
524 // Killed but not reaped: a zombie, which still has a pid.
525 let deadline = Instant::now() + Duration::from_secs(5);
526 while start_time(pid).is_some() {
527 assert!(Instant::now() < deadline, "child still observed as alive");
528 std::thread::sleep(Duration::from_millis(10));
529 }
530 assert_eq!(resource_usage(pid), None, "a zombie reads as unavailable");
531 child.wait().unwrap();
532 // Reaped: the pid names nothing (barring reuse, which would be some
533 // other live process and so still not a reading of zeros).
534 if let Some(usage) = resource_usage(pid) {
535 assert!(usage.memory_bytes > 0, "a reused pid is some live process");
536 }
537 }
538
539 #[test]
540 fn a_pid_with_no_process_reads_as_unavailable() {
541 // Above both kernels' pid limits (Linux caps pid_max at 2^22, macOS at
542 // 99998), so nothing can hold it.
543 assert_eq!(resource_usage(i32::MAX as u32), None);
544 // Not a representable pid at all.
545 assert_eq!(resource_usage(u32::MAX), None);
546 }
547
548 #[cfg(target_os = "linux")]
549 #[test]
550 fn linux_signals_go_through_a_pidfd() {
551 let mut child = spawn_sleep();
552 let process = Process::open(child.id()).unwrap().unwrap();
553 assert!(process.signals_through_pidfd());
554 child.kill().unwrap();
555 child.wait().unwrap();
556 // The pidfd still names the reaped child, so a signal cannot reach anything else.
557 assert!(!process.signal(Signal::Kill).unwrap());
558 }
559}