Skip to main content

running_process_platform_internal/
platform_linux.rs

1//! Linux implementation root for the process capability.
2
3#[path = "platform_linux/foreground.rs"]
4pub(crate) mod foreground;
5
6/// Native niceness for the portable `ProcessPriority::Low` intent.
7pub(crate) const PRIORITY_NICE_LOW: i32 = 10;
8/// Native niceness for the portable `ProcessPriority::High` intent.
9pub(crate) const PRIORITY_NICE_HIGH: i32 = -5;
10
11#[path = "platform_linux/autostart.rs"]
12pub(crate) mod autostart;
13
14#[path = "platform_linux/resources.rs"]
15pub(crate) mod resources;
16pub use resources::{
17    fd_exhaustion_error as resources_fd_exhaustion_error,
18    inode_capacity as resources_inode_capacity,
19    signals_fd_exhaustion as resources_signals_fd_exhaustion,
20    signals_storage_exhaustion as resources_signals_storage_exhaustion,
21    storage_exhaustion_error as resources_storage_exhaustion_error,
22};
23
24pub use autostart::{
25    register as autostart_register,
26    render_registration as autostart_render_registration,
27    unregister as autostart_unregister,
28};
29
30#[path = "platform_linux/process_inspect.rs"]
31pub(crate) mod process_inspect;
32#[cfg(any(feature = "independent-spawn", test))]
33mod resource_placement;
34#[cfg(any(feature = "independent-spawn", test))]
35mod scheduler_launch;
36#[cfg(feature = "independent-spawn")]
37mod independent_spawn;
38#[cfg(feature = "independent-spawn")]
39mod independent_broker;
40#[cfg(feature = "independent-spawn")]
41pub use independent_broker::run as independent_broker_run;
42#[cfg(feature = "independent-spawn")]
43mod independent_broker_wire;
44#[cfg(feature = "independent-spawn")]
45pub use independent_spawn::{spawn as independent_spawn, spawn_broker as independent_broker_spawn, IndependentChild};
46#[cfg(feature = "independent-spawn")]
47mod independent_io;
48#[cfg(feature = "independent-spawn")]
49pub(crate) use independent_io::open_regular as independent_open_regular;
50#[cfg(feature = "independent-spawn")]
51pub(crate) const INDEPENDENT_ZERO_WRITE_PENDING: bool = false;
52pub use process_inspect::{
53    process_executable_path, process_force_kill, process_same_executable_path,
54    process_fault_code_name, process_signal_terminate, ProcessLiveness,
55};
56
57#[path = "platform_linux/loaded_images.rs"]
58mod loaded_images;
59pub use loaded_images::{
60    loaded_images as process_loaded_images,
61    open_loaded_image_file as process_open_loaded_image_file,
62};
63
64#[path = "platform_linux/raw_write.rs"]
65pub(crate) mod raw_write;
66pub use raw_write::write_all_to_descriptor as fs_write_all_to_descriptor;
67
68/// Whether a handle another process holds open keeps a file from being removed.
69///
70/// Linux unlinks a name while descriptors on it stay open, so
71/// callers use the answer to decide whether releasing handles before a
72/// recursive delete means anything here.
73pub const fn fs_open_handles_block_removal() -> bool {
74    false
75}
76
77#[path = "platform_linux/shutdown_request.rs"]
78pub(crate) mod shutdown_request;
79pub use shutdown_request::install_shutdown_request_handler as process_install_shutdown_request_handler;
80
81#[path = "platform_linux/process_owner_death.rs"]
82pub(crate) mod process_owner_death;
83pub use process_owner_death::{
84    install_owner_death_cleanup as process_install_owner_death_cleanup,
85    owner_death_cleanup_target as process_owner_death_cleanup_target,
86};
87
88#[path = "platform_linux/host.rs"]
89pub(crate) mod host;
90pub use host::{
91    boot_id as host_boot_id, current_process_privilege as host_current_process_privilege,
92    environment_keys_are_case_insensitive as host_environment_keys_are_case_insensitive,
93    filesystem_device_id as host_filesystem_device_id, hostname as host_hostname,
94    login_environment as host_login_environment, machine_id as host_machine_id,
95    namespace_id as host_namespace_id, user_machine_identity as host_user_machine_identity,
96    PrivilegedIdentity as HostPrivilegedIdentity,
97};
98pub use host::login_environment_block as host_login_environment_block;
99
100/// This process's control-group membership (`/proc/self/cgroup`). Linux has
101/// cgroups, so the answer is always `Some`; the inner error is a read failure.
102pub fn host_process_cgroup() -> Option<io::Result<String>> {
103    Some(std::fs::read_to_string("/proc/self/cgroup"))
104}
105
106#[cfg(feature = "fs")]
107#[path = "platform_linux/fs.rs"]
108pub(crate) mod fs;
109#[cfg(feature = "fs")]
110pub use fs::{
111    create_private_file as fs_create_private_file,
112    is_link_handle as fs_is_link_handle, open_read_no_follow as fs_open_read_no_follow,
113    decode_path_bytes as fs_decode_path_bytes,
114    replace_file as fs_replace_file, sync_directory as fs_sync_directory,
115    user_config_dir as fs_user_config_dir,
116    user_data_dir as fs_user_data_dir, encode_path_bytes as fs_encode_path_bytes,
117    file_identity as fs_file_identity, is_lock_conflict as fs_is_lock_conflict,
118    open_lock_file as fs_open_lock_file, path_identity as fs_path_identity,
119    try_lock_exclusive as fs_try_lock_exclusive, unlock as fs_unlock,
120    user_run_data_root as fs_user_run_data_root, user_runtime_dir as fs_user_runtime_dir,
121    user_state_dir as fs_user_state_dir,
122    user_state_dir_from_environment as fs_user_state_dir_from_environment,
123    state_home_from_environment as fs_state_home_from_environment,
124    FileIdentity as FsFileIdentity,
125};
126
127#[path = "platform_linux/ape.rs"]
128pub(crate) mod ape;
129pub use ape::{
130    default_loader_dirs as ape_default_loader_dirs, is_exec_format_error as ape_is_exec_format_error,
131    is_executable as ape_is_executable, mark_executable as ape_mark_executable,
132    anonymous_executable as ape_anonymous_executable,
133    private_exec_dir as ape_private_exec_dir,
134    route_through_execvp as ape_route_through_execvp, APE_LOADER_HOST,
135    APE_EXECVP_SHELL_FALLBACK, APE_NEEDS_LOADER, APE_SHELL, APE_SYSTEM_LOADERS,
136};
137#[cfg(feature = "async-process")]
138pub use ape::route_tokio_through_execvp as ape_route_tokio_through_execvp;
139
140#[path = "platform_linux/executable.rs"]
141pub(crate) mod executable;
142pub use executable::{
143    file_name as executable_file_name,
144    sibling_of_current_image as executable_sibling_of_current_image,
145    EXECUTABLE_EXTENSION,
146};
147
148#[cfg(feature = "ipc")]
149#[path = "platform_linux/ipc.rs"]
150pub(crate) mod ipc;
151#[cfg(feature = "private-dir")]
152#[path = "platform_linux/ipc_private_dir.rs"]
153mod ipc_private_dir;
154#[cfg(feature = "ipc")]
155pub use ipc::{
156    current_user_id as ipc_current_user_id, Endpoint as IpcEndpoint,
157    endpoint_is_filesystem_backed as ipc_endpoint_is_filesystem_backed,
158    handoff_transport_available as ipc_handoff_transport_available,
159    nonblocking_zero_read_is_pending as ipc_nonblocking_zero_read_is_pending,
160    select_endpoint_address as ipc_select_endpoint_address,
161    InheritedListener as IpcInheritedListener, Listener as IpcListener,
162    ListenerNonblockingMode as IpcListenerNonblockingMode, PeerIdentity as IpcPeerIdentity,
163    PeerIdentitySource as IpcPeerIdentitySource, Stream as IpcStream,
164};
165#[cfg(feature = "ipc")]
166pub const LEGACY_SCM_RIGHTS_TRANSPORT_SUPPORTED: bool = true;
167#[cfg(feature = "ipc")]
168pub const LEGACY_DUPLICATE_HANDLE_TRANSPORT_SUPPORTED: bool = false;
169#[cfg(feature = "ipc")]
170pub use ipc::{legacy_send_fd_over, legacy_send_fd_to};
171#[cfg(feature = "ipc")]
172pub fn legacy_duplicate_handle(
173    _source_handle: usize,
174    _backend_pid: u32,
175) -> Result<usize, crate::LegacyHandoffError> {
176    Err(crate::LegacyHandoffError::new(
177        crate::platform::ipc::HandoffTransferErrorKind::Unsupported,
178        None,
179    ))
180}
181#[cfg(feature = "private-dir")]
182pub use ipc_private_dir::{
183    ensure_owner_private_directory as private_dir_ensure_owner_private_directory,
184    owner_private_directory as private_dir_owner_private_directory,
185};
186#[cfg(feature = "ipc")]
187pub fn ipc_broker_endpoint_name(bare_name: &str, path_scoped: bool) -> std::io::Result<String> {
188    use std::fmt::Write as _;
189    use std::path::PathBuf;
190
191    if path_scoped {
192        let mut hash = blake3::Hasher::new();
193        hash.update(b"running-process:path-scoped-socket:v1\0");
194        hash.update(bare_name.as_bytes());
195        let mut leaf = String::with_capacity(32);
196        for byte in hash.finalize().as_bytes().iter().take(16) { let _ = write!(leaf, "{byte:02x}"); }
197        return Ok(PathBuf::from("/tmp").join(format!(".rp-path-{leaf}.sock")).to_string_lossy().into_owned());
198    }
199    Ok(ipc_component_endpoint_path("broker-v2", bare_name))
200}
201
202/// Concrete socket path for `bare_name` in the per-user runtime directory of
203/// `component` (`broker-v2`, `probe`, ...). Pure: performs no filesystem write.
204///
205/// The component only picks the directory leaf, so two services never share a
206/// namespace while following one convention (#974).
207#[cfg(feature = "ipc")]
208pub fn ipc_component_endpoint_path(component: &str, bare_name: &str) -> String {
209    // SAFETY: `getuid` reads a process property and cannot fail.
210    let uid = unsafe { libc::getuid() };
211    component_endpoint_path_in(crate::env_vars::XDG_RUNTIME_DIR.os(), uid, component, bare_name)
212}
213
214/// Per-user runtime directory of `component`: the directory that holds its
215/// sockets and any runtime files published beside them. Pure.
216#[cfg(feature = "ipc")]
217pub fn ipc_component_runtime_dir(component: &str) -> std::path::PathBuf {
218    // SAFETY: `getuid` reads a process property and cannot fail.
219    let uid = unsafe { libc::getuid() };
220    component_runtime_dir_in(crate::env_vars::XDG_RUNTIME_DIR.os(), uid, component)
221}
222
223#[cfg(feature = "ipc")]
224fn component_runtime_dir_in(
225    xdg_runtime_dir: Option<std::ffi::OsString>,
226    uid: u32,
227    component: &str,
228) -> std::path::PathBuf {
229    use std::path::PathBuf;
230
231    match xdg_runtime_dir {
232        Some(value) => PathBuf::from(value).join("running-process").join(component),
233        None => PathBuf::from(format!("/tmp/running-process-{uid}/{component}")),
234    }
235}
236
237#[cfg(feature = "ipc")]
238fn component_endpoint_path_in(
239    xdg_runtime_dir: Option<std::ffi::OsString>,
240    uid: u32,
241    component: &str,
242    bare_name: &str,
243) -> String {
244    component_runtime_dir_in(xdg_runtime_dir, uid, component)
245        .join(format!("{bare_name}.sock"))
246        .to_string_lossy()
247        .into_owned()
248}
249
250/// Linux `sun_path` is 108 bytes including the NUL terminator.
251#[cfg(feature = "ipc")]
252const LINUX_SUN_PATH_MAX: usize = 108;
253
254#[cfg(feature = "ipc")]
255pub fn ipc_endpoint_name_limit() -> crate::platform::ipc::EndpointNameLimit {
256    crate::platform::ipc::EndpointNameLimit {
257        max_bytes: LINUX_SUN_PATH_MAX,
258        label: "Linux sun_path",
259    }
260}
261
262/// Directory holding v1 broker sockets.
263///
264/// Deliberately performs no filesystem writes: name derivation stays pure so
265/// the hash and length-limit tests remain deterministic. Callers that bind
266/// create the parent directory themselves.
267#[cfg(feature = "ipc")]
268fn broker_v1_socket_dir() -> std::path::PathBuf {
269    use std::path::PathBuf;
270
271    match crate::env_vars::XDG_RUNTIME_DIR.os() {
272        Some(dir) => PathBuf::from(dir).join("running-process").join("broker"),
273        None => PathBuf::from(format!(
274            "/tmp/running-process-{}/broker",
275            unsafe { libc::getuid() }
276        )),
277    }
278}
279
280#[cfg(feature = "ipc")]
281pub fn ipc_broker_v1_endpoint_path(
282    bare_name: &str,
283) -> Result<String, crate::platform::ipc::EndpointNameTooLong> {
284    // Linux gets 108 bytes and a guaranteed $XDG_RUNTIME_DIR (or a short
285    // /tmp fallback), so the full canonical name survives for debuggability.
286    let candidate = broker_v1_socket_dir().join(format!("{bare_name}.sock"));
287    let candidate = candidate.to_string_lossy();
288    // sockaddr_un is NUL-terminated, so the path itself must be strictly
289    // shorter than the field width.
290    if candidate.len() >= LINUX_SUN_PATH_MAX {
291        return Err(crate::platform::ipc::EndpointNameTooLong {
292            len: candidate.len(),
293            max: LINUX_SUN_PATH_MAX - 1,
294            limit_label: "Linux sun_path",
295        });
296    }
297    Ok(candidate.into_owned())
298}
299
300#[cfg(feature = "ipc")]
301pub fn ipc_endpoint_scope_bytes(path: &std::path::Path) -> Vec<u8> {
302    // Linux paths are opaque byte strings; no spelling difference is
303    // meaningless, so the bytes are hashed exactly as the OS reports them.
304    use std::os::unix::ffi::OsStrExt as _;
305
306    path.as_os_str().as_bytes().to_vec()
307}
308
309#[cfg(feature = "ipc")]
310pub fn ipc_broker_v2_runtime_dir() -> std::path::PathBuf {
311    match crate::env_vars::XDG_RUNTIME_DIR.os() {
312        Some(dir) => std::path::PathBuf::from(dir)
313            .join("running-process")
314            .join("broker-v2"),
315        None => crate::platform::ipc::per_user_runtime_fallback(),
316    }
317}
318#[cfg(feature = "ipc")]
319pub fn into_legacy_ipc_stream(stream: IpcStream) -> interprocess::local_socket::Stream {
320    stream.0
321}
322
323#[cfg(feature = "ipc")]
324pub fn from_legacy_ipc_stream(stream: interprocess::local_socket::Stream) -> IpcStream {
325    ipc::Stream(stream)
326}
327#[cfg(feature = "ipc")]
328pub fn legacy_ipc_name(path: &str) -> Result<interprocess::local_socket::Name<'_>, String> {
329    ipc::legacy_name(path)
330}
331#[cfg(feature = "ipc-async")]
332pub use ipc::{
333    AsyncListener as IpcAsyncListener, AsyncStream as IpcAsyncStream,
334    IntoAsyncListener as IpcIntoAsyncListener, IntoAsyncStream as IpcIntoAsyncStream,
335};
336
337#[cfg(feature = "session-relay")]
338#[path = "platform_linux_session_relay.rs"]
339mod session_relay;
340#[cfg(feature = "session-relay")]
341pub use session_relay::relay_local_socket_session;
342
343#[cfg(feature = "pty")]
344#[path = "platform_linux/terminal.rs"]
345pub mod terminal;
346#[cfg(feature = "terminal-graphics")]
347#[path = "platform_linux/terminal_graphics.rs"]
348mod terminal_graphics;
349#[cfg(feature = "terminal-graphics")]
350pub use terminal_graphics::active_graphics_probe;
351pub use crate::platform::terminal_input;
352
353#[cfg(feature = "window-icon")]
354#[path = "platform_linux/window_icon.rs"]
355mod window_icon;
356#[cfg(feature = "window-icon")]
357pub use window_icon::{icon_support as window_icon_support_impl, set_icon as set_window_icon_impl};
358
359pub fn shell_command(command: &str) -> std::process::Command {
360    let mut shell = std::process::Command::new("/bin/sh");
361    shell.arg("-lc").arg(command);
362    shell
363}
364
365pub fn compat_shell_command(command: &str) -> std::process::Command {
366    let mut shell = std::process::Command::new("/bin/sh");
367    shell.arg("-lc").arg(command);
368    shell
369}
370
371pub fn canonical_environment_pairs(pairs: Vec<(String, String)>) -> Vec<(String, String)> {
372    pairs
373}
374
375pub fn monitor_console_windows(
376    _duration: std::time::Duration,
377) -> Vec<crate::platform::process::ConsoleWindowInfo> {
378    Vec::new()
379}
380
381#[cfg(feature = "async-process")]
382use std::ffi::OsStr;
383use std::io;
384use std::io::Read;
385use std::os::fd::{AsRawFd, RawFd};
386use std::os::unix::net::UnixStream;
387use std::sync::Mutex;
388
389#[cfg(feature = "async-process")]
390use tokio::process::{Child, Command};
391
392#[cfg(feature = "async-process")]
393use crate::SpawnSpec;
394
395#[path = "platform_linux_descendants.rs"]
396mod descendants;
397pub use descendants::start_descendant_monitor;
398
399/// Attach to an already-running root (#1015). On this host the descendant
400/// monitor never depended on the spawn, so attaching is the same monitor.
401pub fn start_attached_descendant_monitor(
402    root_pid: u32,
403    stop: std::sync::Arc<crate::platform::process::DescendantMonitorStop>,
404    emit: Box<dyn Fn(crate::platform::process::DescendantEvent) + Send>,
405) -> std::io::Result<()> {
406    start_descendant_monitor(root_pid, stop, emit)
407}
408
409#[path = "platform_linux_trace.rs"]
410mod exact_trace;
411pub use exact_trace::{configure_exact_trace, start_exact_trace, TracedChild};
412
413pub fn exact_trace_capability() -> crate::platform::process::ExactTraceCapability {
414    crate::platform::process::ExactTraceCapability {
415        available: true,
416        backend: "linux-ptrace",
417        reason: "launch-time PTRACE_TRACEME with follow-fork/clone/exec/exit supervision",
418        non_invasive_backend: "proc-descendant-snapshot",
419        non_invasive_grade:
420            crate::platform::process::NonInvasiveObservationGrade::SnapshotInferred,
421    }
422}
423
424pub struct WindowsJobHandle;
425
426pub fn assign_child_to_windows_job(
427    _child: &std::process::Child,
428    _direct_pid: u32,
429    _address_space_limit_bytes: Option<u64>,
430    _emit: Option<Box<dyn Fn(crate::platform::process::DescendantEvent) + Send>>,
431) -> io::Result<WindowsJobHandle> {
432    Err(io::Error::new(
433        io::ErrorKind::Unsupported,
434        "Windows Job Objects are unavailable on Linux",
435    ))
436}
437
438#[derive(Default)]
439pub struct CaptureCancellation {
440    wakers: Mutex<CaptureWakers>,
441}
442
443#[derive(Default)]
444struct CaptureWakers {
445    stdout: Option<UnixStream>,
446    stderr: Option<UnixStream>,
447}
448
449struct CancelableCaptureReader<R> {
450    reader: R,
451    wake_reader: UnixStream,
452}
453
454impl<R: Read + AsRawFd> Read for CancelableCaptureReader<R> {
455    fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
456        if buf.is_empty() { return Ok(0); }
457        loop {
458            let mut poll_fds = [
459                libc::pollfd { fd: self.reader.as_raw_fd(), events: libc::POLLIN | libc::POLLHUP | libc::POLLERR, revents: 0 },
460                libc::pollfd { fd: self.wake_reader.as_raw_fd(), events: libc::POLLIN | libc::POLLHUP | libc::POLLERR, revents: 0 },
461            ];
462            // SAFETY: both descriptors remain owned by this reader for the call.
463            let polled = unsafe { libc::poll(poll_fds.as_mut_ptr(), poll_fds.len() as _, -1) };
464            if polled < 0 {
465                let error = io::Error::last_os_error();
466                if error.kind() == io::ErrorKind::Interrupted { continue; }
467                return Err(error);
468            }
469            if poll_fds[1].revents != 0 {
470                return Err(io::Error::new(io::ErrorKind::Interrupted, "capture reader cancelled"));
471            }
472            if poll_fds[0].revents != 0 {
473                match self.reader.read(buf) {
474                    Err(error) if error.kind() == io::ErrorKind::WouldBlock => continue,
475                    result => return result,
476                }
477            }
478        }
479    }
480}
481
482pub fn prepare_capture_reader<R>(
483    reader: R,
484    cancellation: &CaptureCancellation,
485    stream: crate::platform::process::CaptureStream,
486) -> io::Result<Box<dyn Read + Send>>
487where R: Read + AsRawFd + Send + 'static {
488    set_nonblocking(reader.as_raw_fd())?;
489    let (wake_reader, wake_writer) = UnixStream::pair()?;
490    wake_writer.set_nonblocking(true)?;
491    let mut wakers = cancellation.wakers.lock().expect("capture wakers mutex poisoned");
492    match stream {
493        crate::platform::process::CaptureStream::Stdout => wakers.stdout = Some(wake_writer),
494        crate::platform::process::CaptureStream::Stderr => wakers.stderr = Some(wake_writer),
495    }
496    Ok(Box::new(CancelableCaptureReader { reader, wake_reader }))
497}
498
499pub fn capture_reader_done(cancellation: &CaptureCancellation, stream: crate::platform::process::CaptureStream) {
500    let mut wakers = cancellation.wakers.lock().expect("capture wakers mutex poisoned");
501    match stream {
502        crate::platform::process::CaptureStream::Stdout => wakers.stdout = None,
503        crate::platform::process::CaptureStream::Stderr => wakers.stderr = None,
504    }
505}
506
507pub fn cancel_capture_reader(cancellation: &CaptureCancellation) {
508    let wakers = cancellation.wakers.lock().expect("capture wakers mutex poisoned");
509    let byte = [1_u8; 1];
510    for writer in [&wakers.stdout, &wakers.stderr].into_iter().flatten() {
511        // SAFETY: the stored wake socket stays alive while the mutex is held.
512        let _ = unsafe { libc::write(writer.as_raw_fd(), byte.as_ptr().cast(), byte.len()) };
513    }
514}
515
516fn set_nonblocking(fd: RawFd) -> io::Result<()> {
517    // SAFETY: `fd` is borrowed from a live reader for both calls.
518    let flags = unsafe { libc::fcntl(fd, libc::F_GETFL) };
519    if flags < 0 { return Err(io::Error::last_os_error()); }
520    // SAFETY: `fd` is borrowed from a live reader for both calls.
521    if unsafe { libc::fcntl(fd, libc::F_SETFL, flags | libc::O_NONBLOCK) } < 0 {
522        return Err(io::Error::last_os_error());
523    }
524    Ok(())
525}
526
527#[path = "platform_linux_file_handles.rs"]
528mod file_handles;
529pub use file_handles::read_process_file_handles;
530#[path = "platform_linux_cmdline.rs"]
531mod cmdline;
532pub use cmdline::{read_process_argv, read_process_cmdline};
533
534#[cfg(feature = "process-inspection")]
535#[path = "platform/process_tree.rs"]
536mod process_tree;
537
538#[cfg(feature = "process-inspection")]
539pub fn kill_tree(pid: u32, timeout: std::time::Duration) -> io::Result<u32> {
540    process_tree::kill_tree(pid, timeout, |_pid, process| Ok(process.start_time()))
541}
542
543pub fn exit_code(status: std::process::ExitStatus) -> i32 {
544    use std::os::unix::process::ExitStatusExt;
545    status.code().unwrap_or_else(|| -status.signal().unwrap_or(1))
546}
547
548/// The signal that terminated `status`'s process, if it died from one.
549pub fn exit_signal(status: &std::process::ExitStatus) -> Option<i32> {
550    use std::os::unix::process::ExitStatusExt;
551    status.signal()
552}
553
554pub fn set_process_name(name: &str) {
555    let truncated: String = name.chars().take(15).collect();
556    let c_name = std::ffi::CString::new(truncated).unwrap_or_default();
557    unsafe { libc::prctl(libc::PR_SET_NAME, c_name.as_ptr() as libc::c_ulong, 0, 0, 0); }
558}
559
560pub fn configure_trampoline_command(_command: &mut std::process::Command) {}
561
562pub fn configure_process_command(
563    command: &mut std::process::Command,
564    config: crate::platform::process::ProcessCommandConfig,
565) -> io::Result<()> {
566    configure_process_command_inner(command, config, false)
567}
568
569/// Root-facade-only launch seam for bounded owner-death containment.
570///
571/// This must be `pub` because `running-process` is a separate package, but
572/// applications should use its semantic bounded-run options instead of this
573/// implementation-detail function.
574#[doc(hidden)]
575pub fn configure_process_command_for_bounded_owner_death(
576    command: &mut std::process::Command,
577    config: crate::platform::process::ProcessCommandConfig,
578) -> io::Result<()> {
579    configure_process_command_inner(command, config, true)
580}
581
582fn configure_process_command_inner(
583    command: &mut std::process::Command,
584    config: crate::platform::process::ProcessCommandConfig,
585    kill_when_owner_dies: bool,
586) -> io::Result<()> {
587    let create_process_group = config.create_process_group;
588    let nice = config.nice;
589    let address_space_limit_bytes = config.address_space_limit_bytes;
590    if !(create_process_group
591        || nice.is_some()
592        || address_space_limit_bytes.is_some()
593        || kill_when_owner_dies)
594    {
595        return Ok(());
596    }
597    let owner_pid = if kill_when_owner_dies {
598        // Read before `Command` forks so the child can detect the narrow race
599        // where this process exits between fork and PR_SET_PDEATHSIG.
600        unsafe { libc::getpid() }
601    } else {
602        0
603    };
604    use std::os::unix::process::CommandExt;
605    unsafe {
606        command.pre_exec(move || {
607            if create_process_group && libc::setpgid(0, 0) == -1 {
608                return Err(io::Error::last_os_error());
609            }
610            if let Some(nice) = nice {
611                if libc::setpriority(libc::PRIO_PROCESS, 0, nice) == -1 {
612                    return Err(io::Error::last_os_error());
613                }
614            }
615            if let Some(limit) = address_space_limit_bytes {
616                let rlim = libc::rlimit { rlim_cur: limit, rlim_max: limit };
617                if libc::setrlimit(libc::RLIMIT_AS, &rlim) == -1 {
618                    return Err(io::Error::last_os_error());
619                }
620            }
621            if kill_when_owner_dies {
622                install_parent_death_signal_with_race_guard(owner_pid)?;
623            }
624            Ok(())
625        });
626    }
627    Ok(())
628}
629
630/// Install PDEATHSIG and close the fork-to-prctl owner-death race.
631///
632/// This runs only in `Command::pre_exec`, after fork and before exec.
633fn install_parent_death_signal_with_race_guard(owner_pid: libc::pid_t) -> io::Result<()> {
634    if unsafe {
635        libc::prctl(
636            libc::PR_SET_PDEATHSIG,
637            libc::SIGTERM as libc::c_ulong,
638            0,
639            0,
640            0,
641        )
642    } == -1
643    {
644        return Err(io::Error::last_os_error());
645    }
646    if unsafe { libc::getppid() } != owner_pid {
647        // The owner died after fork but before PDEATHSIG was armed. A
648        // caller-provided pre_exec hook may have ignored SIGTERM, so sending
649        // that signal then returning would let this orphan exec. SAFETY:
650        // `_exit` is async-signal-safe and bypasses Rust destructors and
651        // allocation in this post-fork child.
652        unsafe { libc::_exit(128 + libc::SIGTERM) };
653    }
654    Ok(())
655}
656
657pub fn trampoline_exit_code(status: std::process::ExitStatus) -> i32 {
658    use std::os::unix::process::ExitStatusExt;
659    status.signal().map_or_else(|| status.code().unwrap_or(1), |signal| 128 + signal)
660}
661
662/// Return the GNU build ID of the running executable without reading the
663/// executable from disk.
664///
665/// The dynamic loader has already mapped the main image's `PT_NOTE` segment,
666/// so callers that only need an image-generation identity do not need to hash
667/// a potentially large unoptimized binary. `None` preserves a clean fallback
668/// for binaries linked without a GNU build ID.
669pub fn current_executable_build_id() -> Option<Vec<u8>> {
670    unsafe extern "C" fn visit(
671        info: *mut libc::dl_phdr_info,
672        _size: libc::size_t,
673        output: *mut libc::c_void,
674    ) -> libc::c_int {
675        const MAX_NOTE_BYTES: usize = 1024 * 1024;
676
677        let info = unsafe { &*info };
678        let is_main_executable = info.dlpi_name.is_null()
679            || unsafe { std::ffi::CStr::from_ptr(info.dlpi_name) }
680                .to_bytes()
681                .is_empty();
682        if !is_main_executable || info.dlpi_phdr.is_null() || info.dlpi_phnum == 0 {
683            return 0;
684        }
685        let headers = unsafe {
686            std::slice::from_raw_parts(info.dlpi_phdr, usize::from(info.dlpi_phnum))
687        };
688        #[allow(clippy::unnecessary_cast)]
689        let load_bias = info.dlpi_addr as u64;
690        for header in headers {
691            if header.p_type != libc::PT_NOTE {
692                continue;
693            }
694            let Ok(length) = usize::try_from(header.p_memsz) else {
695                continue;
696            };
697            if length == 0 || length > MAX_NOTE_BYTES {
698                continue;
699            }
700            let Some(address) = load_bias.checked_add(header.p_vaddr) else {
701                continue;
702            };
703            let Some(note_end) = address.checked_add(length as u64) else {
704                continue;
705            };
706            let mapped_read_only = headers.iter().any(|load| {
707                if load.p_type != libc::PT_LOAD || load.p_flags & libc::PF_R == 0 {
708                    return false;
709                }
710                let Some(start) = load_bias.checked_add(load.p_vaddr) else {
711                    return false;
712                };
713                let Some(end) = start.checked_add(load.p_memsz) else {
714                    return false;
715                };
716                address >= start && note_end <= end
717            });
718            if address == 0 || !mapped_read_only {
719                continue;
720            }
721            let notes = unsafe { std::slice::from_raw_parts(address as *const u8, length) };
722            if let Some(build_id) = gnu_build_id_from_notes(notes) {
723                let output = unsafe { &mut *output.cast::<Option<Vec<u8>>>() };
724                *output = Some(build_id.to_vec());
725                return 1;
726            }
727        }
728        0
729    }
730
731    let mut output = None;
732    unsafe {
733        libc::dl_iterate_phdr(
734            Some(visit),
735            (&mut output as *mut Option<Vec<u8>>).cast::<libc::c_void>(),
736        );
737    }
738    output
739}
740
741fn gnu_build_id_from_notes(mut notes: &[u8]) -> Option<&[u8]> {
742    fn aligned(value: usize) -> Option<usize> {
743        value.checked_add(3).map(|value| value & !3)
744    }
745
746    while notes.len() >= 12 {
747        let name_len = usize::try_from(u32::from_ne_bytes(notes[0..4].try_into().ok()?)).ok()?;
748        let desc_len = usize::try_from(u32::from_ne_bytes(notes[4..8].try_into().ok()?)).ok()?;
749        let kind = u32::from_ne_bytes(notes[8..12].try_into().ok()?);
750        let name_end = 12usize.checked_add(name_len)?;
751        let desc_start = 12usize.checked_add(aligned(name_len)?)?;
752        let desc_end = desc_start.checked_add(desc_len)?;
753        let next = desc_start.checked_add(aligned(desc_len)?)?;
754        if next > notes.len() || name_end > notes.len() || desc_end > notes.len() {
755            return None;
756        }
757        if kind == 3 && notes.get(12..name_end)?.starts_with(b"GNU") && desc_len > 0 {
758            return notes.get(desc_start..desc_end);
759        }
760        notes = &notes[next..];
761    }
762    None
763}
764
765/// Request a graceful shutdown for a child-owned POSIX process group.
766pub fn soft_terminate_process_group(pid: u32) -> io::Result<()> {
767    // SAFETY: `kill` receives only the numeric child-owned group id; no Rust
768    // references or borrowed state cross the OS boundary.
769    let result = unsafe { libc::kill(-(pid as i32), libc::SIGTERM) };
770    if result != 0 {
771        let error = io::Error::last_os_error();
772        if error.raw_os_error() != Some(libc::ESRCH) {
773            return Err(error);
774        }
775    }
776    Ok(())
777}
778
779pub fn process_snapshot() -> Vec<crate::platform::process::ProcessSnapshot> {
780    Vec::new()
781}
782
783pub fn process_snapshot_for_pid(_pid: u32) -> Option<crate::platform::process::ProcessSnapshot> {
784    None
785}
786
787/// Mark inherited descriptors close-on-exec without breaking std's exec-error pipe.
788///
789/// # Safety
790/// This must only be called from a post-fork `pre_exec` closure.
791pub unsafe fn unix_mark_extra_fds_close_on_exec() {
792    #[cfg(any(target_arch = "x86_64", target_arch = "aarch64", target_arch = "x86", target_arch = "arm", target_arch = "riscv64", target_arch = "powerpc64"))]
793    {
794        const SYS_CLOSE_RANGE: libc::c_long = 436;
795        const CLOSE_RANGE_CLOEXEC: libc::c_uint = 4;
796        if libc::syscall(SYS_CLOSE_RANGE, 3u32, libc::c_uint::MAX, CLOSE_RANGE_CLOEXEC) == 0 {
797            return;
798        }
799    }
800    mark_fds_from_directory_or_range();
801}
802
803pub fn configure_sync_daemon_command(command: &mut std::process::Command) -> io::Result<()> {
804    configure_sync_daemon_command_inner(command, None)
805}
806
807pub fn configure_sync_daemon_command_with_inheritance(
808    command: &mut std::process::Command,
809    inheritance: crate::platform::process::DaemonExecInheritance,
810) -> io::Result<()> {
811    configure_sync_daemon_command_inner(command, Some(inheritance))
812}
813
814fn configure_sync_daemon_command_inner(
815    command: &mut std::process::Command,
816    inheritance: Option<crate::platform::process::DaemonExecInheritance>,
817) -> io::Result<()> {
818    use std::os::unix::process::CommandExt;
819    unsafe {
820        command.pre_exec(move || {
821            let _ = libc::setsid();
822            unix_mark_extra_fds_close_on_exec();
823            if let Some(inheritance) = inheritance {
824                clear_cloexec_after_sweep(inheritance.descriptor())?;
825            }
826            Ok(())
827        });
828    }
829    Ok(())
830}
831
832unsafe fn clear_cloexec_after_sweep(fd: libc::c_int) -> io::Result<()> {
833    let flags = libc::fcntl(fd, libc::F_GETFD);
834    if flags == -1 {
835        return Err(io::Error::last_os_error());
836    }
837    if libc::fcntl(fd, libc::F_SETFD, flags & !libc::FD_CLOEXEC) == -1 {
838        return Err(io::Error::last_os_error());
839    }
840    Ok(())
841}
842
843pub fn configure_sync_contained_command(command: &mut std::process::Command) -> io::Result<()> {
844    use std::os::unix::process::CommandExt;
845    let owner_pid = std::process::id() as libc::pid_t;
846    unsafe {
847        command.pre_exec(move || {
848            if libc::setpgid(0, 0) == -1 { return Err(io::Error::last_os_error()); }
849            if libc::prctl(libc::PR_SET_PDEATHSIG, libc::SIGKILL) == -1 {
850                return Err(io::Error::last_os_error());
851            }
852            // PID 1 may be the legitimate owner in a container. Compare the
853            // captured parent identity, not the orphan-reparenting convention.
854            if libc::getppid() != owner_pid { libc::_exit(1); }
855            unix_mark_extra_fds_close_on_exec();
856            Ok(())
857        });
858    }
859    Ok(())
860}
861
862pub fn parent_has_console() -> bool { false }
863
864pub fn sync_child_native_handle(_child: &std::process::Child) -> usize { 0 }
865
866unsafe fn mark_fds_from_directory_or_range() {
867    let dir = libc::opendir(c"/dev/fd".as_ptr());
868    if !dir.is_null() {
869        let dir_fd = libc::dirfd(dir);
870        loop {
871            let entry = libc::readdir(dir);
872            if entry.is_null() { break; }
873            let mut fd: libc::c_int = 0;
874            let mut cursor = (*entry).d_name.as_ptr();
875            let mut numeric = false;
876            while *cursor != 0 {
877                let byte = *cursor as u8;
878                if !byte.is_ascii_digit() { numeric = false; break; }
879                fd = fd * 10 + (byte - b'0') as libc::c_int;
880                cursor = cursor.add(1);
881                numeric = true;
882            }
883            if numeric && fd > 2 && fd != dir_fd { set_cloexec(fd); }
884        }
885        libc::closedir(dir);
886        return;
887    }
888    let maximum = libc::sysconf(libc::_SC_OPEN_MAX);
889    for fd in 3..if maximum < 0 { 4096 } else { maximum as libc::c_int } { set_cloexec(fd); }
890}
891
892unsafe fn set_cloexec(fd: libc::c_int) {
893    let flags = libc::fcntl(fd, libc::F_GETFD);
894    if flags != -1 { libc::fcntl(fd, libc::F_SETFD, flags | libc::FD_CLOEXEC); }
895}
896pub fn observer_backend(scope: crate::platform::process::ObserverScope, category: crate::platform::process::ObserverCategory) -> crate::platform::process::ObserverBackend {
897    use crate::platform::process::{ObserverBackend as B, ObserverCategory as C, ObserverScope as S, ObserverSupport as P};
898    match (scope, category) {
899        (S::SystemWide, C::File) => B { support:P::Unavailable, backend:"seccomp-user-notify", reason:"Phase 3: Linux seccomp user-notify file backend not yet implemented" },
900        (S::SystemWide, C::Network) => B { support:P::Unavailable, backend:"ebpf", reason:"Phase 3: Linux eBPF network backend not yet implemented" },
901        (S::SystemWide, C::Process) => B { support:P::Unavailable, backend:"seccomp-user-notify", reason:"Phase 3: Linux seccomp user-notify process backend not yet implemented" },
902        (S::LaunchedProcessTree, C::File) => B { support:P::Partial, backend:"proc-fd-snapshot", reason:"Linux /proc/<pid>/fd/* snapshot via read_process_file_handles (#539 slice 6 follow-up; no streaming file events)" },
903        (S::LaunchedProcessTree, C::Network) => B { support:P::Unavailable, backend:"none", reason:"#539: no-admin per-child network backend deferred to a follow-up issue" },
904        (S::LaunchedProcessTree, C::Process) => B { support:P::Supported, backend:"subreaper-proc-poll", reason:"Linux PR_SET_CHILD_SUBREAPER + /proc descendant polling (#539 slice 5)" },
905    }
906}
907
908pub fn unix_set_priority(pid: u32, nice: i32) -> io::Result<()> {
909    if unsafe { libc::setpriority(libc::PRIO_PROCESS, pid, nice) } == -1 { Err(io::Error::last_os_error()) } else { Ok(()) }
910}
911pub fn unix_signal_process(pid: u32, signal: crate::platform::process::UnixSignalKind) -> io::Result<()> {
912    if unsafe { libc::kill(pid as i32, unix_signal_raw(signal)) } == -1 { Err(io::Error::last_os_error()) } else { Ok(()) }
913}
914pub(crate) fn observe_owned_child_exit(pid: i32) -> io::Result<Option<i32>> {
915    // SAFETY: siginfo_t is a C output record; zero initializes the no-event PID.
916    let mut info: libc::siginfo_t = unsafe { std::mem::zeroed() };
917    // SAFETY: info is writable and valid for this call. P_PID selects exactly
918    // the owned child; WNOWAIT does not consume its identity or exit status.
919    let result = unsafe {
920        libc::waitid(
921            libc::P_PID,
922            pid as libc::id_t,
923            &mut info,
924            libc::WEXITED | libc::WNOHANG | libc::WNOWAIT,
925        )
926    };
927    if result != 0 {
928        return Err(io::Error::last_os_error());
929    }
930    // SAFETY: successful waitid with WEXITED initializes the child-status fields.
931    if unsafe { info.si_pid() } == 0 {
932        return Ok(None);
933    }
934    // SAFETY: a nonzero child PID identifies the initialized exit-status union.
935    let status = unsafe { info.si_status() };
936    Ok(Some(if info.si_code == libc::CLD_EXITED { status } else { 128 + status }))
937}
938
939/// Apply a process priority expressed as a Unix nice value.
940pub fn apply_process_priority(pid: u32, nice: i32) -> io::Result<()> {
941    unix_set_priority(pid, nice)
942}
943
944/// Deliver the host's interactive-interrupt request to `pid`.
945///
946/// Unix sends SIGINT to the process, or to its group when the child leads one.
947/// `creationflags` only matters on Windows and is ignored here.
948pub fn send_interrupt(pid: u32, _creationflags: Option<u32>, create_process_group: bool) -> io::Result<()> {
949    use crate::platform::process::UnixSignalKind;
950    if create_process_group {
951        unix_signal_process_group(pid as i32, UnixSignalKind::Interrupt)
952    } else {
953        unix_signal_process(pid, UnixSignalKind::Interrupt)
954    }
955}
956pub fn unix_signal_process_group(pid: i32, signal: crate::platform::process::UnixSignalKind) -> io::Result<()> {
957    if unsafe { libc::killpg(pid, unix_signal_raw(signal)) } == -1 { Err(io::Error::last_os_error()) } else { Ok(()) }
958}
959pub fn unix_signal_raw(signal: crate::platform::process::UnixSignalKind) -> i32 {
960    match signal { crate::platform::process::UnixSignalKind::Interrupt => libc::SIGINT, crate::platform::process::UnixSignalKind::Terminate => libc::SIGTERM, crate::platform::process::UnixSignalKind::Kill => libc::SIGKILL }
961}
962
963#[cfg(feature = "async-process")]
964pub fn configure_compat_tokio_command(
965    command: &mut Command,
966    _show_console: bool,
967    kill_when_owner_dies: bool,
968) -> io::Result<()> {
969    configure_command(command, false, kill_when_owner_dies, None)
970}
971
972/// Nothing to do on this host: the parent-death signal is installed in `pre_exec`, before the
973/// child ever runs, so nothing remains to do once it has.
974#[cfg(feature = "async-process")]
975pub fn after_compat_tokio_spawn(
976    _child: &Child,
977    _kill_when_owner_dies: bool,
978) -> io::Result<()> {
979    Ok(())
980}
981
982/// Configure a caller-built command for [`crate::SpawnSpec::from_std_command`].
983///
984/// This is the `NativeProcess` launch mapping (`ProcessCommandConfig`), not
985/// the declarative `SpawnSpec` one, so a command handed over by the sync
986/// engine is configured exactly once and by the same code it uses today.
987#[cfg(feature = "async-process")]
988pub(crate) fn configure_override_command(
989    command: &mut std::process::Command,
990    config: crate::platform::process::ProcessCommandConfig,
991    kill_when_owner_dies: bool,
992) -> io::Result<()> {
993    if kill_when_owner_dies {
994        configure_process_command_for_bounded_owner_death(command, config)
995    } else {
996        configure_process_command(command, config)
997    }
998}
999
1000#[cfg(feature = "async-process")]
1001pub(crate) fn configure_command(
1002    command: &mut Command,
1003    create_process_group: bool,
1004    kill_when_owner_dies: bool,
1005    nice: Option<i32>,
1006) -> io::Result<()> {
1007    if create_process_group {
1008        command.process_group(0);
1009    }
1010    if kill_when_owner_dies {
1011        let owner_pid = unsafe { libc::getpid() };
1012        // SAFETY: the closure invokes only async-signal-safe libc calls.
1013        unsafe {
1014            command.pre_exec(move || {
1015                if let Some(nice) = nice {
1016                    if libc::setpriority(libc::PRIO_PROCESS, 0, nice) == -1 {
1017                        return Err(io::Error::last_os_error());
1018                    }
1019                }
1020                install_parent_death_signal_with_race_guard(owner_pid)?;
1021                Ok(())
1022            });
1023        }
1024    }
1025    Ok(())
1026}
1027
1028/// Niceness that must be applied to the child right after spawn.
1029///
1030/// A `pre_exec` hook makes std abandon `posix_spawn` for `fork` + `exec`,
1031/// whose cost grows with the parent's resident size (~0.5-1 ms per child in
1032/// a large daemon, #1248). Owner-death has to run in the child, so it keeps
1033/// the hook and applies niceness there. Niceness alone does not: it is set
1034/// on the child's pid once `spawn` has returned, before this call returns to
1035/// the caller. The only window is the child's first instructions, and a
1036/// thread or grandchild it starts inside that window inherits the old value.
1037#[cfg(feature = "async-process")]
1038fn nice_after_spawn(kill_when_owner_dies: bool, nice: Option<i32>) -> Option<i32> {
1039    if kill_when_owner_dies {
1040        None
1041    } else {
1042        nice
1043    }
1044}
1045
1046#[cfg(feature = "async-process")]
1047pub(crate) fn after_spawn(
1048    child: &Child,
1049    kill_when_owner_dies: bool,
1050    nice: Option<i32>,
1051) -> io::Result<()> {
1052    let Some(nice) = nice_after_spawn(kill_when_owner_dies, nice) else {
1053        return Ok(());
1054    };
1055    let Some(pid) = child.id() else {
1056        return Ok(());
1057    };
1058    // SAFETY: plain syscall on a pid this process just spawned and has not
1059    // yet reaped, so the pid cannot have been recycled.
1060    if unsafe { libc::setpriority(libc::PRIO_PROCESS, pid as libc::id_t, nice) } == -1 {
1061        let error = io::Error::last_os_error();
1062        // The caller asked for this priority and the child is already running:
1063        // do not leave it running at the wrong one. The caller drops `child`,
1064        // which reaps it.
1065        unsafe { libc::kill(pid as libc::pid_t, libc::SIGKILL) };
1066        return Err(error);
1067    }
1068    Ok(())
1069}
1070
1071/// Launch-bound identity for private async controls.
1072///
1073/// `pidfd_open` supplies the race-free direct-control capability where the
1074/// kernel permits it. `/proc/<pid>/stat` start ticks remain available for CPU
1075/// accounting on older or restricted hosts, but are never a raw-PID control
1076/// fallback.
1077#[cfg(feature = "async-process")]
1078pub(crate) struct AsyncChildIdentity {
1079    pid: u32,
1080    start_ticks: u64,
1081    pidfd: Option<std::os::fd::OwnedFd>,
1082}
1083
1084#[cfg(feature = "async-process")]
1085pub(crate) fn async_child_identity(child: &Child) -> Option<AsyncChildIdentity> {
1086    let pid = child.id()?;
1087    let (start_ticks, _, _) = proc_stat(pid).ok()?;
1088    let fd = unsafe { libc::syscall(libc::SYS_pidfd_open, pid as libc::c_int, 0) } as libc::c_int;
1089    let pidfd = (fd >= 0).then(|| {
1090        // SAFETY: pidfd_open returned a newly owned descriptor above.
1091        unsafe { <std::os::fd::OwnedFd as std::os::fd::FromRawFd>::from_raw_fd(fd) }
1092    });
1093    Some(AsyncChildIdentity {
1094        pid,
1095        start_ticks,
1096        pidfd,
1097    })
1098}
1099
1100#[cfg(feature = "async-process")]
1101pub(crate) fn signal_async_child(identity: &AsyncChildIdentity) -> io::Result<()> {
1102    if identity_matches(identity) {
1103        pidfd_send_signal(identity, libc::SIGKILL)
1104    } else {
1105        Err(io::Error::new(
1106            io::ErrorKind::BrokenPipe,
1107            "child process launch identity no longer matches",
1108        ))
1109    }
1110}
1111
1112#[cfg(feature = "async-process")]
1113pub(crate) fn signal_async_child_group(identity: &AsyncChildIdentity) -> io::Result<()> {
1114    if !identity_matches(identity) || !pidfd_is_live(identity)? {
1115        return Err(io::Error::new(
1116            io::ErrorKind::BrokenPipe,
1117            "child process launch identity no longer matches",
1118        ));
1119    }
1120    if unsafe { libc::kill(-(identity.pid as i32), libc::SIGTERM) } == 0 {
1121        Ok(())
1122    } else {
1123        Err(io::Error::last_os_error())
1124    }
1125}
1126
1127#[cfg(feature = "async-process")]
1128pub(crate) fn async_child_cpu_time(
1129    identity: &AsyncChildIdentity,
1130) -> io::Result<Option<std::time::Duration>> {
1131    let Ok((start_ticks, user_ticks, system_ticks)) = proc_stat(identity.pid) else {
1132        return Ok(None);
1133    };
1134    if start_ticks != identity.start_ticks {
1135        return Ok(None);
1136    }
1137    let ticks_per_second = unsafe { libc::sysconf(libc::_SC_CLK_TCK) };
1138    if ticks_per_second <= 0 {
1139        return Ok(None);
1140    }
1141    let ticks = user_ticks.saturating_add(system_ticks);
1142    let hz = ticks_per_second as u64;
1143    Ok(Some(
1144        std::time::Duration::from_secs(ticks / hz)
1145            + std::time::Duration::from_nanos(
1146                ticks
1147                    % hz
1148                    .saturating_mul(1_000_000_000)
1149                    / hz,
1150            ),
1151    ))
1152}
1153
1154#[cfg(feature = "async-process")]
1155fn identity_matches(identity: &AsyncChildIdentity) -> bool {
1156    matches!(proc_stat(identity.pid), Ok((start_ticks, _, _)) if start_ticks == identity.start_ticks)
1157}
1158
1159#[cfg(feature = "async-process")]
1160fn pidfd_is_live(identity: &AsyncChildIdentity) -> io::Result<bool> {
1161    let Some(pidfd) = identity.pidfd.as_ref() else {
1162        return Err(io::Error::new(
1163            io::ErrorKind::Unsupported,
1164            "pidfd control is unavailable for this child",
1165        ));
1166    };
1167    let result = unsafe {
1168        libc::syscall(
1169            libc::SYS_pidfd_send_signal,
1170            std::os::fd::AsRawFd::as_raw_fd(pidfd),
1171            0,
1172            std::ptr::null::<libc::siginfo_t>(),
1173            0,
1174        )
1175    };
1176    if result == 0 {
1177        return Ok(true);
1178    }
1179    let error = io::Error::last_os_error();
1180    if error.raw_os_error() == Some(libc::ESRCH) {
1181        Ok(false)
1182    } else {
1183        Err(error)
1184    }
1185}
1186
1187#[cfg(feature = "async-process")]
1188fn pidfd_send_signal(identity: &AsyncChildIdentity, signal: libc::c_int) -> io::Result<()> {
1189    let Some(pidfd) = identity.pidfd.as_ref() else {
1190        return Err(io::Error::new(
1191            io::ErrorKind::Unsupported,
1192            "pidfd control is unavailable for this child",
1193        ));
1194    };
1195    let result = unsafe {
1196        libc::syscall(
1197            libc::SYS_pidfd_send_signal,
1198            std::os::fd::AsRawFd::as_raw_fd(pidfd),
1199            signal,
1200            std::ptr::null::<libc::siginfo_t>(),
1201            0,
1202        )
1203    };
1204    if result == 0 {
1205        return Ok(());
1206    }
1207    let error = io::Error::last_os_error();
1208    if error.raw_os_error() == Some(libc::ESRCH) {
1209        Ok(())
1210    } else {
1211        Err(error)
1212    }
1213}
1214
1215#[cfg(feature = "async-process")]
1216fn proc_stat(pid: u32) -> io::Result<(u64, u64, u64)> {
1217    let stat = std::fs::read_to_string(format!("/proc/{pid}/stat"))?;
1218    let fields = stat
1219        .rsplit_once(')')
1220        .map(|(_, fields)| fields.split_ascii_whitespace().collect::<Vec<_>>())
1221        .ok_or_else(|| io::Error::new(io::ErrorKind::InvalidData, "malformed /proc stat"))?;
1222    let parse = |index: usize| -> io::Result<u64> {
1223        fields
1224            .get(index)
1225            .ok_or_else(|| io::Error::new(io::ErrorKind::InvalidData, "short /proc stat"))?
1226            .parse::<u64>()
1227            .map_err(|_| io::Error::new(io::ErrorKind::InvalidData, "invalid /proc stat"))
1228    };
1229    // Fields after the closing command name begin at stat field 3: utime=11,
1230    // stime=12, starttime=19 in this zero-based tail.
1231    Ok((parse(19)?, parse(11)?, parse(12)?))
1232}
1233
1234#[cfg(feature = "async-process")]
1235pub(crate) fn shell_spec(command: &OsStr) -> SpawnSpec {
1236    SpawnSpec::new("/bin/sh").arg("-c").arg(command)
1237}
1238
1239#[cfg(test)]
1240mod tests {
1241    #[test]
1242    fn exit_signal_reports_the_signal_that_killed_the_child() {
1243        use std::os::unix::process::ExitStatusExt;
1244        // A raw wait status whose low bits carry the signal: killed by SIGKILL.
1245        let killed = std::process::ExitStatus::from_raw(libc::SIGKILL);
1246        assert_eq!(super::exit_signal(&killed), Some(libc::SIGKILL));
1247        // Exit code 3 in the high byte: a normal exit, no signal.
1248        let exited = std::process::ExitStatus::from_raw(3 << 8);
1249        assert_eq!(super::exit_signal(&exited), None);
1250    }
1251
1252    #[test]
1253    fn linux_always_reports_a_cgroup_membership() {
1254        assert!(super::host_process_cgroup().is_some());
1255    }
1256
1257    #[cfg(feature = "async-process")]
1258    #[test]
1259    fn async_identity_mismatch_fails_closed_without_pid_signal() {
1260        let pid = unsafe { libc::getpid() as u32 };
1261        let (start_ticks, _, _) = super::proc_stat(pid).expect("read this process start key");
1262        let identity = super::AsyncChildIdentity {
1263            pid,
1264            start_ticks: start_ticks.saturating_add(1),
1265            pidfd: None,
1266        };
1267        assert!(!super::identity_matches(&identity));
1268        let error = super::signal_async_child(&identity)
1269            .expect_err("mismatched launch identity must not signal a reused PID");
1270        assert_eq!(error.kind(), std::io::ErrorKind::BrokenPipe);
1271        assert_eq!(super::async_child_cpu_time(&identity).unwrap(), None);
1272    }
1273
1274    #[cfg(feature = "async-process")]
1275    #[test]
1276    fn async_identity_without_pidfd_keeps_cpu_but_refuses_pid_control() {
1277        let pid = unsafe { libc::getpid() as u32 };
1278        let (start_ticks, _, _) = super::proc_stat(pid).expect("read this process start key");
1279        let identity = super::AsyncChildIdentity {
1280            pid,
1281            start_ticks,
1282            pidfd: None,
1283        };
1284        assert!(super::async_child_cpu_time(&identity).unwrap().is_some());
1285        let error = super::signal_async_child(&identity).expect_err("no raw-PID kill fallback");
1286        assert_eq!(error.kind(), std::io::ErrorKind::Unsupported);
1287    }
1288
1289    #[test]
1290    fn owner_death_race_guard_exits_when_sigterm_is_ignored() {
1291        let child = unsafe { libc::fork() };
1292        assert!(child >= 0, "fork owner-death race fixture");
1293        if child == 0 {
1294            // Model a caller-supplied pre_exec hook which ignored SIGTERM
1295            // before the bounded runner's hook is appended.
1296            if unsafe { libc::signal(libc::SIGTERM, libc::SIG_IGN) } == libc::SIG_ERR {
1297                unsafe { libc::_exit(98) };
1298            }
1299            let owner_pid = unsafe { libc::getppid() }.saturating_add(1);
1300            // A deliberately mismatched owner PID models the post-prctl
1301            // parent-death race. The helper must not return even though the
1302            // SIGTERM disposition above would ignore a signal-based guard.
1303            if super::install_parent_death_signal_with_race_guard(owner_pid).is_err() {
1304                unsafe { libc::_exit(99) };
1305            }
1306            unsafe { libc::_exit(100) };
1307        }
1308
1309        let mut status = 0;
1310        assert_eq!(unsafe { libc::waitpid(child, &mut status, 0) }, child);
1311        assert!(libc::WIFEXITED(status), "race fixture must _exit");
1312        assert_eq!(
1313            libc::WEXITSTATUS(status),
1314            128 + libc::SIGTERM,
1315            "ignored SIGTERM must not permit the owner-dead child to continue"
1316        );
1317    }
1318
1319    #[test]
1320    fn shell_command_preserves_login_shell_contract_and_ignores_child_path() {
1321        use std::ffi::OsStr;
1322
1323        let command_text = "printf '%s' 'alpha beta;\"gamma\"'";
1324        let mut command = super::shell_command(command_text);
1325        assert_eq!(command.get_program(), OsStr::new("/bin/sh"));
1326        assert_eq!(
1327            command.get_args().collect::<Vec<_>>(),
1328            [OsStr::new("-lc"), OsStr::new(command_text)]
1329        );
1330        command
1331            .env_clear()
1332            .env("PATH", "/caller-supplied-path-override");
1333        let output = command
1334            .output()
1335            .expect("absolute shell command should execute independently of child PATH");
1336        assert!(output.status.success());
1337        assert_eq!(output.stdout, b"alpha beta;\"gamma\"");
1338    }
1339
1340    #[test]
1341    #[cfg(not(target_env = "musl"))]
1342    fn current_executable_exposes_a_gnu_build_id() {
1343        let build_id = super::current_executable_build_id()
1344            .expect("Linux test executable should carry a GNU build ID");
1345        assert!(!build_id.is_empty());
1346    }
1347}
1348#[cfg(test)]
1349#[path = "tests/platform_linux_coverage.rs"]
1350mod coverage_tests;
1351#[path = "sync_spawn_group.rs"]
1352mod sync_spawn;
1353pub use sync_spawn::{spawn_sync, spawn_sync_daemon, spawn_sync_daemon_with_inheritance};
1354#[cfg(feature = "independent-spawn")]
1355pub(crate) use sync_spawn::spawn_sync_owned_daemon;
1356
1357#[cfg(all(test, feature = "ipc"))]
1358mod endpoint_naming_tests {
1359    use super::{ipc_broker_v1_endpoint_path, ipc_endpoint_name_limit, LINUX_SUN_PATH_MAX};
1360
1361    #[test]
1362    fn the_v1_address_keeps_the_full_name_for_debuggability() {
1363        let address = ipc_broker_v1_endpoint_path("rpb-v1-abc-shared").expect("derive address");
1364        assert!(address.contains("rpb-v1-abc-shared"));
1365        assert!(address.ends_with("-shared.sock"));
1366        assert!(address.contains("/broker/"));
1367    }
1368
1369    #[test]
1370    fn an_over_long_name_is_refused_against_sun_path() {
1371        let err = ipc_broker_v1_endpoint_path(&"a".repeat(LINUX_SUN_PATH_MAX))
1372            .expect_err("must exceed sun_path");
1373        assert_eq!(err.max, LINUX_SUN_PATH_MAX - 1);
1374        assert_eq!(err.limit_label, "Linux sun_path");
1375    }
1376
1377    #[test]
1378    fn an_accepted_address_is_strictly_shorter_than_the_field() {
1379        // sockaddr_un is NUL-terminated, so equality with the field width
1380        // would truncate the terminator.
1381        let address = ipc_broker_v1_endpoint_path("rpb-v1-abc-shared").expect("derive address");
1382        assert!(address.len() < LINUX_SUN_PATH_MAX);
1383    }
1384
1385    #[test]
1386    fn the_reported_budget_is_sun_path() {
1387        let limit = ipc_endpoint_name_limit();
1388        assert_eq!(limit.max_bytes, LINUX_SUN_PATH_MAX);
1389        assert_eq!(limit.label, "Linux sun_path");
1390    }
1391
1392    #[test]
1393    fn the_scope_spelling_is_the_verbatim_path_bytes() {
1394        // Paths are opaque byte strings here: no spelling difference is
1395        // meaningless, and case is significant. This pins the spelling --
1396        // changing it re-scopes every deployed broker, and the stability
1397        // tests upstream would not notice.
1398        use super::ipc_endpoint_scope_bytes;
1399
1400        let bytes = ipc_endpoint_scope_bytes(std::path::Path::new("/usr/local/bin/Broker"));
1401        assert_eq!(bytes, b"/usr/local/bin/Broker".to_vec());
1402
1403        let lowered = ipc_endpoint_scope_bytes(std::path::Path::new("/usr/local/bin/broker"));
1404        assert_ne!(bytes, lowered, "case must remain significant");
1405    }
1406
1407}
1408
1409/// Replace this process's image with `command`.
1410///
1411/// Returns only on failure: on success `execve` has already replaced the
1412/// program and there is nothing left to return to. That is why the signature
1413/// yields `io::Error` rather than `io::Result<()>` -- an `Ok` would name a
1414/// state that cannot be observed.
1415pub fn process_replace_current_image(command: &mut std::process::Command) -> std::io::Error {
1416    use std::os::unix::process::CommandExt as _;
1417    command.exec()
1418}
1419
1420/// This host replaces a running image in place; see the facade for what that
1421/// means for a caller that cannot accept a successor instead.
1422pub const fn process_can_replace_current_image() -> bool {
1423    true
1424}
1425
1426#[cfg(feature = "async-process")]
1427pub(crate) async fn shutdown_output_reader<R>(reader: R, _pending: bool) -> std::io::Result<()> {
1428    // Tokio's Unix child pipes use readiness I/O, not detached blocking reads.
1429    drop(reader);
1430    Ok(())
1431}
1432
1433/// Pins the per-host answers that facade callers branch on, so a change to
1434/// either is a visible, reviewed edit rather than a silent behaviour change.
1435#[cfg(test)]
1436mod host_semantics_tests {
1437    const ABSENT_PID: u32 = 0x7fff_fffe;
1438
1439    #[test]
1440    fn priority_on_absent_pid_reports_the_os_error() {
1441        assert!(super::apply_process_priority(ABSENT_PID, 0).is_err());
1442    }
1443
1444    #[test]
1445    fn interrupt_ignores_creation_flags_and_reports_absent_pid() {
1446        // Unix has no CREATE_NEW_PROCESS_GROUP prerequisite: flags never
1447        // short-circuit the signal, so the OS answers for a missing pid.
1448        let error = super::send_interrupt(ABSENT_PID, None, false).unwrap_err();
1449        assert_ne!(error.kind(), std::io::ErrorKind::InvalidInput);
1450        assert!(super::send_interrupt(ABSENT_PID, Some(0), true).is_err());
1451    }
1452
1453    #[test]
1454    fn open_handles_block_removal_matches_this_host() {
1455        assert!(!super::fs_open_handles_block_removal());
1456    }
1457
1458    #[cfg(feature = "ipc")]
1459    #[test]
1460    fn handoff_transport_is_available() {
1461        assert!(super::ipc_handoff_transport_available());
1462    }
1463}