Skip to main content

orbit_core/
shm.rs

1//! POSIX shared-memory helpers — V1 substrate for cross-process rings.
2//!
3//! Wraps `shm_open` / `ftruncate` / `mmap` / `munmap` / `shm_unlink`
4//! into a small, RAII-friendly API. Unix-only; Windows support is
5//! a separate concern (Win32 named file mapping) that can land later.
6//!
7//! ## Naming
8//!
9//! Segments are named `/orbit-{fleet}-{kind}-{uid}` — fleet name from
10//! the embedder, KIND from `OrbitTyped::KIND`, UID from `geteuid()`.
11//! UID-scoping avoids the `/dev/shm` sticky-bit cross-user collision
12//! problem (a stale segment owned by one user blocks another from
13//! `shm_unlink`-ing it on next boot).
14//!
15//! Rings that require a process-recoverable writer lock also open a
16//! companion `orbit-{fleet}-{kind}-{uid}.lock` file. It carries no
17//! ring data or state; it only supplies a regular-file inode for `flock`,
18//! because advisory locking on a POSIX SHM descriptor is not uniformly
19//! supported across the Unix targets Orbit serves. An unlocked stale
20//! companion file is safe to reuse.
21//!
22//! Those files live in a per-uid directory — `$XDG_RUNTIME_DIR/orbit-{uid}`
23//! where the session provides one, `/tmp/orbit-{uid}` otherwise — created
24//! `0700` and checked on every lock. They were once in `/tmp` directly, which
25//! made them squattable: see [`lock_dir`].
26//!
27//! ## Lifetime
28//!
29//! [`ShmRegion`] owns the mapped pointer and unmaps on drop. It does
30//! NOT `shm_unlink` on drop — the segment lives until an explicit
31//! [`ShmRegion::unlink`] call. This matches POSIX convention: a
32//! segment with mapped users is not removed; `shm_unlink` only
33//! prevents *new* opens, the current mapping stays valid until the
34//! last process unmaps.
35
36#![cfg(unix)]
37
38use std::ffi::CString;
39use std::fs::OpenOptions;
40use std::io;
41use std::os::fd::{AsRawFd, FromRawFd, OwnedFd};
42use std::os::unix::fs::OpenOptionsExt;
43use std::path::{Path, PathBuf};
44use std::ptr::NonNull;
45
46/// Namespace used by Orbit POSIX shared-memory objects.
47pub const SHM_NAMESPACE: &str = "orbit";
48
49/// Result of physically validating an existing POSIX SHM object.
50///
51/// This check is deliberately below ring semantics: it verifies that the
52/// named object can be opened and is large enough for the requested mapping,
53/// but it does not inspect an owning data structure's magic, version, or
54/// geometry header.
55#[derive(Clone, Copy, Debug, Eq, PartialEq)]
56pub enum ShmValidation {
57    /// No object currently exists under the requested name.
58    Missing,
59    /// The object exists and can safely back at least the requested mapping.
60    Valid { actual_size: usize },
61}
62
63/// A mapped POSIX SHM region. Drop unmaps; `unlink` removes the
64/// underlying name and any companion lock file (only the *creator*
65/// should call it on shutdown).
66pub struct ShmRegion {
67    name: CString,
68    lock_path: PathBuf,
69    /// Whether this region uses the companion file for ordered writes.
70    ///
71    /// The descriptor itself is deliberately not retained: every critical
72    /// section opens its own file description so a later fork does not inherit
73    /// an idle descriptor that can keep a future `flock` alive.
74    process_lock: bool,
75    ptr: NonNull<u8>,
76    len: usize,
77    /// True when this handle was the one that *created* the segment
78    /// (so it knows to `shm_unlink` if asked). Other attachers see
79    /// `false`.
80    created: bool,
81}
82
83impl ShmRegion {
84    /// Validate an existing shared-memory object without creating, mapping,
85    /// resetting, or unlinking it.
86    ///
87    /// Returns [`ShmValidation::Missing`] when the name does not exist. A
88    /// present object must be at least `minimum_size` bytes; larger objects
89    /// are accepted because some platforms report page-rounded SHM sizes.
90    /// The owning ring or table remains responsible for validating its own
91    /// persisted ABI header after mapping.
92    pub fn validate_existing(name: &str, minimum_size: usize) -> io::Result<ShmValidation> {
93        let cname = CString::new(name)
94            .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
95        let raw_fd = loop {
96            // SAFETY: passing a valid C string and well-known POSIX flags.
97            let fd =
98                unsafe { libc::shm_open(cname.as_ptr(), libc::O_RDONLY | libc::O_CLOEXEC, 0o600) };
99            if fd >= 0 {
100                break fd;
101            }
102            let error = io::Error::last_os_error();
103            if error.kind() == io::ErrorKind::Interrupted {
104                continue;
105            }
106            if error.raw_os_error() == Some(libc::ENOENT) {
107                return Ok(ShmValidation::Missing);
108            }
109            return Err(error);
110        };
111        // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
112        // owned by this scope.
113        let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
114        let actual_size = shm_object_size(&fd, name)?;
115        validate_minimum_size(name, actual_size, minimum_size)?;
116        Ok(ShmValidation::Valid { actual_size })
117    }
118
119    /// Map an existing shared-memory object read-only.
120    ///
121    /// This path never creates, sizes, locks, resets, or unlinks the object.
122    /// It is kept crate-private so callers receive a capability such as a
123    /// read-only ring view rather than a [`ShmRegion`] that also exposes
124    /// lifecycle and writable-pointer operations.
125    pub(crate) fn open_existing_read_only(name: &str, minimum_size: usize) -> io::Result<Self> {
126        let cname = CString::new(name)
127            .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
128        let raw_fd = loop {
129            // SAFETY: passing a valid C string and read-only POSIX flags.
130            let fd =
131                unsafe { libc::shm_open(cname.as_ptr(), libc::O_RDONLY | libc::O_CLOEXEC, 0o600) };
132            if fd >= 0 {
133                break fd;
134            }
135            let error = io::Error::last_os_error();
136            if error.kind() == io::ErrorKind::Interrupted {
137                continue;
138            }
139            return Err(error);
140        };
141        // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
142        // owned by this scope.
143        let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
144        let actual_size = shm_object_size(&fd, name)?;
145        validate_minimum_size(name, actual_size, minimum_size)?;
146
147        // Map the complete object so its persisted header can describe the
148        // geometry without the observer reproducing the producer's layout.
149        // SAFETY: fd is valid, actual_size is positive after minimum
150        // validation, and the mapping is read-only.
151        let ptr = unsafe {
152            libc::mmap(
153                std::ptr::null_mut(),
154                actual_size,
155                libc::PROT_READ,
156                libc::MAP_SHARED,
157                fd.as_raw_fd(),
158                0,
159            )
160        };
161        if ptr == libc::MAP_FAILED {
162            return Err(io::Error::last_os_error());
163        }
164        // SAFETY: mmap returned a non-null pointer (checked above).
165        let ptr = NonNull::new(ptr.cast::<u8>()).expect("mmap returned non-null on success");
166
167        Ok(Self {
168            lock_path: lock_file_path(name),
169            name: cname,
170            process_lock: false,
171            ptr,
172            len: actual_size,
173            created: false,
174        })
175    }
176
177    /// Open or create a shared-memory segment of `size` bytes,
178    /// memory-mapped read/write. Idempotent: if the segment already
179    /// exists with the same name and enough mapped bytes, it is reused
180    /// (`created = false`). First creation does `ftruncate(size)`;
181    /// later opens verify the existing object before mapping it. Some
182    /// platforms report a page-rounded SHM size, so a larger `st_size`
183    /// is valid; the owning data structure must verify its own header.
184    pub fn open_or_create(name: &str, size: usize) -> io::Result<Self> {
185        let (region, initialization_lock) = Self::open_or_create_inner(name, size, false)?;
186        debug_assert!(initialization_lock.is_none());
187        Ok(region)
188    }
189
190    /// Open or create a region while holding its process lock through caller
191    /// initialization. This prevents a peer from observing the interval
192    /// between `shm_open` and the owning data structure's initialized header.
193    pub fn open_or_create_locked(name: &str, size: usize) -> io::Result<(Self, ShmRegionLock)> {
194        let (region, initialization_lock) = Self::open_or_create_inner(name, size, true)?;
195        Ok((
196            region,
197            initialization_lock.expect("locked SHM open must return its initialization lock"),
198        ))
199    }
200
201    fn open_or_create_inner(
202        name: &str,
203        size: usize,
204        process_lock: bool,
205    ) -> io::Result<(Self, Option<ShmRegionLock>)> {
206        let cname = CString::new(name)
207            .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
208        let lock_path = lock_file_path(name);
209        let initialization_lock = if process_lock {
210            Some(lock_path_exclusive(&lock_path)?)
211        } else {
212            None
213        };
214
215        // Try create-exclusive first; if it already exists, open.
216        let (raw_fd, created) = unsafe {
217            // SAFETY: passing a valid C string and well-known POSIX flags.
218            let fd = libc::shm_open(
219                cname.as_ptr(),
220                libc::O_RDWR | libc::O_CREAT | libc::O_EXCL,
221                0o600,
222            );
223            if fd >= 0 {
224                (fd, true)
225            } else {
226                // Could be EEXIST (already created by a peer) or another error.
227                let err = io::Error::last_os_error();
228                if err.raw_os_error() != Some(libc::EEXIST) {
229                    return Err(err);
230                }
231                let fd = libc::shm_open(cname.as_ptr(), libc::O_RDWR, 0o600);
232                if fd < 0 {
233                    return Err(io::Error::last_os_error());
234                }
235                (fd, false)
236            }
237        };
238        // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
239        // owned by this scope.
240        let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
241
242        // Size the segment on first creation.
243        if created {
244            // SAFETY: fd is a valid POSIX fd we just received.
245            let rc = unsafe { libc::ftruncate(fd.as_raw_fd(), size as libc::off_t) };
246            if rc != 0 {
247                let err = io::Error::last_os_error();
248                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
249                return Err(err);
250            }
251        }
252
253        // Never mmap beyond the real SHM object: access past it can raise
254        // SIGBUS. A larger reported size is valid on platforms (notably
255        // macOS) that page-round POSIX SHM objects; callers verify their
256        // own ABI metadata after mapping.
257        let actual_size = match shm_object_size(&fd, name) {
258            Ok(actual_size) => actual_size,
259            Err(error) => {
260                if created {
261                    let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
262                }
263                return Err(error);
264            }
265        };
266        if let Err(error) = validate_minimum_size(name, actual_size, size) {
267            if created {
268                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
269            }
270            return Err(error);
271        }
272
273        // Memory-map the segment.
274        // SAFETY: fd valid, size positive, flags well-known.
275        let ptr = unsafe {
276            libc::mmap(
277                std::ptr::null_mut(),
278                size,
279                libc::PROT_READ | libc::PROT_WRITE,
280                libc::MAP_SHARED,
281                fd.as_raw_fd(),
282                0,
283            )
284        };
285
286        if ptr == libc::MAP_FAILED {
287            let err = io::Error::last_os_error();
288            if created {
289                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
290            }
291            return Err(err);
292        }
293
294        // SAFETY: mmap returned a non-null pointer (we just checked).
295        let ptr = NonNull::new(ptr.cast::<u8>()).expect("mmap returned non-null on success");
296
297        Ok((
298            Self {
299                name: cname,
300                lock_path,
301                process_lock,
302                ptr,
303                len: size,
304                created,
305            },
306            initialization_lock,
307        ))
308    }
309
310    /// Raw mapped pointer to the start of the region.
311    pub fn as_ptr(&self) -> *mut u8 {
312        self.ptr.as_ptr()
313    }
314
315    /// Length of the mapped region (the `size` passed to `open_or_create`).
316    pub fn len(&self) -> usize {
317        self.len
318    }
319
320    pub fn is_empty(&self) -> bool {
321        self.len == 0
322    }
323
324    /// True when this handle was the one that created the segment.
325    /// Useful for picking which process performs first-time
326    /// initialization of the header.
327    pub fn created(&self) -> bool {
328        self.created
329    }
330
331    /// Acquire an exclusive cross-process lock tied to this SHM name.
332    ///
333    /// `flock` ownership is held by the kernel and is released when a process
334    /// exits or the descriptor closes, including abnormal termination.
335    pub fn lock_exclusive(&self) -> io::Result<ShmRegionLock> {
336        if !self.process_lock {
337            return Err(io::Error::new(
338                io::ErrorKind::InvalidInput,
339                "SHM region was opened without a process lock",
340            ));
341        }
342        lock_path_exclusive(&self.lock_path)
343    }
344
345    #[cfg(test)]
346    pub(crate) fn try_lock_exclusive(&self) -> io::Result<ShmRegionLock> {
347        if !self.process_lock {
348            return Err(io::Error::new(
349                io::ErrorKind::InvalidInput,
350                "SHM region was opened without a process lock",
351            ));
352        }
353        let lock_fd = open_lock_file(&self.lock_path)?;
354        let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) };
355        if rc == 0 {
356            Ok(ShmRegionLock { lock_fd })
357        } else {
358            Err(io::Error::last_os_error())
359        }
360    }
361
362    /// Remove the underlying segment name. Existing mappings stay
363    /// valid until each process drops its `ShmRegion`. Use only on
364    /// shutdown / fleet teardown by the process that owns lifecycle.
365    pub fn unlink(&self) -> io::Result<()> {
366        // SAFETY: name is a valid C string.
367        let rc = unsafe { libc::shm_unlink(self.name.as_ptr()) };
368        let shm_error = if rc != 0 {
369            let err = io::Error::last_os_error();
370            // ENOENT is fine — segment was already unlinked.
371            if err.raw_os_error() == Some(libc::ENOENT) {
372                None
373            } else {
374                Some(err)
375            }
376        } else {
377            None
378        };
379        let lock_error = match std::fs::remove_file(&self.lock_path) {
380            Ok(()) => None,
381            Err(error) if error.kind() == io::ErrorKind::NotFound => None,
382            Err(error) => Some(error),
383        };
384        if let Some(error) = shm_error.or(lock_error) {
385            return Err(error);
386        }
387        Ok(())
388    }
389}
390
391fn shm_object_size(fd: &OwnedFd, name: &str) -> io::Result<usize> {
392    let mut stat = std::mem::MaybeUninit::<libc::stat>::uninit();
393    // SAFETY: `fd` is valid and `stat` points to writable storage.
394    let stat_rc = unsafe { libc::fstat(fd.as_raw_fd(), stat.as_mut_ptr()) };
395    if stat_rc != 0 {
396        return Err(io::Error::last_os_error());
397    }
398    // SAFETY: `fstat` succeeded and initialized the structure.
399    let actual_size = unsafe { stat.assume_init() }.st_size;
400    usize::try_from(actual_size).map_err(|_| {
401        io::Error::new(
402            io::ErrorKind::InvalidData,
403            format!("SHM segment {name} reported invalid size {actual_size}"),
404        )
405    })
406}
407
408fn validate_minimum_size(name: &str, actual_size: usize, minimum_size: usize) -> io::Result<()> {
409    if actual_size < minimum_size {
410        return Err(io::Error::new(
411            io::ErrorKind::InvalidData,
412            format!(
413                "SHM segment {name} size {actual_size} is smaller than requested mapping {minimum_size}"
414            ),
415        ));
416    }
417    Ok(())
418}
419
420/// RAII guard for a [`ShmRegion`]'s process-recoverable exclusive lock.
421///
422/// Semantic crates use this when a current-state transition must be atomic
423/// across fleet processes. Dropping the guard releases the kernel lock.
424pub struct ShmRegionLock {
425    lock_fd: OwnedFd,
426}
427
428impl Drop for ShmRegionLock {
429    fn drop(&mut self) {
430        let _ = unsafe { libc::flock(self.lock_fd.as_raw_fd(), libc::LOCK_UN) };
431    }
432}
433
434fn lock_path_exclusive(lock_path: &Path) -> io::Result<ShmRegionLock> {
435    lock_fd_exclusive(open_lock_file(lock_path)?)
436}
437
438fn open_lock_file(lock_path: &Path) -> io::Result<OwnedFd> {
439    use std::os::unix::fs::MetadataExt;
440
441    if let Some(dir) = lock_path.parent() {
442        ensure_lock_dir(dir)?;
443    }
444
445    let file = OpenOptions::new()
446        .read(true)
447        .write(true)
448        .create(true)
449        .mode(0o600)
450        .custom_flags(libc::O_CLOEXEC | libc::O_NOFOLLOW)
451        .open(lock_path)?;
452
453    // The directory check makes this unreachable, which is the reason to make
454    // it anyway: it turns a property inferred from the directory's mode into
455    // one this function establishes about the descriptor it is about to lock.
456    let uid = unsafe { libc::geteuid() };
457    let owner = file.metadata()?.uid();
458    if owner != uid {
459        return Err(io::Error::new(
460            io::ErrorKind::PermissionDenied,
461            format!(
462                "{} is owned by uid {owner} rather than {uid}",
463                lock_path.display()
464            ),
465        ));
466    }
467
468    Ok(file.into())
469}
470
471fn lock_fd_exclusive(lock_fd: OwnedFd) -> io::Result<ShmRegionLock> {
472    loop {
473        let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX) };
474        if rc == 0 {
475            return Ok(ShmRegionLock { lock_fd });
476        }
477        let error = io::Error::last_os_error();
478        if error.kind() != io::ErrorKind::Interrupted {
479            return Err(error);
480        }
481    }
482}
483
484impl Drop for ShmRegion {
485    fn drop(&mut self) {
486        // SAFETY: ptr came from mmap of `self.len` bytes; munmap is the inverse.
487        unsafe {
488            libc::munmap(self.ptr.as_ptr().cast(), self.len);
489        }
490    }
491}
492
493// SAFETY: the underlying region is shared memory and synchronization
494// happens at the slot level (atomic seq counters); the handle itself
495// is just a pointer + length, safe to send/share.
496unsafe impl Send for ShmRegion {}
497unsafe impl Sync for ShmRegion {}
498
499/// Build the conventional name for an Orbit ring segment.
500pub fn ring_segment_name(fleet_name: &str, kind: u8) -> String {
501    // SAFETY: `geteuid` always returns a value; no error path.
502    let uid = unsafe { libc::geteuid() };
503    ring_segment_name_for_uid(fleet_name, kind, uid)
504}
505
506/// Build the conventional name for an Orbit ring segment owned by `uid`.
507///
508/// This form is intended for inspection and lifecycle tools that need to
509/// address a user other than their own effective uid.
510pub fn ring_segment_name_for_uid(fleet_name: &str, kind: u8, uid: u32) -> String {
511    format!("/{SHM_NAMESPACE}-{fleet_name}-{kind}-{uid}")
512}
513
514fn lock_file_path(shm_name: &str) -> PathBuf {
515    lock_dir().join(format!("{}.lock", shm_name.trim_start_matches('/')))
516}
517
518/// Where the companion lock files live.
519///
520/// They used to live in `/tmp` directly, as
521/// `/tmp/orbit-{fleet}-{kind}-{uid}.lock`, opened `O_CREAT` without `O_EXCL`
522/// and without asking who owned what the open found. `/tmp` is world-writable,
523/// so any local user could create that file first and then hold `LOCK_EX` on it
524/// for as long as they liked: every process in the fleet would sit in `flock` —
525/// not fail, block — waiting for a lock it was never going to get. The SHM
526/// segment name is uid-scoped and so cannot be squatted this way; the lock path
527/// was not. `O_NOFOLLOW` prevented the symlink version of the trick and nothing
528/// else.
529///
530/// They now live in a per-uid directory created `0700`, which a user who is not
531/// us cannot put a file into. What such a user can still do is create the
532/// directory first, so its owner and mode are checked on every open rather than
533/// assumed from having created it: a directory that is not ours, or not
534/// private, fails the open with the path in the message instead of parking the
535/// process on a lock.
536///
537/// `XDG_RUNTIME_DIR` is preferred where the session provides one, because it is
538/// already per-user and `0700` and so is not inside a world-writable directory
539/// at all. macOS has no such variable but gives each user a private `TMPDIR`;
540/// `/tmp` is the fallback, with the checks above carrying the weight.
541fn lock_dir() -> PathBuf {
542    let uid = unsafe { libc::geteuid() };
543    let base = std::env::var_os("XDG_RUNTIME_DIR")
544        .map(PathBuf::from)
545        .filter(|dir| dir.is_absolute())
546        .unwrap_or_else(|| PathBuf::from("/tmp"));
547
548    base.join(format!("{SHM_NAMESPACE}-{uid}"))
549}
550
551/// Creates the lock directory if it is missing and refuses it if it is not
552/// ours. Called when a lock is actually taken rather than when a region is
553/// opened: a region that never locks has nothing to squat, and failing its open
554/// on a directory it does not use would hand an attacker a wider outage than
555/// the one being closed.
556fn ensure_lock_dir(dir: &Path) -> io::Result<()> {
557    use std::os::unix::fs::DirBuilderExt;
558
559    let uid = unsafe { libc::geteuid() };
560    match std::fs::DirBuilder::new().mode(0o700).create(dir) {
561        Ok(()) => {}
562        Err(error) if error.kind() == io::ErrorKind::AlreadyExists => {}
563        Err(error) => return Err(error),
564    }
565
566    ensure_private_dir(dir, uid)
567}
568
569/// Refuses a lock directory that someone else could write to.
570///
571/// `symlink_metadata` rather than `metadata`: a symlink pointing at a directory
572/// we do own would otherwise pass while the lock files landed somewhere the
573/// attacker chose.
574fn ensure_private_dir(dir: &Path, uid: u32) -> io::Result<()> {
575    use std::os::unix::fs::MetadataExt;
576    use std::os::unix::fs::PermissionsExt;
577
578    let metadata = std::fs::symlink_metadata(dir)?;
579    if !metadata.is_dir() {
580        return Err(io::Error::new(
581            io::ErrorKind::PermissionDenied,
582            format!("{} is not a directory", dir.display()),
583        ));
584    }
585    if metadata.uid() != uid {
586        return Err(io::Error::new(
587            io::ErrorKind::PermissionDenied,
588            format!(
589                "{} is owned by uid {} rather than {uid}; refusing to lock in a directory \
590                 another user controls",
591                dir.display(),
592                metadata.uid()
593            ),
594        ));
595    }
596    if metadata.permissions().mode() & 0o077 != 0 {
597        return Err(io::Error::new(
598            io::ErrorKind::PermissionDenied,
599            format!(
600                "{} is mode {:o}; refusing to lock in a directory others can write to",
601                dir.display(),
602                metadata.permissions().mode() & 0o777
603            ),
604        ));
605    }
606
607    Ok(())
608}