Skip to main content

orbit_core/
shm.rs

1//! POSIX shared-memory helpers — V1 substrate for cross-process rings.
2//!
3//! Wraps `shm_open` / `ftruncate` / `mmap` / `munmap` / `shm_unlink`
4//! into a small, RAII-friendly API. Unix-only; Windows support is
5//! a separate concern (Win32 named file mapping) that can land later.
6//!
7//! ## Naming
8//!
9//! Segments are named `/orbit-{fleet}-{kind}-{uid}` — fleet name from
10//! the embedder, KIND from `OrbitTyped::KIND`, UID from `geteuid()`.
11//! UID-scoping avoids the `/dev/shm` sticky-bit cross-user collision
12//! problem (a stale segment owned by one user blocks another from
13//! `shm_unlink`-ing it on next boot).
14//!
15//! Rings that require a process-recoverable writer lock also open a
16//! companion `orbit-{fleet}-{kind}-{uid}.lock` file. It carries no
17//! ring data or state; it only supplies a regular-file inode for `flock`,
18//! because advisory locking on a POSIX SHM descriptor is not uniformly
19//! supported across the Unix targets Orbit serves. An unlocked stale
20//! companion file is safe to reuse.
21//!
22//! Those files live in a per-uid directory — `$XDG_RUNTIME_DIR/orbit-{uid}`
23//! where the session provides one, `/tmp/orbit-{uid}` otherwise — created
24//! `0700` and checked on every lock. They were once in `/tmp` directly, which
25//! made them squattable: see [`lock_dir`].
26//!
27//! ## Lifetime
28//!
29//! [`ShmRegion`] owns the mapped pointer and unmaps on drop. It does
30//! NOT `shm_unlink` on drop — the segment lives until an explicit
31//! [`ShmRegion::unlink`] call. This matches POSIX convention: a
32//! segment with mapped users is not removed; `shm_unlink` only
33//! prevents *new* opens, the current mapping stays valid until the
34//! last process unmaps.
35
36#![cfg(unix)]
37
38use std::ffi::CString;
39use std::fs::OpenOptions;
40use std::io;
41use std::os::fd::{AsRawFd, FromRawFd, OwnedFd};
42use std::os::unix::fs::OpenOptionsExt;
43use std::path::{Path, PathBuf};
44use std::ptr::NonNull;
45
46/// Namespace used by Orbit POSIX shared-memory objects.
47pub const SHM_NAMESPACE: &str = "orbit";
48
49/// A mapped POSIX SHM region. Drop unmaps; `unlink` removes the
50/// underlying name and any companion lock file (only the *creator*
51/// should call it on shutdown).
52pub struct ShmRegion {
53    name: CString,
54    lock_path: PathBuf,
55    /// Whether this region uses the companion file for ordered writes.
56    ///
57    /// The descriptor itself is deliberately not retained: every critical
58    /// section opens its own file description so a later fork does not inherit
59    /// an idle descriptor that can keep a future `flock` alive.
60    process_lock: bool,
61    ptr: NonNull<u8>,
62    len: usize,
63    /// True when this handle was the one that *created* the segment
64    /// (so it knows to `shm_unlink` if asked). Other attachers see
65    /// `false`.
66    created: bool,
67}
68
69impl ShmRegion {
70    /// Open or create a shared-memory segment of `size` bytes,
71    /// memory-mapped read/write. Idempotent: if the segment already
72    /// exists with the same name and enough mapped bytes, it is reused
73    /// (`created = false`). First creation does `ftruncate(size)`;
74    /// later opens verify the existing object before mapping it. Some
75    /// platforms report a page-rounded SHM size, so a larger `st_size`
76    /// is valid; the owning data structure must verify its own header.
77    pub fn open_or_create(name: &str, size: usize) -> io::Result<Self> {
78        let (region, initialization_lock) = Self::open_or_create_inner(name, size, false)?;
79        debug_assert!(initialization_lock.is_none());
80        Ok(region)
81    }
82
83    /// Open or create a region while holding its process lock through caller
84    /// initialization. This prevents a peer from observing the interval
85    /// between `shm_open` and the owning data structure's initialized header.
86    pub fn open_or_create_locked(name: &str, size: usize) -> io::Result<(Self, ShmRegionLock)> {
87        let (region, initialization_lock) = Self::open_or_create_inner(name, size, true)?;
88        Ok((
89            region,
90            initialization_lock.expect("locked SHM open must return its initialization lock"),
91        ))
92    }
93
94    fn open_or_create_inner(
95        name: &str,
96        size: usize,
97        process_lock: bool,
98    ) -> io::Result<(Self, Option<ShmRegionLock>)> {
99        let cname = CString::new(name)
100            .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
101        let lock_path = lock_file_path(name);
102        let initialization_lock = if process_lock {
103            Some(lock_path_exclusive(&lock_path)?)
104        } else {
105            None
106        };
107
108        // Try create-exclusive first; if it already exists, open.
109        let (raw_fd, created) = unsafe {
110            // SAFETY: passing a valid C string and well-known POSIX flags.
111            let fd = libc::shm_open(
112                cname.as_ptr(),
113                libc::O_RDWR | libc::O_CREAT | libc::O_EXCL,
114                0o600,
115            );
116            if fd >= 0 {
117                (fd, true)
118            } else {
119                // Could be EEXIST (already created by a peer) or another error.
120                let err = io::Error::last_os_error();
121                if err.raw_os_error() != Some(libc::EEXIST) {
122                    return Err(err);
123                }
124                let fd = libc::shm_open(cname.as_ptr(), libc::O_RDWR, 0o600);
125                if fd < 0 {
126                    return Err(io::Error::last_os_error());
127                }
128                (fd, false)
129            }
130        };
131        // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
132        // owned by this scope.
133        let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
134
135        // Size the segment on first creation.
136        if created {
137            // SAFETY: fd is a valid POSIX fd we just received.
138            let rc = unsafe { libc::ftruncate(fd.as_raw_fd(), size as libc::off_t) };
139            if rc != 0 {
140                let err = io::Error::last_os_error();
141                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
142                return Err(err);
143            }
144        }
145
146        // Never mmap beyond the real SHM object: access past it can raise
147        // SIGBUS. A larger reported size is valid on platforms (notably
148        // macOS) that page-round POSIX SHM objects; callers verify their
149        // own ABI metadata after mapping.
150        let mut stat = std::mem::MaybeUninit::<libc::stat>::uninit();
151        let stat_rc = unsafe { libc::fstat(fd.as_raw_fd(), stat.as_mut_ptr()) };
152        if stat_rc != 0 {
153            let err = io::Error::last_os_error();
154            if created {
155                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
156            }
157            return Err(err);
158        }
159        let actual_size = unsafe { stat.assume_init() }.st_size;
160        if actual_size < 0 || (actual_size as usize) < size {
161            if created {
162                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
163            }
164            return Err(io::Error::new(
165                io::ErrorKind::InvalidData,
166                format!(
167                    "SHM segment {name} size {actual_size} is smaller than requested mapping {size}"
168                ),
169            ));
170        }
171
172        // Memory-map the segment.
173        // SAFETY: fd valid, size positive, flags well-known.
174        let ptr = unsafe {
175            libc::mmap(
176                std::ptr::null_mut(),
177                size,
178                libc::PROT_READ | libc::PROT_WRITE,
179                libc::MAP_SHARED,
180                fd.as_raw_fd(),
181                0,
182            )
183        };
184
185        if ptr == libc::MAP_FAILED {
186            let err = io::Error::last_os_error();
187            if created {
188                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
189            }
190            return Err(err);
191        }
192
193        // SAFETY: mmap returned a non-null pointer (we just checked).
194        let ptr = NonNull::new(ptr.cast::<u8>()).expect("mmap returned non-null on success");
195
196        Ok((
197            Self {
198                name: cname,
199                lock_path,
200                process_lock,
201                ptr,
202                len: size,
203                created,
204            },
205            initialization_lock,
206        ))
207    }
208
209    /// Raw mapped pointer to the start of the region.
210    pub fn as_ptr(&self) -> *mut u8 {
211        self.ptr.as_ptr()
212    }
213
214    /// Length of the mapped region (the `size` passed to `open_or_create`).
215    pub fn len(&self) -> usize {
216        self.len
217    }
218
219    pub fn is_empty(&self) -> bool {
220        self.len == 0
221    }
222
223    /// True when this handle was the one that created the segment.
224    /// Useful for picking which process performs first-time
225    /// initialization of the header.
226    pub fn created(&self) -> bool {
227        self.created
228    }
229
230    /// Acquire an exclusive cross-process lock tied to this SHM name.
231    ///
232    /// `flock` ownership is held by the kernel and is released when a process
233    /// exits or the descriptor closes, including abnormal termination.
234    pub fn lock_exclusive(&self) -> io::Result<ShmRegionLock> {
235        if !self.process_lock {
236            return Err(io::Error::new(
237                io::ErrorKind::InvalidInput,
238                "SHM region was opened without a process lock",
239            ));
240        }
241        lock_path_exclusive(&self.lock_path)
242    }
243
244    #[cfg(test)]
245    pub(crate) fn try_lock_exclusive(&self) -> io::Result<ShmRegionLock> {
246        if !self.process_lock {
247            return Err(io::Error::new(
248                io::ErrorKind::InvalidInput,
249                "SHM region was opened without a process lock",
250            ));
251        }
252        let lock_fd = open_lock_file(&self.lock_path)?;
253        let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) };
254        if rc == 0 {
255            Ok(ShmRegionLock { lock_fd })
256        } else {
257            Err(io::Error::last_os_error())
258        }
259    }
260
261    /// Remove the underlying segment name. Existing mappings stay
262    /// valid until each process drops its `ShmRegion`. Use only on
263    /// shutdown / fleet teardown by the process that owns lifecycle.
264    pub fn unlink(&self) -> io::Result<()> {
265        // SAFETY: name is a valid C string.
266        let rc = unsafe { libc::shm_unlink(self.name.as_ptr()) };
267        let shm_error = if rc != 0 {
268            let err = io::Error::last_os_error();
269            // ENOENT is fine — segment was already unlinked.
270            if err.raw_os_error() == Some(libc::ENOENT) {
271                None
272            } else {
273                Some(err)
274            }
275        } else {
276            None
277        };
278        let lock_error = match std::fs::remove_file(&self.lock_path) {
279            Ok(()) => None,
280            Err(error) if error.kind() == io::ErrorKind::NotFound => None,
281            Err(error) => Some(error),
282        };
283        if let Some(error) = shm_error.or(lock_error) {
284            return Err(error);
285        }
286        Ok(())
287    }
288}
289
290/// RAII guard for a [`ShmRegion`]'s process-recoverable exclusive lock.
291///
292/// Semantic crates use this when a current-state transition must be atomic
293/// across fleet processes. Dropping the guard releases the kernel lock.
294pub struct ShmRegionLock {
295    lock_fd: OwnedFd,
296}
297
298impl Drop for ShmRegionLock {
299    fn drop(&mut self) {
300        let _ = unsafe { libc::flock(self.lock_fd.as_raw_fd(), libc::LOCK_UN) };
301    }
302}
303
304fn lock_path_exclusive(lock_path: &Path) -> io::Result<ShmRegionLock> {
305    lock_fd_exclusive(open_lock_file(lock_path)?)
306}
307
308fn open_lock_file(lock_path: &Path) -> io::Result<OwnedFd> {
309    use std::os::unix::fs::MetadataExt;
310
311    if let Some(dir) = lock_path.parent() {
312        ensure_lock_dir(dir)?;
313    }
314
315    let file = OpenOptions::new()
316        .read(true)
317        .write(true)
318        .create(true)
319        .mode(0o600)
320        .custom_flags(libc::O_CLOEXEC | libc::O_NOFOLLOW)
321        .open(lock_path)?;
322
323    // The directory check makes this unreachable, which is the reason to make
324    // it anyway: it turns a property inferred from the directory's mode into
325    // one this function establishes about the descriptor it is about to lock.
326    let uid = unsafe { libc::geteuid() };
327    let owner = file.metadata()?.uid();
328    if owner != uid {
329        return Err(io::Error::new(
330            io::ErrorKind::PermissionDenied,
331            format!(
332                "{} is owned by uid {owner} rather than {uid}",
333                lock_path.display()
334            ),
335        ));
336    }
337
338    Ok(file.into())
339}
340
341fn lock_fd_exclusive(lock_fd: OwnedFd) -> io::Result<ShmRegionLock> {
342    loop {
343        let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX) };
344        if rc == 0 {
345            return Ok(ShmRegionLock { lock_fd });
346        }
347        let error = io::Error::last_os_error();
348        if error.kind() != io::ErrorKind::Interrupted {
349            return Err(error);
350        }
351    }
352}
353
354impl Drop for ShmRegion {
355    fn drop(&mut self) {
356        // SAFETY: ptr came from mmap of `self.len` bytes; munmap is the inverse.
357        unsafe {
358            libc::munmap(self.ptr.as_ptr().cast(), self.len);
359        }
360    }
361}
362
363// SAFETY: the underlying region is shared memory and synchronization
364// happens at the slot level (atomic seq counters); the handle itself
365// is just a pointer + length, safe to send/share.
366unsafe impl Send for ShmRegion {}
367unsafe impl Sync for ShmRegion {}
368
369/// Build the conventional name for an Orbit ring segment.
370pub fn ring_segment_name(fleet_name: &str, kind: u8) -> String {
371    // SAFETY: `geteuid` always returns a value; no error path.
372    let uid = unsafe { libc::geteuid() };
373    ring_segment_name_for_uid(fleet_name, kind, uid)
374}
375
376/// Build the conventional name for an Orbit ring segment owned by `uid`.
377///
378/// This form is intended for inspection and lifecycle tools that need to
379/// address a user other than their own effective uid.
380pub fn ring_segment_name_for_uid(fleet_name: &str, kind: u8, uid: u32) -> String {
381    format!("/{SHM_NAMESPACE}-{fleet_name}-{kind}-{uid}")
382}
383
384fn lock_file_path(shm_name: &str) -> PathBuf {
385    lock_dir().join(format!("{}.lock", shm_name.trim_start_matches('/')))
386}
387
388/// Where the companion lock files live.
389///
390/// They used to live in `/tmp` directly, as
391/// `/tmp/orbit-{fleet}-{kind}-{uid}.lock`, opened `O_CREAT` without `O_EXCL`
392/// and without asking who owned what the open found. `/tmp` is world-writable,
393/// so any local user could create that file first and then hold `LOCK_EX` on it
394/// for as long as they liked: every process in the fleet would sit in `flock` —
395/// not fail, block — waiting for a lock it was never going to get. The SHM
396/// segment name is uid-scoped and so cannot be squatted this way; the lock path
397/// was not. `O_NOFOLLOW` prevented the symlink version of the trick and nothing
398/// else.
399///
400/// They now live in a per-uid directory created `0700`, which a user who is not
401/// us cannot put a file into. What such a user can still do is create the
402/// directory first, so its owner and mode are checked on every open rather than
403/// assumed from having created it: a directory that is not ours, or not
404/// private, fails the open with the path in the message instead of parking the
405/// process on a lock.
406///
407/// `XDG_RUNTIME_DIR` is preferred where the session provides one, because it is
408/// already per-user and `0700` and so is not inside a world-writable directory
409/// at all. macOS has no such variable but gives each user a private `TMPDIR`;
410/// `/tmp` is the fallback, with the checks above carrying the weight.
411fn lock_dir() -> PathBuf {
412    let uid = unsafe { libc::geteuid() };
413    let base = std::env::var_os("XDG_RUNTIME_DIR")
414        .map(PathBuf::from)
415        .filter(|dir| dir.is_absolute())
416        .unwrap_or_else(|| PathBuf::from("/tmp"));
417
418    base.join(format!("{SHM_NAMESPACE}-{uid}"))
419}
420
421/// Creates the lock directory if it is missing and refuses it if it is not
422/// ours. Called when a lock is actually taken rather than when a region is
423/// opened: a region that never locks has nothing to squat, and failing its open
424/// on a directory it does not use would hand an attacker a wider outage than
425/// the one being closed.
426fn ensure_lock_dir(dir: &Path) -> io::Result<()> {
427    use std::os::unix::fs::DirBuilderExt;
428
429    let uid = unsafe { libc::geteuid() };
430    match std::fs::DirBuilder::new().mode(0o700).create(dir) {
431        Ok(()) => {}
432        Err(error) if error.kind() == io::ErrorKind::AlreadyExists => {}
433        Err(error) => return Err(error),
434    }
435
436    ensure_private_dir(dir, uid)
437}
438
439/// Refuses a lock directory that someone else could write to.
440///
441/// `symlink_metadata` rather than `metadata`: a symlink pointing at a directory
442/// we do own would otherwise pass while the lock files landed somewhere the
443/// attacker chose.
444fn ensure_private_dir(dir: &Path, uid: u32) -> io::Result<()> {
445    use std::os::unix::fs::MetadataExt;
446    use std::os::unix::fs::PermissionsExt;
447
448    let metadata = std::fs::symlink_metadata(dir)?;
449    if !metadata.is_dir() {
450        return Err(io::Error::new(
451            io::ErrorKind::PermissionDenied,
452            format!("{} is not a directory", dir.display()),
453        ));
454    }
455    if metadata.uid() != uid {
456        return Err(io::Error::new(
457            io::ErrorKind::PermissionDenied,
458            format!(
459                "{} is owned by uid {} rather than {uid}; refusing to lock in a directory \
460                 another user controls",
461                dir.display(),
462                metadata.uid()
463            ),
464        ));
465    }
466    if metadata.permissions().mode() & 0o077 != 0 {
467        return Err(io::Error::new(
468            io::ErrorKind::PermissionDenied,
469            format!(
470                "{} is mode {:o}; refusing to lock in a directory others can write to",
471                dir.display(),
472                metadata.permissions().mode() & 0o777
473            ),
474        ));
475    }
476
477    Ok(())
478}