Skip to main content

orbit_core/
shm.rs

1//! POSIX shared-memory helpers — V1 substrate for cross-process rings.
2//!
3//! Wraps `shm_open` / `ftruncate` / `mmap` / `munmap` / `shm_unlink`
4//! into a small, RAII-friendly API. Unix-only; Windows support is
5//! a separate concern (Win32 named file mapping) that can land later.
6//!
7//! ## Naming
8//!
9//! Segments are named `/orbit-{fleet}-{kind}-{uid}` — fleet name from
10//! the embedder, KIND from `OrbitTyped::KIND`, UID from `geteuid()`.
11//! UID-scoping avoids the `/dev/shm` sticky-bit cross-user collision
12//! problem (a stale segment owned by one user blocks another from
13//! `shm_unlink`-ing it on next boot).
14//!
15//! Rings that require a process-recoverable writer lock also open a
16//! companion `orbit-{fleet}-{kind}-{uid}.lock` file. It carries no
17//! ring data or state; it only supplies a regular-file inode for `flock`,
18//! because advisory locking on a POSIX SHM descriptor is not uniformly
19//! supported across the Unix targets Orbit serves. An unlocked stale
20//! companion file is safe to reuse.
21//!
22//! Those files live in a per-uid directory — `$XDG_RUNTIME_DIR/orbit-{uid}`
23//! where the session provides one, `/tmp/orbit-{uid}` otherwise — created
24//! `0700` and checked on every lock. They were once in `/tmp` directly, which
25//! made them squattable: see [`lock_dir`].
26//!
27//! ## Lifetime
28//!
29//! [`ShmRegion`] owns the mapped pointer and unmaps on drop. It does
30//! NOT `shm_unlink` on drop — the segment lives until an explicit
31//! [`ShmRegion::unlink`] call. This matches POSIX convention: a
32//! segment with mapped users is not removed; `shm_unlink` only
33//! prevents *new* opens, the current mapping stays valid until the
34//! last process unmaps.
35
36#![cfg(unix)]
37
38use std::ffi::CString;
39use std::fs::OpenOptions;
40use std::io;
41use std::os::fd::{AsRawFd, FromRawFd, OwnedFd};
42use std::os::unix::fs::OpenOptionsExt;
43use std::path::{Path, PathBuf};
44use std::ptr::NonNull;
45
46/// Namespace used by Orbit POSIX shared-memory objects.
47pub const SHM_NAMESPACE: &str = "orbit";
48
49/// Maximum POSIX access allowed for one SHM object, independent of its layout.
50///
51/// New objects are owned by the effective uid. On macOS, the selected group
52/// must be the creator's effective gid and the requested mode is passed directly
53/// to `shm_open` (with the platform's native creation-mask semantics).
54/// Other Unix targets create privately, then set the requested group and exact
55/// group-sharing mode. `OwnerOnly` uses the existing `shm_open(..., 0600)` path.
56/// Orbit never changes `umask` or process credentials, or chmods/chowns an
57/// existing object. This does not
58/// isolate processes running under the same uid.
59#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
60pub enum ShmAccessPolicy {
61    /// Owner read/write, no group or other access (`0600`).
62    #[default]
63    OwnerOnly,
64    /// Owner read/write and selected group read (`0640`).
65    GroupRead { gid: u32 },
66    /// Owner and selected group read/write (`0660`). Trust every group writer.
67    GroupReadWrite { gid: u32 }
68}
69
70impl ShmAccessPolicy {
71    pub const fn mode(self) -> u32 {
72        match self {
73            Self::OwnerOnly => 0o600,
74            Self::GroupRead { .. } => 0o640,
75            Self::GroupReadWrite { .. } => 0o660
76        }
77    }
78
79    pub const fn gid(self) -> Option<u32> {
80        match self {
81            Self::OwnerOnly => None,
82            Self::GroupRead { gid } | Self::GroupReadWrite { gid } => Some(gid)
83        }
84    }
85}
86
87/// Result of physically validating an existing POSIX SHM object.
88///
89/// This check is deliberately below ring semantics: it verifies that the
90/// named object satisfies the access policy and is large enough for the requested mapping,
91/// but it does not inspect an owning data structure's magic, version, or
92/// geometry header.
93#[derive(Clone, Copy, Debug, Eq, PartialEq)]
94pub enum ShmValidation {
95    /// No object currently exists under the requested name.
96    Missing,
97    /// The object exists and can safely back at least the requested mapping.
98    Valid { actual_size: usize }
99}
100
101/// A mapped POSIX SHM region. Drop unmaps; `unlink` removes the
102/// underlying name and any companion lock file (only the *creator*
103/// should call it on shutdown).
104pub struct ShmRegion {
105    name: CString,
106    lock_path: PathBuf,
107    /// Whether this region uses the companion file for ordered writes.
108    ///
109    /// The descriptor itself is deliberately not retained: every critical
110    /// section opens its own file description so a later fork does not inherit
111    /// an idle descriptor that can keep a future `flock` alive.
112    process_lock: bool,
113    ptr: NonNull<u8>,
114    len: usize,
115    /// True when this handle was the one that *created* the segment
116    /// (so it knows to `shm_unlink` if asked). Other attachers see
117    /// `false`.
118    created: bool
119}
120
121impl ShmRegion {
122    /// Validate an existing shared-memory object without creating, mapping,
123    /// resetting, or unlinking it.
124    ///
125    /// Returns [`ShmValidation::Missing`] when the name does not exist. A
126    /// present object must be at least `minimum_size` bytes; larger objects
127    /// are accepted because some platforms report page-rounded SHM sizes.
128    /// The owning ring or table remains responsible for validating its own
129    /// persisted ABI header after mapping.
130    /// The expected owner is the effective uid and the policy is `OwnerOnly`.
131    pub fn validate_existing(
132        name: &str,
133        minimum_size: usize
134    ) -> io::Result<ShmValidation> {
135        Self::validate_existing_with_policy(
136            name,
137            minimum_size,
138            unsafe { libc::geteuid() },
139            ShmAccessPolicy::default()
140        )
141    }
142
143    /// Validate size, expected owner and maximum permissions without mutation.
144    /// `owner_uid` is trusted configuration, not metadata read from the object.
145    pub fn validate_existing_with_policy(
146        name: &str,
147        minimum_size: usize,
148        owner_uid: u32,
149        policy: ShmAccessPolicy
150    ) -> io::Result<ShmValidation> {
151        let cname = CString::new(name)
152            .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
153        let raw_fd = loop {
154            // SAFETY: passing a valid C string and well-known POSIX flags.
155            // macOS `shm_open` rejects `O_CLOEXEC` with EINVAL. The descriptor
156            // is scoped to this validation call and closes before return.
157            let fd = unsafe { libc::shm_open(cname.as_ptr(), libc::O_RDONLY, 0o600) };
158            if fd >= 0 {
159                break fd;
160            }
161            let error = io::Error::last_os_error();
162            if error.kind() == io::ErrorKind::Interrupted {
163                continue;
164            }
165            if error.raw_os_error() == Some(libc::ENOENT) {
166                return Ok(ShmValidation::Missing);
167            }
168            return Err(error);
169        };
170        // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
171        // owned by this scope.
172        let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
173        let actual_size = shm_object_size(&fd, name, owner_uid, policy)?;
174        validate_minimum_size(name, actual_size, minimum_size)?;
175        Ok(ShmValidation::Valid { actual_size })
176    }
177
178    /// Map an existing shared-memory object read-only.
179    ///
180    /// This path never creates, sizes, locks, resets, or unlinks the object.
181    /// It is kept crate-private so callers receive a capability such as a
182    /// read-only ring view rather than a [`ShmRegion`] that also exposes
183    /// lifecycle and writable-pointer operations.
184    pub(crate) fn open_existing_read_only(
185        name: &str,
186        minimum_size: usize,
187        owner_uid: u32,
188        policy: ShmAccessPolicy
189    ) -> io::Result<Self> {
190        let cname = CString::new(name)
191            .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
192        let raw_fd = loop {
193            // SAFETY: passing a valid C string and read-only POSIX flags.
194            // macOS `shm_open` rejects `O_CLOEXEC` with EINVAL. This
195            // descriptor closes immediately after `mmap`, before the view is
196            // returned, so it cannot leak across a later exec.
197            let fd = unsafe { libc::shm_open(cname.as_ptr(), libc::O_RDONLY, 0o600) };
198            if fd >= 0 {
199                break fd;
200            }
201            let error = io::Error::last_os_error();
202            if error.kind() == io::ErrorKind::Interrupted {
203                continue;
204            }
205            return Err(error);
206        };
207        // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
208        // owned by this scope.
209        let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
210        let actual_size = shm_object_size(&fd, name, owner_uid, policy)?;
211        validate_minimum_size(name, actual_size, minimum_size)?;
212
213        // Map the complete object so its persisted header can describe the
214        // geometry without the observer reproducing the producer's layout.
215        // SAFETY: fd is valid, actual_size is positive after minimum
216        // validation, and the mapping is read-only.
217        let ptr = unsafe {
218            libc::mmap(
219                std::ptr::null_mut(),
220                actual_size,
221                libc::PROT_READ,
222                libc::MAP_SHARED,
223                fd.as_raw_fd(),
224                0
225            )
226        };
227        if ptr == libc::MAP_FAILED {
228            return Err(io::Error::last_os_error());
229        }
230        // SAFETY: mmap returned a non-null pointer (checked above).
231        let ptr = NonNull::new(ptr.cast::<u8>()).expect("mmap returned non-null on success");
232
233        Ok(Self {
234            lock_path: lock_file_path(name),
235            name: cname,
236            process_lock: false,
237            ptr,
238            len: actual_size,
239            created: false
240        })
241    }
242
243    /// Open or create a shared-memory segment of `size` bytes,
244    /// memory-mapped read/write. Idempotent: if the segment already
245    /// exists with the same name and enough mapped bytes, it is reused
246    /// (`created = false`). First creation does `ftruncate(size)`;
247    /// later opens verify the existing object before mapping it. Some
248    /// platforms report a page-rounded SHM size, so a larger `st_size`
249    /// is valid; the owning data structure must verify its own header.
250    /// Existing objects must belong to the effective uid and satisfy `OwnerOnly`.
251    pub fn open_or_create(
252        name: &str,
253        size: usize
254    ) -> io::Result<Self> {
255        Self::open_or_create_with_policy(name, size, ShmAccessPolicy::default())
256    }
257
258    /// Open or create using an explicit access policy. Existing objects must
259    /// belong to the effective uid and pass the policy before being mapped.
260    /// Rejection never resizes, chmods, chowns or unlinks an existing object.
261    pub fn open_or_create_with_policy(
262        name: &str,
263        size: usize,
264        policy: ShmAccessPolicy
265    ) -> io::Result<Self> {
266        let (region, initialization_lock) = Self::open_or_create_inner(name, size, false, policy)?;
267        debug_assert!(initialization_lock.is_none());
268        Ok(region)
269    }
270
271    /// Open or create a region while holding its process lock through caller
272    /// initialization. This prevents a peer from observing the interval
273    /// between `shm_open` and the owning data structure's initialized header.
274    pub fn open_or_create_locked(
275        name: &str,
276        size: usize
277    ) -> io::Result<(Self, ShmRegionLock)> {
278        Self::open_or_create_locked_with_policy(name, size, ShmAccessPolicy::default())
279    }
280
281    /// Policy-aware creation with the existing owner-local initialization lock.
282    /// Group permissions do not make that lock or fleet membership cross-user.
283    pub fn open_or_create_locked_with_policy(
284        name: &str,
285        size: usize,
286        policy: ShmAccessPolicy
287    ) -> io::Result<(Self, ShmRegionLock)> {
288        let (region, initialization_lock) = Self::open_or_create_inner(name, size, true, policy)?;
289        Ok((
290            region,
291            initialization_lock.expect("locked SHM open must return its initialization lock")
292        ))
293    }
294
295    fn open_or_create_inner(
296        name: &str,
297        size: usize,
298        process_lock: bool,
299        policy: ShmAccessPolicy
300    ) -> io::Result<(Self, Option<ShmRegionLock>)> {
301        let cname = CString::new(name)
302            .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
303        let lock_path = lock_file_path(name);
304        let initialization_lock =
305            if process_lock { Some(lock_path_exclusive(&lock_path)?) } else { None };
306
307        // Darwin cannot fchmod/fchown a POSIX SHM descriptor. Grant only the
308        // intended group at creation; attaching a matching existing object is OK.
309        #[cfg(target_os = "macos")]
310        let can_create = policy.gid().is_none_or(|gid| gid == unsafe { libc::getegid() });
311        #[cfg(target_os = "macos")]
312        let creation_mode = policy.mode();
313        // Elsewhere, don't expose the object to its initial group before chown.
314        #[cfg(not(target_os = "macos"))]
315        let (can_create, creation_mode) = (true, 0o600u32);
316        // Try create-exclusive first; if it already exists, open.
317        let (raw_fd, created) = unsafe {
318            // SAFETY: passing a valid C string and well-known POSIX flags.
319            let fd = if can_create {
320                libc::shm_open(
321                    cname.as_ptr(),
322                    libc::O_RDWR | libc::O_CREAT | libc::O_EXCL,
323                    creation_mode
324                )
325            } else {
326                libc::shm_open(cname.as_ptr(), libc::O_RDWR, 0)
327            };
328            if fd >= 0 {
329                (fd, can_create)
330            } else {
331                // Could be EEXIST (already created by a peer) or another error.
332                let err = io::Error::last_os_error();
333                if !can_create && err.raw_os_error() == Some(libc::ENOENT) {
334                    return Err(io::Error::new(
335                        io::ErrorKind::InvalidInput,
336                        "creating group-shared SHM requires the selected effective gid"
337                    ));
338                }
339                if err.raw_os_error() != Some(libc::EEXIST) {
340                    return Err(err);
341                }
342                let fd = libc::shm_open(cname.as_ptr(), libc::O_RDWR, 0o600);
343                if fd < 0 {
344                    return Err(io::Error::last_os_error());
345                }
346                (fd, false)
347            }
348        };
349        // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
350        // owned by this scope.
351        let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
352
353        #[cfg(not(target_os = "macos"))]
354        if created && let Err(error) = configure_new_shm(&fd, policy) {
355            let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
356            return Err(error);
357        }
358
359        // Verify ownership and permissions before ftruncate or mmap. Only a
360        // newly created object is ours to unlink on initialization failure.
361        let owner_uid = unsafe { libc::geteuid() };
362        if let Err(error) = shm_object_size(&fd, name, owner_uid, policy) {
363            if created {
364                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
365            }
366            return Err(error);
367        }
368
369        // Size the segment on first creation.
370        if created {
371            // SAFETY: fd is a valid POSIX fd we just received.
372            let rc = unsafe { libc::ftruncate(fd.as_raw_fd(), size as libc::off_t) };
373            if rc != 0 {
374                let err = io::Error::last_os_error();
375                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
376                return Err(err);
377            }
378        }
379
380        // Never mmap beyond the real SHM object: access past it can raise
381        // SIGBUS. A larger reported size is valid on platforms (notably
382        // macOS) that page-round POSIX SHM objects; callers verify their
383        // own ABI metadata after mapping.
384        let actual_size = match shm_object_size(&fd, name, owner_uid, policy) {
385            Ok(actual_size) => actual_size,
386            Err(error) => {
387                if created {
388                    let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
389                }
390                return Err(error);
391            }
392        };
393        if let Err(error) = validate_minimum_size(name, actual_size, size) {
394            if created {
395                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
396            }
397            return Err(error);
398        }
399
400        // Memory-map the segment.
401        // SAFETY: fd valid, size positive, flags well-known.
402        let ptr = unsafe {
403            libc::mmap(
404                std::ptr::null_mut(),
405                size,
406                libc::PROT_READ | libc::PROT_WRITE,
407                libc::MAP_SHARED,
408                fd.as_raw_fd(),
409                0
410            )
411        };
412
413        if ptr == libc::MAP_FAILED {
414            let err = io::Error::last_os_error();
415            if created {
416                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
417            }
418            return Err(err);
419        }
420
421        // SAFETY: mmap returned a non-null pointer (we just checked).
422        let ptr = NonNull::new(ptr.cast::<u8>()).expect("mmap returned non-null on success");
423
424        Ok((
425            Self { name: cname, lock_path, process_lock, ptr, len: size, created },
426            initialization_lock
427        ))
428    }
429
430    /// Raw mapped pointer to the start of the region.
431    pub fn as_ptr(&self) -> *mut u8 {
432        self.ptr.as_ptr()
433    }
434
435    /// Length of the mapped region (the `size` passed to `open_or_create`).
436    pub fn len(&self) -> usize {
437        self.len
438    }
439
440    pub fn is_empty(&self) -> bool {
441        self.len == 0
442    }
443
444    /// True when this handle was the one that created the segment.
445    /// Useful for picking which process performs first-time
446    /// initialization of the header.
447    pub fn created(&self) -> bool {
448        self.created
449    }
450
451    /// Acquire an exclusive cross-process lock tied to this SHM name.
452    ///
453    /// `flock` ownership is held by the kernel and is released when a process
454    /// exits or the descriptor closes, including abnormal termination.
455    pub fn lock_exclusive(&self) -> io::Result<ShmRegionLock> {
456        if !self.process_lock {
457            return Err(io::Error::new(
458                io::ErrorKind::InvalidInput,
459                "SHM region was opened without a process lock"
460            ));
461        }
462        lock_path_exclusive(&self.lock_path)
463    }
464
465    #[cfg(test)]
466    pub(crate) fn try_lock_exclusive(&self) -> io::Result<ShmRegionLock> {
467        if !self.process_lock {
468            return Err(io::Error::new(
469                io::ErrorKind::InvalidInput,
470                "SHM region was opened without a process lock"
471            ));
472        }
473        let lock_fd = open_lock_file(&self.lock_path)?;
474        let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) };
475        if rc == 0 { Ok(ShmRegionLock { lock_fd }) } else { Err(io::Error::last_os_error()) }
476    }
477
478    /// Remove the underlying segment name. Existing mappings stay
479    /// valid until each process drops its `ShmRegion`. Use only on
480    /// shutdown / fleet teardown by the process that owns lifecycle.
481    pub fn unlink(&self) -> io::Result<()> {
482        // SAFETY: name is a valid C string.
483        let rc = unsafe { libc::shm_unlink(self.name.as_ptr()) };
484        let shm_error = if rc != 0 {
485            let err = io::Error::last_os_error();
486            // ENOENT is fine — segment was already unlinked.
487            if err.raw_os_error() == Some(libc::ENOENT) { None } else { Some(err) }
488        } else {
489            None
490        };
491        let mut lock_error = match std::fs::remove_file(&self.lock_path) {
492            Ok(()) => None,
493            Err(error) if error.kind() == io::ErrorKind::NotFound => None,
494            Err(error) => Some(error)
495        };
496        let name = self.name.to_string_lossy();
497        // SAFETY: a plain syscall with no arguments.
498        let uid = unsafe { libc::geteuid() };
499        match companion_lock_files(&name, uid) {
500            Ok(files) => {
501                for file in files {
502                    if let Err(error) = std::fs::remove_file(&file)
503                        && error.kind() != io::ErrorKind::NotFound
504                    {
505                        lock_error.get_or_insert(error);
506                    }
507                }
508            }
509            Err(error) => {
510                lock_error.get_or_insert(error);
511            }
512        }
513        if let Some(error) = shm_error.or(lock_error) {
514            return Err(error);
515        }
516        Ok(())
517    }
518}
519
520#[cfg(not(target_os = "macos"))]
521fn configure_new_shm(
522    fd: &OwnedFd,
523    policy: ShmAccessPolicy
524) -> io::Result<()> {
525    let Some(gid) = policy.gid() else {
526        return Ok(());
527    };
528    if gid == !0 {
529        return Err(io::Error::new(io::ErrorKind::InvalidInput, "invalid SHM group id"));
530    }
531    // SAFETY: this is our new, still owner-only object. Preserve its uid.
532    if unsafe { libc::fchown(fd.as_raw_fd(), !0, gid) } != 0 {
533        return Err(io::Error::last_os_error());
534    }
535    // Explicit group-sharing policy, applied only after group ownership is set.
536    if unsafe { libc::fchmod(fd.as_raw_fd(), policy.mode() as _) } != 0 {
537        return Err(io::Error::last_os_error());
538    }
539    Ok(())
540}
541
542fn shm_object_size(
543    fd: &OwnedFd,
544    name: &str,
545    owner_uid: u32,
546    policy: ShmAccessPolicy
547) -> io::Result<usize> {
548    let mut stat = std::mem::MaybeUninit::<libc::stat>::uninit();
549    // SAFETY: `fd` is valid and `stat` points to writable storage.
550    let stat_rc = unsafe { libc::fstat(fd.as_raw_fd(), stat.as_mut_ptr()) };
551    if stat_rc != 0 {
552        return Err(io::Error::last_os_error());
553    }
554    // SAFETY: `fstat` succeeded and initialized the structure.
555    let stat = unsafe { stat.assume_init() };
556    let mode = u64::from(stat.st_mode) & 0o7777;
557    if stat.st_uid != owner_uid
558        || policy.gid().is_some_and(|gid| gid != stat.st_gid)
559        || mode & !u64::from(policy.mode()) != 0
560    {
561        return Err(io::Error::new(
562            io::ErrorKind::PermissionDenied,
563            format!(
564                "SHM segment {name} owner/group/mode do not satisfy {policy:?} for uid {owner_uid}"
565            )
566        ));
567    }
568    let actual_size = stat.st_size;
569    usize::try_from(actual_size).map_err(|_| {
570        io::Error::new(
571            io::ErrorKind::InvalidData,
572            format!("SHM segment {name} reported invalid size {actual_size}")
573        )
574    })
575}
576
577fn validate_minimum_size(
578    name: &str,
579    actual_size: usize,
580    minimum_size: usize
581) -> io::Result<()> {
582    if actual_size < minimum_size {
583        return Err(io::Error::new(
584            io::ErrorKind::InvalidData,
585            format!(
586                "SHM segment {name} size {actual_size} is smaller than requested mapping {minimum_size}"
587            )
588        ));
589    }
590    Ok(())
591}
592
593/// RAII guard for a [`ShmRegion`]'s process-recoverable exclusive lock.
594///
595/// Semantic crates use this when a current-state transition must be atomic
596/// across fleet processes. Dropping the guard releases the kernel lock.
597pub struct ShmRegionLock {
598    lock_fd: OwnedFd
599}
600
601impl Drop for ShmRegionLock {
602    fn drop(&mut self) {
603        let _ = unsafe { libc::flock(self.lock_fd.as_raw_fd(), libc::LOCK_UN) };
604    }
605}
606
607fn lock_path_exclusive(lock_path: &Path) -> io::Result<ShmRegionLock> {
608    lock_fd_exclusive(open_lock_file(lock_path)?)
609}
610
611fn open_lock_file(lock_path: &Path) -> io::Result<OwnedFd> {
612    use std::os::unix::fs::MetadataExt;
613
614    if let Some(dir) = lock_path.parent() {
615        ensure_lock_dir(dir)?;
616    }
617
618    let file = OpenOptions::new()
619        .read(true)
620        .write(true)
621        .create(true)
622        .mode(0o600)
623        .custom_flags(libc::O_CLOEXEC | libc::O_NOFOLLOW)
624        .open(lock_path)?;
625
626    // The directory check makes this unreachable, which is the reason to make
627    // it anyway: it turns a property inferred from the directory's mode into
628    // one this function establishes about the descriptor it is about to lock.
629    let uid = unsafe { libc::geteuid() };
630    let owner = file.metadata()?.uid();
631    if owner != uid {
632        return Err(io::Error::new(
633            io::ErrorKind::PermissionDenied,
634            format!("{} is owned by uid {owner} rather than {uid}", lock_path.display())
635        ));
636    }
637
638    Ok(file.into())
639}
640
641fn lock_fd_exclusive(lock_fd: OwnedFd) -> io::Result<ShmRegionLock> {
642    loop {
643        let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX) };
644        if rc == 0 {
645            return Ok(ShmRegionLock { lock_fd });
646        }
647        let error = io::Error::last_os_error();
648        if error.kind() != io::ErrorKind::Interrupted {
649            return Err(error);
650        }
651    }
652}
653
654impl Drop for ShmRegion {
655    fn drop(&mut self) {
656        // SAFETY: ptr came from mmap of `self.len` bytes; munmap is the inverse.
657        unsafe {
658            libc::munmap(self.ptr.as_ptr().cast(), self.len);
659        }
660    }
661}
662
663// SAFETY: the underlying region is shared memory and synchronization
664// happens at the slot level (atomic seq counters); the handle itself
665// is just a pointer + length, safe to send/share.
666unsafe impl Send for ShmRegion {}
667unsafe impl Sync for ShmRegion {}
668
669/// Build the conventional name for an Orbit ring segment.
670pub fn ring_segment_name(
671    fleet_name: &str,
672    kind: u8
673) -> String {
674    // SAFETY: `geteuid` always returns a value; no error path.
675    let uid = unsafe { libc::geteuid() };
676    ring_segment_name_for_uid(fleet_name, kind, uid)
677}
678
679/// Build the conventional name for an Orbit ring segment owned by `uid`.
680///
681/// This form is intended for inspection and lifecycle tools that need to
682/// address a user other than their own effective uid.
683pub fn ring_segment_name_for_uid(
684    fleet_name: &str,
685    kind: u8,
686    uid: u32
687) -> String {
688    format!("/{SHM_NAMESPACE}-{fleet_name}-{kind}-{uid}")
689}
690
691/// Where a fleet's members hold their presence: `<lock dir>/orbit-<fleet>.fleet`.
692pub fn fleet_lock_path(fleet_name: &str) -> PathBuf {
693    lock_dir().join(format!("{SHM_NAMESPACE}-{fleet_name}.fleet"))
694}
695
696/// The same for another user's fleet, for lifecycle tools that address a uid
697/// other than their own.
698pub fn fleet_lock_path_for_uid(
699    fleet_name: &str,
700    uid: u32
701) -> PathBuf {
702    lock_dir_for_uid(uid).join(format!("{SHM_NAMESPACE}-{fleet_name}.fleet"))
703}
704
705/// A process's membership in a fleet: a shared `flock` on the fleet's lock
706/// file, held for as long as this lives and released by the kernel when the
707/// process dies, however it dies. It is what [`try_lock_fleet_exclusive`]
708/// contends with, so a tool that removes the fleet's segments cannot do so
709/// while any member is alive.
710pub struct FleetMembership {
711    lock_fd: OwnedFd
712}
713
714impl Drop for FleetMembership {
715    fn drop(&mut self) {
716        let _ = unsafe { libc::flock(self.lock_fd.as_raw_fd(), libc::LOCK_UN) };
717    }
718}
719
720/// Join the fleet's membership. Waits for a lifecycle tool that holds the
721/// exclusive lock at that moment; a clear in progress finishes first.
722pub fn join_fleet_membership(fleet_name: &str) -> io::Result<FleetMembership> {
723    let lock_fd = open_lock_file(&fleet_lock_path(fleet_name))?;
724    // SAFETY: `lock_fd` is an open descriptor owned by this call.
725    let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_SH) };
726    if rc != 0 {
727        return Err(io::Error::last_os_error());
728    }
729    Ok(FleetMembership { lock_fd })
730}
731
732/// Exclusive hold on a fleet, for the tool that removes its segments.
733///
734/// `Ok(None)` means a member is alive and the fleet must not be touched;
735/// there is deliberately no way to force past it. `Ok(Some(_))` keeps the
736/// fleet closed to new members until the guard is dropped, so a removal
737/// cannot interleave with a start. Never waits.
738pub fn try_lock_fleet_exclusive(
739    fleet_name: &str,
740    uid: u32
741) -> io::Result<Option<ShmRegionLock>> {
742    // Creating the file when no member ever joined is right: the guard then
743    // keeps a first member from starting in the middle of a removal.
744    let lock_fd = open_lock_file(&fleet_lock_path_for_uid(fleet_name, uid))?;
745    // SAFETY: `lock_fd` is an open descriptor owned by this call.
746    let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) };
747    if rc == 0 {
748        return Ok(Some(ShmRegionLock { lock_fd }));
749    }
750    let error = io::Error::last_os_error();
751    if error.kind() == io::ErrorKind::WouldBlock {
752        return Ok(None);
753    }
754    Err(error)
755}
756
757/// One process's hold on one lane of a segment: an exclusive `flock` on the
758/// lane's own lock file, kept for as long as this lives and released by the
759/// kernel when the process dies, however it dies.
760///
761/// It is how a lane's owner is known to be gone without a timeout and without
762/// a PID, which another PID namespace would read wrongly: a hold that can be
763/// taken has nobody behind it.
764pub struct LaneHold {
765    lock_fd: OwnedFd
766}
767
768impl Drop for LaneHold {
769    fn drop(&mut self) {
770        let _ = unsafe { libc::flock(self.lock_fd.as_raw_fd(), libc::LOCK_UN) };
771    }
772}
773
774/// Take lane `lane` of segment `shm_name`, if no live process holds it.
775///
776/// `Ok(None)` means a process holding it is alive. Never waits.
777pub fn try_hold_lane(
778    shm_name: &str,
779    lane: usize
780) -> io::Result<Option<LaneHold>> {
781    let name = format!("{}.lane{lane}", shm_name.trim_start_matches('/'));
782    let lock_fd = open_lock_file(&lock_file_path(&name))?;
783    // SAFETY: `lock_fd` is an open descriptor owned by this call.
784    let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) };
785    if rc == 0 {
786        return Ok(Some(LaneHold { lock_fd }));
787    }
788    let error = io::Error::last_os_error();
789    if error.kind() == io::ErrorKind::WouldBlock {
790        return Ok(None);
791    }
792    Err(error)
793}
794
795/// Every lock file beside segment `shm_name` of `uid`: the region's own and
796/// one per lane that was ever held ([`try_hold_lane`]). What removes a segment
797/// removes these with it; an unlocked one left behind is harmless, but it is
798/// one more file per lane for every segment that ever existed.
799pub fn companion_lock_files(
800    shm_name: &str,
801    uid: u32
802) -> io::Result<Vec<PathBuf>> {
803    let base = shm_name.trim_start_matches('/');
804    let own = format!("{base}.lock");
805    let lane = format!("{base}.lane");
806    let dir = lock_dir_for_uid(uid);
807    let entries = match std::fs::read_dir(&dir) {
808        Ok(entries) => entries,
809        Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(Vec::new()),
810        Err(error) => return Err(error)
811    };
812    let mut found = Vec::new();
813    for entry in entries {
814        let entry = entry?;
815        let Some(file) = entry.file_name().to_str().map(str::to_owned) else {
816            continue;
817        };
818        if file == own || (file.starts_with(&lane) && file.ends_with(".lock")) {
819            found.push(entry.path());
820        }
821    }
822    Ok(found)
823}
824
825fn lock_file_path(shm_name: &str) -> PathBuf {
826    lock_dir().join(format!("{}.lock", shm_name.trim_start_matches('/')))
827}
828
829/// Where the companion lock files live.
830///
831/// They used to live in `/tmp` directly, as
832/// `/tmp/orbit-{fleet}-{kind}-{uid}.lock`, opened `O_CREAT` without `O_EXCL`
833/// and without asking who owned what the open found. `/tmp` is world-writable,
834/// so any local user could create that file first and then hold `LOCK_EX` on it
835/// for as long as they liked: every process in the fleet would sit in `flock` —
836/// not fail, block — waiting for a lock it was never going to get. The SHM
837/// segment name is uid-scoped and so cannot be squatted this way; the lock path
838/// was not. `O_NOFOLLOW` prevented the symlink version of the trick and nothing
839/// else.
840///
841/// They now live in a per-uid directory created `0700`, which a user who is not
842/// us cannot put a file into. What such a user can still do is create the
843/// directory first, so its owner and mode are checked on every open rather than
844/// assumed from having created it: a directory that is not ours, or not
845/// private, fails the open with the path in the message instead of parking the
846/// process on a lock.
847///
848/// `XDG_RUNTIME_DIR` is preferred where the session provides one, because it is
849/// already per-user and `0700` and so is not inside a world-writable directory
850/// at all. macOS has no such variable but gives each user a private `TMPDIR`;
851/// `/tmp` is the fallback, with the checks above carrying the weight.
852fn lock_dir() -> PathBuf {
853    let uid = unsafe { libc::geteuid() };
854    lock_dir_for_uid(uid)
855}
856
857fn lock_dir_for_uid(uid: u32) -> PathBuf {
858    let base = std::env::var_os("XDG_RUNTIME_DIR")
859        .map(PathBuf::from)
860        .filter(|dir| dir.is_absolute())
861        .unwrap_or_else(|| PathBuf::from("/tmp"));
862
863    base.join(format!("{SHM_NAMESPACE}-{uid}"))
864}
865
866/// Creates the lock directory if it is missing and refuses it if it is not
867/// ours. Called when a lock is actually taken rather than when a region is
868/// opened: a region that never locks has nothing to squat, and failing its open
869/// on a directory it does not use would hand an attacker a wider outage than
870/// the one being closed.
871fn ensure_lock_dir(dir: &Path) -> io::Result<()> {
872    use std::os::unix::fs::DirBuilderExt;
873
874    let uid = unsafe { libc::geteuid() };
875    match std::fs::DirBuilder::new().mode(0o700).create(dir) {
876        Ok(()) => {}
877        Err(error) if error.kind() == io::ErrorKind::AlreadyExists => {}
878        Err(error) => return Err(error)
879    }
880
881    ensure_private_dir(dir, uid)
882}
883
884/// Refuses a lock directory that someone else could write to.
885///
886/// `symlink_metadata` rather than `metadata`: a symlink pointing at a directory
887/// we do own would otherwise pass while the lock files landed somewhere the
888/// attacker chose.
889fn ensure_private_dir(
890    dir: &Path,
891    uid: u32
892) -> io::Result<()> {
893    use std::os::unix::fs::{MetadataExt, PermissionsExt};
894
895    let metadata = std::fs::symlink_metadata(dir)?;
896    if !metadata.is_dir() {
897        return Err(io::Error::new(
898            io::ErrorKind::PermissionDenied,
899            format!("{} is not a directory", dir.display())
900        ));
901    }
902    if metadata.uid() != uid {
903        return Err(io::Error::new(
904            io::ErrorKind::PermissionDenied,
905            format!(
906                "{} is owned by uid {} rather than {uid}; refusing to lock in a directory \
907                 another user controls",
908                dir.display(),
909                metadata.uid()
910            )
911        ));
912    }
913    if metadata.permissions().mode() & 0o077 != 0 {
914        return Err(io::Error::new(
915            io::ErrorKind::PermissionDenied,
916            format!(
917                "{} is mode {:o}; refusing to lock in a directory others can write to",
918                dir.display(),
919                metadata.permissions().mode() & 0o777
920            )
921        ));
922    }
923
924    Ok(())
925}
926
927#[cfg(test)]
928mod fleet_lock_tests {
929    use super::{join_fleet_membership, try_lock_fleet_exclusive};
930
931    /// A member alive means the fleet cannot be cleared, and there is no
932    /// flag that says otherwise; the member going away is what opens it.
933    #[test]
934    fn a_member_holds_the_fleet_against_exclusive_takers() {
935        let fleet = format!("fl{:x}", std::process::id());
936        let uid = unsafe { libc::geteuid() };
937
938        let member = join_fleet_membership(&fleet).expect("join");
939        assert!(try_lock_fleet_exclusive(&fleet, uid).expect("try").is_none());
940
941        drop(member);
942        let exclusive = try_lock_fleet_exclusive(&fleet, uid).expect("try");
943        assert!(exclusive.is_some());
944        // And a member cannot join while a removal holds the fleet: the
945        // shared lock would block, which is the behaviour, not a test to run.
946        drop(exclusive);
947        let _ = std::fs::remove_file(super::fleet_lock_path_for_uid(&fleet, uid));
948    }
949}