Skip to main content

orbit_core/
shm.rs

1//! POSIX shared-memory helpers — V1 substrate for cross-process rings.
2//!
3//! Wraps `shm_open` / `ftruncate` / `mmap` / `munmap` / `shm_unlink`
4//! into a small, RAII-friendly API. Unix-only; Windows support is
5//! a separate concern (Win32 named file mapping) that can land later.
6//!
7//! ## Naming
8//!
9//! Segments are named `/orbit-{fleet}-{kind}-{uid}` — fleet name from
10//! the embedder, KIND from `OrbitTyped::KIND`, UID from `geteuid()`.
11//! UID-scoping avoids the `/dev/shm` sticky-bit cross-user collision
12//! problem (a stale segment owned by one user blocks another from
13//! `shm_unlink`-ing it on next boot).
14//!
15//! Rings that require a process-recoverable writer lock also open a
16//! companion `orbit-{fleet}-{kind}-{uid}.lock` file. It carries no
17//! ring data or state; it only supplies a regular-file inode for `flock`,
18//! because advisory locking on a POSIX SHM descriptor is not uniformly
19//! supported across the Unix targets Orbit serves. An unlocked stale
20//! companion file is safe to reuse.
21//!
22//! Those files live in a per-uid directory — `$XDG_RUNTIME_DIR/orbit-{uid}`
23//! where the session provides one, `/tmp/orbit-{uid}` otherwise — created
24//! `0700` and checked on every lock. They were once in `/tmp` directly, which
25//! made them squattable: see [`lock_dir`].
26//!
27//! ## Lifetime
28//!
29//! [`ShmRegion`] owns the mapped pointer and unmaps on drop. It does
30//! NOT `shm_unlink` on drop — the segment lives until an explicit
31//! [`ShmRegion::unlink`] call. This matches POSIX convention: a
32//! segment with mapped users is not removed; `shm_unlink` only
33//! prevents *new* opens, the current mapping stays valid until the
34//! last process unmaps.
35
36#![cfg(unix)]
37
38use std::ffi::CString;
39use std::fs::OpenOptions;
40use std::io;
41use std::os::fd::{AsRawFd, FromRawFd, OwnedFd};
42use std::os::unix::fs::OpenOptionsExt;
43use std::path::{Path, PathBuf};
44use std::ptr::NonNull;
45
46/// Namespace used by Orbit POSIX shared-memory objects.
47pub const SHM_NAMESPACE: &str = "orbit";
48
49/// Result of physically validating an existing POSIX SHM object.
50///
51/// This check is deliberately below ring semantics: it verifies that the
52/// named object can be opened and is large enough for the requested mapping,
53/// but it does not inspect an owning data structure's magic, version, or
54/// geometry header.
55#[derive(Clone, Copy, Debug, Eq, PartialEq)]
56pub enum ShmValidation {
57    /// No object currently exists under the requested name.
58    Missing,
59    /// The object exists and can safely back at least the requested mapping.
60    Valid { actual_size: usize },
61}
62
63/// A mapped POSIX SHM region. Drop unmaps; `unlink` removes the
64/// underlying name and any companion lock file (only the *creator*
65/// should call it on shutdown).
66pub struct ShmRegion {
67    name: CString,
68    lock_path: PathBuf,
69    /// Whether this region uses the companion file for ordered writes.
70    ///
71    /// The descriptor itself is deliberately not retained: every critical
72    /// section opens its own file description so a later fork does not inherit
73    /// an idle descriptor that can keep a future `flock` alive.
74    process_lock: bool,
75    ptr: NonNull<u8>,
76    len: usize,
77    /// True when this handle was the one that *created* the segment
78    /// (so it knows to `shm_unlink` if asked). Other attachers see
79    /// `false`.
80    created: bool,
81}
82
83impl ShmRegion {
84    /// Validate an existing shared-memory object without creating, mapping,
85    /// resetting, or unlinking it.
86    ///
87    /// Returns [`ShmValidation::Missing`] when the name does not exist. A
88    /// present object must be at least `minimum_size` bytes; larger objects
89    /// are accepted because some platforms report page-rounded SHM sizes.
90    /// The owning ring or table remains responsible for validating its own
91    /// persisted ABI header after mapping.
92    pub fn validate_existing(name: &str, minimum_size: usize) -> io::Result<ShmValidation> {
93        let cname = CString::new(name)
94            .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
95        let raw_fd = loop {
96            // SAFETY: passing a valid C string and well-known POSIX flags.
97            // macOS `shm_open` rejects `O_CLOEXEC` with EINVAL. The descriptor
98            // is scoped to this validation call and closes before return.
99            let fd = unsafe { libc::shm_open(cname.as_ptr(), libc::O_RDONLY, 0o600) };
100            if fd >= 0 {
101                break fd;
102            }
103            let error = io::Error::last_os_error();
104            if error.kind() == io::ErrorKind::Interrupted {
105                continue;
106            }
107            if error.raw_os_error() == Some(libc::ENOENT) {
108                return Ok(ShmValidation::Missing);
109            }
110            return Err(error);
111        };
112        // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
113        // owned by this scope.
114        let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
115        let actual_size = shm_object_size(&fd, name)?;
116        validate_minimum_size(name, actual_size, minimum_size)?;
117        Ok(ShmValidation::Valid { actual_size })
118    }
119
120    /// Map an existing shared-memory object read-only.
121    ///
122    /// This path never creates, sizes, locks, resets, or unlinks the object.
123    /// It is kept crate-private so callers receive a capability such as a
124    /// read-only ring view rather than a [`ShmRegion`] that also exposes
125    /// lifecycle and writable-pointer operations.
126    pub(crate) fn open_existing_read_only(name: &str, minimum_size: usize) -> io::Result<Self> {
127        let cname = CString::new(name)
128            .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
129        let raw_fd = loop {
130            // SAFETY: passing a valid C string and read-only POSIX flags.
131            // macOS `shm_open` rejects `O_CLOEXEC` with EINVAL. This
132            // descriptor closes immediately after `mmap`, before the view is
133            // returned, so it cannot leak across a later exec.
134            let fd = unsafe { libc::shm_open(cname.as_ptr(), libc::O_RDONLY, 0o600) };
135            if fd >= 0 {
136                break fd;
137            }
138            let error = io::Error::last_os_error();
139            if error.kind() == io::ErrorKind::Interrupted {
140                continue;
141            }
142            return Err(error);
143        };
144        // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
145        // owned by this scope.
146        let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
147        let actual_size = shm_object_size(&fd, name)?;
148        validate_minimum_size(name, actual_size, minimum_size)?;
149
150        // Map the complete object so its persisted header can describe the
151        // geometry without the observer reproducing the producer's layout.
152        // SAFETY: fd is valid, actual_size is positive after minimum
153        // validation, and the mapping is read-only.
154        let ptr = unsafe {
155            libc::mmap(
156                std::ptr::null_mut(),
157                actual_size,
158                libc::PROT_READ,
159                libc::MAP_SHARED,
160                fd.as_raw_fd(),
161                0,
162            )
163        };
164        if ptr == libc::MAP_FAILED {
165            return Err(io::Error::last_os_error());
166        }
167        // SAFETY: mmap returned a non-null pointer (checked above).
168        let ptr = NonNull::new(ptr.cast::<u8>()).expect("mmap returned non-null on success");
169
170        Ok(Self {
171            lock_path: lock_file_path(name),
172            name: cname,
173            process_lock: false,
174            ptr,
175            len: actual_size,
176            created: false,
177        })
178    }
179
180    /// Open or create a shared-memory segment of `size` bytes,
181    /// memory-mapped read/write. Idempotent: if the segment already
182    /// exists with the same name and enough mapped bytes, it is reused
183    /// (`created = false`). First creation does `ftruncate(size)`;
184    /// later opens verify the existing object before mapping it. Some
185    /// platforms report a page-rounded SHM size, so a larger `st_size`
186    /// is valid; the owning data structure must verify its own header.
187    pub fn open_or_create(name: &str, size: usize) -> io::Result<Self> {
188        let (region, initialization_lock) = Self::open_or_create_inner(name, size, false)?;
189        debug_assert!(initialization_lock.is_none());
190        Ok(region)
191    }
192
193    /// Open or create a region while holding its process lock through caller
194    /// initialization. This prevents a peer from observing the interval
195    /// between `shm_open` and the owning data structure's initialized header.
196    pub fn open_or_create_locked(name: &str, size: usize) -> io::Result<(Self, ShmRegionLock)> {
197        let (region, initialization_lock) = Self::open_or_create_inner(name, size, true)?;
198        Ok((
199            region,
200            initialization_lock.expect("locked SHM open must return its initialization lock"),
201        ))
202    }
203
204    fn open_or_create_inner(
205        name: &str,
206        size: usize,
207        process_lock: bool,
208    ) -> io::Result<(Self, Option<ShmRegionLock>)> {
209        let cname = CString::new(name)
210            .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
211        let lock_path = lock_file_path(name);
212        let initialization_lock = if process_lock {
213            Some(lock_path_exclusive(&lock_path)?)
214        } else {
215            None
216        };
217
218        // Try create-exclusive first; if it already exists, open.
219        let (raw_fd, created) = unsafe {
220            // SAFETY: passing a valid C string and well-known POSIX flags.
221            let fd = libc::shm_open(
222                cname.as_ptr(),
223                libc::O_RDWR | libc::O_CREAT | libc::O_EXCL,
224                0o600,
225            );
226            if fd >= 0 {
227                (fd, true)
228            } else {
229                // Could be EEXIST (already created by a peer) or another error.
230                let err = io::Error::last_os_error();
231                if err.raw_os_error() != Some(libc::EEXIST) {
232                    return Err(err);
233                }
234                let fd = libc::shm_open(cname.as_ptr(), libc::O_RDWR, 0o600);
235                if fd < 0 {
236                    return Err(io::Error::last_os_error());
237                }
238                (fd, false)
239            }
240        };
241        // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
242        // owned by this scope.
243        let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
244
245        // Size the segment on first creation.
246        if created {
247            // SAFETY: fd is a valid POSIX fd we just received.
248            let rc = unsafe { libc::ftruncate(fd.as_raw_fd(), size as libc::off_t) };
249            if rc != 0 {
250                let err = io::Error::last_os_error();
251                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
252                return Err(err);
253            }
254        }
255
256        // Never mmap beyond the real SHM object: access past it can raise
257        // SIGBUS. A larger reported size is valid on platforms (notably
258        // macOS) that page-round POSIX SHM objects; callers verify their
259        // own ABI metadata after mapping.
260        let actual_size = match shm_object_size(&fd, name) {
261            Ok(actual_size) => actual_size,
262            Err(error) => {
263                if created {
264                    let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
265                }
266                return Err(error);
267            }
268        };
269        if let Err(error) = validate_minimum_size(name, actual_size, size) {
270            if created {
271                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
272            }
273            return Err(error);
274        }
275
276        // Memory-map the segment.
277        // SAFETY: fd valid, size positive, flags well-known.
278        let ptr = unsafe {
279            libc::mmap(
280                std::ptr::null_mut(),
281                size,
282                libc::PROT_READ | libc::PROT_WRITE,
283                libc::MAP_SHARED,
284                fd.as_raw_fd(),
285                0,
286            )
287        };
288
289        if ptr == libc::MAP_FAILED {
290            let err = io::Error::last_os_error();
291            if created {
292                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
293            }
294            return Err(err);
295        }
296
297        // SAFETY: mmap returned a non-null pointer (we just checked).
298        let ptr = NonNull::new(ptr.cast::<u8>()).expect("mmap returned non-null on success");
299
300        Ok((
301            Self {
302                name: cname,
303                lock_path,
304                process_lock,
305                ptr,
306                len: size,
307                created,
308            },
309            initialization_lock,
310        ))
311    }
312
313    /// Raw mapped pointer to the start of the region.
314    pub fn as_ptr(&self) -> *mut u8 {
315        self.ptr.as_ptr()
316    }
317
318    /// Length of the mapped region (the `size` passed to `open_or_create`).
319    pub fn len(&self) -> usize {
320        self.len
321    }
322
323    pub fn is_empty(&self) -> bool {
324        self.len == 0
325    }
326
327    /// True when this handle was the one that created the segment.
328    /// Useful for picking which process performs first-time
329    /// initialization of the header.
330    pub fn created(&self) -> bool {
331        self.created
332    }
333
334    /// Acquire an exclusive cross-process lock tied to this SHM name.
335    ///
336    /// `flock` ownership is held by the kernel and is released when a process
337    /// exits or the descriptor closes, including abnormal termination.
338    pub fn lock_exclusive(&self) -> io::Result<ShmRegionLock> {
339        if !self.process_lock {
340            return Err(io::Error::new(
341                io::ErrorKind::InvalidInput,
342                "SHM region was opened without a process lock",
343            ));
344        }
345        lock_path_exclusive(&self.lock_path)
346    }
347
348    #[cfg(test)]
349    pub(crate) fn try_lock_exclusive(&self) -> io::Result<ShmRegionLock> {
350        if !self.process_lock {
351            return Err(io::Error::new(
352                io::ErrorKind::InvalidInput,
353                "SHM region was opened without a process lock",
354            ));
355        }
356        let lock_fd = open_lock_file(&self.lock_path)?;
357        let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) };
358        if rc == 0 {
359            Ok(ShmRegionLock { lock_fd })
360        } else {
361            Err(io::Error::last_os_error())
362        }
363    }
364
365    /// Remove the underlying segment name. Existing mappings stay
366    /// valid until each process drops its `ShmRegion`. Use only on
367    /// shutdown / fleet teardown by the process that owns lifecycle.
368    pub fn unlink(&self) -> io::Result<()> {
369        // SAFETY: name is a valid C string.
370        let rc = unsafe { libc::shm_unlink(self.name.as_ptr()) };
371        let shm_error = if rc != 0 {
372            let err = io::Error::last_os_error();
373            // ENOENT is fine — segment was already unlinked.
374            if err.raw_os_error() == Some(libc::ENOENT) {
375                None
376            } else {
377                Some(err)
378            }
379        } else {
380            None
381        };
382        let lock_error = match std::fs::remove_file(&self.lock_path) {
383            Ok(()) => None,
384            Err(error) if error.kind() == io::ErrorKind::NotFound => None,
385            Err(error) => Some(error),
386        };
387        if let Some(error) = shm_error.or(lock_error) {
388            return Err(error);
389        }
390        Ok(())
391    }
392}
393
394fn shm_object_size(fd: &OwnedFd, name: &str) -> io::Result<usize> {
395    let mut stat = std::mem::MaybeUninit::<libc::stat>::uninit();
396    // SAFETY: `fd` is valid and `stat` points to writable storage.
397    let stat_rc = unsafe { libc::fstat(fd.as_raw_fd(), stat.as_mut_ptr()) };
398    if stat_rc != 0 {
399        return Err(io::Error::last_os_error());
400    }
401    // SAFETY: `fstat` succeeded and initialized the structure.
402    let actual_size = unsafe { stat.assume_init() }.st_size;
403    usize::try_from(actual_size).map_err(|_| {
404        io::Error::new(
405            io::ErrorKind::InvalidData,
406            format!("SHM segment {name} reported invalid size {actual_size}"),
407        )
408    })
409}
410
411fn validate_minimum_size(name: &str, actual_size: usize, minimum_size: usize) -> io::Result<()> {
412    if actual_size < minimum_size {
413        return Err(io::Error::new(
414            io::ErrorKind::InvalidData,
415            format!(
416                "SHM segment {name} size {actual_size} is smaller than requested mapping {minimum_size}"
417            ),
418        ));
419    }
420    Ok(())
421}
422
423/// RAII guard for a [`ShmRegion`]'s process-recoverable exclusive lock.
424///
425/// Semantic crates use this when a current-state transition must be atomic
426/// across fleet processes. Dropping the guard releases the kernel lock.
427pub struct ShmRegionLock {
428    lock_fd: OwnedFd,
429}
430
431impl Drop for ShmRegionLock {
432    fn drop(&mut self) {
433        let _ = unsafe { libc::flock(self.lock_fd.as_raw_fd(), libc::LOCK_UN) };
434    }
435}
436
437fn lock_path_exclusive(lock_path: &Path) -> io::Result<ShmRegionLock> {
438    lock_fd_exclusive(open_lock_file(lock_path)?)
439}
440
441fn open_lock_file(lock_path: &Path) -> io::Result<OwnedFd> {
442    use std::os::unix::fs::MetadataExt;
443
444    if let Some(dir) = lock_path.parent() {
445        ensure_lock_dir(dir)?;
446    }
447
448    let file = OpenOptions::new()
449        .read(true)
450        .write(true)
451        .create(true)
452        .mode(0o600)
453        .custom_flags(libc::O_CLOEXEC | libc::O_NOFOLLOW)
454        .open(lock_path)?;
455
456    // The directory check makes this unreachable, which is the reason to make
457    // it anyway: it turns a property inferred from the directory's mode into
458    // one this function establishes about the descriptor it is about to lock.
459    let uid = unsafe { libc::geteuid() };
460    let owner = file.metadata()?.uid();
461    if owner != uid {
462        return Err(io::Error::new(
463            io::ErrorKind::PermissionDenied,
464            format!(
465                "{} is owned by uid {owner} rather than {uid}",
466                lock_path.display()
467            ),
468        ));
469    }
470
471    Ok(file.into())
472}
473
474fn lock_fd_exclusive(lock_fd: OwnedFd) -> io::Result<ShmRegionLock> {
475    loop {
476        let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX) };
477        if rc == 0 {
478            return Ok(ShmRegionLock { lock_fd });
479        }
480        let error = io::Error::last_os_error();
481        if error.kind() != io::ErrorKind::Interrupted {
482            return Err(error);
483        }
484    }
485}
486
487impl Drop for ShmRegion {
488    fn drop(&mut self) {
489        // SAFETY: ptr came from mmap of `self.len` bytes; munmap is the inverse.
490        unsafe {
491            libc::munmap(self.ptr.as_ptr().cast(), self.len);
492        }
493    }
494}
495
496// SAFETY: the underlying region is shared memory and synchronization
497// happens at the slot level (atomic seq counters); the handle itself
498// is just a pointer + length, safe to send/share.
499unsafe impl Send for ShmRegion {}
500unsafe impl Sync for ShmRegion {}
501
502/// Build the conventional name for an Orbit ring segment.
503pub fn ring_segment_name(fleet_name: &str, kind: u8) -> String {
504    // SAFETY: `geteuid` always returns a value; no error path.
505    let uid = unsafe { libc::geteuid() };
506    ring_segment_name_for_uid(fleet_name, kind, uid)
507}
508
509/// Build the conventional name for an Orbit ring segment owned by `uid`.
510///
511/// This form is intended for inspection and lifecycle tools that need to
512/// address a user other than their own effective uid.
513pub fn ring_segment_name_for_uid(fleet_name: &str, kind: u8, uid: u32) -> String {
514    format!("/{SHM_NAMESPACE}-{fleet_name}-{kind}-{uid}")
515}
516
517/// Where a fleet's members hold their presence: `<lock dir>/orbit-<fleet>.fleet`.
518pub fn fleet_lock_path(fleet_name: &str) -> PathBuf {
519    lock_dir().join(format!("{SHM_NAMESPACE}-{fleet_name}.fleet"))
520}
521
522/// The same for another user's fleet, for lifecycle tools that address a uid
523/// other than their own.
524pub fn fleet_lock_path_for_uid(fleet_name: &str, uid: u32) -> PathBuf {
525    lock_dir_for_uid(uid).join(format!("{SHM_NAMESPACE}-{fleet_name}.fleet"))
526}
527
528/// A process's membership in a fleet: a shared `flock` on the fleet's lock
529/// file, held for as long as this lives and released by the kernel when the
530/// process dies, however it dies. It is what [`try_lock_fleet_exclusive`]
531/// contends with, so a tool that removes the fleet's segments cannot do so
532/// while any member is alive.
533pub struct FleetMembership {
534    lock_fd: OwnedFd,
535}
536
537impl Drop for FleetMembership {
538    fn drop(&mut self) {
539        let _ = unsafe { libc::flock(self.lock_fd.as_raw_fd(), libc::LOCK_UN) };
540    }
541}
542
543/// Join the fleet's membership. Waits for a lifecycle tool that holds the
544/// exclusive lock at that moment; a clear in progress finishes first.
545pub fn join_fleet_membership(fleet_name: &str) -> io::Result<FleetMembership> {
546    let lock_fd = open_lock_file(&fleet_lock_path(fleet_name))?;
547    // SAFETY: `lock_fd` is an open descriptor owned by this call.
548    let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_SH) };
549    if rc != 0 {
550        return Err(io::Error::last_os_error());
551    }
552    Ok(FleetMembership { lock_fd })
553}
554
555/// Exclusive hold on a fleet, for the tool that removes its segments.
556///
557/// `Ok(None)` means a member is alive and the fleet must not be touched;
558/// there is deliberately no way to force past it. `Ok(Some(_))` keeps the
559/// fleet closed to new members until the guard is dropped, so a removal
560/// cannot interleave with a start. Never waits.
561pub fn try_lock_fleet_exclusive(fleet_name: &str, uid: u32) -> io::Result<Option<ShmRegionLock>> {
562    // Creating the file when no member ever joined is right: the guard then
563    // keeps a first member from starting in the middle of a removal.
564    let lock_fd = open_lock_file(&fleet_lock_path_for_uid(fleet_name, uid))?;
565    // SAFETY: `lock_fd` is an open descriptor owned by this call.
566    let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) };
567    if rc == 0 {
568        return Ok(Some(ShmRegionLock { lock_fd }));
569    }
570    let error = io::Error::last_os_error();
571    if error.kind() == io::ErrorKind::WouldBlock {
572        return Ok(None);
573    }
574    Err(error)
575}
576
577fn lock_file_path(shm_name: &str) -> PathBuf {
578    lock_dir().join(format!("{}.lock", shm_name.trim_start_matches('/')))
579}
580
581/// Where the companion lock files live.
582///
583/// They used to live in `/tmp` directly, as
584/// `/tmp/orbit-{fleet}-{kind}-{uid}.lock`, opened `O_CREAT` without `O_EXCL`
585/// and without asking who owned what the open found. `/tmp` is world-writable,
586/// so any local user could create that file first and then hold `LOCK_EX` on it
587/// for as long as they liked: every process in the fleet would sit in `flock` —
588/// not fail, block — waiting for a lock it was never going to get. The SHM
589/// segment name is uid-scoped and so cannot be squatted this way; the lock path
590/// was not. `O_NOFOLLOW` prevented the symlink version of the trick and nothing
591/// else.
592///
593/// They now live in a per-uid directory created `0700`, which a user who is not
594/// us cannot put a file into. What such a user can still do is create the
595/// directory first, so its owner and mode are checked on every open rather than
596/// assumed from having created it: a directory that is not ours, or not
597/// private, fails the open with the path in the message instead of parking the
598/// process on a lock.
599///
600/// `XDG_RUNTIME_DIR` is preferred where the session provides one, because it is
601/// already per-user and `0700` and so is not inside a world-writable directory
602/// at all. macOS has no such variable but gives each user a private `TMPDIR`;
603/// `/tmp` is the fallback, with the checks above carrying the weight.
604fn lock_dir() -> PathBuf {
605    let uid = unsafe { libc::geteuid() };
606    lock_dir_for_uid(uid)
607}
608
609fn lock_dir_for_uid(uid: u32) -> PathBuf {
610    let base = std::env::var_os("XDG_RUNTIME_DIR")
611        .map(PathBuf::from)
612        .filter(|dir| dir.is_absolute())
613        .unwrap_or_else(|| PathBuf::from("/tmp"));
614
615    base.join(format!("{SHM_NAMESPACE}-{uid}"))
616}
617
618/// Creates the lock directory if it is missing and refuses it if it is not
619/// ours. Called when a lock is actually taken rather than when a region is
620/// opened: a region that never locks has nothing to squat, and failing its open
621/// on a directory it does not use would hand an attacker a wider outage than
622/// the one being closed.
623fn ensure_lock_dir(dir: &Path) -> io::Result<()> {
624    use std::os::unix::fs::DirBuilderExt;
625
626    let uid = unsafe { libc::geteuid() };
627    match std::fs::DirBuilder::new().mode(0o700).create(dir) {
628        Ok(()) => {}
629        Err(error) if error.kind() == io::ErrorKind::AlreadyExists => {}
630        Err(error) => return Err(error),
631    }
632
633    ensure_private_dir(dir, uid)
634}
635
636/// Refuses a lock directory that someone else could write to.
637///
638/// `symlink_metadata` rather than `metadata`: a symlink pointing at a directory
639/// we do own would otherwise pass while the lock files landed somewhere the
640/// attacker chose.
641fn ensure_private_dir(dir: &Path, uid: u32) -> io::Result<()> {
642    use std::os::unix::fs::MetadataExt;
643    use std::os::unix::fs::PermissionsExt;
644
645    let metadata = std::fs::symlink_metadata(dir)?;
646    if !metadata.is_dir() {
647        return Err(io::Error::new(
648            io::ErrorKind::PermissionDenied,
649            format!("{} is not a directory", dir.display()),
650        ));
651    }
652    if metadata.uid() != uid {
653        return Err(io::Error::new(
654            io::ErrorKind::PermissionDenied,
655            format!(
656                "{} is owned by uid {} rather than {uid}; refusing to lock in a directory \
657                 another user controls",
658                dir.display(),
659                metadata.uid()
660            ),
661        ));
662    }
663    if metadata.permissions().mode() & 0o077 != 0 {
664        return Err(io::Error::new(
665            io::ErrorKind::PermissionDenied,
666            format!(
667                "{} is mode {:o}; refusing to lock in a directory others can write to",
668                dir.display(),
669                metadata.permissions().mode() & 0o777
670            ),
671        ));
672    }
673
674    Ok(())
675}
676
677#[cfg(test)]
678mod fleet_lock_tests {
679    use super::{join_fleet_membership, try_lock_fleet_exclusive};
680
681    /// A member alive means the fleet cannot be cleared, and there is no
682    /// flag that says otherwise; the member going away is what opens it.
683    #[test]
684    fn a_member_holds_the_fleet_against_exclusive_takers() {
685        let fleet = format!("fl{:x}", std::process::id());
686        let uid = unsafe { libc::geteuid() };
687
688        let member = join_fleet_membership(&fleet).expect("join");
689        assert!(
690            try_lock_fleet_exclusive(&fleet, uid)
691                .expect("try")
692                .is_none()
693        );
694
695        drop(member);
696        let exclusive = try_lock_fleet_exclusive(&fleet, uid).expect("try");
697        assert!(exclusive.is_some());
698        // And a member cannot join while a removal holds the fleet: the
699        // shared lock would block, which is the behaviour, not a test to run.
700        drop(exclusive);
701        let _ = std::fs::remove_file(super::fleet_lock_path_for_uid(&fleet, uid));
702    }
703}