Skip to main content

orbit_core/
shm.rs

1//! POSIX shared-memory helpers — V1 substrate for cross-process rings.
2//!
3//! Wraps `shm_open` / `ftruncate` / `mmap` / `munmap` / `shm_unlink`
4//! into a small, RAII-friendly API. Unix-only; Windows support is
5//! a separate concern (Win32 named file mapping) that can land later.
6//!
7//! ## Naming
8//!
9//! Segments are named `/orbit-{fleet}-{kind}-{uid}` — fleet name from
10//! the embedder, KIND from `OrbitTyped::KIND`, UID from `geteuid()`.
11//! UID-scoping avoids the `/dev/shm` sticky-bit cross-user collision
12//! problem (a stale segment owned by one user blocks another from
13//! `shm_unlink`-ing it on next boot).
14//!
15//! Rings that require a process-recoverable writer lock also open a
16//! companion `orbit-{fleet}-{kind}-{uid}.lock` file. It carries no
17//! ring data or state; it only supplies a regular-file inode for `flock`,
18//! because advisory locking on a POSIX SHM descriptor is not uniformly
19//! supported across the Unix targets Orbit serves. An unlocked stale
20//! companion file is safe to reuse.
21//!
22//! Those files live in a per-uid directory — `$XDG_RUNTIME_DIR/orbit-{uid}`
23//! where the session provides one, `/tmp/orbit-{uid}` otherwise — created
24//! `0700` and checked on every lock. They were once in `/tmp` directly, which
25//! made them squattable: see [`lock_dir`].
26//!
27//! ## Lifetime
28//!
29//! [`ShmRegion`] owns the mapped pointer and unmaps on drop. It does
30//! NOT `shm_unlink` on drop — the segment lives until an explicit
31//! [`ShmRegion::unlink`] call. This matches POSIX convention: a
32//! segment with mapped users is not removed; `shm_unlink` only
33//! prevents *new* opens, the current mapping stays valid until the
34//! last process unmaps.
35
36#![cfg(unix)]
37
38use std::ffi::CString;
39use std::fs::OpenOptions;
40use std::io;
41use std::os::fd::{AsRawFd, FromRawFd, OwnedFd};
42use std::os::unix::fs::OpenOptionsExt;
43use std::path::{Path, PathBuf};
44use std::ptr::NonNull;
45
46/// Namespace used by Orbit POSIX shared-memory objects.
47pub const SHM_NAMESPACE: &str = "orbit";
48
49/// Result of physically validating an existing POSIX SHM object.
50///
51/// This check is deliberately below ring semantics: it verifies that the
52/// named object can be opened and is large enough for the requested mapping,
53/// but it does not inspect an owning data structure's magic, version, or
54/// geometry header.
55#[derive(Clone, Copy, Debug, Eq, PartialEq)]
56pub enum ShmValidation {
57    /// No object currently exists under the requested name.
58    Missing,
59    /// The object exists and can safely back at least the requested mapping.
60    Valid { actual_size: usize }
61}
62
63/// A mapped POSIX SHM region. Drop unmaps; `unlink` removes the
64/// underlying name and any companion lock file (only the *creator*
65/// should call it on shutdown).
66pub struct ShmRegion {
67    name: CString,
68    lock_path: PathBuf,
69    /// Whether this region uses the companion file for ordered writes.
70    ///
71    /// The descriptor itself is deliberately not retained: every critical
72    /// section opens its own file description so a later fork does not inherit
73    /// an idle descriptor that can keep a future `flock` alive.
74    process_lock: bool,
75    ptr: NonNull<u8>,
76    len: usize,
77    /// True when this handle was the one that *created* the segment
78    /// (so it knows to `shm_unlink` if asked). Other attachers see
79    /// `false`.
80    created: bool
81}
82
83impl ShmRegion {
84    /// Validate an existing shared-memory object without creating, mapping,
85    /// resetting, or unlinking it.
86    ///
87    /// Returns [`ShmValidation::Missing`] when the name does not exist. A
88    /// present object must be at least `minimum_size` bytes; larger objects
89    /// are accepted because some platforms report page-rounded SHM sizes.
90    /// The owning ring or table remains responsible for validating its own
91    /// persisted ABI header after mapping.
92    pub fn validate_existing(
93        name: &str,
94        minimum_size: usize
95    ) -> io::Result<ShmValidation> {
96        let cname = CString::new(name)
97            .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
98        let raw_fd = loop {
99            // SAFETY: passing a valid C string and well-known POSIX flags.
100            // macOS `shm_open` rejects `O_CLOEXEC` with EINVAL. The descriptor
101            // is scoped to this validation call and closes before return.
102            let fd = unsafe { libc::shm_open(cname.as_ptr(), libc::O_RDONLY, 0o600) };
103            if fd >= 0 {
104                break fd;
105            }
106            let error = io::Error::last_os_error();
107            if error.kind() == io::ErrorKind::Interrupted {
108                continue;
109            }
110            if error.raw_os_error() == Some(libc::ENOENT) {
111                return Ok(ShmValidation::Missing);
112            }
113            return Err(error);
114        };
115        // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
116        // owned by this scope.
117        let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
118        let actual_size = shm_object_size(&fd, name)?;
119        validate_minimum_size(name, actual_size, minimum_size)?;
120        Ok(ShmValidation::Valid { actual_size })
121    }
122
123    /// Map an existing shared-memory object read-only.
124    ///
125    /// This path never creates, sizes, locks, resets, or unlinks the object.
126    /// It is kept crate-private so callers receive a capability such as a
127    /// read-only ring view rather than a [`ShmRegion`] that also exposes
128    /// lifecycle and writable-pointer operations.
129    pub(crate) fn open_existing_read_only(
130        name: &str,
131        minimum_size: usize
132    ) -> io::Result<Self> {
133        let cname = CString::new(name)
134            .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
135        let raw_fd = loop {
136            // SAFETY: passing a valid C string and read-only POSIX flags.
137            // macOS `shm_open` rejects `O_CLOEXEC` with EINVAL. This
138            // descriptor closes immediately after `mmap`, before the view is
139            // returned, so it cannot leak across a later exec.
140            let fd = unsafe { libc::shm_open(cname.as_ptr(), libc::O_RDONLY, 0o600) };
141            if fd >= 0 {
142                break fd;
143            }
144            let error = io::Error::last_os_error();
145            if error.kind() == io::ErrorKind::Interrupted {
146                continue;
147            }
148            return Err(error);
149        };
150        // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
151        // owned by this scope.
152        let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
153        let actual_size = shm_object_size(&fd, name)?;
154        validate_minimum_size(name, actual_size, minimum_size)?;
155
156        // Map the complete object so its persisted header can describe the
157        // geometry without the observer reproducing the producer's layout.
158        // SAFETY: fd is valid, actual_size is positive after minimum
159        // validation, and the mapping is read-only.
160        let ptr = unsafe {
161            libc::mmap(
162                std::ptr::null_mut(),
163                actual_size,
164                libc::PROT_READ,
165                libc::MAP_SHARED,
166                fd.as_raw_fd(),
167                0
168            )
169        };
170        if ptr == libc::MAP_FAILED {
171            return Err(io::Error::last_os_error());
172        }
173        // SAFETY: mmap returned a non-null pointer (checked above).
174        let ptr = NonNull::new(ptr.cast::<u8>()).expect("mmap returned non-null on success");
175
176        Ok(Self {
177            lock_path: lock_file_path(name),
178            name: cname,
179            process_lock: false,
180            ptr,
181            len: actual_size,
182            created: false
183        })
184    }
185
186    /// Open or create a shared-memory segment of `size` bytes,
187    /// memory-mapped read/write. Idempotent: if the segment already
188    /// exists with the same name and enough mapped bytes, it is reused
189    /// (`created = false`). First creation does `ftruncate(size)`;
190    /// later opens verify the existing object before mapping it. Some
191    /// platforms report a page-rounded SHM size, so a larger `st_size`
192    /// is valid; the owning data structure must verify its own header.
193    pub fn open_or_create(
194        name: &str,
195        size: usize
196    ) -> io::Result<Self> {
197        let (region, initialization_lock) = Self::open_or_create_inner(name, size, false)?;
198        debug_assert!(initialization_lock.is_none());
199        Ok(region)
200    }
201
202    /// Open or create a region while holding its process lock through caller
203    /// initialization. This prevents a peer from observing the interval
204    /// between `shm_open` and the owning data structure's initialized header.
205    pub fn open_or_create_locked(
206        name: &str,
207        size: usize
208    ) -> io::Result<(Self, ShmRegionLock)> {
209        let (region, initialization_lock) = Self::open_or_create_inner(name, size, true)?;
210        Ok((
211            region,
212            initialization_lock.expect("locked SHM open must return its initialization lock")
213        ))
214    }
215
216    fn open_or_create_inner(
217        name: &str,
218        size: usize,
219        process_lock: bool
220    ) -> io::Result<(Self, Option<ShmRegionLock>)> {
221        let cname = CString::new(name)
222            .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
223        let lock_path = lock_file_path(name);
224        let initialization_lock =
225            if process_lock { Some(lock_path_exclusive(&lock_path)?) } else { None };
226
227        // Try create-exclusive first; if it already exists, open.
228        let (raw_fd, created) = unsafe {
229            // SAFETY: passing a valid C string and well-known POSIX flags.
230            let fd =
231                libc::shm_open(cname.as_ptr(), libc::O_RDWR | libc::O_CREAT | libc::O_EXCL, 0o600);
232            if fd >= 0 {
233                (fd, true)
234            } else {
235                // Could be EEXIST (already created by a peer) or another error.
236                let err = io::Error::last_os_error();
237                if err.raw_os_error() != Some(libc::EEXIST) {
238                    return Err(err);
239                }
240                let fd = libc::shm_open(cname.as_ptr(), libc::O_RDWR, 0o600);
241                if fd < 0 {
242                    return Err(io::Error::last_os_error());
243                }
244                (fd, false)
245            }
246        };
247        // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
248        // owned by this scope.
249        let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
250
251        // Size the segment on first creation.
252        if created {
253            // SAFETY: fd is a valid POSIX fd we just received.
254            let rc = unsafe { libc::ftruncate(fd.as_raw_fd(), size as libc::off_t) };
255            if rc != 0 {
256                let err = io::Error::last_os_error();
257                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
258                return Err(err);
259            }
260        }
261
262        // Never mmap beyond the real SHM object: access past it can raise
263        // SIGBUS. A larger reported size is valid on platforms (notably
264        // macOS) that page-round POSIX SHM objects; callers verify their
265        // own ABI metadata after mapping.
266        let actual_size = match shm_object_size(&fd, name) {
267            Ok(actual_size) => actual_size,
268            Err(error) => {
269                if created {
270                    let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
271                }
272                return Err(error);
273            }
274        };
275        if let Err(error) = validate_minimum_size(name, actual_size, size) {
276            if created {
277                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
278            }
279            return Err(error);
280        }
281
282        // Memory-map the segment.
283        // SAFETY: fd valid, size positive, flags well-known.
284        let ptr = unsafe {
285            libc::mmap(
286                std::ptr::null_mut(),
287                size,
288                libc::PROT_READ | libc::PROT_WRITE,
289                libc::MAP_SHARED,
290                fd.as_raw_fd(),
291                0
292            )
293        };
294
295        if ptr == libc::MAP_FAILED {
296            let err = io::Error::last_os_error();
297            if created {
298                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
299            }
300            return Err(err);
301        }
302
303        // SAFETY: mmap returned a non-null pointer (we just checked).
304        let ptr = NonNull::new(ptr.cast::<u8>()).expect("mmap returned non-null on success");
305
306        Ok((
307            Self { name: cname, lock_path, process_lock, ptr, len: size, created },
308            initialization_lock
309        ))
310    }
311
312    /// Raw mapped pointer to the start of the region.
313    pub fn as_ptr(&self) -> *mut u8 {
314        self.ptr.as_ptr()
315    }
316
317    /// Length of the mapped region (the `size` passed to `open_or_create`).
318    pub fn len(&self) -> usize {
319        self.len
320    }
321
322    pub fn is_empty(&self) -> bool {
323        self.len == 0
324    }
325
326    /// True when this handle was the one that created the segment.
327    /// Useful for picking which process performs first-time
328    /// initialization of the header.
329    pub fn created(&self) -> bool {
330        self.created
331    }
332
333    /// Acquire an exclusive cross-process lock tied to this SHM name.
334    ///
335    /// `flock` ownership is held by the kernel and is released when a process
336    /// exits or the descriptor closes, including abnormal termination.
337    pub fn lock_exclusive(&self) -> io::Result<ShmRegionLock> {
338        if !self.process_lock {
339            return Err(io::Error::new(
340                io::ErrorKind::InvalidInput,
341                "SHM region was opened without a process lock"
342            ));
343        }
344        lock_path_exclusive(&self.lock_path)
345    }
346
347    #[cfg(test)]
348    pub(crate) fn try_lock_exclusive(&self) -> io::Result<ShmRegionLock> {
349        if !self.process_lock {
350            return Err(io::Error::new(
351                io::ErrorKind::InvalidInput,
352                "SHM region was opened without a process lock"
353            ));
354        }
355        let lock_fd = open_lock_file(&self.lock_path)?;
356        let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) };
357        if rc == 0 { Ok(ShmRegionLock { lock_fd }) } else { Err(io::Error::last_os_error()) }
358    }
359
360    /// Remove the underlying segment name. Existing mappings stay
361    /// valid until each process drops its `ShmRegion`. Use only on
362    /// shutdown / fleet teardown by the process that owns lifecycle.
363    pub fn unlink(&self) -> io::Result<()> {
364        // SAFETY: name is a valid C string.
365        let rc = unsafe { libc::shm_unlink(self.name.as_ptr()) };
366        let shm_error = if rc != 0 {
367            let err = io::Error::last_os_error();
368            // ENOENT is fine — segment was already unlinked.
369            if err.raw_os_error() == Some(libc::ENOENT) { None } else { Some(err) }
370        } else {
371            None
372        };
373        let mut lock_error = match std::fs::remove_file(&self.lock_path) {
374            Ok(()) => None,
375            Err(error) if error.kind() == io::ErrorKind::NotFound => None,
376            Err(error) => Some(error)
377        };
378        let name = self.name.to_string_lossy();
379        // SAFETY: a plain syscall with no arguments.
380        let uid = unsafe { libc::geteuid() };
381        match companion_lock_files(&name, uid) {
382            Ok(files) => {
383                for file in files {
384                    if let Err(error) = std::fs::remove_file(&file)
385                        && error.kind() != io::ErrorKind::NotFound
386                    {
387                        lock_error.get_or_insert(error);
388                    }
389                }
390            }
391            Err(error) => {
392                lock_error.get_or_insert(error);
393            }
394        }
395        if let Some(error) = shm_error.or(lock_error) {
396            return Err(error);
397        }
398        Ok(())
399    }
400}
401
402fn shm_object_size(
403    fd: &OwnedFd,
404    name: &str
405) -> io::Result<usize> {
406    let mut stat = std::mem::MaybeUninit::<libc::stat>::uninit();
407    // SAFETY: `fd` is valid and `stat` points to writable storage.
408    let stat_rc = unsafe { libc::fstat(fd.as_raw_fd(), stat.as_mut_ptr()) };
409    if stat_rc != 0 {
410        return Err(io::Error::last_os_error());
411    }
412    // SAFETY: `fstat` succeeded and initialized the structure.
413    let actual_size = unsafe { stat.assume_init() }.st_size;
414    usize::try_from(actual_size).map_err(|_| {
415        io::Error::new(
416            io::ErrorKind::InvalidData,
417            format!("SHM segment {name} reported invalid size {actual_size}")
418        )
419    })
420}
421
422fn validate_minimum_size(
423    name: &str,
424    actual_size: usize,
425    minimum_size: usize
426) -> io::Result<()> {
427    if actual_size < minimum_size {
428        return Err(io::Error::new(
429            io::ErrorKind::InvalidData,
430            format!(
431                "SHM segment {name} size {actual_size} is smaller than requested mapping {minimum_size}"
432            )
433        ));
434    }
435    Ok(())
436}
437
438/// RAII guard for a [`ShmRegion`]'s process-recoverable exclusive lock.
439///
440/// Semantic crates use this when a current-state transition must be atomic
441/// across fleet processes. Dropping the guard releases the kernel lock.
442pub struct ShmRegionLock {
443    lock_fd: OwnedFd
444}
445
446impl Drop for ShmRegionLock {
447    fn drop(&mut self) {
448        let _ = unsafe { libc::flock(self.lock_fd.as_raw_fd(), libc::LOCK_UN) };
449    }
450}
451
452fn lock_path_exclusive(lock_path: &Path) -> io::Result<ShmRegionLock> {
453    lock_fd_exclusive(open_lock_file(lock_path)?)
454}
455
456fn open_lock_file(lock_path: &Path) -> io::Result<OwnedFd> {
457    use std::os::unix::fs::MetadataExt;
458
459    if let Some(dir) = lock_path.parent() {
460        ensure_lock_dir(dir)?;
461    }
462
463    let file = OpenOptions::new()
464        .read(true)
465        .write(true)
466        .create(true)
467        .mode(0o600)
468        .custom_flags(libc::O_CLOEXEC | libc::O_NOFOLLOW)
469        .open(lock_path)?;
470
471    // The directory check makes this unreachable, which is the reason to make
472    // it anyway: it turns a property inferred from the directory's mode into
473    // one this function establishes about the descriptor it is about to lock.
474    let uid = unsafe { libc::geteuid() };
475    let owner = file.metadata()?.uid();
476    if owner != uid {
477        return Err(io::Error::new(
478            io::ErrorKind::PermissionDenied,
479            format!("{} is owned by uid {owner} rather than {uid}", lock_path.display())
480        ));
481    }
482
483    Ok(file.into())
484}
485
486fn lock_fd_exclusive(lock_fd: OwnedFd) -> io::Result<ShmRegionLock> {
487    loop {
488        let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX) };
489        if rc == 0 {
490            return Ok(ShmRegionLock { lock_fd });
491        }
492        let error = io::Error::last_os_error();
493        if error.kind() != io::ErrorKind::Interrupted {
494            return Err(error);
495        }
496    }
497}
498
499impl Drop for ShmRegion {
500    fn drop(&mut self) {
501        // SAFETY: ptr came from mmap of `self.len` bytes; munmap is the inverse.
502        unsafe {
503            libc::munmap(self.ptr.as_ptr().cast(), self.len);
504        }
505    }
506}
507
508// SAFETY: the underlying region is shared memory and synchronization
509// happens at the slot level (atomic seq counters); the handle itself
510// is just a pointer + length, safe to send/share.
511unsafe impl Send for ShmRegion {}
512unsafe impl Sync for ShmRegion {}
513
514/// Build the conventional name for an Orbit ring segment.
515pub fn ring_segment_name(
516    fleet_name: &str,
517    kind: u8
518) -> String {
519    // SAFETY: `geteuid` always returns a value; no error path.
520    let uid = unsafe { libc::geteuid() };
521    ring_segment_name_for_uid(fleet_name, kind, uid)
522}
523
524/// Build the conventional name for an Orbit ring segment owned by `uid`.
525///
526/// This form is intended for inspection and lifecycle tools that need to
527/// address a user other than their own effective uid.
528pub fn ring_segment_name_for_uid(
529    fleet_name: &str,
530    kind: u8,
531    uid: u32
532) -> String {
533    format!("/{SHM_NAMESPACE}-{fleet_name}-{kind}-{uid}")
534}
535
536/// Where a fleet's members hold their presence: `<lock dir>/orbit-<fleet>.fleet`.
537pub fn fleet_lock_path(fleet_name: &str) -> PathBuf {
538    lock_dir().join(format!("{SHM_NAMESPACE}-{fleet_name}.fleet"))
539}
540
541/// The same for another user's fleet, for lifecycle tools that address a uid
542/// other than their own.
543pub fn fleet_lock_path_for_uid(
544    fleet_name: &str,
545    uid: u32
546) -> PathBuf {
547    lock_dir_for_uid(uid).join(format!("{SHM_NAMESPACE}-{fleet_name}.fleet"))
548}
549
550/// A process's membership in a fleet: a shared `flock` on the fleet's lock
551/// file, held for as long as this lives and released by the kernel when the
552/// process dies, however it dies. It is what [`try_lock_fleet_exclusive`]
553/// contends with, so a tool that removes the fleet's segments cannot do so
554/// while any member is alive.
555pub struct FleetMembership {
556    lock_fd: OwnedFd
557}
558
559impl Drop for FleetMembership {
560    fn drop(&mut self) {
561        let _ = unsafe { libc::flock(self.lock_fd.as_raw_fd(), libc::LOCK_UN) };
562    }
563}
564
565/// Join the fleet's membership. Waits for a lifecycle tool that holds the
566/// exclusive lock at that moment; a clear in progress finishes first.
567pub fn join_fleet_membership(fleet_name: &str) -> io::Result<FleetMembership> {
568    let lock_fd = open_lock_file(&fleet_lock_path(fleet_name))?;
569    // SAFETY: `lock_fd` is an open descriptor owned by this call.
570    let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_SH) };
571    if rc != 0 {
572        return Err(io::Error::last_os_error());
573    }
574    Ok(FleetMembership { lock_fd })
575}
576
577/// Exclusive hold on a fleet, for the tool that removes its segments.
578///
579/// `Ok(None)` means a member is alive and the fleet must not be touched;
580/// there is deliberately no way to force past it. `Ok(Some(_))` keeps the
581/// fleet closed to new members until the guard is dropped, so a removal
582/// cannot interleave with a start. Never waits.
583pub fn try_lock_fleet_exclusive(
584    fleet_name: &str,
585    uid: u32
586) -> io::Result<Option<ShmRegionLock>> {
587    // Creating the file when no member ever joined is right: the guard then
588    // keeps a first member from starting in the middle of a removal.
589    let lock_fd = open_lock_file(&fleet_lock_path_for_uid(fleet_name, uid))?;
590    // SAFETY: `lock_fd` is an open descriptor owned by this call.
591    let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) };
592    if rc == 0 {
593        return Ok(Some(ShmRegionLock { lock_fd }));
594    }
595    let error = io::Error::last_os_error();
596    if error.kind() == io::ErrorKind::WouldBlock {
597        return Ok(None);
598    }
599    Err(error)
600}
601
602/// One process's hold on one lane of a segment: an exclusive `flock` on the
603/// lane's own lock file, kept for as long as this lives and released by the
604/// kernel when the process dies, however it dies.
605///
606/// It is how a lane's owner is known to be gone without a timeout and without
607/// a PID, which another PID namespace would read wrongly: a hold that can be
608/// taken has nobody behind it.
609pub struct LaneHold {
610    lock_fd: OwnedFd
611}
612
613impl Drop for LaneHold {
614    fn drop(&mut self) {
615        let _ = unsafe { libc::flock(self.lock_fd.as_raw_fd(), libc::LOCK_UN) };
616    }
617}
618
619/// Take lane `lane` of segment `shm_name`, if no live process holds it.
620///
621/// `Ok(None)` means a process holding it is alive. Never waits.
622pub fn try_hold_lane(
623    shm_name: &str,
624    lane: usize
625) -> io::Result<Option<LaneHold>> {
626    let name = format!("{}.lane{lane}", shm_name.trim_start_matches('/'));
627    let lock_fd = open_lock_file(&lock_file_path(&name))?;
628    // SAFETY: `lock_fd` is an open descriptor owned by this call.
629    let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) };
630    if rc == 0 {
631        return Ok(Some(LaneHold { lock_fd }));
632    }
633    let error = io::Error::last_os_error();
634    if error.kind() == io::ErrorKind::WouldBlock {
635        return Ok(None);
636    }
637    Err(error)
638}
639
640/// Every lock file beside segment `shm_name` of `uid`: the region's own and
641/// one per lane that was ever held ([`try_hold_lane`]). What removes a segment
642/// removes these with it; an unlocked one left behind is harmless, but it is
643/// one more file per lane for every segment that ever existed.
644pub fn companion_lock_files(
645    shm_name: &str,
646    uid: u32
647) -> io::Result<Vec<PathBuf>> {
648    let base = shm_name.trim_start_matches('/');
649    let own = format!("{base}.lock");
650    let lane = format!("{base}.lane");
651    let dir = lock_dir_for_uid(uid);
652    let entries = match std::fs::read_dir(&dir) {
653        Ok(entries) => entries,
654        Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(Vec::new()),
655        Err(error) => return Err(error)
656    };
657    let mut found = Vec::new();
658    for entry in entries {
659        let entry = entry?;
660        let Some(file) = entry.file_name().to_str().map(str::to_owned) else {
661            continue;
662        };
663        if file == own || (file.starts_with(&lane) && file.ends_with(".lock")) {
664            found.push(entry.path());
665        }
666    }
667    Ok(found)
668}
669
670fn lock_file_path(shm_name: &str) -> PathBuf {
671    lock_dir().join(format!("{}.lock", shm_name.trim_start_matches('/')))
672}
673
674/// Where the companion lock files live.
675///
676/// They used to live in `/tmp` directly, as
677/// `/tmp/orbit-{fleet}-{kind}-{uid}.lock`, opened `O_CREAT` without `O_EXCL`
678/// and without asking who owned what the open found. `/tmp` is world-writable,
679/// so any local user could create that file first and then hold `LOCK_EX` on it
680/// for as long as they liked: every process in the fleet would sit in `flock` —
681/// not fail, block — waiting for a lock it was never going to get. The SHM
682/// segment name is uid-scoped and so cannot be squatted this way; the lock path
683/// was not. `O_NOFOLLOW` prevented the symlink version of the trick and nothing
684/// else.
685///
686/// They now live in a per-uid directory created `0700`, which a user who is not
687/// us cannot put a file into. What such a user can still do is create the
688/// directory first, so its owner and mode are checked on every open rather than
689/// assumed from having created it: a directory that is not ours, or not
690/// private, fails the open with the path in the message instead of parking the
691/// process on a lock.
692///
693/// `XDG_RUNTIME_DIR` is preferred where the session provides one, because it is
694/// already per-user and `0700` and so is not inside a world-writable directory
695/// at all. macOS has no such variable but gives each user a private `TMPDIR`;
696/// `/tmp` is the fallback, with the checks above carrying the weight.
697fn lock_dir() -> PathBuf {
698    let uid = unsafe { libc::geteuid() };
699    lock_dir_for_uid(uid)
700}
701
702fn lock_dir_for_uid(uid: u32) -> PathBuf {
703    let base = std::env::var_os("XDG_RUNTIME_DIR")
704        .map(PathBuf::from)
705        .filter(|dir| dir.is_absolute())
706        .unwrap_or_else(|| PathBuf::from("/tmp"));
707
708    base.join(format!("{SHM_NAMESPACE}-{uid}"))
709}
710
711/// Creates the lock directory if it is missing and refuses it if it is not
712/// ours. Called when a lock is actually taken rather than when a region is
713/// opened: a region that never locks has nothing to squat, and failing its open
714/// on a directory it does not use would hand an attacker a wider outage than
715/// the one being closed.
716fn ensure_lock_dir(dir: &Path) -> io::Result<()> {
717    use std::os::unix::fs::DirBuilderExt;
718
719    let uid = unsafe { libc::geteuid() };
720    match std::fs::DirBuilder::new().mode(0o700).create(dir) {
721        Ok(()) => {}
722        Err(error) if error.kind() == io::ErrorKind::AlreadyExists => {}
723        Err(error) => return Err(error)
724    }
725
726    ensure_private_dir(dir, uid)
727}
728
729/// Refuses a lock directory that someone else could write to.
730///
731/// `symlink_metadata` rather than `metadata`: a symlink pointing at a directory
732/// we do own would otherwise pass while the lock files landed somewhere the
733/// attacker chose.
734fn ensure_private_dir(
735    dir: &Path,
736    uid: u32
737) -> io::Result<()> {
738    use std::os::unix::fs::{MetadataExt, PermissionsExt};
739
740    let metadata = std::fs::symlink_metadata(dir)?;
741    if !metadata.is_dir() {
742        return Err(io::Error::new(
743            io::ErrorKind::PermissionDenied,
744            format!("{} is not a directory", dir.display())
745        ));
746    }
747    if metadata.uid() != uid {
748        return Err(io::Error::new(
749            io::ErrorKind::PermissionDenied,
750            format!(
751                "{} is owned by uid {} rather than {uid}; refusing to lock in a directory \
752                 another user controls",
753                dir.display(),
754                metadata.uid()
755            )
756        ));
757    }
758    if metadata.permissions().mode() & 0o077 != 0 {
759        return Err(io::Error::new(
760            io::ErrorKind::PermissionDenied,
761            format!(
762                "{} is mode {:o}; refusing to lock in a directory others can write to",
763                dir.display(),
764                metadata.permissions().mode() & 0o777
765            )
766        ));
767    }
768
769    Ok(())
770}
771
772#[cfg(test)]
773mod fleet_lock_tests {
774    use super::{join_fleet_membership, try_lock_fleet_exclusive};
775
776    /// A member alive means the fleet cannot be cleared, and there is no
777    /// flag that says otherwise; the member going away is what opens it.
778    #[test]
779    fn a_member_holds_the_fleet_against_exclusive_takers() {
780        let fleet = format!("fl{:x}", std::process::id());
781        let uid = unsafe { libc::geteuid() };
782
783        let member = join_fleet_membership(&fleet).expect("join");
784        assert!(try_lock_fleet_exclusive(&fleet, uid).expect("try").is_none());
785
786        drop(member);
787        let exclusive = try_lock_fleet_exclusive(&fleet, uid).expect("try");
788        assert!(exclusive.is_some());
789        // And a member cannot join while a removal holds the fleet: the
790        // shared lock would block, which is the behaviour, not a test to run.
791        drop(exclusive);
792        let _ = std::fs::remove_file(super::fleet_lock_path_for_uid(&fleet, uid));
793    }
794}