orbit-core 0.3.6

Fleet-aware shared-memory rings over POSIX shared memory.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
//! POSIX shared-memory helpers — V1 substrate for cross-process rings.
//!
//! Wraps `shm_open` / `ftruncate` / `mmap` / `munmap` / `shm_unlink`
//! into a small, RAII-friendly API. Unix-only; Windows support is
//! a separate concern (Win32 named file mapping) that can land later.
//!
//! ## Naming
//!
//! Segments are named `/orbit-{fleet}-{kind}-{uid}` — fleet name from
//! the embedder, KIND from `OrbitTyped::KIND`, UID from `geteuid()`.
//! UID-scoping avoids the `/dev/shm` sticky-bit cross-user collision
//! problem (a stale segment owned by one user blocks another from
//! `shm_unlink`-ing it on next boot).
//!
//! Rings that require a process-recoverable writer lock also open a
//! companion `orbit-{fleet}-{kind}-{uid}.lock` file. It carries no
//! ring data or state; it only supplies a regular-file inode for `flock`,
//! because advisory locking on a POSIX SHM descriptor is not uniformly
//! supported across the Unix targets Orbit serves. An unlocked stale
//! companion file is safe to reuse.
//!
//! Those files live in a per-uid directory — `$XDG_RUNTIME_DIR/orbit-{uid}`
//! where the session provides one, `/tmp/orbit-{uid}` otherwise — created
//! `0700` and checked on every lock. They were once in `/tmp` directly, which
//! made them squattable: see [`lock_dir`].
//!
//! ## Lifetime
//!
//! [`ShmRegion`] owns the mapped pointer and unmaps on drop. It does
//! NOT `shm_unlink` on drop — the segment lives until an explicit
//! [`ShmRegion::unlink`] call. This matches POSIX convention: a
//! segment with mapped users is not removed; `shm_unlink` only
//! prevents *new* opens, the current mapping stays valid until the
//! last process unmaps.

#![cfg(unix)]

use std::ffi::CString;
use std::fs::OpenOptions;
use std::io;
use std::os::fd::{AsRawFd, FromRawFd, OwnedFd};
use std::os::unix::fs::OpenOptionsExt;
use std::path::{Path, PathBuf};
use std::ptr::NonNull;

/// Namespace used by Orbit POSIX shared-memory objects.
pub const SHM_NAMESPACE: &str = "orbit";

/// A mapped POSIX SHM region. Drop unmaps; `unlink` removes the
/// underlying name and any companion lock file (only the *creator*
/// should call it on shutdown).
pub struct ShmRegion {
    name: CString,
    lock_path: PathBuf,
    /// Whether this region uses the companion file for ordered writes.
    ///
    /// The descriptor itself is deliberately not retained: every critical
    /// section opens its own file description so a later fork does not inherit
    /// an idle descriptor that can keep a future `flock` alive.
    process_lock: bool,
    ptr: NonNull<u8>,
    len: usize,
    /// True when this handle was the one that *created* the segment
    /// (so it knows to `shm_unlink` if asked). Other attachers see
    /// `false`.
    created: bool,
}

impl ShmRegion {
    /// Open or create a shared-memory segment of `size` bytes,
    /// memory-mapped read/write. Idempotent: if the segment already
    /// exists with the same name and enough mapped bytes, it is reused
    /// (`created = false`). First creation does `ftruncate(size)`;
    /// later opens verify the existing object before mapping it. Some
    /// platforms report a page-rounded SHM size, so a larger `st_size`
    /// is valid; the owning data structure must verify its own header.
    pub fn open_or_create(name: &str, size: usize) -> io::Result<Self> {
        let (region, initialization_lock) = Self::open_or_create_inner(name, size, false)?;
        debug_assert!(initialization_lock.is_none());
        Ok(region)
    }

    /// Open or create a region while holding its process lock through caller
    /// initialization. This prevents a peer from observing the interval
    /// between `shm_open` and the owning data structure's initialized header.
    pub fn open_or_create_locked(name: &str, size: usize) -> io::Result<(Self, ShmRegionLock)> {
        let (region, initialization_lock) = Self::open_or_create_inner(name, size, true)?;
        Ok((
            region,
            initialization_lock.expect("locked SHM open must return its initialization lock"),
        ))
    }

    fn open_or_create_inner(
        name: &str,
        size: usize,
        process_lock: bool,
    ) -> io::Result<(Self, Option<ShmRegionLock>)> {
        let cname = CString::new(name)
            .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
        let lock_path = lock_file_path(name);
        let initialization_lock = if process_lock {
            Some(lock_path_exclusive(&lock_path)?)
        } else {
            None
        };

        // Try create-exclusive first; if it already exists, open.
        let (raw_fd, created) = unsafe {
            // SAFETY: passing a valid C string and well-known POSIX flags.
            let fd = libc::shm_open(
                cname.as_ptr(),
                libc::O_RDWR | libc::O_CREAT | libc::O_EXCL,
                0o600,
            );
            if fd >= 0 {
                (fd, true)
            } else {
                // Could be EEXIST (already created by a peer) or another error.
                let err = io::Error::last_os_error();
                if err.raw_os_error() != Some(libc::EEXIST) {
                    return Err(err);
                }
                let fd = libc::shm_open(cname.as_ptr(), libc::O_RDWR, 0o600);
                if fd < 0 {
                    return Err(io::Error::last_os_error());
                }
                (fd, false)
            }
        };
        // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
        // owned by this scope.
        let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };

        // Size the segment on first creation.
        if created {
            // SAFETY: fd is a valid POSIX fd we just received.
            let rc = unsafe { libc::ftruncate(fd.as_raw_fd(), size as libc::off_t) };
            if rc != 0 {
                let err = io::Error::last_os_error();
                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
                return Err(err);
            }
        }

        // Never mmap beyond the real SHM object: access past it can raise
        // SIGBUS. A larger reported size is valid on platforms (notably
        // macOS) that page-round POSIX SHM objects; callers verify their
        // own ABI metadata after mapping.
        let mut stat = std::mem::MaybeUninit::<libc::stat>::uninit();
        let stat_rc = unsafe { libc::fstat(fd.as_raw_fd(), stat.as_mut_ptr()) };
        if stat_rc != 0 {
            let err = io::Error::last_os_error();
            if created {
                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
            }
            return Err(err);
        }
        let actual_size = unsafe { stat.assume_init() }.st_size;
        if actual_size < 0 || (actual_size as usize) < size {
            if created {
                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
            }
            return Err(io::Error::new(
                io::ErrorKind::InvalidData,
                format!(
                    "SHM segment {name} size {actual_size} is smaller than requested mapping {size}"
                ),
            ));
        }

        // Memory-map the segment.
        // SAFETY: fd valid, size positive, flags well-known.
        let ptr = unsafe {
            libc::mmap(
                std::ptr::null_mut(),
                size,
                libc::PROT_READ | libc::PROT_WRITE,
                libc::MAP_SHARED,
                fd.as_raw_fd(),
                0,
            )
        };

        if ptr == libc::MAP_FAILED {
            let err = io::Error::last_os_error();
            if created {
                let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
            }
            return Err(err);
        }

        // SAFETY: mmap returned a non-null pointer (we just checked).
        let ptr = NonNull::new(ptr.cast::<u8>()).expect("mmap returned non-null on success");

        Ok((
            Self {
                name: cname,
                lock_path,
                process_lock,
                ptr,
                len: size,
                created,
            },
            initialization_lock,
        ))
    }

    /// Raw mapped pointer to the start of the region.
    pub fn as_ptr(&self) -> *mut u8 {
        self.ptr.as_ptr()
    }

    /// Length of the mapped region (the `size` passed to `open_or_create`).
    pub fn len(&self) -> usize {
        self.len
    }

    pub fn is_empty(&self) -> bool {
        self.len == 0
    }

    /// True when this handle was the one that created the segment.
    /// Useful for picking which process performs first-time
    /// initialization of the header.
    pub fn created(&self) -> bool {
        self.created
    }

    /// Acquire an exclusive cross-process lock tied to this SHM name.
    ///
    /// `flock` ownership is held by the kernel and is released when a process
    /// exits or the descriptor closes, including abnormal termination.
    pub fn lock_exclusive(&self) -> io::Result<ShmRegionLock> {
        if !self.process_lock {
            return Err(io::Error::new(
                io::ErrorKind::InvalidInput,
                "SHM region was opened without a process lock",
            ));
        }
        lock_path_exclusive(&self.lock_path)
    }

    #[cfg(test)]
    pub(crate) fn try_lock_exclusive(&self) -> io::Result<ShmRegionLock> {
        if !self.process_lock {
            return Err(io::Error::new(
                io::ErrorKind::InvalidInput,
                "SHM region was opened without a process lock",
            ));
        }
        let lock_fd = open_lock_file(&self.lock_path)?;
        let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) };
        if rc == 0 {
            Ok(ShmRegionLock { lock_fd })
        } else {
            Err(io::Error::last_os_error())
        }
    }

    /// Remove the underlying segment name. Existing mappings stay
    /// valid until each process drops its `ShmRegion`. Use only on
    /// shutdown / fleet teardown by the process that owns lifecycle.
    pub fn unlink(&self) -> io::Result<()> {
        // SAFETY: name is a valid C string.
        let rc = unsafe { libc::shm_unlink(self.name.as_ptr()) };
        let shm_error = if rc != 0 {
            let err = io::Error::last_os_error();
            // ENOENT is fine — segment was already unlinked.
            if err.raw_os_error() == Some(libc::ENOENT) {
                None
            } else {
                Some(err)
            }
        } else {
            None
        };
        let lock_error = match std::fs::remove_file(&self.lock_path) {
            Ok(()) => None,
            Err(error) if error.kind() == io::ErrorKind::NotFound => None,
            Err(error) => Some(error),
        };
        if let Some(error) = shm_error.or(lock_error) {
            return Err(error);
        }
        Ok(())
    }
}

/// RAII guard for a [`ShmRegion`]'s process-recoverable exclusive lock.
///
/// Semantic crates use this when a current-state transition must be atomic
/// across fleet processes. Dropping the guard releases the kernel lock.
pub struct ShmRegionLock {
    lock_fd: OwnedFd,
}

impl Drop for ShmRegionLock {
    fn drop(&mut self) {
        let _ = unsafe { libc::flock(self.lock_fd.as_raw_fd(), libc::LOCK_UN) };
    }
}

fn lock_path_exclusive(lock_path: &Path) -> io::Result<ShmRegionLock> {
    lock_fd_exclusive(open_lock_file(lock_path)?)
}

fn open_lock_file(lock_path: &Path) -> io::Result<OwnedFd> {
    use std::os::unix::fs::MetadataExt;

    if let Some(dir) = lock_path.parent() {
        ensure_lock_dir(dir)?;
    }

    let file = OpenOptions::new()
        .read(true)
        .write(true)
        .create(true)
        .mode(0o600)
        .custom_flags(libc::O_CLOEXEC | libc::O_NOFOLLOW)
        .open(lock_path)?;

    // The directory check makes this unreachable, which is the reason to make
    // it anyway: it turns a property inferred from the directory's mode into
    // one this function establishes about the descriptor it is about to lock.
    let uid = unsafe { libc::geteuid() };
    let owner = file.metadata()?.uid();
    if owner != uid {
        return Err(io::Error::new(
            io::ErrorKind::PermissionDenied,
            format!(
                "{} is owned by uid {owner} rather than {uid}",
                lock_path.display()
            ),
        ));
    }

    Ok(file.into())
}

fn lock_fd_exclusive(lock_fd: OwnedFd) -> io::Result<ShmRegionLock> {
    loop {
        let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX) };
        if rc == 0 {
            return Ok(ShmRegionLock { lock_fd });
        }
        let error = io::Error::last_os_error();
        if error.kind() != io::ErrorKind::Interrupted {
            return Err(error);
        }
    }
}

impl Drop for ShmRegion {
    fn drop(&mut self) {
        // SAFETY: ptr came from mmap of `self.len` bytes; munmap is the inverse.
        unsafe {
            libc::munmap(self.ptr.as_ptr().cast(), self.len);
        }
    }
}

// SAFETY: the underlying region is shared memory and synchronization
// happens at the slot level (atomic seq counters); the handle itself
// is just a pointer + length, safe to send/share.
unsafe impl Send for ShmRegion {}
unsafe impl Sync for ShmRegion {}

/// Build the conventional name for an Orbit ring segment.
pub fn ring_segment_name(fleet_name: &str, kind: u8) -> String {
    // SAFETY: `geteuid` always returns a value; no error path.
    let uid = unsafe { libc::geteuid() };
    ring_segment_name_for_uid(fleet_name, kind, uid)
}

/// Build the conventional name for an Orbit ring segment owned by `uid`.
///
/// This form is intended for inspection and lifecycle tools that need to
/// address a user other than their own effective uid.
pub fn ring_segment_name_for_uid(fleet_name: &str, kind: u8, uid: u32) -> String {
    format!("/{SHM_NAMESPACE}-{fleet_name}-{kind}-{uid}")
}

fn lock_file_path(shm_name: &str) -> PathBuf {
    lock_dir().join(format!("{}.lock", shm_name.trim_start_matches('/')))
}

/// Where the companion lock files live.
///
/// They used to live in `/tmp` directly, as
/// `/tmp/orbit-{fleet}-{kind}-{uid}.lock`, opened `O_CREAT` without `O_EXCL`
/// and without asking who owned what the open found. `/tmp` is world-writable,
/// so any local user could create that file first and then hold `LOCK_EX` on it
/// for as long as they liked: every process in the fleet would sit in `flock` —
/// not fail, block — waiting for a lock it was never going to get. The SHM
/// segment name is uid-scoped and so cannot be squatted this way; the lock path
/// was not. `O_NOFOLLOW` prevented the symlink version of the trick and nothing
/// else.
///
/// They now live in a per-uid directory created `0700`, which a user who is not
/// us cannot put a file into. What such a user can still do is create the
/// directory first, so its owner and mode are checked on every open rather than
/// assumed from having created it: a directory that is not ours, or not
/// private, fails the open with the path in the message instead of parking the
/// process on a lock.
///
/// `XDG_RUNTIME_DIR` is preferred where the session provides one, because it is
/// already per-user and `0700` and so is not inside a world-writable directory
/// at all. macOS has no such variable but gives each user a private `TMPDIR`;
/// `/tmp` is the fallback, with the checks above carrying the weight.
fn lock_dir() -> PathBuf {
    let uid = unsafe { libc::geteuid() };
    let base = std::env::var_os("XDG_RUNTIME_DIR")
        .map(PathBuf::from)
        .filter(|dir| dir.is_absolute())
        .unwrap_or_else(|| PathBuf::from("/tmp"));

    base.join(format!("{SHM_NAMESPACE}-{uid}"))
}

/// Creates the lock directory if it is missing and refuses it if it is not
/// ours. Called when a lock is actually taken rather than when a region is
/// opened: a region that never locks has nothing to squat, and failing its open
/// on a directory it does not use would hand an attacker a wider outage than
/// the one being closed.
fn ensure_lock_dir(dir: &Path) -> io::Result<()> {
    use std::os::unix::fs::DirBuilderExt;

    let uid = unsafe { libc::geteuid() };
    match std::fs::DirBuilder::new().mode(0o700).create(dir) {
        Ok(()) => {}
        Err(error) if error.kind() == io::ErrorKind::AlreadyExists => {}
        Err(error) => return Err(error),
    }

    ensure_private_dir(dir, uid)
}

/// Refuses a lock directory that someone else could write to.
///
/// `symlink_metadata` rather than `metadata`: a symlink pointing at a directory
/// we do own would otherwise pass while the lock files landed somewhere the
/// attacker chose.
fn ensure_private_dir(dir: &Path, uid: u32) -> io::Result<()> {
    use std::os::unix::fs::MetadataExt;
    use std::os::unix::fs::PermissionsExt;

    let metadata = std::fs::symlink_metadata(dir)?;
    if !metadata.is_dir() {
        return Err(io::Error::new(
            io::ErrorKind::PermissionDenied,
            format!("{} is not a directory", dir.display()),
        ));
    }
    if metadata.uid() != uid {
        return Err(io::Error::new(
            io::ErrorKind::PermissionDenied,
            format!(
                "{} is owned by uid {} rather than {uid}; refusing to lock in a directory \
                 another user controls",
                dir.display(),
                metadata.uid()
            ),
        ));
    }
    if metadata.permissions().mode() & 0o077 != 0 {
        return Err(io::Error::new(
            io::ErrorKind::PermissionDenied,
            format!(
                "{} is mode {:o}; refusing to lock in a directory others can write to",
                dir.display(),
                metadata.permissions().mode() & 0o777
            ),
        ));
    }

    Ok(())
}