orbit_core/shm.rs
1//! POSIX shared-memory helpers — V1 substrate for cross-process rings.
2//!
3//! Wraps `shm_open` / `ftruncate` / `mmap` / `munmap` / `shm_unlink`
4//! into a small, RAII-friendly API. Unix-only; Windows support is
5//! a separate concern (Win32 named file mapping) that can land later.
6//!
7//! ## Naming
8//!
9//! Segments are named `/orbit-{fleet}-{kind}-{uid}` — fleet name from
10//! the embedder, KIND from `OrbitTyped::KIND`, UID from `geteuid()`.
11//! UID-scoping avoids the `/dev/shm` sticky-bit cross-user collision
12//! problem (a stale segment owned by one user blocks another from
13//! `shm_unlink`-ing it on next boot).
14//!
15//! Rings that require a process-recoverable writer lock also open a
16//! companion `orbit-{fleet}-{kind}-{uid}.lock` file. It carries no
17//! ring data or state; it only supplies a regular-file inode for `flock`,
18//! because advisory locking on a POSIX SHM descriptor is not uniformly
19//! supported across the Unix targets Orbit serves. An unlocked stale
20//! companion file is safe to reuse.
21//!
22//! Those files live in a per-uid directory — `$XDG_RUNTIME_DIR/orbit-{uid}`
23//! where the session provides one, `/tmp/orbit-{uid}` otherwise — created
24//! `0700` and checked on every lock. They were once in `/tmp` directly, which
25//! made them squattable: see [`lock_dir`].
26//!
27//! ## Lifetime
28//!
29//! [`ShmRegion`] owns the mapped pointer and unmaps on drop. It does
30//! NOT `shm_unlink` on drop — the segment lives until an explicit
31//! [`ShmRegion::unlink`] call. This matches POSIX convention: a
32//! segment with mapped users is not removed; `shm_unlink` only
33//! prevents *new* opens, the current mapping stays valid until the
34//! last process unmaps.
35
36#![cfg(unix)]
37
38use std::ffi::CString;
39use std::fs::OpenOptions;
40use std::io;
41use std::os::fd::{AsRawFd, FromRawFd, OwnedFd};
42use std::os::unix::fs::OpenOptionsExt;
43use std::path::{Path, PathBuf};
44use std::ptr::NonNull;
45
46/// Namespace used by Orbit POSIX shared-memory objects.
47pub const SHM_NAMESPACE: &str = "orbit";
48
49/// Result of physically validating an existing POSIX SHM object.
50///
51/// This check is deliberately below ring semantics: it verifies that the
52/// named object can be opened and is large enough for the requested mapping,
53/// but it does not inspect an owning data structure's magic, version, or
54/// geometry header.
55#[derive(Clone, Copy, Debug, Eq, PartialEq)]
56pub enum ShmValidation {
57 /// No object currently exists under the requested name.
58 Missing,
59 /// The object exists and can safely back at least the requested mapping.
60 Valid { actual_size: usize },
61}
62
63/// A mapped POSIX SHM region. Drop unmaps; `unlink` removes the
64/// underlying name and any companion lock file (only the *creator*
65/// should call it on shutdown).
66pub struct ShmRegion {
67 name: CString,
68 lock_path: PathBuf,
69 /// Whether this region uses the companion file for ordered writes.
70 ///
71 /// The descriptor itself is deliberately not retained: every critical
72 /// section opens its own file description so a later fork does not inherit
73 /// an idle descriptor that can keep a future `flock` alive.
74 process_lock: bool,
75 ptr: NonNull<u8>,
76 len: usize,
77 /// True when this handle was the one that *created* the segment
78 /// (so it knows to `shm_unlink` if asked). Other attachers see
79 /// `false`.
80 created: bool,
81}
82
83impl ShmRegion {
84 /// Validate an existing shared-memory object without creating, mapping,
85 /// resetting, or unlinking it.
86 ///
87 /// Returns [`ShmValidation::Missing`] when the name does not exist. A
88 /// present object must be at least `minimum_size` bytes; larger objects
89 /// are accepted because some platforms report page-rounded SHM sizes.
90 /// The owning ring or table remains responsible for validating its own
91 /// persisted ABI header after mapping.
92 pub fn validate_existing(name: &str, minimum_size: usize) -> io::Result<ShmValidation> {
93 let cname = CString::new(name)
94 .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
95 let raw_fd = loop {
96 // SAFETY: passing a valid C string and well-known POSIX flags.
97 let fd = unsafe { libc::shm_open(cname.as_ptr(), libc::O_RDWR, 0o600) };
98 if fd >= 0 {
99 break fd;
100 }
101 let error = io::Error::last_os_error();
102 if error.kind() == io::ErrorKind::Interrupted {
103 continue;
104 }
105 if error.raw_os_error() == Some(libc::ENOENT) {
106 return Ok(ShmValidation::Missing);
107 }
108 return Err(error);
109 };
110 // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
111 // owned by this scope.
112 let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
113 let actual_size = shm_object_size(&fd, name)?;
114 validate_minimum_size(name, actual_size, minimum_size)?;
115 Ok(ShmValidation::Valid { actual_size })
116 }
117
118 /// Open or create a shared-memory segment of `size` bytes,
119 /// memory-mapped read/write. Idempotent: if the segment already
120 /// exists with the same name and enough mapped bytes, it is reused
121 /// (`created = false`). First creation does `ftruncate(size)`;
122 /// later opens verify the existing object before mapping it. Some
123 /// platforms report a page-rounded SHM size, so a larger `st_size`
124 /// is valid; the owning data structure must verify its own header.
125 pub fn open_or_create(name: &str, size: usize) -> io::Result<Self> {
126 let (region, initialization_lock) = Self::open_or_create_inner(name, size, false)?;
127 debug_assert!(initialization_lock.is_none());
128 Ok(region)
129 }
130
131 /// Open or create a region while holding its process lock through caller
132 /// initialization. This prevents a peer from observing the interval
133 /// between `shm_open` and the owning data structure's initialized header.
134 pub fn open_or_create_locked(name: &str, size: usize) -> io::Result<(Self, ShmRegionLock)> {
135 let (region, initialization_lock) = Self::open_or_create_inner(name, size, true)?;
136 Ok((
137 region,
138 initialization_lock.expect("locked SHM open must return its initialization lock"),
139 ))
140 }
141
142 fn open_or_create_inner(
143 name: &str,
144 size: usize,
145 process_lock: bool,
146 ) -> io::Result<(Self, Option<ShmRegionLock>)> {
147 let cname = CString::new(name)
148 .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
149 let lock_path = lock_file_path(name);
150 let initialization_lock = if process_lock {
151 Some(lock_path_exclusive(&lock_path)?)
152 } else {
153 None
154 };
155
156 // Try create-exclusive first; if it already exists, open.
157 let (raw_fd, created) = unsafe {
158 // SAFETY: passing a valid C string and well-known POSIX flags.
159 let fd = libc::shm_open(
160 cname.as_ptr(),
161 libc::O_RDWR | libc::O_CREAT | libc::O_EXCL,
162 0o600,
163 );
164 if fd >= 0 {
165 (fd, true)
166 } else {
167 // Could be EEXIST (already created by a peer) or another error.
168 let err = io::Error::last_os_error();
169 if err.raw_os_error() != Some(libc::EEXIST) {
170 return Err(err);
171 }
172 let fd = libc::shm_open(cname.as_ptr(), libc::O_RDWR, 0o600);
173 if fd < 0 {
174 return Err(io::Error::last_os_error());
175 }
176 (fd, false)
177 }
178 };
179 // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
180 // owned by this scope.
181 let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
182
183 // Size the segment on first creation.
184 if created {
185 // SAFETY: fd is a valid POSIX fd we just received.
186 let rc = unsafe { libc::ftruncate(fd.as_raw_fd(), size as libc::off_t) };
187 if rc != 0 {
188 let err = io::Error::last_os_error();
189 let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
190 return Err(err);
191 }
192 }
193
194 // Never mmap beyond the real SHM object: access past it can raise
195 // SIGBUS. A larger reported size is valid on platforms (notably
196 // macOS) that page-round POSIX SHM objects; callers verify their
197 // own ABI metadata after mapping.
198 let actual_size = match shm_object_size(&fd, name) {
199 Ok(actual_size) => actual_size,
200 Err(error) => {
201 if created {
202 let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
203 }
204 return Err(error);
205 }
206 };
207 if let Err(error) = validate_minimum_size(name, actual_size, size) {
208 if created {
209 let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
210 }
211 return Err(error);
212 }
213
214 // Memory-map the segment.
215 // SAFETY: fd valid, size positive, flags well-known.
216 let ptr = unsafe {
217 libc::mmap(
218 std::ptr::null_mut(),
219 size,
220 libc::PROT_READ | libc::PROT_WRITE,
221 libc::MAP_SHARED,
222 fd.as_raw_fd(),
223 0,
224 )
225 };
226
227 if ptr == libc::MAP_FAILED {
228 let err = io::Error::last_os_error();
229 if created {
230 let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
231 }
232 return Err(err);
233 }
234
235 // SAFETY: mmap returned a non-null pointer (we just checked).
236 let ptr = NonNull::new(ptr.cast::<u8>()).expect("mmap returned non-null on success");
237
238 Ok((
239 Self {
240 name: cname,
241 lock_path,
242 process_lock,
243 ptr,
244 len: size,
245 created,
246 },
247 initialization_lock,
248 ))
249 }
250
251 /// Raw mapped pointer to the start of the region.
252 pub fn as_ptr(&self) -> *mut u8 {
253 self.ptr.as_ptr()
254 }
255
256 /// Length of the mapped region (the `size` passed to `open_or_create`).
257 pub fn len(&self) -> usize {
258 self.len
259 }
260
261 pub fn is_empty(&self) -> bool {
262 self.len == 0
263 }
264
265 /// True when this handle was the one that created the segment.
266 /// Useful for picking which process performs first-time
267 /// initialization of the header.
268 pub fn created(&self) -> bool {
269 self.created
270 }
271
272 /// Acquire an exclusive cross-process lock tied to this SHM name.
273 ///
274 /// `flock` ownership is held by the kernel and is released when a process
275 /// exits or the descriptor closes, including abnormal termination.
276 pub fn lock_exclusive(&self) -> io::Result<ShmRegionLock> {
277 if !self.process_lock {
278 return Err(io::Error::new(
279 io::ErrorKind::InvalidInput,
280 "SHM region was opened without a process lock",
281 ));
282 }
283 lock_path_exclusive(&self.lock_path)
284 }
285
286 #[cfg(test)]
287 pub(crate) fn try_lock_exclusive(&self) -> io::Result<ShmRegionLock> {
288 if !self.process_lock {
289 return Err(io::Error::new(
290 io::ErrorKind::InvalidInput,
291 "SHM region was opened without a process lock",
292 ));
293 }
294 let lock_fd = open_lock_file(&self.lock_path)?;
295 let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) };
296 if rc == 0 {
297 Ok(ShmRegionLock { lock_fd })
298 } else {
299 Err(io::Error::last_os_error())
300 }
301 }
302
303 /// Remove the underlying segment name. Existing mappings stay
304 /// valid until each process drops its `ShmRegion`. Use only on
305 /// shutdown / fleet teardown by the process that owns lifecycle.
306 pub fn unlink(&self) -> io::Result<()> {
307 // SAFETY: name is a valid C string.
308 let rc = unsafe { libc::shm_unlink(self.name.as_ptr()) };
309 let shm_error = if rc != 0 {
310 let err = io::Error::last_os_error();
311 // ENOENT is fine — segment was already unlinked.
312 if err.raw_os_error() == Some(libc::ENOENT) {
313 None
314 } else {
315 Some(err)
316 }
317 } else {
318 None
319 };
320 let lock_error = match std::fs::remove_file(&self.lock_path) {
321 Ok(()) => None,
322 Err(error) if error.kind() == io::ErrorKind::NotFound => None,
323 Err(error) => Some(error),
324 };
325 if let Some(error) = shm_error.or(lock_error) {
326 return Err(error);
327 }
328 Ok(())
329 }
330}
331
332fn shm_object_size(fd: &OwnedFd, name: &str) -> io::Result<usize> {
333 let mut stat = std::mem::MaybeUninit::<libc::stat>::uninit();
334 // SAFETY: `fd` is valid and `stat` points to writable storage.
335 let stat_rc = unsafe { libc::fstat(fd.as_raw_fd(), stat.as_mut_ptr()) };
336 if stat_rc != 0 {
337 return Err(io::Error::last_os_error());
338 }
339 // SAFETY: `fstat` succeeded and initialized the structure.
340 let actual_size = unsafe { stat.assume_init() }.st_size;
341 usize::try_from(actual_size).map_err(|_| {
342 io::Error::new(
343 io::ErrorKind::InvalidData,
344 format!("SHM segment {name} reported invalid size {actual_size}"),
345 )
346 })
347}
348
349fn validate_minimum_size(name: &str, actual_size: usize, minimum_size: usize) -> io::Result<()> {
350 if actual_size < minimum_size {
351 return Err(io::Error::new(
352 io::ErrorKind::InvalidData,
353 format!(
354 "SHM segment {name} size {actual_size} is smaller than requested mapping {minimum_size}"
355 ),
356 ));
357 }
358 Ok(())
359}
360
361/// RAII guard for a [`ShmRegion`]'s process-recoverable exclusive lock.
362///
363/// Semantic crates use this when a current-state transition must be atomic
364/// across fleet processes. Dropping the guard releases the kernel lock.
365pub struct ShmRegionLock {
366 lock_fd: OwnedFd,
367}
368
369impl Drop for ShmRegionLock {
370 fn drop(&mut self) {
371 let _ = unsafe { libc::flock(self.lock_fd.as_raw_fd(), libc::LOCK_UN) };
372 }
373}
374
375fn lock_path_exclusive(lock_path: &Path) -> io::Result<ShmRegionLock> {
376 lock_fd_exclusive(open_lock_file(lock_path)?)
377}
378
379fn open_lock_file(lock_path: &Path) -> io::Result<OwnedFd> {
380 use std::os::unix::fs::MetadataExt;
381
382 if let Some(dir) = lock_path.parent() {
383 ensure_lock_dir(dir)?;
384 }
385
386 let file = OpenOptions::new()
387 .read(true)
388 .write(true)
389 .create(true)
390 .mode(0o600)
391 .custom_flags(libc::O_CLOEXEC | libc::O_NOFOLLOW)
392 .open(lock_path)?;
393
394 // The directory check makes this unreachable, which is the reason to make
395 // it anyway: it turns a property inferred from the directory's mode into
396 // one this function establishes about the descriptor it is about to lock.
397 let uid = unsafe { libc::geteuid() };
398 let owner = file.metadata()?.uid();
399 if owner != uid {
400 return Err(io::Error::new(
401 io::ErrorKind::PermissionDenied,
402 format!(
403 "{} is owned by uid {owner} rather than {uid}",
404 lock_path.display()
405 ),
406 ));
407 }
408
409 Ok(file.into())
410}
411
412fn lock_fd_exclusive(lock_fd: OwnedFd) -> io::Result<ShmRegionLock> {
413 loop {
414 let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX) };
415 if rc == 0 {
416 return Ok(ShmRegionLock { lock_fd });
417 }
418 let error = io::Error::last_os_error();
419 if error.kind() != io::ErrorKind::Interrupted {
420 return Err(error);
421 }
422 }
423}
424
425impl Drop for ShmRegion {
426 fn drop(&mut self) {
427 // SAFETY: ptr came from mmap of `self.len` bytes; munmap is the inverse.
428 unsafe {
429 libc::munmap(self.ptr.as_ptr().cast(), self.len);
430 }
431 }
432}
433
434// SAFETY: the underlying region is shared memory and synchronization
435// happens at the slot level (atomic seq counters); the handle itself
436// is just a pointer + length, safe to send/share.
437unsafe impl Send for ShmRegion {}
438unsafe impl Sync for ShmRegion {}
439
440/// Build the conventional name for an Orbit ring segment.
441pub fn ring_segment_name(fleet_name: &str, kind: u8) -> String {
442 // SAFETY: `geteuid` always returns a value; no error path.
443 let uid = unsafe { libc::geteuid() };
444 ring_segment_name_for_uid(fleet_name, kind, uid)
445}
446
447/// Build the conventional name for an Orbit ring segment owned by `uid`.
448///
449/// This form is intended for inspection and lifecycle tools that need to
450/// address a user other than their own effective uid.
451pub fn ring_segment_name_for_uid(fleet_name: &str, kind: u8, uid: u32) -> String {
452 format!("/{SHM_NAMESPACE}-{fleet_name}-{kind}-{uid}")
453}
454
455fn lock_file_path(shm_name: &str) -> PathBuf {
456 lock_dir().join(format!("{}.lock", shm_name.trim_start_matches('/')))
457}
458
459/// Where the companion lock files live.
460///
461/// They used to live in `/tmp` directly, as
462/// `/tmp/orbit-{fleet}-{kind}-{uid}.lock`, opened `O_CREAT` without `O_EXCL`
463/// and without asking who owned what the open found. `/tmp` is world-writable,
464/// so any local user could create that file first and then hold `LOCK_EX` on it
465/// for as long as they liked: every process in the fleet would sit in `flock` —
466/// not fail, block — waiting for a lock it was never going to get. The SHM
467/// segment name is uid-scoped and so cannot be squatted this way; the lock path
468/// was not. `O_NOFOLLOW` prevented the symlink version of the trick and nothing
469/// else.
470///
471/// They now live in a per-uid directory created `0700`, which a user who is not
472/// us cannot put a file into. What such a user can still do is create the
473/// directory first, so its owner and mode are checked on every open rather than
474/// assumed from having created it: a directory that is not ours, or not
475/// private, fails the open with the path in the message instead of parking the
476/// process on a lock.
477///
478/// `XDG_RUNTIME_DIR` is preferred where the session provides one, because it is
479/// already per-user and `0700` and so is not inside a world-writable directory
480/// at all. macOS has no such variable but gives each user a private `TMPDIR`;
481/// `/tmp` is the fallback, with the checks above carrying the weight.
482fn lock_dir() -> PathBuf {
483 let uid = unsafe { libc::geteuid() };
484 let base = std::env::var_os("XDG_RUNTIME_DIR")
485 .map(PathBuf::from)
486 .filter(|dir| dir.is_absolute())
487 .unwrap_or_else(|| PathBuf::from("/tmp"));
488
489 base.join(format!("{SHM_NAMESPACE}-{uid}"))
490}
491
492/// Creates the lock directory if it is missing and refuses it if it is not
493/// ours. Called when a lock is actually taken rather than when a region is
494/// opened: a region that never locks has nothing to squat, and failing its open
495/// on a directory it does not use would hand an attacker a wider outage than
496/// the one being closed.
497fn ensure_lock_dir(dir: &Path) -> io::Result<()> {
498 use std::os::unix::fs::DirBuilderExt;
499
500 let uid = unsafe { libc::geteuid() };
501 match std::fs::DirBuilder::new().mode(0o700).create(dir) {
502 Ok(()) => {}
503 Err(error) if error.kind() == io::ErrorKind::AlreadyExists => {}
504 Err(error) => return Err(error),
505 }
506
507 ensure_private_dir(dir, uid)
508}
509
510/// Refuses a lock directory that someone else could write to.
511///
512/// `symlink_metadata` rather than `metadata`: a symlink pointing at a directory
513/// we do own would otherwise pass while the lock files landed somewhere the
514/// attacker chose.
515fn ensure_private_dir(dir: &Path, uid: u32) -> io::Result<()> {
516 use std::os::unix::fs::MetadataExt;
517 use std::os::unix::fs::PermissionsExt;
518
519 let metadata = std::fs::symlink_metadata(dir)?;
520 if !metadata.is_dir() {
521 return Err(io::Error::new(
522 io::ErrorKind::PermissionDenied,
523 format!("{} is not a directory", dir.display()),
524 ));
525 }
526 if metadata.uid() != uid {
527 return Err(io::Error::new(
528 io::ErrorKind::PermissionDenied,
529 format!(
530 "{} is owned by uid {} rather than {uid}; refusing to lock in a directory \
531 another user controls",
532 dir.display(),
533 metadata.uid()
534 ),
535 ));
536 }
537 if metadata.permissions().mode() & 0o077 != 0 {
538 return Err(io::Error::new(
539 io::ErrorKind::PermissionDenied,
540 format!(
541 "{} is mode {:o}; refusing to lock in a directory others can write to",
542 dir.display(),
543 metadata.permissions().mode() & 0o777
544 ),
545 ));
546 }
547
548 Ok(())
549}