orbit_core/shm.rs
1//! POSIX shared-memory helpers — V1 substrate for cross-process rings.
2//!
3//! Wraps `shm_open` / `ftruncate` / `mmap` / `munmap` / `shm_unlink`
4//! into a small, RAII-friendly API. Unix-only; Windows support is
5//! a separate concern (Win32 named file mapping) that can land later.
6//!
7//! ## Naming
8//!
9//! Segments are named `/orbit-{fleet}-{kind}-{uid}` — fleet name from
10//! the embedder, KIND from `OrbitTyped::KIND`, UID from `geteuid()`.
11//! UID-scoping avoids the `/dev/shm` sticky-bit cross-user collision
12//! problem (a stale segment owned by one user blocks another from
13//! `shm_unlink`-ing it on next boot).
14//!
15//! Rings that require a process-recoverable writer lock also open a
16//! companion `/tmp/orbit-{fleet}-{kind}-{uid}.lock` file. It carries no
17//! ring data or state; it only supplies a regular-file inode for `flock`,
18//! because advisory locking on a POSIX SHM descriptor is not uniformly
19//! supported across the Unix targets Orbit serves. An unlocked stale
20//! companion file is safe to reuse.
21//!
22//! ## Lifetime
23//!
24//! [`ShmRegion`] owns the mapped pointer and unmaps on drop. It does
25//! NOT `shm_unlink` on drop — the segment lives until an explicit
26//! [`ShmRegion::unlink`] call. This matches POSIX convention: a
27//! segment with mapped users is not removed; `shm_unlink` only
28//! prevents *new* opens, the current mapping stays valid until the
29//! last process unmaps.
30
31#![cfg(unix)]
32
33use std::ffi::CString;
34use std::fs::OpenOptions;
35use std::io;
36use std::os::fd::{AsRawFd, FromRawFd, OwnedFd};
37use std::os::unix::fs::OpenOptionsExt;
38use std::path::{Path, PathBuf};
39use std::ptr::NonNull;
40
41/// Namespace used by Orbit POSIX shared-memory objects.
42pub const SHM_NAMESPACE: &str = "orbit";
43
44/// A mapped POSIX SHM region. Drop unmaps; `unlink` removes the
45/// underlying name and any companion lock file (only the *creator*
46/// should call it on shutdown).
47pub struct ShmRegion {
48 name: CString,
49 lock_path: PathBuf,
50 /// Whether this region uses the companion file for ordered writes.
51 ///
52 /// The descriptor itself is deliberately not retained: every critical
53 /// section opens its own file description so a later fork does not inherit
54 /// an idle descriptor that can keep a future `flock` alive.
55 process_lock: bool,
56 ptr: NonNull<u8>,
57 len: usize,
58 /// True when this handle was the one that *created* the segment
59 /// (so it knows to `shm_unlink` if asked). Other attachers see
60 /// `false`.
61 created: bool,
62}
63
64impl ShmRegion {
65 /// Open or create a shared-memory segment of `size` bytes,
66 /// memory-mapped read/write. Idempotent: if the segment already
67 /// exists with the same name and enough mapped bytes, it is reused
68 /// (`created = false`). First creation does `ftruncate(size)`;
69 /// later opens verify the existing object before mapping it. Some
70 /// platforms report a page-rounded SHM size, so a larger `st_size`
71 /// is valid; the owning data structure must verify its own header.
72 pub fn open_or_create(name: &str, size: usize) -> io::Result<Self> {
73 let (region, initialization_lock) = Self::open_or_create_inner(name, size, false)?;
74 debug_assert!(initialization_lock.is_none());
75 Ok(region)
76 }
77
78 /// Open or create a region while holding its process lock through caller
79 /// initialization. This prevents a peer from observing the interval
80 /// between `shm_open` and the owning data structure's initialized header.
81 pub fn open_or_create_locked(name: &str, size: usize) -> io::Result<(Self, ShmRegionLock)> {
82 let (region, initialization_lock) = Self::open_or_create_inner(name, size, true)?;
83 Ok((
84 region,
85 initialization_lock.expect("locked SHM open must return its initialization lock"),
86 ))
87 }
88
89 fn open_or_create_inner(
90 name: &str,
91 size: usize,
92 process_lock: bool,
93 ) -> io::Result<(Self, Option<ShmRegionLock>)> {
94 let cname = CString::new(name)
95 .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "shm name has nul byte"))?;
96 let lock_path = lock_file_path(name);
97 let initialization_lock = if process_lock {
98 Some(lock_path_exclusive(&lock_path)?)
99 } else {
100 None
101 };
102
103 // Try create-exclusive first; if it already exists, open.
104 let (raw_fd, created) = unsafe {
105 // SAFETY: passing a valid C string and well-known POSIX flags.
106 let fd = libc::shm_open(
107 cname.as_ptr(),
108 libc::O_RDWR | libc::O_CREAT | libc::O_EXCL,
109 0o600,
110 );
111 if fd >= 0 {
112 (fd, true)
113 } else {
114 // Could be EEXIST (already created by a peer) or another error.
115 let err = io::Error::last_os_error();
116 if err.raw_os_error() != Some(libc::EEXIST) {
117 return Err(err);
118 }
119 let fd = libc::shm_open(cname.as_ptr(), libc::O_RDWR, 0o600);
120 if fd < 0 {
121 return Err(io::Error::last_os_error());
122 }
123 (fd, false)
124 }
125 };
126 // SAFETY: `raw_fd` was returned by `shm_open` and is now uniquely
127 // owned by this scope.
128 let fd = unsafe { OwnedFd::from_raw_fd(raw_fd) };
129
130 // Size the segment on first creation.
131 if created {
132 // SAFETY: fd is a valid POSIX fd we just received.
133 let rc = unsafe { libc::ftruncate(fd.as_raw_fd(), size as libc::off_t) };
134 if rc != 0 {
135 let err = io::Error::last_os_error();
136 let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
137 return Err(err);
138 }
139 }
140
141 // Never mmap beyond the real SHM object: access past it can raise
142 // SIGBUS. A larger reported size is valid on platforms (notably
143 // macOS) that page-round POSIX SHM objects; callers verify their
144 // own ABI metadata after mapping.
145 let mut stat = std::mem::MaybeUninit::<libc::stat>::uninit();
146 let stat_rc = unsafe { libc::fstat(fd.as_raw_fd(), stat.as_mut_ptr()) };
147 if stat_rc != 0 {
148 let err = io::Error::last_os_error();
149 if created {
150 let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
151 }
152 return Err(err);
153 }
154 let actual_size = unsafe { stat.assume_init() }.st_size;
155 if actual_size < 0 || (actual_size as usize) < size {
156 if created {
157 let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
158 }
159 return Err(io::Error::new(
160 io::ErrorKind::InvalidData,
161 format!(
162 "SHM segment {name} size {actual_size} is smaller than requested mapping {size}"
163 ),
164 ));
165 }
166
167 // Memory-map the segment.
168 // SAFETY: fd valid, size positive, flags well-known.
169 let ptr = unsafe {
170 libc::mmap(
171 std::ptr::null_mut(),
172 size,
173 libc::PROT_READ | libc::PROT_WRITE,
174 libc::MAP_SHARED,
175 fd.as_raw_fd(),
176 0,
177 )
178 };
179
180 if ptr == libc::MAP_FAILED {
181 let err = io::Error::last_os_error();
182 if created {
183 let _ = unsafe { libc::shm_unlink(cname.as_ptr()) };
184 }
185 return Err(err);
186 }
187
188 // SAFETY: mmap returned a non-null pointer (we just checked).
189 let ptr = NonNull::new(ptr.cast::<u8>()).expect("mmap returned non-null on success");
190
191 Ok((
192 Self {
193 name: cname,
194 lock_path,
195 process_lock,
196 ptr,
197 len: size,
198 created,
199 },
200 initialization_lock,
201 ))
202 }
203
204 /// Raw mapped pointer to the start of the region.
205 pub fn as_ptr(&self) -> *mut u8 {
206 self.ptr.as_ptr()
207 }
208
209 /// Length of the mapped region (the `size` passed to `open_or_create`).
210 pub fn len(&self) -> usize {
211 self.len
212 }
213
214 pub fn is_empty(&self) -> bool {
215 self.len == 0
216 }
217
218 /// True when this handle was the one that created the segment.
219 /// Useful for picking which process performs first-time
220 /// initialization of the header.
221 pub fn created(&self) -> bool {
222 self.created
223 }
224
225 /// Acquire an exclusive cross-process lock tied to this SHM name.
226 ///
227 /// `flock` ownership is held by the kernel and is released when a process
228 /// exits or the descriptor closes, including abnormal termination.
229 pub fn lock_exclusive(&self) -> io::Result<ShmRegionLock> {
230 if !self.process_lock {
231 return Err(io::Error::new(
232 io::ErrorKind::InvalidInput,
233 "SHM region was opened without a process lock",
234 ));
235 }
236 lock_path_exclusive(&self.lock_path)
237 }
238
239 #[cfg(test)]
240 pub(crate) fn try_lock_exclusive(&self) -> io::Result<ShmRegionLock> {
241 if !self.process_lock {
242 return Err(io::Error::new(
243 io::ErrorKind::InvalidInput,
244 "SHM region was opened without a process lock",
245 ));
246 }
247 let lock_fd = open_lock_file(&self.lock_path)?;
248 let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX | libc::LOCK_NB) };
249 if rc == 0 {
250 Ok(ShmRegionLock { lock_fd })
251 } else {
252 Err(io::Error::last_os_error())
253 }
254 }
255
256 /// Remove the underlying segment name. Existing mappings stay
257 /// valid until each process drops its `ShmRegion`. Use only on
258 /// shutdown / fleet teardown by the process that owns lifecycle.
259 pub fn unlink(&self) -> io::Result<()> {
260 // SAFETY: name is a valid C string.
261 let rc = unsafe { libc::shm_unlink(self.name.as_ptr()) };
262 let shm_error = if rc != 0 {
263 let err = io::Error::last_os_error();
264 // ENOENT is fine — segment was already unlinked.
265 if err.raw_os_error() == Some(libc::ENOENT) {
266 None
267 } else {
268 Some(err)
269 }
270 } else {
271 None
272 };
273 let lock_error = match std::fs::remove_file(&self.lock_path) {
274 Ok(()) => None,
275 Err(error) if error.kind() == io::ErrorKind::NotFound => None,
276 Err(error) => Some(error),
277 };
278 if let Some(error) = shm_error.or(lock_error) {
279 return Err(error);
280 }
281 Ok(())
282 }
283}
284
285/// RAII guard for a [`ShmRegion`]'s process-recoverable exclusive lock.
286///
287/// Semantic crates use this when a current-state transition must be atomic
288/// across fleet processes. Dropping the guard releases the kernel lock.
289pub struct ShmRegionLock {
290 lock_fd: OwnedFd,
291}
292
293impl Drop for ShmRegionLock {
294 fn drop(&mut self) {
295 let _ = unsafe { libc::flock(self.lock_fd.as_raw_fd(), libc::LOCK_UN) };
296 }
297}
298
299fn lock_path_exclusive(lock_path: &Path) -> io::Result<ShmRegionLock> {
300 lock_fd_exclusive(open_lock_file(lock_path)?)
301}
302
303fn open_lock_file(lock_path: &Path) -> io::Result<OwnedFd> {
304 OpenOptions::new()
305 .read(true)
306 .write(true)
307 .create(true)
308 .mode(0o600)
309 .custom_flags(libc::O_CLOEXEC | libc::O_NOFOLLOW)
310 .open(lock_path)
311 .map(Into::into)
312}
313
314fn lock_fd_exclusive(lock_fd: OwnedFd) -> io::Result<ShmRegionLock> {
315 loop {
316 let rc = unsafe { libc::flock(lock_fd.as_raw_fd(), libc::LOCK_EX) };
317 if rc == 0 {
318 return Ok(ShmRegionLock { lock_fd });
319 }
320 let error = io::Error::last_os_error();
321 if error.kind() != io::ErrorKind::Interrupted {
322 return Err(error);
323 }
324 }
325}
326
327impl Drop for ShmRegion {
328 fn drop(&mut self) {
329 // SAFETY: ptr came from mmap of `self.len` bytes; munmap is the inverse.
330 unsafe {
331 libc::munmap(self.ptr.as_ptr().cast(), self.len);
332 }
333 }
334}
335
336// SAFETY: the underlying region is shared memory and synchronization
337// happens at the slot level (atomic seq counters); the handle itself
338// is just a pointer + length, safe to send/share.
339unsafe impl Send for ShmRegion {}
340unsafe impl Sync for ShmRegion {}
341
342/// Build the conventional name for an Orbit ring segment.
343pub fn ring_segment_name(fleet_name: &str, kind: u8) -> String {
344 // SAFETY: `geteuid` always returns a value; no error path.
345 let uid = unsafe { libc::geteuid() };
346 ring_segment_name_for_uid(fleet_name, kind, uid)
347}
348
349/// Build the conventional name for an Orbit ring segment owned by `uid`.
350///
351/// This form is intended for inspection and lifecycle tools that need to
352/// address a user other than their own effective uid.
353pub fn ring_segment_name_for_uid(fleet_name: &str, kind: u8, uid: u32) -> String {
354 format!("/{SHM_NAMESPACE}-{fleet_name}-{kind}-{uid}")
355}
356
357fn lock_file_path(shm_name: &str) -> PathBuf {
358 PathBuf::from("/tmp").join(format!("{}.lock", shm_name.trim_start_matches('/')))
359}