mnemosyne_arena/segment/alloc/allocate.rs
1//! Handing a segment out: from a pool if one is retained, from the OS
2//! otherwise, then initialized and guarded.
3//!
4//! Returning one is the other direction and lives in [`super::release`],
5//! which reaches `try_return_to_pool` here through the parent.
6
7use super::super::alignment::checked_align_up;
8use super::super::pool::HasSegmentPool;
9#[cfg(feature = "segment-header-guards")]
10use super::SEGMENT_HEADER_GUARD_SIZE;
11use super::{SEGMENT_MAPPING_SIZE, SEGMENT_TAIL_GUARD_SIZE};
12use crate::numa::current_numa_node;
13#[cfg(feature = "segment-header-guards")]
14use mnemosyne_core::constants::PAGE_SIZE;
15use mnemosyne_core::constants::{SEGMENT_ALIGN, SEGMENT_SIZE};
16use mnemosyne_core::types::Segment;
17
18/// Pops a retained free segment from the global segment pool and
19/// re-initializes it, monomorphized per backend `B` (`#[inline(never)]` keeps
20/// this cold pool path out of the hot caller).
21///
22/// Only the free pool is consulted: a retained segment holds no live
23/// allocation, so re-initializing it erases nothing a caller still reads.
24///
25/// # Safety
26///
27/// The caller must ensure that the global segment pool contains valid,
28/// initialized `Segment` structures. The returned segment (if any) is owned by
29/// the caller.
30#[inline(never)]
31unsafe fn pop_free_segment<B: HasSegmentPool>() -> Option<*mut Segment> {
32 let segment = B::global_segment_pool().pop()?;
33 // SAFETY: segment points to a valid allocated Segment. We re-initialize
34 // the segment to erase stale epoch metadata and reset it for new allocations.
35 unsafe {
36 let raw_ptr = (*segment).raw_alloc_ptr;
37 let node = (*segment).numa_node;
38 Segment::initialize(segment, raw_ptr, node);
39 }
40 Some(segment)
41}
42
43/// Helper to return a segment to the global segment pool, monomorphized per
44/// backend `B`.
45///
46/// # Safety
47///
48/// The `segment` pointer must point to a valid, initialized `Segment` exclusively owned
49/// by the caller.
50#[inline(always)]
51pub(super) unsafe fn try_return_to_pool<B: HasSegmentPool>(segment: *mut Segment) -> bool {
52 debug_assert!(
53 !segment.is_null(),
54 "try_return_to_pool received null segment"
55 );
56 // SAFETY: by this function's contract `segment` is a valid, initialized,
57 // exclusively-owned `Segment`, satisfying `try_push_retained`'s own contract.
58 unsafe { B::global_segment_pool().try_push_retained(segment) }
59}
60
61/// Non-generic helper to initialize an allocated segment header and establish alignment bounds.
62///
63/// # Safety
64///
65/// The caller must guarantee:
66/// - `raw_ptr` must point to a valid, exclusive, and page-aligned allocation of size `SEGMENT_MAPPING_SIZE`.
67/// - The memory range must be writable to initialize the `Segment` structure.
68#[inline(never)]
69unsafe fn initialize_allocated_segment(
70 raw_ptr: *mut u8,
71 numa_node: u32,
72) -> Option<(*mut Segment, usize, usize, usize)> {
73 let aligned_addr = checked_align_up(raw_ptr as usize, SEGMENT_ALIGN)?;
74 let aligned_ptr = raw_ptr.map_addr(|_| aligned_addr).cast::<Segment>();
75
76 // SAFETY: aligned_ptr is within the allocated region.
77 unsafe {
78 Segment::initialize(aligned_ptr, raw_ptr, numa_node);
79 }
80
81 let tail_slack_start = if cfg!(feature = "segment-tail-guards") {
82 aligned_addr + SEGMENT_SIZE + SEGMENT_TAIL_GUARD_SIZE
83 } else {
84 aligned_addr + SEGMENT_SIZE
85 };
86 let mapping_end = raw_ptr as usize + SEGMENT_MAPPING_SIZE;
87
88 Some((aligned_ptr, aligned_addr, tail_slack_start, mapping_end))
89}
90
91/// Returns the head (`[raw_ptr, aligned_addr)`) and tail
92/// (`[tail_slack_start, mapping_end)`) alignment-slack subranges of a segment
93/// mapping to the OS via `B::decommit`.
94///
95/// Both slack regions exist only to satisfy `SEGMENT_ALIGN` rounding and never
96/// hold allocator or user data. On Windows `VirtualAlloc` eagerly commits the
97/// whole mapping, so decommitting drops the slack's commit charge (up to
98/// ~`SEGMENT_ALIGN` ≈ 2 MiB of head slack per segment) for the mapping's
99/// lifetime; on Unix the slack is lazily backed, so this is typically a no-op.
100/// Best-effort: a backend without decommit support (`SUPPORTS_DECOMMIT ==
101/// false`) skips entirely, and both subranges stay inside the reservation,
102/// which the base `B::deallocate(raw_ptr, ..)` releases in full.
103///
104/// # Safety
105///
106/// `raw_ptr` must name the base of the live backend mapping that contains both
107/// subranges, with `raw_ptr as usize <= aligned_addr` and
108/// `tail_slack_start <= mapping_end`. Neither subrange may hold allocator or
109/// user data, and the head bounds are page-aligned (`raw_ptr` comes from the
110/// backend allocator and `aligned_addr` is a `SEGMENT_ALIGN` multiple), as
111/// `decommit` requires.
112#[inline]
113pub(crate) unsafe fn decommit_mapping_slack<B: mnemosyne_core::MemoryBackend>(
114 raw_ptr: *mut u8,
115 aligned_addr: usize,
116 tail_slack_start: usize,
117 mapping_end: usize,
118) {
119 if !B::SUPPORTS_DECOMMIT {
120 return;
121 }
122 let head_slack = aligned_addr - raw_ptr as usize;
123 if head_slack > 0 {
124 // SAFETY: per this function's contract, `[raw_ptr, aligned_addr)` is a
125 // page-aligned, data-free subrange of the live mapping.
126 let _ = unsafe { B::decommit(raw_ptr, head_slack) };
127 }
128 if tail_slack_start < mapping_end {
129 // SAFETY: per this function's contract, `[tail_slack_start,
130 // mapping_end)` is a data-free subrange of the live mapping. Deriving
131 // the subrange pointer from `raw_ptr` retains the mapping provenance.
132 let tail_slack = raw_ptr.map_addr(|_| tail_slack_start);
133 let _ = unsafe { B::decommit(tail_slack, mapping_end - tail_slack_start) };
134 }
135}
136
137/// A segment handed out by [`acquire_segment`], tagged with whether it may
138/// still hold live allocations.
139///
140/// The distinction decides what the new owner may do with the segment: a
141/// [`Free`](Self::Free) segment holds nothing, so it may be carved up or handed
142/// back through [`deallocate_segment`](super::deallocate_segment); an
143/// [`Orphan`](Self::Orphan) holds blocks a dead thread cache allocated and
144/// other threads may still read, write, or free, so it may be adopted but
145/// never returned to the free pool or the OS until every one of its pages is
146/// empty.
147#[derive(Debug, Clone, Copy, PartialEq, Eq)]
148pub enum AcquiredSegment {
149 /// A freshly mapped or pool-reinitialized segment with no live allocation.
150 Free(*mut Segment),
151 /// A segment from the orphan pool, handed out as is with its live
152 /// allocations, per-page free lists, and pending cross-thread frees.
153 Orphan(*mut Segment),
154}
155
156/// Allocates an empty, aligned segment, either from the free pool or from the
157/// OS.
158///
159/// Never returns an orphan: an orphaned segment still holds live allocations,
160/// and a caller of this function may hand the segment straight back through
161/// [`deallocate_segment`](super::deallocate_segment), which would put those
162/// allocations in the free pool for reuse or for a purge to unmap. Thread
163/// caches that can adopt an orphan use [`acquire_segment`].
164///
165/// # Monomorphization and ZST Static Routing
166///
167/// The backend parameter `B` acts as a Zero-Sized Type (ZST) policy marker. Calls
168/// to this function are fully monomorphized by the compiler into direct machine-code
169/// calls for the target backend, preserving the zero-cost abstraction invariant.
170///
171/// # Safety
172///
173/// This function is unsafe because it allocates virtual memory from the OS/backend,
174/// aligns and initializes a raw `Segment` pointer. The caller must guarantee:
175/// - The backend `B` must be a valid implementor of `HasSegmentPool`.
176/// - The returned pointer must eventually be returned to the pool via
177/// `deallocate_segment` or released to the OS via `release_segment_mapping`.
178#[inline]
179pub unsafe fn allocate_segment<B: HasSegmentPool>() -> Option<*mut Segment> {
180 // SAFETY: forwarded to the caller — the free pool holds valid segments.
181 if let Some(segment) = unsafe { pop_free_segment::<B>() } {
182 return Some(segment);
183 }
184 // SAFETY: forwarded to the caller.
185 unsafe { map_fresh_segment::<B>() }
186}
187
188/// Acquires a segment for a thread cache: a retained free segment first, then
189/// an orphan to adopt, then a fresh OS mapping.
190///
191/// # Safety
192///
193/// As [`allocate_segment`]. Additionally, an [`AcquiredSegment::Orphan`] must
194/// be adopted — its live pages kept intact — or pushed back to the orphan
195/// pool; it must not reach [`deallocate_segment`](super::deallocate_segment)
196/// or [`release_segment_mapping`](super::release_segment_mapping) while any of
197/// its pages holds a live allocation.
198#[inline]
199pub unsafe fn acquire_segment<B: HasSegmentPool>() -> Option<AcquiredSegment> {
200 // SAFETY: forwarded to the caller — the free pool holds valid segments.
201 if let Some(segment) = unsafe { pop_free_segment::<B>() } {
202 return Some(AcquiredSegment::Free(segment));
203 }
204 if let Some(segment) = B::global_orphan_pool().pop() {
205 return Some(AcquiredSegment::Orphan(segment));
206 }
207 // SAFETY: forwarded to the caller.
208 unsafe { map_fresh_segment::<B>() }.map(AcquiredSegment::Free)
209}
210
211/// Maps, aligns, and initializes a fresh segment from the OS, purging the free
212/// pool and retrying once when the first mapping request fails.
213///
214/// # Safety
215///
216/// As [`allocate_segment`].
217#[inline(never)]
218unsafe fn map_fresh_segment<B: HasSegmentPool>() -> Option<*mut Segment> {
219 // We allocate twice the segment size to ensure we can find an aligned boundary.
220 // SAFETY: SEGMENT_MAPPING_SIZE is non-zero and aligned. We call B::allocate.
221 let mut raw_ptr = unsafe { B::allocate(SEGMENT_MAPPING_SIZE) };
222 if raw_ptr.is_null() {
223 // First OS allocation failed: release every retained free segment
224 // back to the OS to reclaim the address space / commit charge a
225 // transient working-set spike may be holding, then retry exactly
226 // once. Bounded to one purge and one retry: a still-exhausted OS
227 // fails fast on the second call rather than looping.
228 //
229 // SAFETY: segments retained in the pool at this point are
230 // exclusively pool-owned — a segment a thread-local allocator still
231 // references is "active", not "retained" — so releasing them
232 // invalidates no live pointer, and the retried `B::allocate` call
233 // follows the same contract as the first.
234 unsafe { super::release::purge_segment_pool::<B>() };
235 raw_ptr = unsafe { B::allocate(SEGMENT_MAPPING_SIZE) };
236 // Surfaces the recovery attempt and its outcome as a stats counter
237 // (`ArenaMemoryStats::oom_retries` / `oom_retry_successes`) — the
238 // crate has no tracing dependency, so telemetry follows the
239 // existing `purge_calls`/`reset_calls` counter convention.
240 B::global_segment_pool().record_oom_retry(!raw_ptr.is_null());
241 if raw_ptr.is_null() {
242 return None;
243 }
244 }
245
246 let numa_node = current_numa_node();
247 // SAFETY: `raw_ptr` is the non-null `SEGMENT_MAPPING_SIZE` mapping just
248 // returned by `B::allocate`, which is exclusively owned and writable —
249 // exactly `initialize_allocated_segment`'s contract.
250 let (aligned_ptr, aligned_addr, tail_slack_start, mapping_end) =
251 match unsafe { initialize_allocated_segment(raw_ptr, numa_node) } {
252 Some(val) => val,
253 None => {
254 // SAFETY: Releasing raw memory back to the backend because alignment check overflowed.
255 let _released = unsafe { B::deallocate(raw_ptr, SEGMENT_MAPPING_SIZE) };
256 return None;
257 }
258 };
259
260 // Bind the aligned segment mapping to the allocating thread's NUMA node.
261 // `mbind(MPOL_BIND)` enforces first-touch locality so pages are allocated
262 // from the correct socket rather than relying on the OS default policy.
263 // This call goes to the kernel only once per fresh segment (cold path);
264 // segments recycled from the pool retain their prior NUMA binding and never
265 // reach this function (they return early via `pop_free_segment`).
266 //
267 // SAFETY: `aligned_ptr` points to the segment-aligned base of the
268 // `SEGMENT_MAPPING_SIZE` mapping that is exclusively owned at this point
269 // (not yet published to any pool or thread), `SEGMENT_SIZE` is the
270 // usable extent of the aligned portion, and `numa_node` was read from
271 // Themis's thread-local cache immediately before this call.
272 unsafe {
273 crate::numa::bind_segment_to_numa_node(
274 raw_ptr.map_addr(|_| aligned_ptr as usize),
275 SEGMENT_SIZE,
276 numa_node,
277 )
278 };
279
280 #[cfg(feature = "segment-header-guards")]
281 {
282 if B::SUPPORTS_MAKE_GUARD {
283 // Install a header guard at the end of Page 0.
284 // Underflows (backward OOB writes) from Page 1 land in this guard region
285 // instead of overwriting the segment metadata at the start of Page 0.
286 //
287 // SAFETY: aligned_addr + PAGE_SIZE - SEGMENT_HEADER_GUARD_SIZE is inside the mapping
288 // and Page 0 is reserved strictly for Segment metadata (ending far before the guard).
289 let header_guard_addr = aligned_addr + PAGE_SIZE - SEGMENT_HEADER_GUARD_SIZE;
290 let header_guard = raw_ptr.map_addr(|_| header_guard_addr);
291 let _guarded = unsafe { B::make_guard(header_guard, SEGMENT_HEADER_GUARD_SIZE) };
292 }
293 }
294
295 #[cfg(feature = "segment-tail-guards")]
296 {
297 if B::SUPPORTS_MAKE_GUARD {
298 // Install a tail guard immediately after the segment's user-page
299 // region. Forward OOB writes that walk past Page 31 land in this
300 // guard region instead of an unrelated mapping. The address lives
301 // inside the `SEGMENT_MAPPING_SIZE - SEGMENT_SIZE` slack the arena
302 // reserves to satisfy `SEGMENT_ALIGN` rounding, so it is always
303 // part of the same backend allocation and is released together
304 // with the segment by `B::deallocate(raw_ptr, SEGMENT_MAPPING_SIZE)`.
305 // The install is best-effort: a backend without a `make_guard`
306 // implementation (default `false`) or a kernel that declines the
307 // request (e.g. macOS-arm64 where the OS page size exceeds 4 KiB)
308 // silently skips, leaving the slack accessible. Backend telemetry
309 // (`guard_install_calls`) surfaces the actual install count.
310 //
311 // SAFETY: aligned_addr + SEGMENT_SIZE is inside the raw mapping
312 // because slack-after >= OS_PAGE_SIZE >= SEGMENT_TAIL_GUARD_SIZE on
313 // supported targets. `make_guard` never invalidates the mapping.
314 let tail_guard_addr = aligned_addr + SEGMENT_SIZE;
315 let tail_guard = raw_ptr.map_addr(|_| tail_guard_addr);
316 let _guarded = unsafe { B::make_guard(tail_guard, SEGMENT_TAIL_GUARD_SIZE) };
317 }
318 }
319
320 // The mapping over-reserves `SEGMENT_MAPPING_SIZE = 2 * SEGMENT_SIZE` so a
321 // `SEGMENT_ALIGN`-aligned base can always be found; return the resulting
322 // head and tail slack to the OS (guard pages installed above lie outside
323 // both slack subranges, so ordering relative to the guards is immaterial).
324 //
325 // SAFETY: `[raw_ptr, aligned_addr)` precedes the header and
326 // `[tail_slack_start, mapping_end)` succeeds the segment pages (and tail
327 // guard, when enabled), so neither holds allocator data; both stay inside
328 // the live reservation covered by the base release, and the head bounds
329 // are page-aligned (`raw_ptr` from `allocate`, `aligned_addr` a
330 // `SEGMENT_ALIGN` multiple).
331 unsafe { decommit_mapping_slack::<B>(raw_ptr, aligned_addr, tail_slack_start, mapping_end) };
332
333 Some(aligned_ptr)
334}