Skip to main content

mnemosyne_arena/segment/alloc/
allocate.rs

1//! Handing a segment out: from a pool if one is retained, from the OS
2//! otherwise, then initialized and guarded.
3//!
4//! Returning one is the other direction and lives in [`super::release`],
5//! which reaches `try_return_to_pool` here through the parent.
6
7use super::super::alignment::checked_align_up;
8use super::super::pool::HasSegmentPool;
9#[cfg(feature = "segment-header-guards")]
10use super::SEGMENT_HEADER_GUARD_SIZE;
11use super::{SEGMENT_MAPPING_SIZE, SEGMENT_TAIL_GUARD_SIZE};
12use crate::numa::current_numa_node;
13#[cfg(feature = "segment-header-guards")]
14use mnemosyne_core::constants::PAGE_SIZE;
15use mnemosyne_core::constants::{SEGMENT_ALIGN, SEGMENT_SIZE};
16use mnemosyne_core::types::Segment;
17
18/// Pops a retained free segment from the global segment pool and
19/// re-initializes it, monomorphized per backend `B` (`#[inline(never)]` keeps
20/// this cold pool path out of the hot caller).
21///
22/// Only the free pool is consulted: a retained segment holds no live
23/// allocation, so re-initializing it erases nothing a caller still reads.
24///
25/// # Safety
26///
27/// The caller must ensure that the global segment pool contains valid,
28/// initialized `Segment` structures. The returned segment (if any) is owned by
29/// the caller.
30#[inline(never)]
31unsafe fn pop_free_segment<B: HasSegmentPool>() -> Option<*mut Segment> {
32    let segment = B::global_segment_pool().pop()?;
33    // SAFETY: segment points to a valid allocated Segment. We re-initialize
34    // the segment to erase stale epoch metadata and reset it for new allocations.
35    unsafe {
36        let raw_ptr = (*segment).raw_alloc_ptr;
37        let node = (*segment).numa_node;
38        Segment::initialize(segment, raw_ptr, node);
39    }
40    Some(segment)
41}
42
43/// Helper to return a segment to the global segment pool, monomorphized per
44/// backend `B`.
45///
46/// # Safety
47///
48/// The `segment` pointer must point to a valid, initialized `Segment` exclusively owned
49/// by the caller.
50#[inline(always)]
51pub(super) unsafe fn try_return_to_pool<B: HasSegmentPool>(segment: *mut Segment) -> bool {
52    debug_assert!(
53        !segment.is_null(),
54        "try_return_to_pool received null segment"
55    );
56    // SAFETY: by this function's contract `segment` is a valid, initialized,
57    // exclusively-owned `Segment`, satisfying `try_push_retained`'s own contract.
58    unsafe { B::global_segment_pool().try_push_retained(segment) }
59}
60
61/// Non-generic helper to initialize an allocated segment header and establish alignment bounds.
62///
63/// # Safety
64///
65/// The caller must guarantee:
66/// - `raw_ptr` must point to a valid, exclusive, and page-aligned allocation of size `SEGMENT_MAPPING_SIZE`.
67/// - The memory range must be writable to initialize the `Segment` structure.
68#[inline(never)]
69unsafe fn initialize_allocated_segment(
70    raw_ptr: *mut u8,
71    numa_node: u32,
72) -> Option<(*mut Segment, usize, usize, usize)> {
73    let aligned_addr = checked_align_up(raw_ptr as usize, SEGMENT_ALIGN)?;
74    let aligned_ptr = raw_ptr.map_addr(|_| aligned_addr).cast::<Segment>();
75
76    // SAFETY: aligned_ptr is within the allocated region.
77    unsafe {
78        Segment::initialize(aligned_ptr, raw_ptr, numa_node);
79    }
80
81    let tail_slack_start = if cfg!(feature = "segment-tail-guards") {
82        aligned_addr + SEGMENT_SIZE + SEGMENT_TAIL_GUARD_SIZE
83    } else {
84        aligned_addr + SEGMENT_SIZE
85    };
86    let mapping_end = raw_ptr as usize + SEGMENT_MAPPING_SIZE;
87
88    Some((aligned_ptr, aligned_addr, tail_slack_start, mapping_end))
89}
90
91/// Returns the head (`[raw_ptr, aligned_addr)`) and tail
92/// (`[tail_slack_start, mapping_end)`) alignment-slack subranges of a segment
93/// mapping to the OS via `B::decommit`.
94///
95/// Both slack regions exist only to satisfy `SEGMENT_ALIGN` rounding and never
96/// hold allocator or user data. On Windows `VirtualAlloc` eagerly commits the
97/// whole mapping, so decommitting drops the slack's commit charge (up to
98/// ~`SEGMENT_ALIGN` ≈ 2 MiB of head slack per segment) for the mapping's
99/// lifetime; on Unix the slack is lazily backed, so this is typically a no-op.
100/// Best-effort: a backend without decommit support (`SUPPORTS_DECOMMIT ==
101/// false`) skips entirely, and both subranges stay inside the reservation,
102/// which the base `B::deallocate(raw_ptr, ..)` releases in full.
103///
104/// # Safety
105///
106/// `raw_ptr` must name the base of the live backend mapping that contains both
107/// subranges, with `raw_ptr as usize <= aligned_addr` and
108/// `tail_slack_start <= mapping_end`. Neither subrange may hold allocator or
109/// user data, and the head bounds are page-aligned (`raw_ptr` comes from the
110/// backend allocator and `aligned_addr` is a `SEGMENT_ALIGN` multiple), as
111/// `decommit` requires.
112#[inline]
113pub(crate) unsafe fn decommit_mapping_slack<B: mnemosyne_core::MemoryBackend>(
114    raw_ptr: *mut u8,
115    aligned_addr: usize,
116    tail_slack_start: usize,
117    mapping_end: usize,
118) {
119    if !B::SUPPORTS_DECOMMIT {
120        return;
121    }
122    let head_slack = aligned_addr - raw_ptr as usize;
123    if head_slack > 0 {
124        // SAFETY: per this function's contract, `[raw_ptr, aligned_addr)` is a
125        // page-aligned, data-free subrange of the live mapping.
126        let _ = unsafe { B::decommit(raw_ptr, head_slack) };
127    }
128    if tail_slack_start < mapping_end {
129        // SAFETY: per this function's contract, `[tail_slack_start,
130        // mapping_end)` is a data-free subrange of the live mapping. Deriving
131        // the subrange pointer from `raw_ptr` retains the mapping provenance.
132        let tail_slack = raw_ptr.map_addr(|_| tail_slack_start);
133        let _ = unsafe { B::decommit(tail_slack, mapping_end - tail_slack_start) };
134    }
135}
136
137/// A segment handed out by [`acquire_segment`], tagged with whether it may
138/// still hold live allocations.
139///
140/// The distinction decides what the new owner may do with the segment: a
141/// [`Free`](Self::Free) segment holds nothing, so it may be carved up or handed
142/// back through [`deallocate_segment`](super::deallocate_segment); an
143/// [`Orphan`](Self::Orphan) holds blocks a dead thread cache allocated and
144/// other threads may still read, write, or free, so it may be adopted but
145/// never returned to the free pool or the OS until every one of its pages is
146/// empty.
147#[derive(Debug, Clone, Copy, PartialEq, Eq)]
148pub enum AcquiredSegment {
149    /// A freshly mapped or pool-reinitialized segment with no live allocation.
150    Free(*mut Segment),
151    /// A segment from the orphan pool, handed out as is with its live
152    /// allocations, per-page free lists, and pending cross-thread frees.
153    Orphan(*mut Segment),
154}
155
156/// Allocates an empty, aligned segment, either from the free pool or from the
157/// OS.
158///
159/// Never returns an orphan: an orphaned segment still holds live allocations,
160/// and a caller of this function may hand the segment straight back through
161/// [`deallocate_segment`](super::deallocate_segment), which would put those
162/// allocations in the free pool for reuse or for a purge to unmap. Thread
163/// caches that can adopt an orphan use [`acquire_segment`].
164///
165/// # Monomorphization and ZST Static Routing
166///
167/// The backend parameter `B` acts as a Zero-Sized Type (ZST) policy marker. Calls
168/// to this function are fully monomorphized by the compiler into direct machine-code
169/// calls for the target backend, preserving the zero-cost abstraction invariant.
170///
171/// # Safety
172///
173/// This function is unsafe because it allocates virtual memory from the OS/backend,
174/// aligns and initializes a raw `Segment` pointer. The caller must guarantee:
175/// - The backend `B` must be a valid implementor of `HasSegmentPool`.
176/// - The returned pointer must eventually be returned to the pool via
177///   `deallocate_segment` or released to the OS via `release_segment_mapping`.
178#[inline]
179pub unsafe fn allocate_segment<B: HasSegmentPool>() -> Option<*mut Segment> {
180    // SAFETY: forwarded to the caller — the free pool holds valid segments.
181    if let Some(segment) = unsafe { pop_free_segment::<B>() } {
182        return Some(segment);
183    }
184    // SAFETY: forwarded to the caller.
185    unsafe { map_fresh_segment::<B>() }
186}
187
188/// Acquires a segment for a thread cache: a retained free segment first, then
189/// an orphan to adopt, then a fresh OS mapping.
190///
191/// # Safety
192///
193/// As [`allocate_segment`]. Additionally, an [`AcquiredSegment::Orphan`] must
194/// be adopted — its live pages kept intact — or pushed back to the orphan
195/// pool; it must not reach [`deallocate_segment`](super::deallocate_segment)
196/// or [`release_segment_mapping`](super::release_segment_mapping) while any of
197/// its pages holds a live allocation.
198#[inline]
199pub unsafe fn acquire_segment<B: HasSegmentPool>() -> Option<AcquiredSegment> {
200    // SAFETY: forwarded to the caller — the free pool holds valid segments.
201    if let Some(segment) = unsafe { pop_free_segment::<B>() } {
202        return Some(AcquiredSegment::Free(segment));
203    }
204    if let Some(segment) = B::global_orphan_pool().pop() {
205        return Some(AcquiredSegment::Orphan(segment));
206    }
207    // SAFETY: forwarded to the caller.
208    unsafe { map_fresh_segment::<B>() }.map(AcquiredSegment::Free)
209}
210
211/// Maps, aligns, and initializes a fresh segment from the OS, purging the free
212/// pool and retrying once when the first mapping request fails.
213///
214/// # Safety
215///
216/// As [`allocate_segment`].
217#[inline(never)]
218unsafe fn map_fresh_segment<B: HasSegmentPool>() -> Option<*mut Segment> {
219    // We allocate twice the segment size to ensure we can find an aligned boundary.
220    // SAFETY: SEGMENT_MAPPING_SIZE is non-zero and aligned. We call B::allocate.
221    let mut raw_ptr = unsafe { B::allocate(SEGMENT_MAPPING_SIZE) };
222    if raw_ptr.is_null() {
223        // First OS allocation failed: release every retained free segment
224        // back to the OS to reclaim the address space / commit charge a
225        // transient working-set spike may be holding, then retry exactly
226        // once. Bounded to one purge and one retry: a still-exhausted OS
227        // fails fast on the second call rather than looping.
228        //
229        // SAFETY: segments retained in the pool at this point are
230        // exclusively pool-owned — a segment a thread-local allocator still
231        // references is "active", not "retained" — so releasing them
232        // invalidates no live pointer, and the retried `B::allocate` call
233        // follows the same contract as the first.
234        unsafe { super::release::purge_segment_pool::<B>() };
235        raw_ptr = unsafe { B::allocate(SEGMENT_MAPPING_SIZE) };
236        // Surfaces the recovery attempt and its outcome as a stats counter
237        // (`ArenaMemoryStats::oom_retries` / `oom_retry_successes`) — the
238        // crate has no tracing dependency, so telemetry follows the
239        // existing `purge_calls`/`reset_calls` counter convention.
240        B::global_segment_pool().record_oom_retry(!raw_ptr.is_null());
241        if raw_ptr.is_null() {
242            return None;
243        }
244    }
245
246    let numa_node = current_numa_node();
247    // SAFETY: `raw_ptr` is the non-null `SEGMENT_MAPPING_SIZE` mapping just
248    // returned by `B::allocate`, which is exclusively owned and writable —
249    // exactly `initialize_allocated_segment`'s contract.
250    let (aligned_ptr, aligned_addr, tail_slack_start, mapping_end) =
251        match unsafe { initialize_allocated_segment(raw_ptr, numa_node) } {
252            Some(val) => val,
253            None => {
254                // SAFETY: Releasing raw memory back to the backend because alignment check overflowed.
255                let _released = unsafe { B::deallocate(raw_ptr, SEGMENT_MAPPING_SIZE) };
256                return None;
257            }
258        };
259
260    // Bind the aligned segment mapping to the allocating thread's NUMA node.
261    // `mbind(MPOL_BIND)` enforces first-touch locality so pages are allocated
262    // from the correct socket rather than relying on the OS default policy.
263    // This call goes to the kernel only once per fresh segment (cold path);
264    // segments recycled from the pool retain their prior NUMA binding and never
265    // reach this function (they return early via `pop_free_segment`).
266    //
267    // SAFETY: `aligned_ptr` points to the segment-aligned base of the
268    // `SEGMENT_MAPPING_SIZE` mapping that is exclusively owned at this point
269    // (not yet published to any pool or thread), `SEGMENT_SIZE` is the
270    // usable extent of the aligned portion, and `numa_node` was read from
271    // Themis's thread-local cache immediately before this call.
272    unsafe {
273        crate::numa::bind_segment_to_numa_node(
274            raw_ptr.map_addr(|_| aligned_ptr as usize),
275            SEGMENT_SIZE,
276            numa_node,
277        )
278    };
279
280    #[cfg(feature = "segment-header-guards")]
281    {
282        if B::SUPPORTS_MAKE_GUARD {
283            // Install a header guard at the end of Page 0.
284            // Underflows (backward OOB writes) from Page 1 land in this guard region
285            // instead of overwriting the segment metadata at the start of Page 0.
286            //
287            // SAFETY: aligned_addr + PAGE_SIZE - SEGMENT_HEADER_GUARD_SIZE is inside the mapping
288            // and Page 0 is reserved strictly for Segment metadata (ending far before the guard).
289            let header_guard_addr = aligned_addr + PAGE_SIZE - SEGMENT_HEADER_GUARD_SIZE;
290            let header_guard = raw_ptr.map_addr(|_| header_guard_addr);
291            let _guarded = unsafe { B::make_guard(header_guard, SEGMENT_HEADER_GUARD_SIZE) };
292        }
293    }
294
295    #[cfg(feature = "segment-tail-guards")]
296    {
297        if B::SUPPORTS_MAKE_GUARD {
298            // Install a tail guard immediately after the segment's user-page
299            // region. Forward OOB writes that walk past Page 31 land in this
300            // guard region instead of an unrelated mapping. The address lives
301            // inside the `SEGMENT_MAPPING_SIZE - SEGMENT_SIZE` slack the arena
302            // reserves to satisfy `SEGMENT_ALIGN` rounding, so it is always
303            // part of the same backend allocation and is released together
304            // with the segment by `B::deallocate(raw_ptr, SEGMENT_MAPPING_SIZE)`.
305            // The install is best-effort: a backend without a `make_guard`
306            // implementation (default `false`) or a kernel that declines the
307            // request (e.g. macOS-arm64 where the OS page size exceeds 4 KiB)
308            // silently skips, leaving the slack accessible. Backend telemetry
309            // (`guard_install_calls`) surfaces the actual install count.
310            //
311            // SAFETY: aligned_addr + SEGMENT_SIZE is inside the raw mapping
312            // because slack-after >= OS_PAGE_SIZE >= SEGMENT_TAIL_GUARD_SIZE on
313            // supported targets. `make_guard` never invalidates the mapping.
314            let tail_guard_addr = aligned_addr + SEGMENT_SIZE;
315            let tail_guard = raw_ptr.map_addr(|_| tail_guard_addr);
316            let _guarded = unsafe { B::make_guard(tail_guard, SEGMENT_TAIL_GUARD_SIZE) };
317        }
318    }
319
320    // The mapping over-reserves `SEGMENT_MAPPING_SIZE = 2 * SEGMENT_SIZE` so a
321    // `SEGMENT_ALIGN`-aligned base can always be found; return the resulting
322    // head and tail slack to the OS (guard pages installed above lie outside
323    // both slack subranges, so ordering relative to the guards is immaterial).
324    //
325    // SAFETY: `[raw_ptr, aligned_addr)` precedes the header and
326    // `[tail_slack_start, mapping_end)` succeeds the segment pages (and tail
327    // guard, when enabled), so neither holds allocator data; both stay inside
328    // the live reservation covered by the base release, and the head bounds
329    // are page-aligned (`raw_ptr` from `allocate`, `aligned_addr` a
330    // `SEGMENT_ALIGN` multiple).
331    unsafe { decommit_mapping_slack::<B>(raw_ptr, aligned_addr, tail_slack_start, mapping_end) };
332
333    Some(aligned_ptr)
334}