Skip to main content

mnemosyne_local/bin_stats/
api.rs

1//! Public counter-reading API and internal helpers for [`super`]'s
2//! per-size-class allocation telemetry.
3
4use core::sync::atomic::{AtomicU64, Ordering};
5use mnemosyne_core::constants::NUM_SIZE_CLASSES;
6use mnemosyne_core::size_class::class_to_size;
7
8use super::batch;
9use super::snapshot::BinSnapshot;
10
11/// Flushes the TLS batch then sums every entry of a global counter array.
12///
13/// Both [`total_alloc_count`] and [`total_requested_bytes`] share this body —
14/// the only difference between them is which array is summed.
15#[inline]
16fn sum_counter(arr: &[AtomicU64; NUM_SIZE_CLASSES]) -> u64 {
17    batch::flush_current_thread();
18    arr.iter()
19        .map(|c| c.load(Ordering::Relaxed))
20        .fold(0u64, u64::saturating_add)
21}
22
23/// Constructs a [`BinSnapshot`] for a single size class.
24///
25/// Non-generic SSOT shared by [`bin_snapshot`] and [`all_bin_snapshots`],
26/// eliminating the 5-line construction that previously appeared in both.
27/// Callers must flush the per-thread batch counters first.
28#[inline(always)]
29fn make_bin_snapshot(class: usize) -> BinSnapshot {
30    let alloc_count = batch::ALLOC_COUNT[class].load(Ordering::Relaxed);
31    let dealloc_count = batch::DEALLOC_COUNT[class].load(Ordering::Relaxed);
32    let block_size = class_to_size(class);
33    BinSnapshot {
34        alloc_count,
35        dealloc_count,
36        alloc_bytes: batch::allocation_bytes(alloc_count, block_size),
37        requested_bytes: batch::REQUESTED_BYTES[class].load(Ordering::Relaxed),
38        block_size,
39        live_estimate: alloc_count.saturating_sub(dealloc_count),
40    }
41}
42
43/// Returns a snapshot for size class `class`, or `None` if out of range.
44#[must_use]
45pub fn bin_snapshot(class: usize) -> Option<BinSnapshot> {
46    if class >= NUM_SIZE_CLASSES {
47        return None;
48    }
49    batch::flush_current_thread();
50    Some(make_bin_snapshot(class))
51}
52
53/// Returns snapshots for all `NUM_SIZE_CLASSES` size classes.
54#[must_use]
55pub fn all_bin_snapshots() -> [BinSnapshot; NUM_SIZE_CLASSES] {
56    batch::flush_current_thread();
57    core::array::from_fn(make_bin_snapshot)
58}
59
60/// Returns the index of the hottest size class (highest alloc_count), or
61/// `None` when nothing has ever been allocated.
62#[must_use]
63pub fn hottest_class() -> Option<usize> {
64    let snapshots = all_bin_snapshots();
65    snapshots
66        .iter()
67        .enumerate()
68        .max_by_key(|(_, s)| s.alloc_count)
69        .and_then(|(idx, s)| if s.alloc_count > 0 { Some(idx) } else { None })
70}
71
72/// Process-wide live bytes in the small allocator: sum of `live_estimate ×
73/// block_size` across all classes.
74#[must_use]
75pub fn total_live_bytes() -> u64 {
76    all_bin_snapshots()
77        .iter()
78        .map(|s| s.live_bytes())
79        .fold(0u64, u64::saturating_add)
80}
81
82/// Process-wide total allocation count across all small size classes.
83#[must_use]
84pub fn total_alloc_count() -> u64 {
85    sum_counter(&batch::ALLOC_COUNT)
86}
87
88/// Resets all per-class counters to zero.
89///
90/// Useful for marking the start of a profiling window so subsequent
91/// snapshots reflect only activity since the reset.
92pub fn reset_bin_stats() {
93    // Advance the reset generation BEFORE zeroing, so any concurrent
94    // flush that reads the new generation discards its batch rather than
95    // flushing stale pre-reset counts that would be immediately zeroed.
96    //
97    // Relaxed is the whole requirement: the flush guard compares this one
98    // variable against its own stamped value, and single-variable
99    // modification-order coherence -- which Relaxed already guarantees --
100    // is what makes that comparison monotonic. No happens-before edge with
101    // the counters is needed, because every counter access is Relaxed and a
102    // flush that still reads the old generation only adds counts the
103    // zeroing below immediately erases. A stronger ordering here would not
104    // change what the Relaxed loads at the guard can observe.
105    batch::RESET_GENERATION.fetch_add(1, Ordering::Relaxed);
106    batch::flush_current_thread();
107    for class in 0..NUM_SIZE_CLASSES {
108        batch::ALLOC_COUNT[class].store(0, Ordering::Relaxed);
109        batch::DEALLOC_COUNT[class].store(0, Ordering::Relaxed);
110        batch::REQUESTED_BYTES[class].store(0, Ordering::Relaxed);
111    }
112}
113
114/// Flushes the calling thread's pending bin-stats batch to the global counters.
115///
116/// `bin_snapshot` and `all_bin_snapshots` call this automatically; invoke
117/// it explicitly before reading from a different thread.
118#[inline]
119pub fn flush_tls_stats() {
120    batch::flush_current_thread();
121}
122
123/// One-line human-readable summary of process-wide bin stats.
124#[must_use]
125pub fn summary_line() -> std::string::String {
126    let total_allocs = total_alloc_count();
127    let live = total_live_bytes();
128    let int_frag = total_internal_fragmentation();
129    match hottest_class() {
130        Some(cls) => std::format!(
131            "allocs={total_allocs} live_bytes={live} int_frag={int_frag:.1}% hottest_class={cls}({}b)",
132            class_to_size(cls)
133        ),
134        None => std::format!(
135            "allocs={total_allocs} live_bytes={live} int_frag={int_frag:.1}% hottest_class=none"
136        ),
137    }
138}
139
140/// Process-wide cumulative user-requested bytes across all small size classes.
141///
142/// Zero until `record_alloc_with_size` call sites are wired (done in Phase 19).
143#[must_use]
144pub fn total_requested_bytes() -> u64 {
145    sum_counter(&batch::REQUESTED_BYTES)
146}
147
148/// Process-wide internal fragmentation ratio:
149/// `(total_alloc_bytes - total_requested_bytes) / total_alloc_bytes`.
150///
151/// Returns `0.0` when `total_alloc_bytes == 0` or `total_requested_bytes == 0`
152/// (e.g. before any `record_alloc_with_size` calls).
153#[must_use]
154pub fn total_internal_fragmentation() -> f64 {
155    let snapshots = all_bin_snapshots();
156    let alloc: u64 = snapshots
157        .iter()
158        .map(|s| s.alloc_bytes)
159        .fold(0, u64::saturating_add);
160    let requested: u64 = snapshots
161        .iter()
162        .map(|s| s.requested_bytes)
163        .fold(0, u64::saturating_add);
164    if alloc == 0 || requested == 0 {
165        return 0.0;
166    }
167    let waste = alloc.saturating_sub(requested);
168    (waste as f64 / alloc as f64).min(1.0)
169}
170
171/// Returns the current value of the reset generation counter.
172///
173/// Increments monotonically on every [`reset_bin_stats`] call. Useful
174/// for asserting that a snapshot spans exactly one reset-free window.
175#[inline]
176#[must_use]
177pub fn reset_generation_count() -> u32 {
178    batch::RESET_GENERATION.load(Ordering::Relaxed)
179}
180
181/// Returns the fractional distribution of `alloc_count` across all size
182/// classes as an array of `f64` values in `[0.0, 1.0]` that sum to 1.0.
183///
184/// The `i`-th element is `alloc_count[i] / total_alloc_count`. If no
185/// allocations have been recorded, every element is `0.0`.
186///
187/// Useful for understanding which size classes dominate the workload.
188#[must_use]
189pub fn alloc_distribution() -> [f64; NUM_SIZE_CLASSES] {
190    let snapshots = all_bin_snapshots();
191    let total: u64 = snapshots
192        .iter()
193        .map(|s| s.alloc_count)
194        .fold(0, u64::saturating_add);
195    if total == 0 {
196        return [0.0; NUM_SIZE_CLASSES];
197    }
198    let mut dist = [0.0f64; NUM_SIZE_CLASSES];
199    for (i, s) in snapshots.iter().enumerate() {
200        dist[i] = s.alloc_count as f64 / total as f64;
201    }
202    dist
203}
204
205#[cfg(test)]
206mod tests {
207    use core::sync::atomic::AtomicU64;
208    use std::sync::{Arc, Barrier};
209
210    use super::batch::{FLUSH_BATCH, PendingCount, RESET_GENERATION, record_alloc_with_size};
211    use super::{
212        NUM_SIZE_CLASSES, all_bin_snapshots, bin_snapshot, class_to_size, flush_tls_stats,
213        reset_bin_stats, reset_generation_count,
214    };
215
216    #[test]
217    fn pending_counts_flush_without_dropping_observations() {
218        let global = [const { AtomicU64::new(0) }; NUM_SIZE_CLASSES];
219        let mut pending = PendingCount::new();
220
221        for _ in 0..FLUSH_BATCH {
222            pending.record(3, &global);
223        }
224
225        assert_eq!(
226            global[3].load(core::sync::atomic::Ordering::Relaxed),
227            FLUSH_BATCH as u64
228        );
229        assert_eq!(pending.count, 0);
230    }
231
232    #[test]
233    fn generation_counter_discards_stale_batches() {
234        // Build a pending slot with the CURRENT generation.
235        let global = [const { AtomicU64::new(0) }; NUM_SIZE_CLASSES];
236        let mut pending = PendingCount::new();
237        // Prime the slot so `generation` is stamped.
238        pending.record(2, &global);
239        // Advance the global generation (simulates a reset_bin_stats call).
240        RESET_GENERATION.fetch_add(1, core::sync::atomic::Ordering::Relaxed);
241        // Flush: the batch is stale and must be discarded.
242        pending.flush(&global);
243        // The global array must not have received the stale count.
244        assert_eq!(
245            global[2].load(core::sync::atomic::Ordering::Relaxed),
246            0,
247            "stale batch must not flush after generation advance"
248        );
249        // Undo the generation increment to avoid interfering with other tests.
250        RESET_GENERATION.fetch_sub(1, core::sync::atomic::Ordering::Relaxed);
251    }
252
253    /// Multi-threaded boundary proof for MN-BIN-STATS-RESET-BOUNDARY.
254    ///
255    /// Verifies that `reset_bin_stats` advances the generation counter, which
256    /// is the mechanism that makes it a true profiling boundary: TLS batches
257    /// stamped with an older generation are discarded on flush rather than
258    /// added to the fresh post-reset counters. The single-thread property is
259    /// proven by `generation_counter_discards_stale_batches`; this test
260    /// confirms the public API increments the counter monotonically so the
261    /// generation-based discard is actually triggered.
262    #[test]
263    fn reset_bin_stats_monotonically_advances_generation() {
264        let gen_before = reset_generation_count();
265        reset_bin_stats();
266        let gen_after = reset_generation_count();
267        assert!(
268            gen_after > gen_before,
269            "reset_bin_stats must advance the generation counter: \
270             before={gen_before} after={gen_after}"
271        );
272        // Post-reset: all class counters must be zero (this thread has no live
273        // pending batch since `reset_bin_stats` also flushes the calling thread).
274        for (class, snap) in all_bin_snapshots().iter().enumerate() {
275            assert_eq!(
276                snap.alloc_count, 0,
277                "class {class} alloc_count must be zero immediately after reset"
278            );
279            assert_eq!(
280                snap.dealloc_count, 0,
281                "class {class} dealloc_count must be zero immediately after reset"
282            );
283        }
284    }
285
286    #[test]
287    fn derived_allocation_bytes_match_the_class_stride() {
288        for snapshot in all_bin_snapshots() {
289            assert_eq!(
290                snapshot.alloc_bytes,
291                snapshot
292                    .alloc_count
293                    .saturating_mul(snapshot.block_size as u64),
294                "allocation bytes must be derived from the immutable class stride"
295            );
296        }
297    }
298
299    #[test]
300    fn snapshots_preserve_the_public_range_contract() {
301        assert!(bin_snapshot(NUM_SIZE_CLASSES).is_none());
302    }
303
304    /// Cross-thread boundary proof for MN-BIN-STATS-RESET-BOUNDARY: a batch
305    /// accumulated on another thread before `reset_bin_stats` runs must not
306    /// leak into the post-reset counters once that thread finally flushes.
307    #[test]
308    fn reset_excludes_a_batch_pending_on_another_thread() {
309        const CLASS: usize = 5;
310        // Both counts stay below FLUSH_BATCH, so neither batch flushes on its
311        // own and only the explicit flushes move the global counter.
312        const PRE_RESET: u32 = FLUSH_BATCH / 2;
313        const POST_RESET: u32 = FLUSH_BATCH / 4;
314
315        let barrier = Arc::new(Barrier::new(2));
316        let worker = {
317            let barrier = Arc::clone(&barrier);
318            std::thread::spawn(move || {
319                for _ in 0..PRE_RESET {
320                    record_alloc_with_size(CLASS, class_to_size(CLASS));
321                }
322                barrier.wait(); // pre-reset batch is pending
323                barrier.wait(); // reset has completed
324                flush_tls_stats();
325                for _ in 0..POST_RESET {
326                    record_alloc_with_size(CLASS, class_to_size(CLASS));
327                }
328                flush_tls_stats();
329            })
330        };
331
332        barrier.wait();
333        reset_bin_stats();
334        barrier.wait();
335        worker.join().expect("worker thread panicked");
336
337        let snapshot = bin_snapshot(CLASS).expect("invariant: CLASS < NUM_SIZE_CLASSES");
338        assert_eq!(
339            snapshot.alloc_count,
340            u64::from(POST_RESET),
341            "post-reset total must exclude the {PRE_RESET} allocations pending on the worker"
342        );
343        assert_eq!(
344            snapshot.requested_bytes,
345            u64::from(POST_RESET) * class_to_size(CLASS) as u64,
346        );
347    }
348}