Skip to main content

moirai_utils/
cache.rs

1//! Cache alignment utilities for performance optimization.
2//!
3//! This module is the single source of truth for the two cache granularities
4//! the workspace reasons about. They are *different numbers* on the same
5//! target and conflating them is a defect in both directions:
6//!
7//! * [`CACHE_LINE_SIZE`] — the coherence/transfer granularity of one line.
8//!   Correct for prefetch strides and for sizing an iteration chunk so a block
9//!   of elements lands in one line. 64 bytes on x86-64 and aarch64.
10//! * [`DESTRUCTIVE_INTERFERENCE_SIZE`] — the separation two independently
11//!   written atomics need to avoid false sharing. On x86-64 and modern aarch64
12//!   the adjacent-line prefetcher pulls lines in pairs, so two objects 64 bytes
13//!   apart still ping-pong a shared 128-byte sector; separation is 128 bytes.
14//!   Correct for padding and alignment. Never for chunk sizing — doubling a
15//!   chunk width silently doubles a kernel's working set.
16//!
17//! The per-target values follow the table in `crossbeam-utils`'
18//! `CachePadded` (crossbeam-utils 0.8, `src/cache_padded.rs`), which documents
19//! the vendor sources for each architecture.
20
21/// Coherence and transfer granularity of a single cache line, in bytes.
22///
23/// Use for prefetch strides and for deriving how many elements of a type share
24/// one line. For separating concurrently written data, use
25/// [`DESTRUCTIVE_INTERFERENCE_SIZE`] instead — it is larger on the targets
26/// whose prefetcher operates on line pairs.
27pub const CACHE_LINE_SIZE: usize = if cfg!(any(
28    target_arch = "arm",
29    target_arch = "mips",
30    target_arch = "mips32r6",
31    target_arch = "mips64",
32    target_arch = "mips64r6",
33    target_arch = "sparc",
34    target_arch = "hexagon",
35)) {
36    32
37} else if cfg!(target_arch = "m68k") {
38    16
39} else if cfg!(target_arch = "s390x") {
40    256
41} else if cfg!(target_arch = "powerpc64") {
42    128
43} else {
44    // x86, x86-64, aarch64, riscv, loongarch, wasm and every target without a
45    // documented deviation.
46    64
47};
48
49/// Separation, in bytes, at which two concurrently written objects stop
50/// interfering — the C++ `hardware_destructive_interference_size`.
51///
52/// Larger than [`CACHE_LINE_SIZE`] wherever the hardware prefetcher fetches
53/// adjacent lines in pairs (x86-64, aarch64, powerpc64: 128 bytes), so padding
54/// to one line is not enough to stop the ping-pong. This is the value
55/// [`CacheAligned`] aligns to.
56pub const DESTRUCTIVE_INTERFERENCE_SIZE: usize = if cfg!(any(
57    target_arch = "x86_64",
58    target_arch = "aarch64",
59    target_arch = "powerpc64",
60)) {
61    // Intel/AMD L2 spatial ("adjacent line") prefetcher and Apple M-series
62    // both operate on 128-byte sectors.
63    128
64} else if cfg!(any(
65    target_arch = "arm",
66    target_arch = "mips",
67    target_arch = "mips32r6",
68    target_arch = "mips64",
69    target_arch = "mips64r6",
70    target_arch = "sparc",
71    target_arch = "hexagon",
72)) {
73    32
74} else if cfg!(target_arch = "m68k") {
75    16
76} else if cfg!(target_arch = "s390x") {
77    256
78} else {
79    64
80};
81
82// The `#[repr(align(..))]` attribute on `CacheAligned` takes a literal, so the
83// table above is mirrored in `cfg_attr` form below. These assertions make the
84// two representations impossible to drift apart: a target added to one table
85// and not the other fails the build.
86const _: () = assert!(core::mem::align_of::<CacheAligned<u8>>() == DESTRUCTIVE_INTERFERENCE_SIZE);
87const _: () = assert!(CACHE_LINE_SIZE <= DESTRUCTIVE_INTERFERENCE_SIZE);
88const _: () = assert!(CACHE_LINE_SIZE.is_power_of_two());
89const _: () = assert!(DESTRUCTIVE_INTERFERENCE_SIZE.is_power_of_two());
90
91/// Round a size up to the next cache-line boundary.
92///
93/// Uses the transfer granularity, not the false-sharing separation: this
94/// answers "how many lines does this occupy", not "how far apart must these
95/// live".
96#[must_use]
97pub const fn align_to_cache_line(size: usize) -> usize {
98    (size + CACHE_LINE_SIZE - 1) & !(CACHE_LINE_SIZE - 1)
99}
100
101/// A wrapper that pushes its value onto its own false-sharing sector.
102///
103/// Aligns the wrapped value to [`DESTRUCTIVE_INTERFERENCE_SIZE`] so an atomic
104/// written by one core cannot invalidate a neighbouring atomic written by
105/// another. Because a type's size is a multiple of its alignment, the wrapper
106/// also *occupies* that many bytes — which is the point: the padding after the
107/// value is what keeps the next field off this sector.
108///
109/// Structural traits are derived so the wrapper inherits them whenever the
110/// inner `T` supports them.
111#[cfg_attr(
112    any(
113        target_arch = "x86_64",
114        target_arch = "aarch64",
115        target_arch = "powerpc64",
116    ),
117    repr(align(128))
118)]
119#[cfg_attr(
120    any(
121        target_arch = "arm",
122        target_arch = "mips",
123        target_arch = "mips32r6",
124        target_arch = "mips64",
125        target_arch = "mips64r6",
126        target_arch = "sparc",
127        target_arch = "hexagon",
128    ),
129    repr(align(32))
130)]
131#[cfg_attr(target_arch = "m68k", repr(align(16)))]
132#[cfg_attr(target_arch = "s390x", repr(align(256)))]
133#[cfg_attr(
134    not(any(
135        target_arch = "x86_64",
136        target_arch = "aarch64",
137        target_arch = "powerpc64",
138        target_arch = "arm",
139        target_arch = "mips",
140        target_arch = "mips32r6",
141        target_arch = "mips64",
142        target_arch = "mips64r6",
143        target_arch = "sparc",
144        target_arch = "hexagon",
145        target_arch = "m68k",
146        target_arch = "s390x",
147    )),
148    repr(align(64))
149)]
150#[derive(Clone, Copy, Default, PartialEq, Eq, PartialOrd, Ord, Hash)]
151pub struct CacheAligned<T>(pub T);
152
153/// Zero-sized alignment marker.
154///
155/// Embedding one field of this type raises the containing struct's alignment
156/// to [`DESTRUCTIVE_INTERFERENCE_SIZE`] without adding a byte, so a struct that
157/// must start on its own sector single-sources the per-target table instead of
158/// repeating a `#[repr(align(..))]` literal.
159pub type CachePad = CacheAligned<()>;
160
161const _: () = assert!(core::mem::size_of::<CachePad>() == 0);
162
163impl<T> CacheAligned<T> {
164    /// Create a new cache-aligned value.
165    pub const fn new(value: T) -> Self {
166        Self(value)
167    }
168
169    /// Get a reference to the inner value.
170    pub const fn get(&self) -> &T {
171        &self.0
172    }
173
174    /// Get a mutable reference to the inner value.
175    pub fn get_mut(&mut self) -> &mut T {
176        &mut self.0
177    }
178}
179
180impl<T> core::ops::Deref for CacheAligned<T> {
181    type Target = T;
182
183    fn deref(&self) -> &Self::Target {
184        &self.0
185    }
186}
187
188impl<T> core::ops::DerefMut for CacheAligned<T> {
189    fn deref_mut(&mut self) -> &mut Self::Target {
190        &mut self.0
191    }
192}
193
194impl<T: core::fmt::Debug> core::fmt::Debug for CacheAligned<T> {
195    fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
196        f.debug_struct("CacheAligned")
197            .field("value", &self.0)
198            .finish()
199    }
200}
201
202#[cfg(test)]
203mod tests {
204    use super::*;
205
206    #[test]
207    fn align_to_cache_line_rounds_up_to_the_transfer_granularity() {
208        assert_eq!(align_to_cache_line(0), 0);
209        assert_eq!(align_to_cache_line(1), CACHE_LINE_SIZE);
210        assert_eq!(align_to_cache_line(CACHE_LINE_SIZE), CACHE_LINE_SIZE);
211        assert_eq!(
212            align_to_cache_line(CACHE_LINE_SIZE + 1),
213            CACHE_LINE_SIZE * 2
214        );
215    }
216
217    #[test]
218    fn cache_aligned_wrapper_forwards_to_the_inner_value() {
219        let aligned = CacheAligned::new(42);
220        assert_eq!(*aligned, 42);
221        assert_eq!(aligned.get(), &42);
222    }
223
224    #[test]
225    fn cache_aligned_separates_neighbours_by_the_interference_size() {
226        // Two wrapped atomics written by different cores must not share a
227        // sector: the byte distance between consecutive elements is the
228        // separation the wrapper promises.
229        let pair = [CacheAligned::new(0_u8), CacheAligned::new(0_u8)];
230        let first = std::ptr::addr_of!(pair[0].0) as usize;
231        let second = std::ptr::addr_of!(pair[1].0) as usize;
232        assert_eq!(second - first, DESTRUCTIVE_INTERFERENCE_SIZE);
233    }
234
235    #[test]
236    fn cache_pad_is_free_but_raises_alignment() {
237        struct Host {
238            _pad: CachePad,
239            value: u8,
240        }
241
242        assert_eq!(core::mem::size_of::<CachePad>(), 0);
243        assert_eq!(core::mem::align_of::<Host>(), DESTRUCTIVE_INTERFERENCE_SIZE);
244        let host = Host {
245            _pad: CacheAligned::new(()),
246            value: 7,
247        };
248        assert_eq!(host.value, 7);
249    }
250
251    /// The relation between the two constants is pinned by the `const _`
252    /// assertions at module scope; this pins the *values* on the targets where
253    /// a regression to a single 64-byte constant is the likely mistake.
254    #[test]
255    #[cfg(any(target_arch = "x86_64", target_arch = "aarch64"))]
256    fn line_and_interference_sizes_differ_on_this_target() {
257        assert_eq!(CACHE_LINE_SIZE, 64);
258        assert_eq!(DESTRUCTIVE_INTERFERENCE_SIZE, 128);
259        assert_eq!(
260            core::mem::align_of::<CacheAligned<u8>>(),
261            DESTRUCTIVE_INTERFERENCE_SIZE
262        );
263    }
264}