moirai_utils/cache.rs
1//! Cache alignment utilities for performance optimization.
2//!
3//! This module is the single source of truth for the two cache granularities
4//! the workspace reasons about. They are *different numbers* on the same
5//! target and conflating them is a defect in both directions:
6//!
7//! * [`CACHE_LINE_SIZE`] — the coherence/transfer granularity of one line.
8//! Correct for prefetch strides and for sizing an iteration chunk so a block
9//! of elements lands in one line. 64 bytes on x86-64 and aarch64.
10//! * [`DESTRUCTIVE_INTERFERENCE_SIZE`] — the separation two independently
11//! written atomics need to avoid false sharing. On x86-64 and modern aarch64
12//! the adjacent-line prefetcher pulls lines in pairs, so two objects 64 bytes
13//! apart still ping-pong a shared 128-byte sector; separation is 128 bytes.
14//! Correct for padding and alignment. Never for chunk sizing — doubling a
15//! chunk width silently doubles a kernel's working set.
16//!
17//! The per-target values follow the table in `crossbeam-utils`'
18//! `CachePadded` (crossbeam-utils 0.8, `src/cache_padded.rs`), which documents
19//! the vendor sources for each architecture.
20
21/// Coherence and transfer granularity of a single cache line, in bytes.
22///
23/// Use for prefetch strides and for deriving how many elements of a type share
24/// one line. For separating concurrently written data, use
25/// [`DESTRUCTIVE_INTERFERENCE_SIZE`] instead — it is larger on the targets
26/// whose prefetcher operates on line pairs.
27pub const CACHE_LINE_SIZE: usize = if cfg!(any(
28 target_arch = "arm",
29 target_arch = "mips",
30 target_arch = "mips32r6",
31 target_arch = "mips64",
32 target_arch = "mips64r6",
33 target_arch = "sparc",
34 target_arch = "hexagon",
35)) {
36 32
37} else if cfg!(target_arch = "m68k") {
38 16
39} else if cfg!(target_arch = "s390x") {
40 256
41} else if cfg!(target_arch = "powerpc64") {
42 128
43} else {
44 // x86, x86-64, aarch64, riscv, loongarch, wasm and every target without a
45 // documented deviation.
46 64
47};
48
49/// Separation, in bytes, at which two concurrently written objects stop
50/// interfering — the C++ `hardware_destructive_interference_size`.
51///
52/// Larger than [`CACHE_LINE_SIZE`] wherever the hardware prefetcher fetches
53/// adjacent lines in pairs (x86-64, aarch64, powerpc64: 128 bytes), so padding
54/// to one line is not enough to stop the ping-pong. This is the value
55/// [`CacheAligned`] aligns to.
56pub const DESTRUCTIVE_INTERFERENCE_SIZE: usize = if cfg!(any(
57 target_arch = "x86_64",
58 target_arch = "aarch64",
59 target_arch = "powerpc64",
60)) {
61 // Intel/AMD L2 spatial ("adjacent line") prefetcher and Apple M-series
62 // both operate on 128-byte sectors.
63 128
64} else if cfg!(any(
65 target_arch = "arm",
66 target_arch = "mips",
67 target_arch = "mips32r6",
68 target_arch = "mips64",
69 target_arch = "mips64r6",
70 target_arch = "sparc",
71 target_arch = "hexagon",
72)) {
73 32
74} else if cfg!(target_arch = "m68k") {
75 16
76} else if cfg!(target_arch = "s390x") {
77 256
78} else {
79 64
80};
81
82// The `#[repr(align(..))]` attribute on `CacheAligned` takes a literal, so the
83// table above is mirrored in `cfg_attr` form below. These assertions make the
84// two representations impossible to drift apart: a target added to one table
85// and not the other fails the build.
86const _: () = assert!(core::mem::align_of::<CacheAligned<u8>>() == DESTRUCTIVE_INTERFERENCE_SIZE);
87const _: () = assert!(CACHE_LINE_SIZE <= DESTRUCTIVE_INTERFERENCE_SIZE);
88const _: () = assert!(CACHE_LINE_SIZE.is_power_of_two());
89const _: () = assert!(DESTRUCTIVE_INTERFERENCE_SIZE.is_power_of_two());
90
91/// Round a size up to the next cache-line boundary.
92///
93/// Uses the transfer granularity, not the false-sharing separation: this
94/// answers "how many lines does this occupy", not "how far apart must these
95/// live".
96#[must_use]
97pub const fn align_to_cache_line(size: usize) -> usize {
98 (size + CACHE_LINE_SIZE - 1) & !(CACHE_LINE_SIZE - 1)
99}
100
101/// A wrapper that pushes its value onto its own false-sharing sector.
102///
103/// Aligns the wrapped value to [`DESTRUCTIVE_INTERFERENCE_SIZE`] so an atomic
104/// written by one core cannot invalidate a neighbouring atomic written by
105/// another. Because a type's size is a multiple of its alignment, the wrapper
106/// also *occupies* that many bytes — which is the point: the padding after the
107/// value is what keeps the next field off this sector.
108///
109/// Structural traits are derived so the wrapper inherits them whenever the
110/// inner `T` supports them.
111#[cfg_attr(
112 any(
113 target_arch = "x86_64",
114 target_arch = "aarch64",
115 target_arch = "powerpc64",
116 ),
117 repr(align(128))
118)]
119#[cfg_attr(
120 any(
121 target_arch = "arm",
122 target_arch = "mips",
123 target_arch = "mips32r6",
124 target_arch = "mips64",
125 target_arch = "mips64r6",
126 target_arch = "sparc",
127 target_arch = "hexagon",
128 ),
129 repr(align(32))
130)]
131#[cfg_attr(target_arch = "m68k", repr(align(16)))]
132#[cfg_attr(target_arch = "s390x", repr(align(256)))]
133#[cfg_attr(
134 not(any(
135 target_arch = "x86_64",
136 target_arch = "aarch64",
137 target_arch = "powerpc64",
138 target_arch = "arm",
139 target_arch = "mips",
140 target_arch = "mips32r6",
141 target_arch = "mips64",
142 target_arch = "mips64r6",
143 target_arch = "sparc",
144 target_arch = "hexagon",
145 target_arch = "m68k",
146 target_arch = "s390x",
147 )),
148 repr(align(64))
149)]
150#[derive(Clone, Copy, Default, PartialEq, Eq, PartialOrd, Ord, Hash)]
151pub struct CacheAligned<T>(pub T);
152
153/// Zero-sized alignment marker.
154///
155/// Embedding one field of this type raises the containing struct's alignment
156/// to [`DESTRUCTIVE_INTERFERENCE_SIZE`] without adding a byte, so a struct that
157/// must start on its own sector single-sources the per-target table instead of
158/// repeating a `#[repr(align(..))]` literal.
159pub type CachePad = CacheAligned<()>;
160
161const _: () = assert!(core::mem::size_of::<CachePad>() == 0);
162
163impl<T> CacheAligned<T> {
164 /// Create a new cache-aligned value.
165 pub const fn new(value: T) -> Self {
166 Self(value)
167 }
168
169 /// Get a reference to the inner value.
170 pub const fn get(&self) -> &T {
171 &self.0
172 }
173
174 /// Get a mutable reference to the inner value.
175 pub fn get_mut(&mut self) -> &mut T {
176 &mut self.0
177 }
178}
179
180impl<T> core::ops::Deref for CacheAligned<T> {
181 type Target = T;
182
183 fn deref(&self) -> &Self::Target {
184 &self.0
185 }
186}
187
188impl<T> core::ops::DerefMut for CacheAligned<T> {
189 fn deref_mut(&mut self) -> &mut Self::Target {
190 &mut self.0
191 }
192}
193
194impl<T: core::fmt::Debug> core::fmt::Debug for CacheAligned<T> {
195 fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
196 f.debug_struct("CacheAligned")
197 .field("value", &self.0)
198 .finish()
199 }
200}
201
202#[cfg(test)]
203mod tests {
204 use super::*;
205
206 #[test]
207 fn align_to_cache_line_rounds_up_to_the_transfer_granularity() {
208 assert_eq!(align_to_cache_line(0), 0);
209 assert_eq!(align_to_cache_line(1), CACHE_LINE_SIZE);
210 assert_eq!(align_to_cache_line(CACHE_LINE_SIZE), CACHE_LINE_SIZE);
211 assert_eq!(
212 align_to_cache_line(CACHE_LINE_SIZE + 1),
213 CACHE_LINE_SIZE * 2
214 );
215 }
216
217 #[test]
218 fn cache_aligned_wrapper_forwards_to_the_inner_value() {
219 let aligned = CacheAligned::new(42);
220 assert_eq!(*aligned, 42);
221 assert_eq!(aligned.get(), &42);
222 }
223
224 #[test]
225 fn cache_aligned_separates_neighbours_by_the_interference_size() {
226 // Two wrapped atomics written by different cores must not share a
227 // sector: the byte distance between consecutive elements is the
228 // separation the wrapper promises.
229 let pair = [CacheAligned::new(0_u8), CacheAligned::new(0_u8)];
230 let first = std::ptr::addr_of!(pair[0].0) as usize;
231 let second = std::ptr::addr_of!(pair[1].0) as usize;
232 assert_eq!(second - first, DESTRUCTIVE_INTERFERENCE_SIZE);
233 }
234
235 #[test]
236 fn cache_pad_is_free_but_raises_alignment() {
237 struct Host {
238 _pad: CachePad,
239 value: u8,
240 }
241
242 assert_eq!(core::mem::size_of::<CachePad>(), 0);
243 assert_eq!(core::mem::align_of::<Host>(), DESTRUCTIVE_INTERFERENCE_SIZE);
244 let host = Host {
245 _pad: CacheAligned::new(()),
246 value: 7,
247 };
248 assert_eq!(host.value, 7);
249 }
250
251 /// The relation between the two constants is pinned by the `const _`
252 /// assertions at module scope; this pins the *values* on the targets where
253 /// a regression to a single 64-byte constant is the likely mistake.
254 #[test]
255 #[cfg(any(target_arch = "x86_64", target_arch = "aarch64"))]
256 fn line_and_interference_sizes_differ_on_this_target() {
257 assert_eq!(CACHE_LINE_SIZE, 64);
258 assert_eq!(DESTRUCTIVE_INTERFERENCE_SIZE, 128);
259 assert_eq!(
260 core::mem::align_of::<CacheAligned<u8>>(),
261 DESTRUCTIVE_INTERFERENCE_SIZE
262 );
263 }
264}