1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
use melinoe::sync::SyncRegionToken;
use melinoe::{CellSliceExt, MelinoeMut, MelinoeRef};
use super::cell::{ConstPinnedCell, ConstPinnedSlice, NumaPinnedSlice, PinnedCell, PinnedSlice};
use super::static_cell::ConstNumaPinnedSlice;
/// A node-specific placement capability.
pub struct NumaNodePlacement<'brand> {
pub(super) node_id: crate::NumaNodeId,
pub(super) token: SyncRegionToken<'brand>,
}
impl<'brand> NumaNodePlacement<'brand> {
/// Returns the assigned NUMA node ID.
#[must_use]
pub const fn node_id(&self) -> crate::NumaNodeId {
self.node_id
}
/// Reads state from a cell pinned to the same NUMA node.
#[inline]
pub fn read<'a, C, T>(&'a self, cell: &'a C) -> Option<MelinoeRef<'a, 'brand, T>>
where
C: PinnedCell<'brand, T> + ?Sized,
{
if self.node_id == cell.node_id() {
Some(cell.cell().borrow(&self.token))
} else {
None
}
}
/// Writes state to a cell pinned to the same NUMA node.
#[inline]
pub fn write<'a, C, T>(&'a mut self, cell: &'a C) -> Option<MelinoeMut<'a, 'brand, T>>
where
C: PinnedCell<'brand, T> + ?Sized,
{
if self.node_id == cell.node_id() {
Some(cell.cell().borrow_mut(&mut self.token))
} else {
None
}
}
/// Reads a slice pinned to the same NUMA node.
#[inline]
pub fn read_slice<'a, S, T>(&'a self, slice: &'a S) -> Option<&'a [T]>
where
S: PinnedSlice<'brand, T> + ?Sized,
{
if self.node_id == slice.node_id() {
Some(slice.cells().borrow_slice(&self.token))
} else {
None
}
}
/// Writes to a slice pinned to the same NUMA node.
#[inline]
pub fn write_slice<'a, S, T>(&'a mut self, slice: &'a S) -> Option<&'a mut [T]>
where
S: PinnedSlice<'brand, T> + ?Sized,
{
if self.node_id == slice.node_id() {
Some(slice.cells().borrow_slice_mut(&mut self.token))
} else {
None
}
}
/// Mutate disjoint portions of a matching, uniquely owned pinned slice
/// through Melinoe's parallel partition driver.
///
/// The mutable slice borrow establishes unique ownership; this permit adds
/// the dynamic NUMA-node validation before Melinoe executes the shards.
///
/// # Execution is delegated, and the driver is chosen process-wide
///
/// Themis owns the locality law and explicitly does not own scheduling, so
/// the work is handed to Melinoe's partition driver and *where it runs* is
/// decided by whichever executor is registered process-wide at call time.
/// Two cases:
///
/// - **An executor is registered** (for example because the scheduling
/// layer has been initialized): the shards run on that pool and no OS
/// threads are created per call. This is the intended path for
/// NUMA-placement workloads, which are fine-grained enough that
/// per-call thread creation dominates.
/// - **No executor is registered**: Melinoe falls back to spawning and
/// joining `min(parts, len) − 1` OS threads, at roughly 30 µs per shard,
/// with no affinity applied. The parallel call is then usually slower
/// than the serial equivalent for small regions, and binding decisions
/// are not honoured, so this path is not a substitute for the pool.
///
/// Registration is process-global and read per call, and scheduling layers
/// typically register lazily on first access — so initialize the scheduler
/// before partitioning rather than relying on it being primed by luck.
/// See `melinoe::sync::register_parallel_executor`.
#[cfg(feature = "std")]
pub fn partition_for_each_mut_with<T, F>(
&mut self,
slice: &mut NumaPinnedSlice<'brand, T>,
plan: melinoe::sync::PartitionPlan,
f: F,
) -> Option<()>
where
T: Send,
F: Fn(usize, &mut [T]) + Sync,
{
if self.node_id != slice.node_id() {
return None;
}
melinoe::sync::partition_for_each_with(slice.cells_mut(), plan, |start, mut shard| {
f(start, shard.as_mut_slice());
});
Some(())
}
}
/// A node-specific placement capability, verified at compile time.
pub struct ConstNumaNodePlacement<'brand, const NODE_ID: u32> {
pub(super) token: SyncRegionToken<'brand>,
}
impl<'brand, const NODE_ID: u32> ConstNumaNodePlacement<'brand, NODE_ID> {
/// Returns the assigned NUMA node ID.
#[must_use]
pub const fn node_id(&self) -> u32 {
NODE_ID
}
/// Reads state from a cell pinned statically to the same NUMA node.
#[inline]
pub fn read<'a, C, T>(&'a self, cell: &'a C) -> MelinoeRef<'a, 'brand, T>
where
C: ConstPinnedCell<'brand, NODE_ID, T> + ?Sized,
{
cell.cell().borrow(&self.token)
}
/// Writes state to a cell pinned statically to the same NUMA node.
#[inline]
pub fn write<'a, C, T>(&'a mut self, cell: &'a C) -> MelinoeMut<'a, 'brand, T>
where
C: ConstPinnedCell<'brand, NODE_ID, T> + ?Sized,
{
cell.cell().borrow_mut(&mut self.token)
}
/// Reads a slice pinned statically to the same NUMA node.
#[inline]
pub fn read_slice<'a, S, T>(&'a self, slice: &'a S) -> &'a [T]
where
S: ConstPinnedSlice<'brand, NODE_ID, T> + ?Sized,
{
slice.cells().borrow_slice(&self.token)
}
/// Writes to a slice pinned statically to the same NUMA node.
#[inline]
pub fn write_slice<'a, S, T>(&'a mut self, slice: &'a S) -> &'a mut [T]
where
S: ConstPinnedSlice<'brand, NODE_ID, T> + ?Sized,
{
slice.cells().borrow_slice_mut(&mut self.token)
}
/// Mutate disjoint portions of a statically matching, uniquely owned
/// pinned slice through Melinoe's parallel partition driver.
///
/// The mutable slice borrow establishes unique ownership; the const-generic
/// permit supplies the compile-time NUMA placement identity.
///
/// As with [`NumaNodePlacement::partition_for_each_mut_with`], execution is
/// delegated: with a process-wide executor registered the shards run on
/// that pool, and without one Melinoe falls back to per-call OS-thread
/// creation (~30 µs per shard, no affinity), which is usually slower than
/// serial work for small regions. Initialize the scheduler before
/// partitioning rather than relying on registration having happened.
#[cfg(feature = "std")]
pub fn partition_for_each_mut_with<T, F>(
&mut self,
slice: &mut ConstNumaPinnedSlice<'brand, NODE_ID, T>,
plan: melinoe::sync::PartitionPlan,
f: F,
) where
T: Send,
F: Fn(usize, &mut [T]) + Sync,
{
melinoe::sync::partition_for_each_with(slice.cells_mut(), plan, |start, mut shard| {
f(start, shard.as_mut_slice());
});
}
}