corescout_substrate/observation/counters.rs
1//! Hardware performance counters.
2//!
3//! | property | value |
4//! |---|---|
5//! | physical fact | cycles, retired instructions, cache misses and branch mispredictions actually executed by each core |
6//! | source | `perf_event_open(2)` in per-CPU counting mode |
7//! | sample rate | ~1 kHz; each read is one `read(2)` on an open descriptor |
8//! | cost | one syscall per counter per CPU per tick |
9//! | perturbation | **Low** |
10//! | uncertainty | exact within the event's own definition, which varies between microarchitectures |
11//!
12//! # This is the closest thing to direct hardware self-observation
13//!
14//! Every other sensor reads a number the kernel computed. These counters are
15//! read from the PMU registers of the core itself: the machine counting its own
16//! instructions as it executes them. If any observation deserves the word
17//! proprioception, it is this one.
18//!
19//! # Why it is nonetheless `Low` rather than `Negligible`
20//!
21//! Not because reading costs much, but because *holding* the counters does. A
22//! core has a small number of general-purpose PMU counters, typically four to
23//! eight. CoreScout occupying four of them means any other profiler on the
24//! machine gets time-multiplexed onto what is left, and multiplexed counts are
25//! scaled estimates rather than measurements.
26//!
27//! So this sensor degrades other observers' accuracy while improving its own.
28//! That is a genuine perturbation of the machine's observable state, even though
29//! no physical quantity changed, and it is exactly the kind of effect the
30//! perturbation declaration exists to surface.
31//!
32//! # Privilege and availability
33//!
34//! Per-CPU system-wide counting needs `perf_event_paranoid <= 0` or
35//! `CAP_PERFMON`. It is also usually unavailable inside VMs without a virtual
36//! PMU, and inside containers by default. All of these appear as a bind failure,
37//! which marks the sensor inactive and leaves its columns unobserved.
38//!
39//! # File descriptors
40//!
41//! Four events times the number of CPUs. On a large server that is several
42//! hundred descriptors held open for the lifetime of the mirror, which can
43//! exceed a default `RLIMIT_NOFILE` of 1024 on a 256-CPU machine. A partial
44//! open is treated as success for the CPUs that worked.
45
46use crate::observation::{
47 BindContext, Perturbation, Sensor, SensorDescriptor, SensorId, SensorOutcome, StateWriter,
48 Uncertainty,
49};
50use corescout_core::error::{Error, Result};
51#[cfg_attr(not(target_os = "linux"), allow(unused_imports))]
52use corescout_mirror::state::{ChannelId, Semantics, Unit};
53
54/// The events CoreScout counts, as `(channel key, PERF_COUNT_HW_* config)`.
55///
56/// Deliberately four: enough to characterise what a core is doing, few enough
57/// to fit in the general-purpose counters of every current x86 part without
58/// forcing multiplexing on itself.
59#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
60const EVENTS: [(&str, u64); 4] = [
61 ("cpu.pmu.cycles", 0), // PERF_COUNT_HW_CPU_CYCLES
62 ("cpu.pmu.instructions", 1), // PERF_COUNT_HW_INSTRUCTIONS
63 ("cpu.pmu.cache_misses", 3), // PERF_COUNT_HW_CACHE_MISSES
64 ("cpu.pmu.branch_misses", 5), // PERF_COUNT_HW_BRANCH_MISSES
65];
66
67#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
68pub struct CounterSensor {
69 /// `(entity row, event index, file descriptor)`.
70 counters: Vec<(u32, usize, RawFd)>,
71 channels: Vec<ChannelId>,
72}
73
74#[cfg(target_os = "linux")]
75type RawFd = std::os::unix::io::RawFd;
76#[cfg(not(target_os = "linux"))]
77type RawFd = i32;
78
79impl CounterSensor {
80 pub fn new() -> CounterSensor {
81 CounterSensor {
82 counters: Vec::new(),
83 channels: Vec::new(),
84 }
85 }
86}
87
88impl Default for CounterSensor {
89 fn default() -> Self {
90 Self::new()
91 }
92}
93
94impl Drop for CounterSensor {
95 fn drop(&mut self) {
96 #[cfg(target_os = "linux")]
97 for (_, _, fd) in &self.counters {
98 // SAFETY: each fd was returned by perf_event_open and is closed
99 // exactly once, here.
100 unsafe {
101 libc::close(*fd);
102 }
103 }
104 }
105}
106
107impl Sensor for CounterSensor {
108 fn descriptor(&self) -> SensorDescriptor {
109 SensorDescriptor {
110 id: SensorId(7),
111 key: "counters",
112 physical_fact: "cycles, retired instructions, cache misses and branch \
113 mispredictions executed by each core since the counter was opened",
114 source: "perf_event_open(2), PERF_TYPE_HARDWARE, per-CPU counting mode",
115 max_rate_hz: 1000.0,
116 // Holding PMU counters degrades every other profiler on the machine;
117 // see the module docs.
118 perturbation: Perturbation::Low,
119 uncertainty: Uncertainty::unknown(
120 "counts are exact for the event as the microarchitecture defines it, but \
121 event definitions differ between vendors and generations, so absolute \
122 comparison across machines is not meaningful",
123 ),
124 requires_privilege: true,
125 }
126 }
127
128 fn bind(&mut self, ctx: &mut BindContext<'_>) -> Result<()> {
129 #[cfg(not(target_os = "linux"))]
130 {
131 let _ = ctx;
132 Err(Error::unsupported(
133 "hardware performance counters (perf_event_open is Linux only)",
134 ))
135 }
136
137 #[cfg(target_os = "linux")]
138 {
139 let cpus: Vec<u32> = ctx
140 .substrate()
141 .topology
142 .logical_cpus
143 .iter()
144 .filter(|c| c.online)
145 .map(|c| c.id)
146 .collect();
147
148 let mut counters = Vec::new();
149 for cpu in cpus {
150 let Some(row) = ctx.row_of(&corescout_mirror::entity::keys::logical_cpu(cpu))
151 else {
152 continue;
153 };
154 for (index, (_, config)) in EVENTS.iter().enumerate() {
155 match perf::open_counting_event(*config, cpu as i32) {
156 Ok(fd) => counters.push((row, index, fd)),
157 // A partial open is normal: descriptor limits, a CPU
158 // that went offline, or an event this part does not
159 // implement. Whatever opened, we use.
160 Err(_) => continue,
161 }
162 }
163 }
164
165 if counters.is_empty() {
166 return Err(Error::unsupported(
167 "perf_event_open returned nothing usable (perf_event_paranoid may be \
168 above 0, or this machine has no accessible PMU)",
169 ));
170 }
171
172 self.counters = counters;
173 for (key, _) in EVENTS {
174 self.channels
175 .push(ctx.declare_channel(key, Unit::Count, Semantics::Cumulative));
176 }
177 Ok(())
178 }
179 }
180
181 fn observe(&mut self, out: &mut StateWriter<'_>) -> SensorOutcome {
182 #[allow(unused_mut)]
183 let mut outcome = SensorOutcome::default();
184 #[cfg(target_os = "linux")]
185 for (row, event, fd) in &self.counters {
186 match perf::read_counter(*fd) {
187 Some(value) => {
188 if let Some(channel) = self.channels.get(*event) {
189 out.set(*row, *channel, value as f64);
190 outcome.sample();
191 }
192 }
193 None => outcome.error(),
194 }
195 }
196 #[cfg(not(target_os = "linux"))]
197 let _ = out;
198 outcome
199 }
200}
201
202/// The `perf_event_open` ABI, defined here rather than taken from a binding.
203///
204/// The struct below is exactly `PERF_ATTR_SIZE_VER0`, the original 64-byte
205/// layout, and `size` is set to match. The kernel accepts any known size and
206/// zero-fills the rest, so using the oldest layout is both the simplest and the
207/// most portable option: it cannot drift with the kernel, and it avoids
208/// depending on a generated binding whose padding is not part of any stable
209/// contract.
210///
211/// Every flag bit is left zero, which is precisely the configuration wanted:
212/// the counter starts enabled, counts both user and kernel, and does not
213/// inherit across `fork`.
214#[cfg(target_os = "linux")]
215mod perf {
216 /// `PERF_TYPE_HARDWARE`.
217 const PERF_TYPE_HARDWARE: u32 = 0;
218 /// `PERF_FLAG_FD_CLOEXEC`: never leak counters into a child process.
219 const PERF_FLAG_FD_CLOEXEC: u64 = 8;
220
221 /// Size of the attribute struct, which is also the value of its `size`
222 /// field. Exposed so a test can assert the ABI has not drifted.
223 #[allow(dead_code)]
224 pub const ATTR_SIZE: usize = std::mem::size_of::<PerfEventAttrV0>();
225
226 #[repr(C)]
227 #[derive(Default)]
228 struct PerfEventAttrV0 {
229 type_: u32,
230 size: u32,
231 config: u64,
232 sample_period_or_freq: u64,
233 sample_type: u64,
234 read_format: u64,
235 /// The flag bitfield. All zero: enabled, not inherited, counting both
236 /// user and kernel time.
237 flags: u64,
238 wakeup: u32,
239 bp_type: u32,
240 config1: u64,
241 }
242
243 /// Open one counting event on one CPU, across all processes.
244 pub fn open_counting_event(config: u64, cpu: i32) -> Result<i32, i32> {
245 let attr = PerfEventAttrV0 {
246 type_: PERF_TYPE_HARDWARE,
247 size: std::mem::size_of::<PerfEventAttrV0>() as u32,
248 config,
249 ..Default::default()
250 };
251 debug_assert_eq!(std::mem::size_of::<PerfEventAttrV0>(), 64);
252
253 // SAFETY: `attr` is a valid, fully initialised struct of the size it
254 // declares. pid = -1 with a specific cpu means "all processes on this
255 // CPU", which is the system-wide counting mode.
256 let fd = unsafe {
257 libc::syscall(
258 libc::SYS_perf_event_open,
259 &attr as *const PerfEventAttrV0,
260 -1i32, // pid: all processes
261 cpu, // this CPU only
262 -1i32, // group_fd: not in a group
263 PERF_FLAG_FD_CLOEXEC,
264 )
265 };
266 if fd < 0 {
267 return Err(std::io::Error::last_os_error().raw_os_error().unwrap_or(0));
268 }
269 Ok(fd as i32)
270 }
271
272 /// Read a counter's current value.
273 ///
274 /// With `read_format` left at zero, the kernel returns a single `u64`.
275 pub fn read_counter(fd: i32) -> Option<u64> {
276 let mut value: u64 = 0;
277 // SAFETY: reading 8 bytes into an 8-byte stack variable.
278 let read = unsafe {
279 libc::read(
280 fd,
281 &mut value as *mut u64 as *mut libc::c_void,
282 std::mem::size_of::<u64>(),
283 )
284 };
285 if read == std::mem::size_of::<u64>() as isize {
286 Some(value)
287 } else {
288 None
289 }
290 }
291}
292
293#[cfg(test)]
294mod tests {
295 use super::*;
296
297 #[test]
298 fn the_event_set_is_distinct_and_named() {
299 let mut keys: Vec<&str> = EVENTS.iter().map(|(k, _)| *k).collect();
300 let mut configs: Vec<u64> = EVENTS.iter().map(|(_, c)| *c).collect();
301 keys.sort_unstable();
302 configs.sort_unstable();
303 let unique_keys = {
304 let mut v = keys.clone();
305 v.dedup();
306 v.len()
307 };
308 let unique_configs = {
309 let mut v = configs.clone();
310 v.dedup();
311 v.len()
312 };
313 assert_eq!(unique_keys, EVENTS.len());
314 assert_eq!(unique_configs, EVENTS.len());
315 }
316
317 #[test]
318 fn the_event_set_fits_in_the_general_purpose_counters() {
319 // Four events is the budget: more would force the PMU to multiplex,
320 // which would make our own counts estimates rather than measurements.
321 assert!(EVENTS.len() <= 4, "the PMU budget was exceeded");
322 }
323
324 #[test]
325 #[cfg(target_os = "linux")]
326 fn the_attr_struct_matches_the_kernel_abi_size() {
327 // PERF_ATTR_SIZE_VER0. If this ever changes, the kernel will reject
328 // every open with EINVAL, so assert it here rather than discovering it
329 // at runtime on a machine we cannot debug.
330 assert_eq!(super::perf::ATTR_SIZE, 64);
331 }
332}