Skip to main content

corescout_substrate/observation/
counters.rs

1//! Hardware performance counters.
2//!
3//! | property | value |
4//! |---|---|
5//! | physical fact | cycles, retired instructions, cache misses and branch mispredictions actually executed by each core |
6//! | source | `perf_event_open(2)` in per-CPU counting mode |
7//! | sample rate | ~1 kHz; each read is one `read(2)` on an open descriptor |
8//! | cost | one syscall per counter per CPU per tick |
9//! | perturbation | **Low** |
10//! | uncertainty | exact within the event's own definition, which varies between microarchitectures |
11//!
12//! # This is the closest thing to direct hardware self-observation
13//!
14//! Every other sensor reads a number the kernel computed. These counters are
15//! read from the PMU registers of the core itself: the machine counting its own
16//! instructions as it executes them. If any observation deserves the word
17//! proprioception, it is this one.
18//!
19//! # Why it is nonetheless `Low` rather than `Negligible`
20//!
21//! Not because reading costs much, but because *holding* the counters does. A
22//! core has a small number of general-purpose PMU counters, typically four to
23//! eight. CoreScout occupying four of them means any other profiler on the
24//! machine gets time-multiplexed onto what is left, and multiplexed counts are
25//! scaled estimates rather than measurements.
26//!
27//! So this sensor degrades other observers' accuracy while improving its own.
28//! That is a genuine perturbation of the machine's observable state, even though
29//! no physical quantity changed, and it is exactly the kind of effect the
30//! perturbation declaration exists to surface.
31//!
32//! # Privilege and availability
33//!
34//! Per-CPU system-wide counting needs `perf_event_paranoid <= 0` or
35//! `CAP_PERFMON`. It is also usually unavailable inside VMs without a virtual
36//! PMU, and inside containers by default. All of these appear as a bind failure,
37//! which marks the sensor inactive and leaves its columns unobserved.
38//!
39//! # File descriptors
40//!
41//! Four events times the number of CPUs. On a large server that is several
42//! hundred descriptors held open for the lifetime of the mirror, which can
43//! exceed a default `RLIMIT_NOFILE` of 1024 on a 256-CPU machine. A partial
44//! open is treated as success for the CPUs that worked.
45
46use crate::observation::{
47    BindContext, Perturbation, Sensor, SensorDescriptor, SensorId, SensorOutcome, StateWriter,
48    Uncertainty,
49};
50use corescout_core::error::{Error, Result};
51#[cfg_attr(not(target_os = "linux"), allow(unused_imports))]
52use corescout_mirror::state::{ChannelId, Semantics, Unit};
53
54/// The events CoreScout counts, as `(channel key, PERF_COUNT_HW_* config)`.
55///
56/// Deliberately four: enough to characterise what a core is doing, few enough
57/// to fit in the general-purpose counters of every current x86 part without
58/// forcing multiplexing on itself.
59#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
60const EVENTS: [(&str, u64); 4] = [
61    ("cpu.pmu.cycles", 0),        // PERF_COUNT_HW_CPU_CYCLES
62    ("cpu.pmu.instructions", 1),  // PERF_COUNT_HW_INSTRUCTIONS
63    ("cpu.pmu.cache_misses", 3),  // PERF_COUNT_HW_CACHE_MISSES
64    ("cpu.pmu.branch_misses", 5), // PERF_COUNT_HW_BRANCH_MISSES
65];
66
67#[cfg_attr(not(target_os = "linux"), allow(dead_code))]
68pub struct CounterSensor {
69    /// `(entity row, event index, file descriptor)`.
70    counters: Vec<(u32, usize, RawFd)>,
71    channels: Vec<ChannelId>,
72}
73
74#[cfg(target_os = "linux")]
75type RawFd = std::os::unix::io::RawFd;
76#[cfg(not(target_os = "linux"))]
77type RawFd = i32;
78
79impl CounterSensor {
80    pub fn new() -> CounterSensor {
81        CounterSensor {
82            counters: Vec::new(),
83            channels: Vec::new(),
84        }
85    }
86}
87
88impl Default for CounterSensor {
89    fn default() -> Self {
90        Self::new()
91    }
92}
93
94impl Drop for CounterSensor {
95    fn drop(&mut self) {
96        #[cfg(target_os = "linux")]
97        for (_, _, fd) in &self.counters {
98            // SAFETY: each fd was returned by perf_event_open and is closed
99            // exactly once, here.
100            unsafe {
101                libc::close(*fd);
102            }
103        }
104    }
105}
106
107impl Sensor for CounterSensor {
108    fn descriptor(&self) -> SensorDescriptor {
109        SensorDescriptor {
110            id: SensorId(7),
111            key: "counters",
112            physical_fact: "cycles, retired instructions, cache misses and branch \
113                            mispredictions executed by each core since the counter was opened",
114            source: "perf_event_open(2), PERF_TYPE_HARDWARE, per-CPU counting mode",
115            max_rate_hz: 1000.0,
116            // Holding PMU counters degrades every other profiler on the machine;
117            // see the module docs.
118            perturbation: Perturbation::Low,
119            uncertainty: Uncertainty::unknown(
120                "counts are exact for the event as the microarchitecture defines it, but \
121                 event definitions differ between vendors and generations, so absolute \
122                 comparison across machines is not meaningful",
123            ),
124            requires_privilege: true,
125        }
126    }
127
128    fn bind(&mut self, ctx: &mut BindContext<'_>) -> Result<()> {
129        #[cfg(not(target_os = "linux"))]
130        {
131            let _ = ctx;
132            Err(Error::unsupported(
133                "hardware performance counters (perf_event_open is Linux only)",
134            ))
135        }
136
137        #[cfg(target_os = "linux")]
138        {
139            let cpus: Vec<u32> = ctx
140                .substrate()
141                .topology
142                .logical_cpus
143                .iter()
144                .filter(|c| c.online)
145                .map(|c| c.id)
146                .collect();
147
148            let mut counters = Vec::new();
149            for cpu in cpus {
150                let Some(row) = ctx.row_of(&corescout_mirror::entity::keys::logical_cpu(cpu))
151                else {
152                    continue;
153                };
154                for (index, (_, config)) in EVENTS.iter().enumerate() {
155                    match perf::open_counting_event(*config, cpu as i32) {
156                        Ok(fd) => counters.push((row, index, fd)),
157                        // A partial open is normal: descriptor limits, a CPU
158                        // that went offline, or an event this part does not
159                        // implement. Whatever opened, we use.
160                        Err(_) => continue,
161                    }
162                }
163            }
164
165            if counters.is_empty() {
166                return Err(Error::unsupported(
167                    "perf_event_open returned nothing usable (perf_event_paranoid may be \
168                     above 0, or this machine has no accessible PMU)",
169                ));
170            }
171
172            self.counters = counters;
173            for (key, _) in EVENTS {
174                self.channels
175                    .push(ctx.declare_channel(key, Unit::Count, Semantics::Cumulative));
176            }
177            Ok(())
178        }
179    }
180
181    fn observe(&mut self, out: &mut StateWriter<'_>) -> SensorOutcome {
182        #[allow(unused_mut)]
183        let mut outcome = SensorOutcome::default();
184        #[cfg(target_os = "linux")]
185        for (row, event, fd) in &self.counters {
186            match perf::read_counter(*fd) {
187                Some(value) => {
188                    if let Some(channel) = self.channels.get(*event) {
189                        out.set(*row, *channel, value as f64);
190                        outcome.sample();
191                    }
192                }
193                None => outcome.error(),
194            }
195        }
196        #[cfg(not(target_os = "linux"))]
197        let _ = out;
198        outcome
199    }
200}
201
202/// The `perf_event_open` ABI, defined here rather than taken from a binding.
203///
204/// The struct below is exactly `PERF_ATTR_SIZE_VER0`, the original 64-byte
205/// layout, and `size` is set to match. The kernel accepts any known size and
206/// zero-fills the rest, so using the oldest layout is both the simplest and the
207/// most portable option: it cannot drift with the kernel, and it avoids
208/// depending on a generated binding whose padding is not part of any stable
209/// contract.
210///
211/// Every flag bit is left zero, which is precisely the configuration wanted:
212/// the counter starts enabled, counts both user and kernel, and does not
213/// inherit across `fork`.
214#[cfg(target_os = "linux")]
215mod perf {
216    /// `PERF_TYPE_HARDWARE`.
217    const PERF_TYPE_HARDWARE: u32 = 0;
218    /// `PERF_FLAG_FD_CLOEXEC`: never leak counters into a child process.
219    const PERF_FLAG_FD_CLOEXEC: u64 = 8;
220
221    /// Size of the attribute struct, which is also the value of its `size`
222    /// field. Exposed so a test can assert the ABI has not drifted.
223    #[allow(dead_code)]
224    pub const ATTR_SIZE: usize = std::mem::size_of::<PerfEventAttrV0>();
225
226    #[repr(C)]
227    #[derive(Default)]
228    struct PerfEventAttrV0 {
229        type_: u32,
230        size: u32,
231        config: u64,
232        sample_period_or_freq: u64,
233        sample_type: u64,
234        read_format: u64,
235        /// The flag bitfield. All zero: enabled, not inherited, counting both
236        /// user and kernel time.
237        flags: u64,
238        wakeup: u32,
239        bp_type: u32,
240        config1: u64,
241    }
242
243    /// Open one counting event on one CPU, across all processes.
244    pub fn open_counting_event(config: u64, cpu: i32) -> Result<i32, i32> {
245        let attr = PerfEventAttrV0 {
246            type_: PERF_TYPE_HARDWARE,
247            size: std::mem::size_of::<PerfEventAttrV0>() as u32,
248            config,
249            ..Default::default()
250        };
251        debug_assert_eq!(std::mem::size_of::<PerfEventAttrV0>(), 64);
252
253        // SAFETY: `attr` is a valid, fully initialised struct of the size it
254        // declares. pid = -1 with a specific cpu means "all processes on this
255        // CPU", which is the system-wide counting mode.
256        let fd = unsafe {
257            libc::syscall(
258                libc::SYS_perf_event_open,
259                &attr as *const PerfEventAttrV0,
260                -1i32, // pid: all processes
261                cpu,   // this CPU only
262                -1i32, // group_fd: not in a group
263                PERF_FLAG_FD_CLOEXEC,
264            )
265        };
266        if fd < 0 {
267            return Err(std::io::Error::last_os_error().raw_os_error().unwrap_or(0));
268        }
269        Ok(fd as i32)
270    }
271
272    /// Read a counter's current value.
273    ///
274    /// With `read_format` left at zero, the kernel returns a single `u64`.
275    pub fn read_counter(fd: i32) -> Option<u64> {
276        let mut value: u64 = 0;
277        // SAFETY: reading 8 bytes into an 8-byte stack variable.
278        let read = unsafe {
279            libc::read(
280                fd,
281                &mut value as *mut u64 as *mut libc::c_void,
282                std::mem::size_of::<u64>(),
283            )
284        };
285        if read == std::mem::size_of::<u64>() as isize {
286            Some(value)
287        } else {
288            None
289        }
290    }
291}
292
293#[cfg(test)]
294mod tests {
295    use super::*;
296
297    #[test]
298    fn the_event_set_is_distinct_and_named() {
299        let mut keys: Vec<&str> = EVENTS.iter().map(|(k, _)| *k).collect();
300        let mut configs: Vec<u64> = EVENTS.iter().map(|(_, c)| *c).collect();
301        keys.sort_unstable();
302        configs.sort_unstable();
303        let unique_keys = {
304            let mut v = keys.clone();
305            v.dedup();
306            v.len()
307        };
308        let unique_configs = {
309            let mut v = configs.clone();
310            v.dedup();
311            v.len()
312        };
313        assert_eq!(unique_keys, EVENTS.len());
314        assert_eq!(unique_configs, EVENTS.len());
315    }
316
317    #[test]
318    fn the_event_set_fits_in_the_general_purpose_counters() {
319        // Four events is the budget: more would force the PMU to multiplex,
320        // which would make our own counts estimates rather than measurements.
321        assert!(EVENTS.len() <= 4, "the PMU budget was exceeded");
322    }
323
324    #[test]
325    #[cfg(target_os = "linux")]
326    fn the_attr_struct_matches_the_kernel_abi_size() {
327        // PERF_ATTR_SIZE_VER0. If this ever changes, the kernel will reject
328        // every open with EINVAL, so assert it here rather than discovering it
329        // at runtime on a machine we cannot debug.
330        assert_eq!(super::perf::ATTR_SIZE, 64);
331    }
332}