Skip to main content

corescout_substrate/
topology.rs

1//! The topology data model.
2
3use std::collections::BTreeMap;
4
5use serde::{Deserialize, Serialize};
6
7pub use corescout_core::{LogicalId, PhysicalId};
8
9/// Which class of core this is on a hybrid (big.LITTLE style) processor.
10#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
11#[serde(rename_all = "snake_case")]
12pub enum CoreType {
13    /// Performance core: Intel "Core"/P-core, AMD classic Zen core.
14    Performance,
15    /// Efficiency core: Intel "Atom"/E-core, AMD Zen-c dense core.
16    Efficiency,
17    /// Non-hybrid part, or a hybrid part whose class we could not determine.
18    Unknown,
19}
20
21impl CoreType {
22    /// One-character tag used in table output.
23    pub fn short(&self) -> &'static str {
24        match self {
25            CoreType::Performance => "P",
26            CoreType::Efficiency => "E",
27            CoreType::Unknown => "-",
28        }
29    }
30}
31
32/// Level of a cache in the memory hierarchy.
33#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Serialize, Deserialize)]
34#[serde(rename_all = "snake_case")]
35pub enum CacheKind {
36    L1Data,
37    L1Instruction,
38    L2Unified,
39    L3Unified,
40    /// Anything else the kernel reports (L4/eDRAM, unified L1, ...).
41    Other,
42}
43
44/// One cache instance, plus the set of logical CPUs that share it.
45///
46/// Sharing is the load-bearing field. Two threads that share an L2 contend for
47/// it; two threads that *want* to share data are much cheaper to co-locate
48/// under one L3 than to spread across sockets.
49#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
50pub struct Cache {
51    pub level: u8,
52    pub kind: CacheKind,
53    /// Size in bytes. `None` when the kernel did not expose it.
54    pub size_bytes: Option<u64>,
55    pub line_size_bytes: Option<u32>,
56    pub ways_of_associativity: Option<u32>,
57    /// Logical CPUs sharing this exact cache instance, ascending.
58    pub shared_cpus: Vec<LogicalId>,
59}
60
61/// Per-CPU frequency information, in kHz, as reported by cpufreq.
62#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
63pub struct FrequencyInfo {
64    pub current_khz: Option<u64>,
65    pub min_khz: Option<u64>,
66    pub max_khz: Option<u64>,
67    /// Base (non-turbo) frequency where the platform exposes it.
68    pub base_khz: Option<u64>,
69}
70
71/// A manufacturer preferred/favored-core hint.
72///
73/// Intel exposes Turbo Boost Max 3.0 ordering through ACPI CPPC
74/// (`highest_perf`); AMD's Preferred Core works the same way. Both are
75/// *factory bin-sort* results: they describe the silicon as it left the fab,
76/// not how the core behaves in this chassis, at this ambient temperature, with
77/// this interrupt routing and this neighbour thread. CoreScout records the hint
78/// precisely so it can tell you when measurement disagrees with it.
79#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
80pub struct FavoredHint {
81    /// Raw ACPI CPPC "highest performance" value. Higher means the firmware
82    /// believes this core boosts further.
83    pub highest_perf: u32,
84    pub nominal_perf: Option<u32>,
85    /// Rank among all CPUs by `highest_perf`, 1 = the firmware's favourite.
86    pub rank: u32,
87    /// Where the value came from, e.g. `acpi_cppc/highest_perf`.
88    pub source: String,
89}
90
91/// A single logical CPU (an SMT thread, or a whole core on a non-SMT part).
92#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
93pub struct LogicalCpu {
94    pub id: LogicalId,
95    /// Synthetic machine-wide physical core id.
96    pub physical: PhysicalId,
97    /// Socket / package this CPU lives in.
98    pub package_id: u32,
99    /// Kernel `core_id` (unique only within a package).
100    pub core_id: u32,
101    /// NUMA node, when the kernel exposes NUMA at all.
102    pub numa_node: Option<u32>,
103    /// Other logical CPUs on the same physical core, excluding `self.id`.
104    pub smt_siblings: Vec<LogicalId>,
105    pub core_type: CoreType,
106    pub online: bool,
107    pub frequency: FrequencyInfo,
108    /// Manufacturer "this is one of the good ones" hint, if the OS exposes it.
109    pub favored: Option<FavoredHint>,
110}
111
112impl LogicalCpu {
113    /// True when this CPU shares its physical core with another logical CPU.
114    pub fn is_smt(&self) -> bool {
115        !self.smt_siblings.is_empty()
116    }
117}
118
119/// A physical core: one or more logical CPUs sharing execution resources.
120#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
121pub struct PhysicalCore {
122    pub id: PhysicalId,
123    pub package_id: u32,
124    pub core_id: u32,
125    pub numa_node: Option<u32>,
126    pub core_type: CoreType,
127    /// Logical CPUs belonging to this core, ascending. The first entry is the
128    /// one CoreScout benchmarks on.
129    pub logical_cpus: Vec<LogicalId>,
130}
131
132impl PhysicalCore {
133    /// The logical CPU CoreScout pins to when measuring this core.
134    ///
135    /// Which sibling we pick is arbitrary but must be *stable*, otherwise
136    /// re-running the benchmark silently changes what was measured.
137    pub fn primary_cpu(&self) -> LogicalId {
138        self.logical_cpus[0]
139    }
140}
141
142/// A NUMA node and the CPUs attached to it.
143#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
144pub struct NumaNode {
145    pub id: u32,
146    pub cpus: Vec<LogicalId>,
147    /// Total memory on the node in kB, when available.
148    pub memory_kb: Option<u64>,
149}
150
151/// Whole-machine CPU description.
152#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
153pub struct Topology {
154    pub model_name: String,
155    pub vendor: String,
156    /// True when the part mixes core types (Intel hybrid, AMD dense/classic).
157    pub hybrid: bool,
158    pub logical_cpus: Vec<LogicalCpu>,
159    pub physical_cores: Vec<PhysicalCore>,
160    pub caches: Vec<Cache>,
161    pub numa_nodes: Vec<NumaNode>,
162    /// CPUs the kernel currently reports as online.
163    pub online_cpus: Vec<LogicalId>,
164    /// CPUs present but offlined.
165    pub offline_cpus: Vec<LogicalId>,
166    /// Affinity mask of the CoreScout process itself at discovery time. This
167    /// matters more than it looks: inside a container or under a cpuset, most
168    /// of the machine may be off limits, and benchmarking cores we cannot
169    /// actually be scheduled on would produce nonsense.
170    pub process_affinity: Vec<LogicalId>,
171}
172
173impl Topology {
174    pub fn logical_count(&self) -> usize {
175        self.logical_cpus.len()
176    }
177
178    pub fn physical_count(&self) -> usize {
179        self.physical_cores.len()
180    }
181
182    /// True when at least one physical core carries more than one logical CPU.
183    pub fn smt_enabled(&self) -> bool {
184        self.physical_cores.iter().any(|c| c.logical_cpus.len() > 1)
185    }
186
187    /// SMT width, i.e. threads per core. 1 when SMT is off or absent.
188    pub fn smt_width(&self) -> usize {
189        self.physical_cores
190            .iter()
191            .map(|c| c.logical_cpus.len())
192            .max()
193            .unwrap_or(1)
194    }
195
196    pub fn cpu(&self, id: LogicalId) -> Option<&LogicalCpu> {
197        self.logical_cpus.iter().find(|c| c.id == id)
198    }
199
200    pub fn core(&self, id: PhysicalId) -> Option<&PhysicalCore> {
201        self.physical_cores.iter().find(|c| c.id == id)
202    }
203
204    /// The physical core owning a given logical CPU.
205    pub fn core_of_cpu(&self, id: LogicalId) -> Option<&PhysicalCore> {
206        let cpu = self.cpu(id)?;
207        self.core(cpu.physical)
208    }
209
210    /// Caches at `level` that the logical CPU participates in.
211    pub fn caches_for_cpu(&self, id: LogicalId, level: u8) -> Vec<&Cache> {
212        self.caches
213            .iter()
214            .filter(|c| c.level == level && c.shared_cpus.contains(&id))
215            .collect()
216    }
217
218    /// True when the two logical CPUs share a cache at `level`.
219    pub fn shares_cache(&self, a: LogicalId, b: LogicalId, level: u8) -> bool {
220        self.caches
221            .iter()
222            .any(|c| c.level == level && c.shared_cpus.contains(&a) && c.shared_cpus.contains(&b))
223    }
224
225    /// Physical cores that are online *and* inside the process affinity mask:
226    /// the cores CoreScout is actually allowed to benchmark.
227    pub fn benchmarkable_cores(&self) -> Vec<&PhysicalCore> {
228        self.physical_cores
229            .iter()
230            .filter(|core| {
231                core.logical_cpus.iter().any(|cpu| {
232                    self.online_cpus.contains(cpu) && self.process_affinity.contains(cpu)
233                })
234            })
235            .collect()
236    }
237
238    /// Count of cores per type, for hybrid parts.
239    pub fn cores_by_type(&self) -> BTreeMap<&'static str, usize> {
240        let mut map = BTreeMap::new();
241        for core in &self.physical_cores {
242            let label = match core.core_type {
243                CoreType::Performance => "P-core",
244                CoreType::Efficiency => "E-core",
245                CoreType::Unknown => "core",
246            };
247            *map.entry(label).or_insert(0) += 1;
248        }
249        map
250    }
251
252    /// The firmware's favourite CPU, if any hint was found.
253    pub fn firmware_favored_cpu(&self) -> Option<&LogicalCpu> {
254        self.logical_cpus
255            .iter()
256            .filter(|c| c.favored.is_some())
257            .min_by_key(|c| c.favored.as_ref().map(|f| f.rank).unwrap_or(u32::MAX))
258    }
259}
260
261#[cfg(test)]
262mod tests {
263    use super::*;
264    use crate::test_support::fake_topology;
265
266    #[test]
267    fn smt_detection() {
268        let t = fake_topology();
269        assert!(t.smt_enabled());
270        assert_eq!(t.smt_width(), 2);
271        assert_eq!(t.physical_count(), 2);
272        assert_eq!(t.logical_count(), 4);
273    }
274
275    #[test]
276    fn cache_sharing_follows_the_hierarchy() {
277        let t = fake_topology();
278        // SMT siblings share their L2.
279        assert!(t.shares_cache(0, 2, 2));
280        // Different physical cores do not.
281        assert!(!t.shares_cache(0, 1, 2));
282        // ...but they do share the package L3.
283        assert!(t.shares_cache(0, 1, 3));
284    }
285
286    #[test]
287    fn benchmarkable_cores_respect_affinity_and_offline() {
288        let mut t = fake_topology();
289        t.process_affinity = vec![0, 2];
290        t.online_cpus = vec![0, 2];
291        let cores = t.benchmarkable_cores();
292        assert_eq!(cores.len(), 1);
293        assert_eq!(cores[0].id, 0);
294    }
295
296    #[test]
297    fn core_lookup_by_logical_cpu() {
298        let t = fake_topology();
299        assert_eq!(t.core_of_cpu(3).unwrap().id, 1);
300        assert_eq!(t.core(1).unwrap().primary_cpu(), 1);
301    }
302
303    #[test]
304    fn topology_round_trips_through_json() {
305        let t = fake_topology();
306        let s = serde_json::to_string(&t).unwrap();
307        let back: Topology = serde_json::from_str(&s).unwrap();
308        assert_eq!(t, back);
309    }
310}