Skip to main content

corescout_substrate/platform/linux/
sysfs.rs

1//! Topology discovery from Linux `sysfs` and `procfs`.
2//!
3//! # Why sysfs and not CPUID
4//!
5//! CPUID describes the silicon; sysfs describes the *machine as the kernel has
6//! configured it*. Only the kernel knows which CPUs are offline, how a cpuset
7//! or container has restricted us, what cpufreq policy is in force, and how
8//! caches are actually shared after firmware fusing. CoreScout therefore treats
9//! sysfs as authoritative and uses CPUID (see [`crate::platform::cpuid`]) only
10//! to fill gaps such as the brand string.
11//!
12//! # Root injection
13//!
14//! [`Sysfs`] holds the filesystem roots rather than hardcoding `/sys` and
15//! `/proc`. That is not an abstraction for its own sake: it lets the parser be
16//! exercised against captured sysfs trees in unit and integration tests, on any
17//! host OS, including machines with topologies nobody on the project owns.
18//!
19//! This module is compiled on every platform precisely so those tests can run
20//! everywhere; only the syscall layer next door is Linux-only.
21
22use std::collections::{BTreeMap, BTreeSet};
23use std::path::{Path, PathBuf};
24
25use crate::topology::{
26    Cache, CacheKind, CoreType, FavoredHint, FrequencyInfo, LogicalCpu, LogicalId, NumaNode,
27    PhysicalCore, Topology,
28};
29use corescout_core::cpuset::CpuSet;
30use corescout_core::error::{Error, Result};
31
32/// Reader for a (possibly simulated) `/sys` + `/proc` pair.
33#[derive(Debug, Clone)]
34pub struct Sysfs {
35    sys_root: PathBuf,
36    proc_root: PathBuf,
37}
38
39impl Default for Sysfs {
40    fn default() -> Self {
41        Self::system()
42    }
43}
44
45impl Sysfs {
46    /// The real filesystem.
47    pub fn system() -> Self {
48        Sysfs {
49            sys_root: PathBuf::from("/sys"),
50            proc_root: PathBuf::from("/proc"),
51        }
52    }
53
54    /// A simulated tree, used by tests and by `--sysroot` for debugging a
55    /// captured topology from another machine.
56    pub fn with_roots(sys_root: impl Into<PathBuf>, proc_root: impl Into<PathBuf>) -> Self {
57        Sysfs {
58            sys_root: sys_root.into(),
59            proc_root: proc_root.into(),
60        }
61    }
62
63    fn cpu_dir(&self) -> PathBuf {
64        self.sys_root.join("devices/system/cpu")
65    }
66
67    fn node_dir(&self) -> PathBuf {
68        self.sys_root.join("devices/system/node")
69    }
70
71    /// Build the complete [`Topology`].
72    ///
73    /// `process_affinity` is passed in rather than read here because it comes
74    /// from a syscall, not a file. When `None`, the online set is assumed,
75    /// which is what a fresh process without an inherited mask sees.
76    pub fn read_topology(&self, process_affinity: Option<CpuSet>) -> Result<Topology> {
77        let present = self.read_cpu_list("present")?;
78        let online = self.read_cpu_list("online")?;
79        // `offline` is absent on kernels built without CPU hotplug.
80        let offline = self.read_cpu_list("offline").unwrap_or_default();
81
82        if present.is_empty() {
83            return Err(Error::parse(
84                self.cpu_dir().join("present"),
85                "kernel reported no present CPUs",
86            ));
87        }
88
89        let hybrid_map = self.read_hybrid_core_types();
90        let caches = self.read_caches(&online)?;
91
92        // Pass 1: read per-CPU attributes and collect the distinct
93        // (package, core) pairs so physical ids can be assigned densely.
94        struct Raw {
95            id: LogicalId,
96            package_id: u32,
97            core_id: u32,
98            siblings: Vec<LogicalId>,
99            online: bool,
100            frequency: FrequencyInfo,
101            cppc: Option<(u32, Option<u32>)>,
102        }
103
104        let mut raws = Vec::new();
105        for cpu in present.iter() {
106            let is_online = online.contains(cpu);
107            let dir = self.cpu_dir().join(format!("cpu{cpu}"));
108            // An offline CPU has no topology/ directory at all, so its package
109            // and core ids are simply unknown. We keep it in the CPU list
110            // (users want to see it) but it cannot be grouped or benchmarked.
111            let (package_id, core_id, siblings) = if is_online {
112                let package_id = read_u32(dir.join("topology/physical_package_id")).unwrap_or(0);
113                let core_id = read_u32(dir.join("topology/core_id")).unwrap_or(cpu);
114                let siblings = self.read_thread_siblings(&dir, cpu)?;
115                (package_id, core_id, siblings)
116            } else {
117                (u32::MAX, cpu, vec![cpu])
118            };
119
120            raws.push(Raw {
121                id: cpu,
122                package_id,
123                core_id,
124                siblings,
125                online: is_online,
126                frequency: self.read_frequency(&dir),
127                cppc: self.read_cppc(&dir),
128            });
129        }
130
131        // Assign machine-wide physical core ids. Linux `core_id` is unique only
132        // within a package, so a two-socket box would otherwise collide core 0
133        // of socket 0 with core 0 of socket 1.
134        let mut pairs: Vec<(u32, u32)> = raws.iter().map(|r| (r.package_id, r.core_id)).collect();
135        pairs.sort_unstable();
136        pairs.dedup();
137        let phys_index: BTreeMap<(u32, u32), u32> = pairs
138            .iter()
139            .enumerate()
140            .map(|(i, pair)| (*pair, i as u32))
141            .collect();
142
143        let numa_nodes = self.read_numa_nodes()?;
144        let numa_of_cpu = |cpu: LogicalId| -> Option<u32> {
145            numa_nodes
146                .iter()
147                .find(|n| n.cpus.contains(&cpu))
148                .map(|n| n.id)
149        };
150
151        // Pass 2: rank the CPPC hints so the "firmware's favourite" is explicit
152        // rather than something the user has to eyeball out of raw numbers.
153        let mut perf_values: Vec<u32> = raws.iter().filter_map(|r| r.cppc.map(|c| c.0)).collect();
154        perf_values.sort_unstable_by(|a, b| b.cmp(a));
155        perf_values.dedup();
156
157        let mut logical_cpus = Vec::with_capacity(raws.len());
158        for raw in &raws {
159            let physical = *phys_index
160                .get(&(raw.package_id, raw.core_id))
161                .expect("every raw CPU contributed its pair to the index");
162            let smt_siblings: Vec<LogicalId> = raw
163                .siblings
164                .iter()
165                .copied()
166                .filter(|s| *s != raw.id)
167                .collect();
168            let favored = raw.cppc.map(|(highest, nominal)| FavoredHint {
169                highest_perf: highest,
170                nominal_perf: nominal,
171                rank: perf_values
172                    .iter()
173                    .position(|v| *v == highest)
174                    .map(|p| p as u32 + 1)
175                    .unwrap_or(u32::MAX),
176                source: "acpi_cppc/highest_perf".to_string(),
177            });
178
179            logical_cpus.push(LogicalCpu {
180                id: raw.id,
181                physical,
182                package_id: raw.package_id,
183                core_id: raw.core_id,
184                numa_node: numa_of_cpu(raw.id),
185                smt_siblings,
186                core_type: hybrid_map
187                    .get(&raw.id)
188                    .copied()
189                    .unwrap_or(CoreType::Unknown),
190                online: raw.online,
191                frequency: raw.frequency,
192                favored,
193            });
194        }
195        logical_cpus.sort_by_key(|c| c.id);
196
197        // Group logical CPUs into physical cores. Offline CPUs are excluded:
198        // a core we cannot schedule on is not a core we can rank.
199        let mut cores: BTreeMap<u32, PhysicalCore> = BTreeMap::new();
200        for cpu in logical_cpus.iter().filter(|c| c.online) {
201            let entry = cores.entry(cpu.physical).or_insert_with(|| PhysicalCore {
202                id: cpu.physical,
203                package_id: cpu.package_id,
204                core_id: cpu.core_id,
205                numa_node: cpu.numa_node,
206                core_type: cpu.core_type,
207                logical_cpus: Vec::new(),
208            });
209            entry.logical_cpus.push(cpu.id);
210            if entry.core_type == CoreType::Unknown {
211                entry.core_type = cpu.core_type;
212            }
213        }
214        let mut physical_cores: Vec<PhysicalCore> = cores.into_values().collect();
215        for core in &mut physical_cores {
216            core.logical_cpus.sort_unstable();
217        }
218
219        let (vendor, model_name) = self.read_cpu_identity();
220        let hybrid = physical_cores
221            .iter()
222            .any(|c| c.core_type == CoreType::Efficiency)
223            && physical_cores
224                .iter()
225                .any(|c| c.core_type == CoreType::Performance);
226
227        let process_affinity = process_affinity.unwrap_or_else(|| online.clone());
228
229        Ok(Topology {
230            model_name,
231            vendor,
232            hybrid,
233            logical_cpus,
234            physical_cores,
235            caches,
236            numa_nodes,
237            online_cpus: online.to_vec(),
238            offline_cpus: offline.to_vec(),
239            process_affinity: process_affinity.to_vec(),
240        })
241    }
242
243    /// Read one of the `cpu/{present,online,offline,possible}` masks.
244    fn read_cpu_list(&self, name: &str) -> Result<CpuSet> {
245        let path = self.cpu_dir().join(name);
246        let raw = read_string(&path)?;
247        CpuSet::parse_list(&raw).map_err(|e| Error::parse(&path, e.to_string()))
248    }
249
250    /// Thread siblings, i.e. the logical CPUs sharing this physical core.
251    ///
252    /// `core_cpus_list` is the modern name (5.3+); `thread_siblings_list` is
253    /// the legacy one. Both may be missing on kernels without
254    /// `CONFIG_SCHED_SMT` or inside restricted containers, in which case the
255    /// CPU is treated as its own core, which is the correct conservative
256    /// answer: we would rather under-report SMT than pin two benchmark threads
257    /// onto one physical core believing they are independent.
258    fn read_thread_siblings(&self, dir: &Path, cpu: LogicalId) -> Result<Vec<LogicalId>> {
259        for name in ["topology/core_cpus_list", "topology/thread_siblings_list"] {
260            let path = dir.join(name);
261            match read_string(&path) {
262                Ok(raw) => {
263                    let set =
264                        CpuSet::parse_list(&raw).map_err(|e| Error::parse(&path, e.to_string()))?;
265                    if !set.is_empty() {
266                        return Ok(set.to_vec());
267                    }
268                }
269                Err(e) if e.is_not_found() => continue,
270                Err(e) => return Err(e),
271            }
272        }
273        Ok(vec![cpu])
274    }
275
276    fn read_frequency(&self, dir: &Path) -> FrequencyInfo {
277        let cpufreq = dir.join("cpufreq");
278        FrequencyInfo {
279            // `scaling_cur_freq` is the governor's request;
280            // `cpuinfo_cur_freq` is the measured value but needs privileges on
281            // many systems, so the former is the pragmatic default.
282            current_khz: read_u64(cpufreq.join("scaling_cur_freq"))
283                .or_else(|| read_u64(cpufreq.join("cpuinfo_cur_freq"))),
284            min_khz: read_u64(cpufreq.join("cpuinfo_min_freq"))
285                .or_else(|| read_u64(cpufreq.join("scaling_min_freq"))),
286            max_khz: read_u64(cpufreq.join("cpuinfo_max_freq"))
287                .or_else(|| read_u64(cpufreq.join("scaling_max_freq"))),
288            // intel_pstate and amd_pstate expose the non-turbo base clock here.
289            base_khz: read_u64(cpufreq.join("base_frequency")),
290        }
291    }
292
293    /// ACPI CPPC performance hints: the kernel's view of the manufacturer's
294    /// preferred-core ordering (Intel Turbo Boost Max 3.0, AMD Preferred Core).
295    fn read_cppc(&self, dir: &Path) -> Option<(u32, Option<u32>)> {
296        let cppc = dir.join("acpi_cppc");
297        let highest = read_u32(cppc.join("highest_perf"))?;
298        Some((highest, read_u32(cppc.join("nominal_perf"))))
299    }
300
301    /// Hybrid core classification from the per-core-type PMU devices that the
302    /// kernel creates on Intel hybrid parts (`cpu_core` = P, `cpu_atom` = E).
303    ///
304    /// This is preferred over CPUID leaf 0x1A because it needs no pinning and
305    /// works for CPUs we are not allowed to run on.
306    fn read_hybrid_core_types(&self) -> BTreeMap<LogicalId, CoreType> {
307        let mut map = BTreeMap::new();
308        let sources = [
309            ("devices/cpu_core/cpus", CoreType::Performance),
310            ("devices/cpu_atom/cpus", CoreType::Efficiency),
311        ];
312        for (rel, kind) in sources {
313            if let Ok(raw) = read_string(self.sys_root.join(rel)) {
314                if let Ok(set) = CpuSet::parse_list(&raw) {
315                    for cpu in set.iter() {
316                        map.insert(cpu, kind);
317                    }
318                }
319            }
320        }
321        map
322    }
323
324    /// All cache instances, deduplicated by their sharing set.
325    ///
326    /// sysfs lists every cache once *per participating CPU*, so an L3 shared by
327    /// 32 CPUs appears 32 times. We key on (level, kind, shared set) to collapse
328    /// those into one instance, which is what makes "which cores share this
329    /// cache" answerable.
330    fn read_caches(&self, online: &CpuSet) -> Result<Vec<Cache>> {
331        let mut seen: BTreeSet<(u8, CacheKind, Vec<LogicalId>)> = BTreeSet::new();
332        let mut caches = Vec::new();
333
334        for cpu in online.iter() {
335            let cache_dir = self.cpu_dir().join(format!("cpu{cpu}/cache"));
336            // Index numbering is dense but its length is unknown; probe until
337            // a gap. Containers frequently expose no cache directory at all.
338            for index in 0..16u32 {
339                let dir = cache_dir.join(format!("index{index}"));
340                let level = match read_u32(dir.join("level")) {
341                    Some(l) => l as u8,
342                    None => break,
343                };
344                let type_raw = read_string(dir.join("type")).unwrap_or_default();
345                let kind = classify_cache(level, type_raw.trim());
346                let shared = match read_string(dir.join("shared_cpu_list")) {
347                    Ok(raw) => CpuSet::parse_list(&raw).unwrap_or_default().to_vec(),
348                    Err(_) => vec![cpu],
349                };
350                let key = (level, kind, shared.clone());
351                if !seen.insert(key) {
352                    continue;
353                }
354                caches.push(Cache {
355                    level,
356                    kind,
357                    size_bytes: read_string(dir.join("size"))
358                        .ok()
359                        .and_then(|s| parse_size(s.trim())),
360                    line_size_bytes: read_u32(dir.join("coherency_line_size")),
361                    ways_of_associativity: read_u32(dir.join("ways_of_associativity")),
362                    shared_cpus: shared,
363                });
364            }
365        }
366
367        caches.sort_by(|a, b| {
368            a.level
369                .cmp(&b.level)
370                .then(a.kind.cmp(&b.kind))
371                .then(a.shared_cpus.cmp(&b.shared_cpus))
372        });
373        Ok(caches)
374    }
375
376    fn read_numa_nodes(&self) -> Result<Vec<NumaNode>> {
377        let dir = self.node_dir();
378        let entries = match std::fs::read_dir(&dir) {
379            Ok(e) => e,
380            // No CONFIG_NUMA, or a container hiding it. A single implicit node
381            // is the right model in that case, and we signal it with an empty
382            // node list rather than inventing a node 0 that sysfs never showed.
383            Err(_) => return Ok(Vec::new()),
384        };
385
386        let mut nodes = Vec::new();
387        for entry in entries.flatten() {
388            let name = entry.file_name();
389            let name = name.to_string_lossy();
390            let Some(id) = name
391                .strip_prefix("node")
392                .and_then(|n| n.parse::<u32>().ok())
393            else {
394                continue;
395            };
396            let path = dir.join(&*name);
397            let cpus = read_string(path.join("cpulist"))
398                .ok()
399                .and_then(|raw| CpuSet::parse_list(&raw).ok())
400                .unwrap_or_default()
401                .to_vec();
402            nodes.push(NumaNode {
403                id,
404                cpus,
405                memory_kb: read_string(path.join("meminfo"))
406                    .ok()
407                    .and_then(|s| parse_node_mem_total_kb(&s)),
408            });
409        }
410        nodes.sort_by_key(|n| n.id);
411        Ok(nodes)
412    }
413
414    /// Vendor and model name, preferring `/proc/cpuinfo` (which the kernel has
415    /// already normalised) and falling back to the CPUID brand string.
416    fn read_cpu_identity(&self) -> (String, String) {
417        let cpuinfo = read_string(self.proc_root.join("cpuinfo")).unwrap_or_default();
418        let vendor = first_cpuinfo_field(&cpuinfo, "vendor_id")
419            .or_else(|| first_cpuinfo_field(&cpuinfo, "CPU implementer"))
420            .or_else(crate::platform::cpuid::vendor)
421            .unwrap_or_else(|| "unknown".to_string());
422        let model = first_cpuinfo_field(&cpuinfo, "model name")
423            .or_else(|| first_cpuinfo_field(&cpuinfo, "Model"))
424            .or_else(crate::platform::cpuid::brand_string)
425            .unwrap_or_else(|| "unknown".to_string());
426        (vendor, model)
427    }
428}
429
430// ---------------------------------------------------------------------------
431// Small parsing helpers. Kept free-standing so they can be unit tested without
432// constructing a filesystem.
433// ---------------------------------------------------------------------------
434
435fn read_string(path: impl AsRef<Path>) -> Result<String> {
436    let path = path.as_ref();
437    std::fs::read_to_string(path).map_err(|e| Error::io(path, e))
438}
439
440fn read_u64(path: impl AsRef<Path>) -> Option<u64> {
441    read_string(path).ok()?.trim().parse().ok()
442}
443
444fn read_u32(path: impl AsRef<Path>) -> Option<u32> {
445    read_string(path).ok()?.trim().parse().ok()
446}
447
448/// Map sysfs `level` + `type` onto our cache taxonomy.
449fn classify_cache(level: u8, type_raw: &str) -> CacheKind {
450    match (level, type_raw) {
451        (1, "Data") => CacheKind::L1Data,
452        (1, "Instruction") => CacheKind::L1Instruction,
453        (2, "Unified") => CacheKind::L2Unified,
454        (3, "Unified") => CacheKind::L3Unified,
455        _ => CacheKind::Other,
456    }
457}
458
459/// Parse the sysfs cache `size` attribute: `"32K"`, `"1280K"`, `"32M"`.
460fn parse_size(s: &str) -> Option<u64> {
461    let s = s.trim();
462    if s.is_empty() {
463        return None;
464    }
465    let (digits, mult) = match s.as_bytes()[s.len() - 1] {
466        b'K' | b'k' => (&s[..s.len() - 1], 1024),
467        b'M' | b'm' => (&s[..s.len() - 1], 1024 * 1024),
468        b'G' | b'g' => (&s[..s.len() - 1], 1024 * 1024 * 1024),
469        _ => (s, 1),
470    };
471    digits.trim().parse::<u64>().ok().map(|v| v * mult)
472}
473
474/// Pull `MemTotal` out of a NUMA node's `meminfo`, whose lines look like
475/// `Node 0 MemTotal:  16316296 kB`.
476fn parse_node_mem_total_kb(s: &str) -> Option<u64> {
477    for line in s.lines() {
478        if let Some(rest) = line.split("MemTotal:").nth(1) {
479            return rest.split_whitespace().next()?.parse().ok();
480        }
481    }
482    None
483}
484
485/// First value of a `key : value` field in `/proc/cpuinfo`.
486///
487/// "First" because cpuinfo repeats every field per CPU and, on hybrid parts,
488/// the model name genuinely differs between entries; the leading one is the
489/// conventional choice and matches what `lscpu` prints.
490fn first_cpuinfo_field(cpuinfo: &str, key: &str) -> Option<String> {
491    for line in cpuinfo.lines() {
492        let (k, v) = line.split_once(':')?;
493        if k.trim() == key {
494            let v = v.trim();
495            if !v.is_empty() {
496                return Some(v.to_string());
497            }
498        }
499    }
500    None
501}
502
503#[cfg(test)]
504mod tests {
505    use super::*;
506
507    #[test]
508    fn parses_cache_sizes_with_and_without_suffix() {
509        assert_eq!(parse_size("32K"), Some(32 * 1024));
510        assert_eq!(parse_size("1280K"), Some(1280 * 1024));
511        assert_eq!(parse_size("32M"), Some(32 * 1024 * 1024));
512        assert_eq!(parse_size("512"), Some(512));
513        assert_eq!(parse_size(""), None);
514        assert_eq!(parse_size("banana"), None);
515    }
516
517    #[test]
518    fn classifies_cache_levels() {
519        assert_eq!(classify_cache(1, "Data"), CacheKind::L1Data);
520        assert_eq!(classify_cache(1, "Instruction"), CacheKind::L1Instruction);
521        assert_eq!(classify_cache(2, "Unified"), CacheKind::L2Unified);
522        assert_eq!(classify_cache(3, "Unified"), CacheKind::L3Unified);
523        assert_eq!(classify_cache(4, "Unified"), CacheKind::Other);
524        assert_eq!(classify_cache(1, "Unified"), CacheKind::Other);
525    }
526
527    #[test]
528    fn extracts_node_memory() {
529        let s = "Node 0 MemTotal:       16316296 kB\nNode 0 MemFree: 100 kB\n";
530        assert_eq!(parse_node_mem_total_kb(s), Some(16316296));
531        assert_eq!(parse_node_mem_total_kb("Node 0 MemFree: 1 kB"), None);
532    }
533
534    #[test]
535    fn reads_first_cpuinfo_field() {
536        let s = "processor\t: 0\nvendor_id\t: AuthenticAMD\nmodel name\t: AMD Ryzen 9 7950X\n\
537                 processor\t: 1\nmodel name\t: AMD Ryzen 9 7950X\n";
538        assert_eq!(
539            first_cpuinfo_field(s, "vendor_id").as_deref(),
540            Some("AuthenticAMD")
541        );
542        assert_eq!(
543            first_cpuinfo_field(s, "model name").as_deref(),
544            Some("AMD Ryzen 9 7950X")
545        );
546        assert_eq!(first_cpuinfo_field(s, "flags"), None);
547    }
548}