corescout_substrate/platform/mod.rs
1//! Platform abstraction layer.
2//!
3//! Everything OS-specific in CoreScout is reached through the [`Platform`]
4//! trait. Topology discovery, benchmarking, ranking and output are written
5//! against this trait and contain no `cfg(target_os)` of their own.
6//!
7//! # Adding a new platform
8//!
9//! Implement [`Platform`] in `platform/<os>/`, return it from [`detect`] under
10//! the appropriate `cfg`, and nothing above this module needs to change. For
11//! Windows the mapping is:
12//!
13//! | trait method | Win32 |
14//! |-----------------------------|-------|
15//! | `discover_topology` | `GetLogicalProcessorInformationEx` (+ `CallNtPowerInformation` for frequency, `GetSystemCpuSetInformation` for E/P class and `EfficiencyClass`) |
16//! | `pin_current_thread` | `SetThreadAffinityMask` / `SetThreadGroupAffinity` |
17//! | `process_affinity` | `GetProcessAffinityMask` |
18//! | `spawn_with_affinity` | `CreateProcess` suspended + `SetProcessAffinityMask` + `ResumeThread` |
19//! | `thread_switch_counters` | `QueryThreadCycleTime` / ETW (approximate) |
20//!
21//! The one place that needs care on Windows is processor *groups*: a machine
22//! with more than 64 logical CPUs splits into groups, so [`crate::topology::LogicalId`]
23//! must be mapped to a `(group, index)` pair inside the backend, never leaked
24//! upward.
25
26use std::process::{Child, Command};
27
28use crate::topology::{LogicalId, Topology};
29use corescout_core::cpuset::CpuSet;
30use corescout_core::error::Result;
31
32// Compiled everywhere: the sysfs *parser* inside is portable and its tests
33// must run on any developer machine. Only its syscall submodule is gated.
34pub mod linux;
35
36pub mod cpuid;
37pub mod stub;
38#[cfg(target_os = "windows")]
39pub mod windows;
40
41/// Counters describing how much the OS interfered with a thread.
42///
43/// An *involuntary* context switch means the scheduler took the CPU away
44/// mid-computation: the single most common cause of a benchmark outlier that
45/// has nothing to do with the core being measured.
46#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
47pub struct SwitchCounters {
48 pub voluntary: u64,
49 pub involuntary: u64,
50}
51
52impl SwitchCounters {
53 /// Counters accumulated between two samples.
54 pub fn delta(self, earlier: SwitchCounters) -> SwitchCounters {
55 SwitchCounters {
56 voluntary: self.voluntary.saturating_sub(earlier.voluntary),
57 involuntary: self.involuntary.saturating_sub(earlier.involuntary),
58 }
59 }
60}
61
62/// The OS-specific operations CoreScout needs.
63///
64/// Implementations must be `Sync` because the benchmark runner hands the
65/// platform to worker threads.
66pub trait Platform: Send + Sync {
67 /// Human-readable backend name, e.g. `"linux"`.
68 fn name(&self) -> &'static str;
69
70 /// Discover the full machine topology.
71 fn discover_topology(&self) -> Result<Topology>;
72
73 /// Restrict the *calling thread* to a single logical CPU.
74 ///
75 /// This is a hard pin, not a hint: after it returns successfully the
76 /// thread must not run anywhere else, or benchmark attribution is void.
77 fn pin_current_thread(&self, cpu: LogicalId) -> Result<()>;
78
79 /// Set the calling thread's affinity to an arbitrary set.
80 fn set_current_thread_affinity(&self, cpus: &CpuSet) -> Result<()>;
81
82 /// Read the calling thread's current affinity mask.
83 fn current_thread_affinity(&self) -> Result<CpuSet>;
84
85 /// Read the affinity mask of the whole process.
86 fn process_affinity(&self) -> Result<CpuSet>;
87
88 /// Sample the calling thread's context-switch counters, when the platform
89 /// can report them per-thread. `None` means "no interference data", and
90 /// callers degrade to statistical outlier detection alone.
91 fn thread_switch_counters(&self) -> Option<SwitchCounters>;
92
93 /// Cycles the calling thread has actually been executing.
94 ///
95 /// Not wall time: this excludes every moment the thread was not scheduled,
96 /// so it measures silicon consumed rather than time elapsed. That is the
97 /// quantity a productivity ratio wants in its denominator, because a thread
98 /// that waited a long time did not thereby use more of the machine.
99 ///
100 /// `None` where the platform cannot report it, in which case a caller must
101 /// fall back to wall time and say so.
102 fn thread_cycles(&self) -> Option<u64> {
103 None
104 }
105
106 /// Spawn a child process confined to `cpus`.
107 ///
108 /// The affinity must be applied before the child's `main` runs, otherwise
109 /// early allocations and thread spawns land on the wrong cores.
110 fn spawn_with_affinity(&self, command: &mut Command, cpus: &CpuSet) -> Result<Child>;
111}
112
113/// Return the backend for the platform we are running on.
114///
115/// A stub backend is returned on unsupported systems: `corescout info` then
116/// fails with a clear message instead of the binary refusing to build.
117pub fn detect() -> Box<dyn Platform> {
118 #[cfg(target_os = "windows")]
119 {
120 Box::new(windows::WindowsPlatform::new())
121 }
122 #[cfg(target_os = "linux")]
123 {
124 Box::new(linux::LinuxPlatform::new())
125 }
126 #[cfg(not(any(target_os = "linux", target_os = "windows")))]
127 {
128 Box::new(stub::StubPlatform::new())
129 }
130}