Skip to main content

cgp/profilers/
system.rs

1//! System health and VRAM collection via nvidia-smi and /proc.
2//! Spec sections 9.8 (VRAM), 9.10 (System Health), 9.11 (Energy).
3
4use crate::metrics::catalog::{EnergyMetrics, SystemHealthMetrics, VramMetrics};
5use std::process::Command;
6
7/// Collect system health metrics from nvidia-smi (NVML) and /proc.
8pub fn collect_system_health() -> Option<SystemHealthMetrics> {
9    let gpu = query_nvidia_smi(&[
10        "temperature.gpu",
11        "power.draw",
12        "clocks.current.sm",
13        "clocks.current.memory",
14        "memory.used",
15        "memory.total",
16    ])?;
17
18    let fields: Vec<&str> = gpu.split(", ").collect();
19    if fields.len() < 6 {
20        return None;
21    }
22
23    let cpu_freq = read_cpu_frequency().unwrap_or(0.0);
24    let cpu_temp = read_cpu_temperature().unwrap_or(0.0);
25
26    // Unified-memory platforms report memory.total as N/A (parses to 0); fall back to system RAM.
27    let mut gpu_mem_total = parse_nvidia_val(fields[5]);
28    if gpu_mem_total <= 0.0 {
29        gpu_mem_total = read_system_memory_total_mb().unwrap_or(0.0);
30    }
31
32    Some(SystemHealthMetrics {
33        gpu_temperature_celsius: parse_nvidia_val(fields[0]),
34        gpu_power_watts: parse_nvidia_val(fields[1]),
35        gpu_clock_mhz: parse_nvidia_val(fields[2]),
36        gpu_memory_clock_mhz: parse_nvidia_val(fields[3]),
37        cpu_frequency_mhz: cpu_freq,
38        cpu_temperature_celsius: cpu_temp,
39        gpu_memory_used_mb: parse_nvidia_val(fields[4]),
40        gpu_memory_total_mb: gpu_mem_total,
41    })
42}
43
44/// Collect VRAM metrics from nvidia-smi.
45pub fn collect_vram() -> Option<VramMetrics> {
46    let gpu = query_nvidia_smi(&["memory.used", "memory.total", "memory.free"])?;
47
48    let fields: Vec<&str> = gpu.split(", ").collect();
49    if fields.len() < 3 {
50        return None;
51    }
52
53    let used = parse_nvidia_val(fields[0]);
54    let mut total = parse_nvidia_val(fields[1]);
55    let free = parse_nvidia_val(fields[2]);
56    // Unified-memory NVIDIA platforms (GB10/GH200/Jetson) report VRAM total as [N/A] via nvidia-smi
57    // because the GPU shares system RAM. Fall back to total system memory so the profiler reports the
58    // (unified) memory budget instead of 0.
59    if total <= 0.0 {
60        total = read_system_memory_total_mb().unwrap_or(0.0);
61    }
62    let utilization = if total > 0.0 {
63        used / total * 100.0
64    } else {
65        0.0
66    };
67
68    Some(VramMetrics {
69        vram_used_mb: used,
70        vram_total_mb: total,
71        vram_free_mb: free,
72        vram_utilization_pct: utilization,
73        vram_peak_mb: used, // snapshot — no tracking history
74        vram_allocation_count: 0,
75        vram_fragmentation_pct: 0.0,
76    })
77}
78
79/// Compute energy efficiency from power and throughput.
80pub fn compute_energy(power_watts: f64, tflops: f64, duration_us: f64) -> Option<EnergyMetrics> {
81    if power_watts <= 0.0 {
82        return None;
83    }
84    let tflops_per_watt = if power_watts > 0.0 {
85        tflops / power_watts
86    } else {
87        0.0
88    };
89    let joules = power_watts * duration_us * 1e-6;
90    Some(EnergyMetrics {
91        tflops_per_watt,
92        joules_per_inference: joules,
93    })
94}
95
96/// Run nvidia-smi --query-gpu and return the CSV row.
97fn query_nvidia_smi(fields: &[&str]) -> Option<String> {
98    let query = fields.join(",");
99    let output = Command::new("nvidia-smi")
100        .args(["--query-gpu", &query, "--format=csv,noheader,nounits"])
101        .output()
102        .ok()?;
103
104    if !output.status.success() {
105        return None;
106    }
107    let stdout = String::from_utf8_lossy(&output.stdout);
108    let line = stdout.trim();
109    if line.is_empty() || line.contains("[N/A]") && line.chars().all(|c| c == ',' || c == ' ') {
110        return None;
111    }
112    Some(line.to_string())
113}
114
115/// Parse a numeric value from nvidia-smi output (handles "123 W", "45 MiB", etc.)
116fn parse_nvidia_val(s: &str) -> f64 {
117    let s = s.trim();
118    if s == "[N/A]" || s == "N/A" {
119        return 0.0;
120    }
121    // Take the first token that looks numeric
122    s.split_whitespace()
123        .next()
124        .and_then(|token| token.parse::<f64>().ok())
125        .unwrap_or(0.0)
126}
127
128/// Total system RAM in MB, read from `/proc/meminfo` (`MemTotal`).
129///
130/// Used as a fallback for GPU memory total on unified-memory NVIDIA platforms (GB10 / Grace-Blackwell,
131/// GH200, Jetson), where `nvidia-smi --query-gpu=memory.total` reports `[N/A]` because the GPU shares
132/// system RAM rather than exposing dedicated VRAM. On dedicated GPUs nvidia-smi returns a real value,
133/// so this fallback is never reached and behavior is unchanged.
134fn read_system_memory_total_mb() -> Option<f64> {
135    let content = std::fs::read_to_string("/proc/meminfo").ok()?;
136    for line in content.lines() {
137        // Format: "MemTotal:       65780480 kB"
138        if let Some(rest) = line.strip_prefix("MemTotal:") {
139            let kb = rest.split_whitespace().next()?.parse::<f64>().ok()?;
140            return Some(kb / 1024.0);
141        }
142    }
143    None
144}
145
146/// Read current CPU frequency from /proc/cpuinfo (MHz).
147fn read_cpu_frequency() -> Option<f64> {
148    let content = std::fs::read_to_string("/proc/cpuinfo").ok()?;
149    // Take average across all cores
150    let mut total = 0.0;
151    let mut count = 0;
152    for line in content.lines() {
153        if line.starts_with("cpu MHz") {
154            if let Some(val) = line.split(':').nth(1) {
155                if let Ok(mhz) = val.trim().parse::<f64>() {
156                    total += mhz;
157                    count += 1;
158                }
159            }
160        }
161    }
162    if count > 0 {
163        Some(total / count as f64)
164    } else {
165        None
166    }
167}
168
169/// Read CPU temperature from /sys thermal zones.
170fn read_cpu_temperature() -> Option<f64> {
171    // Try thermal_zone0 first (usually CPU package)
172    for i in 0..10 {
173        let path = format!("/sys/class/thermal/thermal_zone{i}/temp");
174        if let Ok(content) = std::fs::read_to_string(&path) {
175            if let Ok(millidegrees) = content.trim().parse::<f64>() {
176                return Some(millidegrees / 1000.0);
177            }
178        }
179    }
180    None
181}
182
183#[cfg(test)]
184mod tests {
185    use super::*;
186
187    #[test]
188    fn test_parse_nvidia_val() {
189        assert!((parse_nvidia_val("285.32 W") - 285.32).abs() < 0.01);
190        assert!((parse_nvidia_val("24564 MiB") - 24564.0).abs() < 1.0);
191        assert!((parse_nvidia_val("62") - 62.0).abs() < 0.01);
192        assert!((parse_nvidia_val("[N/A]")).abs() < 0.01);
193        assert!((parse_nvidia_val("N/A")).abs() < 0.01);
194    }
195
196    #[test]
197    fn test_compute_energy() {
198        let e = compute_energy(300.0, 11.6, 23.2).unwrap();
199        assert!((e.tflops_per_watt - 11.6 / 300.0).abs() < 0.001);
200        assert!((e.joules_per_inference - 300.0 * 23.2e-6).abs() < 0.001);
201    }
202
203    #[test]
204    fn test_compute_energy_zero_power() {
205        assert!(compute_energy(0.0, 11.6, 23.2).is_none());
206    }
207
208    /// System health collection should not panic even without nvidia-smi.
209    #[test]
210    fn test_collect_system_health_no_panic() {
211        let _ = collect_system_health();
212    }
213
214    /// VRAM collection should not panic even without nvidia-smi.
215    #[test]
216    fn test_collect_vram_no_panic() {
217        let _ = collect_vram();
218    }
219
220    #[test]
221    fn test_read_cpu_frequency_no_panic() {
222        let _ = read_cpu_frequency();
223    }
224
225    /// On Linux `/proc/meminfo` always reports a positive `MemTotal`. This backs the
226    /// unified-memory VRAM fallback (GB10/GH200/Jetson report memory.total = N/A).
227    #[test]
228    fn test_read_system_memory_total_mb() {
229        let total = read_system_memory_total_mb();
230        assert!(total.is_some(), "/proc/meminfo MemTotal should be readable");
231        assert!(total.unwrap() > 0.0, "system memory total should be > 0 MB");
232    }
233
234    #[test]
235    fn test_read_cpu_temperature_no_panic() {
236        let _ = read_cpu_temperature();
237    }
238
239    /// How many GPUs `nvidia-smi` actually REPORTS -- not whether the binary exists,
240    /// and never at the cost of hanging the suite.
241    ///
242    /// `which::which("nvidia-smi").is_ok()` was the precondition for both GPU tests, and it
243    /// is the wrong question. The intel clean-room runner has the NVIDIA userland installed
244    /// and no visible device, so `nvidia-smi` resolved, the tests decided a GPU was present,
245    /// the collectors correctly returned `None`, and Coverage Nightly failed with
246    /// "nvidia-smi exists but no health data" -- a red build reporting an ENVIRONMENT fact as
247    /// a code defect. A binary on `PATH` is not a device on the bus.
248    ///
249    /// THREE THINGS AN INDEPENDENT REVIEW REFUSED THE FIRST VERSION OVER:
250    ///
251    /// 1. A WEDGED DRIVER MUST NOT HANG CI. `Command::output()` blocks forever, and a hung
252    ///    `nvidia-smi` is a real state -- `which` could never hang, so a naive probe is a
253    ///    REGRESSION in failure mode. We spawn, poll `try_wait` against a deadline, and kill.
254    /// 2. THE OUTPUT IS PARSED AS A WHITELIST. Counting "non-empty lines" accepts a licence
255    ///    banner or an update notice as a GPU. `--query-gpu=index` emits integers; a line
256    ///    counts only if it PARSES as one. Blacklists fail open on their complement.
257    /// 3. ONE PROBE, ONE CALL. Calling it from two tests admits a TOCTOU where a transient
258    ///    makes both skip and the pair asserts nothing. There is now one test.
259    fn gpus_reported() -> usize {
260        use std::process::{Command, Stdio};
261        use std::time::{Duration, Instant};
262
263        let Ok(mut child) = Command::new("nvidia-smi")
264            .args(["--query-gpu=index", "--format=csv,noheader"])
265            .stdout(Stdio::piped())
266            .stderr(Stdio::null())
267            .spawn()
268        else {
269            return 0; // not installed, or not executable: no device either way
270        };
271
272        let deadline = Instant::now() + Duration::from_secs(10);
273        loop {
274            match child.try_wait() {
275                Ok(Some(status)) => {
276                    if !status.success() {
277                        return 0; // installed, but the driver has nothing to report
278                    }
279                    break;
280                }
281                Ok(None) => {
282                    if Instant::now() >= deadline {
283                        let _ = child.kill();
284                        let _ = child.wait();
285                        return 0; // wedged driver: treat as no device, never hang
286                    }
287                    std::thread::sleep(Duration::from_millis(50));
288                }
289                Err(_) => return 0,
290            }
291        }
292
293        let Ok(out) = child.wait_with_output() else {
294            return 0;
295        };
296        String::from_utf8_lossy(&out.stdout)
297            .lines()
298            .filter(|l| l.trim().parse::<u32>().is_ok())
299            .count()
300    }
301
302    /// One probe, one branch, no skips.
303    ///
304    /// An earlier version was a PAIR of tests, each calling the probe and early-returning
305    /// when the host was the other kind. A review found two defects in that shape: the two
306    /// probe calls admit a TOCTOU where a transient failure makes BOTH tests skip and the
307    /// pair asserts nothing at all, and the `eprintln!` explaining a skip is swallowed by
308    /// `cargo test` without `--nocapture`, so a silent skip is invisible in CI.
309    ///
310    /// One test, probing once, cannot skip: exactly one branch executes on every host.
311    ///
312    /// WHAT EACH BRANCH PROVES, AND WHAT IT DOES NOT. The GPU branch proves the collectors
313    /// PARSE -- it is the only branch that can, and a parser regression fails it here. The
314    /// no-GPU branch proves the ABSENCE path returns `None` rather than fabricating a
315    /// reading or panicking. It cannot distinguish "no GPU" from "broken parser", because
316    /// both yield `None`; that is inherent to the branch and is why parse correctness is
317    /// asserted on the other side rather than claimed on this one.
318    #[test]
319    fn test_gpu_collectors_match_what_nvidia_smi_reports() {
320        if gpus_reported() > 0 {
321            let health = collect_system_health().expect("a reporting GPU must yield health data");
322            assert!(
323                health.gpu_temperature_celsius > 0.0,
324                "GPU temp should be > 0"
325            );
326            assert!(
327                health.gpu_memory_total_mb > 0.0,
328                "GPU memory total should be > 0"
329            );
330
331            let vram = collect_vram().expect("a reporting GPU must yield VRAM data");
332            assert!(vram.vram_total_mb > 0.0, "VRAM total should be > 0");
333            assert!(vram.vram_utilization_pct >= 0.0 && vram.vram_utilization_pct <= 100.0);
334        } else {
335            assert!(
336                collect_system_health().is_none(),
337                "with no GPU reported, health must be None rather than a fabricated reading"
338            );
339            assert!(
340                collect_vram().is_none(),
341                "with no GPU reported, VRAM must be None rather than a fabricated reading"
342            );
343        }
344    }
345
346    /// The probe must not count a banner, a notice, or an error string as a GPU.
347    ///
348    /// This is the whitelist from `gpus_reported` doc-comment point 2, asserted directly so
349    /// that loosening the filter back to "non-empty lines" turns it RED on a host with no
350    /// GPU as well as on one with a GPU. `nvidia-smi` writes most notices to stderr, which
351    /// the probe discards -- but not all of them, and a blacklist would fail open on the
352    /// first one it had not seen.
353    #[test]
354    fn test_the_probe_counts_only_lines_that_parse_as_an_index() {
355        fn count(stdout: &str) -> usize {
356            stdout
357                .lines()
358                .filter(|l| l.trim().parse::<u32>().is_ok())
359                .count()
360        }
361        assert_eq!(count("0\n1\n"), 2, "two indices are two GPUs");
362        assert_eq!(count(""), 0, "no output is no GPU");
363        assert_eq!(count("\n  \n"), 0, "blank lines are no GPU");
364        assert_eq!(
365            count("NVIDIA-SMI has failed because it couldn't communicate with the driver\n"),
366            0,
367            "an error banner on stdout is NOT a GPU -- the defect a non-empty-lines filter has"
368        );
369        assert_eq!(
370            count("Please update your driver\n0\n"),
371            1,
372            "a notice beside a real index counts the index only"
373        );
374    }
375}