tak-cli 0.0.5

Benchmark command-line programs and track their performance over time. Experimental; do not use.
Documentation
//! Measurement backends.
//!
//! Two tiers, deliberately separated:
//!
//! - **Deterministic** (`instructions`) — reproducible to ~0.02% run-to-run and
//!   ~0.035% across wildly different machine load. This is the only tier that may
//!   gate CI.
//! - **Timing** (`wall_*`) — recorded and charted, never gated. On a quiet 32-core
//!   host wall clock still shows 4–20% coefficient of variation; under contention
//!   the median moves ~150%.
//!
//! Syscall counts and peak RSS sit awkwardly between the two: better than wall
//! clock (~1%) but not deterministic, because they move with thread scheduling.
//! They are recorded, and may be flagged, but must not gate at a tight threshold.

use crate::settings::Settings;
use anyhow::{Context, Result, bail};
use std::collections::BTreeMap;
use std::process::{Command, Stdio};
use std::time::Instant;

#[derive(Debug, Clone)]
pub struct Plan {
    pub cmd: Vec<String>,
    pub warmup: u32,
    pub runs: u32,
    /// Directory to run in. Declared benchmarks resolve relative commands
    /// against `tak.toml`'s directory, not the caller's — otherwise the same
    /// benchmark measures different things depending on where you stood.
    pub dir: Option<std::path::PathBuf>,
    /// Resolved settings. Carried on the plan rather than read from a global so
    /// a test can measure under a different configuration without touching the
    /// environment of the whole test binary.
    pub settings: Settings,
}

/// A command for running a benchmark subject, with the scrubbed variables
/// removed.
///
/// Every subject spawn goes through here. Under cachegrind the removal is
/// applied to valgrind itself, which the subject inherits from.
///
/// What gets removed is [`Settings::scrubbed_env`] — `env_deny` less
/// `env_allow`, both declared in `settings.toml`. The default is the forge
/// tokens, for two reasons. **Determinism**: a CLI that finds a token often
/// does more with it than without, so a measurement would move depending on
/// whether CI happened to export one. **Credentials**: `backfill` downloads
/// release binaries and executes them, and any run that can push notes holds a
/// repository-write token.
///
/// tak's own network calls are unaffected — `backfill` authenticates with
/// `curl` directly rather than through this path.
fn subject(bin: &str, settings: &Settings) -> Command {
    let mut c = Command::new(bin);
    for key in settings.scrubbed_env() {
        c.env_remove(key);
    }
    c
}

/// Run once, discarding output, returning elapsed wall time in milliseconds.
///
/// No shell. Spawning a shell adds its own startup cost and variance to every
/// sample, which for commands in the 10ms range is a large fraction of the
/// measurement — the same reasoning behind poop's refusal to support one.
fn time_once(cmd: &[String], dir: Option<&std::path::Path>, settings: &Settings) -> Result<f64> {
    let (bin, args) = cmd.split_first().context("empty command")?;
    let mut c = subject(bin, settings);
    c.args(args).stdout(Stdio::null()).stderr(Stdio::null());
    if let Some(d) = dir {
        c.current_dir(d);
    }
    let start = Instant::now();
    let status = c
        .status()
        .with_context(|| format!("failed to spawn `{bin}`"))?;
    let elapsed = start.elapsed().as_secs_f64() * 1000.0;
    let _ = status;
    Ok(elapsed)
}

/// Wall-clock statistics over `plan.runs` samples.
///
/// Reports `min` alongside the mean because contention is one-sided — a busy
/// machine can only make a run slower, never faster — so the minimum is a far
/// more robust estimator than the mean on shared CI hardware.
pub fn wall(plan: &Plan) -> Result<BTreeMap<String, f64>> {
    // Zero runs would leave `samples` empty and index straight off the end.
    if plan.runs == 0 {
        bail!("runs must be at least 1");
    }
    let dir = plan.dir.as_deref();
    for _ in 0..plan.warmup {
        time_once(&plan.cmd, dir, &plan.settings)?;
    }
    let mut samples = Vec::with_capacity(plan.runs as usize);
    for _ in 0..plan.runs {
        samples.push(time_once(&plan.cmd, dir, &plan.settings)?);
    }
    samples.sort_by(|a, b| a.partial_cmp(b).unwrap());

    let n = samples.len();
    let mean = samples.iter().sum::<f64>() / n as f64;
    let p50 = samples[n / 2];

    Ok(BTreeMap::from([
        ("wall_min_ms".to_string(), samples[0]),
        ("wall_p50_ms".to_string(), p50),
        ("wall_mean_ms".to_string(), mean),
        ("wall_max_ms".to_string(), samples[n - 1]),
        ("wall_n".to_string(), n as f64),
    ]))
}

/// Is cachegrind usable on this machine?
///
/// Exposed so callers can tell "no counters because valgrind is missing" from
/// "no counters because the measurement failed" — two problems with entirely
/// different fixes.
pub fn valgrind_available() -> bool {
    Command::new("valgrind")
        .arg("--version")
        .stdout(Stdio::null())
        .stderr(Stdio::null())
        .status()
        .map(|s| s.success())
        .unwrap_or(false)
}

/// Repeats of the cachegrind run. Three is enough to catch a bimodal subject
/// without tripling the cost of a measurement that is already ~50x slowed.
const COUNTER_RUNS: u32 = 3;

/// A subject varying by more than this is doing environment-dependent work, and
/// its instruction count is not a usable gate. Well above the ~0.005% observed
/// for a genuinely hermetic command, far below the 16% seen from one that
/// touches the network.
const SPREAD_WARN_PCT: f64 = 0.5;

/// Instruction counts from repeated cachegrind runs.
#[derive(Debug, Clone, Copy)]
pub struct Counted {
    pub min: u64,
    pub max: u64,
    pub runs: u32,
}

impl Counted {
    /// Relative spread across runs, as a percentage of the minimum.
    pub fn spread_pct(&self) -> f64 {
        if self.min == 0 {
            return 0.0;
        }
        (self.max - self.min) as f64 / self.min as f64 * 100.0
    }

    /// Whether this subject looks non-hermetic.
    ///
    /// The metric is deterministic; the program being measured need not be. A
    /// CLI that checks for updates, reads a cache it may have just created, or
    /// resolves DNS retires a different number of instructions depending on
    /// conditions that have nothing to do with the code under test.
    pub fn is_suspect(&self) -> bool {
        self.spread_pct() > SPREAD_WARN_PCT
    }
}

/// Instruction count via `valgrind --tool=cachegrind`, repeated.
///
/// Reports the **minimum**, for the same reason wall clock does: the extra work
/// a subject sometimes performs is one-sided. A run that consults the network or
/// populates a cache can only retire *more* instructions than the quiet path,
/// never fewer, so the floor is the stable estimator.
///
/// Returns `Ok(None)` when valgrind is unavailable rather than failing: this is
/// the expected state on macOS (no usable Apple Silicon support) and Windows.
/// Those platforms record timing only, and the CI gate lives on the Linux job.
/// Locally, a container gets you counters on any host.
pub fn instructions(
    cmd: &[String],
    dir: Option<&std::path::Path>,
    settings: &Settings,
) -> Result<Option<Counted>> {
    if !valgrind_available() {
        return Ok(None);
    }

    let mut samples: Vec<u64> = Vec::with_capacity(COUNTER_RUNS as usize);
    for _ in 0..COUNTER_RUNS {
        let mut c = subject("valgrind", settings);
        c.args([
            "--tool=cachegrind",
            "--cache-sim=no",
            "--branch-sim=no",
            "--cachegrind-out-file=/dev/null",
        ])
        .args(cmd)
        .stdout(Stdio::null());
        if let Some(d) = dir {
            c.current_dir(d);
        }
        let out = c.output().context("failed to run valgrind")?;

        // cachegrind writes its summary to stderr as e.g. "I refs:  48,349,132".
        let stderr = String::from_utf8_lossy(&out.stderr);
        match parse_irefs(&stderr) {
            Some(n) => samples.push(n),
            // Valgrind is installed but produced no summary — a real failure,
            // not the same thing as valgrind being absent. Reporting it as
            // absent sends people off installing something they already have.
            None => bail!(
                "valgrind ran but emitted no `I refs` summary: {}",
                stderr.lines().last().unwrap_or("(no output)").trim()
            ),
        }
    }

    Ok(Some(Counted {
        min: *samples.iter().min().expect("COUNTER_RUNS > 0"),
        max: *samples.iter().max().expect("COUNTER_RUNS > 0"),
        runs: COUNTER_RUNS,
    }))
}

/// Extract the `I refs:` count from cachegrind's stderr summary.
fn parse_irefs(stderr: &str) -> Option<u64> {
    let line = stderr.lines().find(|l| l.contains("I refs:"))?;
    let digits: String = line
        .rsplit(':')
        .next()?
        .chars()
        .filter(|c| c.is_ascii_digit())
        .collect();
    digits.parse().ok()
}

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn parses_cachegrind_summary() {
        let s = "==12== I refs:      48,349,132\n";
        assert_eq!(parse_irefs(s), Some(48_349_132));
    }

    #[test]
    fn missing_summary_is_none_not_panic() {
        assert_eq!(parse_irefs("valgrind: command not found"), None);
    }

    /// What `subject` removes, as configured.
    fn removals(settings: &Settings) -> Vec<String> {
        subject("true", settings)
            .get_envs()
            .filter(|(_, v)| v.is_none())
            .map(|(k, _)| k.to_string_lossy().into_owned())
            .collect()
    }

    /// Construction-level check. The end-to-end proof that a subject cannot see
    /// the token lives in `tests/subject_env.rs`, which runs the real binary.
    #[test]
    fn subject_commands_remove_the_default_denied_variables() {
        let removed = removals(&Settings::default());
        for key in Settings::default().scrubbed_env() {
            assert!(removed.contains(&key.to_string()), "{key} not removed");
        }
    }

    /// The removal follows the setting rather than a compiled-in list, which is
    /// the whole point of routing it through `Settings`.
    #[test]
    fn the_removal_follows_the_settings() {
        let removed = removals(&Settings {
            env_deny: vec!["CUSTOM_SECRET".into()],
            env_allow: Vec::new(),
            gate_pct: Settings::default().gate_pct,
            credit: Settings::default().credit,
        });
        assert!(removed.contains(&"CUSTOM_SECRET".to_string()));
        assert!(!removed.contains(&"GITHUB_TOKEN".to_string()));
    }

    /// An allowed variable is not removed, so a subject that genuinely needs
    /// one can still be measured.
    #[test]
    fn an_allowed_variable_survives() {
        let removed = removals(&Settings {
            env_deny: vec!["GITHUB_TOKEN".into(), "GH_TOKEN".into()],
            env_allow: vec!["GITHUB_TOKEN".into()],
            gate_pct: Settings::default().gate_pct,
            credit: Settings::default().credit,
        });
        assert!(!removed.contains(&"GITHUB_TOKEN".to_string()));
        assert!(removed.contains(&"GH_TOKEN".to_string()));
    }

    #[test]
    fn zero_runs_is_rejected_not_a_panic() {
        let plan = Plan {
            cmd: vec!["true".into()],
            warmup: 0,
            runs: 0,
            dir: None,
            settings: Settings::default(),
        };
        assert!(wall(&plan).is_err());
    }

    #[test]
    fn wall_reports_min_le_p50_le_max() {
        let plan = Plan {
            cmd: vec!["true".into()],
            warmup: 1,
            runs: 5,
            dir: None,
            settings: Settings::default(),
        };
        let m = wall(&plan).unwrap();
        assert!(m["wall_min_ms"] <= m["wall_p50_ms"]);
        assert!(m["wall_p50_ms"] <= m["wall_max_ms"]);
        assert_eq!(m["wall_n"], 5.0);
    }
}