perf-sentinel-core 0.9.15

Core library for perf-sentinel: polyglot performance anti-pattern detector
Documentation
//! Alumet interval-energy math + state update.
//!
//! Methodology (why the reading is neither a power gauge nor a
//! cumulative counter) lives in `docs/design/05-GREENOPS-AND-CARBON.md`
//! "Alumet interval-energy attribution".

use std::collections::HashMap;

use super::config::AlumetConfig;
use super::state::{AlumetState, DbEnergyState, ServiceEnergy};
use crate::score::energy_state::upsert_row;
use crate::score::prom_parser::{PromSample, sum_by_label};

/// Convert one Alumet energy reading + observed op count into an
/// energy-per-op coefficient (kWh per op).
///
/// Formula:
/// ```text
///   watts            = joules_per_interval / energy_interval_secs
///   window_joules    = watts × scrape_interval_secs
///   energy_per_op_kwh = window_joules / (ops × 3_600_000)
/// ```
///
/// The division by `energy_interval_secs` is what makes an Alumet
/// reading comparable to a Scaphandre one: Alumet publishes the joules
/// burned during one source `poll_interval`, so the raw number is
/// meaningless until it is turned back into a rate. Summing raw readings
/// across scrapes would be wrong in both directions (double-counting
/// when scraping faster than Alumet flushes, dropped intervals when
/// scraping slower).
///
/// Returns `None` when the math is meaningless (zero ops, non-finite or
/// negative energy, non-positive intervals, or an overflowing product)
/// so the caller keeps the previous entry rather than publishing a
/// division-by-zero or a flapping coefficient for an idle service.
#[must_use]
pub fn compute_energy_per_op_kwh(
    joules_per_interval: f64,
    energy_interval_secs: f64,
    scrape_interval_secs: f64,
    ops: u64,
) -> Option<f64> {
    // `<= 0.0` rather than `< 0.0`: a zero reading means the exporter's
    // last flush caught the consumer idle, not that the work in this
    // scrape window was free. Publishing 0.0 would override every
    // lower-tier backend with a measured zero for a service that
    // demonstrably did I/O. Mirrors Kepler's `delta > 0.0` filter, the
    // caller keeps the previous entry instead.
    if ops == 0 {
        return None;
    }
    let kwh = compute_window_kwh(
        joules_per_interval,
        energy_interval_secs,
        scrape_interval_secs,
    )?;
    let per_op = kwh / ops as f64;
    per_op.is_finite().then_some(per_op)
}

/// Base case of [`compute_energy_per_op_kwh`]: the scrape window's
/// energy in kWh, without the per-op division. Also used directly for
/// the database cgroup (no ops).
#[must_use]
pub fn compute_window_kwh(
    joules_per_interval: f64,
    energy_interval_secs: f64,
    scrape_interval_secs: f64,
) -> Option<f64> {
    if !joules_per_interval.is_finite() || joules_per_interval <= 0.0 {
        return None;
    }
    // Config validation already rejects these, re-checked here because
    // the math is a public entry point and a zero interval would divide
    // by zero into an infinite coefficient.
    if !energy_interval_secs.is_finite()
        || energy_interval_secs <= 0.0
        || !scrape_interval_secs.is_finite()
        || scrape_interval_secs <= 0.0
    {
        return None;
    }
    let watts = joules_per_interval / energy_interval_secs;
    // 1 kWh = 3.6e6 J.
    let kwh = watts * scrape_interval_secs / 3_600_000.0;
    kwh.is_finite().then_some(kwh)
}

/// Apply a freshly-scraped Alumet batch to an [`AlumetState`]. Services
/// with no ops this window keep their previous entry.
///
/// Unlike Kepler, there is no delta bookkeeping: the reading already is
/// an interval delta, so each scrape stands alone and an exporter
/// restart needs no counter-reset guard.
///
/// Returns how many mapped services found their label on the wire,
/// independent of whether they had ops. The caller uses it to tell
/// "the endpoint answered but nothing maps to my services" (a config
/// error) from "everything matched but the services were idle" (fine).
#[allow(clippy::implicit_hasher)]
pub fn apply_scrape(
    state: &AlumetState,
    db_state: Option<&DbEnergyState>,
    samples: &[PromSample],
    op_deltas: &HashMap<String, u64>,
    cfg: &AlumetConfig,
    now_ms: u64,
) -> usize {
    // O(N) index over samples so the service loop stays O(N + M) on
    // endpoints exposing hundreds of series. Label collisions are
    // routine for Alumet (one row per RAPL domain per pod, one row per
    // socket under `label_key = "domain"`), the summing and per-row
    // validation semantics live in [`sum_by_label`].
    let by_label = sum_by_label(samples);
    let scrape_interval_secs = cfg.scrape_interval.as_secs_f64();
    // Declared database cgroup: no ops, so its energy accumulates for
    // the waste figure instead of the per-op path below. Liveness is
    // marked by the scraper on every successful scrape, not here, so
    // an idle database or a renamed label does not strand banked energy.
    if let (Some(db_cfg), Some(db)) = (cfg.database.as_ref(), db_state)
        && let Some(&joules) = by_label.get(db_cfg.label_value.as_str())
        && let Some(kwh) =
            compute_window_kwh(joules, cfg.energy_interval_secs, scrape_interval_secs)
    {
        db.add_window_kwh(kwh, now_ms);
    }
    let mut next = state.current_owned();
    let mut any_change = false;
    let mut matched = 0usize;
    for (service, label_value) in &cfg.service_mappings {
        let Some(&joules) = by_label.get(label_value.as_str()) else {
            continue;
        };
        // Counted before the ops gate: the label exists on the wire, so
        // the mapping is right even if the service happens to be idle.
        matched += 1;
        let Some(ops) = op_deltas.get(service).copied() else {
            continue;
        };
        let Some(energy_per_op) =
            compute_energy_per_op_kwh(joules, cfg.energy_interval_secs, scrape_interval_secs, ops)
        else {
            continue;
        };
        let row = ServiceEnergy {
            energy_per_op_kwh: energy_per_op,
            last_update_ms: now_ms,
        };
        upsert_row(&mut next, service, row);
        any_change = true;
    }
    if any_change {
        state.publish(next);
    }
    matched
}