Skip to main content

amont_runtime/
gate_evidence.rs

1//! What the stamps ADD UP TO — the statistics side of
2//! [`crate::gate_stamp`].
3//!
4//! Every push that runs a gate leaves a `run` line in a note keyed by the
5//! tree the gate ran against (see that module's "second half of the note").
6//! Across a few weeks that is a per-gate outcome dataset, already on the
7//! machine, that nothing read. This module reads it, and answers two
8//! questions with it:
9//!
10//! 1. **Is a gate still doing its job?** The fleet audit of 2026-09-19 found
11//!    four mechanisms that had stopped checking anything and said nothing
12//!    about it: a nested lockfile audited from the repository root, a
13//!    `govulncheck` built with a Go too old for the database, `uv` invoked
14//!    without a `.venv`, and an `npm` that silently timed out. Each one went
15//!    green. Each one, on the record here, also went from minutes to
16//!    milliseconds on the day it broke. [`summarise`] is that comparison,
17//!    made explicit.
18//! 2. **Which gate should run first?** With `order = evidence`, the push
19//!    gates are ordered by what the record says they cost and how often they
20//!    catch something, so a push that is going to fail fails early. See
21//!    [`order_by_evidence`].
22//!
23//! # Three rules this module obeys
24//!
25//! **It never skips a check.** Ordering is a permutation; nothing here
26//! removes a gate, shortens one or lets one be assumed to pass. A prediction
27//! that a gate will pass is not evidence that it did, and a check that does
28//! not run cannot be stamped or attested — the whole chain from
29//! [`crate::gate_stamp`] to [`crate::attest`] rests on a record of a run that
30//! happened, and a statistical model of one is not that.
31//!
32//! **It abstains rather than guessing.** Under [`Thresholds::min_runs`]
33//! verdicts a gate is reported with its counts and NO flags, and the report
34//! says why in words (`insufficient history (2 runs)`). A dashboard that
35//! calls a two-run gate "flaky" teaches people to ignore the column.
36//!
37//! **Its thresholds are numbers, written down.** Every flag below names the
38//! comparison it made and the number it made it against, both in the report
39//! and in `docs/gate-evidence.md`, and every one of them is a flag on the
40//! command line. A heuristic nobody can see the inside of is a heuristic
41//! nobody can argue with.
42
43use std::collections::BTreeMap;
44use std::path::Path;
45
46use crate::gate_stamp::{Note, Run, RunOutcome, NOTES_REF};
47
48/// Seconds since the epoch, or 0 if the clock is before it. The record is
49/// local and single-machine, so this is the same clock on both sides of every
50/// comparison made here.
51pub fn now() -> u64 {
52    std::time::SystemTime::now()
53        .duration_since(std::time::UNIX_EPOCH)
54        .map(|d| d.as_secs())
55        .unwrap_or(0)
56}
57
58/// Every run recorded in one repository, with the fingerprint each was
59/// recorded against.
60///
61/// The fingerprint is the note's key — the git tree the gate ran on. Two runs
62/// with the same fingerprint read identical content, which is what makes
63/// disagreement between them evidence of flakiness rather than of a change.
64#[derive(Debug, Clone, Default, PartialEq, Eq)]
65pub struct History {
66    pub runs: Vec<(String, Run)>,
67}
68
69impl History {
70    pub fn is_empty(&self) -> bool {
71        self.runs.is_empty()
72    }
73
74    /// The runs inside the window, newest last. `window_days` of 0 means
75    /// "everything recorded".
76    fn within(&self, now: u64, window_days: u64) -> Vec<(&str, &Run)> {
77        let floor = if window_days == 0 {
78            0
79        } else {
80            now.saturating_sub(window_days.saturating_mul(86_400))
81        };
82        let mut runs: Vec<(&str, &Run)> = self
83            .runs
84            .iter()
85            .filter(|(_, r)| r.at >= floor)
86            .map(|(fp, r)| (fp.as_str(), r))
87            .collect();
88        // Ties broken by gate name so the order is total, and the report of
89        // one repository is the same on two machines reading one ref.
90        runs.sort_by(|a, b| a.1.at.cmp(&b.1.at).then_with(|| a.1.gate.cmp(&b.1.gate)));
91        runs
92    }
93}
94
95/// Read one repository's recorded runs.
96///
97/// Two spawns whatever the size of the history: `notes list` names every note
98/// and its object, and one `cat-file --batch` reads all the bodies. A git
99/// that will not answer, an absent ref and a repository with no history are
100/// the same empty answer — a report that cannot be built is reported as
101/// missing, never as zero.
102pub fn history_in(repo: &Path) -> History {
103    let Some(list) = crate::git::stdout_in(repo, &["notes", "--ref", NOTES_REF, "list"]) else {
104        return History::default();
105    };
106    // `<note blob> <annotated object>` per line. The annotated object is the
107    // fingerprint; the blob is what `cat-file` is about to be asked for.
108    let mut keyed: BTreeMap<String, String> = BTreeMap::new();
109    let mut stdin = String::new();
110    for line in list.lines() {
111        let mut t = line.split_whitespace();
112        if let (Some(blob), Some(object)) = (t.next(), t.next()) {
113            keyed.insert(blob.to_string(), object.to_string());
114            stdin.push_str(blob);
115            stdin.push('\n');
116        }
117    }
118    if keyed.is_empty() {
119        return History::default();
120    }
121    let Some(batch) = crate::git::stdout_piped_in(repo, &["cat-file", "--batch"], stdin.as_bytes())
122    else {
123        return History::default();
124    };
125    let mut history = History::default();
126    for (blob, body) in parse_batch(&batch) {
127        let Some(fingerprint) = keyed.get(&blob) else {
128            continue;
129        };
130        for run in Note::parse(&body).runs {
131            history.runs.push((fingerprint.clone(), run));
132        }
133    }
134    history
135}
136
137/// Split `git cat-file --batch` output into `(object id, body)`.
138///
139/// Line-oriented rather than byte-counted, and that is a deliberate trade.
140/// The header git writes is `<oid> <type> <size>`, and a body line could in
141/// principle look like one — but every body this ref holds was written by
142/// [`crate::gate_stamp`], whose lines begin `amont-gate-` or `run `. The
143/// alternative, honouring the byte count, cannot be written against a helper
144/// that trims its output, and a note this reader misreads costs a row in a
145/// report rather than a verdict.
146fn parse_batch(out: &str) -> Vec<(String, String)> {
147    let mut found = Vec::new();
148    let mut current: Option<(String, Vec<&str>)> = None;
149    for line in out.lines() {
150        let mut t = line.split_whitespace();
151        let (oid, kind, size) = (t.next(), t.next(), t.next());
152        let is_header = matches!((oid, kind, size), (Some(o), Some("blob"), Some(s))
153            if o.len() >= 40
154                && o.chars().all(|c| c.is_ascii_hexdigit())
155                && s.parse::<u64>().is_ok()
156                && t.next().is_none());
157        if is_header {
158            if let Some((oid, body)) = current.take() {
159                found.push((oid, body.join("\n")));
160            }
161            current = Some((oid.unwrap_or_default().to_string(), Vec::new()));
162            continue;
163        }
164        if let Some((_, body)) = current.as_mut() {
165            body.push(line);
166        }
167    }
168    if let Some((oid, body)) = current.take() {
169        found.push((oid, body.join("\n")));
170    }
171    found
172}
173
174/// The numbers every flag below is decided against. Defaults are documented
175/// in `docs/gate-evidence.md` and overridable per invocation.
176#[derive(Debug, Clone, Copy, PartialEq, Eq)]
177pub struct Thresholds {
178    /// How far back the report looks, in days. 0 means the whole record.
179    pub window_days: u64,
180    /// Verdicts below which no flag is computed and the report says so.
181    pub min_runs: usize,
182    /// A last duration below this percentage of the median is a collapse.
183    pub noop_ratio_percent: u64,
184    /// …but only for a gate whose median is at least this many seconds. A
185    /// gate that has always taken 200 ms has no collapse to detect.
186    pub noop_median_secs: u64,
187    /// A pass faster than this, from a gate that has never once passed this
188    /// fast, is the other shape of the same failure.
189    pub fast_pass_ms: u64,
190    /// How many later fingerprints may go by without this gate running before
191    /// it is called stale.
192    pub stale_runs: usize,
193    /// …or how many days, when other gates have run more recently.
194    pub stale_days: u64,
195}
196
197impl Default for Thresholds {
198    fn default() -> Self {
199        Thresholds {
200            window_days: 90,
201            min_runs: 5,
202            noop_ratio_percent: 10,
203            noop_median_secs: 30,
204            fast_pass_ms: 1_000,
205            stale_runs: 10,
206            stale_days: 30,
207        }
208    }
209}
210
211#[derive(Debug, Clone, Copy, PartialEq, Eq)]
212pub enum FlagKind {
213    /// The gate stopped doing the work it used to do.
214    NoOpSuspect,
215    /// One fingerprint, two answers.
216    Flaky,
217    /// It has not run while its neighbours have.
218    Stale,
219}
220
221impl FlagKind {
222    pub fn as_str(self) -> &'static str {
223        match self {
224            FlagKind::NoOpSuspect => "no-op suspect",
225            FlagKind::Flaky => "flaky",
226            FlagKind::Stale => "stale",
227        }
228    }
229}
230
231/// A flag and the comparison that raised it. The reason travels WITH the
232/// flag: a dashboard cell saying "no-op suspect" and nothing else is an
233/// accusation, and nobody can check an accusation.
234#[derive(Debug, Clone, PartialEq, Eq)]
235pub struct Flag {
236    pub kind: FlagKind,
237    pub why: String,
238}
239
240/// What the record says about one gate in one repository.
241#[derive(Debug, Clone, PartialEq, Eq)]
242pub struct GateReport {
243    pub gate: String,
244    /// Runs in the window, of every outcome.
245    pub runs: usize,
246    pub passed: usize,
247    pub failed: usize,
248    /// Ran and did not judge: the tool was missing, or the repository lacks
249    /// what turns the gate on. Counted apart from a pass, always — this is
250    /// the number the fleet audit was blind to.
251    pub unavailable: usize,
252    /// Median wall-clock of the runs that reached a verdict, milliseconds.
253    pub median_ms: u64,
254    pub last_ms: u64,
255    pub last_at: u64,
256    pub last_outcome: Option<RunOutcome>,
257    /// Distinct fingerprints this gate ran against.
258    pub fingerprints: usize,
259    /// Fingerprints that other gates judged AFTER this gate last ran, and
260    /// that this gate did not. The "in pushes" half of staleness.
261    pub missed: usize,
262    pub flags: Vec<Flag>,
263    /// Set when the history is too short to say anything, with the count in
264    /// words. A report that abstains says so; it does not print an empty
265    /// flags column that reads as "all clear".
266    pub abstained: Option<String>,
267}
268
269impl GateReport {
270    /// Age of the last run in whole days.
271    pub fn last_age_days(&self, now: u64) -> u64 {
272        now.saturating_sub(self.last_at) / 86_400
273    }
274}
275
276/// The per-gate report for one repository's history.
277///
278/// Pure over `now`, so a test can hand it a clock. Gates appear in name
279/// order; a gate that has never run does not appear at all, because this
280/// record only knows what ran — `amont list` is the answer to what is
281/// declared.
282pub fn summarise(history: &History, now: u64, t: &Thresholds) -> Vec<GateReport> {
283    let runs = history.within(now, t.window_days);
284    let mut by_gate: BTreeMap<&str, Vec<(&str, &Run)>> = BTreeMap::new();
285    for &(fp, run) in &runs {
286        by_gate
287            .entry(run.gate.as_str())
288            .or_default()
289            .push((fp, run));
290    }
291    by_gate
292        .into_iter()
293        .map(|(gate, mine)| one_gate(gate, &mine, &runs, now, t))
294        .collect()
295}
296
297fn one_gate(
298    gate: &str,
299    mine: &[(&str, &Run)],
300    all: &[(&str, &Run)],
301    now: u64,
302    t: &Thresholds,
303) -> GateReport {
304    let verdicts: Vec<(&str, &Run)> = mine
305        .iter()
306        .filter(|(_, r)| r.outcome.is_verdict())
307        .copied()
308        .collect();
309    let passed = verdicts
310        .iter()
311        .filter(|(_, r)| r.outcome == RunOutcome::Passed)
312        .count();
313    let failed = verdicts.len() - passed;
314    let unavailable = mine
315        .iter()
316        .filter(|(_, r)| r.outcome == RunOutcome::Unavailable)
317        .count();
318    let median_ms = median(&verdicts.iter().map(|(_, r)| r.ms).collect::<Vec<_>>());
319    let last = mine.last().copied();
320    let mut fingerprints: Vec<&str> = mine.iter().map(|(fp, _)| *fp).collect();
321    fingerprints.sort_unstable();
322    fingerprints.dedup();
323
324    // Fingerprints judged by ANY gate after this one last ran, minus the ones
325    // this gate was part of. "Other gates kept working and this one did not"
326    // is the whole claim, so it is counted rather than inferred from a date.
327    let last_at = last.map(|(_, r)| r.at).unwrap_or(0);
328    let mut later: Vec<&str> = all
329        .iter()
330        .filter(|(_, r)| r.at > last_at && r.gate != gate)
331        .map(|(fp, _)| *fp)
332        .collect();
333    later.sort_unstable();
334    later.dedup();
335    let missed = later
336        .iter()
337        .filter(|fp| !fingerprints.contains(*fp))
338        .count();
339
340    let mut report = GateReport {
341        gate: gate.to_string(),
342        runs: mine.len(),
343        passed,
344        failed,
345        unavailable,
346        median_ms,
347        last_ms: last.map(|(_, r)| r.ms).unwrap_or(0),
348        last_at,
349        last_outcome: last.map(|(_, r)| r.outcome),
350        fingerprints: fingerprints.len(),
351        missed,
352        flags: Vec::new(),
353        abstained: None,
354    };
355
356    // Staleness is decided first, because it is the one question a short
357    // history can still answer: it is a claim about the OTHER gates' record,
358    // not about the distribution of this one's.
359    if let Some(flag) = stale_flag(&report, all, now, t) {
360        report.flags.push(flag);
361    }
362
363    if verdicts.len() < t.min_runs {
364        report.abstained = Some(format!(
365            "insufficient history ({} verdict{})",
366            verdicts.len(),
367            if verdicts.len() == 1 { "" } else { "s" }
368        ));
369        return report;
370    }
371
372    if let Some(flag) = noop_flag(&verdicts, median_ms, t) {
373        report.flags.push(flag);
374    }
375    if let Some(flag) = flaky_flag(gate, &verdicts) {
376        report.flags.push(flag);
377    }
378    report
379}
380
381/// Did the gate's cost collapse?
382///
383/// Two shapes of the same failure, and both are needed: a suite that used to
384/// take eleven minutes and now takes 0.4 s is caught by the ratio; a gate
385/// installed broken — passing in milliseconds from its very first run — has
386/// no earlier median to collapse from, and is caught by the second rule only
387/// once it has enough verdicts to be sure it never once did real work.
388fn noop_flag(verdicts: &[(&str, &Run)], median_ms: u64, t: &Thresholds) -> Option<Flag> {
389    let (_, last) = *verdicts.last()?;
390    if median_ms >= t.noop_median_secs.saturating_mul(1_000)
391        && last.ms.saturating_mul(100) < median_ms.saturating_mul(t.noop_ratio_percent)
392    {
393        return Some(Flag {
394            kind: FlagKind::NoOpSuspect,
395            why: format!(
396                "last run {} against a median of {} ({}% of it, threshold {}%)",
397                duration(last.ms),
398                duration(median_ms),
399                last.ms.saturating_mul(100) / median_ms.max(1),
400                t.noop_ratio_percent
401            ),
402        });
403    }
404    if last.outcome == RunOutcome::Passed
405        && last.ms < t.fast_pass_ms
406        && verdicts[..verdicts.len() - 1]
407            .iter()
408            .all(|(_, r)| r.ms >= t.fast_pass_ms)
409    {
410        return Some(Flag {
411            kind: FlagKind::NoOpSuspect,
412            why: format!(
413                "passed in {}, and none of the {} earlier runs ever finished under {}",
414                duration(last.ms),
415                verdicts.len() - 1,
416                duration(t.fast_pass_ms)
417            ),
418        });
419    }
420    None
421}
422
423/// One fingerprint, two answers. The content did not change between them, so
424/// something else decided the verdict.
425fn flaky_flag(gate: &str, verdicts: &[(&str, &Run)]) -> Option<Flag> {
426    let mut by_fp: BTreeMap<&str, (usize, usize)> = BTreeMap::new();
427    for &(fp, run) in verdicts {
428        let e = by_fp.entry(fp).or_insert((0, 0));
429        if run.outcome == RunOutcome::Passed {
430            e.0 += 1;
431        } else {
432            e.1 += 1;
433        }
434    }
435    let split: Vec<(&str, usize, usize)> = by_fp
436        .into_iter()
437        .filter(|(_, (pass, fail))| *pass > 0 && *fail > 0)
438        .map(|(fp, (pass, fail))| (fp, pass, fail))
439        .collect();
440    let (fp, pass, fail) = *split.first()?;
441    Some(Flag {
442        kind: FlagKind::Flaky,
443        why: format!(
444            "{gate} both passed ({pass}) and failed ({fail}) on tree {}{}",
445            short(fp),
446            if split.len() > 1 {
447                format!(", and on {} other tree(s)", split.len() - 1)
448            } else {
449                String::new()
450            }
451        ),
452    })
453}
454
455/// Has it stopped running while its neighbours kept going?
456fn stale_flag(report: &GateReport, all: &[(&str, &Run)], now: u64, t: &Thresholds) -> Option<Flag> {
457    if report.runs == 0 {
458        return None;
459    }
460    let newest_other = all
461        .iter()
462        .filter(|(_, r)| r.gate != report.gate)
463        .map(|(_, r)| r.at)
464        .max()?;
465    if newest_other <= report.last_at {
466        return None;
467    }
468    if report.missed >= t.stale_runs {
469        return Some(Flag {
470            kind: FlagKind::Stale,
471            why: format!(
472                "{} later tree(s) were judged by other gates and not by this one (threshold {})",
473                report.missed, t.stale_runs
474            ),
475        });
476    }
477    let age = report.last_age_days(now);
478    if age >= t.stale_days {
479        return Some(Flag {
480            kind: FlagKind::Stale,
481            why: format!(
482                "last ran {age} days ago (threshold {}), while another gate ran {} days ago",
483                t.stale_days,
484                now.saturating_sub(newest_other) / 86_400
485            ),
486        });
487    }
488    None
489}
490
491/// The order the push gates should be attempted in, as a permutation of the
492/// indices of `gates`.
493///
494/// **What this is not.** It is not a prediction that a gate will fail, and
495/// nothing is skipped on the strength of it: every gate in `gates` appears in
496/// the result exactly once. Fail-fast is what makes the order worth anything
497/// — pre-push already stops at the first blocking failure — so ordering only
498/// changes WHEN the news arrives, never WHETHER it does.
499///
500/// **The rule.** Gates that have actually failed in the window are promoted,
501/// ordered by failures per unit of time: `failures / runs / median duration`,
502/// compared as integers by cross-multiplication so the order is exact and the
503/// same everywhere. That ratio, and not the failure rate alone, is what
504/// minimises the expected time spent before a failing push fails — a gate
505/// that fails one push in ten and takes five seconds is worth attempting
506/// before one that fails one in three and takes twenty minutes.
507///
508/// Everything else — a gate that has never failed, and every gate with no
509/// record at all — keeps its declared order, after the promoted ones. The
510/// declared index is the final tie-break, so the result is deterministic for
511/// a given history.
512pub fn order_by_evidence(
513    gates: &[String],
514    history: &History,
515    now: u64,
516    window_days: u64,
517) -> Vec<usize> {
518    let runs = history.within(now, window_days);
519    let mut stats: BTreeMap<&str, (u64, u64, Vec<u64>)> = BTreeMap::new();
520    for (_, run) in &runs {
521        if !run.outcome.is_verdict() {
522            continue;
523        }
524        let e = stats.entry(run.gate.as_str()).or_insert((0, 0, Vec::new()));
525        e.0 += 1;
526        if run.outcome == RunOutcome::Failed {
527            e.1 += 1;
528        }
529        e.2.push(run.ms);
530    }
531    // (failures, runs, median ms) per declared gate, or None with no record.
532    let weight = |name: &String| -> Option<(u64, u64, u64)> {
533        let (runs, fails, ms) = stats.get(name.as_str())?;
534        (*fails > 0).then(|| (*fails, *runs, median(ms).max(1)))
535    };
536    let mut order: Vec<usize> = (0..gates.len()).collect();
537    order.sort_by(|&a, &b| {
538        match (weight(&gates[a]), weight(&gates[b])) {
539            (Some((fa, ra, ma)), Some((fb, rb, mb))) => {
540                // fa/(ra*ma) vs fb/(rb*mb), cross-multiplied in u128 so no
541                // float ever decides the order of a hook.
542                let left = u128::from(fa) * u128::from(rb) * u128::from(mb);
543                let right = u128::from(fb) * u128::from(ra) * u128::from(ma);
544                right.cmp(&left).then(a.cmp(&b))
545            }
546            (Some(_), None) => std::cmp::Ordering::Less,
547            (None, Some(_)) => std::cmp::Ordering::Greater,
548            (None, None) => a.cmp(&b),
549        }
550    });
551    order
552}
553
554fn median(values: &[u64]) -> u64 {
555    if values.is_empty() {
556        return 0;
557    }
558    let mut v = values.to_vec();
559    v.sort_unstable();
560    let mid = v.len() / 2;
561    if v.len() % 2 == 1 {
562        v[mid]
563    } else {
564        // The lower of the two middles rather than their mean: the numbers
565        // are durations, and an average of two invents a duration nothing
566        // ever took.
567        v[mid - 1]
568    }
569}
570
571/// A duration a person can read. Milliseconds up to a second, then seconds,
572/// then minutes — the report is scanned, not computed with.
573pub fn duration(ms: u64) -> String {
574    if ms < 1_000 {
575        return format!("{ms} ms");
576    }
577    if ms < 60_000 {
578        return format!("{}.{} s", ms / 1000, (ms % 1000) / 100);
579    }
580    format!("{} m {:02} s", ms / 60_000, (ms % 60_000) / 1000)
581}
582
583/// A fingerprint, shortened the way git shortens one.
584fn short(oid: &str) -> &str {
585    oid.get(..8).unwrap_or(oid)
586}
587
588#[cfg(test)]
589mod tests {
590    use super::*;
591
592    fn run(at: u64, gate: &str, outcome: RunOutcome, ms: u64) -> Run {
593        Run {
594            at,
595            gate: gate.to_string(),
596            outcome,
597            ms,
598        }
599    }
600
601    fn history(rows: &[(&str, Run)]) -> History {
602        History {
603            runs: rows
604                .iter()
605                .map(|(fp, r)| ((*fp).to_string(), r.clone()))
606                .collect(),
607        }
608    }
609
610    /// The same, where a test builds its fingerprints rather than spelling
611    /// them out.
612    fn owned_history(rows: Vec<(String, Run)>) -> History {
613        History { runs: rows }
614    }
615
616    const NOW: u64 = 1_800_000_000;
617    const DAY: u64 = 86_400;
618
619    fn of<'a>(reports: &'a [GateReport], gate: &str) -> &'a GateReport {
620        reports
621            .iter()
622            .find(|r| r.gate == gate)
623            .unwrap_or_else(|| panic!("no report for {gate}: {reports:?}"))
624    }
625
626    fn kinds(r: &GateReport) -> Vec<FlagKind> {
627        r.flags.iter().map(|f| f.kind).collect()
628    }
629
630    /// The signature the fleet audit of 2026-09-19 found four times: a gate
631    /// that used to take minutes returns in milliseconds and still passes.
632    /// Nothing in the hook can see it — the exit code is 0 — and this is the
633    /// only place the collapse is visible.
634    #[test]
635    fn a_suite_that_collapsed_to_milliseconds_is_a_no_op_suspect() {
636        let mut rows: Vec<(&str, Run)> = (0..8)
637            .map(|i| {
638                (
639                    "tree0",
640                    run(
641                        NOW - (10 - i) * DAY,
642                        "pre-push-cargo-test",
643                        RunOutcome::Passed,
644                        600_000,
645                    ),
646                )
647            })
648            .collect();
649        rows.push((
650            "tree9",
651            run(NOW - DAY, "pre-push-cargo-test", RunOutcome::Passed, 400),
652        ));
653        let reports = summarise(&history(&rows), NOW, &Thresholds::default());
654        let r = of(&reports, "pre-push-cargo-test");
655        assert_eq!(kinds(r), vec![FlagKind::NoOpSuspect]);
656        assert!(
657            r.flags[0].why.contains("400 ms") && r.flags[0].why.contains("threshold 10%"),
658            "the flag must carry the comparison it made: {:?}",
659            r.flags[0].why
660        );
661    }
662
663    /// The other shape: a gate that has ALWAYS returned instantly has no
664    /// collapse to measure, so the rule is "it has never once done real
665    /// work", and it only fires once there is enough history to say so.
666    #[test]
667    fn a_gate_that_never_once_took_real_time_is_also_suspect() {
668        let rows: Vec<(&str, Run)> = (0..6)
669            .map(|i| {
670                (
671                    "tree0",
672                    run(
673                        NOW - (7 - i) * DAY,
674                        "pre-push-audit-python",
675                        RunOutcome::Passed,
676                        if i == 5 { 30 } else { 4_000 },
677                    ),
678                )
679            })
680            .collect();
681        let reports = summarise(&history(&rows), NOW, &Thresholds::default());
682        assert_eq!(
683            kinds(of(&reports, "pre-push-audit-python")),
684            vec![FlagKind::NoOpSuspect]
685        );
686    }
687
688    /// Two verdicts, one tree. The content could not have changed between
689    /// them, so the gate did.
690    #[test]
691    fn one_fingerprint_with_two_answers_is_flaky() {
692        let rows = [
693            (
694                "treeA",
695                run(NOW - 6 * DAY, "pre-push-pytest", RunOutcome::Passed, 90_000),
696            ),
697            (
698                "treeA",
699                run(NOW - 5 * DAY, "pre-push-pytest", RunOutcome::Failed, 88_000),
700            ),
701            (
702                "treeA",
703                run(NOW - 4 * DAY, "pre-push-pytest", RunOutcome::Passed, 91_000),
704            ),
705            (
706                "treeB",
707                run(NOW - 3 * DAY, "pre-push-pytest", RunOutcome::Passed, 92_000),
708            ),
709            (
710                "treeB",
711                run(NOW - 2 * DAY, "pre-push-pytest", RunOutcome::Passed, 90_500),
712            ),
713        ];
714        let r = summarise(&history(&rows), NOW, &Thresholds::default());
715        let r = of(&r, "pre-push-pytest");
716        assert_eq!(kinds(r), vec![FlagKind::Flaky]);
717        assert!(r.flags[0].why.contains("treeA"), "{:?}", r.flags[0].why);
718    }
719
720    /// Stale is a claim about the NEIGHBOURS: other gates kept judging trees
721    /// this one did not. A quiet repository where nothing ran is not stale.
722    #[test]
723    fn a_gate_left_behind_by_its_neighbours_is_stale() {
724        let mut rows = vec![(
725            "tree0".to_string(),
726            run(
727                NOW - 40 * DAY,
728                "pre-push-go-test",
729                RunOutcome::Passed,
730                5_000,
731            ),
732        )];
733        for i in 0..12u64 {
734            rows.push((
735                format!("tree{}", i + 1),
736                run(
737                    NOW - (12 - i) * DAY,
738                    "pre-push-cargo-test",
739                    RunOutcome::Passed,
740                    5_000,
741                ),
742            ));
743        }
744        let reports = summarise(&owned_history(rows), NOW, &Thresholds::default());
745        let stale = of(&reports, "pre-push-go-test");
746        assert!(kinds(stale).contains(&FlagKind::Stale), "{:?}", stale.flags);
747        assert!(
748            of(&reports, "pre-push-cargo-test").flags.is_empty(),
749            "the gate that kept running is not stale"
750        );
751    }
752
753    /// The refusal that matters most: two runs say nothing, and the report
754    /// says THAT rather than an empty flags column.
755    #[test]
756    fn too_little_history_abstains_out_loud() {
757        let rows = [
758            (
759                "tree0",
760                run(
761                    NOW - 2 * DAY,
762                    "pre-push-cargo-test",
763                    RunOutcome::Passed,
764                    600_000,
765                ),
766            ),
767            (
768                "tree1",
769                run(NOW - DAY, "pre-push-cargo-test", RunOutcome::Passed, 300),
770            ),
771        ];
772        let reports = summarise(&history(&rows), NOW, &Thresholds::default());
773        let r = of(&reports, "pre-push-cargo-test");
774        assert_eq!(
775            r.abstained.as_deref(),
776            Some("insufficient history (2 verdicts)")
777        );
778        assert!(
779            r.flags.is_empty(),
780            "a two-run history must not raise a flag: {:?}",
781            r.flags
782        );
783        assert_eq!((r.runs, r.passed), (2, 2), "the counts are still reported");
784    }
785
786    /// A gate that could not run is not a gate that passed. The count is
787    /// kept separately, because the audit this module exists for found four
788    /// mechanisms whose whole failure mode was that nobody counted it.
789    #[test]
790    fn an_unavailable_run_is_never_counted_as_a_pass() {
791        let rows = [
792            (
793                "tree0",
794                run(
795                    NOW - 2 * DAY,
796                    "pre-push-audit-go",
797                    RunOutcome::Unavailable,
798                    200,
799                ),
800            ),
801            (
802                "tree1",
803                run(NOW - DAY, "pre-push-audit-go", RunOutcome::Unavailable, 210),
804            ),
805        ];
806        let reports = summarise(&history(&rows), NOW, &Thresholds::default());
807        let r = of(&reports, "pre-push-audit-go");
808        assert_eq!((r.passed, r.failed, r.unavailable), (0, 0, 2));
809    }
810
811    /// Outside the window is outside the report.
812    #[test]
813    fn the_window_excludes_what_is_older_than_it() {
814        let rows = [
815            (
816                "tree0",
817                run(
818                    NOW - 200 * DAY,
819                    "pre-push-cargo-test",
820                    RunOutcome::Failed,
821                    500,
822                ),
823            ),
824            (
825                "tree1",
826                run(NOW - DAY, "pre-push-cargo-test", RunOutcome::Passed, 500),
827            ),
828        ];
829        let reports = summarise(&history(&rows), NOW, &Thresholds::default());
830        assert_eq!(of(&reports, "pre-push-cargo-test").runs, 1);
831    }
832
833    /// Ordering promotes the gate that catches the most per second spent —
834    /// not the one that fails most often. A five-second audit failing one
835    /// push in six beats a twenty-minute suite failing one in three.
836    #[test]
837    fn ordering_prefers_failures_per_second_not_failure_rate() {
838        let gates: Vec<String> = ["suite", "audit"].iter().map(|s| s.to_string()).collect();
839        let mut rows: Vec<(&str, Run)> = Vec::new();
840        for i in 0..6u64 {
841            rows.push((
842                "tree0",
843                run(
844                    NOW - (10 - i) * DAY,
845                    "suite",
846                    if i < 2 {
847                        RunOutcome::Failed
848                    } else {
849                        RunOutcome::Passed
850                    },
851                    1_200_000,
852                ),
853            ));
854            rows.push((
855                "tree0",
856                run(
857                    NOW - (10 - i) * DAY,
858                    "audit",
859                    if i < 1 {
860                        RunOutcome::Failed
861                    } else {
862                        RunOutcome::Passed
863                    },
864                    5_000,
865                ),
866            ));
867        }
868        let order = order_by_evidence(&gates, &history(&rows), NOW, 90);
869        assert_eq!(order, vec![1, 0], "the cheap audit runs first");
870    }
871
872    /// With no record at all, the declared order is the order. This is the
873    /// fallback every repository starts in, and it must be exact.
874    #[test]
875    fn with_no_history_the_declared_order_is_kept() {
876        let gates: Vec<String> = ["a", "b", "c"].iter().map(|s| s.to_string()).collect();
877        assert_eq!(
878            order_by_evidence(&gates, &History::default(), NOW, 90),
879            vec![0, 1, 2]
880        );
881    }
882
883    /// A gate that has never failed is not promoted over one that has, and
884    /// the never-failed ones keep their order among themselves.
885    #[test]
886    fn only_gates_that_actually_failed_are_promoted() {
887        let gates: Vec<String> = ["a", "b", "c"].iter().map(|s| s.to_string()).collect();
888        let rows = [
889            ("t", run(NOW - 3 * DAY, "a", RunOutcome::Passed, 1_000)),
890            ("t", run(NOW - 3 * DAY, "b", RunOutcome::Passed, 1_000)),
891            ("t", run(NOW - 2 * DAY, "c", RunOutcome::Failed, 1_000)),
892            ("t", run(NOW - DAY, "c", RunOutcome::Passed, 1_000)),
893        ];
894        assert_eq!(
895            order_by_evidence(&gates, &history(&rows), NOW, 90),
896            vec![2, 0, 1]
897        );
898    }
899
900    /// Every gate handed in comes back out, exactly once — the property that
901    /// makes this a permutation and not a filter.
902    #[test]
903    fn ordering_is_a_permutation_and_never_drops_a_gate() {
904        let gates: Vec<String> = ["a", "b", "c", "d"].iter().map(|s| s.to_string()).collect();
905        let rows = [
906            ("t", run(NOW - DAY, "b", RunOutcome::Failed, 10)),
907            ("t", run(NOW - DAY, "d", RunOutcome::Failed, 10_000)),
908        ];
909        let mut order = order_by_evidence(&gates, &history(&rows), NOW, 90);
910        assert_eq!(order.len(), gates.len());
911        order.sort_unstable();
912        assert_eq!(order, vec![0, 1, 2, 3]);
913    }
914
915    /// The batch reader, against the shape git actually prints.
916    #[test]
917    fn cat_file_batch_output_splits_into_bodies() {
918        let oid = "0123456789abcdef0123456789abcdef01234567";
919        let other = "89abcdef0123456789abcdef0123456789abcdef";
920        let out = format!(
921            "{oid} blob 42\namont-gate-v1 pre-push-cargo-test\nrun 10 pre-push-cargo-test pass 5\n\
922             {other} blob 14\namont-gate-v1\n"
923        );
924        let got = parse_batch(&out);
925        assert_eq!(got.len(), 2);
926        assert_eq!(got[0].0, oid);
927        assert!(got[0].1.contains("run 10"));
928        assert_eq!(got[1].0, other);
929    }
930
931    #[test]
932    fn durations_read_like_durations() {
933        assert_eq!(duration(400), "400 ms");
934        assert_eq!(duration(5_400), "5.4 s");
935        assert_eq!(duration(662_000), "11 m 02 s");
936    }
937}