Skip to main content

turnframe_eval/
baseline.rs

1//! Comparing a run against a previous one — and refusing to, when the two runs
2//! stopped being the same experiment.
3//!
4//! Two things look like "the evaluation got worse" and must not be treated the
5//! same. A **deterministic regression** is an item that used to satisfy its
6//! assertions and no longer does: something the agent *does* changed, it is
7//! reproducible, and it can be a merge blocker. A **judge drift** is the same
8//! behaviour graded differently: a signal about the measurement, and gating a
9//! merge on it is how a team learns to ignore its own evaluation. [`ChangeKind`]
10//! keeps them in separate variants.
11//!
12//! The third thing is an item that changed because the code under test generates
13//! part of it, which silently unpairs the comparison. Exclusion is then as loud
14//! as the score, there is no headline to read past
15//! [`ComparisonPolicy::max_excluded_share`], and the three kinds of item change
16//! have three different names. [`NoiseFloor`] makes a difference news only when
17//! it is bigger than the one the unchanged system produces.
18//!
19//! The reasoning, and what each of the three names means, is in
20//! [`docs/evaluation.md`](https://github.com/turnframe-rs/turnframe/blob/main/docs/evaluation.md).
21//!
22
23use serde::{Deserialize, Serialize};
24
25use crate::assertions::ExpectationName;
26use crate::control::NoiseFloor;
27use crate::corpus::{ItemId, ItemPart, PartProvenance};
28use crate::judge::{JudgeCriterion, ratio};
29use crate::report::{EvalReport, ItemReport, ReliabilityCategory};
30
31/// How much movement is noise rather than news.
32#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
33#[serde(deny_unknown_fields, default)]
34pub struct DriftTolerance {
35    /// Change in deterministic pass rate below which nothing is reported. Zero
36    /// by default: a deterministic assertion either holds or it does not, and
37    /// pretending a drop of one sample in twenty is noise is how a regression
38    /// gets shipped.
39    pub pass_rate_epsilon: f64,
40    /// Change in mean judge score below which nothing is reported. A judge
41    /// moving by a tenth of a point is the judge breathing.
42    pub judge_score_epsilon: f64,
43}
44
45impl Default for DriftTolerance {
46    fn default() -> Self {
47        Self {
48            pass_rate_epsilon: 0.0,
49            judge_score_epsilon: 0.25,
50        }
51    }
52}
53
54/// Everything [`compare`] needs beyond the two reports.
55///
56/// ```
57/// use turnframe_eval::baseline::ComparisonPolicy;
58///
59/// // A tenth of the corpus may go unpaired before a headline is refused.
60/// assert!((ComparisonPolicy::default().max_excluded_share - 0.1).abs() < 1e-9);
61/// ```
62#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
63#[serde(deny_unknown_fields, default)]
64pub struct ComparisonPolicy {
65    /// How much movement is noise rather than news.
66    pub tolerance: DriftTolerance,
67    /// The fraction of the paired corpus that may be excluded before the
68    /// comparison refuses to produce a headline number.
69    ///
70    /// A tenth by default, and deliberately low. The number exists because a
71    /// report over a third of a corpus reads exactly like a report over all of
72    /// it, and nothing in the shape of the output would ever tell a reader
73    /// otherwise. Raise it only when you have looked at the exclusions and
74    /// decided they are acceptable.
75    pub max_excluded_share: f64,
76    /// What the same code produced against itself, when a
77    /// [control run](crate::control) measured it.
78    ///
79    /// `None` leaves every change [`NoiseVerdict::Unmeasured`]: without a
80    /// control run there is no honest way to say whether a difference is signal.
81    #[serde(default, skip_serializing_if = "Option::is_none")]
82    pub noise_floor: Option<NoiseFloor>,
83}
84
85impl Default for ComparisonPolicy {
86    fn default() -> Self {
87        Self {
88            tolerance: DriftTolerance::default(),
89            max_excluded_share: 0.1,
90            noise_floor: None,
91        }
92    }
93}
94
95impl ComparisonPolicy {
96    /// A policy with this tolerance and the default exclusion ceiling.
97    #[must_use]
98    pub fn new(tolerance: DriftTolerance) -> Self {
99        Self {
100            tolerance,
101            ..Self::default()
102        }
103    }
104
105    /// Sets how much of the paired corpus may be excluded before the headline is
106    /// withheld.
107    #[must_use]
108    pub const fn with_max_excluded_share(mut self, share: f64) -> Self {
109        self.max_excluded_share = share;
110        self
111    }
112
113    /// Attaches the floor a [control run](crate::control) measured.
114    #[must_use]
115    pub fn with_noise_floor(mut self, floor: NoiseFloor) -> Self {
116        self.noise_floor = Some(floor);
117        self
118    }
119}
120
121/// What changed between two runs of the same suite.
122///
123/// The field order is the reading order, and it is deliberate: the headline (or
124/// the refusal to produce one) and the exclusions come before the changes, in
125/// the machine-readable form as much as in [`Comparison::summary`].
126#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
127#[serde(deny_unknown_fields)]
128pub struct Comparison {
129    /// The suite the baseline reported on.
130    pub baseline_suite: String,
131    /// The suite the current run reported on.
132    pub current_suite: String,
133    /// The figures, or the reason there are none.
134    pub headline: Headline,
135    /// Items that appeared in both runs and could not be paired, with why.
136    pub excluded: Vec<ExcludedItem>,
137    /// Every change on the items that stayed paired, in current-run item order
138    /// followed by removals.
139    pub changes: Vec<Change>,
140}
141
142impl Comparison {
143    /// Returns `true` when nothing changed and nothing was excluded.
144    #[must_use]
145    pub fn is_unchanged(&self) -> bool {
146        self.changes.is_empty() && self.excluded.is_empty()
147    }
148
149    /// The deterministic regressions — behaviour that got worse.
150    #[must_use]
151    pub fn deterministic_regressions(&self) -> Vec<&Change> {
152        self.changes
153            .iter()
154            .filter(|change| matches!(change.kind, ChangeKind::DeterministicRegression { .. }))
155            .collect()
156    }
157
158    /// The deterministic regressions that a [control run](crate::control) says
159    /// are larger than the harness's own variation.
160    ///
161    /// Without a noise floor this is every regression: an
162    /// [`Unmeasured`](NoiseVerdict::Unmeasured) movement is not known to be
163    /// noise, and treating the unknown as harmless is how a regression ships.
164    #[must_use]
165    pub fn signal_regressions(&self) -> Vec<&Change> {
166        self.deterministic_regressions()
167            .into_iter()
168            .filter(|change| change.against_noise != NoiseVerdict::WithinNoise)
169            .collect()
170    }
171
172    /// The judge drifts — the same behaviour, graded differently.
173    #[must_use]
174    pub fn judge_drifts(&self) -> Vec<&Change> {
175        self.changes
176            .iter()
177            .filter(|change| matches!(change.kind, ChangeKind::JudgeDrift { .. }))
178            .collect()
179    }
180
181    /// The items whose declared-derived parts changed: the intended effect of
182    /// the change being measured, and not a broken pairing.
183    #[must_use]
184    pub fn derived_input_changes(&self) -> Vec<&Change> {
185        self.changes
186            .iter()
187            .filter(|change| matches!(change.kind, ChangeKind::DerivedInput { .. }))
188            .collect()
189    }
190
191    /// Returns `true` when at least one item's deterministic assertions got
192    /// worse.
193    #[must_use]
194    pub fn has_deterministic_regression(&self) -> bool {
195        !self.deterministic_regressions().is_empty()
196    }
197
198    /// Returns `true` when at least one regression exceeds the measured noise
199    /// floor. This is the question a merge gate should ask.
200    #[must_use]
201    pub fn has_signal_regression(&self) -> bool {
202        !self.signal_regressions().is_empty()
203    }
204
205    /// Returns `true` when at least one judge score moved beyond the tolerance.
206    /// This is a question about the measurement, not about the agent.
207    #[must_use]
208    pub fn has_judge_drift(&self) -> bool {
209        !self.judge_drifts().is_empty()
210    }
211
212    /// The excluded items whose `recorded` parts changed.
213    ///
214    /// These are not a result about the model and must never be read as one:
215    /// something regenerated testimony about what the system actually emitted,
216    /// which is a defect in the corpus or in the tooling that touched it. They
217    /// are excluded from every figure, and this is where they are named.
218    #[must_use]
219    pub fn corpus_defects(&self) -> Vec<&ExcludedItem> {
220        self.excluded
221            .iter()
222            .filter(|excluded| excluded.is_corpus_defect())
223            .collect()
224    }
225
226    /// Returns `true` when a recorded part changed anywhere in the corpus.
227    #[must_use]
228    pub fn has_corpus_defect(&self) -> bool {
229        !self.corpus_defects().is_empty()
230    }
231
232    /// The fraction of the paired corpus that could not be compared.
233    #[must_use]
234    pub fn excluded_share(&self) -> f64 {
235        self.headline.excluded_share()
236    }
237
238    /// Returns `true` when there is no headline figure to read, because too much
239    /// of the corpus went unpaired.
240    #[must_use]
241    pub fn is_withheld(&self) -> bool {
242        matches!(self.headline, Headline::Withheld(_))
243    }
244
245    /// A readable rendering.
246    ///
247    /// The first line is the pairing: how much of the corpus was compared and
248    /// how much was dropped. The second is the headline, or the refusal. Only
249    /// then come the individual changes, regressions first.
250    #[must_use]
251    pub fn summary(&self) -> String {
252        use std::fmt::Write as _;
253        let mut out = String::new();
254        let counts = self.headline.counts();
255        let _ = writeln!(
256            out,
257            "{} → {}: {} of {} paired item(s) compared, {} excluded ({:.0}% of the paired corpus)",
258            self.baseline_suite,
259            self.current_suite,
260            counts.compared,
261            counts.compared + counts.excluded,
262            counts.excluded,
263            self.excluded_share() * 100.0
264        );
265        if counts.unverified > 0 {
266            let _ = writeln!(
267                out,
268                "  {} item(s) carried no fingerprint on one side, so their pairing was not checked",
269                counts.unverified
270            );
271        }
272        let defects = self.corpus_defects();
273        if !defects.is_empty() {
274            // Said before any figure, because it is not a figure: something
275            // rewrote testimony, and no number in this report answers that.
276            let _ = writeln!(
277                out,
278                "CORPUS DEFECT in {} item(s): a `recorded` part changed. Testimony about what the \
279                 system emitted must never be regenerated — this is a problem with the corpus or \
280                 the tooling, not a result about the model.",
281                defects.len()
282            );
283        }
284        match &self.headline {
285            Headline::Measured(figures) => {
286                let _ = writeln!(
287                    out,
288                    "deterministic pass rate {:.2} → {:.2} ({:+.2}) over the compared items",
289                    figures.baseline_pass_rate, figures.current_pass_rate, figures.pass_rate_delta
290                );
291            }
292            Headline::Withheld(withheld) => {
293                let _ = writeln!(out, "NO HEADLINE FIGURE: {}", withheld.reason);
294            }
295        }
296        let _ = writeln!(
297            out,
298            "{} change(s), {} deterministic regression(s) of which {} exceed the noise floor, \
299             {} judge drift(s), {} item(s) changed as the projection intended",
300            self.changes.len(),
301            self.deterministic_regressions().len(),
302            self.signal_regressions().len(),
303            self.judge_drifts().len(),
304            self.derived_input_changes().len()
305        );
306        for excluded in &self.excluded {
307            let _ = writeln!(out, "  EXCLUDED {excluded}");
308        }
309        let mut ordered: Vec<&Change> = self.deterministic_regressions();
310        ordered.extend(
311            self.changes.iter().filter(|change| {
312                !matches!(change.kind, ChangeKind::DeterministicRegression { .. })
313            }),
314        );
315        for change in ordered {
316            let _ = writeln!(out, "  {change}");
317        }
318        out
319    }
320}
321
322/// The figures of a comparison, or the reason there are none.
323///
324/// There is no accessor anywhere that produces a pass-rate delta outside
325/// [`HeadlineFigures`], which is what makes the refusal a refusal rather than a
326/// warning a caller can step over.
327#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
328#[serde(rename_all = "snake_case")]
329#[non_exhaustive]
330pub enum Headline {
331    /// Enough of the corpus stayed paired for the figures to mean something.
332    Measured(HeadlineFigures),
333    /// Too much of the corpus went unpaired. No figure is produced.
334    Withheld(WithheldHeadline),
335}
336
337/// The counts every headline carries, whatever it decided.
338struct HeadlineCounts {
339    compared: usize,
340    excluded: usize,
341    unverified: usize,
342}
343
344impl Headline {
345    /// The figures, when there are any.
346    #[must_use]
347    pub const fn figures(&self) -> Option<&HeadlineFigures> {
348        match self {
349            Self::Measured(figures) => Some(figures),
350            Self::Withheld(_) => None,
351        }
352    }
353
354    /// The refusal, when the headline was withheld.
355    #[must_use]
356    pub const fn withheld(&self) -> Option<&WithheldHeadline> {
357        match self {
358            Self::Withheld(reason) => Some(reason),
359            Self::Measured(_) => None,
360        }
361    }
362
363    /// The fraction of the paired corpus that was excluded — reported whether or
364    /// not there is a figure beside it.
365    #[must_use]
366    pub const fn excluded_share(&self) -> f64 {
367        match self {
368            Self::Measured(figures) => figures.excluded_share,
369            Self::Withheld(withheld) => withheld.excluded_share,
370        }
371    }
372
373    const fn counts(&self) -> HeadlineCounts {
374        match self {
375            Self::Measured(figures) => HeadlineCounts {
376                compared: figures.items_compared,
377                excluded: figures.items_excluded,
378                unverified: figures.unverified_pairings,
379            },
380            Self::Withheld(withheld) => HeadlineCounts {
381                compared: withheld.items_compared,
382                excluded: withheld.items_excluded,
383                unverified: withheld.unverified_pairings,
384            },
385        }
386    }
387}
388
389/// The figures of a comparison that stood.
390///
391/// Every rate is over the **compared** items only. An excluded item contributes
392/// nothing to either side, because a measurement it is not part of is not a
393/// measurement of it.
394#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
395#[serde(deny_unknown_fields)]
396pub struct HeadlineFigures {
397    /// Deterministic pass rate of the baseline over the compared items.
398    pub baseline_pass_rate: f64,
399    /// Deterministic pass rate of the current run over the compared items.
400    pub current_pass_rate: f64,
401    /// Current minus baseline.
402    pub pass_rate_delta: f64,
403    /// How many items were compared.
404    pub items_compared: usize,
405    /// How many appeared in both runs and could not be paired.
406    pub items_excluded: usize,
407    /// Excluded over compared plus excluded.
408    pub excluded_share: f64,
409    /// How many items carried no fingerprint on one side, so their pairing could
410    /// not be checked at all.
411    pub unverified_pairings: usize,
412}
413
414/// Why a comparison produced no figure.
415#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
416#[serde(deny_unknown_fields)]
417pub struct WithheldHeadline {
418    /// Excluded over compared plus excluded.
419    pub excluded_share: f64,
420    /// The ceiling it crossed.
421    pub max_excluded_share: f64,
422    /// How many items were still comparable.
423    pub items_compared: usize,
424    /// How many appeared in both runs and could not be paired.
425    pub items_excluded: usize,
426    /// How many items carried no fingerprint on one side.
427    pub unverified_pairings: usize,
428    /// One sentence a person reads instead of a number.
429    pub reason: String,
430}
431
432/// One item that appeared in both runs and could not be paired.
433#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
434#[serde(deny_unknown_fields)]
435pub struct ExcludedItem {
436    /// Which item.
437    pub item: ItemId,
438    /// Why its pairing did not survive. An item can hit more than one reason at
439    /// once — an edited fixture *and* rewritten testimony — and both are said.
440    pub reasons: Vec<ExclusionReason>,
441}
442
443impl ExcludedItem {
444    /// Returns `true` when one of the reasons is evidence that something
445    /// rewrote a recorded part.
446    #[must_use]
447    pub fn is_corpus_defect(&self) -> bool {
448        self.reasons
449            .iter()
450            .any(|reason| matches!(reason, ExclusionReason::RecordedPartChanged { .. }))
451    }
452}
453
454impl std::fmt::Display for ExcludedItem {
455    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
456        write!(f, "{}: ", self.item)?;
457        for (index, reason) in self.reasons.iter().enumerate() {
458            if index > 0 {
459                f.write_str("; ")?;
460            }
461            write!(f, "{reason}")?;
462        }
463        Ok(())
464    }
465}
466
467/// Why an item could not be paired between two runs.
468#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
469#[serde(rename_all = "snake_case")]
470#[non_exhaustive]
471pub enum ExclusionReason {
472    /// Authored parts of the item changed, so the two runs measured two
473    /// different scenarios under one identifier. Nothing about this item is
474    /// compared.
475    PairingBroken {
476        /// The parts a person wrote that are no longer the same.
477        parts: Vec<ItemPart>,
478    },
479    /// Parts declared `recorded` changed.
480    ///
481    /// Testimony about what the system actually emitted does not change on its
482    /// own, so something regenerated it — and regenerating a recording destroys
483    /// the only property that made it worth keeping. This is a defect in the
484    /// corpus or in the tooling that touched it, and it is deliberately *not*
485    /// a result about the model: the item is excluded from every figure and
486    /// reported on its own terms.
487    RecordedPartChanged {
488        /// The parts that were supposed to be evidence.
489        parts: Vec<ItemPart>,
490    },
491}
492
493impl std::fmt::Display for ExclusionReason {
494    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
495        match self {
496            Self::PairingBroken { parts } => write!(
497                f,
498                "pairing broken — authored `{}` changed, so the two runs did not measure the same \
499                 scenario",
500                joined(parts)
501            ),
502            Self::RecordedPartChanged { parts } => write!(
503                f,
504                "CORPUS DEFECT — recorded `{}` changed; testimony must never be regenerated, so \
505                 fix the corpus or the tooling rather than reading this as a result",
506                joined(parts)
507            ),
508        }
509    }
510}
511
512fn joined(parts: &[ItemPart]) -> String {
513    parts
514        .iter()
515        .map(|part| part.as_str())
516        .collect::<Vec<&str>>()
517        .join("`, `")
518}
519
520/// One item's change.
521#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
522#[serde(deny_unknown_fields)]
523pub struct Change {
524    /// Which item.
525    pub item: ItemId,
526    /// What changed about it.
527    pub kind: ChangeKind,
528    /// How the movement sits against a measured noise floor.
529    #[serde(default)]
530    pub against_noise: NoiseVerdict,
531}
532
533impl std::fmt::Display for Change {
534    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
535        write!(f, "{}: {}", self.item, self.kind)?;
536        match self.against_noise {
537            NoiseVerdict::Unmeasured => Ok(()),
538            NoiseVerdict::WithinNoise => f.write_str(" [within the measured noise floor]"),
539            NoiseVerdict::ExceedsNoise => f.write_str(" [exceeds the measured noise floor]"),
540        }
541    }
542}
543
544/// How a movement compares with what the unchanged system produced against
545/// itself.
546#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Hash, Serialize, Deserialize)]
547#[serde(rename_all = "snake_case")]
548#[non_exhaustive]
549pub enum NoiseVerdict {
550    /// No control run was supplied, so nothing can be said. The default, because
551    /// assuming a difference is noise without measuring one is the habit this
552    /// vocabulary exists to break.
553    #[default]
554    Unmeasured,
555    /// No larger than the movement the same code produced against itself.
556    WithinNoise,
557    /// Larger than the movement the same code produced against itself.
558    ExceedsNoise,
559}
560
561/// The kinds of change a comparison reports.
562#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
563#[serde(tag = "kind", rename_all = "snake_case")]
564#[non_exhaustive]
565pub enum ChangeKind {
566    /// The item is new in the current run.
567    Added {
568        /// Its deterministic pass rate.
569        pass_rate: f64,
570    },
571    /// The item was in the baseline and is not in the current run.
572    Removed {
573        /// Its deterministic pass rate in the baseline.
574        pass_rate: f64,
575    },
576    /// The item's content changed, in parts the corpus declares are written by
577    /// the code under test.
578    ///
579    /// This is the intended effect of the change being measured, not a broken
580    /// pairing: the item is still compared, and whatever deterministic or judge
581    /// change it also produced is reported beside this one. It is the difference
582    /// between "the state block changed because we changed the projector" and
583    /// "somebody edited the fixture".
584    DerivedInput {
585        /// The declared parts whose content changed.
586        parts: Vec<ItemPart>,
587    },
588    /// Deterministic assertions that used to hold no longer do. Behaviour
589    /// changed.
590    DeterministicRegression {
591        /// Pass rate before.
592        before: f64,
593        /// Pass rate now.
594        after: f64,
595        /// Expectations that fail now and did not before.
596        newly_failing: Vec<ExpectationName>,
597        /// The reliability categories those expectations belong to (§26.3).
598        categories: Vec<ReliabilityCategory>,
599    },
600    /// Deterministic assertions that used to fail now hold.
601    DeterministicImprovement {
602        /// Pass rate before.
603        before: f64,
604        /// Pass rate now.
605        after: f64,
606    },
607    /// The same behaviour, graded differently. Not a regression.
608    JudgeDrift {
609        /// Which criterion moved.
610        criterion: JudgeCriterion,
611        /// Mean score before.
612        before: f64,
613        /// Mean score now.
614        after: f64,
615    },
616}
617
618impl std::fmt::Display for ChangeKind {
619    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
620        match self {
621            Self::Added { pass_rate } => write!(f, "added, pass rate {pass_rate:.2}"),
622            Self::Removed { pass_rate } => write!(f, "removed, was {pass_rate:.2}"),
623            Self::DerivedInput { parts } => {
624                let names: Vec<&str> = parts.iter().map(|part| part.as_str()).collect();
625                write!(
626                    f,
627                    "derived input changed as intended (`{}`)",
628                    names.join("`, `")
629                )
630            }
631            Self::DeterministicRegression {
632                before,
633                after,
634                newly_failing,
635                ..
636            } => {
637                let names: Vec<&str> = newly_failing
638                    .iter()
639                    .map(|expectation| expectation.as_str())
640                    .collect();
641                write!(
642                    f,
643                    "DETERMINISTIC REGRESSION {before:.2} → {after:.2} ({})",
644                    names.join(", ")
645                )
646            }
647            Self::DeterministicImprovement { before, after } => {
648                write!(f, "improved {before:.2} → {after:.2}")
649            }
650            Self::JudgeDrift {
651                criterion,
652                before,
653                after,
654            } => write!(f, "judge drift on {criterion}: {before:.2} → {after:.2}"),
655        }
656    }
657}
658
659/// Compares a current run against a baseline.
660///
661/// Items are joined by identifier, and then by fingerprint: an item whose
662/// undeclared content changed between the two runs is **excluded**, not
663/// compared, because the two runs did not measure the same scenario. An item
664/// whose *declared derived* content changed is compared and reports a
665/// [`ChangeKind::DerivedInput`] beside whatever else it produced.
666///
667/// An item that changed both its behaviour and its judge score produces two
668/// changes, one of each kind, because they are two different pieces of news.
669///
670/// The headline is withheld — no pass-rate figure at all — when the excluded
671/// share crosses [`ComparisonPolicy::max_excluded_share`], or when nothing is
672/// left to compare.
673#[must_use]
674pub fn compare(
675    baseline: &EvalReport,
676    current: &EvalReport,
677    policy: &ComparisonPolicy,
678) -> Comparison {
679    let mut changes = Vec::new();
680    let mut excluded = Vec::new();
681    let mut compared: Vec<ItemId> = Vec::new();
682    let mut unverified = 0_usize;
683
684    for item in &current.items {
685        let Some(before) = baseline.item(&item.id) else {
686            changes.push(change(
687                item.id.clone(),
688                ChangeKind::Added {
689                    pass_rate: item.deterministic_pass_rate(),
690                },
691                policy,
692            ));
693            continue;
694        };
695        if before.fingerprint.is_unknown() || item.fingerprint.is_unknown() {
696            unverified += 1;
697        }
698
699        let sorted = sort_by_provenance(
700            &before.fingerprint.differing_parts(&item.fingerprint),
701            before,
702            item,
703        );
704        let mut reasons = Vec::new();
705        if !sorted.recorded.is_empty() {
706            reasons.push(ExclusionReason::RecordedPartChanged {
707                parts: sorted.recorded,
708            });
709        }
710        if !sorted.authored.is_empty() {
711            reasons.push(ExclusionReason::PairingBroken {
712                parts: sorted.authored,
713            });
714        }
715        if !reasons.is_empty() {
716            excluded.push(ExcludedItem {
717                item: item.id.clone(),
718                reasons,
719            });
720            continue;
721        }
722
723        compared.push(item.id.clone());
724        if !sorted.derived.is_empty() {
725            changes.push(change(
726                item.id.clone(),
727                ChangeKind::DerivedInput {
728                    parts: sorted.derived,
729                },
730                policy,
731            ));
732        }
733        compare_deterministic(before, item, policy, &mut changes);
734        compare_judge(before, item, policy, &mut changes);
735    }
736
737    for item in &baseline.items {
738        if current.item(&item.id).is_none() {
739            changes.push(change(
740                item.id.clone(),
741                ChangeKind::Removed {
742                    pass_rate: item.deterministic_pass_rate(),
743                },
744                policy,
745            ));
746        }
747    }
748
749    Comparison {
750        baseline_suite: baseline.suite.clone(),
751        current_suite: current.suite.clone(),
752        headline: headline(baseline, current, &compared, &excluded, unverified, policy),
753        excluded,
754        changes,
755    }
756}
757
758/// The changed parts of one item, split by what the two corpora said they were.
759struct SortedParts {
760    /// Declared `recorded` on at least one side: evidence that moved.
761    recorded: Vec<ItemPart>,
762    /// Declared `derived` on both sides: the intended effect of the change.
763    derived: Vec<ItemPart>,
764    /// Everything else, which is a person's fixture that no longer matches.
765    authored: Vec<ItemPart>,
766}
767
768/// Splits the parts that differ by the provenance the two runs recorded.
769///
770/// `recorded` wins over everything, and takes the vote of **either** side: a
771/// part one corpus calls testimony is testimony, and the reading that raises a
772/// defect is the one worth being wrong about. `derived` needs **both** sides,
773/// so adding the declaration cannot retroactively excuse a difference against a
774/// baseline that never knew about it. Everything left over is authored.
775fn sort_by_provenance(parts: &[ItemPart], before: &ItemReport, after: &ItemReport) -> SortedParts {
776    let mut sorted = SortedParts {
777        recorded: Vec::new(),
778        derived: Vec::new(),
779        authored: Vec::new(),
780    };
781    for part in parts {
782        let was = before.fingerprint.provenance_of(*part);
783        let now = after.fingerprint.provenance_of(*part);
784        if was == PartProvenance::Recorded || now == PartProvenance::Recorded {
785            sorted.recorded.push(*part);
786        } else if was == PartProvenance::Derived && now == PartProvenance::Derived {
787            sorted.derived.push(*part);
788        } else {
789            sorted.authored.push(*part);
790        }
791    }
792    sorted
793}
794
795fn headline(
796    baseline: &EvalReport,
797    current: &EvalReport,
798    compared: &[ItemId],
799    excluded: &[ExcludedItem],
800    unverified: usize,
801    policy: &ComparisonPolicy,
802) -> Headline {
803    let paired = compared.len() + excluded.len();
804    let share = ratio(excluded.len(), paired);
805    if compared.is_empty() {
806        return Headline::Withheld(WithheldHeadline {
807            excluded_share: share,
808            max_excluded_share: policy.max_excluded_share,
809            items_compared: 0,
810            items_excluded: excluded.len(),
811            unverified_pairings: unverified,
812            reason: if paired == 0 {
813                "no item appeared in both runs, so there was nothing to compare".to_owned()
814            } else {
815                format!(
816                    "all {} paired item(s) were excluded, so there was nothing left to compare",
817                    excluded.len()
818                )
819            },
820        });
821    }
822    if share > policy.max_excluded_share {
823        return Headline::Withheld(WithheldHeadline {
824            excluded_share: share,
825            max_excluded_share: policy.max_excluded_share,
826            items_compared: compared.len(),
827            items_excluded: excluded.len(),
828            unverified_pairings: unverified,
829            reason: format!(
830                "{:.0}% of the paired corpus was excluded, above the {:.0}% ceiling; a figure over \
831                 the remaining {} of {} item(s) would read like a measurement of the suite and \
832                 would not be one",
833                share * 100.0,
834                policy.max_excluded_share * 100.0,
835                compared.len(),
836                paired
837            ),
838        });
839    }
840    let before = pass_rate_over(baseline, compared);
841    let after = pass_rate_over(current, compared);
842    Headline::Measured(HeadlineFigures {
843        baseline_pass_rate: before,
844        current_pass_rate: after,
845        pass_rate_delta: after - before,
846        items_compared: compared.len(),
847        items_excluded: excluded.len(),
848        excluded_share: share,
849        unverified_pairings: unverified,
850    })
851}
852
853/// The deterministic pass rate of a report over a subset of its items.
854fn pass_rate_over(report: &EvalReport, items: &[ItemId]) -> f64 {
855    let (passed, total) = report
856        .items
857        .iter()
858        .filter(|item| items.contains(&item.id))
859        .fold((0_usize, 0_usize), |(passed, total), item| {
860            (passed + item.samples_passed(), total + item.total_samples())
861        });
862    ratio(passed, total)
863}
864
865/// Builds a change, labelling it against the noise floor when there is one.
866fn change(item: ItemId, kind: ChangeKind, policy: &ComparisonPolicy) -> Change {
867    let against_noise = policy
868        .noise_floor
869        .as_ref()
870        .map_or(NoiseVerdict::Unmeasured, |floor| match &kind {
871            ChangeKind::DeterministicRegression { before, after, .. }
872            | ChangeKind::DeterministicImprovement { before, after } => {
873                if floor.covers_pass_rate(after - before) {
874                    NoiseVerdict::WithinNoise
875                } else {
876                    NoiseVerdict::ExceedsNoise
877                }
878            }
879            ChangeKind::JudgeDrift { before, after, .. } => {
880                if floor.covers_judge_score(after - before) {
881                    NoiseVerdict::WithinNoise
882                } else {
883                    NoiseVerdict::ExceedsNoise
884                }
885            }
886            // An item that appeared, vanished or had its inputs rewritten has no
887            // movement to weigh: it is not a smaller or larger difference than
888            // the harness's own, it is a different kind of fact.
889            _ => NoiseVerdict::Unmeasured,
890        });
891    Change {
892        item,
893        kind,
894        against_noise,
895    }
896}
897
898fn compare_deterministic(
899    before: &ItemReport,
900    after: &ItemReport,
901    policy: &ComparisonPolicy,
902    changes: &mut Vec<Change>,
903) {
904    let previous = before.deterministic_pass_rate();
905    let now = after.deterministic_pass_rate();
906    let delta = now - previous;
907    if delta.abs() <= policy.tolerance.pass_rate_epsilon {
908        return;
909    }
910    if delta > 0.0 {
911        changes.push(change(
912            after.id.clone(),
913            ChangeKind::DeterministicImprovement {
914                before: previous,
915                after: now,
916            },
917            policy,
918        ));
919        return;
920    }
921    let newly_failing = newly_failing(before, after);
922    let mut categories: Vec<ReliabilityCategory> = newly_failing
923        .iter()
924        .map(|expectation| ReliabilityCategory::of(*expectation))
925        .collect();
926    categories.sort_unstable();
927    categories.dedup();
928    changes.push(change(
929        after.id.clone(),
930        ChangeKind::DeterministicRegression {
931            before: previous,
932            after: now,
933            newly_failing,
934            categories,
935        },
936        policy,
937    ));
938}
939
940fn newly_failing(before: &ItemReport, after: &ItemReport) -> Vec<ExpectationName> {
941    let was: Vec<ExpectationName> = before
942        .failures()
943        .into_iter()
944        .map(|failure| failure.expectation)
945        .collect();
946    let mut now: Vec<ExpectationName> = after
947        .failures()
948        .into_iter()
949        .map(|failure| failure.expectation)
950        .filter(|expectation| !was.contains(expectation))
951        .collect();
952    now.sort_unstable();
953    now.dedup();
954    now
955}
956
957fn compare_judge(
958    before: &ItemReport,
959    after: &ItemReport,
960    policy: &ComparisonPolicy,
961    changes: &mut Vec<Change>,
962) {
963    for current in after.judge_summaries() {
964        let Some(previous) = before
965            .judge_summaries()
966            .into_iter()
967            .find(|summary| summary.criterion == current.criterion)
968        else {
969            continue;
970        };
971        let (Some(was), Some(now)) = (previous.mean_score, current.mean_score) else {
972            continue;
973        };
974        if (now - was).abs() > policy.tolerance.judge_score_epsilon {
975            changes.push(change(
976                after.id.clone(),
977                ChangeKind::JudgeDrift {
978                    criterion: current.criterion,
979                    before: was,
980                    after: now,
981                },
982                policy,
983            ));
984        }
985    }
986}
987
988#[cfg(test)]
989mod tests {
990    use chrono::DateTime;
991
992    use super::*;
993    use crate::assertions::AssertionFailure;
994    use crate::config::EvalConfig;
995    use crate::corpus::{ItemFingerprint, PartDigest};
996    use crate::judge::{CriterionOutcome, JudgeVerdict, JudgeVote};
997    use crate::report::SampleReport;
998
999    fn sample(failing: bool, score: Option<u8>) -> SampleReport {
1000        SampleReport {
1001            sample: 1,
1002            failures: if failing {
1003                vec![AssertionFailure::new(
1004                    ExpectationName::Commands,
1005                    "[trip.set_name]",
1006                    "[]",
1007                )]
1008            } else {
1009                Vec::new()
1010            },
1011            harness_error: None,
1012            signature: "sig".to_owned(),
1013            judge: score
1014                .map(|score| {
1015                    vec![CriterionOutcome {
1016                        criterion: JudgeCriterion::Tone,
1017                        votes: vec![JudgeVote {
1018                            vote: 1,
1019                            verdict: Some(JudgeVerdict {
1020                                score,
1021                                reason: "r".to_owned(),
1022                            }),
1023                            error: None,
1024                        }],
1025                    }]
1026                })
1027                .unwrap_or_default(),
1028            acts_proposed: 0,
1029            acts_refused: 0,
1030            commands_journaled: 0,
1031            provider_failures: 0,
1032            cards_created: 0,
1033            abandoned: false,
1034            discarded_answers: Vec::new(),
1035            answer: String::new(),
1036            tasks: Default::default(),
1037        }
1038    }
1039
1040    /// A fingerprint whose `setup` digest is `state`, with that provenance.
1041    fn fingerprint(state: &str, setup: PartProvenance) -> ItemFingerprint {
1042        ItemFingerprint {
1043            parts: ItemPart::ALL
1044                .into_iter()
1045                .map(|part| PartDigest {
1046                    part,
1047                    digest: if part == ItemPart::Setup {
1048                        state.to_owned()
1049                    } else {
1050                        "same".to_owned()
1051                    },
1052                    provenance: if part == ItemPart::Setup {
1053                        setup
1054                    } else {
1055                        PartProvenance::Authored
1056                    },
1057                })
1058                .collect(),
1059        }
1060    }
1061
1062    fn item(id: &str, sample: SampleReport, fingerprint: ItemFingerprint) -> ItemReport {
1063        ItemReport {
1064            id: ItemId::new(id),
1065            name: "An item".to_owned(),
1066            tags: Vec::new(),
1067            fingerprint,
1068            samples: vec![sample],
1069        }
1070    }
1071
1072    fn report(items: Vec<ItemReport>) -> EvalReport {
1073        EvalReport::new(
1074            "suite",
1075            DateTime::from_timestamp(0, 0).unwrap_or_default(),
1076            EvalConfig::default(),
1077            items,
1078        )
1079    }
1080
1081    fn one(sample: SampleReport) -> EvalReport {
1082        report(vec![item(
1083            "i",
1084            sample,
1085            fingerprint("state", PartProvenance::Authored),
1086        )])
1087    }
1088
1089    #[test]
1090    fn a_broken_assertion_is_a_regression_not_a_drift() {
1091        let comparison = compare(
1092            &one(sample(false, Some(5))),
1093            &one(sample(true, Some(5))),
1094            &ComparisonPolicy::default(),
1095        );
1096        assert!(comparison.has_deterministic_regression());
1097        assert!(!comparison.has_judge_drift());
1098        let regression = comparison.deterministic_regressions()[0];
1099        let ChangeKind::DeterministicRegression {
1100            newly_failing,
1101            categories,
1102            ..
1103        } = &regression.kind
1104        else {
1105            panic!("expected a regression, got {:?}", regression.kind);
1106        };
1107        assert_eq!(newly_failing, &[ExpectationName::Commands]);
1108        assert_eq!(categories, &[ReliabilityCategory::SideEffectIntegrity]);
1109    }
1110
1111    #[test]
1112    fn a_moved_score_on_unchanged_behaviour_is_a_drift_not_a_regression() {
1113        let comparison = compare(
1114            &one(sample(false, Some(5))),
1115            &one(sample(false, Some(3))),
1116            &ComparisonPolicy::default(),
1117        );
1118        assert!(!comparison.has_deterministic_regression());
1119        assert!(comparison.has_judge_drift());
1120        assert!(comparison.summary().contains("judge drift"));
1121    }
1122
1123    #[test]
1124    fn a_score_inside_the_tolerance_is_not_news() {
1125        let comparison = compare(
1126            &one(sample(false, Some(5))),
1127            &one(sample(false, Some(5))),
1128            &ComparisonPolicy::default(),
1129        );
1130        assert!(comparison.is_unchanged());
1131    }
1132
1133    #[test]
1134    fn the_headline_reports_the_excluded_share_even_when_it_stands() {
1135        let comparison = compare(
1136            &one(sample(false, None)),
1137            &one(sample(false, None)),
1138            &ComparisonPolicy::default(),
1139        );
1140        let figures = comparison
1141            .headline
1142            .figures()
1143            .expect("nothing was excluded, so the headline stands");
1144        assert_eq!(figures.items_compared, 1);
1145        assert_eq!(figures.items_excluded, 0);
1146        assert!((figures.excluded_share - 0.0).abs() < 1e-9);
1147        assert!(comparison.summary().contains("0 excluded"));
1148    }
1149
1150    /// Two reports of one item whose `setup` changed, with that provenance on
1151    /// each side.
1152    fn pair(before: PartProvenance, after: PartProvenance) -> (EvalReport, EvalReport) {
1153        (
1154            report(vec![item(
1155                "i",
1156                sample(false, None),
1157                fingerprint("before", before),
1158            )]),
1159            report(vec![item(
1160                "i",
1161                sample(true, None),
1162                fingerprint("after", after),
1163            )]),
1164        )
1165    }
1166
1167    #[test]
1168    fn an_authored_item_change_is_excluded_and_produces_no_regression() {
1169        // The item's setup changed and a person wrote that setup. Under the old
1170        // rule this would have compared two different scenarios and reported the
1171        // difference as a regression.
1172        let (baseline, current) = pair(PartProvenance::Authored, PartProvenance::Authored);
1173        let comparison = compare(&baseline, &current, &ComparisonPolicy::default());
1174
1175        assert!(!comparison.has_deterministic_regression());
1176        assert!(!comparison.has_corpus_defect());
1177        assert_eq!(comparison.excluded.len(), 1);
1178        assert!(matches!(
1179            comparison.excluded[0].reasons.as_slice(),
1180            [ExclusionReason::PairingBroken { parts }] if parts == &[ItemPart::Setup]
1181        ));
1182    }
1183
1184    #[test]
1185    fn a_declared_derived_change_stays_compared_and_is_named_as_intended() {
1186        let (baseline, current) = pair(PartProvenance::Derived, PartProvenance::Derived);
1187        let comparison = compare(&baseline, &current, &ComparisonPolicy::default());
1188
1189        assert!(comparison.excluded.is_empty());
1190        assert_eq!(comparison.derived_input_changes().len(), 1);
1191        assert!(comparison.has_deterministic_regression());
1192        assert!(
1193            comparison.summary().contains("derived input changed"),
1194            "{}",
1195            comparison.summary()
1196        );
1197    }
1198
1199    #[test]
1200    fn a_recorded_change_is_a_corpus_defect_not_a_result() {
1201        let (baseline, current) = pair(PartProvenance::Recorded, PartProvenance::Recorded);
1202        let comparison = compare(&baseline, &current, &ComparisonPolicy::default());
1203
1204        assert!(comparison.has_corpus_defect());
1205        assert_eq!(comparison.corpus_defects().len(), 1);
1206        // Never folded into a figure, and never a regression.
1207        assert!(!comparison.has_deterministic_regression());
1208        assert!(comparison.derived_input_changes().is_empty());
1209        assert!(matches!(
1210            comparison.excluded[0].reasons.as_slice(),
1211            [ExclusionReason::RecordedPartChanged { parts }] if parts == &[ItemPart::Setup]
1212        ));
1213        assert!(
1214            comparison.summary().contains("CORPUS DEFECT"),
1215            "{}",
1216            comparison.summary()
1217        );
1218    }
1219
1220    #[test]
1221    fn one_side_calling_a_part_recorded_is_enough_to_raise_a_defect() {
1222        // Whichever run is right, something rewrote a recording, and the reading
1223        // that says so is the one worth being wrong about.
1224        let (baseline, current) = pair(PartProvenance::Derived, PartProvenance::Recorded);
1225        assert!(compare(&baseline, &current, &ComparisonPolicy::default()).has_corpus_defect());
1226    }
1227
1228    #[test]
1229    fn a_derived_declaration_on_one_side_only_does_not_rescue_a_pairing() {
1230        let (baseline, current) = pair(PartProvenance::Authored, PartProvenance::Derived);
1231        let comparison = compare(&baseline, &current, &ComparisonPolicy::default());
1232        assert_eq!(comparison.excluded.len(), 1);
1233        assert!(!comparison.has_corpus_defect());
1234    }
1235
1236    #[test]
1237    fn an_unknown_fingerprint_pairs_by_identifier_and_is_counted() {
1238        let baseline = one(sample(false, None));
1239        let current = report(vec![item(
1240            "i",
1241            sample(true, None),
1242            ItemFingerprint::default(),
1243        )]);
1244        let comparison = compare(&baseline, &current, &ComparisonPolicy::default());
1245        assert!(comparison.has_deterministic_regression());
1246        let figures = comparison.headline.figures().expect("still comparable");
1247        assert_eq!(figures.unverified_pairings, 1);
1248        assert!(
1249            comparison.summary().contains("pairing was not checked"),
1250            "{}",
1251            comparison.summary()
1252        );
1253    }
1254}