Skip to main content

turnframe_eval/
report.rs

1//! What a run produced: per item, per suite, and in a form a build can gate on
2//! (spec §26.3, §27.6).
3//!
4//! # One number would be a lie
5//!
6//! §26.3 is blunt about it: "a single 'agent accuracy' percentage hides the
7//! most important distinctions". A corpus in which the assistant sent a
8//! rebooking it should not have, but phrased three replies beautifully, can
9//! average to a very healthy figure. So the report keeps the dashboard's
10//! categories apart — side-effect integrity, operational claim integrity,
11//! semantic interpretation, clarification, abandonment, provider failure and
12//! the judge's user-experience scores — and refuses to combine them into one.
13//!
14//! # Three numbers, and each one sees what the others cannot
15//!
16//! The pass rate is about **effects**: what the turn did and did not do. It is
17//! the number a release blocks on, and on its own it flatters the design,
18//! because every reading the runtime refused to carry out reads as a clean
19//! turn. [`ItemReport::refused_proposals`] is the correction: the model
20//! proposed an act and nothing was journaled, which is a fact about the model's
21//! reading rather than about the effect.
22//!
23//! Neither of them can see a turn whose answer never became a plan at all.
24//! [`ItemReport::samples_with_discards`] is that one: the runtime read what the
25//! model produced and threw it away whole — an invented citation, a question
26//! quoting nothing, a document of the wrong shape — and asked again. It is
27//! usually invisible, since the repair round recovers and the effects come out
28//! right, and when it is not invisible the turn simply has no effects, which
29//! looks exactly like a turn that correctly had nothing to do.
30//!
31//! # Samples and votes stay apart too
32//!
33//! [`ItemReport::deterministic_pass_rate`] counts **samples**, never votes. A
34//! judge score never enters it. [`CriterionSummary`] carries the judge's
35//! numbers with their vote spread, so a reader can see whether a low score is
36//! the model's fault or the judge's disagreement with itself.
37
38use chrono::{DateTime, Utc};
39use serde::{Deserialize, Serialize};
40
41use crate::assertions::{AssertionFailure, ExpectationName};
42use crate::config::EvalConfig;
43use crate::corpus::{ItemFingerprint, ItemId, Tag};
44use crate::judge::{CriterionOutcome, JudgeCriterion, ratio};
45
46/// The reliability categories of the specification's dashboard (§26.3).
47///
48/// A failure belongs to exactly one, derived from the expectation it broke, so
49/// a forbidden command that fired is never averaged into anything.
50#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)]
51#[serde(rename_all = "snake_case")]
52#[non_exhaustive]
53pub enum ReliabilityCategory {
54    /// An effect happened that should not have, or did not happen that should
55    /// have. The category a release blocks on.
56    SideEffectIntegrity,
57    /// What the turn said about itself did not match what it did.
58    OperationalClaimIntegrity,
59    /// The model's reading of the message was wrong, before any effect.
60    SemanticInterpretation,
61    /// The turn asked, or failed to ask, for a clarification or a confirmation.
62    Clarification,
63    /// The turn produced nothing usable at all.
64    Abandonment,
65    /// The provider layer failed.
66    ProviderFailure,
67    /// What a judge thought of the wording.
68    UserExperience,
69}
70
71impl ReliabilityCategory {
72    /// The category a broken expectation belongs to.
73    #[must_use]
74    pub const fn of(expectation: ExpectationName) -> Self {
75        match expectation {
76            ExpectationName::Commands
77            | ExpectationName::ForbiddenCommand
78            | ExpectationName::Events
79            | ExpectationName::ForbiddenEvent
80            | ExpectationName::TruncatedLedger
81            | ExpectationName::CaseRevision
82            // A value that ended up wrong, or moved when it should not have, is
83            // a side effect like any other: something happened to a record. It
84            // is not a claim about what the assistant SAID, which is the other
85            // category and a different kind of harm.
86            | ExpectationName::CaseState
87            | ExpectationName::WorkflowState
88            | ExpectationName::CaseCount => Self::SideEffectIntegrity,
89            ExpectationName::Outcome
90            | ExpectationName::ResponseBlocks
91            | ExpectationName::TurnPhase => Self::OperationalClaimIntegrity,
92            ExpectationName::Acts | ExpectationName::TargetResolution => {
93                Self::SemanticInterpretation
94            }
95            ExpectationName::InteractionStatus => Self::Clarification,
96        }
97    }
98
99    /// The snake-case label.
100    #[must_use]
101    pub const fn as_str(self) -> &'static str {
102        match self {
103            Self::SideEffectIntegrity => "side_effect_integrity",
104            Self::OperationalClaimIntegrity => "operational_claim_integrity",
105            Self::SemanticInterpretation => "semantic_interpretation",
106            Self::Clarification => "clarification",
107            Self::Abandonment => "abandonment",
108            Self::ProviderFailure => "provider_failure",
109            Self::UserExperience => "user_experience",
110        }
111    }
112}
113
114impl std::fmt::Display for ReliabilityCategory {
115    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
116        f.write_str(self.as_str())
117    }
118}
119
120/// One execution of one item.
121#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
122#[serde(deny_unknown_fields)]
123pub struct SampleReport {
124    /// 1-based position in the item's samples.
125    pub sample: u32,
126    /// Deterministic expectations that did not hold.
127    #[serde(default)]
128    pub failures: Vec<AssertionFailure>,
129    /// The harness could not even set the sample up. Distinct from a failing
130    /// assertion: nothing was measured.
131    #[serde(default, skip_serializing_if = "Option::is_none")]
132    pub harness_error: Option<String>,
133    /// Canonical rendering of what this run did, for counting distinct
134    /// behaviours across samples.
135    pub signature: String,
136    /// Judge opinions about this sample, when the item asked for any.
137    #[serde(default)]
138    pub judge: Vec<CriterionOutcome>,
139    /// How many provider attempts failed or fell back.
140    #[serde(default)]
141    pub provider_failures: usize,
142    /// How many cards the turn created.
143    #[serde(default)]
144    pub cards_created: usize,
145    /// How many acts the message was understood to ask for.
146    ///
147    /// The pair below is the whole point of recording it: a turn where the model
148    /// proposed something and nothing was journaled is a turn the runtime
149    /// REFUSED, and that is a fact about the model's reading rather than about
150    /// the effect. A report that shows only the effect says an assistant is
151    /// perfect exactly where its reading is worst — every refusal reads as a
152    /// clean turn — which flatters a design whose whole claim is that it makes
153    /// bad readings harmless.
154    #[serde(default)]
155    pub acts_proposed: usize,
156    /// How many of them the reduction refused outright.
157    ///
158    /// Only `rejected`: an act the domain turned down. Not the one that
159    /// changes nothing, not the one awaiting a confirmation, not the one a
160    /// later act superseded — see
161    /// [`refused_proposals`](ItemReport::refused_proposals).
162    #[serde(default)]
163    pub acts_refused: usize,
164    /// How many commands the turn journaled.
165    ///
166    /// Not comparable one-to-one with [`Self::acts_proposed`]: one act can
167    /// produce several commands and a command type is not an operation name. It
168    /// is here to be read against zero — nothing journaled after something was
169    /// proposed — and not as a ratio of the two.
170    #[serde(default)]
171    pub commands_journaled: usize,
172    /// The turn produced no usable answer at all.
173    #[serde(default)]
174    pub abandoned: bool,
175    /// The stable code of every answer the runtime threw away whole this turn,
176    /// in order and with repeats.
177    ///
178    /// The third number of a reliability report, and the one neither of the
179    /// other two can reach. The pass rate is about effects and the refused
180    /// proposals are about a plan the runtime declined to carry out; this is
181    /// about a plan that never became one, because the runtime read the model's
182    /// answer and refused it whole — a citation the user's message does not
183    /// contain, a question that does not quote what it answers, a document that
184    /// does not match its schema.
185    ///
186    /// Codes rather than the full records, because a report is read in
187    /// aggregate and the runtime's own wording names positions inside one
188    /// turn's plan. The full records, reasons included, are on
189    /// [`Observation::discarded_answers`](crate::observation::Observation::discarded_answers).
190    ///
191    /// [`Observation`]: crate::observation::Observation
192    #[serde(default, skip_serializing_if = "Vec::is_empty")]
193    pub discarded_answers: Vec<String>,
194    /// What the turn actually said, when the run was configured to keep it.
195    ///
196    /// Empty unless
197    /// [`ExecutionConfig::record_answers`](crate::config::ExecutionConfig::record_answers)
198    /// is on — see there for why that is the default. It is here for a person
199    /// curating a corpus, never for an assertion: nothing in this crate reads
200    /// it, and an expectation that did would be measuring prose.
201    #[serde(default, skip_serializing_if = "String::is_empty")]
202    pub answer: String,
203    /// How each understanding task did, where the item says what it expects of it.
204    #[serde(default)]
205    pub tasks: crate::understanding::TaskScores,
206}
207
208impl SampleReport {
209    /// Returns `true` when every deterministic expectation held.
210    #[must_use]
211    pub fn passed(&self) -> bool {
212        self.failures.is_empty() && self.harness_error.is_none()
213    }
214}
215
216/// How much the agent under test varied across the samples of one item.
217///
218/// This is the number `samples_per_item` exists to produce. It is about the
219/// **model**; no judge vote contributes to it.
220#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
221#[serde(deny_unknown_fields)]
222pub struct Variance {
223    /// How many samples ran.
224    pub samples: usize,
225    /// How many distinct behaviours were observed. One means the model was
226    /// perfectly repeatable — whether or not it was right.
227    pub distinct_behaviours: usize,
228    /// Fraction of samples that satisfied every deterministic expectation.
229    pub pass_rate: f64,
230    /// Variance of the pass indicator, `p * (1 - p)`. Zero when every sample
231    /// agreed; at its maximum when half of them did.
232    pub pass_variance: f64,
233    /// How often each distinct behaviour occurred, most frequent first.
234    pub behaviours: Vec<BehaviourCount>,
235}
236
237impl Variance {
238    /// Returns `true` when the samples disagreed with each other.
239    #[must_use]
240    pub const fn is_flaky(&self) -> bool {
241        self.distinct_behaviours > 1
242    }
243}
244
245/// One distinct behaviour and how often it happened.
246#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
247#[serde(deny_unknown_fields)]
248pub struct BehaviourCount {
249    /// The canonical signature of the behaviour.
250    pub signature: String,
251    /// How many samples produced it.
252    pub count: usize,
253    /// Whether samples with this behaviour passed.
254    pub passed: bool,
255}
256
257/// The judge's numbers for one criterion of one item.
258#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
259#[serde(deny_unknown_fields)]
260pub struct CriterionSummary {
261    /// What was graded.
262    pub criterion: JudgeCriterion,
263    /// How many samples were judged.
264    pub samples_judged: usize,
265    /// Mean of the per-sample majority scores.
266    #[serde(default, skip_serializing_if = "Option::is_none")]
267    pub mean_score: Option<f64>,
268    /// Lowest per-sample majority score.
269    #[serde(default, skip_serializing_if = "Option::is_none")]
270    pub min_score: Option<u8>,
271    /// Highest per-sample majority score.
272    #[serde(default, skip_serializing_if = "Option::is_none")]
273    pub max_score: Option<u8>,
274    /// Mean spread between the votes *within* a sample: the judge's own
275    /// variance, which is what `votes_per_sample` reduces.
276    pub mean_vote_spread: f64,
277    /// How many votes returned nothing at all.
278    pub failed_votes: usize,
279}
280
281/// Everything one item produced.
282#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
283#[serde(deny_unknown_fields)]
284pub struct ItemReport {
285    /// The item.
286    pub id: ItemId,
287    /// Its name.
288    pub name: String,
289    /// Its tags.
290    #[serde(default)]
291    pub tags: Vec<Tag>,
292    /// What the item contained when this run measured it, part by part, with
293    /// the corpus's `derived` declaration recorded alongside.
294    ///
295    /// [`crate::baseline::compare`] reads it to answer "is this still the same
296    /// experiment?". A report archived before fingerprints existed carries
297    /// [`ItemFingerprint::is_unknown`], and a comparison against it counts the
298    /// item as an unverified pairing rather than pretending it checked one.
299    #[serde(default)]
300    pub fingerprint: ItemFingerprint,
301    /// Every sample, in order.
302    pub samples: Vec<SampleReport>,
303}
304
305impl ItemReport {
306    /// How many samples ran.
307    #[must_use]
308    pub fn total_samples(&self) -> usize {
309        self.samples.len()
310    }
311
312    /// Samples where the model proposed something and the runtime journaled
313    /// nothing.
314    ///
315    /// The second of the three numbers described at the top of this module,
316    /// and the one a corpus of
317    /// forbidden effects cannot produce on its own. A refused proposal is not a
318    /// failure — the item may well pass, and should, because nothing happened —
319    /// but it is the model reading the turn wrongly, and a design that claims to
320    /// make wrong readings harmless has to be able to say how often it is doing
321    /// that work.
322    ///
323    /// Samples the harness could not set up are not counted: nothing was
324    /// proposed there because nothing ran.
325    ///
326    /// Counted from the REDUCTION's verdict on each act, not from the command
327    /// count. «Proposed something and journaled nothing» reads like the same
328    /// question and is not: an act that is valid and changes nothing, and one
329    /// that is waiting for a person to confirm it, both journal zero commands
330    /// and neither was refused — so the design working reported as the model
331    /// failing. It went the other way too: a plan holding one refusal beside
332    /// one act that did write journals a command, and the refusal disappeared.
333    #[must_use]
334    pub fn refused_proposals(&self) -> usize {
335        self.samples
336            .iter()
337            .filter(|sample| sample.harness_error.is_none() && sample.acts_refused > 0)
338            .count()
339    }
340
341    /// Samples where the runtime threw away at least one model answer.
342    ///
343    /// Counted per sample and not per answer, so the number is comparable with
344    /// [`total_samples`](Self::total_samples): a turn that lost three answers
345    /// in a row is one turn that struggled, not three.
346    ///
347    /// Samples the harness could not set up are not counted, for the same
348    /// reason as in [`refused_proposals`](Self::refused_proposals): nothing
349    /// ran, so nothing was discarded.
350    #[must_use]
351    pub fn samples_with_discards(&self) -> usize {
352        self.samples
353            .iter()
354            .filter(|sample| sample.harness_error.is_none() && !sample.discarded_answers.is_empty())
355            .count()
356    }
357
358    /// How many satisfied every deterministic expectation.
359    #[must_use]
360    pub fn samples_passed(&self) -> usize {
361        self.samples.iter().filter(|s| s.passed()).count()
362    }
363
364    /// Fraction of samples that satisfied every deterministic expectation.
365    ///
366    /// Judge scores never enter this number.
367    #[must_use]
368    pub fn deterministic_pass_rate(&self) -> f64 {
369        ratio(self.samples_passed(), self.total_samples())
370    }
371
372    /// Returns `true` when some samples passed and others did not. A flaky item
373    /// is a result, not an error.
374    #[must_use]
375    pub fn is_flaky(&self) -> bool {
376        let passed = self.samples_passed();
377        passed > 0 && passed < self.total_samples()
378    }
379
380    /// How much the agent varied across samples.
381    #[must_use]
382    pub fn variance(&self) -> Variance {
383        let mut behaviours: Vec<BehaviourCount> = Vec::new();
384        for sample in &self.samples {
385            match behaviours
386                .iter_mut()
387                .find(|found| found.signature == sample.signature)
388            {
389                Some(found) => found.count += 1,
390                None => behaviours.push(BehaviourCount {
391                    signature: sample.signature.clone(),
392                    count: 1,
393                    passed: sample.passed(),
394                }),
395            }
396        }
397        behaviours.sort_by(|left, right| {
398            right
399                .count
400                .cmp(&left.count)
401                .then_with(|| left.signature.cmp(&right.signature))
402        });
403        let pass_rate = self.deterministic_pass_rate();
404        Variance {
405            samples: self.total_samples(),
406            distinct_behaviours: behaviours.len(),
407            pass_rate,
408            pass_variance: pass_rate * (1.0 - pass_rate),
409            behaviours,
410        }
411    }
412
413    /// Every deterministic failure of every sample.
414    #[must_use]
415    pub fn failures(&self) -> Vec<&AssertionFailure> {
416        self.samples
417            .iter()
418            .flat_map(|sample| sample.failures.iter())
419            .collect()
420    }
421
422    /// How many samples failed at least one expectation of `category`.
423    #[must_use]
424    pub fn samples_failing(&self, category: ReliabilityCategory) -> usize {
425        self.samples
426            .iter()
427            .filter(|sample| {
428                sample
429                    .failures
430                    .iter()
431                    .any(|failure| ReliabilityCategory::of(failure.expectation) == category)
432            })
433            .count()
434    }
435
436    /// The judge's numbers, one row per criterion that was graded.
437    #[must_use]
438    pub fn judge_summaries(&self) -> Vec<CriterionSummary> {
439        let mut summaries = Vec::new();
440        for criterion in JudgeCriterion::ALL {
441            let outcomes: Vec<&CriterionOutcome> = self
442                .samples
443                .iter()
444                .flat_map(|sample| sample.judge.iter())
445                .filter(|outcome| outcome.criterion == criterion)
446                .collect();
447            if outcomes.is_empty() {
448                continue;
449            }
450            summaries.push(summarize(criterion, &outcomes));
451        }
452        summaries
453    }
454}
455
456fn summarize(criterion: JudgeCriterion, outcomes: &[&CriterionOutcome]) -> CriterionSummary {
457    let majorities: Vec<u8> = outcomes
458        .iter()
459        .filter_map(|outcome| outcome.majority_score())
460        .collect();
461    let mean_score = if majorities.is_empty() {
462        None
463    } else {
464        let total: u32 = majorities.iter().map(|score| u32::from(*score)).sum();
465        Some(f64::from(total) / precise(majorities.len()))
466    };
467    let spread_total: u32 = outcomes
468        .iter()
469        .map(|outcome| u32::from(outcome.spread()))
470        .sum();
471    CriterionSummary {
472        criterion,
473        samples_judged: outcomes.len(),
474        mean_score,
475        min_score: majorities.iter().copied().min(),
476        max_score: majorities.iter().copied().max(),
477        mean_vote_spread: f64::from(spread_total) / precise(outcomes.len()),
478        failed_votes: outcomes.iter().map(|o| o.failed_votes()).sum(),
479    }
480}
481
482fn precise(value: usize) -> f64 {
483    if value == 0 {
484        return 1.0;
485    }
486    #[allow(clippy::cast_precision_loss)]
487    {
488        value as f64
489    }
490}
491
492/// The dashboard of §26.3, with nothing averaged across its rows.
493#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
494#[serde(deny_unknown_fields)]
495pub struct Reliability {
496    /// Samples whose effects were wrong.
497    pub side_effect_integrity: CategoryStats,
498    /// Samples whose claims about themselves were wrong.
499    pub operational_claim_integrity: CategoryStats,
500    /// Samples whose reading of the message was wrong.
501    pub semantic_interpretation: CategoryStats,
502    /// Samples whose cards were not what the corpus expected.
503    pub clarification_integrity: CategoryStats,
504    /// Fraction of samples in which the turn raised at least one card.
505    pub clarification_rate: f64,
506    /// Fraction of samples in which the turn produced nothing usable.
507    pub abandonment_rate: f64,
508    /// Fraction of samples in which a provider attempt failed or fell back.
509    pub provider_failure_rate: f64,
510    /// What the judges thought, never mixed into anything above.
511    pub user_experience: Vec<CriterionSummary>,
512}
513
514/// Samples counted for one category.
515#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
516#[serde(deny_unknown_fields)]
517pub struct CategoryStats {
518    /// Total samples in the run.
519    pub samples: usize,
520    /// Samples with at least one failure in this category.
521    pub failing_samples: usize,
522    /// Individual failures in this category.
523    pub failures: usize,
524}
525
526impl CategoryStats {
527    /// Fraction of samples with at least one failure in this category.
528    #[must_use]
529    pub fn failure_rate(&self) -> f64 {
530        ratio(self.failing_samples, self.samples)
531    }
532}
533
534/// One whole run.
535#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
536#[serde(deny_unknown_fields)]
537pub struct EvalReport {
538    /// The suite that ran.
539    pub suite: String,
540    /// When the run finished.
541    pub generated_at: DateTime<Utc>,
542    /// The configuration it ran under, so a reader can see how many samples and
543    /// how many votes produced these numbers.
544    pub config: EvalConfig,
545    /// Every item, in suite order.
546    pub items: Vec<ItemReport>,
547}
548
549impl EvalReport {
550    /// Builds a report.
551    #[must_use]
552    pub fn new(
553        suite: impl Into<String>,
554        generated_at: DateTime<Utc>,
555        config: EvalConfig,
556        items: Vec<ItemReport>,
557    ) -> Self {
558        Self {
559            suite: suite.into(),
560            generated_at,
561            config,
562            items,
563        }
564    }
565
566    /// One item by identifier.
567    #[must_use]
568    pub fn item(&self, id: &ItemId) -> Option<&ItemReport> {
569        self.items.iter().find(|item| &item.id == id)
570    }
571
572    /// Total samples across the run.
573    #[must_use]
574    pub fn total_samples(&self) -> usize {
575        self.items.iter().map(ItemReport::total_samples).sum()
576    }
577
578    /// Samples that satisfied every deterministic expectation.
579    #[must_use]
580    pub fn samples_passed(&self) -> usize {
581        self.items.iter().map(ItemReport::samples_passed).sum()
582    }
583
584    /// Fraction of samples that satisfied every deterministic expectation.
585    #[must_use]
586    pub fn deterministic_pass_rate(&self) -> f64 {
587        ratio(self.samples_passed(), self.total_samples())
588    }
589
590    /// Items whose samples disagreed with each other.
591    #[must_use]
592    pub fn flaky_items(&self) -> Vec<&ItemReport> {
593        self.items.iter().filter(|item| item.is_flaky()).collect()
594    }
595
596    /// The dashboard of §26.3.
597    #[must_use]
598    pub fn reliability(&self) -> Reliability {
599        let samples = self.total_samples();
600        let all: Vec<&SampleReport> = self
601            .items
602            .iter()
603            .flat_map(|item| item.samples.iter())
604            .collect();
605        Reliability {
606            side_effect_integrity: self.stats(ReliabilityCategory::SideEffectIntegrity),
607            operational_claim_integrity: self.stats(ReliabilityCategory::OperationalClaimIntegrity),
608            semantic_interpretation: self.stats(ReliabilityCategory::SemanticInterpretation),
609            clarification_integrity: self.stats(ReliabilityCategory::Clarification),
610            clarification_rate: ratio(all.iter().filter(|s| s.cards_created > 0).count(), samples),
611            abandonment_rate: ratio(all.iter().filter(|s| s.abandoned).count(), samples),
612            provider_failure_rate: ratio(
613                all.iter().filter(|s| s.provider_failures > 0).count(),
614                samples,
615            ),
616            user_experience: self.judge_summaries(),
617        }
618    }
619
620    /// The judge's numbers across the whole suite.
621    #[must_use]
622    pub fn judge_summaries(&self) -> Vec<CriterionSummary> {
623        let mut summaries = Vec::new();
624        for criterion in JudgeCriterion::ALL {
625            let outcomes: Vec<&CriterionOutcome> = self
626                .items
627                .iter()
628                .flat_map(|item| item.samples.iter())
629                .flat_map(|sample| sample.judge.iter())
630                .filter(|outcome| outcome.criterion == criterion)
631                .collect();
632            if outcomes.is_empty() {
633                continue;
634            }
635            summaries.push(summarize(criterion, &outcomes));
636        }
637        summaries
638    }
639
640    fn stats(&self, category: ReliabilityCategory) -> CategoryStats {
641        let mut stats = CategoryStats {
642            samples: self.total_samples(),
643            ..CategoryStats::default()
644        };
645        for item in &self.items {
646            stats.failing_samples += item.samples_failing(category);
647            stats.failures += item
648                .failures()
649                .into_iter()
650                .filter(|failure| ReliabilityCategory::of(failure.expectation) == category)
651                .count();
652        }
653        stats
654    }
655
656    /// The machine-readable form, for a continuous integration gate to archive
657    /// and for [`crate::baseline`] to read back.
658    ///
659    /// # Errors
660    ///
661    /// [`ReportError::Serialize`] when the report cannot be rendered as JSON.
662    pub fn to_json(&self) -> Result<String, ReportError> {
663        serde_json::to_string_pretty(self).map_err(|error| ReportError::Serialize {
664            message: error.to_string(),
665        })
666    }
667
668    /// Reads a report back from its machine-readable form.
669    ///
670    /// # Errors
671    ///
672    /// [`ReportError::Deserialize`] when the document is not a report.
673    pub fn from_json(source: &str) -> Result<Self, ReportError> {
674        serde_json::from_str(source).map_err(|error| ReportError::Deserialize {
675            message: error.to_string(),
676        })
677    }
678
679    /// A readable summary, one block per item plus the dashboard.
680    #[must_use]
681    pub fn summary(&self) -> String {
682        use std::fmt::Write as _;
683        let mut out = String::new();
684        let _ = writeln!(
685            out,
686            "suite {}: {} items, {} samples/item, {} votes/sample",
687            self.suite,
688            self.items.len(),
689            self.config.execution.samples_per_item,
690            self.config.judging.votes_per_sample
691        );
692        let _ = writeln!(
693            out,
694            "deterministic pass rate {:.0}% ({}/{} samples)",
695            self.deterministic_pass_rate() * 100.0,
696            self.samples_passed(),
697            self.total_samples()
698        );
699        // The second number, on its own line and never folded into the first.
700        // A refused proposal is a passing sample — nothing happened, which is
701        // what the item asked — so a report that showed only the pass rate would
702        // hide exactly the turns where the model read the situation wrongly and
703        // the runtime carried it.
704        let refused: usize = self.items.iter().map(ItemReport::refused_proposals).sum();
705        let measured: usize = self
706            .items
707            .iter()
708            .flat_map(|item| &item.samples)
709            .filter(|sample| sample.harness_error.is_none())
710            .count();
711        if measured > 0 {
712            #[allow(clippy::cast_precision_loss)] // counts, not quantities
713            let share = refused as f64 / measured as f64 * 100.0;
714            let _ = writeln!(
715                out,
716                "refused proposals {share:.0}% ({refused}/{measured} measured samples): \
717                 the model proposed an act and nothing was journaled"
718            );
719            // The third number. A discarded answer is invisible in both of the
720            // others: the sample usually passes, because the repair round
721            // recovered, and when it does not the turn simply has no effects to
722            // report. This is where a turn that closed without proposing or
723            // asking anything finally shows up as something other than silence.
724            let discarded: usize = self
725                .items
726                .iter()
727                .map(ItemReport::samples_with_discards)
728                .sum();
729            #[allow(clippy::cast_precision_loss)] // counts, not quantities
730            let share = discarded as f64 / measured as f64 * 100.0;
731            let _ = writeln!(
732                out,
733                "discarded answers {share:.0}% ({discarded}/{measured} measured samples): \
734                 the runtime threw the model's answer away whole{}",
735                match self.discard_codes() {
736                    codes if codes.is_empty() => String::new(),
737                    codes => format!(": {codes}"),
738                }
739            );
740        }
741        if let Some(line) = self.task_accuracy() {
742            let _ = writeln!(out, "understanding by task: {line}");
743        }
744        for item in &self.items {
745            write_item(&mut out, item);
746        }
747        write_reliability(&mut out, &self.reliability());
748        out
749    }
750
751    /// Each understanding task's accuracy over the samples whose item says what it
752    /// expects of that task, as `segment 40/46 · route 30/31`; `None` when no item
753    /// says.
754    #[must_use]
755    pub fn task_accuracy(&self) -> Option<String> {
756        let samples: Vec<&SampleReport> = self
757            .items
758            .iter()
759            .flat_map(|item| &item.samples)
760            .filter(|sample| sample.harness_error.is_none())
761            .collect();
762        let tasks = crate::understanding::TaskScores::default().by_task();
763        let line: Vec<String> = tasks
764            .iter()
765            .enumerate()
766            .filter_map(|(at, (task, _))| {
767                let scored: Vec<bool> = samples
768                    .iter()
769                    .filter_map(|sample| sample.tasks.by_task()[at].1)
770                    .collect();
771                if scored.is_empty() {
772                    return None;
773                }
774                let passed = scored.iter().filter(|pass| **pass).count();
775                Some(format!("{task} {passed}/{}", scored.len()))
776            })
777            .collect();
778        (!line.is_empty()).then(|| line.join(" · "))
779    }
780
781    /// The discarded-answer codes of the whole run, commonest first, rendered
782    /// as `code×n`.
783    ///
784    /// Which code it is decides what to do about it, and the two that this
785    /// corpus produces want opposite answers: `evidence` means the model cited
786    /// something the turn does not contain, which is a reading the grounding
787    /// rule caught, while `schema_violation` means the document was the wrong
788    /// shape, which is a schema the model cannot satisfy. Reporting only a rate
789    /// would leave the reader unable to tell them apart.
790    #[must_use]
791    pub fn discard_codes(&self) -> String {
792        let mut counts: std::collections::BTreeMap<&str, usize> = std::collections::BTreeMap::new();
793        for sample in self.items.iter().flat_map(|item| &item.samples) {
794            for code in &sample.discarded_answers {
795                *counts.entry(code.as_str()).or_default() += 1;
796            }
797        }
798        let mut ordered: Vec<(&str, usize)> = counts.into_iter().collect();
799        // Commonest first, and alphabetical within a tie so two runs of the same
800        // numbers render the same string.
801        ordered.sort_by(|left, right| right.1.cmp(&left.1).then_with(|| left.0.cmp(right.0)));
802        ordered
803            .iter()
804            .map(|(code, count)| format!("{code}×{count}"))
805            .collect::<Vec<_>>()
806            .join(", ")
807    }
808
809    /// Applies a gate's thresholds.
810    #[must_use]
811    pub fn gate(&self, thresholds: &GateThresholds) -> GateOutcome {
812        let mut violations = Vec::new();
813        let rate = self.deterministic_pass_rate();
814        if rate < thresholds.min_deterministic_pass_rate {
815            violations.push(GateViolation {
816                scope: self.suite.clone(),
817                rule: "min_deterministic_pass_rate".to_owned(),
818                detail: format!("{rate:.3} < {:.3}", thresholds.min_deterministic_pass_rate),
819            });
820        }
821        let reliability = self.reliability();
822        if thresholds.forbid_side_effect_failures
823            && reliability.side_effect_integrity.failing_samples > 0
824        {
825            violations.push(GateViolation {
826                scope: self.suite.clone(),
827                rule: "forbid_side_effect_failures".to_owned(),
828                detail: format!(
829                    "{} samples failed a side-effect integrity assertion",
830                    reliability.side_effect_integrity.failing_samples
831                ),
832            });
833        }
834        if reliability.abandonment_rate > thresholds.max_abandonment_rate {
835            violations.push(GateViolation {
836                scope: self.suite.clone(),
837                rule: "max_abandonment_rate".to_owned(),
838                detail: format!(
839                    "{:.3} > {:.3}",
840                    reliability.abandonment_rate, thresholds.max_abandonment_rate
841                ),
842            });
843        }
844        if !thresholds.allow_flaky_items {
845            for item in self.flaky_items() {
846                violations.push(GateViolation {
847                    scope: item.id.to_string(),
848                    rule: "allow_flaky_items".to_owned(),
849                    detail: format!(
850                        "{}/{} samples passed",
851                        item.samples_passed(),
852                        item.total_samples()
853                    ),
854                });
855            }
856        }
857        if let Some(minimum) = thresholds.min_judge_score {
858            for summary in &reliability.user_experience {
859                if summary.mean_score.is_some_and(|score| score < minimum) {
860                    violations.push(GateViolation {
861                        scope: summary.criterion.to_string(),
862                        rule: "min_judge_score".to_owned(),
863                        detail: format!(
864                            "{:.2} < {minimum:.2}",
865                            summary.mean_score.unwrap_or_default()
866                        ),
867                    });
868                }
869            }
870        }
871        GateOutcome {
872            passed: violations.is_empty(),
873            violations,
874        }
875    }
876}
877
878fn write_item(out: &mut String, item: &ItemReport) {
879    use std::fmt::Write as _;
880    let variance = item.variance();
881    let _ = writeln!(
882        out,
883        "  {}: {}/{} samples passed, {} distinct behaviour(s){}",
884        item.id,
885        item.samples_passed(),
886        item.total_samples(),
887        variance.distinct_behaviours,
888        if item.is_flaky() { ", FLAKY" } else { "" }
889    );
890    // An unmeasured sample is not a failing one, and a summary that hid the
891    // difference would read as "the agent got it wrong" when the truth is that
892    // nobody asked it anything.
893    for sample in &item.samples {
894        if let Some(error) = &sample.harness_error {
895            let _ = writeln!(out, "      sample {} not measured: {error}", sample.sample);
896        }
897    }
898    // Only when it happened: an ordinary item's block is unchanged, and an
899    // item whose answers were being thrown away says so next to its own
900    // numbers rather than only in the run-wide total.
901    let discarded = item.samples_with_discards();
902    if discarded > 0 {
903        let mut codes: Vec<&str> = item
904            .samples
905            .iter()
906            .flat_map(|sample| sample.discarded_answers.iter().map(String::as_str))
907            .collect();
908        codes.sort_unstable();
909        codes.dedup();
910        // The measured samples, not every sample: the numerator already
911        // excludes the ones the harness could not set up, and mixing the two
912        // printed `1/2` under a run-wide line that correctly said `1/1`.
913        let measured = item
914            .samples
915            .iter()
916            .filter(|sample| sample.harness_error.is_none())
917            .count();
918        let _ = writeln!(
919            out,
920            "      {discarded}/{measured} sample(s) had an answer discarded whole: {}",
921            codes.join(", ")
922        );
923    }
924    for failure in item.failures() {
925        let _ = writeln!(
926            out,
927            "      [{}] {failure}",
928            ReliabilityCategory::of(failure.expectation)
929        );
930    }
931    for summary in item.judge_summaries() {
932        let _ = writeln!(
933            out,
934            "      judge {}: mean {:.2} over {} sample(s), vote spread {:.2}",
935            summary.criterion,
936            summary.mean_score.unwrap_or_default(),
937            summary.samples_judged,
938            summary.mean_vote_spread
939        );
940    }
941}
942
943fn write_reliability(out: &mut String, reliability: &Reliability) {
944    use std::fmt::Write as _;
945    let _ = writeln!(out, "reliability (spec §26.3, categories kept apart):");
946    for (label, stats) in [
947        ("side-effect integrity", reliability.side_effect_integrity),
948        (
949            "operational claim integrity",
950            reliability.operational_claim_integrity,
951        ),
952        (
953            "semantic interpretation",
954            reliability.semantic_interpretation,
955        ),
956        (
957            "clarification integrity",
958            reliability.clarification_integrity,
959        ),
960    ] {
961        let _ = writeln!(
962            out,
963            "  {label}: {} failing sample(s), {} failure(s)",
964            stats.failing_samples, stats.failures
965        );
966    }
967    let _ = writeln!(
968        out,
969        "  clarification rate {:.0}%, abandonment rate {:.0}%, provider failure rate {:.0}%",
970        reliability.clarification_rate * 100.0,
971        reliability.abandonment_rate * 100.0,
972        reliability.provider_failure_rate * 100.0
973    );
974    for summary in &reliability.user_experience {
975        let _ = writeln!(
976            out,
977            "  user experience, {}: mean {:.2}, vote spread {:.2}",
978            summary.criterion,
979            summary.mean_score.unwrap_or_default(),
980            summary.mean_vote_spread
981        );
982    }
983}
984
985/// What a continuous integration gate refuses to merge.
986#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
987#[serde(deny_unknown_fields, default)]
988pub struct GateThresholds {
989    /// Minimum fraction of samples that must satisfy every deterministic
990    /// expectation.
991    pub min_deterministic_pass_rate: f64,
992    /// Maximum fraction of samples that may produce nothing usable.
993    pub max_abandonment_rate: f64,
994    /// Whether an item whose samples disagree may still pass.
995    pub allow_flaky_items: bool,
996    /// Whether any side-effect integrity failure fails the gate outright,
997    /// whatever the pass rate is. On by default: §26.3 exists so this is not a
998    /// percentage.
999    pub forbid_side_effect_failures: bool,
1000    /// Minimum mean judge score, when the run judged anything.
1001    pub min_judge_score: Option<f64>,
1002}
1003
1004impl Default for GateThresholds {
1005    fn default() -> Self {
1006        Self {
1007            min_deterministic_pass_rate: 1.0,
1008            max_abandonment_rate: 0.0,
1009            allow_flaky_items: false,
1010            forbid_side_effect_failures: true,
1011            min_judge_score: None,
1012        }
1013    }
1014}
1015
1016/// The gate's verdict.
1017#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
1018#[serde(deny_unknown_fields)]
1019pub struct GateOutcome {
1020    /// Whether the run may merge.
1021    pub passed: bool,
1022    /// Every threshold that was not met.
1023    pub violations: Vec<GateViolation>,
1024}
1025
1026/// One unmet threshold.
1027#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
1028#[serde(deny_unknown_fields)]
1029pub struct GateViolation {
1030    /// The suite, item or criterion it is about.
1031    pub scope: String,
1032    /// Which threshold.
1033    pub rule: String,
1034    /// The numbers.
1035    pub detail: String,
1036}
1037
1038impl std::fmt::Display for GateViolation {
1039    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
1040        write!(f, "{} [{}]: {}", self.rule, self.scope, self.detail)
1041    }
1042}
1043
1044/// Why a report could not be rendered or read back.
1045#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)]
1046#[non_exhaustive]
1047pub enum ReportError {
1048    /// The report could not be rendered as JSON.
1049    #[error("report could not be serialized: {message}")]
1050    Serialize {
1051        /// What `serde` said.
1052        message: String,
1053    },
1054    /// The document is not a report.
1055    #[error("report could not be read: {message}")]
1056    Deserialize {
1057        /// What `serde` said.
1058        message: String,
1059    },
1060}
1061
1062#[cfg(test)]
1063mod tests {
1064
1065    #[test]
1066    fn a_refused_proposal_is_counted_and_a_carried_one_is_not() {
1067        let mut item = ItemReport {
1068            id: ItemId::new("i"),
1069            name: "An item".to_owned(),
1070            tags: Vec::new(),
1071            fingerprint: ItemFingerprint::default(),
1072            samples: vec![sample(1, "A", false)],
1073        };
1074        // Proposed and executed: the model read it right.
1075        item.samples[0].acts_proposed = 1;
1076        item.samples[0].acts_refused = 0;
1077        item.samples[0].commands_journaled = 1;
1078        assert_eq!(item.refused_proposals(), 0);
1079
1080        // Refused: the structure turned the reading down. The sample may still
1081        // pass — nothing happened, which is often what the item asked — and
1082        // that is exactly why this is counted separately.
1083        item.samples[0].acts_refused = 1;
1084        item.samples[0].commands_journaled = 0;
1085        assert_eq!(item.refused_proposals(), 1);
1086
1087        // The two that journal nothing and are NOT refusals, which is the whole
1088        // reason this is not counted off the command total: an act valid on a
1089        // record already in the requested state, and one waiting for a person
1090        // to confirm it. Counting either reports the design working as the
1091        // model failing.
1092        item.samples[0].acts_refused = 0;
1093        assert_eq!(item.refused_proposals(), 0);
1094
1095        // And the other way round: a plan that refused one act while another
1096        // one wrote. The command count hides this entirely.
1097        item.samples[0].acts_proposed = 2;
1098        item.samples[0].acts_refused = 1;
1099        item.samples[0].commands_journaled = 1;
1100        assert_eq!(item.refused_proposals(), 1);
1101
1102        // Proposed nothing: an answer, not a refusal.
1103        item.samples[0].acts_proposed = 0;
1104        item.samples[0].acts_refused = 0;
1105        assert_eq!(item.refused_proposals(), 0);
1106
1107        // Never ran: the harness could not build the world, so there was no
1108        // reading to be right or wrong about.
1109        item.samples[0].acts_proposed = 1;
1110        item.samples[0].acts_refused = 1;
1111        item.samples[0].harness_error = Some("no world".to_owned());
1112        assert_eq!(item.refused_proposals(), 0);
1113    }
1114    use super::*;
1115    use crate::judge::{JudgeVerdict, JudgeVote};
1116
1117    fn sample(index: u32, signature: &str, failing: bool) -> SampleReport {
1118        SampleReport {
1119            sample: index,
1120            failures: if failing {
1121                vec![AssertionFailure::new(
1122                    ExpectationName::Commands,
1123                    "[a]",
1124                    "[b]",
1125                )]
1126            } else {
1127                Vec::new()
1128            },
1129            harness_error: None,
1130            signature: signature.to_owned(),
1131            judge: Vec::new(),
1132            acts_proposed: 0,
1133            acts_refused: 0,
1134            commands_journaled: 0,
1135            provider_failures: 0,
1136            cards_created: 0,
1137            abandoned: false,
1138            discarded_answers: Vec::new(),
1139            answer: String::new(),
1140            tasks: Default::default(),
1141        }
1142    }
1143
1144    fn report(samples: Vec<SampleReport>) -> EvalReport {
1145        EvalReport::new(
1146            "s",
1147            DateTime::from_timestamp(0, 0).unwrap_or_default(),
1148            EvalConfig::default(),
1149            vec![ItemReport {
1150                id: ItemId::new("i"),
1151                name: "An item".to_owned(),
1152                tags: Vec::new(),
1153                fingerprint: ItemFingerprint::default(),
1154                samples,
1155            }],
1156        )
1157    }
1158
1159    /// The third number reaches the summary, and says which code it was.
1160    ///
1161    /// The scenario is the one it exists for: every sample PASSED. An answer
1162    /// was thrown away and the repair round recovered, so the effects are right
1163    /// and the pass rate is 100% — and a report that stopped there would
1164    /// describe a turn that struggled as a turn that did not.
1165    #[test]
1166    fn a_discarded_answer_is_reported_even_when_every_sample_passed() {
1167        let mut passing = sample(1, "A", false);
1168        passing.discarded_answers = vec!["evidence".to_owned()];
1169        let mut clean = sample(2, "A", false);
1170        clean.discarded_answers.clear();
1171        let report = report(vec![passing, clean]);
1172
1173        assert_eq!(report.items[0].samples_with_discards(), 1);
1174        assert!(
1175            (report.deterministic_pass_rate() - 1.0).abs() < 1e-9,
1176            "every sample passed, which is the whole point of the number"
1177        );
1178
1179        let summary = report.summary();
1180        assert!(
1181            summary.contains("discarded answers 50% (1/2 measured samples)"),
1182            "the run-wide line states the rate: {summary}"
1183        );
1184        assert!(
1185            summary.contains("evidence×1"),
1186            "and which code it was, because the answer differs by code: {summary}"
1187        );
1188        assert!(
1189            summary.contains("1/2 sample(s) had an answer discarded whole: evidence"),
1190            "the item says it next to its own numbers too: {summary}"
1191        );
1192    }
1193
1194    /// Counted per sample, not per answer.
1195    ///
1196    /// A turn that lost three answers in a row is one turn that struggled. The
1197    /// per-answer count is still there in the codes, which is where it belongs:
1198    /// it says what went wrong, not how many turns went wrong.
1199    #[test]
1200    fn a_sample_that_lost_three_answers_is_one_sample() {
1201        let mut struggled = sample(1, "A", false);
1202        struggled.discarded_answers = vec![
1203            "evidence".to_owned(),
1204            "evidence".to_owned(),
1205            "schema_violation".to_owned(),
1206        ];
1207        let report = report(vec![struggled, sample(2, "A", false)]);
1208
1209        assert_eq!(report.items[0].samples_with_discards(), 1);
1210        assert_eq!(report.discard_codes(), "evidence×2, schema_violation×1");
1211    }
1212
1213    /// A sample that never ran discarded nothing.
1214    ///
1215    /// Same rule as the refused proposals above it: the harness could not build
1216    /// the world, so there was no answer to throw away.
1217    #[test]
1218    fn an_unmeasured_sample_discards_nothing() {
1219        let mut never_ran = sample(1, "A", false);
1220        never_ran.discarded_answers = vec!["evidence".to_owned()];
1221        never_ran.harness_error = Some("no world".to_owned());
1222        let report = report(vec![never_ran]);
1223        assert_eq!(report.items[0].samples_with_discards(), 0);
1224    }
1225
1226    #[test]
1227    fn variance_counts_distinct_behaviours_not_failures() {
1228        let report = report(vec![
1229            sample(1, "A", false),
1230            sample(2, "B", true),
1231            sample(3, "A", false),
1232        ]);
1233        let variance = report.items[0].variance();
1234        assert_eq!(variance.distinct_behaviours, 2);
1235        assert!((variance.pass_rate - 2.0 / 3.0).abs() < 1e-9);
1236        assert!(variance.pass_variance > 0.0);
1237        assert_eq!(variance.behaviours[0].count, 2);
1238        assert!(report.items[0].is_flaky());
1239    }
1240
1241    #[test]
1242    fn a_side_effect_failure_is_not_averaged_into_a_language_score() {
1243        let mut failing = sample(1, "A", true);
1244        failing.judge = vec![CriterionOutcome {
1245            criterion: JudgeCriterion::Tone,
1246            votes: vec![JudgeVote {
1247                vote: 1,
1248                verdict: Some(JudgeVerdict {
1249                    score: 5,
1250                    reason: "fine".to_owned(),
1251                }),
1252                error: None,
1253            }],
1254        }];
1255        let report = report(vec![failing]);
1256        let reliability = report.reliability();
1257        assert_eq!(reliability.side_effect_integrity.failing_samples, 1);
1258        assert_eq!(reliability.user_experience[0].mean_score, Some(5.0));
1259        assert!((report.deterministic_pass_rate() - 0.0).abs() < 1e-9);
1260    }
1261
1262    #[test]
1263    fn the_default_gate_refuses_a_side_effect_failure() {
1264        let outcome = report(vec![sample(1, "A", true)]).gate(&GateThresholds::default());
1265        assert!(!outcome.passed);
1266        assert!(
1267            outcome
1268                .violations
1269                .iter()
1270                .any(|v| v.rule == "forbid_side_effect_failures"),
1271            "{:?}",
1272            outcome.violations
1273        );
1274    }
1275
1276    #[test]
1277    fn a_report_round_trips_through_its_machine_readable_form() {
1278        let report = report(vec![sample(1, "A", false)]);
1279        let json = report.to_json().unwrap();
1280        assert_eq!(EvalReport::from_json(&json).unwrap(), report);
1281    }
1282}