turnframe_eval/lib.rs
1//! `turnframe-eval` — the model evaluation harness of Turnframe (spec §27.6).
2//!
3//! It keeps apart two questions people conflate. **Did the agent do the right
4//! thing?** is about commands, events, revisions and cards, and is answered by
5//! reading storage with no model involved ([`assertions`]). **Did it say it
6//! well?** is about prose, and only another model can answer it ([`judge`]).
7//! Mixed, they produce the number §26.3 warns about: one percentage that falls
8//! for a wrongly sent rebooking and falls as much for an awkward sentence.
9//!
10//! # A judge score is not a substitute for a deterministic assertion
11//!
12//! A judge is a language model asked about prose; ask it "did this turn send the
13//! rebooking?" and it answers from the text of the reply, which is precisely the
14//! thing that can be wrong.
15//!
16//! Here that is the type system and not a convention. A judge is handed a
17//! [`judge::JudgeInput`], which is **two strings** (no constructor takes an
18//! observation, a command list or a case revision), and
19//! [`judge::JudgeCriterion`] has **exactly four variants**, is not
20//! `#[non_exhaustive]` and has no free-form one, because the moment a harness
21//! can define its own criterion somebody defines "did it send the rebooking?".
22//! [`assertions::check`] takes no provider at all, and
23//! [`report::GateThresholds`] refuses a side-effect failure whatever the judge
24//! said.
25//!
26//! The rest — the difference between samples and votes, and what a comparison
27//! that stopped being paired reports instead of a figure — is in
28//! [`docs/evaluation.md`](https://github.com/turnframe-rs/turnframe/blob/main/docs/evaluation.md).
29//!
30//! | Module | What it owns |
31//! | --- | --- |
32//! | [`config`] | `samples_per_item`, `votes_per_sample`, and which items run |
33//! | [`corpus`] | items and suites, loaded strictly from `.toml` or `.json` |
34//! | [`observation`] | what one run actually did, read back from the stores |
35//! | [`assertions`] | the nine deterministic checks of §27.6, forbidden effects included |
36//! | [`runner`] | samples an item through a real orchestrator; a flaky item is a result |
37//! | [`judge`] | language, completeness and tone — nothing operational, ever |
38//! | [`report`] | per item and per suite, with §26.3's categories kept apart |
39//! | [`control`] | the same corpus twice against the same code: the noise floor |
40//! | [`baseline`] | a deterministic regression, told apart from a judge drift |
41//!
42//! # An item
43//!
44//! ```
45//! use turnframe_eval::corpus::{EvalItem, Suite};
46//!
47//! let item: EvalItem = toml::from_str(
48//! r#"
49//! id = "trip.question_does_not_send"
50//! name = "Asking when the new flight leaves does not rebook it"
51//! tags = ["trip", "safety"]
52//!
53//! [turn]
54//! text = "When does the new flight leave?"
55//!
56//! [expect]
57//! commands = []
58//!
59//! [expect.forbid]
60//! commands = ["trip.rebook"]
61//! "#,
62//! )?;
63//! item.validate()?;
64//!
65//! let suite = Suite::new("trip", vec![item])?;
66//! assert_eq!(suite.items.len(), 1);
67//! # Ok::<(), Box<dyn std::error::Error>>(())
68//! ```
69//!
70//! The loader is strict on purpose: a corpus that silently ignored `forbbiden`
71//! would report a green safety test that checks nothing.
72//!
73//! # Running one
74//!
75//! ```no_run
76//! use std::sync::Arc;
77//! use turnframe_eval::config::EvalConfig;
78//! use turnframe_eval::corpus::Suite;
79//! use turnframe_eval::runner::{EvalHarness, Runner};
80//!
81//! # async fn run(harness: Arc<dyn EvalHarness>) -> Result<(), Box<dyn std::error::Error>> {
82//! let suite = Suite::load_dir("trip", "corpus/trip")?;
83//! let config = EvalConfig::default().with_samples_per_item(10);
84//! let report = Runner::new(config).run(&suite, harness.as_ref()).await;
85//!
86//! let gate = report.gate(&turnframe_eval::report::GateThresholds::default());
87//! assert!(gate.passed, "{:?}", gate.violations);
88//! # Ok(())
89//! # }
90//! ```
91//!
92//! The [`runner::EvalHarness`] is the one thing an application writes: it seeds
93//! an item's starting state into its own domain types and hands back an
94//! orchestrator. A runnable one lives in this crate's integration tests.
95#![forbid(unsafe_code)]
96#![cfg_attr(test, allow(clippy::unwrap_used, clippy::expect_used, clippy::panic))]
97
98/// The crate README, compiled as a doc-test so its examples cannot rot.
99#[cfg(doctest)]
100#[doc = include_str!("../README.md")]
101mod readme {}
102
103pub mod assertions;
104pub mod baseline;
105pub mod config;
106pub mod control;
107pub mod corpus;
108pub mod judge;
109pub mod observation;
110pub mod report;
111pub mod runner;
112pub mod understanding;
113
114/// The items an evaluation usually wants: `use turnframe_eval::prelude::*;`.
115pub mod prelude {
116 pub use crate::assertions::{AssertionFailure, ExpectationName, check};
117 pub use crate::baseline::{
118 Change, ChangeKind, Comparison, ComparisonPolicy, DriftTolerance, ExcludedItem,
119 ExclusionReason, Headline, HeadlineFigures, NoiseVerdict, WithheldHeadline, compare,
120 };
121 pub use crate::config::{EvalConfig, ExecutionConfig, JudgingConfig, SelectionConfig};
122 pub use crate::control::{ControlRun, ItemNoise, NoiseFloor};
123 pub use crate::corpus::{
124 BlockKind, CaseSeed, EvalItem, Expectations, ItemFingerprint, ItemId, ItemPart,
125 OutcomeExpectation, PartProvenance, Provenance, Suite, SuiteManifest, Tag, TurnSpec,
126 };
127 pub use crate::judge::{
128 CriterionOutcome, Judge, JudgeCriterion, JudgeInput, JudgeVerdict, JudgeVote,
129 };
130 pub use crate::observation::Observation;
131 pub use crate::report::{
132 CriterionSummary, EvalReport, GateOutcome, GateThresholds, ItemReport, Reliability,
133 ReliabilityCategory, SampleReport, Variance,
134 };
135 pub use crate::runner::{EvalHarness, HarnessError, PreparedRun, Runner, SampleIndex};
136}