turnframe_eval/lib.rs
1//! `turnframe-eval` — the model evaluation harness of Turnframe (spec §27.6).
2//!
3//! It keeps apart two questions people conflate. **Did the agent do the right
4//! thing?** is about commands, events, revisions and cards, and is answered by
5//! reading storage with no model involved ([`assertions`]). **Did it say it
6//! well?** is about prose, and only another model can answer it ([`judge`]).
7//! Mixed, they produce the number §26.3 warns about: one percentage that falls
8//! for a wrongly sent rebooking and falls as much for an awkward sentence.
9//!
10//! # A judge score is not a substitute for a deterministic assertion
11//!
12//! A judge is a language model asked about prose; ask it "did this turn send the
13//! rebooking?" and it answers from the text of the reply, which is precisely the
14//! thing that can be wrong.
15//!
16//! Here that is the type system and not a convention. A judge is handed a
17//! [`judge::JudgeInput`], which is **two strings** (no constructor takes an
18//! observation, a command list or a case revision), and
19//! [`judge::JudgeCriterion`] has **exactly four variants**, is not
20//! `#[non_exhaustive]` and has no free-form one, because the moment a harness
21//! can define its own criterion somebody defines "did it send the rebooking?".
22//! [`assertions::check`] takes no provider at all, and
23//! [`report::GateThresholds`] refuses a side-effect failure whatever the judge
24//! said.
25//!
26//! The rest — the difference between samples and votes, and what a comparison
27//! that stopped being paired reports instead of a figure — is in
28//! [`docs/evaluation.md`](https://github.com/turnframe-rs/turnframe/blob/main/docs/evaluation.md).
29//!
30//! | Module | What it owns |
31//! | --- | --- |
32//! | [`config`] | `samples_per_item`, `votes_per_sample`, and which items run |
33//! | [`corpus`] | items and suites, loaded strictly from `.toml` or `.json` |
34//! | [`observation`] | what one run actually did, read back from the stores |
35//! | [`assertions`] | the nine deterministic checks of §27.6, forbidden effects included |
36//! | [`runner`] | samples an item through a real orchestrator; a flaky item is a result |
37//! | [`judge`] | language, completeness and tone — nothing operational, ever |
38//! | [`report`] | per item and per suite, with §26.3's categories kept apart |
39//! | [`control`] | the same corpus twice against the same code: the noise floor |
40//! | [`baseline`] | a deterministic regression, told apart from a judge drift |
41//! | [`simulate`] | goal-driven conversations with a simulated user, scored by code |
42//!
43//! # An item
44//!
45//! ```
46//! use turnframe_eval::corpus::{EvalItem, Suite};
47//!
48//! let item: EvalItem = toml::from_str(
49//! r#"
50//! id = "trip.question_does_not_send"
51//! name = "Asking when the new flight leaves does not rebook it"
52//! tags = ["trip", "safety"]
53//!
54//! [turn]
55//! text = "When does the new flight leave?"
56//!
57//! [expect]
58//! commands = []
59//!
60//! [expect.forbid]
61//! commands = ["trip.rebook"]
62//! "#,
63//! )?;
64//! item.validate()?;
65//!
66//! let suite = Suite::new("trip", vec![item])?;
67//! assert_eq!(suite.items.len(), 1);
68//! # Ok::<(), Box<dyn std::error::Error>>(())
69//! ```
70//!
71//! The loader is strict on purpose: a corpus that silently ignored `forbbiden`
72//! would report a green safety test that checks nothing.
73//!
74//! # Running one
75//!
76//! ```no_run
77//! use std::sync::Arc;
78//! use turnframe_eval::config::EvalConfig;
79//! use turnframe_eval::corpus::Suite;
80//! use turnframe_eval::runner::{EvalHarness, Runner};
81//!
82//! # async fn run(harness: Arc<dyn EvalHarness>) -> Result<(), Box<dyn std::error::Error>> {
83//! let suite = Suite::load_dir("trip", "corpus/trip")?;
84//! let config = EvalConfig::default().with_samples_per_item(10);
85//! let report = Runner::new(config).run(&suite, harness.as_ref()).await;
86//!
87//! let gate = report.gate(&turnframe_eval::report::GateThresholds::default());
88//! assert!(gate.passed, "{:?}", gate.violations);
89//! # Ok(())
90//! # }
91//! ```
92//!
93//! The [`runner::EvalHarness`] is the one thing an application writes: it seeds
94//! an item's starting state into its own domain types and hands back an
95//! orchestrator. A runnable one lives in this crate's integration tests.
96#![forbid(unsafe_code)]
97#![cfg_attr(test, allow(clippy::unwrap_used, clippy::expect_used, clippy::panic))]
98
99/// The crate README, compiled as a doc-test so its examples cannot rot.
100#[cfg(doctest)]
101#[doc = include_str!("../README.md")]
102mod readme {}
103
104pub mod assertions;
105pub mod baseline;
106pub mod config;
107pub mod control;
108pub mod corpus;
109pub mod judge;
110pub mod observation;
111pub mod report;
112pub mod runner;
113pub mod simulate;
114pub mod understanding;
115
116/// The items an evaluation usually wants: `use turnframe_eval::prelude::*;`.
117pub mod prelude {
118 pub use crate::assertions::{AssertionFailure, ExpectationName, check};
119 pub use crate::baseline::{
120 Change, ChangeKind, Comparison, ComparisonPolicy, DriftTolerance, ExcludedItem,
121 ExclusionReason, Headline, HeadlineFigures, NoiseVerdict, WithheldHeadline, compare,
122 };
123 pub use crate::config::{EvalConfig, ExecutionConfig, JudgingConfig, SelectionConfig};
124 pub use crate::control::{ControlRun, ItemNoise, NoiseFloor};
125 pub use crate::corpus::{
126 BlockKind, CaseSeed, EvalItem, Expectations, ItemFingerprint, ItemId, ItemPart,
127 OutcomeExpectation, PartProvenance, Provenance, Suite, SuiteManifest, Tag, TurnSpec,
128 };
129 pub use crate::judge::{
130 CriterionOutcome, Judge, JudgeCriterion, JudgeInput, JudgeVerdict, JudgeVote,
131 };
132 pub use crate::observation::Observation;
133 pub use crate::report::{
134 CriterionSummary, EvalReport, GateOutcome, GateThresholds, ItemReport, Reliability,
135 ReliabilityCategory, SampleReport, Variance,
136 };
137 pub use crate::runner::{EvalHarness, HarnessError, PreparedRun, Runner, SampleIndex};
138}