Skip to main content

turnframe_eval/
config.rs

1//! How a run is configured (spec §27.6).
2//!
3//! The whole point of this module is one distinction the specification insists
4//! on, and it is worth stating before any type appears:
5//!
6//! * **A sample is one execution of the agent under test.** More samples
7//!   measure how much the *model* varies. Ten samples of the same item are ten
8//!   chances for understanding to read something different, and the spread
9//!   between them is the number a release gate cares about.
10//! * **A vote is one judge opinion about one sample.** More votes measure how
11//!   much the *judge* varies. Three votes on one sample tell you nothing about
12//!   the agent; they tell you whether the judge would have said the same thing
13//!   twice.
14//!
15//! Averaging the two together produces a number that moves when either the
16//! model or the judge wobbles and cannot say which — which is exactly the
17//! failure §26.3 warns about. So they are two settings, they are counted
18//! separately, and they are reported separately.
19//!
20//! The file format is the one printed in the specification:
21//!
22//! ```toml
23//! [execution]
24//! samples_per_item = 10
25//!
26//! [judging]
27//! votes_per_sample = 3
28//! ```
29
30use std::path::{Path, PathBuf};
31
32use serde::{Deserialize, Serialize};
33
34use crate::corpus::{ItemId, Tag};
35
36/// Everything one evaluation run needs to know.
37///
38/// ```
39/// use turnframe_eval::config::EvalConfig;
40///
41/// let config = EvalConfig::from_toml_str(
42///     r#"
43///     [execution]
44///     samples_per_item = 10
45///
46///     [judging]
47///     votes_per_sample = 3
48///     "#,
49/// )?;
50///
51/// assert_eq!(config.execution.samples_per_item, 10);
52/// assert_eq!(config.judging.votes_per_sample, 3);
53/// config.validate()?;
54/// # Ok::<(), turnframe_eval::config::ConfigError>(())
55/// ```
56#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
57#[serde(deny_unknown_fields, default)]
58pub struct EvalConfig {
59    /// How the agent under test is exercised.
60    pub execution: ExecutionConfig,
61    /// How the judge is polled, when an item asks for one.
62    pub judging: JudgingConfig,
63    /// Which subset of a suite runs.
64    pub selection: SelectionConfig,
65}
66
67impl EvalConfig {
68    /// A run of one sample and one vote: the cheapest configuration that still
69    /// exercises every stage. Use it for a smoke run; a release gate wants the
70    /// numbers of spec §27.6 instead.
71    #[must_use]
72    pub fn single() -> Self {
73        Self::default()
74    }
75
76    /// Sets the number of executions of the agent under test per item.
77    #[must_use]
78    pub const fn with_samples_per_item(mut self, samples: u32) -> Self {
79        self.execution.samples_per_item = samples;
80        self
81    }
82
83    /// Keeps the turn's own words on each sample, for curating a corpus.
84    ///
85    /// See [`ExecutionConfig::record_answers`] for why this is off by default.
86    #[must_use]
87    pub const fn recording_answers(mut self) -> Self {
88        self.execution.record_answers = true;
89        self
90    }
91
92    /// Sets the number of judge opinions collected per *sample*.
93    #[must_use]
94    pub const fn with_votes_per_sample(mut self, votes: u32) -> Self {
95        self.judging.votes_per_sample = votes;
96        self
97    }
98
99    /// Sets how many samples may be in flight at once.
100    ///
101    /// Leave it at one for a reproducible in-memory run; raise it only for a
102    /// corpus against a real endpoint. See
103    /// [`ExecutionConfig::sample_concurrency`].
104    #[must_use]
105    pub const fn with_sample_concurrency(mut self, samples: u32) -> Self {
106        self.execution.sample_concurrency = samples;
107        self
108    }
109
110    /// Caps how many of a turn's events one observation records, loudly.
111    ///
112    /// See [`ExecutionConfig::max_observed_events`]: an observation that hits
113    /// the cap fails every assertion about events rather than passing over a
114    /// ledger it only half read.
115    #[must_use]
116    pub const fn with_max_observed_events(mut self, events: usize) -> Self {
117        self.execution.max_observed_events = Some(events);
118        self
119    }
120
121    /// Restricts the run to items carrying `tag`.
122    #[must_use]
123    pub fn including_tag(mut self, tag: impl Into<Tag>) -> Self {
124        self.selection.include_tags.push(tag.into());
125        self
126    }
127
128    /// Excludes items carrying `tag`, whatever else selects them.
129    #[must_use]
130    pub fn excluding_tag(mut self, tag: impl Into<Tag>) -> Self {
131        self.selection.exclude_tags.push(tag.into());
132        self
133    }
134
135    /// Parses a configuration from TOML.
136    ///
137    /// # Errors
138    ///
139    /// [`ConfigError::Parse`] when the document is malformed or carries a key
140    /// this crate does not understand.
141    pub fn from_toml_str(source: &str) -> Result<Self, ConfigError> {
142        toml::from_str(source).map_err(|error| ConfigError::Parse {
143            message: error.to_string(),
144        })
145    }
146
147    /// Reads a configuration from a TOML file.
148    ///
149    /// # Errors
150    ///
151    /// * [`ConfigError::Read`] when the file cannot be read;
152    /// * [`ConfigError::Parse`] when its contents are not a configuration.
153    pub fn load(path: impl AsRef<Path>) -> Result<Self, ConfigError> {
154        let path = path.as_ref();
155        let source = std::fs::read_to_string(path).map_err(|error| ConfigError::Read {
156            path: path.to_path_buf(),
157            message: error.to_string(),
158        })?;
159        Self::from_toml_str(&source)
160    }
161
162    /// Checks the configuration is one this crate will run.
163    ///
164    /// # Errors
165    ///
166    /// * [`ConfigError::ZeroSamples`] — an item that never executes has no
167    ///   result, not an empty one;
168    /// * [`ConfigError::ZeroVotes`] — a judge with no votes has no verdict.
169    pub const fn validate(&self) -> Result<(), ConfigError> {
170        if self.execution.samples_per_item == 0 {
171            return Err(ConfigError::ZeroSamples);
172        }
173        if self.judging.votes_per_sample == 0 {
174            return Err(ConfigError::ZeroVotes);
175        }
176        Ok(())
177    }
178}
179
180/// How the agent under test is exercised.
181///
182/// `#[non_exhaustive]`: this 0.1 grows a field whenever a run needs to say
183/// something new, and each one would be source-breaking for anybody constructing
184/// this with a struct literal — `#[serde(default)]` keeps a FILE readable and
185/// does nothing for Rust. Build it from [`Default`] and the `with_*` methods,
186/// which is what the next added field will not break.
187#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
188#[serde(deny_unknown_fields, default)]
189#[non_exhaustive]
190pub struct ExecutionConfig {
191    /// How many times each item runs end to end.
192    ///
193    /// This is the only knob that measures **model** variance. Raising it makes
194    /// a flaky item visible; raising [`JudgingConfig::votes_per_sample`] never
195    /// will.
196    pub samples_per_item: u32,
197    /// Stop the whole run after this many samples have failed their
198    /// deterministic assertions. `None` runs the corpus to the end, which is
199    /// what a nightly job wants; a pull-request gate may prefer to stop early.
200    pub stop_after_failures: Option<u32>,
201    /// How many samples may be in flight at once.
202    ///
203    /// One — the default — runs the corpus strictly one sample at a time, which
204    /// is what a reproducible in-memory run wants: the harness sees the samples
205    /// in index order, and a scripted provider keyed on call order sees exactly
206    /// the sequence it was written for.
207    ///
208    /// Raising it is for a corpus against a **real endpoint**, where a hundred
209    /// samples in series is hours of waiting on a network. What it changes is
210    /// the *execution* order: samples start and finish interleaved, so a
211    /// harness that shares anything between them — a counter, a queue of
212    /// scripted answers, a rate limit — will see a different order every run.
213    /// The report does not move: samples are still reported in index order,
214    /// and the same set of results comes back whatever this is set to. Zero is
215    /// read as one.
216    pub sample_concurrency: u32,
217    /// How many of a turn's events one observation may record before it stops
218    /// and says so.
219    ///
220    /// `None` — the default — reads the ledger to the end, paging the journal
221    /// by its sequence cursor. A `Some(limit)` is a deliberate ceiling for a
222    /// run where one item could commit an unbounded number of events, and it is
223    /// never silent: an observation that hit it is marked truncated, and every
224    /// assertion that reads the event list then fails loudly instead of passing
225    /// over the half of the ledger nobody read.
226    pub max_observed_events: Option<usize>,
227    /// Keep the turn's own words on each sample of the report.
228    ///
229    /// Off by default, and the default is the careful one: the reply is the
230    /// only model-authored prose a run produces, it is the one field that can
231    /// carry whatever a person typed, and a report is a file that gets attached
232    /// to things. A measurement does not need it — every assertion reads
233    /// storage, and the judge is handed the text directly whether this is on or
234    /// off.
235    ///
236    /// Turn it on to CURATE. Writing the expectations of an item means deciding
237    /// what the right reply would have been, and that cannot be done from a
238    /// signature: two runs whose effects are identical can differ entirely in
239    /// whether the assistant asked the question the turn needed. Reading them
240    /// is how a corpus of scenes carried over from another engine gets its
241    /// assertions.
242    pub record_answers: bool,
243}
244
245impl Default for ExecutionConfig {
246    fn default() -> Self {
247        Self {
248            samples_per_item: 1,
249            stop_after_failures: None,
250            sample_concurrency: 1,
251            max_observed_events: None,
252            record_answers: false,
253        }
254    }
255}
256
257/// How the judge is polled.
258#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
259#[serde(deny_unknown_fields, default)]
260pub struct JudgingConfig {
261    /// How many judge opinions are collected about **one sample**.
262    ///
263    /// Votes reduce the judge's own variance. They say nothing about the agent
264    /// under test, so they never enter a deterministic pass rate. An odd number
265    /// avoids ties; on a tie the lower score wins, because a judge harness that
266    /// rounds up in its own favour is not a measurement.
267    pub votes_per_sample: u32,
268    /// Judge every sample, or only the first one of each item. Judging one
269    /// sample per item is the usual choice: the judge is there to grade
270    /// language, and language costs money to grade.
271    pub judge_every_sample: bool,
272}
273
274impl Default for JudgingConfig {
275    fn default() -> Self {
276        Self {
277            votes_per_sample: 1,
278            judge_every_sample: false,
279        }
280    }
281}
282
283/// Which items of a suite a run selects.
284#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
285#[serde(deny_unknown_fields, default)]
286pub struct SelectionConfig {
287    /// Run only items carrying at least one of these tags. Empty means every
288    /// item is a candidate.
289    pub include_tags: Vec<Tag>,
290    /// Skip items carrying any of these tags, even when `include_tags` selected
291    /// them.
292    pub exclude_tags: Vec<Tag>,
293    /// Run only these items, by identifier. Applied after the tag filters.
294    pub items: Vec<ItemId>,
295}
296
297impl SelectionConfig {
298    /// Returns `true` when nothing is filtered and every item runs.
299    #[must_use]
300    pub fn is_empty(&self) -> bool {
301        self.include_tags.is_empty() && self.exclude_tags.is_empty() && self.items.is_empty()
302    }
303
304    /// Returns `true` when an item with these tags and identifier runs.
305    #[must_use]
306    pub fn selects(&self, id: &ItemId, tags: &[Tag]) -> bool {
307        if !self.include_tags.is_empty() && !self.include_tags.iter().any(|t| tags.contains(t)) {
308            return false;
309        }
310        if self.exclude_tags.iter().any(|t| tags.contains(t)) {
311            return false;
312        }
313        self.items.is_empty() || self.items.contains(id)
314    }
315}
316
317/// Why a configuration could not be loaded or run.
318#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)]
319#[non_exhaustive]
320pub enum ConfigError {
321    /// The file could not be read.
322    #[error("evaluation configuration at {path} could not be read: {message}")]
323    Read {
324        /// The path that was tried.
325        path: PathBuf,
326        /// What the filesystem said.
327        message: String,
328    },
329    /// The document is not a configuration this crate understands.
330    #[error("evaluation configuration could not be parsed: {message}")]
331    Parse {
332        /// What the parser said, including the unknown key when there was one.
333        message: String,
334    },
335    /// `samples_per_item` is zero.
336    #[error("samples_per_item must be at least 1: an item that never runs has no result")]
337    ZeroSamples,
338    /// `votes_per_sample` is zero.
339    #[error("votes_per_sample must be at least 1: a judge with no votes has no verdict")]
340    ZeroVotes,
341}
342
343#[cfg(test)]
344mod tests {
345    use super::*;
346
347    #[test]
348    fn the_specification_snippet_parses() {
349        let config = EvalConfig::from_toml_str(
350            "[execution]\nsamples_per_item = 10\n\n[judging]\nvotes_per_sample = 3\n",
351        )
352        .unwrap();
353        assert_eq!(config.execution.samples_per_item, 10);
354        assert_eq!(config.judging.votes_per_sample, 3);
355    }
356
357    #[test]
358    fn an_unknown_key_is_refused_rather_than_ignored() {
359        let error =
360            EvalConfig::from_toml_str("[execution]\nsamples = 10\n").expect_err("unknown key");
361        assert!(matches!(error, ConfigError::Parse { .. }), "{error}");
362    }
363
364    #[test]
365    fn votes_cannot_stand_in_for_samples() {
366        // The type system cannot stop someone setting one and meaning the
367        // other, but the validation can at least refuse the degenerate values.
368        assert!(matches!(
369            EvalConfig::default().with_samples_per_item(0).validate(),
370            Err(ConfigError::ZeroSamples)
371        ));
372        assert!(matches!(
373            EvalConfig::default().with_votes_per_sample(0).validate(),
374            Err(ConfigError::ZeroVotes)
375        ));
376    }
377
378    #[test]
379    fn selection_filters_by_tag_then_by_identifier() {
380        let selection = SelectionConfig {
381            include_tags: vec![Tag::new("trip")],
382            exclude_tags: vec![Tag::new("slow")],
383            items: Vec::new(),
384        };
385        assert!(selection.selects(&ItemId::new("a"), &[Tag::new("trip")]));
386        assert!(!selection.selects(&ItemId::new("a"), &[Tag::new("traveler")]));
387        assert!(!selection.selects(&ItemId::new("a"), &[Tag::new("trip"), Tag::new("slow")]));
388    }
389}