turnframe_eval/config.rs
1//! How a run is configured (spec §27.6).
2//!
3//! The whole point of this module is one distinction the specification insists
4//! on, and it is worth stating before any type appears:
5//!
6//! * **A sample is one execution of the agent under test.** More samples
7//! measure how much the *model* varies. Ten samples of the same item are ten
8//! chances for understanding to read something different, and the spread
9//! between them is the number a release gate cares about.
10//! * **A vote is one judge opinion about one sample.** More votes measure how
11//! much the *judge* varies. Three votes on one sample tell you nothing about
12//! the agent; they tell you whether the judge would have said the same thing
13//! twice.
14//!
15//! Averaging the two together produces a number that moves when either the
16//! model or the judge wobbles and cannot say which — which is exactly the
17//! failure §26.3 warns about. So they are two settings, they are counted
18//! separately, and they are reported separately.
19//!
20//! The file format is the one printed in the specification:
21//!
22//! ```toml
23//! [execution]
24//! samples_per_item = 10
25//!
26//! [judging]
27//! votes_per_sample = 3
28//! ```
29
30use std::path::{Path, PathBuf};
31
32use serde::{Deserialize, Serialize};
33
34use crate::corpus::{ItemId, Tag};
35
36/// Everything one evaluation run needs to know.
37///
38/// ```
39/// use turnframe_eval::config::EvalConfig;
40///
41/// let config = EvalConfig::from_toml_str(
42/// r#"
43/// [execution]
44/// samples_per_item = 10
45///
46/// [judging]
47/// votes_per_sample = 3
48/// "#,
49/// )?;
50///
51/// assert_eq!(config.execution.samples_per_item, 10);
52/// assert_eq!(config.judging.votes_per_sample, 3);
53/// config.validate()?;
54/// # Ok::<(), turnframe_eval::config::ConfigError>(())
55/// ```
56#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
57#[serde(deny_unknown_fields, default)]
58pub struct EvalConfig {
59 /// How the agent under test is exercised.
60 pub execution: ExecutionConfig,
61 /// How the judge is polled, when an item asks for one.
62 pub judging: JudgingConfig,
63 /// Which subset of a suite runs.
64 pub selection: SelectionConfig,
65}
66
67impl EvalConfig {
68 /// A run of one sample and one vote: the cheapest configuration that still
69 /// exercises every stage. Use it for a smoke run; a release gate wants the
70 /// numbers of spec §27.6 instead.
71 #[must_use]
72 pub fn single() -> Self {
73 Self::default()
74 }
75
76 /// Sets the number of executions of the agent under test per item.
77 #[must_use]
78 pub const fn with_samples_per_item(mut self, samples: u32) -> Self {
79 self.execution.samples_per_item = samples;
80 self
81 }
82
83 /// Keeps the turn's own words on each sample, for curating a corpus.
84 ///
85 /// See [`ExecutionConfig::record_answers`] for why this is off by default.
86 #[must_use]
87 pub const fn recording_answers(mut self) -> Self {
88 self.execution.record_answers = true;
89 self
90 }
91
92 /// Sets the number of judge opinions collected per *sample*.
93 #[must_use]
94 pub const fn with_votes_per_sample(mut self, votes: u32) -> Self {
95 self.judging.votes_per_sample = votes;
96 self
97 }
98
99 /// Sets how many samples may be in flight at once.
100 ///
101 /// Leave it at one for a reproducible in-memory run; raise it only for a
102 /// corpus against a real endpoint. See
103 /// [`ExecutionConfig::sample_concurrency`].
104 #[must_use]
105 pub const fn with_sample_concurrency(mut self, samples: u32) -> Self {
106 self.execution.sample_concurrency = samples;
107 self
108 }
109
110 /// Caps how many of a turn's events one observation records, loudly.
111 ///
112 /// See [`ExecutionConfig::max_observed_events`]: an observation that hits
113 /// the cap fails every assertion about events rather than passing over a
114 /// ledger it only half read.
115 #[must_use]
116 pub const fn with_max_observed_events(mut self, events: usize) -> Self {
117 self.execution.max_observed_events = Some(events);
118 self
119 }
120
121 /// Restricts the run to items carrying `tag`.
122 #[must_use]
123 pub fn including_tag(mut self, tag: impl Into<Tag>) -> Self {
124 self.selection.include_tags.push(tag.into());
125 self
126 }
127
128 /// Excludes items carrying `tag`, whatever else selects them.
129 #[must_use]
130 pub fn excluding_tag(mut self, tag: impl Into<Tag>) -> Self {
131 self.selection.exclude_tags.push(tag.into());
132 self
133 }
134
135 /// Parses a configuration from TOML.
136 ///
137 /// # Errors
138 ///
139 /// [`ConfigError::Parse`] when the document is malformed or carries a key
140 /// this crate does not understand.
141 pub fn from_toml_str(source: &str) -> Result<Self, ConfigError> {
142 toml::from_str(source).map_err(|error| ConfigError::Parse {
143 message: error.to_string(),
144 })
145 }
146
147 /// Reads a configuration from a TOML file.
148 ///
149 /// # Errors
150 ///
151 /// * [`ConfigError::Read`] when the file cannot be read;
152 /// * [`ConfigError::Parse`] when its contents are not a configuration.
153 pub fn load(path: impl AsRef<Path>) -> Result<Self, ConfigError> {
154 let path = path.as_ref();
155 let source = std::fs::read_to_string(path).map_err(|error| ConfigError::Read {
156 path: path.to_path_buf(),
157 message: error.to_string(),
158 })?;
159 Self::from_toml_str(&source)
160 }
161
162 /// Checks the configuration is one this crate will run.
163 ///
164 /// # Errors
165 ///
166 /// * [`ConfigError::ZeroSamples`] — an item that never executes has no
167 /// result, not an empty one;
168 /// * [`ConfigError::ZeroVotes`] — a judge with no votes has no verdict.
169 pub const fn validate(&self) -> Result<(), ConfigError> {
170 if self.execution.samples_per_item == 0 {
171 return Err(ConfigError::ZeroSamples);
172 }
173 if self.judging.votes_per_sample == 0 {
174 return Err(ConfigError::ZeroVotes);
175 }
176 Ok(())
177 }
178}
179
180/// How the agent under test is exercised.
181///
182/// `#[non_exhaustive]`: this 0.1 grows a field whenever a run needs to say
183/// something new, and each one would be source-breaking for anybody constructing
184/// this with a struct literal — `#[serde(default)]` keeps a FILE readable and
185/// does nothing for Rust. Build it from [`Default`] and the `with_*` methods,
186/// which is what the next added field will not break.
187#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
188#[serde(deny_unknown_fields, default)]
189#[non_exhaustive]
190pub struct ExecutionConfig {
191 /// How many times each item runs end to end.
192 ///
193 /// This is the only knob that measures **model** variance. Raising it makes
194 /// a flaky item visible; raising [`JudgingConfig::votes_per_sample`] never
195 /// will.
196 pub samples_per_item: u32,
197 /// Stop the whole run after this many samples have failed their
198 /// deterministic assertions. `None` runs the corpus to the end, which is
199 /// what a nightly job wants; a pull-request gate may prefer to stop early.
200 pub stop_after_failures: Option<u32>,
201 /// How many samples may be in flight at once.
202 ///
203 /// One — the default — runs the corpus strictly one sample at a time, which
204 /// is what a reproducible in-memory run wants: the harness sees the samples
205 /// in index order, and a scripted provider keyed on call order sees exactly
206 /// the sequence it was written for.
207 ///
208 /// Raising it is for a corpus against a **real endpoint**, where a hundred
209 /// samples in series is hours of waiting on a network. What it changes is
210 /// the *execution* order: samples start and finish interleaved, so a
211 /// harness that shares anything between them — a counter, a queue of
212 /// scripted answers, a rate limit — will see a different order every run.
213 /// The report does not move: samples are still reported in index order,
214 /// and the same set of results comes back whatever this is set to. Zero is
215 /// read as one.
216 pub sample_concurrency: u32,
217 /// How many of a turn's events one observation may record before it stops
218 /// and says so.
219 ///
220 /// `None` — the default — reads the ledger to the end, paging the journal
221 /// by its sequence cursor. A `Some(limit)` is a deliberate ceiling for a
222 /// run where one item could commit an unbounded number of events, and it is
223 /// never silent: an observation that hit it is marked truncated, and every
224 /// assertion that reads the event list then fails loudly instead of passing
225 /// over the half of the ledger nobody read.
226 pub max_observed_events: Option<usize>,
227 /// Keep the turn's own words on each sample of the report.
228 ///
229 /// Off by default, and the default is the careful one: the reply is the
230 /// only model-authored prose a run produces, it is the one field that can
231 /// carry whatever a person typed, and a report is a file that gets attached
232 /// to things. A measurement does not need it — every assertion reads
233 /// storage, and the judge is handed the text directly whether this is on or
234 /// off.
235 ///
236 /// Turn it on to CURATE. Writing the expectations of an item means deciding
237 /// what the right reply would have been, and that cannot be done from a
238 /// signature: two runs whose effects are identical can differ entirely in
239 /// whether the assistant asked the question the turn needed. Reading them
240 /// is how a corpus of scenes carried over from another engine gets its
241 /// assertions.
242 pub record_answers: bool,
243}
244
245impl Default for ExecutionConfig {
246 fn default() -> Self {
247 Self {
248 samples_per_item: 1,
249 stop_after_failures: None,
250 sample_concurrency: 1,
251 max_observed_events: None,
252 record_answers: false,
253 }
254 }
255}
256
257/// How the judge is polled.
258#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
259#[serde(deny_unknown_fields, default)]
260pub struct JudgingConfig {
261 /// How many judge opinions are collected about **one sample**.
262 ///
263 /// Votes reduce the judge's own variance. They say nothing about the agent
264 /// under test, so they never enter a deterministic pass rate. An odd number
265 /// avoids ties; on a tie the lower score wins, because a judge harness that
266 /// rounds up in its own favour is not a measurement.
267 pub votes_per_sample: u32,
268 /// Judge every sample, or only the first one of each item. Judging one
269 /// sample per item is the usual choice: the judge is there to grade
270 /// language, and language costs money to grade.
271 pub judge_every_sample: bool,
272}
273
274impl Default for JudgingConfig {
275 fn default() -> Self {
276 Self {
277 votes_per_sample: 1,
278 judge_every_sample: false,
279 }
280 }
281}
282
283/// Which items of a suite a run selects.
284#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
285#[serde(deny_unknown_fields, default)]
286pub struct SelectionConfig {
287 /// Run only items carrying at least one of these tags. Empty means every
288 /// item is a candidate.
289 pub include_tags: Vec<Tag>,
290 /// Skip items carrying any of these tags, even when `include_tags` selected
291 /// them.
292 pub exclude_tags: Vec<Tag>,
293 /// Run only these items, by identifier. Applied after the tag filters.
294 pub items: Vec<ItemId>,
295}
296
297impl SelectionConfig {
298 /// Returns `true` when nothing is filtered and every item runs.
299 #[must_use]
300 pub fn is_empty(&self) -> bool {
301 self.include_tags.is_empty() && self.exclude_tags.is_empty() && self.items.is_empty()
302 }
303
304 /// Returns `true` when an item with these tags and identifier runs.
305 #[must_use]
306 pub fn selects(&self, id: &ItemId, tags: &[Tag]) -> bool {
307 if !self.include_tags.is_empty() && !self.include_tags.iter().any(|t| tags.contains(t)) {
308 return false;
309 }
310 if self.exclude_tags.iter().any(|t| tags.contains(t)) {
311 return false;
312 }
313 self.items.is_empty() || self.items.contains(id)
314 }
315}
316
317/// Why a configuration could not be loaded or run.
318#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)]
319#[non_exhaustive]
320pub enum ConfigError {
321 /// The file could not be read.
322 #[error("evaluation configuration at {path} could not be read: {message}")]
323 Read {
324 /// The path that was tried.
325 path: PathBuf,
326 /// What the filesystem said.
327 message: String,
328 },
329 /// The document is not a configuration this crate understands.
330 #[error("evaluation configuration could not be parsed: {message}")]
331 Parse {
332 /// What the parser said, including the unknown key when there was one.
333 message: String,
334 },
335 /// `samples_per_item` is zero.
336 #[error("samples_per_item must be at least 1: an item that never runs has no result")]
337 ZeroSamples,
338 /// `votes_per_sample` is zero.
339 #[error("votes_per_sample must be at least 1: a judge with no votes has no verdict")]
340 ZeroVotes,
341}
342
343#[cfg(test)]
344mod tests {
345 use super::*;
346
347 #[test]
348 fn the_specification_snippet_parses() {
349 let config = EvalConfig::from_toml_str(
350 "[execution]\nsamples_per_item = 10\n\n[judging]\nvotes_per_sample = 3\n",
351 )
352 .unwrap();
353 assert_eq!(config.execution.samples_per_item, 10);
354 assert_eq!(config.judging.votes_per_sample, 3);
355 }
356
357 #[test]
358 fn an_unknown_key_is_refused_rather_than_ignored() {
359 let error =
360 EvalConfig::from_toml_str("[execution]\nsamples = 10\n").expect_err("unknown key");
361 assert!(matches!(error, ConfigError::Parse { .. }), "{error}");
362 }
363
364 #[test]
365 fn votes_cannot_stand_in_for_samples() {
366 // The type system cannot stop someone setting one and meaning the
367 // other, but the validation can at least refuse the degenerate values.
368 assert!(matches!(
369 EvalConfig::default().with_samples_per_item(0).validate(),
370 Err(ConfigError::ZeroSamples)
371 ));
372 assert!(matches!(
373 EvalConfig::default().with_votes_per_sample(0).validate(),
374 Err(ConfigError::ZeroVotes)
375 ));
376 }
377
378 #[test]
379 fn selection_filters_by_tag_then_by_identifier() {
380 let selection = SelectionConfig {
381 include_tags: vec![Tag::new("trip")],
382 exclude_tags: vec![Tag::new("slow")],
383 items: Vec::new(),
384 };
385 assert!(selection.selects(&ItemId::new("a"), &[Tag::new("trip")]));
386 assert!(!selection.selects(&ItemId::new("a"), &[Tag::new("traveler")]));
387 assert!(!selection.selects(&ItemId::new("a"), &[Tag::new("trip"), Tag::new("slow")]));
388 }
389}