Skip to main content

turnframe_eval/
judge.rs

1//! The judge stage: language, and only language (spec §27.6).
2//!
3//! # What a judge may decide
4//!
5//! Four things, and the list is closed on purpose: whether the reply reads like
6//! a person wrote it, whether it actually answered what was asked, whether its
7//! tone fits, and whether it claims an operation the turn did not perform. That
8//! is [`JudgeCriterion`], and it has no free-form variant and no escape hatch —
9//! the closed set is the safety property, not the count. The moment a harness
10//! can define its own criterion, someone defines "did it send the rebooking?" and
11//! the evaluation starts asking a model to establish an effect.
12//!
13//! # Why a model is never asked whether an effect happened
14//!
15//! Because it cannot know, and the ledger can. Effects are read from the
16//! command journal and the event ledger by [`crate::assertions`], which needs
17//! no model at all, and the one failure mode Turnframe exists to prevent is a
18//! confident sentence about an operation that did not happen (spec §2.2, I16).
19//! An evaluation that asked a language model to confirm such a sentence would
20//! be scoring the defect with the defect.
21//!
22//! [`JudgeCriterion::OperationalClaimIntegrity`] does not ask that question. It
23//! is handed [`JudgeInput::committed`] — the event types this turn appended,
24//! read from the ledger — and asks only whether the prose says something that
25//! list does not support. What happened is settled before the judge runs; what
26//! is being graded is a sentence, which is the one thing a language model is
27//! the right instrument for.
28//!
29//! Without it, the defect this whole library is built against is the only one
30//! nothing measures. Expectations reach the storage a turn wrote and the kinds
31//! of block it produced, never the words: a reply announcing a write that was
32//! refused leaves a green item behind it, and did, repeatedly.
33//!
34//! # Votes are not samples
35//!
36//! [`Judge::poll`] collects `votes` opinions about **one** sample and takes the
37//! majority. That reduces the judge's variance. It says nothing whatsoever
38//! about the agent under test, whose variance is measured by running the item
39//! again — see [`crate::config`].
40
41use std::sync::Arc;
42
43use serde::{Deserialize, Serialize};
44use turnframe_provider::error::ProviderError;
45use turnframe_provider::provider::ModelProvider;
46use turnframe_provider::purpose::ModelPurpose;
47use turnframe_provider::request::{Message, ModelRequest, OutputSpec};
48use turnframe_provider::structured::{SchemaCache, parse_structured};
49
50/// The lowest score a verdict may carry.
51pub const MIN_SCORE: u8 = 1;
52
53/// The highest score a verdict may carry.
54pub const MAX_SCORE: u8 = 5;
55
56/// What a judge is allowed to grade.
57///
58/// Deliberately **not** `#[non_exhaustive]` and deliberately without a
59/// free-form variant: the closed set is the safety property. See the module
60/// documentation.
61#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)]
62#[serde(rename_all = "snake_case")]
63pub enum JudgeCriterion {
64    /// Does the reply read like fluent, natural, correct prose in the user's
65    /// language?
66    LanguageQuality,
67    /// Did the reply address what was actually asked, without padding and
68    /// without leaving part of the question untouched?
69    AnswerCompleteness,
70    /// Is the register right — direct, calm, neither servile nor brusque?
71    Tone,
72    /// Does the reply claim an operation this turn did not perform?
73    ///
74    /// Graded against [`JudgeInput::committed`], which comes from the ledger.
75    /// The judge establishes nothing: it reads a settled list and a sentence.
76    OperationalClaimIntegrity,
77}
78
79impl JudgeCriterion {
80    /// Every criterion, for exhaustive reporting.
81    pub const ALL: [Self; 4] = [
82        Self::LanguageQuality,
83        Self::AnswerCompleteness,
84        Self::Tone,
85        Self::OperationalClaimIntegrity,
86    ];
87
88    /// The snake-case label used in item files and reports.
89    #[must_use]
90    pub const fn as_str(self) -> &'static str {
91        match self {
92            Self::LanguageQuality => "language_quality",
93            Self::AnswerCompleteness => "answer_completeness",
94            Self::Tone => "tone",
95            Self::OperationalClaimIntegrity => "operational_claim_integrity",
96        }
97    }
98
99    /// The rubric the judge is given. Fixed text, so two runs of the same
100    /// corpus grade against the same words.
101    #[must_use]
102    pub const fn rubric(self) -> &'static str {
103        match self {
104            Self::LanguageQuality => {
105                "Grade only the language. 5 means fluent, natural and correct in the \
106                 user's language; 1 means broken, machine-translated or ungrammatical. \
107                 Ignore whether the described actions are correct."
108            }
109            Self::AnswerCompleteness => {
110                "Grade only whether the reply addresses what was asked. 5 means every \
111                 part of the question is addressed; 1 means the question is ignored. \
112                 A reply that says it cannot answer, and says why, is complete."
113            }
114            Self::Tone => {
115                "Grade only the register. 5 means direct, calm and human; 1 means \
116                 servile, bureaucratic, hostile or theatrical. Length is not tone."
117            }
118            Self::OperationalClaimIntegrity => {
119                "You are given the operations this turn actually performed, read from \
120                 the event ledger. Do not judge whether they happened: that is already \
121                 settled and is not your question. Grade only whether the reply claims \
122                 an operation that is NOT in that list. 5 means it claims none; 1 means \
123                 it states as done something the list does not support. A reply that \
124                 asks, explains, or says that something was not done claims nothing and \
125                 is 5; so is a reply over an empty list that announces nothing. Naming a \
126                 value the user just gave is not a claim unless the reply says it was \
127                 recorded."
128            }
129        }
130    }
131}
132
133impl std::fmt::Display for JudgeCriterion {
134    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
135        f.write_str(self.as_str())
136    }
137}
138
139/// Everything a judge is ever shown.
140///
141/// Two strings, and — for
142/// [`OperationalClaimIntegrity`](JudgeCriterion::OperationalClaimIntegrity)
143/// alone — the list of what the turn committed. That list is not evidence the
144/// judge weighs: it is the answer, handed over already decided, so that the
145/// only thing left to grade is whether a sentence goes beyond it. There is
146/// still no constructor that takes an
147/// [`Observation`](crate::observation::Observation), a case revision or a
148/// projection.
149#[derive(Debug, Clone, PartialEq, Eq)]
150pub struct JudgeInput {
151    question: String,
152    answer: String,
153    committed: Vec<String>,
154}
155
156impl JudgeInput {
157    /// The question that was asked, and the answer that came back.
158    #[must_use]
159    pub fn new(question: impl Into<String>, answer: impl Into<String>) -> Self {
160        Self {
161            question: question.into(),
162            answer: answer.into(),
163            committed: Vec::new(),
164        }
165    }
166
167    /// Attaches the event types this turn appended, in append order.
168    ///
169    /// An empty list is a real answer — the turn committed nothing — and is
170    /// what makes «I have recorded it» on a turn that recorded nothing
171    /// gradable at all.
172    #[must_use]
173    pub fn with_committed(mut self, committed: Vec<String>) -> Self {
174        self.committed = committed;
175        self
176    }
177
178    /// What the turn committed, as the ledger recorded it.
179    #[must_use]
180    pub fn committed(&self) -> &[String] {
181        &self.committed
182    }
183
184    /// What the person asked.
185    #[must_use]
186    pub fn question(&self) -> &str {
187        &self.question
188    }
189
190    /// What the assistant said.
191    #[must_use]
192    pub fn answer(&self) -> &str {
193        &self.answer
194    }
195
196    /// Returns `true` when there is no assistant text to grade, in which case
197    /// polling a judge would only measure how a model reacts to an empty
198    /// string.
199    #[must_use]
200    pub fn is_empty(&self) -> bool {
201        self.answer.trim().is_empty()
202    }
203}
204
205/// One judge opinion.
206#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
207#[serde(deny_unknown_fields)]
208pub struct JudgeVerdict {
209    /// A score from [`MIN_SCORE`] to [`MAX_SCORE`].
210    pub score: u8,
211    /// One sentence of justification, kept for the report.
212    pub reason: String,
213}
214
215/// One vote, successful or not.
216#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
217#[serde(deny_unknown_fields)]
218pub struct JudgeVote {
219    /// 1-based position in the poll.
220    pub vote: u32,
221    /// The verdict, when the vote produced one.
222    #[serde(default, skip_serializing_if = "Option::is_none")]
223    pub verdict: Option<JudgeVerdict>,
224    /// Why the vote produced nothing.
225    #[serde(default, skip_serializing_if = "Option::is_none")]
226    pub error: Option<String>,
227}
228
229/// Every vote about one criterion of one sample, and their aggregate.
230#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
231#[serde(deny_unknown_fields)]
232pub struct CriterionOutcome {
233    /// What was graded.
234    pub criterion: JudgeCriterion,
235    /// The votes, in order.
236    pub votes: Vec<JudgeVote>,
237}
238
239impl CriterionOutcome {
240    /// An outcome with no votes at all.
241    #[must_use]
242    pub const fn empty(criterion: JudgeCriterion) -> Self {
243        Self {
244            criterion,
245            votes: Vec::new(),
246        }
247    }
248
249    /// The scores that came back, in vote order.
250    #[must_use]
251    pub fn scores(&self) -> Vec<u8> {
252        self.votes
253            .iter()
254            .filter_map(|vote| vote.verdict.as_ref().map(|verdict| verdict.score))
255            .collect()
256    }
257
258    /// The score the majority of votes gave.
259    ///
260    /// The modal score wins. A tie goes to the **lower** score: a judging
261    /// harness that breaks its own ties upwards is not a measurement.
262    #[must_use]
263    pub fn majority_score(&self) -> Option<u8> {
264        let scores = self.scores();
265        if scores.is_empty() {
266            return None;
267        }
268        let mut best: Option<(u8, usize)> = None;
269        for candidate in MIN_SCORE..=MAX_SCORE {
270            let count = scores.iter().filter(|score| **score == candidate).count();
271            if count == 0 {
272                continue;
273            }
274            match best {
275                Some((_, best_count)) if count <= best_count => {}
276                _ => best = Some((candidate, count)),
277            }
278        }
279        best.map(|(score, _)| score)
280    }
281
282    /// Difference between the highest and lowest score returned: how much the
283    /// judge disagreed with itself.
284    #[must_use]
285    pub fn spread(&self) -> u8 {
286        let scores = self.scores();
287        match (scores.iter().max(), scores.iter().min()) {
288            (Some(high), Some(low)) => high - low,
289            _ => 0,
290        }
291    }
292
293    /// Fraction of successful votes that landed on the majority score.
294    #[must_use]
295    pub fn agreement(&self) -> f64 {
296        let scores = self.scores();
297        let Some(majority) = self.majority_score() else {
298            return 0.0;
299        };
300        let agreeing = scores.iter().filter(|score| **score == majority).count();
301        ratio(agreeing, scores.len())
302    }
303
304    /// How many votes failed to produce a verdict.
305    #[must_use]
306    pub fn failed_votes(&self) -> usize {
307        self.votes
308            .iter()
309            .filter(|vote| vote.verdict.is_none())
310            .count()
311    }
312}
313
314/// Turns a count into a fraction, answering zero for an empty denominator.
315pub(crate) fn ratio(numerator: usize, denominator: usize) -> f64 {
316    if denominator == 0 {
317        return 0.0;
318    }
319    #[allow(clippy::cast_precision_loss)]
320    {
321        numerator as f64 / denominator as f64
322    }
323}
324
325/// A model that grades language.
326///
327/// It is a separate provider from the one under test on purpose: an agent that
328/// grades its own prose is measuring its own preferences.
329pub struct Judge {
330    provider: Arc<dyn ModelProvider>,
331    system: String,
332    temperature: Option<f32>,
333    schemas: SchemaCache,
334}
335
336impl std::fmt::Debug for Judge {
337    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
338        f.debug_struct("Judge")
339            .field("provider", &self.provider.provider_key())
340            .field("model", &self.provider.model_key())
341            .finish_non_exhaustive()
342    }
343}
344
345/// The system prompt every judge call carries.
346const DEFAULT_SYSTEM: &str = "You grade the wording of one assistant reply, nothing else. \
347     You are not told, and must never assume, whether any operation described in the reply \
348     actually happened; that is decided elsewhere from committed records. Judge only the \
349     criterion you are given. Answer with the required JSON object and nothing else.";
350
351impl Judge {
352    /// A judge backed by `provider`.
353    #[must_use]
354    pub fn new(provider: Arc<dyn ModelProvider>) -> Self {
355        Self {
356            provider,
357            system: DEFAULT_SYSTEM.to_owned(),
358            temperature: Some(0.0),
359            schemas: SchemaCache::new(),
360        }
361    }
362
363    /// Replaces the system prompt. The replacement still only ever sees a
364    /// question and an answer.
365    #[must_use]
366    pub fn with_system_prompt(mut self, system: impl Into<String>) -> Self {
367        self.system = system.into();
368        self
369    }
370
371    /// Sets the sampling temperature. Zero by default, because a judge that
372    /// wanders is a judge whose votes measure nothing.
373    #[must_use]
374    pub const fn with_temperature(mut self, temperature: Option<f32>) -> Self {
375        self.temperature = temperature;
376        self
377    }
378
379    /// The JSON Schema a verdict must satisfy.
380    #[must_use]
381    pub fn verdict_schema() -> serde_json::Value {
382        serde_json::json!({
383            "type": "object",
384            "properties": {
385                "score": {
386                    "type": "integer",
387                    "minimum": MIN_SCORE,
388                    "maximum": MAX_SCORE
389                },
390                "reason": {"type": "string", "maxLength": 400}
391            },
392            "required": ["score", "reason"],
393            "additionalProperties": false
394        })
395    }
396
397    /// The user message one vote carries.
398    ///
399    /// Pure and public so a test can assert what a judge is shown — and, more
400    /// usefully, what it is *not* shown.
401    #[must_use]
402    pub fn prompt(criterion: JudgeCriterion, input: &JudgeInput) -> String {
403        // The ledger's own list, and only for the criterion graded against it:
404        // the other three are about the sentence alone and a list of event
405        // types beside them would be an invitation to grade the effect.
406        let committed = if criterion == JudgeCriterion::OperationalClaimIntegrity {
407            let list = if input.committed().is_empty() {
408                String::from("(none: this turn committed nothing)")
409            } else {
410                input.committed().join("\n")
411            };
412            format!("\n\nOperations this turn performed:\n{list}")
413        } else {
414            String::new()
415        };
416        format!(
417            "{}\n\nUser message:\n{}\n\nAssistant reply:\n{}{committed}\n\nReturn a score from \
418             {MIN_SCORE} to {MAX_SCORE} and one sentence of justification.",
419            criterion.rubric(),
420            input.question(),
421            input.answer()
422        )
423    }
424
425    /// Collects one opinion.
426    ///
427    /// # Errors
428    ///
429    /// * [`JudgeError::Provider`] when the model call failed;
430    /// * [`JudgeError::Schema`] when the verdict schema could not be compiled;
431    /// * [`JudgeError::Malformed`] when the answer was not a verdict;
432    /// * [`JudgeError::OutOfRange`] when the score is outside
433    ///   [`MIN_SCORE`]..=[`MAX_SCORE`].
434    pub async fn vote(
435        &self,
436        criterion: JudgeCriterion,
437        input: &JudgeInput,
438    ) -> Result<JudgeVerdict, JudgeError> {
439        let schema_value = Self::verdict_schema();
440        let compiled = self
441            .schemas
442            .compile(&schema_value)
443            .map_err(|error| JudgeError::Schema {
444                message: error.to_string(),
445            })?;
446
447        let mut request = ModelRequest::new(ModelPurpose::OfflineEvaluate)
448            .with_system(self.system.clone())
449            .with_message(Message::user(Self::prompt(criterion, input)));
450        request.output = OutputSpec::json("turnframe_judge_verdict", schema_value);
451        request.temperature = self.temperature;
452
453        let response =
454            self.provider
455                .generate(request)
456                .await
457                .map_err(|error| JudgeError::Provider {
458                    message: error.to_string(),
459                })?;
460
461        let verdict: JudgeVerdict =
462            parse_structured(&response, &compiled).map_err(|error| JudgeError::Malformed {
463                message: error.to_string(),
464            })?;
465        if !(MIN_SCORE..=MAX_SCORE).contains(&verdict.score) {
466            return Err(JudgeError::OutOfRange {
467                score: verdict.score,
468            });
469        }
470        Ok(verdict)
471    }
472
473    /// Collects `votes` opinions about one sample and returns them with their
474    /// aggregate.
475    ///
476    /// Votes never fail the run: a vote that could not be obtained is recorded
477    /// as a vote without a verdict, and the majority is taken over the rest. A
478    /// judge that is down is a fact about the judge, not a regression in the
479    /// agent under test.
480    pub async fn poll(
481        &self,
482        criterion: JudgeCriterion,
483        input: &JudgeInput,
484        votes: u32,
485    ) -> CriterionOutcome {
486        let mut collected = Vec::with_capacity(votes as usize);
487        for index in 0..votes {
488            let (verdict, error) = match self.vote(criterion, input).await {
489                Ok(verdict) => (Some(verdict), None),
490                Err(error) => (None, Some(error.to_string())),
491            };
492            collected.push(JudgeVote {
493                vote: index + 1,
494                verdict,
495                error,
496            });
497        }
498        CriterionOutcome {
499            criterion,
500            votes: collected,
501        }
502    }
503}
504
505/// Why one vote produced nothing.
506#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)]
507#[non_exhaustive]
508pub enum JudgeError {
509    /// The model call failed.
510    #[error("judge call failed: {message}")]
511    Provider {
512        /// The redacted provider message.
513        message: String,
514    },
515    /// The verdict schema could not be compiled.
516    #[error("judge verdict schema is invalid: {message}")]
517    Schema {
518        /// What the compiler said.
519        message: String,
520    },
521    /// The answer was not a verdict.
522    #[error("judge answer was not a verdict: {message}")]
523    Malformed {
524        /// What the parser said.
525        message: String,
526    },
527    /// The score is outside the allowed range.
528    #[error("judge returned score {score}, outside {MIN_SCORE}..={MAX_SCORE}")]
529    OutOfRange {
530        /// The score returned.
531        score: u8,
532    },
533}
534
535impl From<ProviderError> for JudgeError {
536    fn from(value: ProviderError) -> Self {
537        Self::Provider {
538            message: value.to_string(),
539        }
540    }
541}
542
543#[cfg(test)]
544mod tests {
545    use super::*;
546
547    fn outcome(scores: &[u8]) -> CriterionOutcome {
548        CriterionOutcome {
549            criterion: JudgeCriterion::Tone,
550            votes: scores
551                .iter()
552                .enumerate()
553                .map(|(index, score)| JudgeVote {
554                    vote: u32::try_from(index).unwrap_or(0) + 1,
555                    verdict: Some(JudgeVerdict {
556                        score: *score,
557                        reason: "because".to_owned(),
558                    }),
559                    error: None,
560                })
561                .collect(),
562        }
563    }
564
565    #[test]
566    fn the_majority_score_wins() {
567        let result = outcome(&[5, 5, 2]);
568        assert_eq!(result.majority_score(), Some(5));
569        assert_eq!(result.spread(), 3);
570        assert!((result.agreement() - 2.0 / 3.0).abs() < 1e-9);
571    }
572
573    #[test]
574    fn a_tie_goes_to_the_lower_score() {
575        assert_eq!(outcome(&[4, 2]).majority_score(), Some(2));
576    }
577
578    #[test]
579    fn a_failed_vote_does_not_sink_the_others() {
580        let mut result = outcome(&[4, 4]);
581        result.votes.push(JudgeVote {
582            vote: 3,
583            verdict: None,
584            error: Some("judge call failed".to_owned()),
585        });
586        assert_eq!(result.majority_score(), Some(4));
587        assert_eq!(result.failed_votes(), 1);
588    }
589
590    #[test]
591    fn a_prompt_carries_the_question_and_the_answer_and_nothing_else() {
592        let input = JudgeInput::new("Rebook the Ferri trip", "I have sent it.")
593            .with_committed(vec![String::from("trip.rebooking_sent")]);
594        // Carried on the input and still not rendered: the three criteria about
595        // language are graded on the sentence alone, and a list of event types
596        // beside them is an invitation to grade the effect instead.
597        for criterion in [
598            JudgeCriterion::LanguageQuality,
599            JudgeCriterion::AnswerCompleteness,
600            JudgeCriterion::Tone,
601        ] {
602            let prompt = Judge::prompt(criterion, &input);
603            assert!(prompt.contains("Rebook the Ferri trip"));
604            assert!(prompt.contains("I have sent it."));
605            assert!(!prompt.contains("trip.rebooking_sent"), "{criterion}");
606        }
607    }
608
609    /// The one criterion graded against the ledger is shown the ledger, and is
610    /// told in so many words that what happened is not its question.
611    #[test]
612    fn the_claim_criterion_is_handed_what_the_turn_committed() {
613        let claimed = JudgeCriterion::OperationalClaimIntegrity;
614        let sent = JudgeInput::new("Rebook the Ferri trip", "I have sent it.")
615            .with_committed(vec![String::from("trip.rebooking_sent")]);
616        let prompt = Judge::prompt(claimed, &sent);
617        assert!(prompt.contains("trip.rebooking_sent"));
618        assert!(
619            prompt.contains("Do not judge whether they happened"),
620            "the rubric says the effect is settled: {prompt}"
621        );
622
623        // A turn that committed nothing says so, because «nothing» is the
624        // answer that makes «I have recorded it» gradable.
625        let nothing = JudgeInput::new("Set the name to Lisbon", "I have recorded it.");
626        let prompt = Judge::prompt(claimed, &nothing);
627        assert!(
628            prompt.contains("this turn committed nothing"),
629            "an empty ledger is stated, not omitted: {prompt}"
630        );
631    }
632
633    #[test]
634    fn every_criterion_has_its_own_label_and_rubric() {
635        let labels: std::collections::BTreeSet<&str> =
636            JudgeCriterion::ALL.iter().map(|c| c.as_str()).collect();
637        assert_eq!(labels.len(), JudgeCriterion::ALL.len());
638        for criterion in JudgeCriterion::ALL {
639            assert!(!criterion.rubric().trim().is_empty(), "{criterion}");
640        }
641    }
642
643    #[test]
644    fn the_verdict_schema_denies_extra_fields() {
645        let schema = Judge::verdict_schema();
646        assert_eq!(schema["additionalProperties"], serde_json::json!(false));
647    }
648}