Skip to main content

areev_loop/
eval.rs

1//! Evalset runs — the one reader for the `evalset:<hash> mg:eval_run`
2//! summaries that `areev eval run` journals into `agent:harness`.
3//!
4//! Two edges consume these, and they deliberately share this code:
5//!
6//! - the **gating** edge (`areev loop apply --gating-run <id>`), which names one
7//!   run and reads its numbers rather than trusting the command line, and
8//! - the **outcome** edge (`measure_metric`'s `evalset:` kind), which re-reads
9//!   the newest run at each checkpoint.
10//!
11//! Sharing matters because the alternative is two parsers of the same JSON
12//! drifting apart, so a rule could be *admitted* on one reading of an evalset
13//! and *judged* on another.
14
15use crate::error::Result;
16use crate::substrate::{ReadOpts, SubstrateRead};
17use serde_json::Value;
18
19/// The relation an eval-run summary is recorded under.
20pub const EVAL_RUN_RELATION: &str = "mg:eval_run";
21
22/// The namespace those summaries live in.
23pub const HARNESS_NS: &str = "agent:harness";
24
25/// One recorded execution of an evalset.
26#[derive(Debug, Clone)]
27pub struct EvalRun {
28    /// The `eval-` run id the cases were journaled under.
29    pub run_id: String,
30    pub passed: u64,
31    pub failed: u64,
32    /// The whole summary object, so host-defined fields (`category_accuracy`,
33    /// …) are reachable without this module having to know them.
34    pub summary: serde_json::Map<String, Value>,
35    /// When the summary grain was written.
36    pub recorded_ms: i64,
37    /// What the run spent, when the runtime journaled it: the `run_outcome`
38    /// Observation `areev run` writes for the same run id. Read only for a
39    /// cost key the summary does not carry, so an eval run that IS an
40    /// `areev run` run quotes one spend to both the Verify gate and the
41    /// `run_outcome` analyzer.
42    pub spend: Option<RunSpend>,
43}
44
45/// The spent figures of a terminal `run_outcome` Observation
46/// (`spent_input_tokens` … `spent_wall_ms`), integers as the runtime wrote
47/// them.
48#[derive(Debug, Clone, Copy, PartialEq, Eq)]
49pub struct RunSpend {
50    pub input_tokens: u64,
51    pub output_tokens: u64,
52    pub usd_micros: u64,
53    pub wall_ms: u64,
54}
55
56/// The cost fields `run_value` promotes, beside the quality ones. `tokens`
57/// is input + output; `usd` is `usd_micros / 1e6`; `cost_per_pass` is
58/// `usd / passed`, undefined when nothing passed.
59pub const COST_FIELDS: [&str; 5] = ["effects", "tokens", "usd", "wall_ms", "cost_per_pass"];
60
61impl EvalRun {
62    /// A numeric field of the summary. `passed`/`failed` are promoted to typed
63    /// fields but stay readable here too, so a metric string may name any of
64    /// them uniformly.
65    pub fn field(&self, name: &str) -> Option<f64> {
66        self.summary.get(name).and_then(Value::as_f64)
67    }
68
69    /// Cases that ran. `0` when the summary records neither.
70    pub fn total(&self) -> u64 {
71        self.passed.saturating_add(self.failed)
72    }
73
74    /// An integer cost key, fail-closed: a summary that carries the key as
75    /// anything but a non-negative integer makes the cost NOT measurable —
76    /// never zero, and never the runtime's figure either, because a present
77    /// wrong value is a malformed record, not a missing one. A key the
78    /// summary does not carry at all falls back to the runtime's spend.
79    fn cost_key(&self, key: &str, from_spend: impl Fn(&RunSpend) -> u64) -> Option<u64> {
80        match self.summary.get(key) {
81            Some(v) => v.as_u64(),
82            None => self.spend.as_ref().map(from_spend),
83        }
84    }
85
86    /// Tool calls the cases made. Summary key `effects` only — the runtime's
87    /// Observation counts supersteps, not effects, so there is no fallback.
88    pub fn effects(&self) -> Option<u64> {
89        self.summary.get("effects").and_then(Value::as_u64)
90    }
91
92    /// Input + output tokens; both keys must be integers.
93    pub fn tokens(&self) -> Option<u64> {
94        let i = self.cost_key("input_tokens", |s| s.input_tokens)?;
95        let o = self.cost_key("output_tokens", |s| s.output_tokens)?;
96        Some(i.saturating_add(o))
97    }
98
99    pub fn usd_micros(&self) -> Option<u64> {
100        self.cost_key("usd_micros", |s| s.usd_micros)
101    }
102
103    pub fn wall_ms(&self) -> Option<u64> {
104        self.cost_key("wall_ms", |s| s.wall_ms)
105    }
106}
107
108/// Every recorded run of `evalset_hash`, oldest first.
109///
110/// `since_ms` bounds the scan to summaries written at or after a moment —
111/// which is what keeps an outcome honest: a run journaled *before* a
112/// recommendation was applied cannot be evidence of what applying it did.
113pub fn eval_runs<S: SubstrateRead + ?Sized>(
114    sub: &S,
115    evalset_hash: &str,
116    since_ms: Option<i64>,
117) -> Result<Vec<EvalRun>> {
118    let subject = format!("evalset:{evalset_hash}");
119    let facts = sub.grains_of_type(
120        crate::model::grain_type::FACT,
121        Some(HARNESS_NS),
122        ReadOpts { live_only: true, since_ms },
123    )?;
124    let spends = run_spends(sub)?;
125    let mut out: Vec<EvalRun> = facts
126        .iter()
127        .filter(|f| f.str_field("relation") == Some(EVAL_RUN_RELATION))
128        .filter(|f| f.str_field("subject") == Some(subject.as_str()))
129        .filter_map(|f| {
130            let obj = f.str_field("object")?;
131            let Ok(Value::Object(summary)) = serde_json::from_str::<Value>(obj) else {
132                // A summary we cannot parse is skipped, never guessed at: a
133                // fabricated number here would become a receipt.
134                return None;
135            };
136            // `areev eval run` always writes all three, so a summary missing
137            // one is malformed. Dropping it rather than defaulting keeps every
138            // consumer fail-CLOSED: at the apply gate an absent `failed` must
139            // never read as "zero failures", and an outcome must never score
140            // against numbers nobody recorded.
141            let run_id = summary.get("run_id").and_then(Value::as_str)?.to_string();
142            Some(EvalRun {
143                spend: spends.get(&run_id).copied(),
144                run_id,
145                passed: summary.get("passed").and_then(Value::as_u64)?,
146                failed: summary.get("failed").and_then(Value::as_u64)?,
147                recorded_ms: f.created_at_ms,
148                summary,
149            })
150        })
151        .collect();
152    // Recording order, with the hash as a deterministic tiebreak for two
153    // summaries written in the same millisecond.
154    out.sort_by(|a, b| {
155        a.recorded_ms
156            .cmp(&b.recorded_ms)
157            .then_with(|| a.run_id.cmp(&b.run_id))
158    });
159    Ok(out)
160}
161
162/// The spent figures of every terminal `run_outcome` Observation, by run id
163/// — the runtime's own record, read here so a harness that journals an
164/// evalset run under the run id it executed does not have to copy the
165/// numbers (and cannot copy them wrong). All four keys must be integers, or
166/// the run has no spend here.
167fn run_spends<S: SubstrateRead + ?Sized>(sub: &S) -> Result<std::collections::BTreeMap<String, RunSpend>> {
168    let obs = match sub.grains_of_type(
169        crate::model::grain_type::OBSERVATION,
170        Some(HARNESS_NS),
171        ReadOpts { live_only: true, since_ms: None },
172    ) {
173        Ok(rows) => rows,
174        // No harness observations / no read grant: no spend — the summary's
175        // own keys still work, and a missing cost stays not measurable.
176        Err(_) => return Ok(Default::default()),
177    };
178    let mut out = std::collections::BTreeMap::new();
179    for g in &obs {
180        if g.str_field("observation_kind") != Some("run_outcome") {
181            continue;
182        }
183        let Some(run_id) = g.str_field("run_id") else { continue };
184        let int = |k: &str| g.fields.get(k).and_then(Value::as_u64);
185        let (Some(i), Some(o), Some(u), Some(w)) = (
186            int("spent_input_tokens"),
187            int("spent_output_tokens"),
188            int("spent_usd_micros"),
189            int("spent_wall_ms"),
190        ) else {
191            continue;
192        };
193        out.insert(
194            run_id.to_string(),
195            RunSpend { input_tokens: i, output_tokens: o, usd_micros: u, wall_ms: w },
196        );
197    }
198    Ok(out)
199}
200
201/// The newest recorded run of `evalset_hash` at or after `since_ms`.
202pub fn newest_eval_run<S: SubstrateRead + ?Sized>(
203    sub: &S,
204    evalset_hash: &str,
205    since_ms: Option<i64>,
206) -> Result<Option<EvalRun>> {
207    Ok(eval_runs(sub, evalset_hash, since_ms)?.pop())
208}
209
210/// The value a metric field takes on one run. `failed`/`passed`/`total` are
211/// promoted so a metric can be written against any evalset without the host
212/// having to add fields; `error_rate` is derived (undefined, not zero, when
213/// no case ran); the cost fields (`COST_FIELDS`) read the integer cost keys
214/// fail-closed, with `cost_per_pass` undefined — not zero, not a division —
215/// when nothing passed; anything else is read from the summary the host did
216/// write. The proposal-time baseline and every later measurement go through
217/// this one reader, so a rule cannot be admitted on one reading of a run and
218/// judged on another.
219pub fn run_value(run: &EvalRun, field: &str) -> Option<f64> {
220    match field {
221        "failed" => Some(run.failed as f64),
222        "passed" => Some(run.passed as f64),
223        "total" => Some(run.total() as f64),
224        "error_rate" => match run.total() {
225            0 => None,
226            t => Some(run.failed as f64 / t as f64),
227        },
228        "effects" => run.effects().map(|n| n as f64),
229        "tokens" => run.tokens().map(|n| n as f64),
230        "usd" => run.usd_micros().map(|n| n as f64 / 1e6),
231        "wall_ms" => run.wall_ms().map(|n| n as f64),
232        "cost_per_pass" => match run.passed {
233            0 => None,
234            p => run.usd_micros().map(|n| n as f64 / 1e6 / p as f64),
235        },
236        other => run.field(other),
237    }
238}
239
240/// One recorded run by id, over all of history.
241pub fn eval_run_by_id<S: SubstrateRead + ?Sized>(
242    sub: &S,
243    evalset_hash: &str,
244    run_id: &str,
245) -> Result<Option<EvalRun>> {
246    Ok(eval_runs(sub, evalset_hash, None)?
247        .into_iter()
248        .find(|r| r.run_id == run_id))
249}
250
251/// Parse an `evalset:<hash>:<field>` metric string into its parts.
252///
253/// The hash is hex, so splitting on the LAST colon is unambiguous and lets a
254/// field name contain no colon by construction.
255pub fn parse_evalset_metric(metric: &str) -> Option<(&str, &str)> {
256    let rest = metric.strip_prefix("evalset:")?;
257    let (hash, field) = rest.rsplit_once(':')?;
258    if hash.is_empty() || field.is_empty() || hash.contains(':') {
259        return None;
260    }
261    Some((hash, field))
262}
263
264#[cfg(test)]
265mod tests {
266    use super::*;
267
268    fn run(summary: serde_json::Value, spend: Option<RunSpend>) -> EvalRun {
269        let summary = summary.as_object().unwrap().clone();
270        EvalRun {
271            run_id: "eval-1".into(),
272            passed: summary.get("passed").and_then(Value::as_u64).unwrap_or(0),
273            failed: summary.get("failed").and_then(Value::as_u64).unwrap_or(0),
274            summary,
275            recorded_ms: 0,
276            spend,
277        }
278    }
279
280    /// The cost keys read fail-closed, exactly like `passed`/`failed`: a
281    /// string, a float or a negative number is NOT measurable — never zero —
282    /// and `cost_per_pass` is undefined when nothing passed.
283    #[test]
284    fn cost_fields_are_integers_or_not_measurable() {
285        let ok = run(
286            serde_json::json!({"passed": 4, "failed": 1, "effects": 12, "input_tokens": 1000,
287                               "output_tokens": 200, "usd_micros": 2_000_000, "wall_ms": 340}),
288            None,
289        );
290        assert_eq!(run_value(&ok, "effects"), Some(12.0));
291        assert_eq!(run_value(&ok, "tokens"), Some(1200.0));
292        assert_eq!(run_value(&ok, "usd"), Some(2.0));
293        assert_eq!(run_value(&ok, "wall_ms"), Some(340.0));
294        assert_eq!(run_value(&ok, "cost_per_pass"), Some(0.5));
295
296        for bad in [
297            serde_json::json!({"passed": 4, "failed": 1, "input_tokens": "1000", "output_tokens": 200}),
298            serde_json::json!({"passed": 4, "failed": 1, "input_tokens": 1000.5, "output_tokens": 200}),
299            serde_json::json!({"passed": 4, "failed": 1, "input_tokens": -1, "output_tokens": 200}),
300            serde_json::json!({"passed": 4, "failed": 1, "output_tokens": 200}),
301            serde_json::json!({"passed": 4, "failed": 1}),
302        ] {
303            let r = run(bad.clone(), None);
304            assert_eq!(run_value(&r, "tokens"), None, "{bad}");
305            assert_eq!(run_value(&r, "cost_per_pass"), None, "{bad}");
306        }
307        // Nothing passed: undefined, not a division by zero, not zero.
308        let none = run(serde_json::json!({"passed": 0, "failed": 5, "usd_micros": 2_000_000}), None);
309        assert_eq!(run_value(&none, "usd"), Some(2.0));
310        assert_eq!(run_value(&none, "cost_per_pass"), None);
311        // The quality fields are untouched by a malformed cost key.
312        let r = run(serde_json::json!({"passed": 4, "failed": 1, "wall_ms": "fast"}), None);
313        assert_eq!(run_value(&r, "passed"), Some(4.0));
314        assert_eq!(run_value(&r, "wall_ms"), None);
315    }
316
317    /// A summary that carries no cost key reads the runtime's spend for the
318    /// same run id; one that carries the key — right or wrong — does not.
319    #[test]
320    fn a_missing_cost_key_falls_back_to_the_runtime_spend_and_a_present_one_does_not() {
321        let spend = RunSpend { input_tokens: 700, output_tokens: 300, usd_micros: 4_000_000, wall_ms: 9_000 };
322        let r = run(serde_json::json!({"passed": 2, "failed": 0}), Some(spend));
323        assert_eq!(run_value(&r, "tokens"), Some(1000.0));
324        assert_eq!(run_value(&r, "usd"), Some(4.0));
325        assert_eq!(run_value(&r, "wall_ms"), Some(9000.0));
326        assert_eq!(run_value(&r, "cost_per_pass"), Some(2.0));
327        assert_eq!(run_value(&r, "effects"), None, "the runtime records supersteps, not effects");
328        let wrong = run(serde_json::json!({"passed": 2, "failed": 0, "input_tokens": "700"}), Some(spend));
329        assert_eq!(run_value(&wrong, "tokens"), None, "a malformed key is malformed, not missing");
330        let own = run(serde_json::json!({"passed": 2, "failed": 0, "input_tokens": 1, "output_tokens": 1}), Some(spend));
331        assert_eq!(run_value(&own, "tokens"), Some(2.0), "the summary's own figure wins");
332    }
333
334    #[test]
335    fn metric_strings_parse_into_hash_and_field() {
336        assert_eq!(
337            parse_evalset_metric("evalset:abc123:category_accuracy"),
338            Some(("abc123", "category_accuracy"))
339        );
340        assert_eq!(parse_evalset_metric("evalset:abc123:failed"), Some(("abc123", "failed")));
341        // Not an evalset metric at all.
342        assert_eq!(parse_evalset_metric("tool_error_recurrence"), None);
343        // Malformed shapes must fail rather than half-parse into a lookup that
344        // silently finds nothing.
345        assert_eq!(parse_evalset_metric("evalset:abc123"), None);
346        assert_eq!(parse_evalset_metric("evalset::field"), None);
347        assert_eq!(parse_evalset_metric("evalset:abc123:"), None);
348        assert_eq!(parse_evalset_metric("evalset:a:b:c"), None);
349    }
350}