Skip to main content

areev_loop/
eval.rs

1//! Evalset runs — the one reader for the `evalset:<hash> mg:eval_run`
2//! summaries that `areev eval run` journals into `agent:harness`.
3//!
4//! Two edges consume these, and they deliberately share this code:
5//!
6//! - the **gating** edge (`areev loop apply --gating-run <id>`), which names one
7//!   run and reads its numbers rather than trusting the command line, and
8//! - the **outcome** edge (`measure_metric`'s `evalset:` kind), which re-reads
9//!   the newest run at each checkpoint.
10//!
11//! Sharing matters because the alternative is two parsers of the same JSON
12//! drifting apart, so a rule could be *admitted* on one reading of an evalset
13//! and *judged* on another.
14
15use crate::error::Result;
16use crate::substrate::{ReadOpts, SubstrateRead};
17use serde_json::Value;
18
19/// The relation an eval-run summary is recorded under.
20pub const EVAL_RUN_RELATION: &str = "mg:eval_run";
21
22/// The namespace those summaries live in.
23pub const HARNESS_NS: &str = "agent:harness";
24
25/// One recorded execution of an evalset.
26#[derive(Debug, Clone)]
27pub struct EvalRun {
28    /// The `eval-` run id the cases were journaled under.
29    pub run_id: String,
30    pub passed: u64,
31    pub failed: u64,
32    /// The whole summary object, so host-defined fields (`category_accuracy`,
33    /// …) are reachable without this module having to know them.
34    pub summary: serde_json::Map<String, Value>,
35    /// When the summary grain was written.
36    pub recorded_ms: i64,
37}
38
39impl EvalRun {
40    /// A numeric field of the summary. `passed`/`failed` are promoted to typed
41    /// fields but stay readable here too, so a metric string may name any of
42    /// them uniformly.
43    pub fn field(&self, name: &str) -> Option<f64> {
44        self.summary.get(name).and_then(Value::as_f64)
45    }
46
47    /// Cases that ran. `0` when the summary records neither.
48    pub fn total(&self) -> u64 {
49        self.passed.saturating_add(self.failed)
50    }
51}
52
53/// Every recorded run of `evalset_hash`, oldest first.
54///
55/// `since_ms` bounds the scan to summaries written at or after a moment —
56/// which is what keeps an outcome honest: a run journaled *before* a
57/// recommendation was applied cannot be evidence of what applying it did.
58pub fn eval_runs<S: SubstrateRead + ?Sized>(
59    sub: &S,
60    evalset_hash: &str,
61    since_ms: Option<i64>,
62) -> Result<Vec<EvalRun>> {
63    let subject = format!("evalset:{evalset_hash}");
64    let facts = sub.grains_of_type(
65        crate::model::grain_type::FACT,
66        Some(HARNESS_NS),
67        ReadOpts { live_only: true, since_ms },
68    )?;
69    let mut out: Vec<EvalRun> = facts
70        .iter()
71        .filter(|f| f.str_field("relation") == Some(EVAL_RUN_RELATION))
72        .filter(|f| f.str_field("subject") == Some(subject.as_str()))
73        .filter_map(|f| {
74            let obj = f.str_field("object")?;
75            let Ok(Value::Object(summary)) = serde_json::from_str::<Value>(obj) else {
76                // A summary we cannot parse is skipped, never guessed at: a
77                // fabricated number here would become a receipt.
78                return None;
79            };
80            // `areev eval run` always writes all three, so a summary missing
81            // one is malformed. Dropping it rather than defaulting keeps every
82            // consumer fail-CLOSED: at the apply gate an absent `failed` must
83            // never read as "zero failures", and an outcome must never score
84            // against numbers nobody recorded.
85            Some(EvalRun {
86                run_id: summary.get("run_id").and_then(Value::as_str)?.to_string(),
87                passed: summary.get("passed").and_then(Value::as_u64)?,
88                failed: summary.get("failed").and_then(Value::as_u64)?,
89                recorded_ms: f.created_at_ms,
90                summary,
91            })
92        })
93        .collect();
94    // Recording order, with the hash as a deterministic tiebreak for two
95    // summaries written in the same millisecond.
96    out.sort_by(|a, b| {
97        a.recorded_ms
98            .cmp(&b.recorded_ms)
99            .then_with(|| a.run_id.cmp(&b.run_id))
100    });
101    Ok(out)
102}
103
104/// The newest recorded run of `evalset_hash` at or after `since_ms`.
105pub fn newest_eval_run<S: SubstrateRead + ?Sized>(
106    sub: &S,
107    evalset_hash: &str,
108    since_ms: Option<i64>,
109) -> Result<Option<EvalRun>> {
110    Ok(eval_runs(sub, evalset_hash, since_ms)?.pop())
111}
112
113/// The newest recorded run of `evalset_hash` strictly before `before_ms` —
114/// the state of the world when something was about to be applied.
115pub fn newest_eval_run_before<S: SubstrateRead + ?Sized>(
116    sub: &S,
117    evalset_hash: &str,
118    before_ms: i64,
119) -> Result<Option<EvalRun>> {
120    Ok(eval_runs(sub, evalset_hash, None)?
121        .into_iter()
122        .rfind(|r| r.recorded_ms < before_ms))
123}
124
125/// The value a metric field takes on one run. `failed`/`passed`/`total` are
126/// promoted so a metric can be written against any evalset without the host
127/// having to add fields; `error_rate` is derived (undefined, not zero, when
128/// no case ran); anything else is read from the summary the host did write.
129/// The proposal-time baseline and every later measurement go through this one
130/// reader, so a rule cannot be admitted on one reading of a run and judged
131/// on another.
132pub fn run_value(run: &EvalRun, field: &str) -> Option<f64> {
133    match field {
134        "failed" => Some(run.failed as f64),
135        "passed" => Some(run.passed as f64),
136        "total" => Some(run.total() as f64),
137        "error_rate" => match run.total() {
138            0 => None,
139            t => Some(run.failed as f64 / t as f64),
140        },
141        other => run.field(other),
142    }
143}
144
145/// One recorded run by id, over all of history.
146pub fn eval_run_by_id<S: SubstrateRead + ?Sized>(
147    sub: &S,
148    evalset_hash: &str,
149    run_id: &str,
150) -> Result<Option<EvalRun>> {
151    Ok(eval_runs(sub, evalset_hash, None)?
152        .into_iter()
153        .find(|r| r.run_id == run_id))
154}
155
156/// Parse an `evalset:<hash>:<field>` metric string into its parts.
157///
158/// The hash is hex, so splitting on the LAST colon is unambiguous and lets a
159/// field name contain no colon by construction.
160pub fn parse_evalset_metric(metric: &str) -> Option<(&str, &str)> {
161    let rest = metric.strip_prefix("evalset:")?;
162    let (hash, field) = rest.rsplit_once(':')?;
163    if hash.is_empty() || field.is_empty() || hash.contains(':') {
164        return None;
165    }
166    Some((hash, field))
167}
168
169#[cfg(test)]
170mod tests {
171    use super::*;
172
173    #[test]
174    fn metric_strings_parse_into_hash_and_field() {
175        assert_eq!(
176            parse_evalset_metric("evalset:abc123:category_accuracy"),
177            Some(("abc123", "category_accuracy"))
178        );
179        assert_eq!(parse_evalset_metric("evalset:abc123:failed"), Some(("abc123", "failed")));
180        // Not an evalset metric at all.
181        assert_eq!(parse_evalset_metric("tool_error_recurrence"), None);
182        // Malformed shapes must fail rather than half-parse into a lookup that
183        // silently finds nothing.
184        assert_eq!(parse_evalset_metric("evalset:abc123"), None);
185        assert_eq!(parse_evalset_metric("evalset::field"), None);
186        assert_eq!(parse_evalset_metric("evalset:abc123:"), None);
187        assert_eq!(parse_evalset_metric("evalset:a:b:c"), None);
188    }
189}