areev_loop/eval.rs
1//! Evalset runs — the one reader for the `evalset:<hash> mg:eval_run`
2//! summaries that `areev eval run` journals into `agent:harness`.
3//!
4//! Two edges consume these, and they deliberately share this code:
5//!
6//! - the **gating** edge (`areev loop apply --gating-run <id>`), which names one
7//! run and reads its numbers rather than trusting the command line, and
8//! - the **outcome** edge (`measure_metric`'s `evalset:` kind), which re-reads
9//! the newest run at each checkpoint.
10//!
11//! Sharing matters because the alternative is two parsers of the same JSON
12//! drifting apart, so a rule could be *admitted* on one reading of an evalset
13//! and *judged* on another.
14
15use crate::error::Result;
16use crate::substrate::{ReadOpts, SubstrateRead};
17use serde_json::Value;
18
19/// The relation an eval-run summary is recorded under.
20pub const EVAL_RUN_RELATION: &str = "mg:eval_run";
21
22/// The namespace those summaries live in.
23pub const HARNESS_NS: &str = "agent:harness";
24
25/// One recorded execution of an evalset.
26#[derive(Debug, Clone)]
27pub struct EvalRun {
28 /// The `eval-` run id the cases were journaled under.
29 pub run_id: String,
30 pub passed: u64,
31 pub failed: u64,
32 /// The whole summary object, so host-defined fields (`category_accuracy`,
33 /// …) are reachable without this module having to know them.
34 pub summary: serde_json::Map<String, Value>,
35 /// When the summary grain was written.
36 pub recorded_ms: i64,
37}
38
39impl EvalRun {
40 /// A numeric field of the summary. `passed`/`failed` are promoted to typed
41 /// fields but stay readable here too, so a metric string may name any of
42 /// them uniformly.
43 pub fn field(&self, name: &str) -> Option<f64> {
44 self.summary.get(name).and_then(Value::as_f64)
45 }
46
47 /// Cases that ran. `0` when the summary records neither.
48 pub fn total(&self) -> u64 {
49 self.passed.saturating_add(self.failed)
50 }
51}
52
53/// Every recorded run of `evalset_hash`, oldest first.
54///
55/// `since_ms` bounds the scan to summaries written at or after a moment —
56/// which is what keeps an outcome honest: a run journaled *before* a
57/// recommendation was applied cannot be evidence of what applying it did.
58pub fn eval_runs<S: SubstrateRead + ?Sized>(
59 sub: &S,
60 evalset_hash: &str,
61 since_ms: Option<i64>,
62) -> Result<Vec<EvalRun>> {
63 let subject = format!("evalset:{evalset_hash}");
64 let facts = sub.grains_of_type(
65 crate::model::grain_type::FACT,
66 Some(HARNESS_NS),
67 ReadOpts { live_only: true, since_ms },
68 )?;
69 let mut out: Vec<EvalRun> = facts
70 .iter()
71 .filter(|f| f.str_field("relation") == Some(EVAL_RUN_RELATION))
72 .filter(|f| f.str_field("subject") == Some(subject.as_str()))
73 .filter_map(|f| {
74 let obj = f.str_field("object")?;
75 let Ok(Value::Object(summary)) = serde_json::from_str::<Value>(obj) else {
76 // A summary we cannot parse is skipped, never guessed at: a
77 // fabricated number here would become a receipt.
78 return None;
79 };
80 // `areev eval run` always writes all three, so a summary missing
81 // one is malformed. Dropping it rather than defaulting keeps every
82 // consumer fail-CLOSED: at the apply gate an absent `failed` must
83 // never read as "zero failures", and an outcome must never score
84 // against numbers nobody recorded.
85 Some(EvalRun {
86 run_id: summary.get("run_id").and_then(Value::as_str)?.to_string(),
87 passed: summary.get("passed").and_then(Value::as_u64)?,
88 failed: summary.get("failed").and_then(Value::as_u64)?,
89 recorded_ms: f.created_at_ms,
90 summary,
91 })
92 })
93 .collect();
94 // Recording order, with the hash as a deterministic tiebreak for two
95 // summaries written in the same millisecond.
96 out.sort_by(|a, b| {
97 a.recorded_ms
98 .cmp(&b.recorded_ms)
99 .then_with(|| a.run_id.cmp(&b.run_id))
100 });
101 Ok(out)
102}
103
104/// The newest recorded run of `evalset_hash` at or after `since_ms`.
105pub fn newest_eval_run<S: SubstrateRead + ?Sized>(
106 sub: &S,
107 evalset_hash: &str,
108 since_ms: Option<i64>,
109) -> Result<Option<EvalRun>> {
110 Ok(eval_runs(sub, evalset_hash, since_ms)?.pop())
111}
112
113/// One recorded run by id, over all of history.
114pub fn eval_run_by_id<S: SubstrateRead + ?Sized>(
115 sub: &S,
116 evalset_hash: &str,
117 run_id: &str,
118) -> Result<Option<EvalRun>> {
119 Ok(eval_runs(sub, evalset_hash, None)?
120 .into_iter()
121 .find(|r| r.run_id == run_id))
122}
123
124/// Parse an `evalset:<hash>:<field>` metric string into its parts.
125///
126/// The hash is hex, so splitting on the LAST colon is unambiguous and lets a
127/// field name contain no colon by construction.
128pub fn parse_evalset_metric(metric: &str) -> Option<(&str, &str)> {
129 let rest = metric.strip_prefix("evalset:")?;
130 let (hash, field) = rest.rsplit_once(':')?;
131 if hash.is_empty() || field.is_empty() || hash.contains(':') {
132 return None;
133 }
134 Some((hash, field))
135}
136
137#[cfg(test)]
138mod tests {
139 use super::*;
140
141 #[test]
142 fn metric_strings_parse_into_hash_and_field() {
143 assert_eq!(
144 parse_evalset_metric("evalset:abc123:category_accuracy"),
145 Some(("abc123", "category_accuracy"))
146 );
147 assert_eq!(parse_evalset_metric("evalset:abc123:failed"), Some(("abc123", "failed")));
148 // Not an evalset metric at all.
149 assert_eq!(parse_evalset_metric("tool_error_recurrence"), None);
150 // Malformed shapes must fail rather than half-parse into a lookup that
151 // silently finds nothing.
152 assert_eq!(parse_evalset_metric("evalset:abc123"), None);
153 assert_eq!(parse_evalset_metric("evalset::field"), None);
154 assert_eq!(parse_evalset_metric("evalset:abc123:"), None);
155 assert_eq!(parse_evalset_metric("evalset:a:b:c"), None);
156 }
157}