areev_loop/eval.rs
1//! Evalset runs — the one reader for the `evalset:<hash> mg:eval_run`
2//! summaries that `areev eval run` journals into `agent:harness`.
3//!
4//! Two edges consume these, and they deliberately share this code:
5//!
6//! - the **gating** edge (`areev loop apply --gating-run <id>`), which names one
7//! run and reads its numbers rather than trusting the command line, and
8//! - the **outcome** edge (`measure_metric`'s `evalset:` kind), which re-reads
9//! the newest run at each checkpoint.
10//!
11//! Sharing matters because the alternative is two parsers of the same JSON
12//! drifting apart, so a rule could be *admitted* on one reading of an evalset
13//! and *judged* on another.
14
15use crate::error::Result;
16use crate::substrate::{ReadOpts, SubstrateRead};
17use serde_json::Value;
18
19/// The relation an eval-run summary is recorded under.
20pub const EVAL_RUN_RELATION: &str = "mg:eval_run";
21
22/// The namespace those summaries live in.
23pub const HARNESS_NS: &str = "agent:harness";
24
25/// One recorded execution of an evalset.
26#[derive(Debug, Clone)]
27pub struct EvalRun {
28 /// The `eval-` run id the cases were journaled under.
29 pub run_id: String,
30 pub passed: u64,
31 pub failed: u64,
32 /// The whole summary object, so host-defined fields (`category_accuracy`,
33 /// …) are reachable without this module having to know them.
34 pub summary: serde_json::Map<String, Value>,
35 /// When the summary grain was written.
36 pub recorded_ms: i64,
37}
38
39impl EvalRun {
40 /// A numeric field of the summary. `passed`/`failed` are promoted to typed
41 /// fields but stay readable here too, so a metric string may name any of
42 /// them uniformly.
43 pub fn field(&self, name: &str) -> Option<f64> {
44 self.summary.get(name).and_then(Value::as_f64)
45 }
46
47 /// Cases that ran. `0` when the summary records neither.
48 pub fn total(&self) -> u64 {
49 self.passed.saturating_add(self.failed)
50 }
51}
52
53/// Every recorded run of `evalset_hash`, oldest first.
54///
55/// `since_ms` bounds the scan to summaries written at or after a moment —
56/// which is what keeps an outcome honest: a run journaled *before* a
57/// recommendation was applied cannot be evidence of what applying it did.
58pub fn eval_runs<S: SubstrateRead + ?Sized>(
59 sub: &S,
60 evalset_hash: &str,
61 since_ms: Option<i64>,
62) -> Result<Vec<EvalRun>> {
63 let subject = format!("evalset:{evalset_hash}");
64 let facts = sub.grains_of_type(
65 crate::model::grain_type::FACT,
66 Some(HARNESS_NS),
67 ReadOpts { live_only: true, since_ms },
68 )?;
69 let mut out: Vec<EvalRun> = facts
70 .iter()
71 .filter(|f| f.str_field("relation") == Some(EVAL_RUN_RELATION))
72 .filter(|f| f.str_field("subject") == Some(subject.as_str()))
73 .filter_map(|f| {
74 let obj = f.str_field("object")?;
75 let Ok(Value::Object(summary)) = serde_json::from_str::<Value>(obj) else {
76 // A summary we cannot parse is skipped, never guessed at: a
77 // fabricated number here would become a receipt.
78 return None;
79 };
80 // `areev eval run` always writes all three, so a summary missing
81 // one is malformed. Dropping it rather than defaulting keeps every
82 // consumer fail-CLOSED: at the apply gate an absent `failed` must
83 // never read as "zero failures", and an outcome must never score
84 // against numbers nobody recorded.
85 Some(EvalRun {
86 run_id: summary.get("run_id").and_then(Value::as_str)?.to_string(),
87 passed: summary.get("passed").and_then(Value::as_u64)?,
88 failed: summary.get("failed").and_then(Value::as_u64)?,
89 recorded_ms: f.created_at_ms,
90 summary,
91 })
92 })
93 .collect();
94 // Recording order, with the hash as a deterministic tiebreak for two
95 // summaries written in the same millisecond.
96 out.sort_by(|a, b| {
97 a.recorded_ms
98 .cmp(&b.recorded_ms)
99 .then_with(|| a.run_id.cmp(&b.run_id))
100 });
101 Ok(out)
102}
103
104/// The newest recorded run of `evalset_hash` at or after `since_ms`.
105pub fn newest_eval_run<S: SubstrateRead + ?Sized>(
106 sub: &S,
107 evalset_hash: &str,
108 since_ms: Option<i64>,
109) -> Result<Option<EvalRun>> {
110 Ok(eval_runs(sub, evalset_hash, since_ms)?.pop())
111}
112
113/// The newest recorded run of `evalset_hash` strictly before `before_ms` —
114/// the state of the world when something was about to be applied.
115pub fn newest_eval_run_before<S: SubstrateRead + ?Sized>(
116 sub: &S,
117 evalset_hash: &str,
118 before_ms: i64,
119) -> Result<Option<EvalRun>> {
120 Ok(eval_runs(sub, evalset_hash, None)?
121 .into_iter()
122 .rfind(|r| r.recorded_ms < before_ms))
123}
124
125/// The value a metric field takes on one run. `failed`/`passed`/`total` are
126/// promoted so a metric can be written against any evalset without the host
127/// having to add fields; `error_rate` is derived (undefined, not zero, when
128/// no case ran); anything else is read from the summary the host did write.
129/// The proposal-time baseline and every later measurement go through this one
130/// reader, so a rule cannot be admitted on one reading of a run and judged
131/// on another.
132pub fn run_value(run: &EvalRun, field: &str) -> Option<f64> {
133 match field {
134 "failed" => Some(run.failed as f64),
135 "passed" => Some(run.passed as f64),
136 "total" => Some(run.total() as f64),
137 "error_rate" => match run.total() {
138 0 => None,
139 t => Some(run.failed as f64 / t as f64),
140 },
141 other => run.field(other),
142 }
143}
144
145/// One recorded run by id, over all of history.
146pub fn eval_run_by_id<S: SubstrateRead + ?Sized>(
147 sub: &S,
148 evalset_hash: &str,
149 run_id: &str,
150) -> Result<Option<EvalRun>> {
151 Ok(eval_runs(sub, evalset_hash, None)?
152 .into_iter()
153 .find(|r| r.run_id == run_id))
154}
155
156/// Parse an `evalset:<hash>:<field>` metric string into its parts.
157///
158/// The hash is hex, so splitting on the LAST colon is unambiguous and lets a
159/// field name contain no colon by construction.
160pub fn parse_evalset_metric(metric: &str) -> Option<(&str, &str)> {
161 let rest = metric.strip_prefix("evalset:")?;
162 let (hash, field) = rest.rsplit_once(':')?;
163 if hash.is_empty() || field.is_empty() || hash.contains(':') {
164 return None;
165 }
166 Some((hash, field))
167}
168
169#[cfg(test)]
170mod tests {
171 use super::*;
172
173 #[test]
174 fn metric_strings_parse_into_hash_and_field() {
175 assert_eq!(
176 parse_evalset_metric("evalset:abc123:category_accuracy"),
177 Some(("abc123", "category_accuracy"))
178 );
179 assert_eq!(parse_evalset_metric("evalset:abc123:failed"), Some(("abc123", "failed")));
180 // Not an evalset metric at all.
181 assert_eq!(parse_evalset_metric("tool_error_recurrence"), None);
182 // Malformed shapes must fail rather than half-parse into a lookup that
183 // silently finds nothing.
184 assert_eq!(parse_evalset_metric("evalset:abc123"), None);
185 assert_eq!(parse_evalset_metric("evalset::field"), None);
186 assert_eq!(parse_evalset_metric("evalset:abc123:"), None);
187 assert_eq!(parse_evalset_metric("evalset:a:b:c"), None);
188 }
189}