1use crate::error::Result;
16use crate::substrate::{ReadOpts, SubstrateRead};
17use serde_json::Value;
18
19pub const EVAL_RUN_RELATION: &str = "mg:eval_run";
21
22pub const HARNESS_NS: &str = "agent:harness";
24
25#[derive(Debug, Clone)]
27pub struct EvalRun {
28 pub run_id: String,
30 pub passed: u64,
31 pub failed: u64,
32 pub summary: serde_json::Map<String, Value>,
35 pub recorded_ms: i64,
37 pub spend: Option<RunSpend>,
43}
44
45#[derive(Debug, Clone, Copy, PartialEq, Eq)]
49pub struct RunSpend {
50 pub input_tokens: u64,
51 pub output_tokens: u64,
52 pub usd_micros: u64,
53 pub wall_ms: u64,
54}
55
56pub const COST_FIELDS: [&str; 5] = ["effects", "tokens", "usd", "wall_ms", "cost_per_pass"];
60
61impl EvalRun {
62 pub fn field(&self, name: &str) -> Option<f64> {
66 self.summary.get(name).and_then(Value::as_f64)
67 }
68
69 pub fn total(&self) -> u64 {
71 self.passed.saturating_add(self.failed)
72 }
73
74 fn cost_key(&self, key: &str, from_spend: impl Fn(&RunSpend) -> u64) -> Option<u64> {
80 match self.summary.get(key) {
81 Some(v) => v.as_u64(),
82 None => self.spend.as_ref().map(from_spend),
83 }
84 }
85
86 pub fn effects(&self) -> Option<u64> {
89 self.summary.get("effects").and_then(Value::as_u64)
90 }
91
92 pub fn tokens(&self) -> Option<u64> {
94 let i = self.cost_key("input_tokens", |s| s.input_tokens)?;
95 let o = self.cost_key("output_tokens", |s| s.output_tokens)?;
96 Some(i.saturating_add(o))
97 }
98
99 pub fn usd_micros(&self) -> Option<u64> {
100 self.cost_key("usd_micros", |s| s.usd_micros)
101 }
102
103 pub fn wall_ms(&self) -> Option<u64> {
104 self.cost_key("wall_ms", |s| s.wall_ms)
105 }
106}
107
108pub fn eval_runs<S: SubstrateRead + ?Sized>(
114 sub: &S,
115 evalset_hash: &str,
116 since_ms: Option<i64>,
117) -> Result<Vec<EvalRun>> {
118 let subject = format!("evalset:{evalset_hash}");
119 let facts = sub.grains_of_type(
120 crate::model::grain_type::FACT,
121 Some(HARNESS_NS),
122 ReadOpts { live_only: true, since_ms },
123 )?;
124 let spends = run_spends(sub)?;
125 let mut out: Vec<EvalRun> = facts
126 .iter()
127 .filter(|f| f.str_field("relation") == Some(EVAL_RUN_RELATION))
128 .filter(|f| f.str_field("subject") == Some(subject.as_str()))
129 .filter_map(|f| {
130 let obj = f.str_field("object")?;
131 let Ok(Value::Object(summary)) = serde_json::from_str::<Value>(obj) else {
132 return None;
135 };
136 let run_id = summary.get("run_id").and_then(Value::as_str)?.to_string();
142 Some(EvalRun {
143 spend: spends.get(&run_id).copied(),
144 run_id,
145 passed: summary.get("passed").and_then(Value::as_u64)?,
146 failed: summary.get("failed").and_then(Value::as_u64)?,
147 recorded_ms: f.created_at_ms,
148 summary,
149 })
150 })
151 .collect();
152 out.sort_by(|a, b| {
155 a.recorded_ms
156 .cmp(&b.recorded_ms)
157 .then_with(|| a.run_id.cmp(&b.run_id))
158 });
159 Ok(out)
160}
161
162fn run_spends<S: SubstrateRead + ?Sized>(sub: &S) -> Result<std::collections::BTreeMap<String, RunSpend>> {
168 let obs = match sub.grains_of_type(
169 crate::model::grain_type::OBSERVATION,
170 Some(HARNESS_NS),
171 ReadOpts { live_only: true, since_ms: None },
172 ) {
173 Ok(rows) => rows,
174 Err(_) => return Ok(Default::default()),
177 };
178 let mut out = std::collections::BTreeMap::new();
179 for g in &obs {
180 if g.str_field("observation_kind") != Some("run_outcome") {
181 continue;
182 }
183 let Some(run_id) = g.str_field("run_id") else { continue };
184 let int = |k: &str| g.fields.get(k).and_then(Value::as_u64);
185 let (Some(i), Some(o), Some(u), Some(w)) = (
186 int("spent_input_tokens"),
187 int("spent_output_tokens"),
188 int("spent_usd_micros"),
189 int("spent_wall_ms"),
190 ) else {
191 continue;
192 };
193 out.insert(
194 run_id.to_string(),
195 RunSpend { input_tokens: i, output_tokens: o, usd_micros: u, wall_ms: w },
196 );
197 }
198 Ok(out)
199}
200
201pub fn newest_eval_run<S: SubstrateRead + ?Sized>(
203 sub: &S,
204 evalset_hash: &str,
205 since_ms: Option<i64>,
206) -> Result<Option<EvalRun>> {
207 Ok(eval_runs(sub, evalset_hash, since_ms)?.pop())
208}
209
210pub fn run_value(run: &EvalRun, field: &str) -> Option<f64> {
220 match field {
221 "failed" => Some(run.failed as f64),
222 "passed" => Some(run.passed as f64),
223 "total" => Some(run.total() as f64),
224 "error_rate" => match run.total() {
225 0 => None,
226 t => Some(run.failed as f64 / t as f64),
227 },
228 "effects" => run.effects().map(|n| n as f64),
229 "tokens" => run.tokens().map(|n| n as f64),
230 "usd" => run.usd_micros().map(|n| n as f64 / 1e6),
231 "wall_ms" => run.wall_ms().map(|n| n as f64),
232 "cost_per_pass" => match run.passed {
233 0 => None,
234 p => run.usd_micros().map(|n| n as f64 / 1e6 / p as f64),
235 },
236 other => run.field(other),
237 }
238}
239
240pub fn eval_run_by_id<S: SubstrateRead + ?Sized>(
242 sub: &S,
243 evalset_hash: &str,
244 run_id: &str,
245) -> Result<Option<EvalRun>> {
246 Ok(eval_runs(sub, evalset_hash, None)?
247 .into_iter()
248 .find(|r| r.run_id == run_id))
249}
250
251pub fn parse_evalset_metric(metric: &str) -> Option<(&str, &str)> {
256 let rest = metric.strip_prefix("evalset:")?;
257 let (hash, field) = rest.rsplit_once(':')?;
258 if hash.is_empty() || field.is_empty() || hash.contains(':') {
259 return None;
260 }
261 Some((hash, field))
262}
263
264#[cfg(test)]
265mod tests {
266 use super::*;
267
268 fn run(summary: serde_json::Value, spend: Option<RunSpend>) -> EvalRun {
269 let summary = summary.as_object().unwrap().clone();
270 EvalRun {
271 run_id: "eval-1".into(),
272 passed: summary.get("passed").and_then(Value::as_u64).unwrap_or(0),
273 failed: summary.get("failed").and_then(Value::as_u64).unwrap_or(0),
274 summary,
275 recorded_ms: 0,
276 spend,
277 }
278 }
279
280 #[test]
284 fn cost_fields_are_integers_or_not_measurable() {
285 let ok = run(
286 serde_json::json!({"passed": 4, "failed": 1, "effects": 12, "input_tokens": 1000,
287 "output_tokens": 200, "usd_micros": 2_000_000, "wall_ms": 340}),
288 None,
289 );
290 assert_eq!(run_value(&ok, "effects"), Some(12.0));
291 assert_eq!(run_value(&ok, "tokens"), Some(1200.0));
292 assert_eq!(run_value(&ok, "usd"), Some(2.0));
293 assert_eq!(run_value(&ok, "wall_ms"), Some(340.0));
294 assert_eq!(run_value(&ok, "cost_per_pass"), Some(0.5));
295
296 for bad in [
297 serde_json::json!({"passed": 4, "failed": 1, "input_tokens": "1000", "output_tokens": 200}),
298 serde_json::json!({"passed": 4, "failed": 1, "input_tokens": 1000.5, "output_tokens": 200}),
299 serde_json::json!({"passed": 4, "failed": 1, "input_tokens": -1, "output_tokens": 200}),
300 serde_json::json!({"passed": 4, "failed": 1, "output_tokens": 200}),
301 serde_json::json!({"passed": 4, "failed": 1}),
302 ] {
303 let r = run(bad.clone(), None);
304 assert_eq!(run_value(&r, "tokens"), None, "{bad}");
305 assert_eq!(run_value(&r, "cost_per_pass"), None, "{bad}");
306 }
307 let none = run(serde_json::json!({"passed": 0, "failed": 5, "usd_micros": 2_000_000}), None);
309 assert_eq!(run_value(&none, "usd"), Some(2.0));
310 assert_eq!(run_value(&none, "cost_per_pass"), None);
311 let r = run(serde_json::json!({"passed": 4, "failed": 1, "wall_ms": "fast"}), None);
313 assert_eq!(run_value(&r, "passed"), Some(4.0));
314 assert_eq!(run_value(&r, "wall_ms"), None);
315 }
316
317 #[test]
320 fn a_missing_cost_key_falls_back_to_the_runtime_spend_and_a_present_one_does_not() {
321 let spend = RunSpend { input_tokens: 700, output_tokens: 300, usd_micros: 4_000_000, wall_ms: 9_000 };
322 let r = run(serde_json::json!({"passed": 2, "failed": 0}), Some(spend));
323 assert_eq!(run_value(&r, "tokens"), Some(1000.0));
324 assert_eq!(run_value(&r, "usd"), Some(4.0));
325 assert_eq!(run_value(&r, "wall_ms"), Some(9000.0));
326 assert_eq!(run_value(&r, "cost_per_pass"), Some(2.0));
327 assert_eq!(run_value(&r, "effects"), None, "the runtime records supersteps, not effects");
328 let wrong = run(serde_json::json!({"passed": 2, "failed": 0, "input_tokens": "700"}), Some(spend));
329 assert_eq!(run_value(&wrong, "tokens"), None, "a malformed key is malformed, not missing");
330 let own = run(serde_json::json!({"passed": 2, "failed": 0, "input_tokens": 1, "output_tokens": 1}), Some(spend));
331 assert_eq!(run_value(&own, "tokens"), Some(2.0), "the summary's own figure wins");
332 }
333
334 #[test]
335 fn metric_strings_parse_into_hash_and_field() {
336 assert_eq!(
337 parse_evalset_metric("evalset:abc123:category_accuracy"),
338 Some(("abc123", "category_accuracy"))
339 );
340 assert_eq!(parse_evalset_metric("evalset:abc123:failed"), Some(("abc123", "failed")));
341 assert_eq!(parse_evalset_metric("tool_error_recurrence"), None);
343 assert_eq!(parse_evalset_metric("evalset:abc123"), None);
346 assert_eq!(parse_evalset_metric("evalset::field"), None);
347 assert_eq!(parse_evalset_metric("evalset:abc123:"), None);
348 assert_eq!(parse_evalset_metric("evalset:a:b:c"), None);
349 }
350}