1use crate::trace_analyzer::HarnessTraceSummary;
2use crate::{EvalMetric, task::EvalCategory};
3use serde::Serialize;
4
5#[derive(Debug, Clone, Serialize)]
6pub struct TaskReport {
7 task_id: String,
8 pub(crate) category: String,
9 pub(crate) metric: EvalMetric,
10}
11
12#[derive(Debug, Clone, Serialize)]
13pub struct SuiteReport {
14 pub(crate) suite_id: String,
15 pub(crate) suite_name: String,
16 pub(crate) task_reports: Vec<TaskReport>,
17 pub(crate) aggregate: EvalMetric,
18 pub(crate) capability_metrics: EvalMetric,
19 pub(crate) regression_metrics: EvalMetric,
20 pub(crate) cost_usd: Option<f64>,
23 pub(crate) unpriced_runs: u32,
24 pub(crate) duration_secs: f64,
25 pub(crate) trace_summary: Option<HarnessTraceSummary>,
27 pub(crate) mean_cost_per_attempt: Option<f64>,
29 pub(crate) cost_per_solve: Option<f64>,
32 pub(crate) mean_tokens_per_attempt: Option<f64>,
34 pub(crate) mean_turns_per_attempt: Option<f64>,
36}
37
38#[derive(Debug, Clone, Serialize)]
39pub struct EvalReport {
40 pub(crate) generated_at: String,
41 pub(crate) suites: Vec<SuiteReport>,
42}
43
44impl EvalReport {
45 pub fn new(generated_at: impl Into<String>, suites: Vec<SuiteReport>) -> Self {
47 Self { generated_at: generated_at.into(), suites }
48 }
49
50 pub fn to_markdown(&self) -> String {
51 let mut out = String::new();
52 out.push_str("# Eval Report\n\n");
53 for s in &self.suites {
54 out.push_str(&format!("## {}\n\n", s.suite_name));
55 out.push_str(&format!(
56 "- Aggregate: pass@k={:.1}%; pass^k={:.1}%\n",
57 s.aggregate.pass_at_k * 100.0,
58 s.aggregate.pass_power_k * 100.0
59 ));
60 if let Some(cost) = s.cost_usd {
61 out.push_str(&format!("- Cost (known): ${cost:.6}; unpriced attempts: {}\n", s.unpriced_runs));
62 } else if s.unpriced_runs > 0 {
63 out.push_str(&format!("- Cost (known): unavailable; unpriced attempts: {}\n", s.unpriced_runs));
64 }
65 if s.mean_cost_per_attempt.is_some()
66 || s.cost_per_solve.is_some()
67 || s.mean_tokens_per_attempt.is_some()
68 || s.mean_turns_per_attempt.is_some()
69 {
70 let mut parts = Vec::new();
71 if let Some(cost_per_solve) = s.cost_per_solve {
72 parts.push(format!("cost/solve ${cost_per_solve:.4}"));
73 }
74 if let Some(mean_cost) = s.mean_cost_per_attempt {
75 parts.push(format!("${mean_cost:.4} per priced attempt"));
76 }
77 if let Some(mean_tokens) = s.mean_tokens_per_attempt {
78 parts.push(format!("{mean_tokens:.0} tokens"));
79 }
80 if let Some(mean_turns) = s.mean_turns_per_attempt {
81 parts.push(format!("{mean_turns:.1} turns"));
82 }
83 out.push_str(&format!("- Efficiency: {}\n", parts.join(" · ")));
84 }
85 if let Some(trace) = &s.trace_summary {
86 if let Some(mean_ms) = trace.latency.mean_ms {
87 out.push_str(&format!(
88 "- Trace: {} turns, {} tool calls, mean latency {:.1} ms\n",
89 trace.turns, trace.tool_calls, mean_ms
90 ));
91 } else {
92 out.push_str(&format!("- Trace: {} turns, {} tool calls\n", trace.turns, trace.tool_calls));
93 }
94 }
95 out.push_str("| Task | Category | pass@k | pass^k | passed/total |\n");
96 out.push_str("|------|----------|--------|--------|-------------|\n");
97 for t in &s.task_reports {
98 out.push_str(&format!(
99 "| {} | {} | {:.1}% | {:.1}% | {}/{} |\n",
100 t.task_id,
101 t.category.as_str(),
102 t.metric.pass_at_k * 100.0,
103 t.metric.pass_power_k * 100.0,
104 t.metric.passed_runs,
105 t.metric.total_runs
106 ));
107 }
108 out.push('\n');
109 }
110 out
111 }
112}
113
114pub fn build_task_report(task_id: &str, _name: &str, category: EvalCategory, metric: EvalMetric) -> TaskReport {
115 TaskReport {
116 task_id: task_id.into(),
117 category: category.label().into(),
118 metric,
119 }
120}
121
122#[derive(Debug, Clone, Copy, Default, PartialEq)]
127pub struct CostEfficiency {
128 pub mean_cost_per_attempt: Option<f64>,
129 pub cost_per_solve: Option<f64>,
130 pub mean_tokens_per_attempt: Option<f64>,
131 pub mean_turns_per_attempt: Option<f64>,
132}
133
134pub fn priced_cost(cost_usd: Option<f64>) -> Option<f64> {
136 cost_usd.filter(|value| value.is_finite() && *value >= 0.0)
137}
138
139impl CostEfficiency {
140 pub fn from_runs<'a>(results: impl IntoIterator<Item = &'a crate::task::EvalRunResult>) -> Self {
145 let mut total_cost = 0.0_f64;
146 let mut known_cost_runs = 0_u32;
147 let mut unpriced_runs = 0_u32;
148 let mut passed_runs = 0_u32;
149 let mut token_sum = 0.0_f64;
150 let mut token_count = 0_u32;
151 let mut turn_sum = 0.0_f64;
152 let mut turn_count = 0_u32;
153
154 for result in results {
155 if result.outcome == crate::task::RunOutcome::Pass {
156 passed_runs = passed_runs.saturating_add(1);
157 }
158 match priced_cost(result.cost_usd) {
159 Some(cost) => {
160 total_cost += cost;
161 known_cost_runs = known_cost_runs.saturating_add(1);
162 }
163 None => {
164 unpriced_runs = unpriced_runs.saturating_add(1);
165 }
166 }
167 if let Some(summary) = &result.trace_summary {
168 let gross = (summary
169 .token_usage
170 .input_tokens
171 .saturating_add(summary.token_usage.output_tokens)) as f64;
172 token_sum += gross;
173 token_count = token_count.saturating_add(1);
174 turn_sum += summary.turns as f64;
175 turn_count = turn_count.saturating_add(1);
176 }
177 }
178
179 let mean_cost_per_attempt = (known_cost_runs > 0).then(|| total_cost / f64::from(known_cost_runs));
180 let cost_per_solve =
181 (unpriced_runs == 0 && passed_runs > 0 && known_cost_runs > 0).then(|| total_cost / f64::from(passed_runs));
182 let mean_tokens_per_attempt = (token_count > 0).then(|| token_sum / f64::from(token_count));
183 let mean_turns_per_attempt = (turn_count > 0).then(|| turn_sum / f64::from(turn_count));
184
185 Self {
186 mean_cost_per_attempt,
187 cost_per_solve,
188 mean_tokens_per_attempt,
189 mean_turns_per_attempt,
190 }
191 }
192}
193
194#[cfg(test)]
195mod tests {
196 use super::*;
197 use crate::metric::EvalMetric;
198 use crate::task::EvalCategory;
199
200 #[test]
201 fn to_markdown_renders_tasks_and_aggregate() {
202 let report = EvalReport {
203 generated_at: "2026-01-01".into(),
204 suites: vec![SuiteReport {
205 suite_id: "s1".into(),
206 suite_name: "demo".into(),
207 task_reports: vec![TaskReport {
208 task_id: "t1".into(),
209 category: "Capability".into(),
210 metric: EvalMetric {
211 pass_at_k: 0.5,
212 pass_power_k: 0.25,
213 pass_all_k: 0.0,
214 k: 1,
215 total_runs: 2,
216 passed_runs: 1,
217 task_id: "t1".into(),
218 },
219 }],
220 aggregate: EvalMetric {
221 pass_at_k: 0.5,
222 pass_power_k: 0.25,
223 pass_all_k: 0.0,
224 k: 1,
225 total_runs: 2,
226 passed_runs: 1,
227 task_id: "aggregate".into(),
228 },
229 capability_metrics: EvalMetric {
230 pass_at_k: 0.5,
231 pass_power_k: 0.25,
232 pass_all_k: 0.0,
233 k: 1,
234 total_runs: 2,
235 passed_runs: 1,
236 task_id: "cap".into(),
237 },
238 regression_metrics: EvalMetric {
239 pass_at_k: 0.0,
240 pass_power_k: 0.0,
241 pass_all_k: 0.0,
242 k: 0,
243 total_runs: 0,
244 passed_runs: 0,
245 task_id: "reg".into(),
246 },
247 cost_usd: Some(0.0),
248 unpriced_runs: 0,
249 duration_secs: 0.0,
250 trace_summary: None,
251 mean_cost_per_attempt: None,
252 cost_per_solve: None,
253 mean_tokens_per_attempt: None,
254 mean_turns_per_attempt: None,
255 }],
256 };
257 let md = report.to_markdown();
258 assert!(md.contains("# Eval Report"));
259 assert!(md.contains("demo"));
260 assert!(md.contains("t1"));
261 assert!(md.contains("Capability"));
262 assert!(md.contains("pass^k"));
263 }
264
265 #[test]
266 fn build_task_report_maps_category() {
267 let tr = build_task_report(
268 "t1",
269 "name",
270 EvalCategory::Regression,
271 EvalMetric {
272 pass_at_k: 1.0,
273 pass_power_k: 1.0,
274 pass_all_k: 1.0,
275 k: 1,
276 total_runs: 1,
277 passed_runs: 1,
278 task_id: "t1".into(),
279 },
280 );
281 assert_eq!(tr.task_id, "t1");
282 assert_eq!(tr.category, "Regression");
283 }
284
285 fn run(outcome: crate::task::RunOutcome, cost: Option<f64>, turns: u64, tokens: u64) -> crate::task::EvalRunResult {
286 use crate::trace_analyzer::{HarnessTraceSummary, TokenUsage};
287 crate::task::EvalRunResult {
288 task_id: "t1".into(),
289 outcome,
290 error_message: None,
291 duration_secs: 1.0,
292 attempt: 1,
293 cost_usd: cost,
294 transcript_path: None,
295 trace_summary: Some(HarnessTraceSummary {
296 turns,
297 token_usage: TokenUsage {
298 input_tokens: tokens / 2,
299 output_tokens: tokens - tokens / 2,
300 ..TokenUsage::default()
301 },
302 ..HarnessTraceSummary::default()
303 }),
304 }
305 }
306
307 #[test]
308 fn cost_efficiency_uses_total_cost_over_passed_attempts() {
309 let results = vec![
310 run(crate::task::RunOutcome::Pass, Some(0.02), 10, 1_000),
311 run(crate::task::RunOutcome::Fail, Some(0.04), 20, 3_000),
312 ];
313 let eff = CostEfficiency::from_runs(&results);
314 assert_eq!(eff.cost_per_solve, Some(0.06));
315 assert_eq!(eff.mean_cost_per_attempt, Some(0.03));
316 assert_eq!(eff.mean_tokens_per_attempt, Some(2_000.0));
317 assert_eq!(eff.mean_turns_per_attempt, Some(15.0));
318 }
319
320 #[test]
321 fn cost_per_solve_is_none_when_any_attempt_is_unpriced() {
322 let results = vec![
323 run(crate::task::RunOutcome::Pass, Some(0.02), 5, 100),
324 run(crate::task::RunOutcome::Pass, None, 5, 100),
325 ];
326 let eff = CostEfficiency::from_runs(&results);
327 assert_eq!(eff.cost_per_solve, None);
328 assert_eq!(eff.mean_cost_per_attempt, Some(0.02));
330 }
331
332 #[test]
333 fn cost_per_solve_is_none_when_nothing_passed() {
334 let results = vec![run(crate::task::RunOutcome::Fail, Some(0.10), 3, 50)];
335 let eff = CostEfficiency::from_runs(&results);
336 assert_eq!(eff.cost_per_solve, None);
337 assert_eq!(eff.mean_cost_per_attempt, Some(0.10));
338 }
339
340 #[test]
341 fn token_and_turn_means_are_none_without_traces() {
342 let mut result = run(crate::task::RunOutcome::Pass, Some(0.01), 1, 1);
343 result.trace_summary = None;
344 let eff = CostEfficiency::from_runs(std::slice::from_ref(&result));
345 assert_eq!(eff.mean_tokens_per_attempt, None);
346 assert_eq!(eff.mean_turns_per_attempt, None);
347 }
348
349 #[test]
350 fn markdown_renders_efficiency_line() {
351 let report = EvalReport {
352 generated_at: "2026-01-01".into(),
353 suites: vec![SuiteReport {
354 suite_id: "s1".into(),
355 suite_name: "demo".into(),
356 task_reports: Vec::new(),
357 aggregate: EvalMetric {
358 pass_at_k: 1.0,
359 pass_power_k: 1.0,
360 pass_all_k: 1.0,
361 k: 1,
362 total_runs: 1,
363 passed_runs: 1,
364 task_id: "aggregate".into(),
365 },
366 capability_metrics: EvalMetric {
367 pass_at_k: 1.0,
368 pass_power_k: 1.0,
369 pass_all_k: 1.0,
370 k: 1,
371 total_runs: 1,
372 passed_runs: 1,
373 task_id: "cap".into(),
374 },
375 regression_metrics: EvalMetric {
376 pass_at_k: 0.0,
377 pass_power_k: 0.0,
378 pass_all_k: 0.0,
379 k: 0,
380 total_runs: 0,
381 passed_runs: 0,
382 task_id: "reg".into(),
383 },
384 cost_usd: Some(0.05),
385 unpriced_runs: 0,
386 duration_secs: 1.0,
387 trace_summary: None,
388 mean_cost_per_attempt: Some(0.05),
389 cost_per_solve: Some(0.05),
390 mean_tokens_per_attempt: Some(12_000.0),
391 mean_turns_per_attempt: Some(8.5),
392 }],
393 };
394 let md = report.to_markdown();
395 assert!(
396 md.contains("- Efficiency: cost/solve $0.0500 · $0.0500 per priced attempt · 12000 tokens · 8.5 turns"),
397 "missing efficiency line: {md}"
398 );
399 }
400}