Skip to main content

vtcode_eval/
report.rs

1use crate::trace_analyzer::HarnessTraceSummary;
2use crate::{EvalMetric, task::EvalCategory};
3use serde::Serialize;
4
5#[derive(Debug, Clone, Serialize)]
6pub struct TaskReport {
7    task_id: String,
8    pub(crate) category: String,
9    pub(crate) metric: EvalMetric,
10}
11
12#[derive(Debug, Clone, Serialize)]
13pub struct SuiteReport {
14    pub(crate) suite_id: String,
15    pub(crate) suite_name: String,
16    pub(crate) task_reports: Vec<TaskReport>,
17    pub(crate) aggregate: EvalMetric,
18    pub(crate) capability_metrics: EvalMetric,
19    pub(crate) regression_metrics: EvalMetric,
20    /// Sum of known per-attempt costs. Unknown-pricing attempts are counted
21    /// separately instead of being treated as zero-cost successes.
22    pub(crate) cost_usd: Option<f64>,
23    pub(crate) unpriced_runs: u32,
24    pub(crate) duration_secs: f64,
25    /// Aggregate privacy-preserving trace facts joined by task and attempt.
26    pub(crate) trace_summary: Option<HarnessTraceSummary>,
27    /// Mean cost over priced attempts only. `None` when every attempt is unpriced.
28    pub(crate) mean_cost_per_attempt: Option<f64>,
29    /// Total known cost divided by successful attempts. `None` when any
30    /// attempt is unpriced (unknown cost is not free) or nothing passed.
31    pub(crate) cost_per_solve: Option<f64>,
32    /// Mean gross tokens (input + output) over attempts that carry a trace.
33    pub(crate) mean_tokens_per_attempt: Option<f64>,
34    /// Mean model turns over attempts that carry a trace.
35    pub(crate) mean_turns_per_attempt: Option<f64>,
36}
37
38#[derive(Debug, Clone, Serialize)]
39pub struct EvalReport {
40    pub(crate) generated_at: String,
41    pub(crate) suites: Vec<SuiteReport>,
42}
43
44impl EvalReport {
45    /// Create a report from generated-at metadata and suite reports.
46    pub fn new(generated_at: impl Into<String>, suites: Vec<SuiteReport>) -> Self {
47        Self { generated_at: generated_at.into(), suites }
48    }
49
50    pub fn to_markdown(&self) -> String {
51        let mut out = String::new();
52        out.push_str("# Eval Report\n\n");
53        for s in &self.suites {
54            out.push_str(&format!("## {}\n\n", s.suite_name));
55            out.push_str(&format!(
56                "- Aggregate: pass@k={:.1}%; pass^k={:.1}%\n",
57                s.aggregate.pass_at_k * 100.0,
58                s.aggregate.pass_power_k * 100.0
59            ));
60            if let Some(cost) = s.cost_usd {
61                out.push_str(&format!("- Cost (known): ${cost:.6}; unpriced attempts: {}\n", s.unpriced_runs));
62            } else if s.unpriced_runs > 0 {
63                out.push_str(&format!("- Cost (known): unavailable; unpriced attempts: {}\n", s.unpriced_runs));
64            }
65            if s.mean_cost_per_attempt.is_some()
66                || s.cost_per_solve.is_some()
67                || s.mean_tokens_per_attempt.is_some()
68                || s.mean_turns_per_attempt.is_some()
69            {
70                let mut parts = Vec::new();
71                if let Some(cost_per_solve) = s.cost_per_solve {
72                    parts.push(format!("cost/solve ${cost_per_solve:.4}"));
73                }
74                if let Some(mean_cost) = s.mean_cost_per_attempt {
75                    parts.push(format!("${mean_cost:.4} per priced attempt"));
76                }
77                if let Some(mean_tokens) = s.mean_tokens_per_attempt {
78                    parts.push(format!("{mean_tokens:.0} tokens"));
79                }
80                if let Some(mean_turns) = s.mean_turns_per_attempt {
81                    parts.push(format!("{mean_turns:.1} turns"));
82                }
83                out.push_str(&format!("- Efficiency: {}\n", parts.join(" · ")));
84            }
85            if let Some(trace) = &s.trace_summary {
86                if let Some(mean_ms) = trace.latency.mean_ms {
87                    out.push_str(&format!(
88                        "- Trace: {} turns, {} tool calls, mean latency {:.1} ms\n",
89                        trace.turns, trace.tool_calls, mean_ms
90                    ));
91                } else {
92                    out.push_str(&format!("- Trace: {} turns, {} tool calls\n", trace.turns, trace.tool_calls));
93                }
94            }
95            out.push_str("| Task | Category | pass@k | pass^k | passed/total |\n");
96            out.push_str("|------|----------|--------|--------|-------------|\n");
97            for t in &s.task_reports {
98                out.push_str(&format!(
99                    "| {} | {} | {:.1}% | {:.1}% | {}/{} |\n",
100                    t.task_id,
101                    t.category.as_str(),
102                    t.metric.pass_at_k * 100.0,
103                    t.metric.pass_power_k * 100.0,
104                    t.metric.passed_runs,
105                    t.metric.total_runs
106                ));
107            }
108            out.push('\n');
109        }
110        out
111    }
112}
113
114pub fn build_task_report(task_id: &str, _name: &str, category: EvalCategory, metric: EvalMetric) -> TaskReport {
115    TaskReport {
116        task_id: task_id.into(),
117        category: category.label().into(),
118        metric,
119    }
120}
121
122/// Cost-efficiency aggregates for a suite (HarnessTax-style frontier metrics).
123///
124/// Unknown pricing is never treated as zero: any unpriced attempt makes
125/// `cost_per_solve` `None` so the report cannot claim a complete cost basis.
126#[derive(Debug, Clone, Copy, Default, PartialEq)]
127pub struct CostEfficiency {
128    pub mean_cost_per_attempt: Option<f64>,
129    pub cost_per_solve: Option<f64>,
130    pub mean_tokens_per_attempt: Option<f64>,
131    pub mean_turns_per_attempt: Option<f64>,
132}
133
134/// Finite, non-negative attempt cost. Unpriced is not free.
135pub fn priced_cost(cost_usd: Option<f64>) -> Option<f64> {
136    cost_usd.filter(|value| value.is_finite() && *value >= 0.0)
137}
138
139impl CostEfficiency {
140    /// Compute efficiency metrics from attempt results.
141    ///
142    /// `cost_usd` / `trace_summary` come from each `EvalRunResult`. Token and
143    /// turn means use only attempts that carry a trace summary.
144    pub fn from_runs<'a>(results: impl IntoIterator<Item = &'a crate::task::EvalRunResult>) -> Self {
145        let mut total_cost = 0.0_f64;
146        let mut known_cost_runs = 0_u32;
147        let mut unpriced_runs = 0_u32;
148        let mut passed_runs = 0_u32;
149        let mut token_sum = 0.0_f64;
150        let mut token_count = 0_u32;
151        let mut turn_sum = 0.0_f64;
152        let mut turn_count = 0_u32;
153
154        for result in results {
155            if result.outcome == crate::task::RunOutcome::Pass {
156                passed_runs = passed_runs.saturating_add(1);
157            }
158            match priced_cost(result.cost_usd) {
159                Some(cost) => {
160                    total_cost += cost;
161                    known_cost_runs = known_cost_runs.saturating_add(1);
162                }
163                None => {
164                    unpriced_runs = unpriced_runs.saturating_add(1);
165                }
166            }
167            if let Some(summary) = &result.trace_summary {
168                let gross = (summary
169                    .token_usage
170                    .input_tokens
171                    .saturating_add(summary.token_usage.output_tokens)) as f64;
172                token_sum += gross;
173                token_count = token_count.saturating_add(1);
174                turn_sum += summary.turns as f64;
175                turn_count = turn_count.saturating_add(1);
176            }
177        }
178
179        let mean_cost_per_attempt = (known_cost_runs > 0).then(|| total_cost / f64::from(known_cost_runs));
180        let cost_per_solve =
181            (unpriced_runs == 0 && passed_runs > 0 && known_cost_runs > 0).then(|| total_cost / f64::from(passed_runs));
182        let mean_tokens_per_attempt = (token_count > 0).then(|| token_sum / f64::from(token_count));
183        let mean_turns_per_attempt = (turn_count > 0).then(|| turn_sum / f64::from(turn_count));
184
185        Self {
186            mean_cost_per_attempt,
187            cost_per_solve,
188            mean_tokens_per_attempt,
189            mean_turns_per_attempt,
190        }
191    }
192}
193
194#[cfg(test)]
195mod tests {
196    use super::*;
197    use crate::metric::EvalMetric;
198    use crate::task::EvalCategory;
199
200    #[test]
201    fn to_markdown_renders_tasks_and_aggregate() {
202        let report = EvalReport {
203            generated_at: "2026-01-01".into(),
204            suites: vec![SuiteReport {
205                suite_id: "s1".into(),
206                suite_name: "demo".into(),
207                task_reports: vec![TaskReport {
208                    task_id: "t1".into(),
209                    category: "Capability".into(),
210                    metric: EvalMetric {
211                        pass_at_k: 0.5,
212                        pass_power_k: 0.25,
213                        pass_all_k: 0.0,
214                        k: 1,
215                        total_runs: 2,
216                        passed_runs: 1,
217                        task_id: "t1".into(),
218                    },
219                }],
220                aggregate: EvalMetric {
221                    pass_at_k: 0.5,
222                    pass_power_k: 0.25,
223                    pass_all_k: 0.0,
224                    k: 1,
225                    total_runs: 2,
226                    passed_runs: 1,
227                    task_id: "aggregate".into(),
228                },
229                capability_metrics: EvalMetric {
230                    pass_at_k: 0.5,
231                    pass_power_k: 0.25,
232                    pass_all_k: 0.0,
233                    k: 1,
234                    total_runs: 2,
235                    passed_runs: 1,
236                    task_id: "cap".into(),
237                },
238                regression_metrics: EvalMetric {
239                    pass_at_k: 0.0,
240                    pass_power_k: 0.0,
241                    pass_all_k: 0.0,
242                    k: 0,
243                    total_runs: 0,
244                    passed_runs: 0,
245                    task_id: "reg".into(),
246                },
247                cost_usd: Some(0.0),
248                unpriced_runs: 0,
249                duration_secs: 0.0,
250                trace_summary: None,
251                mean_cost_per_attempt: None,
252                cost_per_solve: None,
253                mean_tokens_per_attempt: None,
254                mean_turns_per_attempt: None,
255            }],
256        };
257        let md = report.to_markdown();
258        assert!(md.contains("# Eval Report"));
259        assert!(md.contains("demo"));
260        assert!(md.contains("t1"));
261        assert!(md.contains("Capability"));
262        assert!(md.contains("pass^k"));
263    }
264
265    #[test]
266    fn build_task_report_maps_category() {
267        let tr = build_task_report(
268            "t1",
269            "name",
270            EvalCategory::Regression,
271            EvalMetric {
272                pass_at_k: 1.0,
273                pass_power_k: 1.0,
274                pass_all_k: 1.0,
275                k: 1,
276                total_runs: 1,
277                passed_runs: 1,
278                task_id: "t1".into(),
279            },
280        );
281        assert_eq!(tr.task_id, "t1");
282        assert_eq!(tr.category, "Regression");
283    }
284
285    fn run(outcome: crate::task::RunOutcome, cost: Option<f64>, turns: u64, tokens: u64) -> crate::task::EvalRunResult {
286        use crate::trace_analyzer::{HarnessTraceSummary, TokenUsage};
287        crate::task::EvalRunResult {
288            task_id: "t1".into(),
289            outcome,
290            error_message: None,
291            duration_secs: 1.0,
292            attempt: 1,
293            cost_usd: cost,
294            transcript_path: None,
295            trace_summary: Some(HarnessTraceSummary {
296                turns,
297                token_usage: TokenUsage {
298                    input_tokens: tokens / 2,
299                    output_tokens: tokens - tokens / 2,
300                    ..TokenUsage::default()
301                },
302                ..HarnessTraceSummary::default()
303            }),
304        }
305    }
306
307    #[test]
308    fn cost_efficiency_uses_total_cost_over_passed_attempts() {
309        let results = vec![
310            run(crate::task::RunOutcome::Pass, Some(0.02), 10, 1_000),
311            run(crate::task::RunOutcome::Fail, Some(0.04), 20, 3_000),
312        ];
313        let eff = CostEfficiency::from_runs(&results);
314        assert_eq!(eff.cost_per_solve, Some(0.06));
315        assert_eq!(eff.mean_cost_per_attempt, Some(0.03));
316        assert_eq!(eff.mean_tokens_per_attempt, Some(2_000.0));
317        assert_eq!(eff.mean_turns_per_attempt, Some(15.0));
318    }
319
320    #[test]
321    fn cost_per_solve_is_none_when_any_attempt_is_unpriced() {
322        let results = vec![
323            run(crate::task::RunOutcome::Pass, Some(0.02), 5, 100),
324            run(crate::task::RunOutcome::Pass, None, 5, 100),
325        ];
326        let eff = CostEfficiency::from_runs(&results);
327        assert_eq!(eff.cost_per_solve, None);
328        // Mean cost still reflects only the priced attempts.
329        assert_eq!(eff.mean_cost_per_attempt, Some(0.02));
330    }
331
332    #[test]
333    fn cost_per_solve_is_none_when_nothing_passed() {
334        let results = vec![run(crate::task::RunOutcome::Fail, Some(0.10), 3, 50)];
335        let eff = CostEfficiency::from_runs(&results);
336        assert_eq!(eff.cost_per_solve, None);
337        assert_eq!(eff.mean_cost_per_attempt, Some(0.10));
338    }
339
340    #[test]
341    fn token_and_turn_means_are_none_without_traces() {
342        let mut result = run(crate::task::RunOutcome::Pass, Some(0.01), 1, 1);
343        result.trace_summary = None;
344        let eff = CostEfficiency::from_runs(std::slice::from_ref(&result));
345        assert_eq!(eff.mean_tokens_per_attempt, None);
346        assert_eq!(eff.mean_turns_per_attempt, None);
347    }
348
349    #[test]
350    fn markdown_renders_efficiency_line() {
351        let report = EvalReport {
352            generated_at: "2026-01-01".into(),
353            suites: vec![SuiteReport {
354                suite_id: "s1".into(),
355                suite_name: "demo".into(),
356                task_reports: Vec::new(),
357                aggregate: EvalMetric {
358                    pass_at_k: 1.0,
359                    pass_power_k: 1.0,
360                    pass_all_k: 1.0,
361                    k: 1,
362                    total_runs: 1,
363                    passed_runs: 1,
364                    task_id: "aggregate".into(),
365                },
366                capability_metrics: EvalMetric {
367                    pass_at_k: 1.0,
368                    pass_power_k: 1.0,
369                    pass_all_k: 1.0,
370                    k: 1,
371                    total_runs: 1,
372                    passed_runs: 1,
373                    task_id: "cap".into(),
374                },
375                regression_metrics: EvalMetric {
376                    pass_at_k: 0.0,
377                    pass_power_k: 0.0,
378                    pass_all_k: 0.0,
379                    k: 0,
380                    total_runs: 0,
381                    passed_runs: 0,
382                    task_id: "reg".into(),
383                },
384                cost_usd: Some(0.05),
385                unpriced_runs: 0,
386                duration_secs: 1.0,
387                trace_summary: None,
388                mean_cost_per_attempt: Some(0.05),
389                cost_per_solve: Some(0.05),
390                mean_tokens_per_attempt: Some(12_000.0),
391                mean_turns_per_attempt: Some(8.5),
392            }],
393        };
394        let md = report.to_markdown();
395        assert!(
396            md.contains("- Efficiency: cost/solve $0.0500 · $0.0500 per priced attempt · 12000 tokens · 8.5 turns"),
397            "missing efficiency line: {md}"
398        );
399    }
400}