Skip to main content

vtcode_eval/trace_analyzer/
model.rs

1use std::collections::BTreeMap;
2
3use serde::{Deserialize, Serialize};
4
5/// Aggregate token and prompt-cache usage found in a trace.
6#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq, Eq)]
7pub struct TokenUsage {
8    /// Total prompt/input tokens.
9    pub input_tokens: u64,
10    /// Total generated/output tokens.
11    pub output_tokens: u64,
12    /// Total prompt tokens served from cache.
13    pub cached_input_tokens: u64,
14    /// Total tokens used to create cache entries.
15    pub cache_creation_tokens: u64,
16    /// Total generated reasoning tokens when the provider reports them.
17    pub reasoning_tokens: u64,
18}
19
20/// Statistics over recorded latency samples, in milliseconds.
21#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
22pub struct LatencyStatistics {
23    /// Number of latency samples.
24    pub count: u64,
25    /// Sum of latency samples.
26    pub total_ms: u64,
27    /// Arithmetic mean, or `None` when no samples were recorded.
28    pub mean_ms: Option<f64>,
29    /// Median from the bounded latency reservoir, or `None` when no samples were recorded.
30    pub p50_ms: Option<u64>,
31    /// 95th percentile from the bounded latency reservoir, or `None` when no samples were recorded.
32    pub p95_ms: Option<u64>,
33    /// Largest recorded sample, or `None` when no samples were recorded.
34    pub max_ms: Option<u64>,
35}
36
37/// Redacted aggregate facts extracted from DeepSeek or VT Code JSONL traces.
38#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq)]
39pub struct HarnessTraceSummary {
40    /// Number of execution turns.
41    pub turns: u64,
42    /// Number of agent steps.
43    pub steps: u64,
44    /// Number of tool calls.
45    pub tool_calls: u64,
46    /// Tool name to invocation count.
47    pub tool_counts: BTreeMap<String, u64>,
48    /// Canonical error category to count.
49    pub error_categories: BTreeMap<String, u64>,
50    /// Latency aggregate for all recognized samples.
51    pub latency: LatencyStatistics,
52    /// Total UTF-8 byte length of tool outputs, without retaining output text.
53    pub output_bytes: u64,
54    /// Number of calls after the first call for each tool name.
55    pub repeated_calls: u64,
56    /// Repeated calls grouped by tool name.
57    pub repeated_tool_counts: BTreeMap<String, u64>,
58    /// Aggregate model token usage.
59    pub token_usage: TokenUsage,
60    /// Lines that were not valid JSON objects.
61    pub malformed_lines: u64,
62    /// Valid JSON objects with no recognized trace shape.
63    pub unrecognized_lines: u64,
64}
65
66impl HarnessTraceSummary {
67    /// Merge another privacy-preserving trace summary into this aggregate.
68    /// Percentiles are retained only for a single source because they cannot
69    /// be combined exactly without retaining raw latency samples.
70    pub fn merge(&mut self, other: &Self) {
71        self.turns = self.turns.saturating_add(other.turns);
72        self.steps = self.steps.saturating_add(other.steps);
73        self.tool_calls = self.tool_calls.saturating_add(other.tool_calls);
74        self.output_bytes = self.output_bytes.saturating_add(other.output_bytes);
75        self.repeated_calls = self.repeated_calls.saturating_add(other.repeated_calls);
76        self.malformed_lines = self.malformed_lines.saturating_add(other.malformed_lines);
77        self.unrecognized_lines = self.unrecognized_lines.saturating_add(other.unrecognized_lines);
78
79        for (tool, count) in &other.tool_counts {
80            let entry = self.tool_counts.entry(tool.clone()).or_default();
81            *entry = entry.saturating_add(*count);
82        }
83        for (tool, count) in &other.repeated_tool_counts {
84            let entry = self.repeated_tool_counts.entry(tool.clone()).or_default();
85            *entry = entry.saturating_add(*count);
86        }
87        for (category, count) in &other.error_categories {
88            let entry = self.error_categories.entry(category.clone()).or_default();
89            *entry = entry.saturating_add(*count);
90        }
91
92        let previous_latency_count = self.latency.count;
93        let combined_count = previous_latency_count.saturating_add(other.latency.count);
94        self.latency.total_ms = self.latency.total_ms.saturating_add(other.latency.total_ms);
95        self.latency.count = combined_count;
96        self.latency.mean_ms = (combined_count > 0).then_some(self.latency.total_ms as f64 / combined_count as f64);
97        self.latency.max_ms = match (self.latency.max_ms, other.latency.max_ms) {
98            (Some(left), Some(right)) => Some(left.max(right)),
99            (left, right) => left.or(right),
100        };
101        if previous_latency_count > 0 && other.latency.count > 0 {
102            self.latency.p50_ms = None;
103            self.latency.p95_ms = None;
104        } else if previous_latency_count == 0 {
105            self.latency.p50_ms = other.latency.p50_ms;
106            self.latency.p95_ms = other.latency.p95_ms;
107        }
108
109        self.token_usage.input_tokens = self.token_usage.input_tokens.saturating_add(other.token_usage.input_tokens);
110        self.token_usage.output_tokens = self.token_usage.output_tokens.saturating_add(other.token_usage.output_tokens);
111        self.token_usage.cached_input_tokens = self
112            .token_usage
113            .cached_input_tokens
114            .saturating_add(other.token_usage.cached_input_tokens);
115        self.token_usage.cache_creation_tokens = self
116            .token_usage
117            .cache_creation_tokens
118            .saturating_add(other.token_usage.cache_creation_tokens);
119        self.token_usage.reasoning_tokens = self
120            .token_usage
121            .reasoning_tokens
122            .saturating_add(other.token_usage.reasoning_tokens);
123    }
124}