llm_browser_testkit/events.rs
1//! Typed run events emitted by the runner.
2//!
3//! The runner no longer prints directly: every interesting point in a run
4//! (test start/end, step start/end, LLM call, budget warning) is emitted as a
5//! [`TestEvent`]. Sinks attached to the [`crate::reporting::Reporter`] render
6//! these events for humans (console), machines (NDJSON log file, JUnit XML),
7//! CI systems (GitHub workflow commands) and profilers (Perfetto trace).
8//!
9//! The event stream doubles as the run's source of truth: JUnit and trace
10//! files are derived from it, and a `--log-file` NDJSON capture alone is
11//! enough to replay or analyze a run.
12
13use serde::Serialize;
14
15/// Outcome of a single step.
16#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)]
17#[serde(rename_all = "snake_case")]
18pub enum StepStatus {
19 /// Step executed successfully and all assertions passed.
20 Passed,
21 /// Step execution or assertion failed.
22 Failed,
23 /// Step was skipped.
24 Skipped,
25}
26
27/// One event in the run's lifecycle.
28///
29/// Serialized as a flat JSON object with a `type` discriminator, e.g.
30/// `{"type":"step_finished","test":"login","status":"passed",...}`. The
31/// reporter adds a `ts` (epoch milliseconds) field on emission.
32#[derive(Debug, Clone, Serialize)]
33#[serde(tag = "type", rename_all = "snake_case")]
34pub enum TestEvent {
35 /// The run is about to execute `total_tests` tests.
36 RunStarted {
37 /// Number of tests in the run.
38 total_tests: u32,
39 },
40 /// A test started.
41 TestStarted {
42 /// Test name (includes viewport-matrix suffix when expanded).
43 test: String,
44 },
45 /// A step inside a test started.
46 StepStarted {
47 /// Enclosing test name.
48 test: String,
49 /// Zero-based step index within the test.
50 index: u32,
51 /// Human-readable label, e.g. `[click] the sign-in button`.
52 label: String,
53 },
54 /// A step finished.
55 StepFinished {
56 /// Enclosing test name.
57 test: String,
58 /// Zero-based step index within the test.
59 index: u32,
60 /// Human-readable label, e.g. `[click] the sign-in button`.
61 label: String,
62 /// Outcome of the step.
63 status: StepStatus,
64 /// Wall-clock duration of the step.
65 duration_ms: u64,
66 /// Human-readable result message (includes the page-state excerpt
67 /// and the failure message for failed steps).
68 message: String,
69 /// Multi-line failure diagnostics (URL, title, visible text, alerts)
70 /// — `None` for non-failed steps.
71 diagnostics: Option<String>,
72 /// Path of the failure screenshot, if one was written.
73 screenshot: Option<String>,
74 },
75 /// An LLM call started (one per attempted endpoint chain).
76 LlmCallStarted {
77 /// Enclosing test name.
78 test: String,
79 /// Zero-based step index within the test.
80 index: u32,
81 /// Endpoint name (or chain entry) being called.
82 endpoint: String,
83 /// Model name sent in the request.
84 model: String,
85 /// Call purpose: `targeting` or `assertion`.
86 purpose: String,
87 },
88 /// An LLM call finished (success or exhaustion of all attempts).
89 LlmCallFinished {
90 /// Enclosing test name.
91 test: String,
92 /// Zero-based step index within the test.
93 index: u32,
94 /// Endpoint that answered (or the last one tried on failure).
95 endpoint: String,
96 /// Model name used.
97 model: String,
98 /// Call purpose: `targeting` or `assertion`.
99 purpose: String,
100 /// Whether an answer was obtained.
101 ok: bool,
102 /// Wall-clock duration of the call including retries.
103 duration_ms: u64,
104 /// Input (prompt) tokens billed.
105 input_tokens: u64,
106 /// Output (completion) tokens billed.
107 output_tokens: u64,
108 /// Input tokens served from the provider's prompt cache.
109 cached_input_tokens: u64,
110 /// Input tokens written to the provider's prompt cache.
111 cache_creation_input_tokens: u64,
112 /// Computed cost in USD.
113 cost: f64,
114 /// Error message when `ok` is false.
115 error: Option<String>,
116 },
117 /// A test finished.
118 TestFinished {
119 /// Test name.
120 test: String,
121 /// Number of passed steps.
122 passed: u32,
123 /// Number of failed steps.
124 failed: u32,
125 /// Number of skipped steps.
126 skipped: u32,
127 /// Wall-clock duration of the whole test.
128 duration_ms: u64,
129 /// Total cost of the test in USD.
130 cost: f64,
131 /// Total tokens consumed by the test.
132 tokens: u64,
133 /// Input (prompt) tokens consumed by the test.
134 input_tokens: u64,
135 /// Output (completion) tokens consumed by the test.
136 output_tokens: u64,
137 /// Input tokens served from provider prompt caches.
138 cached_input_tokens: u64,
139 /// Input tokens written to provider prompt caches.
140 cache_creation_input_tokens: u64,
141 /// Models used by the test, sorted and deduplicated.
142 models: Vec<String>,
143 /// Total LLM/MCP/agent calls made by the test.
144 calls: u64,
145 },
146 /// The whole run finished.
147 RunFinished {
148 /// Tests that passed.
149 tests_passed: u32,
150 /// Tests that failed.
151 tests_failed: u32,
152 /// Steps that passed.
153 steps_passed: u32,
154 /// Steps that failed.
155 steps_failed: u32,
156 /// Steps that were skipped.
157 steps_skipped: u32,
158 /// Total cost in USD across all tests.
159 total_cost: f64,
160 /// Total tokens consumed.
161 total_tokens: u64,
162 /// Total input (prompt) tokens consumed.
163 total_input_tokens: u64,
164 /// Total output (completion) tokens consumed.
165 total_output_tokens: u64,
166 /// Total input tokens served from provider prompt caches.
167 total_cached_input_tokens: u64,
168 /// Total input tokens written to provider prompt caches.
169 total_cache_creation_input_tokens: u64,
170 /// Models used across the run, sorted and deduplicated.
171 models: Vec<String>,
172 /// Total calls made.
173 total_calls: u64,
174 },
175 /// A non-fatal warning (budget soft-exceeded, feature not enabled, ...).
176 Warning {
177 /// Human-readable warning text.
178 message: String,
179 },
180}
181
182#[cfg(test)]
183mod tests {
184 use super::{StepStatus, TestEvent};
185
186 #[test]
187 fn test_step_status_serializes_snake_case() {
188 assert_eq!(
189 serde_json::to_value(StepStatus::Passed).unwrap(),
190 serde_json::json!("passed")
191 );
192 assert_eq!(
193 serde_json::to_value(StepStatus::Failed).unwrap(),
194 serde_json::json!("failed")
195 );
196 assert_eq!(
197 serde_json::to_value(StepStatus::Skipped).unwrap(),
198 serde_json::json!("skipped")
199 );
200 }
201
202 #[test]
203 fn test_event_serializes_with_type_tag() {
204 let event = TestEvent::StepFinished {
205 test: "login".into(),
206 index: 2,
207 label: "[click] the button".into(),
208 status: StepStatus::Failed,
209 duration_ms: 1234,
210 message: "element #btn not found".into(),
211 diagnostics: Some(" │ url: http://x".into()),
212 screenshot: Some("artifacts/x.png".into()),
213 };
214 let value = serde_json::to_value(&event).unwrap();
215 assert_eq!(value["type"], "step_finished");
216 assert_eq!(value["test"], "login");
217 assert_eq!(value["index"], 2);
218 assert_eq!(value["label"], "[click] the button");
219 assert_eq!(value["status"], "failed");
220 assert_eq!(value["duration_ms"], 1234);
221 assert_eq!(value["message"], "element #btn not found");
222 assert_eq!(value["diagnostics"], " │ url: http://x");
223 assert_eq!(value["screenshot"], "artifacts/x.png");
224 }
225
226 #[test]
227 fn test_run_started_shape() {
228 let value = serde_json::to_value(TestEvent::RunStarted { total_tests: 3 }).unwrap();
229 assert_eq!(value["type"], "run_started");
230 assert_eq!(value["total_tests"], 3);
231 }
232
233 #[test]
234 fn test_llm_call_finished_shape() {
235 let value = serde_json::to_value(TestEvent::LlmCallFinished {
236 test: "login".into(),
237 index: 0,
238 endpoint: "default".into(),
239 model: "deepseek".into(),
240 purpose: "targeting".into(),
241 ok: false,
242 duration_ms: 900,
243 input_tokens: 100,
244 output_tokens: 0,
245 cached_input_tokens: 40,
246 cache_creation_input_tokens: 7,
247 cost: 0.0012,
248 error: Some("HTTP 429: slow down".into()),
249 })
250 .unwrap();
251 assert_eq!(value["type"], "llm_call_finished");
252 assert_eq!(value["endpoint"], "default");
253 assert_eq!(value["model"], "deepseek");
254 assert_eq!(value["purpose"], "targeting");
255 assert_eq!(value["ok"], false);
256 assert_eq!(value["cached_input_tokens"], 40);
257 assert_eq!(value["cache_creation_input_tokens"], 7);
258 assert_eq!(value["cost"], 0.0012);
259 assert_eq!(value["error"], "HTTP 429: slow down");
260 }
261
262 #[test]
263 fn test_run_finished_shape() {
264 let value = serde_json::to_value(TestEvent::RunFinished {
265 tests_passed: 2,
266 tests_failed: 1,
267 steps_passed: 9,
268 steps_failed: 1,
269 steps_skipped: 2,
270 total_cost: 0.05,
271 total_tokens: 5000,
272 total_input_tokens: 4000,
273 total_output_tokens: 1000,
274 total_cached_input_tokens: 500,
275 total_cache_creation_input_tokens: 250,
276 models: vec!["deepseek".into()],
277 total_calls: 7,
278 })
279 .unwrap();
280 assert_eq!(value["type"], "run_finished");
281 assert_eq!(value["tests_passed"], 2);
282 assert_eq!(value["steps_failed"], 1);
283 assert_eq!(value["total_cost"], 0.05);
284 assert_eq!(value["total_input_tokens"], 4000);
285 assert_eq!(value["total_cached_input_tokens"], 500);
286 assert_eq!(value["total_cache_creation_input_tokens"], 250);
287 assert_eq!(value["models"][0], "deepseek");
288 }
289
290 #[test]
291 fn test_warning_shape() {
292 let value = serde_json::to_value(TestEvent::Warning {
293 message: "budget soft".into(),
294 })
295 .unwrap();
296 assert_eq!(value["type"], "warning");
297 assert_eq!(value["message"], "budget soft");
298 }
299}