Skip to main content

llm_browser_testkit/
events.rs

1//! Typed run events emitted by the runner.
2//!
3//! The runner no longer prints directly: every interesting point in a run
4//! (test start/end, step start/end, LLM call, budget warning) is emitted as a
5//! [`TestEvent`]. Sinks attached to the [`crate::reporting::Reporter`] render
6//! these events for humans (console), machines (NDJSON log file, JUnit XML),
7//! CI systems (GitHub workflow commands) and profilers (Perfetto trace).
8//!
9//! The event stream doubles as the run's source of truth: JUnit and trace
10//! files are derived from it, and a `--log-file` NDJSON capture alone is
11//! enough to replay or analyze a run.
12
13use serde::Serialize;
14
15/// Outcome of a single step.
16#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)]
17#[serde(rename_all = "snake_case")]
18pub enum StepStatus {
19    /// Step executed successfully and all assertions passed.
20    Passed,
21    /// Step execution or assertion failed.
22    Failed,
23    /// Step was skipped.
24    Skipped,
25}
26
27/// One event in the run's lifecycle.
28///
29/// Serialized as a flat JSON object with a `type` discriminator, e.g.
30/// `{"type":"step_finished","test":"login","status":"passed",...}`. The
31/// reporter adds a `ts` (epoch milliseconds) field on emission.
32#[derive(Debug, Clone, Serialize)]
33#[serde(tag = "type", rename_all = "snake_case")]
34pub enum TestEvent {
35    /// The run is about to execute `total_tests` tests.
36    RunStarted {
37        /// Number of tests in the run.
38        total_tests: u32,
39    },
40    /// A test started.
41    TestStarted {
42        /// Test name (includes viewport-matrix suffix when expanded).
43        test: String,
44    },
45    /// A step inside a test started.
46    StepStarted {
47        /// Enclosing test name.
48        test: String,
49        /// Zero-based step index within the test.
50        index: u32,
51        /// Human-readable label, e.g. `[click] the sign-in button`.
52        label: String,
53    },
54    /// A step finished.
55    StepFinished {
56        /// Enclosing test name.
57        test: String,
58        /// Zero-based step index within the test.
59        index: u32,
60        /// Human-readable label, e.g. `[click] the sign-in button`.
61        label: String,
62        /// Outcome of the step.
63        status: StepStatus,
64        /// Wall-clock duration of the step.
65        duration_ms: u64,
66        /// Human-readable result message (includes the page-state excerpt
67        /// and the failure message for failed steps).
68        message: String,
69        /// Multi-line failure diagnostics (URL, title, visible text, alerts)
70        /// — `None` for non-failed steps.
71        diagnostics: Option<String>,
72        /// Path of the failure screenshot, if one was written.
73        screenshot: Option<String>,
74    },
75    /// An LLM call started (one per attempted endpoint chain).
76    LlmCallStarted {
77        /// Enclosing test name.
78        test: String,
79        /// Zero-based step index within the test.
80        index: u32,
81        /// Endpoint name (or chain entry) being called.
82        endpoint: String,
83        /// Model name sent in the request.
84        model: String,
85        /// Call purpose: `targeting` or `assertion`.
86        purpose: String,
87    },
88    /// An LLM call finished (success or exhaustion of all attempts).
89    LlmCallFinished {
90        /// Enclosing test name.
91        test: String,
92        /// Zero-based step index within the test.
93        index: u32,
94        /// Endpoint that answered (or the last one tried on failure).
95        endpoint: String,
96        /// Model name used.
97        model: String,
98        /// Call purpose: `targeting` or `assertion`.
99        purpose: String,
100        /// Whether an answer was obtained.
101        ok: bool,
102        /// Wall-clock duration of the call including retries.
103        duration_ms: u64,
104        /// Input (prompt) tokens billed.
105        input_tokens: u64,
106        /// Output (completion) tokens billed.
107        output_tokens: u64,
108        /// Input tokens served from the provider's prompt cache.
109        cached_input_tokens: u64,
110        /// Input tokens written to the provider's prompt cache.
111        cache_creation_input_tokens: u64,
112        /// Computed cost in USD.
113        cost: f64,
114        /// Error message when `ok` is false.
115        error: Option<String>,
116    },
117    /// A test finished.
118    TestFinished {
119        /// Test name.
120        test: String,
121        /// Number of passed steps.
122        passed: u32,
123        /// Number of failed steps.
124        failed: u32,
125        /// Number of skipped steps.
126        skipped: u32,
127        /// Wall-clock duration of the whole test.
128        duration_ms: u64,
129        /// Total cost of the test in USD.
130        cost: f64,
131        /// Total tokens consumed by the test.
132        tokens: u64,
133        /// Input (prompt) tokens consumed by the test.
134        input_tokens: u64,
135        /// Output (completion) tokens consumed by the test.
136        output_tokens: u64,
137        /// Input tokens served from provider prompt caches.
138        cached_input_tokens: u64,
139        /// Input tokens written to provider prompt caches.
140        cache_creation_input_tokens: u64,
141        /// Models used by the test, sorted and deduplicated.
142        models: Vec<String>,
143        /// Total LLM/MCP/agent calls made by the test.
144        calls: u64,
145    },
146    /// The whole run finished.
147    RunFinished {
148        /// Tests that passed.
149        tests_passed: u32,
150        /// Tests that failed.
151        tests_failed: u32,
152        /// Steps that passed.
153        steps_passed: u32,
154        /// Steps that failed.
155        steps_failed: u32,
156        /// Steps that were skipped.
157        steps_skipped: u32,
158        /// Total cost in USD across all tests.
159        total_cost: f64,
160        /// Total tokens consumed.
161        total_tokens: u64,
162        /// Total input (prompt) tokens consumed.
163        total_input_tokens: u64,
164        /// Total output (completion) tokens consumed.
165        total_output_tokens: u64,
166        /// Total input tokens served from provider prompt caches.
167        total_cached_input_tokens: u64,
168        /// Total input tokens written to provider prompt caches.
169        total_cache_creation_input_tokens: u64,
170        /// Models used across the run, sorted and deduplicated.
171        models: Vec<String>,
172        /// Total calls made.
173        total_calls: u64,
174    },
175    /// A non-fatal warning (budget soft-exceeded, feature not enabled, ...).
176    Warning {
177        /// Human-readable warning text.
178        message: String,
179    },
180}
181
182#[cfg(test)]
183mod tests {
184    use super::{StepStatus, TestEvent};
185
186    #[test]
187    fn test_step_status_serializes_snake_case() {
188        assert_eq!(
189            serde_json::to_value(StepStatus::Passed).unwrap(),
190            serde_json::json!("passed")
191        );
192        assert_eq!(
193            serde_json::to_value(StepStatus::Failed).unwrap(),
194            serde_json::json!("failed")
195        );
196        assert_eq!(
197            serde_json::to_value(StepStatus::Skipped).unwrap(),
198            serde_json::json!("skipped")
199        );
200    }
201
202    #[test]
203    fn test_event_serializes_with_type_tag() {
204        let event = TestEvent::StepFinished {
205            test: "login".into(),
206            index: 2,
207            label: "[click] the button".into(),
208            status: StepStatus::Failed,
209            duration_ms: 1234,
210            message: "element #btn not found".into(),
211            diagnostics: Some("    │ url: http://x".into()),
212            screenshot: Some("artifacts/x.png".into()),
213        };
214        let value = serde_json::to_value(&event).unwrap();
215        assert_eq!(value["type"], "step_finished");
216        assert_eq!(value["test"], "login");
217        assert_eq!(value["index"], 2);
218        assert_eq!(value["label"], "[click] the button");
219        assert_eq!(value["status"], "failed");
220        assert_eq!(value["duration_ms"], 1234);
221        assert_eq!(value["message"], "element #btn not found");
222        assert_eq!(value["diagnostics"], "    │ url: http://x");
223        assert_eq!(value["screenshot"], "artifacts/x.png");
224    }
225
226    #[test]
227    fn test_run_started_shape() {
228        let value = serde_json::to_value(TestEvent::RunStarted { total_tests: 3 }).unwrap();
229        assert_eq!(value["type"], "run_started");
230        assert_eq!(value["total_tests"], 3);
231    }
232
233    #[test]
234    fn test_llm_call_finished_shape() {
235        let value = serde_json::to_value(TestEvent::LlmCallFinished {
236            test: "login".into(),
237            index: 0,
238            endpoint: "default".into(),
239            model: "deepseek".into(),
240            purpose: "targeting".into(),
241            ok: false,
242            duration_ms: 900,
243            input_tokens: 100,
244            output_tokens: 0,
245            cached_input_tokens: 40,
246            cache_creation_input_tokens: 7,
247            cost: 0.0012,
248            error: Some("HTTP 429: slow down".into()),
249        })
250        .unwrap();
251        assert_eq!(value["type"], "llm_call_finished");
252        assert_eq!(value["endpoint"], "default");
253        assert_eq!(value["model"], "deepseek");
254        assert_eq!(value["purpose"], "targeting");
255        assert_eq!(value["ok"], false);
256        assert_eq!(value["cached_input_tokens"], 40);
257        assert_eq!(value["cache_creation_input_tokens"], 7);
258        assert_eq!(value["cost"], 0.0012);
259        assert_eq!(value["error"], "HTTP 429: slow down");
260    }
261
262    #[test]
263    fn test_run_finished_shape() {
264        let value = serde_json::to_value(TestEvent::RunFinished {
265            tests_passed: 2,
266            tests_failed: 1,
267            steps_passed: 9,
268            steps_failed: 1,
269            steps_skipped: 2,
270            total_cost: 0.05,
271            total_tokens: 5000,
272            total_input_tokens: 4000,
273            total_output_tokens: 1000,
274            total_cached_input_tokens: 500,
275            total_cache_creation_input_tokens: 250,
276            models: vec!["deepseek".into()],
277            total_calls: 7,
278        })
279        .unwrap();
280        assert_eq!(value["type"], "run_finished");
281        assert_eq!(value["tests_passed"], 2);
282        assert_eq!(value["steps_failed"], 1);
283        assert_eq!(value["total_cost"], 0.05);
284        assert_eq!(value["total_input_tokens"], 4000);
285        assert_eq!(value["total_cached_input_tokens"], 500);
286        assert_eq!(value["total_cache_creation_input_tokens"], 250);
287        assert_eq!(value["models"][0], "deepseek");
288    }
289
290    #[test]
291    fn test_warning_shape() {
292        let value = serde_json::to_value(TestEvent::Warning {
293            message: "budget soft".into(),
294        })
295        .unwrap();
296        assert_eq!(value["type"], "warning");
297        assert_eq!(value["message"], "budget soft");
298    }
299}