llm_browser_testkit/scenario.rs
1//! Scenario types for human-readable browser test case definitions.
2//!
3//! Scenarios are written in [TOML](https://toml.io) and describe groups of
4//! browser interaction tests with reusable assertion definitions and
5//! configurable test-level overrides.
6//!
7//! # Structure
8//!
9//! ```toml
10//! [config] # Global defaults
11//! start_url = "/dashboard"
12//!
13//! [[definitions]] # Reusable assertion definitions
14//! name = "no_errors"
15//! preset = "no_error_on_page"
16//!
17//! [[test]] # Test group
18//! name = "Dashboard Smoke"
19//! start_url = "/dashboard" # Override global start_url (optional)
20//!
21//! [[test.steps]] # Ordered steps — the `kind` field
22//! kind = "navigate" # determines which other fields apply
23//! url = "/dashboard"
24//!
25//! [[test.steps]]
26//! kind = "click"
27//! target = "the Login button" # Natural language — LLM resolves to selector
28//!
29//! [[test.steps]]
30//! kind = "assert"
31//! definition = "no_errors"
32//! ```
33//!
34//! ## Step Kinds
35//!
36//! | `kind` | Required fields | Optional fields |
37//! |---------------|-----------------|------------------------------------------|
38//! | `navigate` | `url` | `wait_after_ms` |
39//! | `click` | `target` | `selector`, `wait_after_ms` |
40//! | `type` | `target`, `text`| `selector`, `wait_after_ms` |
41//! | `wait` | `target` | `selector`, `timeout_ms` |
42//! | `assert` | *one of below* | — |
43//! | `screenshot` | — | `path` |
44//! | `agent` | `agent`, `task` | — |
45//! | `mcp` | `server`, `tool`| `args` |
46//!
47//! Assert steps require one of: `definition` (references a named
48//! `[[definitions]]` entry), `preset` (built-in preset name), or `prompt`
49//! (custom LLM evaluation prompt).
50
51use std::collections::HashMap;
52
53use serde::Deserialize;
54use serde_json::Value;
55
56/// Top-level scenario file, deserialized from TOML.
57#[derive(Debug, Deserialize)]
58pub struct Scenario {
59 /// Global configuration (overridable per test).
60 #[serde(default)]
61 pub config: ScenarioConfig,
62
63 /// Reusable assertion definitions referenced by name in `assert` steps.
64 #[serde(default)]
65 pub definitions: Vec<AssertDefinition>,
66
67 /// Ordered test groups to execute.
68 #[serde(default)]
69 pub test: Vec<TestGroup>,
70}
71
72/// Global scenario configuration with per-test overridable fields.
73#[derive(Debug, Deserialize, Default, Clone)]
74pub struct ScenarioConfig {
75 /// Base URL for relative navigation.
76 #[serde(default)]
77 pub base_url: Option<String>,
78 /// LLM server base URL (deprecated; prefer `[config.endpoints]`).
79 #[serde(default)]
80 pub llm_url: Option<String>,
81 /// LLM model name (deprecated; prefer `[config.endpoints]`).
82 #[serde(default)]
83 pub llm_model: Option<String>,
84 /// LLM API key (Bearer token).
85 #[serde(default)]
86 pub llm_api_key: Option<String>,
87 /// Custom HTTP headers as JSON key-value pairs.
88 #[serde(default, deserialize_with = "deserialize_headers")]
89 pub llm_headers: HashMap<String, String>,
90 /// Run browser in headless mode.
91 #[serde(default)]
92 pub browser_headless: Option<bool>,
93 /// HTTP / browser action timeout in seconds.
94 #[serde(default)]
95 pub timeout_secs: Option<u64>,
96 /// Browser viewport width.
97 #[serde(default)]
98 pub viewport_width: Option<u32>,
99 /// Browser viewport height.
100 #[serde(default)]
101 pub viewport_height: Option<u32>,
102 /// Default URL every test auto-navigates to before running its steps.
103 #[serde(default)]
104 pub start_url: Option<String>,
105 /// Whether to auto-navigate to `start_url` before test steps.
106 ///
107 /// Disable when a test starts with click-based navigation.
108 #[serde(default = "default_auto_navigate")]
109 pub auto_navigate: bool,
110 /// LLM temperature (0.0–1.0). Lower = more deterministic.
111 #[serde(default = "default_temperature")]
112 pub temperature: f64,
113 /// Enable thinking/reasoning tokens. `None` means the provider default
114 /// is used (no `thinking` key is sent). Set to `true`/`false` to
115 /// explicitly enable or disable.
116 #[serde(default)]
117 pub thinking: Option<bool>,
118 /// Provider-specific model parameters merged into the chat completion
119 /// request body (e.g. `effort = "high"` for Anthropic).
120 #[serde(default, deserialize_with = "deserialize_model_params")]
121 pub model_params: HashMap<String, Value>,
122 /// Named endpoints (LLM, MCP, A2A agents) with pricing.
123 #[serde(default)]
124 pub endpoints: HashMap<String, EndpointConfig>,
125 /// Global and per-test budgets for cost/token/call limits.
126 #[serde(default)]
127 pub budgets: BudgetsConfig,
128 /// MCP server exposure configuration.
129 #[serde(default)]
130 pub mcp_server: Option<McpServerConfig>,
131 /// A2A agent server exposure configuration.
132 #[serde(default)]
133 pub a2a_server: Option<A2aServerConfig>,
134 /// Whether to continue running the remaining steps of a test after a
135 /// step fails. Default `false` = fail fast: the first failed step ends
136 /// the test and the rest are reported as skipped. Set to `true` to run
137 /// every step (more diagnostics, more LLM cost on broken apps).
138 #[serde(default)]
139 pub continue_on_failure: bool,
140 /// Longest edge (px) of screenshots attached to `screenshot = true`
141 /// assert steps, before they are JPEG-encoded and sent to the vision
142 /// endpoint. Downscaling keeps vision token cost/quality sane.
143 /// Default: 1400.
144 #[serde(default)]
145 pub screenshot_max_dimension: Option<u32>,
146 /// Height cap (px) of page coverage for screenshots attached to
147 /// `screenshot = true` assert steps. The full scrollable page is
148 /// captured, then split into viewport-tall tiles (each sent as its own
149 /// image part, ordered from the top) covering at most this many pixels
150 /// — below-the-fold content stays visible to the vision model at 1:1
151 /// detail while token cost stays bounded. Accepts an absolute pixel
152 /// count (`2880`) or a viewport multiple (`"20x"` = twenty times the
153 /// currently applied viewport height, which follows viewport-matrix
154 /// and per-test overrides). A value below the viewport height is
155 /// raised to it, so the visible viewport is always fully included
156 /// (`0` / `"0x"` therefore means "viewport only", the pre-full-page
157 /// behavior). Default: `"20x"`.
158 #[serde(default, deserialize_with = "deserialize_screenshot_max_height")]
159 pub screenshot_max_height: Option<ScreenshotHeight>,
160 /// Directory for failure artifacts (screenshots, page snapshots).
161 /// Defaults to `artifacts`.
162 #[serde(default)]
163 pub artifacts_dir: Option<String>,
164 /// Optional viewport matrix: when set, every test in the scenario is
165 /// expanded into one variant per viewport (e.g. mobile/tablet/desktop).
166 /// Each variant overrides the test's viewport and gets a ` — <name>`
167 /// suffix on the test name. Per-test budgets apply per variant.
168 #[serde(default)]
169 pub viewport_matrix: Option<ViewportMatrix>,
170 /// Class-name prefixes the `layout_no_issues` scan skips: elements
171 /// whose class matches any prefix are ignored by the fixed-element,
172 /// text-clipped, and overlap checks. Defaults cover the Angular CDK
173 /// screen-reader helpers (`cdk-visually-hidden`,
174 /// `cdk-describedby-message-container`, `cdk-overlay-container`),
175 /// which are intentionally 1x1 / off-screen.
176 #[serde(default = "default_layout_ignore_classes")]
177 pub layout_ignore_classes: Vec<String>,
178 /// Concurrency group for parallel runs across scenario files.
179 /// When several scenario files are run together (`--parallel > 1`),
180 /// files that declare the **same** `concurrency_group` are never
181 /// executed at the same time — use this for files that touch the same
182 /// shared backend state and would interfere if run concurrently. A file
183 /// with no group gets its own implicit group, so distinct files run in
184 /// parallel by default. Only honored when files are passed to the
185 /// runner as a batch; has no effect on the steps within a single file,
186 /// which always run sequentially.
187 #[serde(default)]
188 pub concurrency_group: Option<String>,
189 /// Default for per-endpoint `cache` across all endpoints (default
190 /// `true`). Set to `false` to disable provider-side prompt-cache
191 /// markers globally.
192 #[serde(default)]
193 pub cache: Option<bool>,
194 /// Default for per-endpoint `pricing.cache_pricing` across all
195 /// endpoints (default `true`). Set to `false` to bill all prompt
196 /// tokens at the flat input price instead of cache rates.
197 #[serde(default)]
198 pub cache_pricing: Option<bool>,
199}
200
201/// A list of named viewports a scenario is expanded across.
202#[derive(Debug, Deserialize, Clone, Default)]
203pub struct ViewportMatrix {
204 /// The viewport variants (`{name, width, height}`).
205 #[serde(default)]
206 pub viewports: Vec<ViewportDef>,
207}
208
209/// Height cap for screenshots attached to `screenshot = true` assert
210/// steps: either an absolute pixel count or a multiple of the currently
211/// applied viewport height.
212#[derive(Debug, Clone, PartialEq)]
213pub enum ScreenshotHeight {
214 /// Absolute height in pixels.
215 Pixels(u32),
216 /// Multiple of the active viewport height (e.g. `2x` = twice the
217 /// current viewport's pixel height).
218 ViewportTimes(f64),
219}
220
221/// Ceiling (px) applied when resolving screenshot height specs, so an
222/// absurdly large multiplier can never overflow or produce an unusable
223/// capture.
224const MAX_SCREENSHOT_HEIGHT: u32 = 4_000_000;
225
226impl ScreenshotHeight {
227 /// Resolves this height spec to a pixel value for the given viewport
228 /// height. Multipliers are rounded and clamped to
229 /// [`MAX_SCREENSHOT_HEIGHT`].
230 #[must_use]
231 pub fn to_px(&self, viewport_height: u32) -> u32 {
232 match self {
233 Self::Pixels(px) => *px,
234 Self::ViewportTimes(mult) => {
235 let scaled = f64::from(viewport_height) * mult;
236 #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)]
237 let px = scaled
238 .round()
239 .max(1.0)
240 .min(f64::from(MAX_SCREENSHOT_HEIGHT)) as u32;
241 px
242 }
243 }
244 }
245}
246
247/// One named viewport size in a matrix.
248#[derive(Debug, Deserialize, Clone)]
249pub struct ViewportDef {
250 /// Human-readable variant name (appended to test names, e.g.
251 /// `— mobile`).
252 pub name: String,
253 /// Browser viewport width in pixels.
254 pub width: u32,
255 /// Browser viewport height in pixels.
256 pub height: u32,
257}
258
259/// A named endpoint definition with pricing.
260#[derive(Debug, Deserialize, Clone, Default)]
261pub struct EndpointConfig {
262 /// Endpoint type: `llm`, `mcp`, or `a2a`.
263 #[serde(rename = "type")]
264 pub endpoint_type: EndpointType,
265 /// Base URL for the endpoint.
266 #[serde(default)]
267 pub url: Option<String>,
268 /// Model name (LLM endpoints only).
269 #[serde(default)]
270 pub model: Option<String>,
271 /// API key / bearer token.
272 #[serde(default)]
273 pub api_key: Option<String>,
274 /// Custom HTTP headers as JSON key-value pairs.
275 #[serde(default, deserialize_with = "deserialize_headers")]
276 pub headers: HashMap<String, String>,
277 /// Pricing configuration.
278 #[serde(default)]
279 pub pricing: Option<PricingConfig>,
280 /// Automatically fetch exact per-token pricing for this endpoint from a
281 /// provider's public pricing API, filling in any pricing fields left
282 /// unset. `"openrouter"` always uses the `OpenRouter` models API;
283 /// `"auto"` does so only when the endpoint URL host is `openrouter.ai`.
284 /// Explicit `pricing` values win over fetched ones.
285 #[serde(default)]
286 pub pricing_source: Option<String>,
287 /// Task types this endpoint serves by default
288 /// (e.g. `["targeting", "assertion"]`).
289 #[serde(default)]
290 pub default_for: Vec<String>,
291 /// Command to launch an MCP server subprocess (stdio transport).
292 #[serde(default)]
293 pub command: Option<String>,
294 /// Arguments for the MCP server command.
295 #[serde(default)]
296 pub args: Vec<String>,
297 /// Whether this LLM endpoint accepts image parts (vision) in addition
298 /// to text. `assert` steps with `screenshot = true` require a vision
299 /// endpoint.
300 #[serde(default)]
301 pub vision: bool,
302 /// How often a single chat completion against this endpoint is retried
303 /// on transient failures (HTTP 429/5xx, empty 200 bodies, network
304 /// errors) before the fallback chain is tried. Default: 3 (override
305 /// globally with `HARNESS_LLM_CALL_ATTEMPTS`).
306 #[serde(default)]
307 pub max_attempts: Option<u32>,
308 /// Ordered names of other endpoints to try when this endpoint exhausts
309 /// its attempts. Only LLM endpoints are eligible. Useful for pairing a
310 /// cheap primary model with a more powerful/expensive fallback.
311 #[serde(default)]
312 pub fallbacks: Vec<String>,
313 /// LLM provider protocol: `openai` (default, OpenAI-compatible chat
314 /// completions), `azure` (`Azure` `OpenAI`), or `bedrock` (AWS Bedrock
315 /// Converse API; requires the `aws` cargo feature).
316 #[serde(default)]
317 pub provider: Provider,
318 /// `Azure` `OpenAI` deployment name (`provider = "azure"`). Defaults to
319 /// `model` when unset.
320 #[serde(default)]
321 pub deployment: Option<String>,
322 /// `Azure` `OpenAI` API version (`provider = "azure"`). Defaults to
323 /// `2024-10-21`.
324 #[serde(default)]
325 pub api_version: Option<String>,
326 /// Authentication configuration for LLM endpoints (API key, token
327 /// command, Entra ID client credentials / managed identity).
328 #[serde(default)]
329 pub auth: AuthConfig,
330 /// Extra HTTP headers produced by running a command per call, keyed by
331 /// header name. The command's stdout (first line) becomes the header
332 /// value. Provider-agnostic — applies to every LLM provider.
333 #[serde(default, deserialize_with = "deserialize_headers")]
334 pub header_commands: HashMap<String, String>,
335 /// AWS credential settings (`provider = "bedrock"`).
336 #[serde(default)]
337 pub aws: AwsConfig,
338 /// Send provider-side prompt-cache markers from this endpoint
339 /// (default `true`). Only providers that require explicit markers are
340 /// affected: AWS Bedrock gets a `cachePoint` block, and Anthropic-style
341 /// OpenAI-compatible models (model name contains `claude`/`anthropic`,
342 /// e.g. via `OpenRouter`) get a `cache_control: ephemeral` block on the
343 /// system message. `OpenAI`, `Azure`, Groq, xAI and `DeepSeek` cache
344 /// automatically and need no markers. Set to `false` to disable.
345 #[serde(default)]
346 pub cache: Option<bool>,
347}
348
349/// Type discriminator for endpoint configuration.
350#[derive(Debug, Deserialize, Clone, PartialEq, Eq, Default)]
351#[serde(rename_all = "lowercase")]
352pub enum EndpointType {
353 /// OpenAI-compatible LLM API.
354 #[default]
355 Llm,
356 /// Model Context Protocol server.
357 Mcp,
358 /// Agent-to-Agent protocol agent.
359 A2a,
360}
361
362/// LLM provider protocol used by an LLM endpoint.
363#[derive(Debug, Deserialize, Clone, Copy, PartialEq, Eq, Default)]
364#[serde(rename_all = "lowercase")]
365pub enum Provider {
366 /// OpenAI-compatible Chat Completions API
367 /// (`POST <url>/v1/chat/completions`).
368 #[default]
369 Openai,
370 /// `Azure` `OpenAI`
371 /// (`POST <url>/openai/deployments/<deployment>/chat/completions`).
372 Azure,
373 /// AWS Bedrock Converse API (`POST https://bedrock-runtime.<region>
374 /// .amazonaws.com/model/<model>/converse`). Requires the `aws` cargo
375 /// feature; uses the standard AWS credential chain unless overridden.
376 Bedrock,
377}
378
379/// Authentication mode for an LLM endpoint.
380#[derive(Debug, Deserialize, Clone, Copy, PartialEq, Eq, Default)]
381#[serde(rename_all = "kebab-case")]
382pub enum AuthMode {
383 /// Static API key. Sent as `Authorization: Bearer <key>` by default;
384 /// set [`AuthConfig::api_key_header`] (or use the `azure` provider)
385 /// to send it in a different header.
386 #[default]
387 ApiKey,
388 /// Execute a command and use its stdout (first line) as the bearer
389 /// token. Uses the `token_command` field. Provider-agnostic escape
390 /// hatch — e.g. `az account get-access-token …`.
391 TokenCommand,
392 /// Entra ID (`Azure` AD) OAuth 2.0 client-credentials grant: exchanges
393 /// `client_id` + `client_secret` in `tenant_id` for a bearer token.
394 EntraClientCredentials,
395 /// Entra ID managed identity: fetches a bearer token from the IMDS
396 /// endpoint (works on `Azure` VMs / App Service / ACI with a
397 /// system-assigned identity; zero credentials in the config).
398 EntraManagedIdentity,
399}
400
401/// Authentication settings for an LLM endpoint.
402#[derive(Debug, Deserialize, Clone, Default)]
403pub struct AuthConfig {
404 /// Which authentication mode to use. Default: `api-key`.
405 #[serde(default)]
406 pub mode: AuthMode,
407 /// Header name that receives the API key instead of
408 /// `Authorization: Bearer` (`mode = "api-key"`). For example the `Azure`
409 /// `OpenAI` `api-key` header.
410 #[serde(default)]
411 pub api_key_header: Option<String>,
412 /// Command whose stdout (first line) is used as the bearer token
413 /// (`mode = "token-command"`).
414 #[serde(default)]
415 pub token_command: Option<String>,
416 /// Entra tenant id (`mode = "entra-client-credentials"`,
417 /// `"entra-managed-identity"`).
418 #[serde(default)]
419 pub tenant_id: Option<String>,
420 /// Entra client id (`mode = "entra-client-credentials"`).
421 #[serde(default)]
422 pub client_id: Option<String>,
423 /// Entra client secret (`mode = "entra-client-credentials"`).
424 #[serde(default)]
425 pub client_secret: Option<String>,
426 /// Entra token scope. Defaults to
427 /// `https://cognitiveservices.azure.com/.default`.
428 #[serde(default)]
429 pub scope: Option<String>,
430 /// Override for the Entra token endpoint
431 /// (`mode = "entra-client-credentials"`), default
432 /// `https://login.microsoftonline.com`. Also used as the IMDS identity
433 /// endpoint base for `"entra-managed-identity"` (default
434 /// `http://169.254.169.254`). Mainly useful for tests and proxies.
435 #[serde(default)]
436 pub token_url: Option<String>,
437 /// Seconds to reuse a fetched (non-API-key) token before refetching.
438 /// For Entra modes the server-issued expiry is used when available;
439 /// this caps the reuse window. Default 300.
440 #[serde(default)]
441 pub cache_ttl_secs: Option<u64>,
442}
443
444/// AWS credential and region settings for a `bedrock` provider endpoint.
445///
446/// When no explicit access key is given, the standard AWS credential chain
447/// is used (env vars, shared config `~/.aws/config` + `~/.aws/credentials`,
448/// SSO, ECS/IMDS) — the same behavior as the AWS CLI. `profile` selects a
449/// named profile from the shared config, and `region` overrides the chain
450/// default.
451#[derive(Debug, Deserialize, Clone, Default)]
452pub struct AwsConfig {
453 /// Named profile from `~/.aws/config` / `~/.aws/credentials` to use
454 /// (default: the AWS CLI default selection via `AWS_PROFILE` or
455 /// `default`).
456 #[serde(default)]
457 pub profile: Option<String>,
458 /// AWS region (default: `AWS_REGION` env or the profile's region;
459 /// required if neither is set).
460 #[serde(default)]
461 pub region: Option<String>,
462 /// Explicit access key id (bypasses the credential chain).
463 #[serde(default)]
464 pub access_key_id: Option<String>,
465 /// Explicit secret access key (with `access_key_id`).
466 #[serde(default)]
467 pub secret_access_key: Option<String>,
468 /// Optional session token for explicit temporary credentials.
469 #[serde(default)]
470 pub session_token: Option<String>,
471}
472
473/// Pricing configuration for an endpoint.
474#[derive(Debug, Deserialize, Clone, Default)]
475pub struct PricingConfig {
476 /// Cost per 1M input tokens (USD).
477 #[serde(default)]
478 pub input_per_1m_tokens: f64,
479 /// Cost per 1M output tokens (USD).
480 #[serde(default)]
481 pub output_per_1m_tokens: f64,
482 /// Flat cost per call (USD), used for MCP/agent endpoints.
483 #[serde(default)]
484 pub per_call: f64,
485 /// Cost per 1M cached-input (prompt cache read) tokens (USD). When
486 /// unset, `input_per_1m_tokens * cache_read_multiplier` is used.
487 #[serde(default)]
488 pub cached_input_per_1m_tokens: Option<f64>,
489 /// Cost per 1M cache-write (cache creation) input tokens (USD). When
490 /// unset, `input_per_1m_tokens * cache_write_multiplier` is used.
491 #[serde(default)]
492 pub cache_write_per_1m_tokens: Option<f64>,
493 /// Multiplier applied to `input_per_1m_tokens` for cache reads when
494 /// `cached_input_per_1m_tokens` is unset. Default: 0.1.
495 #[serde(default)]
496 pub cache_read_multiplier: Option<f64>,
497 /// Multiplier applied to `input_per_1m_tokens` for cache writes when
498 /// `cache_write_per_1m_tokens` is unset. Default: 1.25.
499 #[serde(default)]
500 pub cache_write_multiplier: Option<f64>,
501 /// Bill cache reads/writes at their cache rates (default `true`). Set
502 /// to `false` to bill every prompt token at the flat input price.
503 #[serde(default)]
504 pub cache_pricing: Option<bool>,
505}
506
507/// Budget limits for test execution.
508#[derive(Debug, Deserialize, Clone, Default)]
509pub struct BudgetsConfig {
510 /// Global budget across all tests in the scenario.
511 #[serde(default)]
512 pub global: Option<BudgetDef>,
513 /// Default per-test budget. Individual tests can override.
514 #[serde(default)]
515 pub per_test_default: Option<BudgetDef>,
516}
517
518/// A budget definition with limits and enforcement mode.
519#[derive(Debug, Deserialize, Clone)]
520pub struct BudgetDef {
521 /// Maximum cost in USD.
522 #[serde(default)]
523 pub max_cost: Option<f64>,
524 /// Maximum total tokens (input + output).
525 #[serde(default)]
526 pub max_tokens: Option<u64>,
527 /// Maximum number of calls (LLM, MCP, agent combined).
528 #[serde(default)]
529 pub max_calls: Option<u64>,
530 /// Enforcement mode: `hard` (abort) or `soft` (warn and continue).
531 #[serde(default)]
532 pub enforcement: Option<BudgetEnforcement>,
533}
534
535/// Budget enforcement strategy.
536#[derive(Debug, Deserialize, Clone, PartialEq, Eq)]
537#[serde(rename_all = "lowercase")]
538pub enum BudgetEnforcement {
539 /// Abort the test or run when budget is exceeded.
540 Hard,
541 /// Log a warning but continue execution.
542 Soft,
543}
544
545/// MCP server exposure configuration.
546#[derive(Debug, Deserialize, Clone)]
547pub struct McpServerConfig {
548 /// Whether to enable the embedded MCP server.
549 #[serde(default)]
550 pub enabled: bool,
551 /// Port to listen on.
552 #[serde(default = "default_mcp_port")]
553 pub port: u16,
554}
555
556const fn default_mcp_port() -> u16 {
557 3000
558}
559
560/// A2A agent server exposure configuration.
561#[derive(Debug, Deserialize, Clone)]
562pub struct A2aServerConfig {
563 /// Whether to enable the embedded A2A agent server.
564 #[serde(default)]
565 pub enabled: bool,
566 /// Port to listen on.
567 #[serde(default = "default_a2a_port")]
568 pub port: u16,
569}
570
571const fn default_a2a_port() -> u16 {
572 3100
573}
574
575fn deserialize_headers<'de, D>(deserializer: D) -> Result<HashMap<String, String>, D::Error>
576where
577 D: serde::Deserializer<'de>,
578{
579 let raw: Option<serde_json::Value> = Option::deserialize(deserializer)?;
580 let Some(json) = raw else {
581 return Ok(HashMap::new());
582 };
583 let serde_json::Value::Object(obj) = json else {
584 return Ok(HashMap::new());
585 };
586 Ok(obj
587 .into_iter()
588 .filter_map(|(k, v)| v.as_str().map(|s| (k, s.to_owned())))
589 .collect())
590}
591
592const fn default_auto_navigate() -> bool {
593 true
594}
595
596fn default_layout_ignore_classes() -> Vec<String> {
597 vec![
598 "cdk-visually-hidden".to_owned(),
599 "cdk-describedby-message-container".to_owned(),
600 "cdk-overlay-container".to_owned(),
601 ]
602}
603
604const fn default_temperature() -> f64 {
605 0.0
606}
607
608fn deserialize_model_params<'de, D>(deserializer: D) -> Result<HashMap<String, Value>, D::Error>
609where
610 D: serde::Deserializer<'de>,
611{
612 #[derive(Deserialize)]
613 #[serde(untagged)]
614 enum Raw {
615 Map(HashMap<String, Value>),
616 Table(HashMap<String, Value>),
617 }
618 let raw: Option<Raw> = Option::deserialize(deserializer)?;
619 Ok(match raw {
620 Some(Raw::Map(m) | Raw::Table(m)) => m,
621 None => HashMap::new(),
622 })
623}
624
625fn deserialize_screenshot_max_height<'de, D>(
626 deserializer: D,
627) -> Result<Option<ScreenshotHeight>, D::Error>
628where
629 D: serde::Deserializer<'de>,
630{
631 let raw: Option<serde_json::Value> = Option::deserialize(deserializer)?;
632 match raw {
633 None => Ok(None),
634 Some(value) => match value.as_u64() {
635 // Integer: absolute pixel count.
636 Some(px) if px <= u64::from(MAX_SCREENSHOT_HEIGHT) => {
637 #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)]
638 Ok(Some(ScreenshotHeight::Pixels(px as u32)))
639 }
640 // String: viewport multiple like "2x" or "1.5x".
641 None => match value.as_str() {
642 Some(s) => {
643 let lower = s.trim().to_lowercase();
644 if lower.ends_with('x') {
645 let num = lower.strip_suffix("x");
646 if let Some(num) = num {
647 if let Ok(m) = num.parse::<f64>() {
648 if m > 0.0 {
649 return Ok(Some(ScreenshotHeight::ViewportTimes(m)));
650 }
651 }
652 }
653 }
654 Err(serde::de::Error::custom(format_args!(
655 "screenshot_max_height must be a pixel count (e.g. 2880) or a viewport multiple like \"2x\", found {s:?}"
656 )))
657 }
658 None => Err(serde::de::Error::custom(format_args!(
659 "screenshot_max_height must be a pixel count (e.g. 2880) or a viewport multiple like \"2x\", found {value:?}"
660 ))),
661 },
662 Some(_) => Err(serde::de::Error::custom(format_args!(
663 "screenshot_max_height value too large (max {MAX_SCREENSHOT_HEIGHT} px)"
664 ))),
665 },
666 }
667}
668
669/// Reusable assertion definition referenced by name from `assert` steps.
670///
671/// Definitions can either reference a built-in preset via `preset`, supply a
672/// custom LLM `prompt`, or define a **custom preset** by providing both
673/// `system` and `user_template`. Custom presets support the same template
674/// variables as built-in presets: `{url}`, `{title}`, `{content}`,
675/// `{expected_text}`, `{description}`.
676#[derive(Debug, Deserialize, Clone)]
677pub struct AssertDefinition {
678 /// Unique name used to reference this definition from steps.
679 pub name: String,
680 /// Predefined assertion preset name
681 /// (e.g. `no_error_on_page`, `text_visible`).
682 #[serde(default)]
683 pub preset: Option<String>,
684 /// Custom LLM prompt for assertion evaluation.
685 #[serde(default)]
686 pub prompt: Option<String>,
687 /// System prompt for a custom preset.
688 #[serde(default)]
689 pub system: Option<String>,
690 /// User template (with `{placeholders}`) for a custom preset.
691 #[serde(default)]
692 pub user_template: Option<String>,
693 /// Text that the `text_visible` preset checks for, or the
694 /// `{expected_text}` placeholder value for custom presets.
695 #[serde(default)]
696 pub assert_text: Option<String>,
697 /// Agent endpoint to call for this assertion.
698 #[serde(default)]
699 pub agent: Option<String>,
700 /// Agent task template for this assertion.
701 #[serde(default)]
702 pub task_template: Option<String>,
703}
704
705/// A group of steps that form a single test scenario.
706#[derive(Debug, Deserialize, Clone)]
707pub struct TestGroup {
708 /// Human-readable test name.
709 pub name: String,
710 /// Override the global `start_url` for this test.
711 #[serde(default)]
712 pub start_url: Option<String>,
713 /// Override the global `auto_navigate` for this test.
714 #[serde(default)]
715 pub auto_navigate: Option<bool>,
716 /// Override the global `base_url` for this test.
717 #[serde(default)]
718 pub base_url: Option<String>,
719 /// Override the global `timeout_secs` for this test.
720 #[serde(default)]
721 pub timeout_secs: Option<u64>,
722 /// Override the global `browser_headless` for this test.
723 #[serde(default)]
724 pub browser_headless: Option<bool>,
725 /// Override the global viewport width for this test (applied via CDP
726 /// `Emulation.setDeviceMetricsOverride` before the test runs).
727 #[serde(default)]
728 pub viewport_width: Option<u32>,
729 /// Override the global viewport height for this test.
730 #[serde(default)]
731 pub viewport_height: Option<u32>,
732 /// Per-test budget override.
733 #[serde(default)]
734 pub budget: Option<BudgetDef>,
735 /// Endpoint to use for all steps in this test (can be overridden
736 /// per-step).
737 #[serde(default)]
738 pub endpoint: Option<String>,
739 /// Ordered steps to execute.
740 #[serde(default)]
741 pub steps: Vec<TestStep>,
742}
743
744/// A single step in a test. The `kind` field determines which variant is
745/// deserialized and which field constraints apply.
746#[derive(Debug, Deserialize, Clone)]
747#[serde(tag = "kind")]
748pub enum TestStep {
749 /// Navigate the browser to a URL.
750 #[serde(rename = "navigate")]
751 Navigate {
752 /// URL to navigate to (absolute, or relative to the test's
753 /// base URL).
754 url: String,
755 /// Milliseconds to wait after navigation completes.
756 #[serde(default)]
757 wait_after_ms: Option<u64>,
758 },
759
760 /// Click an element described in natural language.
761 #[serde(rename = "click")]
762 Click {
763 /// Natural language description of the element. The LLM resolves
764 /// this to a CSS selector at runtime.
765 target: String,
766 /// Explicit CSS selector override (bypasses LLM resolution).
767 #[serde(default)]
768 selector: Option<String>,
769 /// Milliseconds to wait after the click.
770 #[serde(default)]
771 wait_after_ms: Option<u64>,
772 /// Endpoint to use for LLM element targeting.
773 #[serde(default)]
774 endpoint: Option<String>,
775 /// Idempotent: when the target element is absent the step is
776 /// reported skipped instead of failed (the action was already
777 /// done / not applicable).
778 #[serde(default)]
779 idempotent: bool,
780 },
781
782 /// Type text into an input element.
783 #[serde(rename = "type")]
784 Type {
785 /// Natural language description of the target input element.
786 target: String,
787 /// Text to type into the element.
788 text: String,
789 /// Explicit CSS selector override (bypasses LLM resolution).
790 #[serde(default)]
791 selector: Option<String>,
792 /// Milliseconds to wait after typing.
793 #[serde(default)]
794 wait_after_ms: Option<u64>,
795 /// Endpoint to use for LLM element targeting.
796 #[serde(default)]
797 endpoint: Option<String>,
798 /// Idempotent: when the target element is absent the step is
799 /// reported skipped instead of failed (the action was already
800 /// done / not applicable).
801 #[serde(default)]
802 idempotent: bool,
803 },
804
805 /// Wait for an element to appear on the page.
806 #[serde(rename = "wait")]
807 Wait {
808 /// Natural language description of the element to wait for.
809 target: String,
810 /// Explicit CSS selector override (bypasses LLM resolution).
811 #[serde(default)]
812 selector: Option<String>,
813 /// Wait until the page's visible text contains this substring
814 /// (alternative to `selector`; either or both may be set — both are
815 /// required to hold when both are set).
816 #[serde(default)]
817 text: Option<String>,
818 /// Maximum milliseconds to wait (default: 10000).
819 #[serde(default)]
820 timeout_ms: Option<u64>,
821 /// Endpoint to use for LLM element targeting.
822 #[serde(default)]
823 endpoint: Option<String>,
824 /// Idempotent: when the condition never becomes true within the
825 /// timeout the step is reported skipped instead of failed (the
826 /// condition was not applicable, e.g. already-authenticated
827 /// pages in a viewport matrix).
828 #[serde(default)]
829 idempotent: bool,
830 },
831
832 /// Evaluate an assertion against the current page content.
833 #[serde(rename = "assert")]
834 Assert {
835 /// Reference to a named `[[definitions]]` entry.
836 #[serde(default)]
837 definition: Option<String>,
838 /// Inline predefined assertion preset (e.g. `no_error_on_page`).
839 #[serde(default)]
840 preset: Option<String>,
841 /// Inline custom LLM prompt for assertion evaluation.
842 #[serde(default)]
843 prompt: Option<String>,
844 /// Text that the `text_visible` preset checks for.
845 #[serde(default)]
846 assert_text: Option<String>,
847 /// Endpoint to use for this assertion's LLM call.
848 #[serde(default)]
849 endpoint: Option<String>,
850 /// Attach a screenshot of the current viewport to the assertion so
851 /// the LLM can evaluate visuals (overlaps, clipping, layout).
852 /// Requires the resolved endpoint to declare `vision = true`.
853 #[serde(default)]
854 screenshot: bool,
855 },
856
857 /// Take a screenshot of the current page.
858 #[serde(rename = "screenshot")]
859 Screenshot {
860 /// File path to save the screenshot (default: `screenshot.png`).
861 #[serde(default)]
862 path: Option<String>,
863 },
864
865 /// Call an A2A agent with a task.
866 #[serde(rename = "agent")]
867 Agent {
868 /// Name of the agent endpoint to call.
869 agent: String,
870 /// Task description / prompt for the agent.
871 task: String,
872 /// Optional definition name with a task template.
873 #[serde(default)]
874 definition: Option<String>,
875 },
876
877 /// Call an MCP server tool.
878 #[serde(rename = "mcp")]
879 Mcp {
880 /// Name of the MCP server endpoint.
881 server: String,
882 /// Tool name to invoke on the server.
883 tool: String,
884 /// Tool arguments as JSON.
885 #[serde(default)]
886 args: Option<serde_json::Value>,
887 },
888}