llm_browser_testkit/scenario.rs
1//! Scenario types for human-readable browser test case definitions.
2//!
3//! Scenarios are written in [TOML](https://toml.io) and describe groups of
4//! browser interaction tests with reusable assertion definitions and
5//! configurable test-level overrides.
6//!
7//! # Structure
8//!
9//! ```toml
10//! [config] # Global defaults
11//! start_url = "/dashboard"
12//!
13//! [[definitions]] # Reusable assertion definitions
14//! name = "no_errors"
15//! preset = "no_error_on_page"
16//!
17//! [[test]] # Test group
18//! name = "Dashboard Smoke"
19//! start_url = "/dashboard" # Override global start_url (optional)
20//!
21//! [[test.steps]] # Ordered steps — the `kind` field
22//! kind = "navigate" # determines which other fields apply
23//! url = "/dashboard"
24//!
25//! [[test.steps]]
26//! kind = "click"
27//! target = "the Login button" # Natural language — LLM resolves to selector
28//!
29//! [[test.steps]]
30//! kind = "assert"
31//! definition = "no_errors"
32//! ```
33//!
34//! ## Step Kinds
35//!
36//! | `kind` | Required fields | Optional fields |
37//! |---------------|-----------------|------------------------------------------|
38//! | `navigate` | `url` | `wait_after_ms` |
39//! | `click` | `target` | `selector`, `wait_after_ms` |
40//! | `type` | `target`, `text`| `selector`, `wait_after_ms` |
41//! | `wait` | `target` | `selector`, `timeout_ms` |
42//! | `assert` | *one of below* | — |
43//! | `screenshot` | — | `path` |
44//! | `agent` | `agent`, `task` | — |
45//! | `mcp` | `server`, `tool`| `args` |
46//!
47//! Assert steps require one of: `definition` (references a named
48//! `[[definitions]]` entry), `preset` (built-in preset name), or `prompt`
49//! (custom LLM evaluation prompt).
50
51use std::collections::HashMap;
52
53use serde::Deserialize;
54use serde_json::Value;
55
56/// Top-level scenario file, deserialized from TOML.
57#[derive(Debug, Deserialize)]
58pub struct Scenario {
59 /// Global configuration (overridable per test).
60 #[serde(default)]
61 pub config: ScenarioConfig,
62
63 /// Reusable assertion definitions referenced by name in `assert` steps.
64 #[serde(default)]
65 pub definitions: Vec<AssertDefinition>,
66
67 /// Ordered test groups to execute.
68 #[serde(default)]
69 pub test: Vec<TestGroup>,
70}
71
72/// Global scenario configuration with per-test overridable fields.
73#[derive(Debug, Deserialize, Default, Clone)]
74pub struct ScenarioConfig {
75 /// Base URL for relative navigation.
76 #[serde(default)]
77 pub base_url: Option<String>,
78 /// LLM server base URL (deprecated; prefer `[config.endpoints]`).
79 #[serde(default)]
80 pub llm_url: Option<String>,
81 /// LLM model name (deprecated; prefer `[config.endpoints]`).
82 #[serde(default)]
83 pub llm_model: Option<String>,
84 /// LLM API key (Bearer token).
85 #[serde(default)]
86 pub llm_api_key: Option<String>,
87 /// Custom HTTP headers as JSON key-value pairs.
88 #[serde(default, deserialize_with = "deserialize_headers")]
89 pub llm_headers: HashMap<String, String>,
90 /// Run browser in headless mode.
91 #[serde(default)]
92 pub browser_headless: Option<bool>,
93 /// HTTP / browser action timeout in seconds.
94 #[serde(default)]
95 pub timeout_secs: Option<u64>,
96 /// Browser viewport width.
97 #[serde(default)]
98 pub viewport_width: Option<u32>,
99 /// Browser viewport height.
100 #[serde(default)]
101 pub viewport_height: Option<u32>,
102 /// Default URL every test auto-navigates to before running its steps.
103 #[serde(default)]
104 pub start_url: Option<String>,
105 /// Whether to auto-navigate to `start_url` before test steps.
106 ///
107 /// Disable when a test starts with click-based navigation.
108 #[serde(default = "default_auto_navigate")]
109 pub auto_navigate: bool,
110 /// LLM temperature (0.0–1.0). Lower = more deterministic.
111 #[serde(default = "default_temperature")]
112 pub temperature: f64,
113 /// Enable thinking/reasoning tokens. `None` means the provider default
114 /// is used (no `thinking` key is sent). Set to `true`/`false` to
115 /// explicitly enable or disable.
116 #[serde(default)]
117 pub thinking: Option<bool>,
118 /// Provider-specific model parameters merged into the chat completion
119 /// request body (e.g. `effort = "high"` for Anthropic).
120 #[serde(default, deserialize_with = "deserialize_model_params")]
121 pub model_params: HashMap<String, Value>,
122 /// Named endpoints (LLM, MCP, A2A agents) with pricing.
123 #[serde(default)]
124 pub endpoints: HashMap<String, EndpointConfig>,
125 /// Global and per-test budgets for cost/token/call limits.
126 #[serde(default)]
127 pub budgets: BudgetsConfig,
128 /// MCP server exposure configuration.
129 #[serde(default)]
130 pub mcp_server: Option<McpServerConfig>,
131 /// A2A agent server exposure configuration.
132 #[serde(default)]
133 pub a2a_server: Option<A2aServerConfig>,
134 /// Whether to continue running the remaining steps of a test after a
135 /// step fails. Default `false` = fail fast: the first failed step ends
136 /// the test and the rest are reported as skipped. Set to `true` to run
137 /// every step (more diagnostics, more LLM cost on broken apps).
138 #[serde(default)]
139 pub continue_on_failure: bool,
140 /// Longest edge (px) of screenshots attached to `screenshot = true`
141 /// assert steps, before they are JPEG-encoded and sent to the vision
142 /// endpoint. Downscaling keeps vision token cost/quality sane.
143 /// Default: 1400.
144 #[serde(default)]
145 pub screenshot_max_dimension: Option<u32>,
146 /// Height cap (px) of page coverage for screenshots attached to
147 /// `screenshot = true` assert steps. The full scrollable page is
148 /// captured, then split into viewport-tall tiles (each sent as its own
149 /// image part, ordered from the top) covering at most this many pixels
150 /// — below-the-fold content stays visible to the vision model at 1:1
151 /// detail while token cost stays bounded. Accepts an absolute pixel
152 /// count (`2880`) or a viewport multiple (`"20x"` = twenty times the
153 /// currently applied viewport height, which follows viewport-matrix
154 /// and per-test overrides). A value below the viewport height is
155 /// raised to it, so the visible viewport is always fully included
156 /// (`0` / `"0x"` therefore means "viewport only", the pre-full-page
157 /// behavior). Default: `"20x"`.
158 #[serde(default, deserialize_with = "deserialize_screenshot_max_height")]
159 pub screenshot_max_height: Option<ScreenshotHeight>,
160 /// Directory for failure artifacts (screenshots, page snapshots).
161 /// Defaults to `artifacts`.
162 #[serde(default)]
163 pub artifacts_dir: Option<String>,
164 /// Optional viewport matrix: when set, every test in the scenario is
165 /// expanded into one variant per viewport (e.g. mobile/tablet/desktop).
166 /// Each variant overrides the test's viewport and gets a ` — <name>`
167 /// suffix on the test name. Per-test budgets apply per variant.
168 #[serde(default)]
169 pub viewport_matrix: Option<ViewportMatrix>,
170 /// Class-name prefixes the `layout_no_issues` scan skips: elements
171 /// whose class matches any prefix are ignored by the fixed-element,
172 /// text-clipped, and overlap checks. Defaults cover the Angular CDK
173 /// screen-reader helpers (`cdk-visually-hidden`,
174 /// `cdk-describedby-message-container`, `cdk-overlay-container`),
175 /// which are intentionally 1x1 / off-screen.
176 #[serde(default = "default_layout_ignore_classes")]
177 pub layout_ignore_classes: Vec<String>,
178 /// Concurrency group for parallel runs across scenario files.
179 /// When several scenario files are run together (`--parallel > 1`),
180 /// files that declare the **same** `concurrency_group` are never
181 /// executed at the same time — use this for files that touch the same
182 /// shared backend state and would interfere if run concurrently. A file
183 /// with no group gets its own implicit group, so distinct files run in
184 /// parallel by default. Only honored when files are passed to the
185 /// runner as a batch; has no effect on the steps within a single file,
186 /// which always run sequentially.
187 #[serde(default)]
188 pub concurrency_group: Option<String>,
189 /// Default for per-endpoint `cache` across all endpoints (default
190 /// `true`). Set to `false` to disable provider-side prompt-cache
191 /// markers globally.
192 #[serde(default)]
193 pub cache: Option<bool>,
194 /// Default for per-endpoint `pricing.cache_pricing` across all
195 /// endpoints (default `true`). Set to `false` to bill all prompt
196 /// tokens at the flat input price instead of cache rates.
197 #[serde(default)]
198 pub cache_pricing: Option<bool>,
199}
200
201/// A list of named viewports a scenario is expanded across.
202#[derive(Debug, Deserialize, Clone, Default)]
203pub struct ViewportMatrix {
204 /// The viewport variants (`{name, width, height}`).
205 #[serde(default)]
206 pub viewports: Vec<ViewportDef>,
207}
208
209/// Height cap for screenshots attached to `screenshot = true` assert
210/// steps: either an absolute pixel count or a multiple of the currently
211/// applied viewport height.
212#[derive(Debug, Clone, PartialEq)]
213pub enum ScreenshotHeight {
214 /// Absolute height in pixels.
215 Pixels(u32),
216 /// Multiple of the active viewport height (e.g. `2x` = twice the
217 /// current viewport's pixel height).
218 ViewportTimes(f64),
219}
220
221/// Ceiling (px) applied when resolving screenshot height specs, so an
222/// absurdly large multiplier can never overflow or produce an unusable
223/// capture.
224const MAX_SCREENSHOT_HEIGHT: u32 = 4_000_000;
225
226impl ScreenshotHeight {
227 /// Resolves this height spec to a pixel value for the given viewport
228 /// height. Multipliers are rounded and clamped to
229 /// [`MAX_SCREENSHOT_HEIGHT`].
230 #[must_use]
231 pub fn to_px(&self, viewport_height: u32) -> u32 {
232 match self {
233 Self::Pixels(px) => *px,
234 Self::ViewportTimes(mult) => {
235 let scaled = f64::from(viewport_height) * mult;
236 #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)]
237 let px = scaled
238 .round()
239 .max(1.0)
240 .min(f64::from(MAX_SCREENSHOT_HEIGHT)) as u32;
241 px
242 }
243 }
244 }
245}
246
247/// One named viewport size in a matrix.
248#[derive(Debug, Deserialize, Clone)]
249pub struct ViewportDef {
250 /// Human-readable variant name (appended to test names, e.g.
251 /// `— mobile`).
252 pub name: String,
253 /// Browser viewport width in pixels.
254 pub width: u32,
255 /// Browser viewport height in pixels.
256 pub height: u32,
257}
258
259/// A named endpoint definition with pricing.
260#[derive(Debug, Deserialize, Clone, Default)]
261pub struct EndpointConfig {
262 /// Endpoint type: `llm`, `mcp`, or `a2a`.
263 #[serde(rename = "type")]
264 pub endpoint_type: EndpointType,
265 /// Base URL for the endpoint.
266 #[serde(default)]
267 pub url: Option<String>,
268 /// Model name (LLM endpoints only).
269 #[serde(default)]
270 pub model: Option<String>,
271 /// API key / bearer token.
272 #[serde(default)]
273 pub api_key: Option<String>,
274 /// Custom HTTP headers as JSON key-value pairs.
275 #[serde(default, deserialize_with = "deserialize_headers")]
276 pub headers: HashMap<String, String>,
277 /// Pricing configuration.
278 #[serde(default)]
279 pub pricing: Option<PricingConfig>,
280 /// Automatically fetch exact per-token pricing for this endpoint at
281 /// startup from a provider's public pricing API, filling in any pricing
282 /// fields left unset (explicit `pricing` values win). Defaults to
283 /// `"auto"`: Bedrock endpoints use the AWS Price List and `openrouter.ai`
284 /// URLs use the `OpenRouter` models API; other providers are a no-op.
285 /// Force a source with `"bedrock"` / `"openrouter"`, or disable the
286 /// lookup with `"off"` / `"none"` / `"disabled"`.
287 #[serde(default)]
288 pub pricing_source: Option<String>,
289 /// Task types this endpoint serves by default
290 /// (e.g. `["targeting", "assertion"]`).
291 #[serde(default)]
292 pub default_for: Vec<String>,
293 /// Command to launch an MCP server subprocess (stdio transport).
294 #[serde(default)]
295 pub command: Option<String>,
296 /// Arguments for the MCP server command.
297 #[serde(default)]
298 pub args: Vec<String>,
299 /// Whether this LLM endpoint accepts image parts (vision) in addition
300 /// to text. `assert` steps with `screenshot = true` require a vision
301 /// endpoint.
302 #[serde(default)]
303 pub vision: bool,
304 /// How often a single chat completion against this endpoint is retried
305 /// on transient failures (HTTP 429/5xx, empty 200 bodies, network
306 /// errors) before the fallback chain is tried. Default: 3 (override
307 /// globally with `HARNESS_LLM_CALL_ATTEMPTS`).
308 #[serde(default)]
309 pub max_attempts: Option<u32>,
310 /// Ordered names of other endpoints to try when this endpoint exhausts
311 /// its attempts. Only LLM endpoints are eligible. Useful for pairing a
312 /// cheap primary model with a more powerful/expensive fallback.
313 #[serde(default)]
314 pub fallbacks: Vec<String>,
315 /// LLM provider protocol: `openai` (default, OpenAI-compatible chat
316 /// completions), `azure` (`Azure` `OpenAI`), or `bedrock` (AWS Bedrock
317 /// Converse API; requires the `aws` cargo feature).
318 #[serde(default)]
319 pub provider: Provider,
320 /// `Azure` `OpenAI` deployment name (`provider = "azure"`). Defaults to
321 /// `model` when unset.
322 #[serde(default)]
323 pub deployment: Option<String>,
324 /// `Azure` `OpenAI` API version (`provider = "azure"`). Defaults to
325 /// `2024-10-21`.
326 #[serde(default)]
327 pub api_version: Option<String>,
328 /// Authentication configuration for LLM endpoints (API key, token
329 /// command, Entra ID client credentials / managed identity).
330 #[serde(default)]
331 pub auth: AuthConfig,
332 /// Extra HTTP headers produced by running a command per call, keyed by
333 /// header name. The command's stdout (first line) becomes the header
334 /// value. Provider-agnostic — applies to every LLM provider.
335 #[serde(default, deserialize_with = "deserialize_headers")]
336 pub header_commands: HashMap<String, String>,
337 /// AWS credential settings (`provider = "bedrock"`).
338 #[serde(default)]
339 pub aws: AwsConfig,
340 /// Send provider-side prompt-cache markers from this endpoint
341 /// (default `true`). Only providers that require explicit markers are
342 /// affected: AWS Bedrock gets a `cachePoint` block, and Anthropic-style
343 /// OpenAI-compatible models (model name contains `claude`/`anthropic`,
344 /// e.g. via `OpenRouter`) get a `cache_control: ephemeral` block on the
345 /// system message. `OpenAI`, `Azure`, Groq, xAI and `DeepSeek` cache
346 /// automatically and need no markers. Set to `false` to disable.
347 #[serde(default)]
348 pub cache: Option<bool>,
349}
350
351/// Type discriminator for endpoint configuration.
352#[derive(Debug, Deserialize, Clone, PartialEq, Eq, Default)]
353#[serde(rename_all = "lowercase")]
354pub enum EndpointType {
355 /// OpenAI-compatible LLM API.
356 #[default]
357 Llm,
358 /// Model Context Protocol server.
359 Mcp,
360 /// Agent-to-Agent protocol agent.
361 A2a,
362}
363
364/// LLM provider protocol used by an LLM endpoint.
365#[derive(Debug, Deserialize, Clone, Copy, PartialEq, Eq, Default)]
366#[serde(rename_all = "lowercase")]
367pub enum Provider {
368 /// OpenAI-compatible Chat Completions API
369 /// (`POST <url>/v1/chat/completions`).
370 #[default]
371 Openai,
372 /// `Azure` `OpenAI`
373 /// (`POST <url>/openai/deployments/<deployment>/chat/completions`).
374 Azure,
375 /// AWS Bedrock Converse API (`POST https://bedrock-runtime.<region>
376 /// .amazonaws.com/model/<model>/converse`). Requires the `aws` cargo
377 /// feature; uses the standard AWS credential chain unless overridden.
378 Bedrock,
379}
380
381/// Authentication mode for an LLM endpoint.
382#[derive(Debug, Deserialize, Clone, Copy, PartialEq, Eq, Default)]
383#[serde(rename_all = "kebab-case")]
384pub enum AuthMode {
385 /// Static API key. Sent as `Authorization: Bearer <key>` by default;
386 /// set [`AuthConfig::api_key_header`] (or use the `azure` provider)
387 /// to send it in a different header.
388 #[default]
389 ApiKey,
390 /// Execute a command and use its stdout (first line) as the bearer
391 /// token. Uses the `token_command` field. Provider-agnostic escape
392 /// hatch — e.g. `az account get-access-token …`.
393 TokenCommand,
394 /// Entra ID (`Azure` AD) OAuth 2.0 client-credentials grant: exchanges
395 /// `client_id` + `client_secret` in `tenant_id` for a bearer token.
396 EntraClientCredentials,
397 /// Entra ID managed identity: fetches a bearer token from the IMDS
398 /// endpoint (works on `Azure` VMs / App Service / ACI with a
399 /// system-assigned identity; zero credentials in the config).
400 EntraManagedIdentity,
401}
402
403/// Authentication settings for an LLM endpoint.
404#[derive(Debug, Deserialize, Clone, Default)]
405pub struct AuthConfig {
406 /// Which authentication mode to use. Default: `api-key`.
407 #[serde(default)]
408 pub mode: AuthMode,
409 /// Header name that receives the API key instead of
410 /// `Authorization: Bearer` (`mode = "api-key"`). For example the `Azure`
411 /// `OpenAI` `api-key` header.
412 #[serde(default)]
413 pub api_key_header: Option<String>,
414 /// Command whose stdout (first line) is used as the bearer token
415 /// (`mode = "token-command"`).
416 #[serde(default)]
417 pub token_command: Option<String>,
418 /// Entra tenant id (`mode = "entra-client-credentials"`,
419 /// `"entra-managed-identity"`).
420 #[serde(default)]
421 pub tenant_id: Option<String>,
422 /// Entra client id (`mode = "entra-client-credentials"`).
423 #[serde(default)]
424 pub client_id: Option<String>,
425 /// Entra client secret (`mode = "entra-client-credentials"`).
426 #[serde(default)]
427 pub client_secret: Option<String>,
428 /// Entra token scope. Defaults to
429 /// `https://cognitiveservices.azure.com/.default`.
430 #[serde(default)]
431 pub scope: Option<String>,
432 /// Override for the Entra token endpoint
433 /// (`mode = "entra-client-credentials"`), default
434 /// `https://login.microsoftonline.com`. Also used as the IMDS identity
435 /// endpoint base for `"entra-managed-identity"` (default
436 /// `http://169.254.169.254`). Mainly useful for tests and proxies.
437 #[serde(default)]
438 pub token_url: Option<String>,
439 /// Seconds to reuse a fetched (non-API-key) token before refetching.
440 /// For Entra modes the server-issued expiry is used when available;
441 /// this caps the reuse window. Default 300.
442 #[serde(default)]
443 pub cache_ttl_secs: Option<u64>,
444}
445
446/// AWS credential and region settings for a `bedrock` provider endpoint.
447///
448/// When no explicit access key is given, the standard AWS credential chain
449/// is used (env vars, shared config `~/.aws/config` + `~/.aws/credentials`,
450/// SSO, ECS/IMDS) — the same behavior as the AWS CLI. `profile` selects a
451/// named profile from the shared config, and `region` overrides the chain
452/// default.
453#[derive(Debug, Deserialize, Clone, Default)]
454pub struct AwsConfig {
455 /// Named profile from `~/.aws/config` / `~/.aws/credentials` to use
456 /// (default: the AWS CLI default selection via `AWS_PROFILE` or
457 /// `default`).
458 #[serde(default)]
459 pub profile: Option<String>,
460 /// AWS region (default: `AWS_REGION` env or the profile's region;
461 /// required if neither is set).
462 #[serde(default)]
463 pub region: Option<String>,
464 /// Explicit access key id (bypasses the credential chain).
465 #[serde(default)]
466 pub access_key_id: Option<String>,
467 /// Explicit secret access key (with `access_key_id`).
468 #[serde(default)]
469 pub secret_access_key: Option<String>,
470 /// Optional session token for explicit temporary credentials.
471 #[serde(default)]
472 pub session_token: Option<String>,
473}
474
475/// Pricing configuration for an endpoint.
476#[derive(Debug, Deserialize, Clone, Default)]
477pub struct PricingConfig {
478 /// Cost per 1M input tokens (USD).
479 #[serde(default)]
480 pub input_per_1m_tokens: f64,
481 /// Cost per 1M output tokens (USD).
482 #[serde(default)]
483 pub output_per_1m_tokens: f64,
484 /// Flat cost per call (USD), used for MCP/agent endpoints.
485 #[serde(default)]
486 pub per_call: f64,
487 /// Cost per 1M cached-input (prompt cache read) tokens (USD). When
488 /// unset, `input_per_1m_tokens * cache_read_multiplier` is used.
489 #[serde(default)]
490 pub cached_input_per_1m_tokens: Option<f64>,
491 /// Cost per 1M cache-write (cache creation) input tokens (USD). When
492 /// unset, `input_per_1m_tokens * cache_write_multiplier` is used.
493 #[serde(default)]
494 pub cache_write_per_1m_tokens: Option<f64>,
495 /// Multiplier applied to `input_per_1m_tokens` for cache reads when
496 /// `cached_input_per_1m_tokens` is unset. Default: 0.1.
497 #[serde(default)]
498 pub cache_read_multiplier: Option<f64>,
499 /// Multiplier applied to `input_per_1m_tokens` for cache writes when
500 /// `cache_write_per_1m_tokens` is unset. Default: 1.25.
501 #[serde(default)]
502 pub cache_write_multiplier: Option<f64>,
503 /// Bill cache reads/writes at their cache rates (default `true`). Set
504 /// to `false` to bill every prompt token at the flat input price.
505 #[serde(default)]
506 pub cache_pricing: Option<bool>,
507}
508
509/// Budget limits for test execution.
510#[derive(Debug, Deserialize, Clone, Default)]
511pub struct BudgetsConfig {
512 /// Global budget across all tests in the scenario.
513 #[serde(default)]
514 pub global: Option<BudgetDef>,
515 /// Default per-test budget. Individual tests can override.
516 #[serde(default)]
517 pub per_test_default: Option<BudgetDef>,
518}
519
520/// A budget definition with limits and enforcement mode.
521#[derive(Debug, Deserialize, Clone)]
522pub struct BudgetDef {
523 /// Maximum cost in USD.
524 #[serde(default)]
525 pub max_cost: Option<f64>,
526 /// Maximum total tokens (input + output).
527 #[serde(default)]
528 pub max_tokens: Option<u64>,
529 /// Maximum number of calls (LLM, MCP, agent combined).
530 #[serde(default)]
531 pub max_calls: Option<u64>,
532 /// Enforcement mode: `hard` (abort) or `soft` (warn and continue).
533 #[serde(default)]
534 pub enforcement: Option<BudgetEnforcement>,
535}
536
537/// Budget enforcement strategy.
538#[derive(Debug, Deserialize, Clone, PartialEq, Eq)]
539#[serde(rename_all = "lowercase")]
540pub enum BudgetEnforcement {
541 /// Abort the test or run when budget is exceeded.
542 Hard,
543 /// Log a warning but continue execution.
544 Soft,
545}
546
547/// MCP server exposure configuration.
548#[derive(Debug, Deserialize, Clone)]
549pub struct McpServerConfig {
550 /// Whether to enable the embedded MCP server.
551 #[serde(default)]
552 pub enabled: bool,
553 /// Port to listen on.
554 #[serde(default = "default_mcp_port")]
555 pub port: u16,
556}
557
558const fn default_mcp_port() -> u16 {
559 3000
560}
561
562/// A2A agent server exposure configuration.
563#[derive(Debug, Deserialize, Clone)]
564pub struct A2aServerConfig {
565 /// Whether to enable the embedded A2A agent server.
566 #[serde(default)]
567 pub enabled: bool,
568 /// Port to listen on.
569 #[serde(default = "default_a2a_port")]
570 pub port: u16,
571}
572
573const fn default_a2a_port() -> u16 {
574 3100
575}
576
577fn deserialize_headers<'de, D>(deserializer: D) -> Result<HashMap<String, String>, D::Error>
578where
579 D: serde::Deserializer<'de>,
580{
581 let raw: Option<serde_json::Value> = Option::deserialize(deserializer)?;
582 let Some(json) = raw else {
583 return Ok(HashMap::new());
584 };
585 let serde_json::Value::Object(obj) = json else {
586 return Ok(HashMap::new());
587 };
588 Ok(obj
589 .into_iter()
590 .filter_map(|(k, v)| v.as_str().map(|s| (k, s.to_owned())))
591 .collect())
592}
593
594const fn default_auto_navigate() -> bool {
595 true
596}
597
598fn default_layout_ignore_classes() -> Vec<String> {
599 vec![
600 "cdk-visually-hidden".to_owned(),
601 "cdk-describedby-message-container".to_owned(),
602 "cdk-overlay-container".to_owned(),
603 ]
604}
605
606const fn default_temperature() -> f64 {
607 0.0
608}
609
610fn deserialize_model_params<'de, D>(deserializer: D) -> Result<HashMap<String, Value>, D::Error>
611where
612 D: serde::Deserializer<'de>,
613{
614 #[derive(Deserialize)]
615 #[serde(untagged)]
616 enum Raw {
617 Map(HashMap<String, Value>),
618 Table(HashMap<String, Value>),
619 }
620 let raw: Option<Raw> = Option::deserialize(deserializer)?;
621 Ok(match raw {
622 Some(Raw::Map(m) | Raw::Table(m)) => m,
623 None => HashMap::new(),
624 })
625}
626
627fn deserialize_screenshot_max_height<'de, D>(
628 deserializer: D,
629) -> Result<Option<ScreenshotHeight>, D::Error>
630where
631 D: serde::Deserializer<'de>,
632{
633 let raw: Option<serde_json::Value> = Option::deserialize(deserializer)?;
634 match raw {
635 None => Ok(None),
636 Some(value) => match value.as_u64() {
637 // Integer: absolute pixel count.
638 Some(px) if px <= u64::from(MAX_SCREENSHOT_HEIGHT) => {
639 #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)]
640 Ok(Some(ScreenshotHeight::Pixels(px as u32)))
641 }
642 // String: viewport multiple like "2x" or "1.5x".
643 None => match value.as_str() {
644 Some(s) => {
645 let lower = s.trim().to_lowercase();
646 if lower.ends_with('x') {
647 let num = lower.strip_suffix("x");
648 if let Some(num) = num {
649 if let Ok(m) = num.parse::<f64>() {
650 if m > 0.0 {
651 return Ok(Some(ScreenshotHeight::ViewportTimes(m)));
652 }
653 }
654 }
655 }
656 Err(serde::de::Error::custom(format_args!(
657 "screenshot_max_height must be a pixel count (e.g. 2880) or a viewport multiple like \"2x\", found {s:?}"
658 )))
659 }
660 None => Err(serde::de::Error::custom(format_args!(
661 "screenshot_max_height must be a pixel count (e.g. 2880) or a viewport multiple like \"2x\", found {value:?}"
662 ))),
663 },
664 Some(_) => Err(serde::de::Error::custom(format_args!(
665 "screenshot_max_height value too large (max {MAX_SCREENSHOT_HEIGHT} px)"
666 ))),
667 },
668 }
669}
670
671/// Reusable assertion definition referenced by name from `assert` steps.
672///
673/// Definitions can either reference a built-in preset via `preset`, supply a
674/// custom LLM `prompt`, or define a **custom preset** by providing both
675/// `system` and `user_template`. Custom presets support the same template
676/// variables as built-in presets: `{url}`, `{title}`, `{content}`,
677/// `{expected_text}`, `{description}`.
678#[derive(Debug, Deserialize, Clone)]
679pub struct AssertDefinition {
680 /// Unique name used to reference this definition from steps.
681 pub name: String,
682 /// Predefined assertion preset name
683 /// (e.g. `no_error_on_page`, `text_visible`).
684 #[serde(default)]
685 pub preset: Option<String>,
686 /// Custom LLM prompt for assertion evaluation.
687 #[serde(default)]
688 pub prompt: Option<String>,
689 /// System prompt for a custom preset.
690 #[serde(default)]
691 pub system: Option<String>,
692 /// User template (with `{placeholders}`) for a custom preset.
693 #[serde(default)]
694 pub user_template: Option<String>,
695 /// Text that the `text_visible` preset checks for, or the
696 /// `{expected_text}` placeholder value for custom presets.
697 #[serde(default)]
698 pub assert_text: Option<String>,
699 /// Agent endpoint to call for this assertion.
700 #[serde(default)]
701 pub agent: Option<String>,
702 /// Agent task template for this assertion.
703 #[serde(default)]
704 pub task_template: Option<String>,
705}
706
707/// A group of steps that form a single test scenario.
708#[derive(Debug, Deserialize, Clone)]
709pub struct TestGroup {
710 /// Human-readable test name.
711 pub name: String,
712 /// Override the global `start_url` for this test.
713 #[serde(default)]
714 pub start_url: Option<String>,
715 /// Override the global `auto_navigate` for this test.
716 #[serde(default)]
717 pub auto_navigate: Option<bool>,
718 /// Override the global `base_url` for this test.
719 #[serde(default)]
720 pub base_url: Option<String>,
721 /// Override the global `timeout_secs` for this test.
722 #[serde(default)]
723 pub timeout_secs: Option<u64>,
724 /// Override the global `browser_headless` for this test.
725 #[serde(default)]
726 pub browser_headless: Option<bool>,
727 /// Override the global viewport width for this test (applied via CDP
728 /// `Emulation.setDeviceMetricsOverride` before the test runs).
729 #[serde(default)]
730 pub viewport_width: Option<u32>,
731 /// Override the global viewport height for this test.
732 #[serde(default)]
733 pub viewport_height: Option<u32>,
734 /// Per-test budget override.
735 #[serde(default)]
736 pub budget: Option<BudgetDef>,
737 /// Endpoint to use for all steps in this test (can be overridden
738 /// per-step).
739 #[serde(default)]
740 pub endpoint: Option<String>,
741 /// Ordered steps to execute.
742 #[serde(default)]
743 pub steps: Vec<TestStep>,
744}
745
746/// A single step in a test. The `kind` field determines which variant is
747/// deserialized and which field constraints apply.
748#[derive(Debug, Deserialize, Clone)]
749#[serde(tag = "kind")]
750pub enum TestStep {
751 /// Navigate the browser to a URL.
752 #[serde(rename = "navigate")]
753 Navigate {
754 /// URL to navigate to (absolute, or relative to the test's
755 /// base URL).
756 url: String,
757 /// Milliseconds to wait after navigation completes.
758 #[serde(default)]
759 wait_after_ms: Option<u64>,
760 },
761
762 /// Click an element described in natural language.
763 #[serde(rename = "click")]
764 Click {
765 /// Natural language description of the element. The LLM resolves
766 /// this to a CSS selector at runtime.
767 target: String,
768 /// Explicit CSS selector override (bypasses LLM resolution).
769 #[serde(default)]
770 selector: Option<String>,
771 /// Milliseconds to wait after the click.
772 #[serde(default)]
773 wait_after_ms: Option<u64>,
774 /// Endpoint to use for LLM element targeting.
775 #[serde(default)]
776 endpoint: Option<String>,
777 /// Idempotent: when the target element is absent the step is
778 /// reported skipped instead of failed (the action was already
779 /// done / not applicable).
780 #[serde(default)]
781 idempotent: bool,
782 },
783
784 /// Type text into an input element.
785 #[serde(rename = "type")]
786 Type {
787 /// Natural language description of the target input element.
788 target: String,
789 /// Text to type into the element.
790 text: String,
791 /// Explicit CSS selector override (bypasses LLM resolution).
792 #[serde(default)]
793 selector: Option<String>,
794 /// Milliseconds to wait after typing.
795 #[serde(default)]
796 wait_after_ms: Option<u64>,
797 /// Endpoint to use for LLM element targeting.
798 #[serde(default)]
799 endpoint: Option<String>,
800 /// Idempotent: when the target element is absent the step is
801 /// reported skipped instead of failed (the action was already
802 /// done / not applicable).
803 #[serde(default)]
804 idempotent: bool,
805 },
806
807 /// Wait for an element to appear on the page.
808 #[serde(rename = "wait")]
809 Wait {
810 /// Natural language description of the element to wait for.
811 target: String,
812 /// Explicit CSS selector override (bypasses LLM resolution).
813 #[serde(default)]
814 selector: Option<String>,
815 /// Wait until the page's visible text contains this substring
816 /// (alternative to `selector`; either or both may be set — both are
817 /// required to hold when both are set).
818 #[serde(default)]
819 text: Option<String>,
820 /// Maximum milliseconds to wait (default: 10000).
821 #[serde(default)]
822 timeout_ms: Option<u64>,
823 /// Endpoint to use for LLM element targeting.
824 #[serde(default)]
825 endpoint: Option<String>,
826 /// Idempotent: when the condition never becomes true within the
827 /// timeout the step is reported skipped instead of failed (the
828 /// condition was not applicable, e.g. already-authenticated
829 /// pages in a viewport matrix).
830 #[serde(default)]
831 idempotent: bool,
832 },
833
834 /// Evaluate an assertion against the current page content.
835 #[serde(rename = "assert")]
836 Assert {
837 /// Reference to a named `[[definitions]]` entry.
838 #[serde(default)]
839 definition: Option<String>,
840 /// Inline predefined assertion preset (e.g. `no_error_on_page`).
841 #[serde(default)]
842 preset: Option<String>,
843 /// Inline custom LLM prompt for assertion evaluation.
844 #[serde(default)]
845 prompt: Option<String>,
846 /// Text that the `text_visible` preset checks for.
847 #[serde(default)]
848 assert_text: Option<String>,
849 /// Endpoint to use for this assertion's LLM call.
850 #[serde(default)]
851 endpoint: Option<String>,
852 /// Attach a screenshot of the current viewport to the assertion so
853 /// the LLM can evaluate visuals (overlaps, clipping, layout).
854 /// Requires the resolved endpoint to declare `vision = true`.
855 #[serde(default)]
856 screenshot: bool,
857 },
858
859 /// Take a screenshot of the current page.
860 #[serde(rename = "screenshot")]
861 Screenshot {
862 /// File path to save the screenshot (default: `screenshot.png`).
863 #[serde(default)]
864 path: Option<String>,
865 },
866
867 /// Call an A2A agent with a task.
868 #[serde(rename = "agent")]
869 Agent {
870 /// Name of the agent endpoint to call.
871 agent: String,
872 /// Task description / prompt for the agent.
873 task: String,
874 /// Optional definition name with a task template.
875 #[serde(default)]
876 definition: Option<String>,
877 },
878
879 /// Call an MCP server tool.
880 #[serde(rename = "mcp")]
881 Mcp {
882 /// Name of the MCP server endpoint.
883 server: String,
884 /// Tool name to invoke on the server.
885 tool: String,
886 /// Tool arguments as JSON.
887 #[serde(default)]
888 args: Option<serde_json::Value>,
889 },
890}