llm_browser_testkit/scenario.rs
1//! Scenario types for human-readable browser test case definitions.
2//!
3//! Scenarios are written in [TOML](https://toml.io) and describe groups of
4//! browser interaction tests with reusable assertion definitions and
5//! configurable test-level overrides.
6//!
7//! # Structure
8//!
9//! ```toml
10//! [config] # Global defaults
11//! start_url = "/dashboard"
12//!
13//! [[definitions]] # Reusable assertion definitions
14//! name = "no_errors"
15//! preset = "no_error_on_page"
16//!
17//! [[test]] # Test group
18//! name = "Dashboard Smoke"
19//! start_url = "/dashboard" # Override global start_url (optional)
20//!
21//! [[test.steps]] # Ordered steps — the `kind` field
22//! kind = "navigate" # determines which other fields apply
23//! url = "/dashboard"
24//!
25//! [[test.steps]]
26//! kind = "click"
27//! target = "the Login button" # Natural language — LLM resolves to selector
28//!
29//! [[test.steps]]
30//! kind = "assert"
31//! definition = "no_errors"
32//! ```
33//!
34//! ## Step Kinds
35//!
36//! | `kind` | Required fields | Optional fields |
37//! |---------------|-----------------|------------------------------------------|
38//! | `navigate` | `url` | `wait_after_ms` |
39//! | `click` | `target` | `selector`, `wait_after_ms` |
40//! | `type` | `target`, `text`| `selector`, `wait_after_ms` |
41//! | `wait` | `target` | `selector`, `timeout_ms` |
42//! | `assert` | *one of below* | — |
43//! | `screenshot` | — | `path` |
44//! | `agent` | `agent`, `task` | — |
45//! | `mcp` | `server`, `tool`| `args` |
46//!
47//! Assert steps require one of: `definition` (references a named
48//! `[[definitions]]` entry), `preset` (built-in preset name), or `prompt`
49//! (custom LLM evaluation prompt).
50
51use std::collections::HashMap;
52
53use serde::Deserialize;
54use serde_json::Value;
55
56/// Top-level scenario file, deserialized from TOML.
57#[derive(Debug, Deserialize)]
58pub struct Scenario {
59 /// Global configuration (overridable per test).
60 #[serde(default)]
61 pub config: ScenarioConfig,
62
63 /// Reusable assertion definitions referenced by name in `assert` steps.
64 #[serde(default)]
65 pub definitions: Vec<AssertDefinition>,
66
67 /// Ordered test groups to execute.
68 #[serde(default)]
69 pub test: Vec<TestGroup>,
70}
71
72/// Global scenario configuration with per-test overridable fields.
73#[derive(Debug, Deserialize, Default, Clone)]
74pub struct ScenarioConfig {
75 /// Base URL for relative navigation.
76 #[serde(default)]
77 pub base_url: Option<String>,
78 /// LLM server base URL (deprecated; prefer `[config.endpoints]`).
79 #[serde(default)]
80 pub llm_url: Option<String>,
81 /// LLM model name (deprecated; prefer `[config.endpoints]`).
82 #[serde(default)]
83 pub llm_model: Option<String>,
84 /// LLM API key (Bearer token).
85 #[serde(default)]
86 pub llm_api_key: Option<String>,
87 /// Custom HTTP headers as JSON key-value pairs.
88 #[serde(default, deserialize_with = "deserialize_headers")]
89 pub llm_headers: HashMap<String, String>,
90 /// Run browser in headless mode.
91 #[serde(default)]
92 pub browser_headless: Option<bool>,
93 /// HTTP Basic Auth username for browser navigation.
94 #[serde(default)]
95 pub browser_basic_auth_user: Option<String>,
96 /// HTTP Basic Auth password for browser navigation.
97 #[serde(default)]
98 pub browser_basic_auth_password: Option<String>,
99 /// HTTP / browser action timeout in seconds.
100 #[serde(default)]
101 pub timeout_secs: Option<u64>,
102 /// Browser viewport width.
103 #[serde(default)]
104 pub viewport_width: Option<u32>,
105 /// Browser viewport height.
106 #[serde(default)]
107 pub viewport_height: Option<u32>,
108 /// Default URL every test auto-navigates to before running its steps.
109 #[serde(default)]
110 pub start_url: Option<String>,
111 /// Whether to auto-navigate to `start_url` before test steps.
112 ///
113 /// Disable when a test starts with click-based navigation.
114 #[serde(default = "default_auto_navigate")]
115 pub auto_navigate: bool,
116 /// LLM temperature (0.0–1.0). Lower = more deterministic.
117 #[serde(default = "default_temperature")]
118 pub temperature: f64,
119 /// Enable thinking/reasoning tokens. `None` means the provider default
120 /// is used (no `thinking` key is sent). Set to `true`/`false` to
121 /// explicitly enable or disable.
122 #[serde(default)]
123 pub thinking: Option<bool>,
124 /// Provider-specific model parameters merged into the chat completion
125 /// request body (e.g. `effort = "high"` for Anthropic).
126 #[serde(default, deserialize_with = "deserialize_model_params")]
127 pub model_params: HashMap<String, Value>,
128 /// Named endpoints (LLM, MCP, A2A agents) with pricing.
129 #[serde(default)]
130 pub endpoints: HashMap<String, EndpointConfig>,
131 /// Global and per-test budgets for cost/token/call limits.
132 #[serde(default)]
133 pub budgets: BudgetsConfig,
134 /// MCP server exposure configuration.
135 #[serde(default)]
136 pub mcp_server: Option<McpServerConfig>,
137 /// A2A agent server exposure configuration.
138 #[serde(default)]
139 pub a2a_server: Option<A2aServerConfig>,
140 /// Whether to continue running the remaining steps of a test after a
141 /// step fails. Default `false` = fail fast: the first failed step ends
142 /// the test and the rest are reported as skipped. Set to `true` to run
143 /// every step (more diagnostics, more LLM cost on broken apps).
144 #[serde(default)]
145 pub continue_on_failure: bool,
146 /// Longest edge (px) of screenshots attached to `screenshot = true`
147 /// assert steps, before they are JPEG-encoded and sent to the vision
148 /// endpoint. Downscaling keeps vision token cost/quality sane.
149 /// Default: 1400.
150 #[serde(default)]
151 pub screenshot_max_dimension: Option<u32>,
152 /// Height cap (px) of page coverage for screenshots attached to
153 /// `screenshot = true` assert steps. The full scrollable page is
154 /// captured, then split into viewport-tall tiles (each sent as its own
155 /// image part, ordered from the top) covering at most this many pixels
156 /// — below-the-fold content stays visible to the vision model at 1:1
157 /// detail while token cost stays bounded. Accepts an absolute pixel
158 /// count (`2880`) or a viewport multiple (`"20x"` = twenty times the
159 /// currently applied viewport height, which follows viewport-matrix
160 /// and per-test overrides). A value below the viewport height is
161 /// raised to it, so the visible viewport is always fully included
162 /// (`0` / `"0x"` therefore means "viewport only", the pre-full-page
163 /// behavior). Default: `"20x"`.
164 #[serde(default, deserialize_with = "deserialize_screenshot_max_height")]
165 pub screenshot_max_height: Option<ScreenshotHeight>,
166 /// Directory for failure artifacts (screenshots, page snapshots).
167 /// Defaults to `artifacts`.
168 #[serde(default)]
169 pub artifacts_dir: Option<String>,
170 /// Optional viewport matrix: when set, every test in the scenario is
171 /// expanded into one variant per viewport (e.g. mobile/tablet/desktop).
172 /// Each variant overrides the test's viewport and gets a ` — <name>`
173 /// suffix on the test name. Per-test budgets apply per variant.
174 #[serde(default)]
175 pub viewport_matrix: Option<ViewportMatrix>,
176 /// Class-name prefixes the `layout_no_issues` scan skips: elements
177 /// whose class matches any prefix are ignored by the fixed-element,
178 /// text-clipped, and overlap checks. Defaults cover the Angular CDK
179 /// screen-reader helpers (`cdk-visually-hidden`,
180 /// `cdk-describedby-message-container`, `cdk-overlay-container`),
181 /// which are intentionally 1x1 / off-screen.
182 #[serde(default = "default_layout_ignore_classes")]
183 pub layout_ignore_classes: Vec<String>,
184 /// Concurrency group for parallel runs across scenario files.
185 /// When several scenario files are run together (`--parallel > 1`),
186 /// files that declare the **same** `concurrency_group` are never
187 /// executed at the same time — use this for files that touch the same
188 /// shared backend state and would interfere if run concurrently. A file
189 /// with no group gets its own implicit group, so distinct files run in
190 /// parallel by default. Only honored when files are passed to the
191 /// runner as a batch; has no effect on the steps within a single file,
192 /// which always run sequentially.
193 #[serde(default)]
194 pub concurrency_group: Option<String>,
195 /// Default for per-endpoint `cache` across all endpoints (default
196 /// `true`). Set to `false` to disable provider-side prompt-cache
197 /// markers globally.
198 #[serde(default)]
199 pub cache: Option<bool>,
200 /// Default for per-endpoint `pricing.cache_pricing` across all
201 /// endpoints (default `true`). Set to `false` to bill all prompt
202 /// tokens at the flat input price instead of cache rates.
203 #[serde(default)]
204 pub cache_pricing: Option<bool>,
205}
206
207/// A list of named viewports a scenario is expanded across.
208#[derive(Debug, Deserialize, Clone, Default)]
209pub struct ViewportMatrix {
210 /// The viewport variants (`{name, width, height}`).
211 #[serde(default)]
212 pub viewports: Vec<ViewportDef>,
213}
214
215/// Height cap for screenshots attached to `screenshot = true` assert
216/// steps: either an absolute pixel count or a multiple of the currently
217/// applied viewport height.
218#[derive(Debug, Clone, PartialEq)]
219pub enum ScreenshotHeight {
220 /// Absolute height in pixels.
221 Pixels(u32),
222 /// Multiple of the active viewport height (e.g. `2x` = twice the
223 /// current viewport's pixel height).
224 ViewportTimes(f64),
225}
226
227/// Ceiling (px) applied when resolving screenshot height specs, so an
228/// absurdly large multiplier can never overflow or produce an unusable
229/// capture.
230const MAX_SCREENSHOT_HEIGHT: u32 = 4_000_000;
231
232impl ScreenshotHeight {
233 /// Resolves this height spec to a pixel value for the given viewport
234 /// height. Multipliers are rounded and clamped to
235 /// [`MAX_SCREENSHOT_HEIGHT`].
236 #[must_use]
237 pub fn to_px(&self, viewport_height: u32) -> u32 {
238 match self {
239 Self::Pixels(px) => *px,
240 Self::ViewportTimes(mult) => {
241 let scaled = f64::from(viewport_height) * mult;
242 #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)]
243 let px = scaled
244 .round()
245 .max(1.0)
246 .min(f64::from(MAX_SCREENSHOT_HEIGHT)) as u32;
247 px
248 }
249 }
250 }
251}
252
253/// One named viewport size in a matrix.
254#[derive(Debug, Deserialize, Clone)]
255pub struct ViewportDef {
256 /// Human-readable variant name (appended to test names, e.g.
257 /// `— mobile`).
258 pub name: String,
259 /// Browser viewport width in pixels.
260 pub width: u32,
261 /// Browser viewport height in pixels.
262 pub height: u32,
263}
264
265/// A named endpoint definition with pricing.
266#[derive(Debug, Deserialize, Clone, Default)]
267pub struct EndpointConfig {
268 /// Endpoint type: `llm`, `mcp`, or `a2a`.
269 #[serde(rename = "type")]
270 pub endpoint_type: EndpointType,
271 /// Base URL for the endpoint.
272 #[serde(default)]
273 pub url: Option<String>,
274 /// Model name (LLM endpoints only).
275 #[serde(default)]
276 pub model: Option<String>,
277 /// API key / bearer token.
278 #[serde(default)]
279 pub api_key: Option<String>,
280 /// Custom HTTP headers as JSON key-value pairs.
281 #[serde(default, deserialize_with = "deserialize_headers")]
282 pub headers: HashMap<String, String>,
283 /// Pricing configuration.
284 #[serde(default)]
285 pub pricing: Option<PricingConfig>,
286 /// Automatically fetch exact per-token pricing for this endpoint at
287 /// startup from a provider's public pricing API, filling in any pricing
288 /// fields left unset (explicit `pricing` values win). Defaults to
289 /// `"auto"`: Bedrock endpoints use the AWS Price List and `openrouter.ai`
290 /// URLs use the `OpenRouter` models API; other providers are a no-op.
291 /// Force a source with `"bedrock"` / `"openrouter"`, or disable the
292 /// lookup with `"off"` / `"none"` / `"disabled"`.
293 #[serde(default)]
294 pub pricing_source: Option<String>,
295 /// Task types this endpoint serves by default
296 /// (e.g. `["targeting", "assertion"]`).
297 #[serde(default)]
298 pub default_for: Vec<String>,
299 /// Command to launch an MCP server subprocess (stdio transport).
300 #[serde(default)]
301 pub command: Option<String>,
302 /// Arguments for the MCP server command.
303 #[serde(default)]
304 pub args: Vec<String>,
305 /// Whether this LLM endpoint accepts image parts (vision) in addition
306 /// to text. `assert` steps with `screenshot = true` require a vision
307 /// endpoint.
308 #[serde(default)]
309 pub vision: bool,
310 /// How often a single chat completion against this endpoint is retried
311 /// on transient failures (HTTP 429/5xx, empty 200 bodies, network
312 /// errors) before the fallback chain is tried. Default: 3 (override
313 /// globally with `HARNESS_LLM_CALL_ATTEMPTS`).
314 #[serde(default)]
315 pub max_attempts: Option<u32>,
316 /// Ordered names of other endpoints to try when this endpoint exhausts
317 /// its attempts. Only LLM endpoints are eligible. Useful for pairing a
318 /// cheap primary model with a more powerful/expensive fallback.
319 #[serde(default)]
320 pub fallbacks: Vec<String>,
321 /// LLM provider protocol: `openai` (default, OpenAI-compatible chat
322 /// completions), `azure` (`Azure` `OpenAI`), or `bedrock` (AWS Bedrock
323 /// Converse API; requires the `aws` cargo feature).
324 #[serde(default)]
325 pub provider: Provider,
326 /// `Azure` `OpenAI` deployment name (`provider = "azure"`). Defaults to
327 /// `model` when unset.
328 #[serde(default)]
329 pub deployment: Option<String>,
330 /// `Azure` `OpenAI` API version (`provider = "azure"`). Defaults to
331 /// `2024-10-21`.
332 #[serde(default)]
333 pub api_version: Option<String>,
334 /// Authentication configuration for LLM endpoints (API key, token
335 /// command, Entra ID client credentials / managed identity).
336 #[serde(default)]
337 pub auth: AuthConfig,
338 /// Extra HTTP headers produced by running a command per call, keyed by
339 /// header name. The command's stdout (first line) becomes the header
340 /// value. Provider-agnostic — applies to every LLM provider.
341 #[serde(default, deserialize_with = "deserialize_headers")]
342 pub header_commands: HashMap<String, String>,
343 /// AWS credential settings (`provider = "bedrock"`).
344 #[serde(default)]
345 pub aws: AwsConfig,
346 /// Send provider-side prompt-cache markers from this endpoint
347 /// (default `true`). Only providers that require explicit markers are
348 /// affected: AWS Bedrock gets a `cachePoint` block, and Anthropic-style
349 /// OpenAI-compatible models (model name contains `claude`/`anthropic`,
350 /// e.g. via `OpenRouter`) get a `cache_control: ephemeral` block on the
351 /// system message. `OpenAI`, `Azure`, Groq, xAI and `DeepSeek` cache
352 /// automatically and need no markers. Set to `false` to disable.
353 #[serde(default)]
354 pub cache: Option<bool>,
355}
356
357/// Type discriminator for endpoint configuration.
358#[derive(Debug, Deserialize, Clone, PartialEq, Eq, Default)]
359#[serde(rename_all = "lowercase")]
360pub enum EndpointType {
361 /// OpenAI-compatible LLM API.
362 #[default]
363 Llm,
364 /// Model Context Protocol server.
365 Mcp,
366 /// Agent-to-Agent protocol agent.
367 A2a,
368}
369
370/// LLM provider protocol used by an LLM endpoint.
371#[derive(Debug, Deserialize, Clone, Copy, PartialEq, Eq, Default)]
372#[serde(rename_all = "lowercase")]
373pub enum Provider {
374 /// OpenAI-compatible Chat Completions API
375 /// (`POST <url>/v1/chat/completions`).
376 #[default]
377 Openai,
378 /// `Azure` `OpenAI`
379 /// (`POST <url>/openai/deployments/<deployment>/chat/completions`).
380 Azure,
381 /// AWS Bedrock Converse API (`POST https://bedrock-runtime.<region>
382 /// .amazonaws.com/model/<model>/converse`). Requires the `aws` cargo
383 /// feature; uses the standard AWS credential chain unless overridden.
384 Bedrock,
385}
386
387/// Authentication mode for an LLM endpoint.
388#[derive(Debug, Deserialize, Clone, Copy, PartialEq, Eq, Default)]
389#[serde(rename_all = "kebab-case")]
390pub enum AuthMode {
391 /// Static API key. Sent as `Authorization: Bearer <key>` by default;
392 /// set [`AuthConfig::api_key_header`] (or use the `azure` provider)
393 /// to send it in a different header.
394 #[default]
395 ApiKey,
396 /// Execute a command and use its stdout (first line) as the bearer
397 /// token. Uses the `token_command` field. Provider-agnostic escape
398 /// hatch — e.g. `az account get-access-token …`.
399 TokenCommand,
400 /// Entra ID (`Azure` AD) OAuth 2.0 client-credentials grant: exchanges
401 /// `client_id` + `client_secret` in `tenant_id` for a bearer token.
402 EntraClientCredentials,
403 /// Entra ID managed identity: fetches a bearer token from the IMDS
404 /// endpoint (works on `Azure` VMs / App Service / ACI with a
405 /// system-assigned identity; zero credentials in the config).
406 EntraManagedIdentity,
407}
408
409/// Authentication settings for an LLM endpoint.
410#[derive(Debug, Deserialize, Clone, Default)]
411pub struct AuthConfig {
412 /// Which authentication mode to use. Default: `api-key`.
413 #[serde(default)]
414 pub mode: AuthMode,
415 /// Header name that receives the API key instead of
416 /// `Authorization: Bearer` (`mode = "api-key"`). For example the `Azure`
417 /// `OpenAI` `api-key` header.
418 #[serde(default)]
419 pub api_key_header: Option<String>,
420 /// Command whose stdout (first line) is used as the bearer token
421 /// (`mode = "token-command"`).
422 #[serde(default)]
423 pub token_command: Option<String>,
424 /// Entra tenant id (`mode = "entra-client-credentials"`,
425 /// `"entra-managed-identity"`).
426 #[serde(default)]
427 pub tenant_id: Option<String>,
428 /// Entra client id (`mode = "entra-client-credentials"`).
429 #[serde(default)]
430 pub client_id: Option<String>,
431 /// Entra client secret (`mode = "entra-client-credentials"`).
432 #[serde(default)]
433 pub client_secret: Option<String>,
434 /// Entra token scope. Defaults to
435 /// `https://cognitiveservices.azure.com/.default`.
436 #[serde(default)]
437 pub scope: Option<String>,
438 /// Override for the Entra token endpoint
439 /// (`mode = "entra-client-credentials"`), default
440 /// `https://login.microsoftonline.com`. Also used as the IMDS identity
441 /// endpoint base for `"entra-managed-identity"` (default
442 /// `http://169.254.169.254`). Mainly useful for tests and proxies.
443 #[serde(default)]
444 pub token_url: Option<String>,
445 /// Seconds to reuse a fetched (non-API-key) token before refetching.
446 /// For Entra modes the server-issued expiry is used when available;
447 /// this caps the reuse window. Default 300.
448 #[serde(default)]
449 pub cache_ttl_secs: Option<u64>,
450}
451
452/// AWS credential and region settings for a `bedrock` provider endpoint.
453///
454/// When no explicit access key is given, the standard AWS credential chain
455/// is used (env vars, shared config `~/.aws/config` + `~/.aws/credentials`,
456/// SSO, ECS/IMDS) — the same behavior as the AWS CLI. `profile` selects a
457/// named profile from the shared config, and `region` overrides the chain
458/// default.
459#[derive(Debug, Deserialize, Clone, Default)]
460pub struct AwsConfig {
461 /// Named profile from `~/.aws/config` / `~/.aws/credentials` to use
462 /// (default: the AWS CLI default selection via `AWS_PROFILE` or
463 /// `default`).
464 #[serde(default)]
465 pub profile: Option<String>,
466 /// AWS region (default: `AWS_REGION` env or the profile's region;
467 /// required if neither is set).
468 #[serde(default)]
469 pub region: Option<String>,
470 /// Explicit access key id (bypasses the credential chain).
471 #[serde(default)]
472 pub access_key_id: Option<String>,
473 /// Explicit secret access key (with `access_key_id`).
474 #[serde(default)]
475 pub secret_access_key: Option<String>,
476 /// Optional session token for explicit temporary credentials.
477 #[serde(default)]
478 pub session_token: Option<String>,
479}
480
481/// Pricing configuration for an endpoint.
482#[derive(Debug, Deserialize, Clone, Default)]
483pub struct PricingConfig {
484 /// Cost per 1M input tokens (USD).
485 #[serde(default)]
486 pub input_per_1m_tokens: f64,
487 /// Cost per 1M output tokens (USD).
488 #[serde(default)]
489 pub output_per_1m_tokens: f64,
490 /// Flat cost per call (USD), used for MCP/agent endpoints.
491 #[serde(default)]
492 pub per_call: f64,
493 /// Cost per 1M cached-input (prompt cache read) tokens (USD). When
494 /// unset, `input_per_1m_tokens * cache_read_multiplier` is used.
495 #[serde(default)]
496 pub cached_input_per_1m_tokens: Option<f64>,
497 /// Cost per 1M cache-write (cache creation) input tokens (USD). When
498 /// unset, `input_per_1m_tokens * cache_write_multiplier` is used.
499 #[serde(default)]
500 pub cache_write_per_1m_tokens: Option<f64>,
501 /// Multiplier applied to `input_per_1m_tokens` for cache reads when
502 /// `cached_input_per_1m_tokens` is unset. Default: 0.1.
503 #[serde(default)]
504 pub cache_read_multiplier: Option<f64>,
505 /// Multiplier applied to `input_per_1m_tokens` for cache writes when
506 /// `cache_write_per_1m_tokens` is unset. Default: 1.25.
507 #[serde(default)]
508 pub cache_write_multiplier: Option<f64>,
509 /// Bill cache reads/writes at their cache rates (default `true`). Set
510 /// to `false` to bill every prompt token at the flat input price.
511 #[serde(default)]
512 pub cache_pricing: Option<bool>,
513}
514
515/// Budget limits for test execution.
516#[derive(Debug, Deserialize, Clone, Default)]
517pub struct BudgetsConfig {
518 /// Global budget across all tests in the scenario.
519 #[serde(default)]
520 pub global: Option<BudgetDef>,
521 /// Default per-test budget. Individual tests can override.
522 #[serde(default)]
523 pub per_test_default: Option<BudgetDef>,
524}
525
526/// A budget definition with limits and enforcement mode.
527#[derive(Debug, Deserialize, Clone)]
528pub struct BudgetDef {
529 /// Maximum cost in USD.
530 #[serde(default)]
531 pub max_cost: Option<f64>,
532 /// Maximum total tokens (input + output).
533 #[serde(default)]
534 pub max_tokens: Option<u64>,
535 /// Maximum number of calls (LLM, MCP, agent combined).
536 #[serde(default)]
537 pub max_calls: Option<u64>,
538 /// Enforcement mode: `hard` (abort) or `soft` (warn and continue).
539 #[serde(default)]
540 pub enforcement: Option<BudgetEnforcement>,
541}
542
543/// Budget enforcement strategy.
544#[derive(Debug, Deserialize, Clone, PartialEq, Eq)]
545#[serde(rename_all = "lowercase")]
546pub enum BudgetEnforcement {
547 /// Abort the test or run when budget is exceeded.
548 Hard,
549 /// Log a warning but continue execution.
550 Soft,
551}
552
553/// MCP server exposure configuration.
554#[derive(Debug, Deserialize, Clone)]
555pub struct McpServerConfig {
556 /// Whether to enable the embedded MCP server.
557 #[serde(default)]
558 pub enabled: bool,
559 /// Port to listen on.
560 #[serde(default = "default_mcp_port")]
561 pub port: u16,
562}
563
564const fn default_mcp_port() -> u16 {
565 3000
566}
567
568/// A2A agent server exposure configuration.
569#[derive(Debug, Deserialize, Clone)]
570pub struct A2aServerConfig {
571 /// Whether to enable the embedded A2A agent server.
572 #[serde(default)]
573 pub enabled: bool,
574 /// Port to listen on.
575 #[serde(default = "default_a2a_port")]
576 pub port: u16,
577}
578
579const fn default_a2a_port() -> u16 {
580 3100
581}
582
583fn deserialize_headers<'de, D>(deserializer: D) -> Result<HashMap<String, String>, D::Error>
584where
585 D: serde::Deserializer<'de>,
586{
587 let raw: Option<serde_json::Value> = Option::deserialize(deserializer)?;
588 let Some(json) = raw else {
589 return Ok(HashMap::new());
590 };
591 let serde_json::Value::Object(obj) = json else {
592 return Ok(HashMap::new());
593 };
594 Ok(obj
595 .into_iter()
596 .filter_map(|(k, v)| v.as_str().map(|s| (k, s.to_owned())))
597 .collect())
598}
599
600const fn default_auto_navigate() -> bool {
601 true
602}
603
604fn default_layout_ignore_classes() -> Vec<String> {
605 vec![
606 "cdk-visually-hidden".to_owned(),
607 "cdk-describedby-message-container".to_owned(),
608 "cdk-overlay-container".to_owned(),
609 ]
610}
611
612const fn default_temperature() -> f64 {
613 0.0
614}
615
616fn deserialize_model_params<'de, D>(deserializer: D) -> Result<HashMap<String, Value>, D::Error>
617where
618 D: serde::Deserializer<'de>,
619{
620 #[derive(Deserialize)]
621 #[serde(untagged)]
622 enum Raw {
623 Map(HashMap<String, Value>),
624 Table(HashMap<String, Value>),
625 }
626 let raw: Option<Raw> = Option::deserialize(deserializer)?;
627 Ok(match raw {
628 Some(Raw::Map(m) | Raw::Table(m)) => m,
629 None => HashMap::new(),
630 })
631}
632
633fn deserialize_screenshot_max_height<'de, D>(
634 deserializer: D,
635) -> Result<Option<ScreenshotHeight>, D::Error>
636where
637 D: serde::Deserializer<'de>,
638{
639 let raw: Option<serde_json::Value> = Option::deserialize(deserializer)?;
640 match raw {
641 None => Ok(None),
642 Some(value) => match value.as_u64() {
643 // Integer: absolute pixel count.
644 Some(px) if px <= u64::from(MAX_SCREENSHOT_HEIGHT) => {
645 #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)]
646 Ok(Some(ScreenshotHeight::Pixels(px as u32)))
647 }
648 // String: viewport multiple like "2x" or "1.5x".
649 None => match value.as_str() {
650 Some(s) => {
651 let lower = s.trim().to_lowercase();
652 if lower.ends_with('x') {
653 let num = lower.strip_suffix("x");
654 if let Some(num) = num {
655 if let Ok(m) = num.parse::<f64>() {
656 if m > 0.0 {
657 return Ok(Some(ScreenshotHeight::ViewportTimes(m)));
658 }
659 }
660 }
661 }
662 Err(serde::de::Error::custom(format_args!(
663 "screenshot_max_height must be a pixel count (e.g. 2880) or a viewport multiple like \"2x\", found {s:?}"
664 )))
665 }
666 None => Err(serde::de::Error::custom(format_args!(
667 "screenshot_max_height must be a pixel count (e.g. 2880) or a viewport multiple like \"2x\", found {value:?}"
668 ))),
669 },
670 Some(_) => Err(serde::de::Error::custom(format_args!(
671 "screenshot_max_height value too large (max {MAX_SCREENSHOT_HEIGHT} px)"
672 ))),
673 },
674 }
675}
676
677/// Reusable assertion definition referenced by name from `assert` steps.
678///
679/// Definitions can either reference a built-in preset via `preset`, supply a
680/// custom LLM `prompt`, or define a **custom preset** by providing both
681/// `system` and `user_template`. Custom presets support the same template
682/// variables as built-in presets: `{url}`, `{title}`, `{content}`,
683/// `{expected_text}`, `{description}`.
684#[derive(Debug, Deserialize, Clone)]
685pub struct AssertDefinition {
686 /// Unique name used to reference this definition from steps.
687 pub name: String,
688 /// Predefined assertion preset name
689 /// (e.g. `no_error_on_page`, `text_visible`).
690 #[serde(default)]
691 pub preset: Option<String>,
692 /// Custom LLM prompt for assertion evaluation.
693 #[serde(default)]
694 pub prompt: Option<String>,
695 /// System prompt for a custom preset.
696 #[serde(default)]
697 pub system: Option<String>,
698 /// User template (with `{placeholders}`) for a custom preset.
699 #[serde(default)]
700 pub user_template: Option<String>,
701 /// Text that the `text_visible` preset checks for, or the
702 /// `{expected_text}` placeholder value for custom presets.
703 #[serde(default)]
704 pub assert_text: Option<String>,
705 /// Agent endpoint to call for this assertion.
706 #[serde(default)]
707 pub agent: Option<String>,
708 /// Agent task template for this assertion.
709 #[serde(default)]
710 pub task_template: Option<String>,
711}
712
713/// A group of steps that form a single test scenario.
714#[derive(Debug, Deserialize, Clone)]
715pub struct TestGroup {
716 /// Human-readable test name.
717 pub name: String,
718 /// Override the global `start_url` for this test.
719 #[serde(default)]
720 pub start_url: Option<String>,
721 /// Override the global `auto_navigate` for this test.
722 #[serde(default)]
723 pub auto_navigate: Option<bool>,
724 /// Override the global `base_url` for this test.
725 #[serde(default)]
726 pub base_url: Option<String>,
727 /// Override the global `timeout_secs` for this test.
728 #[serde(default)]
729 pub timeout_secs: Option<u64>,
730 /// Override the global `browser_headless` for this test.
731 #[serde(default)]
732 pub browser_headless: Option<bool>,
733 /// Override the global viewport width for this test (applied via CDP
734 /// `Emulation.setDeviceMetricsOverride` before the test runs).
735 #[serde(default)]
736 pub viewport_width: Option<u32>,
737 /// Override the global viewport height for this test.
738 #[serde(default)]
739 pub viewport_height: Option<u32>,
740 /// Per-test budget override.
741 #[serde(default)]
742 pub budget: Option<BudgetDef>,
743 /// Endpoint to use for all steps in this test (can be overridden
744 /// per-step).
745 #[serde(default)]
746 pub endpoint: Option<String>,
747 /// Ordered steps to execute.
748 #[serde(default)]
749 pub steps: Vec<TestStep>,
750}
751
752/// A single step in a test. The `kind` field determines which variant is
753/// deserialized and which field constraints apply.
754#[derive(Debug, Deserialize, Clone)]
755#[serde(tag = "kind")]
756pub enum TestStep {
757 /// Navigate the browser to a URL.
758 #[serde(rename = "navigate")]
759 Navigate {
760 /// URL to navigate to (absolute, or relative to the test's
761 /// base URL).
762 url: String,
763 /// Milliseconds to wait after navigation completes.
764 #[serde(default)]
765 wait_after_ms: Option<u64>,
766 },
767
768 /// Click an element described in natural language.
769 #[serde(rename = "click")]
770 Click {
771 /// Natural language description of the element. The LLM resolves
772 /// this to a CSS selector at runtime.
773 target: String,
774 /// Explicit CSS selector override (bypasses LLM resolution).
775 #[serde(default)]
776 selector: Option<String>,
777 /// Milliseconds to wait after the click.
778 #[serde(default)]
779 wait_after_ms: Option<u64>,
780 /// Endpoint to use for LLM element targeting.
781 #[serde(default)]
782 endpoint: Option<String>,
783 /// Idempotent: when the target element is absent the step is
784 /// reported skipped instead of failed (the action was already
785 /// done / not applicable).
786 #[serde(default)]
787 idempotent: bool,
788 },
789
790 /// Type text into an input element.
791 #[serde(rename = "type")]
792 Type {
793 /// Natural language description of the target input element.
794 target: String,
795 /// Text to type into the element.
796 text: String,
797 /// Explicit CSS selector override (bypasses LLM resolution).
798 #[serde(default)]
799 selector: Option<String>,
800 /// Milliseconds to wait after typing.
801 #[serde(default)]
802 wait_after_ms: Option<u64>,
803 /// Endpoint to use for LLM element targeting.
804 #[serde(default)]
805 endpoint: Option<String>,
806 /// Idempotent: when the target element is absent the step is
807 /// reported skipped instead of failed (the action was already
808 /// done / not applicable).
809 #[serde(default)]
810 idempotent: bool,
811 },
812
813 /// Wait for an element to appear on the page.
814 #[serde(rename = "wait")]
815 Wait {
816 /// Natural language description of the element to wait for.
817 target: String,
818 /// Explicit CSS selector override (bypasses LLM resolution).
819 #[serde(default)]
820 selector: Option<String>,
821 /// Wait until the page's visible text contains this substring
822 /// (alternative to `selector`; either or both may be set — both are
823 /// required to hold when both are set).
824 #[serde(default)]
825 text: Option<String>,
826 /// Maximum milliseconds to wait (default: 10000).
827 #[serde(default)]
828 timeout_ms: Option<u64>,
829 /// Endpoint to use for LLM element targeting.
830 #[serde(default)]
831 endpoint: Option<String>,
832 /// Idempotent: when the condition never becomes true within the
833 /// timeout the step is reported skipped instead of failed (the
834 /// condition was not applicable, e.g. already-authenticated
835 /// pages in a viewport matrix).
836 #[serde(default)]
837 idempotent: bool,
838 },
839
840 /// Evaluate an assertion against the current page content.
841 #[serde(rename = "assert")]
842 Assert {
843 /// Reference to a named `[[definitions]]` entry.
844 #[serde(default)]
845 definition: Option<String>,
846 /// Inline predefined assertion preset (e.g. `no_error_on_page`).
847 #[serde(default)]
848 preset: Option<String>,
849 /// Inline custom LLM prompt for assertion evaluation.
850 #[serde(default)]
851 prompt: Option<String>,
852 /// Text that the `text_visible` preset checks for.
853 #[serde(default)]
854 assert_text: Option<String>,
855 /// Endpoint to use for this assertion's LLM call.
856 #[serde(default)]
857 endpoint: Option<String>,
858 /// Attach a screenshot of the current viewport to the assertion so
859 /// the LLM can evaluate visuals (overlaps, clipping, layout).
860 /// Requires the resolved endpoint to declare `vision = true`.
861 #[serde(default)]
862 screenshot: bool,
863 },
864
865 /// Take a screenshot of the current page.
866 #[serde(rename = "screenshot")]
867 Screenshot {
868 /// File path to save the screenshot (default: `screenshot.png`).
869 #[serde(default)]
870 path: Option<String>,
871 },
872
873 /// Call an A2A agent with a task.
874 #[serde(rename = "agent")]
875 Agent {
876 /// Name of the agent endpoint to call.
877 agent: String,
878 /// Task description / prompt for the agent.
879 task: String,
880 /// Optional definition name with a task template.
881 #[serde(default)]
882 definition: Option<String>,
883 },
884
885 /// Call an MCP server tool.
886 #[serde(rename = "mcp")]
887 Mcp {
888 /// Name of the MCP server endpoint.
889 server: String,
890 /// Tool name to invoke on the server.
891 tool: String,
892 /// Tool arguments as JSON.
893 #[serde(default)]
894 args: Option<serde_json::Value>,
895 },
896}