Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::UsageTracker;
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::llm_chat_vision_with_usage_chain;
15use crate::llm_chat_with_usage_chain;
16use crate::mcp_client::McpClient;
17use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
18use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
19use crate::truncate;
20use crate::LlmConfig;
21use crate::DOM_EXTRACT_JS;
22
23/// One detected layout defect (`layout_no_issues` preset).
24#[derive(Debug, serde::Deserialize)]
25struct LayoutIssue {
26    #[serde(rename = "type")]
27    issue_type: String,
28    element: String,
29    detail: String,
30}
31
32/// In-browser DOM layout scan for `layout_no_issues`.
33///
34/// Geometry-only checks (no LLM, no pixels):
35/// 1. page horizontal overflow (`scrollWidth` > viewport width);
36/// 2. elements outside the viewport that scrolling cannot reveal
37///    (fixed elements off-screen, left/negative overflow, right-edge
38///    overflow beyond the horizontally scrollable area, and bottom
39///    overflow on a page that cannot scroll down) — below-the-fold
40///    content on a tall scrollable page is normal flow, NOT a defect;
41/// 3. text clipped by `overflow: hidden` containers whose content
42///    is measurably larger than the box;
43/// 4. interactive elements (buttons/links/inputs) whose center point
44///    is covered by a different element that would intercept the click.
45///
46/// Elements whose class matches a configured ignore prefix (default:
47/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
48/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
49/// skipped — they are intentionally 1x1 / off-screen. The prefix list
50/// is injected at `__IGNORE_CLASSES__` from
51/// [`ScenarioConfig::layout_ignore_classes`].
52///
53/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
54/// sticky headers) is excluded by the position/relation filters.
55const LAYOUT_SCAN_JS: &str = r#"
56(() => {
57  const issues = [];
58  const push = (type, el, detail) => {
59    if (issues.length >= 30) return;
60    let element = el.tagName.toLowerCase();
61    if (el.id) element += '#' + el.id;
62    else if (typeof el.className === 'string' && el.className.trim())
63      element += '.' + el.className.trim().split(/\s+/).join('.');
64    issues.push({ type, element, detail: String(detail).slice(0, 220) });
65  };
66  const vw = document.documentElement.clientWidth || window.innerWidth;
67  const vh = document.documentElement.clientHeight || window.innerHeight;
68  if (!vw || !vh) return JSON.stringify(issues);
69  const de = document.documentElement;
70  // 1. Page-level horizontal overflow.
71  if (de.scrollWidth > vw + 2)
72    push('page-overflow-x', de,
73      'page scrollWidth ' + de.scrollWidth + ' exceeds viewport width ' + vw);
74  // Class-name prefixes to skip (injected; default = Angular CDK
75  // screen-reader helpers, which are intentionally 1x1 / off-screen).
76  const ignorePrefixes = __IGNORE_CLASSES__;
77  const isIgnored = (el) => {
78    if (typeof el.className !== 'string' || !el.className.trim()) return false;
79    const classes = el.className.trim().split(/\s+/);
80    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
81  };
82  const vScrollable = de.scrollHeight > vh + 2;
83  const hScrollable = de.scrollWidth > vw + 2;
84  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
85  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
86  const hasContent = (el) =>
87    ((el.textContent || '').trim().length > 0) ||
88    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
89  // 2. Elements outside the viewport that scrolling cannot reveal.
90  for (const el of all) {
91    const cs = getComputedStyle(el);
92    if (!visible(cs)) continue;
93    if (isIgnored(el)) continue;
94    const r = el.getBoundingClientRect();
95    if (r.width < 2 || r.height < 2) continue;
96    if (!hasContent(el) && el.children.length === 0) continue;
97    if (cs.position === 'fixed') {
98      // Fixed elements never move with the scroll: any edge outside the
99      // viewport is unreachable content and therefore a defect.
100      const overTop = -r.top;
101      const overLeft = -r.left;
102      const overRight = r.right - vw;
103      const overBottom = r.bottom - vh;
104      if (overTop > 2 || overLeft > 2 || overRight > 2 || overBottom > 2) {
105        let where = '';
106        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
107        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
108        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
109        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
110        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
111        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
112        push('element-out-of-viewport', el,
113          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
114      }
115      continue;
116    }
117    if (cs.position === 'sticky') continue;
118    // Negative left overflow cannot be reached by scrolling (scrollLeft
119    // never goes below 0).
120    if (r.left < -2) {
121      push('element-out-of-viewport', el,
122        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
123      continue;
124    }
125    // Negative top with the page at the top means the element sits above
126    // the document origin — also unreachable.
127    if (r.top < -2 && de.scrollTop <= 2) {
128      push('element-out-of-viewport', el,
129        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
130      continue;
131    }
132    const overRight = r.right - vw;
133    // Right-edge overflow is only a defect when the page cannot scroll
134    // horizontally to reveal it (or the element sticks out past the
135    // scrollable content width itself).
136    if (overRight > 2 && (!hScrollable || r.right > de.scrollWidth + 2)) {
137      push('element-out-of-viewport', el,
138        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
139      continue;
140    }
141    // Below-the-fold content on a scrollable page is normal (tall landing
142    // pages); only flag bottom overflow the user can never scroll to.
143    const overBottom = r.bottom - vh;
144    if (overBottom > 2 && (!vScrollable || r.bottom > de.scrollHeight + 2)) {
145      push('element-out-of-viewport', el,
146        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
147    }
148  }
149  // 3. Text clipped by overflow:hidden containers.
150  for (const el of all) {
151    if (isIgnored(el)) continue;
152    const cs = getComputedStyle(el);
153    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
154    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
155    if (!(el.textContent || '').trim()) continue;
156    push('text-clipped', el,
157      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
158      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
159  }
160  // 4. Interactive elements covered by a different element.
161  const interactive =
162    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
163  const targets = document.querySelectorAll(interactive);
164  for (const el of targets) {
165    if (isIgnored(el)) continue;
166    const r = el.getBoundingClientRect();
167    if (r.width < 6 || r.height < 6) continue;
168    const cs = getComputedStyle(el);
169    if (!visible(cs)) continue;
170    const cx = r.left + r.width / 2;
171    const cy = r.top + r.height / 2;
172    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
173    const top = document.elementFromPoint(cx, cy);
174    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
175    if (isIgnored(top)) continue;
176    const tcs = getComputedStyle(top);
177    if (!visible(tcs)) continue;
178    if (tcs.pointerEvents === 'none') continue;
179    const tr = top.getBoundingClientRect();
180    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
181    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
182      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
183    push('element-overlap', el,
184      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
185      ' is covered by <' + tname + '>');
186  }
187  return JSON.stringify(issues);
188})()
189"#;
190
191/// How long the CDP connection stays open after the browser goes quiet.
192///
193/// `headless_chrome` ships a 30s default and tears down the entire connection
194/// when no traffic arrives for that long; a run must own its connection for
195/// its full duration instead.
196const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
197
198/// Executes a [`Scenario`] against a real browser with optional LLM
199/// assistance for element targeting and assertions.
200pub struct ScenarioRunner {
201    config: ScenarioConfig,
202    definitions: HashMap<String, AssertDefinition>,
203    llm: LlmConfig,
204    timeout: Duration,
205    viewport_width: u32,
206    viewport_height: u32,
207    /// The viewport currently applied in the browser (CDP emulation).
208    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
209    applied_viewport: std::cell::Cell<(u32, u32)>,
210    endpoints: EndpointRegistry,
211    usage: Arc<UsageTracker>,
212    budgets: BudgetTracker,
213    /// Directory for failure artifacts (screenshots).
214    artifacts_dir: PathBuf,
215}
216
217/// Aggregated results from a scenario run.
218#[derive(Debug, Default)]
219pub struct RunReport {
220    /// Number of tests that passed.
221    pub tests_passed: u32,
222    /// Number of tests that failed.
223    pub tests_failed: u32,
224    /// Number of steps that passed.
225    pub passed: u32,
226    /// Number of steps that failed.
227    pub failed: u32,
228    /// Number of steps that were skipped.
229    pub skipped: u32,
230    /// Per-step details.
231    pub details: Vec<StepResult>,
232}
233
234/// Result of a single step execution.
235#[derive(Debug)]
236pub struct StepResult {
237    /// The step name.
238    pub name: String,
239    /// Whether the step passed, failed, or was skipped.
240    pub status: StepStatus,
241    /// Human-readable result message.
242    pub message: String,
243}
244
245/// Outcome for a single step.
246#[derive(Debug, PartialEq, Eq)]
247pub enum StepStatus {
248    /// Step executed successfully and all assertions passed.
249    Passed,
250    /// Step execution or assertion failed.
251    Failed,
252    /// Step was skipped.
253    Skipped,
254}
255
256/// Predefined assertion preset definition.
257struct AssertPreset {
258    name: &'static str,
259    system: &'static str,
260    user_template: &'static str,
261}
262
263/// Built-in assertion presets.
264#[allow(clippy::literal_string_with_formatting_args)]
265const ASSERTION_PRESETS: &[AssertPreset] = &[
266    AssertPreset {
267        name: "no_error_on_page",
268        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
269        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
270    },
271    AssertPreset {
272        name: "text_visible",
273        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
274        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
275    },
276    AssertPreset {
277        name: "element_exists",
278        system: "You are a QA tester. Check if a described UI element exists on a web page.",
279        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
280    },
281    AssertPreset {
282        name: "visual_no_issues",
283        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
284        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
285    },
286    AssertPreset {
287        name: "visual_no_overlaps",
288        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
289        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
290    },
291    AssertPreset {
292        name: "visual_text_visible",
293        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
294        user_template: "Inspect the attached screenshot and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
295    },
296    AssertPreset {
297        name: "layout_no_issues",
298        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
299        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
300    },
301];
302
303impl ScenarioRunner {
304    /// Creates a new runner with the given scenario configuration and
305    /// assertion definitions.
306    #[must_use]
307    #[allow(clippy::needless_pass_by_value)]
308    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
309        let llm = LlmConfig {
310            url: scenario_config
311                .llm_url
312                .clone()
313                .unwrap_or_else(crate::llm_base_url),
314            model: scenario_config
315                .llm_model
316                .clone()
317                .unwrap_or_else(crate::llm_model),
318            api_key: scenario_config
319                .llm_api_key
320                .clone()
321                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
322            headers: if scenario_config.llm_headers.is_empty() {
323                crate::parse_headers_env()
324            } else {
325                scenario_config.llm_headers.clone()
326            },
327            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
328            temperature: scenario_config.temperature,
329            thinking: scenario_config.thinking,
330            model_params: scenario_config.model_params.clone(),
331            max_attempts: crate::default_llm_attempts(),
332        };
333        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
334        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
335        let defs_map: HashMap<String, AssertDefinition> = definitions
336            .into_iter()
337            .map(|d| (d.name.clone(), d))
338            .collect();
339
340        Self {
341            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
342            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
343            viewport_height: scenario_config.viewport_height.unwrap_or(720),
344            applied_viewport: std::cell::Cell::new((0, 0)),
345            config: scenario_config.clone(),
346            definitions: defs_map,
347            llm,
348            endpoints,
349            usage: Arc::new(UsageTracker::new()),
350            budgets,
351            artifacts_dir: PathBuf::from(
352                scenario_config
353                    .artifacts_dir
354                    .unwrap_or_else(|| "artifacts".to_owned()),
355            ),
356        }
357    }
358
359    /// Returns a clone of the [`UsageTracker`] for reporting.
360    #[must_use]
361    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
362        Arc::clone(&self.usage)
363    }
364
365    /// Returns a reference to the [`BudgetTracker`].
366    #[must_use]
367    pub const fn budget_tracker(&self) -> &BudgetTracker {
368        &self.budgets
369    }
370
371    /// Executes all test groups in the scenario and returns a report.
372    ///
373    /// # Errors
374    ///
375    /// Returns an error if the browser fails to launch.
376    #[allow(clippy::too_many_lines)]
377    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
378        let mut report = RunReport::default();
379
380        if tests.is_empty() {
381            eprintln!("No tests defined in scenario.");
382            return Ok(report);
383        }
384
385        let browser_headless = self.config.browser_headless.unwrap_or(true);
386
387        let launch_opts = LaunchOptions {
388            headless: browser_headless,
389            window_size: Some((self.viewport_width, self.viewport_height)),
390            sandbox: false,
391            // headless_chrome defaults this to 30s and shuts down the whole CDP
392            // connection when no messages arrive for that long. A scenario can
393            // easily exceed 30s of browser silence (slow LLM targeting/assertion
394            // calls, page waits, budget checks between steps), after which every
395            // remaining step fails with "Unable to make method calls because
396            // underlying connection is closed" — one quiet gap kills the run.
397            // Open-ended scenarios must own the connection for their full
398            // duration, so keep it alive for 6 hours.
399            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
400            ..LaunchOptions::default()
401        };
402
403        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
404        let tab = browser.new_tab().context("failed to open browser tab")?;
405        let _ = tab.set_default_timeout(self.timeout);
406
407        // Start MCP server if configured
408        #[cfg(feature = "mcp-server")]
409        if let Some(ref mcp_cfg) = self.config.mcp_server {
410            if mcp_cfg.enabled {
411                let port = mcp_cfg.port;
412                std::thread::spawn(move || {
413                    let _ = crate::mcp_server::start_mcp_server(port);
414                });
415            }
416        }
417        #[cfg(not(feature = "mcp-server"))]
418        if let Some(mcp_cfg) = &self.config.mcp_server {
419            if mcp_cfg.enabled {
420                eprintln!("  ⚠️  MCP server configured but 'mcp-server' feature not enabled");
421            }
422        }
423
424        // Start A2A agent server if configured
425        #[cfg(feature = "a2a-server")]
426        if let Some(ref a2a_cfg) = self.config.a2a_server {
427            if a2a_cfg.enabled {
428                let port = a2a_cfg.port;
429                tokio::spawn(crate::a2a_server::start_a2a_server(port));
430            }
431        }
432        #[cfg(not(feature = "a2a-server"))]
433        if let Some(a2a_cfg) = &self.config.a2a_server {
434            if a2a_cfg.enabled {
435                eprintln!("  ⚠️  A2A server configured but 'a2a-server' feature not enabled");
436            }
437        }
438
439        for test in tests {
440            eprintln!("\n╔══════════════════════════════");
441            eprintln!("║  Test: {}", test.name);
442            eprintln!("╚══════════════════════════════");
443
444            self.usage.reset_per_test();
445
446            let test_result = self.run_test(test, &tab);
447            self.usage.commit_test(&test.name);
448
449            if test_result.failed == 0 && test_result.total > 0 {
450                report.tests_passed += 1;
451                eprintln!("  Test ✅ Passed");
452            } else if test_result.total > 0 {
453                report.tests_failed += 1;
454                eprintln!("  Test ❌ Failed");
455            }
456
457            report.passed += test_result.passed;
458            report.failed += test_result.failed;
459            report.skipped += test_result.skipped;
460            report.details.extend(test_result.details);
461        }
462
463        Ok(report)
464    }
465
466    #[allow(clippy::too_many_lines)]
467    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
468        let base_url = test
469            .base_url
470            .clone()
471            .or_else(|| self.config.base_url.clone())
472            .unwrap_or_else(crate::base_url);
473
474        // Per-test viewport override: switch the browser via CDP
475        // device-metrics emulation before this test runs.
476        let vw = test.viewport_width.unwrap_or(self.viewport_width);
477        let vh = test.viewport_height.unwrap_or(self.viewport_height);
478        if self.applied_viewport.get() != (vw, vh) {
479            self.apply_viewport(tab, vw, vh);
480            self.applied_viewport.set((vw, vh));
481        }
482
483        // Per-test isolation: every test starts from its own start_url
484        // (unless auto_navigate is disabled), so a test never inherits the
485        // previous test's page state.
486        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
487
488        let start_url = test
489            .start_url
490            .clone()
491            .or_else(|| self.config.start_url.clone())
492            .unwrap_or_else(|| "/dashboard".to_owned());
493
494        if auto_navigate {
495            let full_url = resolve_url(&start_url, &base_url);
496            eprintln!("  → auto-navigate: {full_url}");
497            let _ = tab.navigate_to(&full_url);
498            let _ = tab.wait_until_navigated();
499            std::thread::sleep(Duration::from_secs(4));
500        }
501
502        let mut result = TestRunResult::default();
503
504        for (step_index, step) in test.steps.iter().enumerate() {
505            result.total += 1;
506
507            let wait_ms = match step {
508                TestStep::Navigate { wait_after_ms, .. }
509                | TestStep::Click { wait_after_ms, .. }
510                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
511                _ => None,
512            };
513
514            let mut step_result = match step {
515                TestStep::Navigate { url, .. } => {
516                    let full_url = resolve_url(url, &base_url);
517                    run_navigate_step(&full_url, tab)
518                }
519                TestStep::Click {
520                    target,
521                    selector,
522                    endpoint,
523                    idempotent,
524                    ..
525                } => self.run_click(
526                    target,
527                    selector.as_deref(),
528                    endpoint.as_deref(),
529                    test.endpoint.as_deref(),
530                    *idempotent,
531                    tab,
532                ),
533                TestStep::Type {
534                    target,
535                    text,
536                    selector,
537                    endpoint,
538                    idempotent,
539                    ..
540                } => self.run_type(
541                    target,
542                    text,
543                    selector.as_deref(),
544                    endpoint.as_deref(),
545                    test.endpoint.as_deref(),
546                    *idempotent,
547                    tab,
548                ),
549                TestStep::Wait {
550                    target,
551                    selector,
552                    text,
553                    timeout_ms,
554                    endpoint,
555                    idempotent,
556                } => self.run_wait(
557                    target,
558                    selector.as_deref(),
559                    text.as_deref(),
560                    *timeout_ms,
561                    endpoint.as_deref(),
562                    test.endpoint.as_deref(),
563                    *idempotent,
564                    tab,
565                ),
566                TestStep::Assert {
567                    definition,
568                    preset,
569                    prompt,
570                    assert_text,
571                    endpoint,
572                    screenshot,
573                } => self.run_assert(
574                    definition.as_deref(),
575                    preset.as_deref(),
576                    prompt.as_deref(),
577                    assert_text.as_deref(),
578                    *screenshot,
579                    endpoint.as_deref(),
580                    test.endpoint.as_deref(),
581                    tab,
582                ),
583                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
584                TestStep::Agent {
585                    agent,
586                    task,
587                    definition,
588                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
589                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
590            };
591
592            // Failure diagnostics: capture the page state and a screenshot so
593            // CI logs say WHAT the page looked like when the step failed,
594            // instead of a bare "timed out: The event waited for never came".
595            if step_result.status == StepStatus::Failed {
596                let state = diagnostics::capture(tab);
597                let screenshot = diagnostics::save_screenshot(
598                    tab,
599                    &self.artifacts_dir,
600                    &test.name,
601                    &test.name,
602                    step_index,
603                    step_kind_label(step),
604                );
605                step_result.message = format!(
606                    "{base} — {excerpt}",
607                    base = step_result.message,
608                    excerpt = diagnostics::inline_excerpt(&state),
609                );
610                eprintln!(
611                    "{}",
612                    diagnostics::full_context(&state, screenshot.as_deref())
613                );
614            }
615
616            eprintln!(
617                "    {} {} — {}",
618                if step_result.status == StepStatus::Passed {
619                    "✅"
620                } else if step_result.status == StepStatus::Failed {
621                    "❌"
622                } else {
623                    "⏭️"
624                },
625                step_result.name,
626                step_result.message,
627            );
628
629            match step_result.status {
630                StepStatus::Passed => result.passed += 1,
631                StepStatus::Failed => result.failed += 1,
632                StepStatus::Skipped => result.skipped += 1,
633            }
634
635            // Fail fast: the first failed step ends the test and the
636            // remaining steps are reported as skipped (no LLM budget is
637            // burned asserting against a page that is already known broken).
638            if step_result.status == StepStatus::Failed
639                && !self.config.continue_on_failure
640                && step_index + 1 < test.steps.len()
641            {
642                eprintln!(
643                    "      ⏭️  failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
644                    test.steps.len() - step_index - 1
645                );
646                for skipped in &test.steps[step_index + 1..] {
647                    result.total += 1;
648                    result.skipped += 1;
649                    eprintln!(
650                        "    ⏭️  {} — skipped: previous step failed",
651                        step_label(skipped)
652                    );
653                    result.details.push(StepResult {
654                        name: step_label(skipped),
655                        status: StepStatus::Skipped,
656                        message: "skipped: previous step failed".into(),
657                    });
658                }
659                result.details.push(step_result);
660                return result;
661            }
662
663            // Check per-test budget after each step
664            let test_usage = self.usage.current_test_snapshot();
665            let global_usage = self.usage.global_snapshot();
666            let budget_status = self.budgets.check_all(
667                &test.name,
668                &test_usage,
669                &global_usage,
670                test.budget.as_ref(),
671            );
672            match budget_status {
673                BudgetStatus::HardExceeded { message, .. } => {
674                    crate::reporting::print_budget_error(&message);
675                    result.details.push(StepResult {
676                        name: "[budget]".into(),
677                        status: StepStatus::Failed,
678                        message,
679                    });
680                    result.failed += 1;
681                    return result;
682                }
683                BudgetStatus::SoftExceeded { message, .. } => {
684                    crate::reporting::print_budget_warning(&message);
685                }
686                BudgetStatus::Ok => {}
687            }
688
689            if let Some(ms) = wait_ms {
690                std::thread::sleep(Duration::from_millis(ms));
691            }
692
693            result.details.push(step_result);
694        }
695
696        result
697    }
698
699    /// Applies a viewport size to the current tab via CDP
700    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
701    /// overrides and the viewport matrix. The initial window size set at
702    /// browser launch is replaced by emulation; failures are logged but
703    /// do not fail the test (a mismatched viewport only weakens coverage).
704    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
705        use headless_chrome::protocol::cdp::Emulation;
706        let _ = self;
707        let params = Emulation::SetDeviceMetricsOverride {
708            width,
709            height,
710            device_scale_factor: 1.0,
711            mobile: false,
712            scale: None,
713            screen_width: Some(width),
714            screen_height: Some(height),
715            position_x: None,
716            position_y: None,
717            dont_set_visible_size: None,
718            screen_orientation: None,
719            viewport: None,
720            display_feature: None,
721            device_posture: None,
722        };
723        eprintln!("      ↻ viewport: {width}x{height}");
724        if let Err(e) = tab.call_method(params) {
725            eprintln!("      ⚠️  viewport switch to {width}x{height} failed: {e}");
726        }
727    }
728
729    // ── step handlers ───────────────────────────────────────────────────
730
731    #[allow(clippy::too_many_lines)]
732    fn run_click(
733        &self,
734        target: &str,
735        selector_override: Option<&str>,
736        step_endpoint: Option<&str>,
737        test_endpoint: Option<&str>,
738        idempotent: bool,
739        tab: &Tab,
740    ) -> StepResult {
741        let name = format!("[click] {target}");
742        let selector = match self.resolve_selector(
743            selector_override,
744            target,
745            step_endpoint,
746            test_endpoint,
747            tab,
748        ) {
749            Ok(s) => s,
750            Err(msg) => {
751                if idempotent {
752                    return StepResult {
753                        name,
754                        status: StepStatus::Skipped,
755                        message: format!("skipped (idempotent): no target found — {msg}"),
756                    };
757                }
758                return StepResult {
759                    name,
760                    status: StepStatus::Failed,
761                    message: msg,
762                };
763            }
764        };
765
766        // Idempotent steps probe briefly: a missing target means the
767        // action was already done / not applicable (e.g. an
768        // already-authenticated session), and skipping is the success
769        // path, not a failure.
770        let probe_secs = if idempotent { 5 } else { 10 };
771        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
772            Ok(element) => match element.click() {
773                Ok(_) => StepResult {
774                    name,
775                    status: StepStatus::Passed,
776                    message: format!("clicked {selector}"),
777                },
778                Err(e) => StepResult {
779                    name,
780                    status: StepStatus::Failed,
781                    message: format!("click failed on {selector}: {e}"),
782                },
783            },
784            Err(e) if idempotent => StepResult {
785                name,
786                status: StepStatus::Skipped,
787                message: format!("skipped (idempotent): element {selector} not present — {e}"),
788            },
789            Err(e) => StepResult {
790                name,
791                status: StepStatus::Failed,
792                message: format!("element {selector} not found: {e}"),
793            },
794        }
795    }
796
797    #[allow(clippy::too_many_arguments)]
798    fn run_type(
799        &self,
800        target: &str,
801        text: &str,
802        selector_override: Option<&str>,
803        step_endpoint: Option<&str>,
804        test_endpoint: Option<&str>,
805        idempotent: bool,
806        tab: &Tab,
807    ) -> StepResult {
808        let name = format!("[type] {target}");
809        let selector = match self.resolve_selector(
810            selector_override,
811            target,
812            step_endpoint,
813            test_endpoint,
814            tab,
815        ) {
816            Ok(s) => s,
817            Err(msg) => {
818                if idempotent {
819                    return StepResult {
820                        name,
821                        status: StepStatus::Skipped,
822                        message: format!("skipped (idempotent): no target found — {msg}"),
823                    };
824                }
825                return StepResult {
826                    name,
827                    status: StepStatus::Failed,
828                    message: msg,
829                };
830            }
831        };
832
833        let probe_secs = if idempotent { 5 } else { 10 };
834        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
835            Ok(element) => {
836                if let Err(e) = element.click() {
837                    return StepResult {
838                        name,
839                        status: StepStatus::Failed,
840                        message: format!("click to focus {selector} failed: {e}"),
841                    };
842                }
843
844                let js = format!(
845                    "document.querySelector('{}').value = '';",
846                    selector.replace('\'', "\\'")
847                );
848                let _ = tab.evaluate(&js, false);
849
850                match element.type_into(text) {
851                    Ok(_) => StepResult {
852                        name,
853                        status: StepStatus::Passed,
854                        message: format!("typed {text:?} into {selector}"),
855                    },
856                    Err(e) => StepResult {
857                        name,
858                        status: StepStatus::Failed,
859                        message: format!("type into {selector} failed: {e}"),
860                    },
861                }
862            }
863            Err(e) if idempotent => StepResult {
864                name,
865                status: StepStatus::Skipped,
866                message: format!("skipped (idempotent): element {selector} not present — {e}"),
867            },
868            Err(e) => StepResult {
869                name,
870                status: StepStatus::Failed,
871                message: format!("element {selector} not found: {e}"),
872            },
873        }
874    }
875
876    #[allow(clippy::too_many_arguments)]
877    #[allow(clippy::too_many_lines)]
878    fn run_wait(
879        &self,
880        target: &str,
881        selector_override: Option<&str>,
882        text: Option<&str>,
883        timeout_ms: Option<u64>,
884        step_endpoint: Option<&str>,
885        test_endpoint: Option<&str>,
886        idempotent: bool,
887        tab: &Tab,
888    ) -> StepResult {
889        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
890        let step_name = format!("[wait] {target}");
891
892        // Resolve an explicit selector only (text-only waits are LLM-free).
893        let selector = match selector_override {
894            Some(s) => Some(s.to_owned()),
895            None if text.is_some() => None,
896            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
897                Ok(s) => Some(s),
898                Err(msg) => {
899                    if idempotent {
900                        return StepResult {
901                            name: step_name,
902                            status: StepStatus::Skipped,
903                            message: format!("skipped (idempotent): no target found — {msg}"),
904                        };
905                    }
906                    return StepResult {
907                        name: step_name,
908                        status: StepStatus::Failed,
909                        message: msg,
910                    };
911                }
912            },
913        };
914
915        if text.is_some() {
916            let sel_js = selector
917                .as_deref()
918                .map(crate::selectors::selector_matches_js);
919            let text_js = text.map(|t| {
920                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
921                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
922            });
923
924            let deadline = Instant::now() + timeout;
925            loop {
926                let sel_ok = sel_js
927                    .as_ref()
928                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
929                let text_ok = text_js
930                    .as_ref()
931                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
932                if sel_ok && text_ok {
933                    let mut what = Vec::new();
934                    if let Some(sel) = &selector {
935                        what.push(format!("found {sel}"));
936                    }
937                    if let Some(t) = text {
938                        what.push(format!("text {t:?} visible"));
939                    }
940                    return StepResult {
941                        name: step_name,
942                        status: StepStatus::Passed,
943                        message: what.join(" and "),
944                    };
945                }
946                if Instant::now() >= deadline {
947                    let mut what = Vec::new();
948                    if let Some(sel) = &selector {
949                        what.push(sel.clone());
950                    }
951                    if let Some(t) = text {
952                        what.push(format!("text {t:?}"));
953                    }
954                    let message = format!(
955                        "wait for {} timed out after {}ms: the event waited for never came",
956                        what.join(" / "),
957                        timeout.as_millis(),
958                    );
959                    if idempotent {
960                        return StepResult {
961                            name: step_name,
962                            status: StepStatus::Skipped,
963                            message: format!("skipped (idempotent): {message}"),
964                        };
965                    }
966                    return StepResult {
967                        name: step_name,
968                        status: StepStatus::Failed,
969                        message,
970                    };
971                }
972                std::thread::sleep(Duration::from_millis(250));
973            }
974        }
975
976        match selector.as_deref() {
977            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
978                Ok(_) => StepResult {
979                    name: step_name,
980                    status: StepStatus::Passed,
981                    message: format!("found {sel}"),
982                },
983                Err(e) if idempotent => StepResult {
984                    name: step_name,
985                    status: StepStatus::Skipped,
986                    message: format!(
987                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
988                        timeout.as_millis()
989                    ),
990                },
991                Err(e) => StepResult {
992                    name: step_name,
993                    status: StepStatus::Failed,
994                    message: format!(
995                        "wait for {sel} timed out after {}ms: {e}",
996                        timeout.as_millis()
997                    ),
998                },
999            },
1000            None => StepResult {
1001                name: step_name,
1002                status: StepStatus::Failed,
1003                message: "wait step has neither selector nor text".into(),
1004            },
1005        }
1006    }
1007
1008    #[allow(clippy::too_many_arguments)]
1009    fn run_assert(
1010        &self,
1011        definition: Option<&str>,
1012        preset: Option<&str>,
1013        prompt: Option<&str>,
1014        assert_text: Option<&str>,
1015        screenshot: bool,
1016        step_endpoint: Option<&str>,
1017        test_endpoint: Option<&str>,
1018        tab: &Tab,
1019    ) -> StepResult {
1020        std::thread::sleep(Duration::from_millis(500));
1021
1022        let page_content = get_page_text(tab);
1023
1024        // Vision attach: capture the viewport once per assert step and hand
1025        // the JPEG data URL to the preset/prompt evaluation below.
1026        let image = if screenshot {
1027            let endpoint = self
1028                .endpoints
1029                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1030            if !endpoint.vision {
1031                return StepResult {
1032                    name: "[assert]".into(),
1033                    status: StepStatus::Failed,
1034                    message: format!(
1035                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1036                        name = endpoint.name
1037                    ),
1038                };
1039            }
1040            match crate::vision::capture_screenshot_data_url(
1041                tab,
1042                self.config
1043                    .screenshot_max_dimension
1044                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1045            ) {
1046                Ok(data_url) => Some(data_url),
1047                Err(e) => {
1048                    return StepResult {
1049                        name: "[assert]".into(),
1050                        status: StepStatus::Failed,
1051                        message: format!("screenshot capture failed: {e}"),
1052                    };
1053                }
1054            }
1055        } else {
1056            None
1057        };
1058
1059        if let Some(def_name) = definition {
1060            if let Some(def) = self.definitions.get(def_name) {
1061                return self.run_assert_def(
1062                    def,
1063                    &page_content,
1064                    image.as_deref(),
1065                    step_endpoint,
1066                    test_endpoint,
1067                    tab,
1068                );
1069            }
1070            return StepResult {
1071                name: format!("[assert] {def_name}"),
1072                status: StepStatus::Failed,
1073                message: format!("definition '{def_name}' not found"),
1074            };
1075        }
1076
1077        if let Some(preset_name) = preset {
1078            // Deterministic DOM layout scan — runs JS in the browser and
1079            // never calls the LLM (free, fast, no pixel budget).
1080            if preset_name == "layout_no_issues" {
1081                return self.run_layout_preset(tab);
1082            }
1083            return self.run_preset(
1084                preset_name,
1085                assert_text,
1086                &page_content,
1087                image.as_deref(),
1088                step_endpoint,
1089                test_endpoint,
1090            );
1091        }
1092
1093        if let Some(prompt_text) = prompt {
1094            return self.run_custom(
1095                prompt_text,
1096                &page_content,
1097                image.as_deref(),
1098                step_endpoint,
1099                test_endpoint,
1100            );
1101        }
1102
1103        StepResult {
1104            name: "[assert]".into(),
1105            status: StepStatus::Skipped,
1106            message: "no definition, preset, or prompt specified".into(),
1107        }
1108    }
1109
1110    fn run_assert_def(
1111        &self,
1112        def: &AssertDefinition,
1113        page_content: &PageContent,
1114        image: Option<&str>,
1115        step_endpoint: Option<&str>,
1116        test_endpoint: Option<&str>,
1117        tab: &Tab,
1118    ) -> StepResult {
1119        // Agent-based definition: delegate to an A2A agent
1120        if let Some(ref agent) = def.agent {
1121            if image.is_some() {
1122                return StepResult {
1123                    name: format!("[assert] {}", def.name),
1124                    status: StepStatus::Failed,
1125                    message: "agent-backed assertions do not support screenshots".into(),
1126                };
1127            }
1128            let task = def
1129                .task_template
1130                .as_deref()
1131                .unwrap_or("Evaluate the assertion")
1132                .replace("{url}", &page_content.url)
1133                .replace("{title}", &page_content.title)
1134                .replace("{content}", &page_content.body_text)
1135                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1136
1137            return self.run_agent_step(agent, &task, &def.name);
1138        }
1139
1140        // Custom preset: system + user_template provided in the definition
1141        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1142            return self.run_custom_preset(
1143                &def.name,
1144                system,
1145                template,
1146                def.assert_text.as_deref(),
1147                page_content,
1148                image,
1149                step_endpoint,
1150                test_endpoint,
1151            );
1152        }
1153
1154        def.preset.as_ref().map_or_else(
1155            || {
1156                def.prompt.as_ref().map_or_else(
1157                    || StepResult {
1158                        name: format!("[assert] {}", def.name),
1159                        status: StepStatus::Failed,
1160                        message: "definition has no preset, prompt, or system+user_template".into(),
1161                    },
1162                    |prompt| {
1163                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1164                    },
1165                )
1166            },
1167            |preset_name| {
1168                if preset_name == "layout_no_issues" {
1169                    return self.run_layout_preset(tab);
1170                }
1171                self.run_preset(
1172                    preset_name,
1173                    def.assert_text.as_deref(),
1174                    page_content,
1175                    image,
1176                    step_endpoint,
1177                    test_endpoint,
1178                )
1179            },
1180        )
1181    }
1182
1183    #[allow(clippy::too_many_arguments)]
1184    fn run_custom_preset(
1185        &self,
1186        name: &str,
1187        system: &str,
1188        template: &str,
1189        assert_text: Option<&str>,
1190        page_content: &PageContent,
1191        image: Option<&str>,
1192        step_endpoint: Option<&str>,
1193        test_endpoint: Option<&str>,
1194    ) -> StepResult {
1195        let user_prompt = template
1196            .replace("{url}", &page_content.url)
1197            .replace("{title}", &page_content.title)
1198            .replace("{content}", &page_content.body_text)
1199            .replace("{expected_text}", assert_text.unwrap_or(""))
1200            .replace("{description}", "");
1201
1202        // Custom preset definitions frequently forget the {content}
1203        // placeholder — without it the LLM has no page to evaluate and
1204        // answers "I can't determine that without seeing the page". Always
1205        // append the page context unless the template already references it.
1206        let user_prompt = if template.contains("{content}") {
1207            user_prompt
1208        } else {
1209            format!(
1210                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1211                url = page_content.url,
1212                title = page_content.title,
1213                content = page_content.body_text,
1214            )
1215        };
1216
1217        eprintln!("      assert: {name} (custom preset)");
1218
1219        let chain = self
1220            .endpoints
1221            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1222        let sys = system.to_owned();
1223
1224        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image);
1225
1226        response.map_or_else(
1227            |e| StepResult {
1228                name: format!("[assert] {name}"),
1229                status: StepStatus::Failed,
1230                message: format!("LLM assertion call failed: {e}"),
1231            },
1232            |(lr, idx)| {
1233                self.usage.record_llm_call(
1234                    &chain[idx].name,
1235                    chain[idx],
1236                    lr.usage.prompt_tokens,
1237                    lr.usage.completion_tokens,
1238                );
1239                let content_lower = lr.content.to_lowercase().trim().to_owned();
1240                if content_lower.starts_with("pass") {
1241                    StepResult {
1242                        name: format!("[assert] {name}"),
1243                        status: StepStatus::Passed,
1244                        message: "PASS".into(),
1245                    }
1246                } else {
1247                    StepResult {
1248                        name: format!("[assert] {name}"),
1249                        status: StepStatus::Failed,
1250                        message: lr.content,
1251                    }
1252                }
1253            },
1254        )
1255    }
1256
1257    fn run_preset(
1258        &self,
1259        preset_name: &str,
1260        assert_text: Option<&str>,
1261        page_content: &PageContent,
1262        image: Option<&str>,
1263        step_endpoint: Option<&str>,
1264        test_endpoint: Option<&str>,
1265    ) -> StepResult {
1266        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1267            return StepResult {
1268                name: format!("[assert] {preset_name}"),
1269                status: StepStatus::Failed,
1270                message: format!("unknown assertion preset: {preset_name}"),
1271            };
1272        };
1273        if preset_name.starts_with("visual_") && image.is_none() {
1274            return StepResult {
1275                name: format!("[assert] {preset_name}"),
1276                status: StepStatus::Failed,
1277                message: format!(
1278                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1279                ),
1280            };
1281        }
1282
1283        let user_prompt = preset
1284            .user_template
1285            .replace("{url}", &page_content.url)
1286            .replace("{title}", &page_content.title)
1287            .replace("{content}", &page_content.body_text)
1288            .replace("{expected_text}", assert_text.unwrap_or(""))
1289            .replace("{description}", "");
1290
1291        // Same safety net as custom presets: never let the LLM answer with
1292        // no page context at all.
1293        let user_prompt = if preset.user_template.contains("{content}") {
1294            user_prompt
1295        } else {
1296            format!(
1297                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1298                url = page_content.url,
1299                title = page_content.title,
1300                content = page_content.body_text,
1301            )
1302        };
1303
1304        eprintln!("      assert: {preset_name}");
1305
1306        let chain = self
1307            .endpoints
1308            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1309        let sys = preset.system.to_owned();
1310
1311        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image);
1312
1313        response.map_or_else(
1314            |e| StepResult {
1315                name: format!("[assert] {preset_name}"),
1316                status: StepStatus::Failed,
1317                message: format!("LLM assertion call failed: {e}"),
1318            },
1319            |(lr, idx)| {
1320                self.usage.record_llm_call(
1321                    &chain[idx].name,
1322                    chain[idx],
1323                    lr.usage.prompt_tokens,
1324                    lr.usage.completion_tokens,
1325                );
1326                let content_lower = lr.content.to_lowercase().trim().to_owned();
1327                if content_lower.starts_with("pass") {
1328                    StepResult {
1329                        name: format!("[assert] {preset_name}"),
1330                        status: StepStatus::Passed,
1331                        message: "PASS".into(),
1332                    }
1333                } else {
1334                    StepResult {
1335                        name: format!("[assert] {preset_name}"),
1336                        status: StepStatus::Failed,
1337                        message: lr.content,
1338                    }
1339                }
1340            },
1341        )
1342    }
1343
1344    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1345    ///
1346    /// Evaluates the layout-scan JS in the page and fails with the list of
1347    /// detected issues: horizontal page overflow, visible elements sticking
1348    /// out of the viewport, text clipped by `overflow: hidden` containers,
1349    /// and interactive elements covered by other elements. No LLM call —
1350    /// checks are geometry-based so the check is free, deterministic, and
1351    /// safe to run on every page × viewport variant.
1352    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1353        let name = "[assert] layout_no_issues".to_owned();
1354        eprintln!("      assert: layout_no_issues (DOM layout scan)");
1355        let js = LAYOUT_SCAN_JS.replace(
1356            "__IGNORE_CLASSES__",
1357            &serde_json::to_string(&self.config.layout_ignore_classes)
1358                .unwrap_or_else(|_| "[]".to_owned()),
1359        );
1360        let result = tab.evaluate(&js, false);
1361        let json_str = match result {
1362            Ok(r) => r
1363                .value
1364                .as_ref()
1365                .and_then(|v| v.as_str().map(String::from))
1366                .unwrap_or_else(|| "[]".to_owned()),
1367            Err(e) => {
1368                return StepResult {
1369                    name,
1370                    status: StepStatus::Failed,
1371                    message: format!("layout scan JS failed: {e}"),
1372                };
1373            }
1374        };
1375        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1376        if issues.is_empty() {
1377            return StepResult {
1378                name,
1379                status: StepStatus::Passed,
1380                message: "PASS — no layout defects detected".into(),
1381            };
1382        }
1383        let mut lines: Vec<String> = issues
1384            .iter()
1385            .take(10)
1386            .map(|i| {
1387                format!(
1388                    "- [{type_}] {element}: {detail}",
1389                    type_ = i.issue_type,
1390                    element = i.element,
1391                    detail = i.detail
1392                )
1393            })
1394            .collect();
1395        if issues.len() > 10 {
1396            lines.push(format!("- … and {} more", issues.len() - 10));
1397        }
1398        StepResult {
1399            name,
1400            status: StepStatus::Failed,
1401            message: format!(
1402                "FAIL — {} layout defect(s) detected:\n{}",
1403                issues.len(),
1404                lines.join("\n")
1405            ),
1406        }
1407    }
1408
1409    fn run_custom(
1410        &self,
1411        prompt: &str,
1412        page_content: &PageContent,
1413        image: Option<&str>,
1414        step_endpoint: Option<&str>,
1415        test_endpoint: Option<&str>,
1416    ) -> StepResult {
1417        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1418
1419        let mut user = format!(
1420            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1421            url = page_content.url,
1422            title = page_content.title,
1423            content = page_content.body_text,
1424        );
1425        if image.is_some() {
1426            user.push_str(
1427                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1428            );
1429        }
1430
1431        eprintln!("      custom assert");
1432
1433        let chain = self
1434            .endpoints
1435            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1436        let sys = system.to_owned();
1437
1438        let response = self.llm_call_chain(&chain, &sys, &user, image);
1439
1440        response.map_or_else(
1441            |e| StepResult {
1442                name: "[assert] custom".into(),
1443                status: StepStatus::Failed,
1444                message: format!("LLM assertion call failed: {e}"),
1445            },
1446            |(lr, idx)| {
1447                self.usage.record_llm_call(
1448                    &chain[idx].name,
1449                    chain[idx],
1450                    lr.usage.prompt_tokens,
1451                    lr.usage.completion_tokens,
1452                );
1453                let content_lower = lr.content.to_lowercase().trim().to_owned();
1454                if content_lower.starts_with("pass") {
1455                    StepResult {
1456                        name: "[assert] custom".into(),
1457                        status: StepStatus::Passed,
1458                        message: "PASS".into(),
1459                    }
1460                } else {
1461                    StepResult {
1462                        name: "[assert] custom".into(),
1463                        status: StepStatus::Failed,
1464                        message: lr.content,
1465                    }
1466                }
1467            },
1468        )
1469    }
1470
1471    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1472        let path = path.unwrap_or("screenshot.png");
1473
1474        match tab.capture_screenshot(
1475            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1476            None,
1477            None,
1478            true,
1479        ) {
1480            Ok(data) => {
1481                if let Err(e) = std::fs::write(path, &data) {
1482                    return StepResult {
1483                        name: format!("[screenshot] {path}"),
1484                        status: StepStatus::Failed,
1485                        message: format!("failed to write screenshot: {e}"),
1486                    };
1487                }
1488                StepResult {
1489                    name: format!("[screenshot] {path}"),
1490                    status: StepStatus::Passed,
1491                    message: format!("saved to {path}"),
1492                }
1493            }
1494            Err(e) => StepResult {
1495                name: format!("[screenshot] {path}"),
1496                status: StepStatus::Failed,
1497                message: format!("screenshot failed: {e}"),
1498            },
1499        }
1500    }
1501
1502    /// Runs an A2A agent step.
1503    #[allow(clippy::literal_string_with_formatting_args)]
1504    fn run_agent(
1505        &self,
1506        agent_name: &str,
1507        task: &str,
1508        definition: Option<&str>,
1509        _test_endpoint: Option<&str>,
1510    ) -> StepResult {
1511        // If a definition is specified, look up the task template
1512        let resolved_task = if let Some(def_name) = definition {
1513            if let Some(def) = self.definitions.get(def_name) {
1514                let tmpl = def.task_template.as_deref().unwrap_or(task);
1515                tmpl.replace("{task}", task)
1516            } else {
1517                return StepResult {
1518                    name: format!("[agent] {def_name}"),
1519                    status: StepStatus::Failed,
1520                    message: format!("definition '{def_name}' not found"),
1521                };
1522            }
1523        } else {
1524            task.to_owned()
1525        };
1526
1527        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1528    }
1529
1530    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1531        let Some(ep) = self.endpoints.get(agent_name) else {
1532            return StepResult {
1533                name: format!("[agent] {display_name}"),
1534                status: StepStatus::Failed,
1535                message: format!("agent endpoint '{agent_name}' not found"),
1536            };
1537        };
1538
1539        if ep.url.is_empty() {
1540            return StepResult {
1541                name: format!("[agent] {display_name}"),
1542                status: StepStatus::Failed,
1543                message: format!("agent endpoint '{agent_name}' has no URL"),
1544            };
1545        }
1546
1547        eprintln!("      → agent {agent_name}: {task}");
1548
1549        let url = ep.url.clone();
1550        let client = A2aClient::new(&url, self.timeout);
1551        let task_clone = task.to_owned();
1552
1553        let response = std::thread::spawn(move || {
1554            let rt = tokio::runtime::Builder::new_current_thread()
1555                .enable_all()
1556                .build()
1557                .unwrap();
1558            rt.block_on(client.send_task(&task_clone))
1559        })
1560        .join()
1561        .unwrap();
1562
1563        // Record the flat-cost call
1564        self.usage.record_flat_call(agent_name, ep);
1565
1566        match response {
1567            Ok(text) => {
1568                let clean = text.trim().to_owned();
1569                let lower = clean.to_lowercase();
1570                if lower.starts_with("pass") {
1571                    StepResult {
1572                        name: format!("[agent] {display_name}"),
1573                        status: StepStatus::Passed,
1574                        message: format!("PASS: {clean}"),
1575                    }
1576                } else if lower.starts_with("fail") {
1577                    StepResult {
1578                        name: format!("[agent] {display_name}"),
1579                        status: StepStatus::Failed,
1580                        message: clean,
1581                    }
1582                } else {
1583                    StepResult {
1584                        name: format!("[agent] {display_name}"),
1585                        status: StepStatus::Passed,
1586                        message: format!("response: {clean}"),
1587                    }
1588                }
1589            }
1590            Err(e) => StepResult {
1591                name: format!("[agent] {display_name}"),
1592                status: StepStatus::Failed,
1593                message: format!("agent call failed: {e}"),
1594            },
1595        }
1596    }
1597
1598    /// Runs an MCP tool call step.
1599    fn run_mcp(
1600        &self,
1601        server_name: &str,
1602        tool_name: &str,
1603        args: Option<&serde_json::Value>,
1604    ) -> StepResult {
1605        let Some(ep) = self.endpoints.get(server_name) else {
1606            return StepResult {
1607                name: format!("[mcp] {server_name}:{tool_name}"),
1608                status: StepStatus::Failed,
1609                message: format!("MCP server endpoint '{server_name}' not found"),
1610            };
1611        };
1612
1613        let cmd = ep.command.as_deref().unwrap_or("");
1614        if cmd.is_empty() {
1615            return StepResult {
1616                name: format!("[mcp] {server_name}:{tool_name}"),
1617                status: StepStatus::Failed,
1618                message: format!("MCP server '{server_name}' has no command configured"),
1619            };
1620        }
1621
1622        eprintln!("      → mcp {server_name} {tool_name}");
1623
1624        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1625
1626        let command = cmd.to_owned();
1627        let args_vec = ep.args.clone();
1628        let tool = tool_name.to_owned();
1629
1630        let response = std::thread::spawn(move || {
1631            let mut mcp_client =
1632                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1633            mcp_client
1634                .call_tool(&tool, &args_val)
1635                .map_err(|e| e.to_string())
1636        })
1637        .join()
1638        .unwrap();
1639
1640        // Record the flat-cost call
1641        self.usage.record_flat_call(server_name, ep);
1642
1643        match response {
1644            Ok(result) => {
1645                if result.isError {
1646                    StepResult {
1647                        name: format!("[mcp] {server_name}:{tool_name}"),
1648                        status: StepStatus::Failed,
1649                        message: result.to_string(),
1650                    }
1651                } else {
1652                    StepResult {
1653                        name: format!("[mcp] {server_name}:{tool_name}"),
1654                        status: StepStatus::Passed,
1655                        message: result.to_string(),
1656                    }
1657                }
1658            }
1659            Err(e) => StepResult {
1660                name: format!("[mcp] {server_name}:{tool_name}"),
1661                status: StepStatus::Failed,
1662                message: format!("MCP call failed: {e}"),
1663            },
1664        }
1665    }
1666
1667    // ── helpers ──────────────────────────────────────────────────────────
1668
1669    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1670    /// the runner's default LLM config for any unset fields.
1671    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1672        LlmConfig {
1673            url: if endpoint.url.is_empty() {
1674                self.llm.url.clone()
1675            } else {
1676                endpoint.url.clone()
1677            },
1678            model: endpoint
1679                .model
1680                .clone()
1681                .unwrap_or_else(|| self.llm.model.clone()),
1682            api_key: endpoint
1683                .api_key
1684                .clone()
1685                .or_else(|| self.llm.api_key.clone()),
1686            headers: if endpoint.headers.is_empty() {
1687                self.llm.headers.clone()
1688            } else {
1689                endpoint.headers.clone()
1690            },
1691            timeout: self.llm.timeout,
1692            temperature: self.llm.temperature,
1693            thinking: self.llm.thinking,
1694            model_params: self.llm.model_params.clone(),
1695            max_attempts: endpoint.max_attempts.max(1),
1696        }
1697    }
1698
1699    /// Runs a single LLM call against an ordered endpoint chain (primary +
1700    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
1701    /// the first endpoint that answers wins. Returns the response together
1702    /// with the chain index of the answering endpoint (0 = primary) so the
1703    /// caller can attribute usage to the correct endpoint.
1704    fn llm_call_chain(
1705        &self,
1706        chain: &[&ResolvedEndpoint],
1707        system: &str,
1708        user: &str,
1709        image: Option<&str>,
1710    ) -> Result<(crate::costs::LlmResponse, usize), String> {
1711        if chain.is_empty() {
1712            return Err("empty LLM endpoint chain".into());
1713        }
1714        let primary = self.build_llm_for_endpoint(chain[0]);
1715        let fallbacks: Vec<LlmConfig> = chain[1..]
1716            .iter()
1717            .map(|e| self.build_llm_for_endpoint(e))
1718            .collect();
1719        let sys = system.to_owned();
1720        let user = user.to_owned();
1721        let image = image.map(str::to_owned);
1722
1723        std::thread::spawn(move || {
1724            let rt = tokio::runtime::Builder::new_current_thread()
1725                .enable_all()
1726                .build()
1727                .unwrap();
1728            let call = async {
1729                match image.as_deref() {
1730                    Some(img) => {
1731                        llm_chat_vision_with_usage_chain(&primary, &fallbacks, &sys, &user, img)
1732                            .await
1733                    }
1734                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
1735                }
1736            };
1737            rt.block_on(call)
1738        })
1739        .join()
1740        .unwrap()
1741    }
1742
1743    /// Resolves a CSS selector for the target element. Uses the explicit
1744    /// `selector` if provided, otherwise asks the LLM to find the element
1745    /// from the natural language `target` description and page DOM.
1746    ///
1747    /// LLM responses are sanitized and verified against the live page: a
1748    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
1749    /// immediately with the raw LLM output, and a selector that matches
1750    /// nothing triggers one retry with feedback before failing.
1751    #[allow(clippy::too_many_lines)]
1752    fn resolve_selector(
1753        &self,
1754        css_override: Option<&str>,
1755        target: &str,
1756        step_endpoint: Option<&str>,
1757        test_endpoint: Option<&str>,
1758        tab: &Tab,
1759    ) -> Result<String, String> {
1760        if let Some(explicit) = css_override {
1761            return Ok(explicit.to_owned());
1762        }
1763
1764        let dom_info = extract_dom_info(tab)?;
1765        let page_content = get_page_text(tab);
1766
1767        let system = concat!(
1768            "You are a browser automation selector generator. ",
1769            "Given a web page's content and interactive elements, ",
1770            "return ONLY the best CSS selector for the described element. ",
1771            "Output nothing except the CSS selector. ",
1772            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
1773            "[name=\"...\"], tag.class, tag. ",
1774            "Never output explanations, markdown, or extra text."
1775        );
1776
1777        let user = format!(
1778            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
1779            page_content.url,
1780            page_content.title,
1781            truncate(&page_content.body_text, 4000),
1782            dom_info,
1783            target,
1784        );
1785
1786        let retry_user = format!(
1787            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
1788            "The selector must match at least one element currently present on the page.",
1789            page_content.url,
1790            page_content.title,
1791            truncate(&page_content.body_text, 4000),
1792            dom_info,
1793            target,
1794        );
1795
1796        eprintln!("      LLM targeting: {target}");
1797
1798        let chain = self
1799            .endpoints
1800            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
1801        let sys = system.to_owned();
1802
1803        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None);
1804
1805        let first = call_llm(&user);
1806        let (lr, idx) = match first {
1807            Ok(lr) => lr,
1808            Err(e) => {
1809                return Err(format!("LLM element targeting failed: {e}"));
1810            }
1811        };
1812        self.usage.record_llm_call(
1813            &chain[idx].name,
1814            chain[idx],
1815            lr.usage.prompt_tokens,
1816            lr.usage.completion_tokens,
1817        );
1818        let clean = sanitize_selector(&lr.content);
1819        eprintln!("      resolved selector: {clean}");
1820
1821        if selector_is_useless(&clean) {
1822            return Err(format!(
1823                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
1824                raw = lr.content.trim(),
1825            ));
1826        }
1827        if let Err(reason) = validate_selector(&clean) {
1828            return Err(format!(
1829                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
1830                raw = lr.content.trim(),
1831            ));
1832        }
1833        if !selector_matches(tab, &clean).unwrap_or(false) {
1834            // One retry with feedback: flaky models occasionally invent a
1835            // selector that does not exist on the page.
1836            eprintln!(
1837                "      selector {clean} matches nothing — retrying LLM targeting with feedback"
1838            );
1839            let second = call_llm(&retry_user);
1840            let (lr2, idx2) = match second {
1841                Ok(lr2) => lr2,
1842                Err(e) => {
1843                    return Err(format!(
1844                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
1845                    ));
1846                }
1847            };
1848            self.usage.record_llm_call(
1849                &chain[idx2].name,
1850                chain[idx2],
1851                lr2.usage.prompt_tokens,
1852                lr2.usage.completion_tokens,
1853            );
1854            let clean2 = sanitize_selector(&lr2.content);
1855            eprintln!("      resolved selector (retry): {clean2}");
1856            if selector_is_useless(&clean2) {
1857                return Err(format!(
1858                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
1859                    raw = lr2.content.trim(),
1860                    excerpt = truncate(&page_content.body_text, 300),
1861                ));
1862            }
1863            if !selector_matches(tab, &clean2).unwrap_or(false) {
1864                return Err(format!(
1865                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
1866                ));
1867            }
1868            return Ok(clean2);
1869        }
1870
1871        Ok(clean)
1872    }
1873}
1874
1875/// Evaluates a JS expression that is expected to return a boolean.
1876fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
1877    tab.evaluate(js, false)
1878        .map_err(|e| format!("evaluate failed: {e}"))?
1879        .value
1880        .and_then(|v| v.as_bool())
1881        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
1882}
1883
1884/// Checks whether a CSS selector matches at least one current element.
1885fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
1886    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
1887}
1888
1889// ── Free helper functions ──────────────────────────────────────────────
1890
1891fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
1892    let name = format!("[navigate] {full_url}");
1893    match tab.navigate_to(full_url) {
1894        Ok(_) => {
1895            let _ = tab.wait_until_navigated();
1896            StepResult {
1897                name,
1898                status: StepStatus::Passed,
1899                message: format!("navigated to {full_url}"),
1900            }
1901        }
1902        Err(e) => StepResult {
1903            name,
1904            status: StepStatus::Failed,
1905            message: format!("navigation failed: {e}"),
1906        },
1907    }
1908}
1909
1910fn extract_dom_info(tab: &Tab) -> Result<String, String> {
1911    let result = tab
1912        .evaluate(DOM_EXTRACT_JS, false)
1913        .map_err(|e| format!("DOM extraction failed: {e}"))?;
1914
1915    let json_str = result
1916        .value
1917        .as_ref()
1918        .and_then(|v| v.as_str())
1919        .unwrap_or("[]");
1920
1921    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
1922
1923    if elements.is_empty() {
1924        return Ok("(no interactive elements found)".to_owned());
1925    }
1926
1927    Ok(elements.join("\n"))
1928}
1929
1930fn get_page_text(tab: &Tab) -> PageContent {
1931    let url = tab.get_url();
1932
1933    let title = tab
1934        .evaluate("document.title", false)
1935        .ok()
1936        .and_then(|r| r.value)
1937        .and_then(|v| v.as_str().map(String::from))
1938        .unwrap_or_else(|| "unknown".to_owned());
1939
1940    let body_text = tab
1941        .evaluate(
1942            "document.body ? document.body.innerText : document.documentElement.innerText",
1943            false,
1944        )
1945        .ok()
1946        .and_then(|r| r.value)
1947        .and_then(|v| v.as_str().map(String::from))
1948        .unwrap_or_default();
1949
1950    PageContent {
1951        url,
1952        title,
1953        body_text: truncate(&body_text, 8000),
1954    }
1955}
1956
1957fn resolve_url(url: &str, base_url: &str) -> String {
1958    if url.starts_with("http://") || url.starts_with("https://") {
1959        return url.to_owned();
1960    }
1961    let base = base_url.trim_end_matches('/');
1962    if url.starts_with('/') {
1963        format!("{base}{url}")
1964    } else {
1965        format!("{base}/{url}")
1966    }
1967}
1968
1969/// Human-readable label for a step, used when steps are skipped after an
1970/// earlier failure.
1971fn step_label(step: &TestStep) -> String {
1972    match step {
1973        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
1974        TestStep::Click { target, .. } => format!("[click] {target}"),
1975        TestStep::Type { target, .. } => format!("[type] {target}"),
1976        TestStep::Wait { target, .. } => format!("[wait] {target}"),
1977        TestStep::Assert {
1978            definition,
1979            preset,
1980            prompt,
1981            ..
1982        } => definition.as_ref().map_or_else(
1983            || {
1984                preset.as_ref().map_or_else(
1985                    || {
1986                        prompt.as_ref().map_or_else(
1987                            || "[assert]".to_owned(),
1988                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
1989                        )
1990                    },
1991                    |p| format!("[assert] {p}"),
1992                )
1993            },
1994            |d| format!("[assert] {d}"),
1995        ),
1996        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
1997        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
1998        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
1999    }
2000}
2001
2002/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2003#[must_use]
2004const fn step_kind_label(step: &TestStep) -> &'static str {
2005    match step {
2006        TestStep::Navigate { .. } => "navigate",
2007        TestStep::Click { .. } => "click",
2008        TestStep::Type { .. } => "type",
2009        TestStep::Wait { .. } => "wait",
2010        TestStep::Assert { .. } => "assert",
2011        TestStep::Screenshot { .. } => "screenshot",
2012        TestStep::Agent { .. } => "agent",
2013        TestStep::Mcp { .. } => "mcp",
2014    }
2015}
2016
2017// ── Support types ──────────────────────────────────────────────────────
2018
2019#[derive(Default)]
2020struct TestRunResult {
2021    passed: u32,
2022    failed: u32,
2023    skipped: u32,
2024    total: u32,
2025    details: Vec<StepResult>,
2026}
2027
2028struct PageContent {
2029    url: String,
2030    title: String,
2031    body_text: String,
2032}