Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::UsageTracker;
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, TaskType};
14use crate::llm_chat_vision_with_usage;
15use crate::llm_chat_with_usage;
16use crate::mcp_client::McpClient;
17use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
18use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
19use crate::truncate;
20use crate::LlmConfig;
21use crate::DOM_EXTRACT_JS;
22
23/// One detected layout defect (`layout_no_issues` preset).
24#[derive(Debug, serde::Deserialize)]
25struct LayoutIssue {
26    #[serde(rename = "type")]
27    issue_type: String,
28    element: String,
29    detail: String,
30}
31
32/// In-browser DOM layout scan for `layout_no_issues`.
33///
34/// Geometry-only checks (no LLM, no pixels):
35/// 1. page horizontal overflow (`scrollWidth` > viewport width);
36/// 2. elements outside the viewport that scrolling cannot reveal
37///    (fixed elements off-screen, left/negative overflow, right-edge
38///    overflow beyond the horizontally scrollable area, and bottom
39///    overflow on a page that cannot scroll down) — below-the-fold
40///    content on a tall scrollable page is normal flow, NOT a defect;
41/// 3. text clipped by `overflow: hidden` containers whose content
42///    is measurably larger than the box;
43/// 4. interactive elements (buttons/links/inputs) whose center point
44///    is covered by a different element that would intercept the click.
45///
46/// Elements whose class matches a configured ignore prefix (default:
47/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
48/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
49/// skipped — they are intentionally 1x1 / off-screen. The prefix list
50/// is injected at `__IGNORE_CLASSES__` from
51/// [`ScenarioConfig::layout_ignore_classes`].
52///
53/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
54/// sticky headers) is excluded by the position/relation filters.
55const LAYOUT_SCAN_JS: &str = r#"
56(() => {
57  const issues = [];
58  const push = (type, el, detail) => {
59    if (issues.length >= 30) return;
60    let element = el.tagName.toLowerCase();
61    if (el.id) element += '#' + el.id;
62    else if (typeof el.className === 'string' && el.className.trim())
63      element += '.' + el.className.trim().split(/\s+/).join('.');
64    issues.push({ type, element, detail: String(detail).slice(0, 220) });
65  };
66  const vw = document.documentElement.clientWidth || window.innerWidth;
67  const vh = document.documentElement.clientHeight || window.innerHeight;
68  if (!vw || !vh) return JSON.stringify(issues);
69  const de = document.documentElement;
70  // 1. Page-level horizontal overflow.
71  if (de.scrollWidth > vw + 2)
72    push('page-overflow-x', de,
73      'page scrollWidth ' + de.scrollWidth + ' exceeds viewport width ' + vw);
74  // Class-name prefixes to skip (injected; default = Angular CDK
75  // screen-reader helpers, which are intentionally 1x1 / off-screen).
76  const ignorePrefixes = __IGNORE_CLASSES__;
77  const isIgnored = (el) => {
78    if (typeof el.className !== 'string' || !el.className.trim()) return false;
79    const classes = el.className.trim().split(/\s+/);
80    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
81  };
82  const vScrollable = de.scrollHeight > vh + 2;
83  const hScrollable = de.scrollWidth > vw + 2;
84  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
85  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
86  const hasContent = (el) =>
87    ((el.textContent || '').trim().length > 0) ||
88    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
89  // 2. Elements outside the viewport that scrolling cannot reveal.
90  for (const el of all) {
91    const cs = getComputedStyle(el);
92    if (!visible(cs)) continue;
93    if (isIgnored(el)) continue;
94    const r = el.getBoundingClientRect();
95    if (r.width < 2 || r.height < 2) continue;
96    if (!hasContent(el) && el.children.length === 0) continue;
97    if (cs.position === 'fixed') {
98      // Fixed elements never move with the scroll: any edge outside the
99      // viewport is unreachable content and therefore a defect.
100      const overTop = -r.top;
101      const overLeft = -r.left;
102      const overRight = r.right - vw;
103      const overBottom = r.bottom - vh;
104      if (overTop > 2 || overLeft > 2 || overRight > 2 || overBottom > 2) {
105        let where = '';
106        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
107        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
108        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
109        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
110        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
111        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
112        push('element-out-of-viewport', el,
113          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
114      }
115      continue;
116    }
117    if (cs.position === 'sticky') continue;
118    // Negative left overflow cannot be reached by scrolling (scrollLeft
119    // never goes below 0).
120    if (r.left < -2) {
121      push('element-out-of-viewport', el,
122        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
123      continue;
124    }
125    // Negative top with the page at the top means the element sits above
126    // the document origin — also unreachable.
127    if (r.top < -2 && de.scrollTop <= 2) {
128      push('element-out-of-viewport', el,
129        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
130      continue;
131    }
132    const overRight = r.right - vw;
133    // Right-edge overflow is only a defect when the page cannot scroll
134    // horizontally to reveal it (or the element sticks out past the
135    // scrollable content width itself).
136    if (overRight > 2 && (!hScrollable || r.right > de.scrollWidth + 2)) {
137      push('element-out-of-viewport', el,
138        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
139      continue;
140    }
141    // Below-the-fold content on a scrollable page is normal (tall landing
142    // pages); only flag bottom overflow the user can never scroll to.
143    const overBottom = r.bottom - vh;
144    if (overBottom > 2 && (!vScrollable || r.bottom > de.scrollHeight + 2)) {
145      push('element-out-of-viewport', el,
146        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
147    }
148  }
149  // 3. Text clipped by overflow:hidden containers.
150  for (const el of all) {
151    if (isIgnored(el)) continue;
152    const cs = getComputedStyle(el);
153    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
154    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
155    if (!(el.textContent || '').trim()) continue;
156    push('text-clipped', el,
157      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
158      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
159  }
160  // 4. Interactive elements covered by a different element.
161  const interactive =
162    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
163  const targets = document.querySelectorAll(interactive);
164  for (const el of targets) {
165    if (isIgnored(el)) continue;
166    const r = el.getBoundingClientRect();
167    if (r.width < 6 || r.height < 6) continue;
168    const cs = getComputedStyle(el);
169    if (!visible(cs)) continue;
170    const cx = r.left + r.width / 2;
171    const cy = r.top + r.height / 2;
172    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
173    const top = document.elementFromPoint(cx, cy);
174    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
175    if (isIgnored(top)) continue;
176    const tcs = getComputedStyle(top);
177    if (!visible(tcs)) continue;
178    if (tcs.pointerEvents === 'none') continue;
179    const tr = top.getBoundingClientRect();
180    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
181    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
182      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
183    push('element-overlap', el,
184      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
185      ' is covered by <' + tname + '>');
186  }
187  return JSON.stringify(issues);
188})()
189"#;
190
191/// How long the CDP connection stays open after the browser goes quiet.
192///
193/// `headless_chrome` ships a 30s default and tears down the entire connection
194/// when no traffic arrives for that long; a run must own its connection for
195/// its full duration instead.
196const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
197
198/// Executes a [`Scenario`] against a real browser with optional LLM
199/// assistance for element targeting and assertions.
200pub struct ScenarioRunner {
201    config: ScenarioConfig,
202    definitions: HashMap<String, AssertDefinition>,
203    llm: LlmConfig,
204    timeout: Duration,
205    viewport_width: u32,
206    viewport_height: u32,
207    /// The viewport currently applied in the browser (CDP emulation).
208    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
209    applied_viewport: std::cell::Cell<(u32, u32)>,
210    endpoints: EndpointRegistry,
211    usage: Arc<UsageTracker>,
212    budgets: BudgetTracker,
213    /// Directory for failure artifacts (screenshots).
214    artifacts_dir: PathBuf,
215}
216
217/// Aggregated results from a scenario run.
218#[derive(Debug, Default)]
219pub struct RunReport {
220    /// Number of tests that passed.
221    pub tests_passed: u32,
222    /// Number of tests that failed.
223    pub tests_failed: u32,
224    /// Number of steps that passed.
225    pub passed: u32,
226    /// Number of steps that failed.
227    pub failed: u32,
228    /// Number of steps that were skipped.
229    pub skipped: u32,
230    /// Per-step details.
231    pub details: Vec<StepResult>,
232}
233
234/// Result of a single step execution.
235#[derive(Debug)]
236pub struct StepResult {
237    /// The step name.
238    pub name: String,
239    /// Whether the step passed, failed, or was skipped.
240    pub status: StepStatus,
241    /// Human-readable result message.
242    pub message: String,
243}
244
245/// Outcome for a single step.
246#[derive(Debug, PartialEq, Eq)]
247pub enum StepStatus {
248    /// Step executed successfully and all assertions passed.
249    Passed,
250    /// Step execution or assertion failed.
251    Failed,
252    /// Step was skipped.
253    Skipped,
254}
255
256/// Predefined assertion preset definition.
257struct AssertPreset {
258    name: &'static str,
259    system: &'static str,
260    user_template: &'static str,
261}
262
263/// Built-in assertion presets.
264#[allow(clippy::literal_string_with_formatting_args)]
265const ASSERTION_PRESETS: &[AssertPreset] = &[
266    AssertPreset {
267        name: "no_error_on_page",
268        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
269        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
270    },
271    AssertPreset {
272        name: "text_visible",
273        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
274        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
275    },
276    AssertPreset {
277        name: "element_exists",
278        system: "You are a QA tester. Check if a described UI element exists on a web page.",
279        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
280    },
281    AssertPreset {
282        name: "visual_no_issues",
283        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
284        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
285    },
286    AssertPreset {
287        name: "visual_no_overlaps",
288        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
289        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
290    },
291    AssertPreset {
292        name: "visual_text_visible",
293        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
294        user_template: "Inspect the attached screenshot and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
295    },
296    AssertPreset {
297        name: "layout_no_issues",
298        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
299        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
300    },
301];
302
303impl ScenarioRunner {
304    /// Creates a new runner with the given scenario configuration and
305    /// assertion definitions.
306    #[must_use]
307    #[allow(clippy::needless_pass_by_value)]
308    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
309        let llm = LlmConfig {
310            url: scenario_config
311                .llm_url
312                .clone()
313                .unwrap_or_else(crate::llm_base_url),
314            model: scenario_config
315                .llm_model
316                .clone()
317                .unwrap_or_else(crate::llm_model),
318            api_key: scenario_config
319                .llm_api_key
320                .clone()
321                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
322            headers: if scenario_config.llm_headers.is_empty() {
323                crate::parse_headers_env()
324            } else {
325                scenario_config.llm_headers.clone()
326            },
327            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
328            temperature: scenario_config.temperature,
329            thinking: scenario_config.thinking,
330            model_params: scenario_config.model_params.clone(),
331        };
332        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
333        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
334        let defs_map: HashMap<String, AssertDefinition> = definitions
335            .into_iter()
336            .map(|d| (d.name.clone(), d))
337            .collect();
338
339        Self {
340            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
341            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
342            viewport_height: scenario_config.viewport_height.unwrap_or(720),
343            applied_viewport: std::cell::Cell::new((0, 0)),
344            config: scenario_config.clone(),
345            definitions: defs_map,
346            llm,
347            endpoints,
348            usage: Arc::new(UsageTracker::new()),
349            budgets,
350            artifacts_dir: PathBuf::from(
351                scenario_config
352                    .artifacts_dir
353                    .unwrap_or_else(|| "artifacts".to_owned()),
354            ),
355        }
356    }
357
358    /// Returns a clone of the [`UsageTracker`] for reporting.
359    #[must_use]
360    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
361        Arc::clone(&self.usage)
362    }
363
364    /// Returns a reference to the [`BudgetTracker`].
365    #[must_use]
366    pub const fn budget_tracker(&self) -> &BudgetTracker {
367        &self.budgets
368    }
369
370    /// Executes all test groups in the scenario and returns a report.
371    ///
372    /// # Errors
373    ///
374    /// Returns an error if the browser fails to launch.
375    #[allow(clippy::too_many_lines)]
376    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
377        let mut report = RunReport::default();
378
379        if tests.is_empty() {
380            eprintln!("No tests defined in scenario.");
381            return Ok(report);
382        }
383
384        let browser_headless = self.config.browser_headless.unwrap_or(true);
385
386        let launch_opts = LaunchOptions {
387            headless: browser_headless,
388            window_size: Some((self.viewport_width, self.viewport_height)),
389            sandbox: false,
390            // headless_chrome defaults this to 30s and shuts down the whole CDP
391            // connection when no messages arrive for that long. A scenario can
392            // easily exceed 30s of browser silence (slow LLM targeting/assertion
393            // calls, page waits, budget checks between steps), after which every
394            // remaining step fails with "Unable to make method calls because
395            // underlying connection is closed" — one quiet gap kills the run.
396            // Open-ended scenarios must own the connection for their full
397            // duration, so keep it alive for 6 hours.
398            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
399            ..LaunchOptions::default()
400        };
401
402        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
403        let tab = browser.new_tab().context("failed to open browser tab")?;
404        let _ = tab.set_default_timeout(self.timeout);
405
406        // Start MCP server if configured
407        #[cfg(feature = "mcp-server")]
408        if let Some(ref mcp_cfg) = self.config.mcp_server {
409            if mcp_cfg.enabled {
410                let port = mcp_cfg.port;
411                std::thread::spawn(move || {
412                    let _ = crate::mcp_server::start_mcp_server(port);
413                });
414            }
415        }
416        #[cfg(not(feature = "mcp-server"))]
417        if let Some(mcp_cfg) = &self.config.mcp_server {
418            if mcp_cfg.enabled {
419                eprintln!("  ⚠️  MCP server configured but 'mcp-server' feature not enabled");
420            }
421        }
422
423        // Start A2A agent server if configured
424        #[cfg(feature = "a2a-server")]
425        if let Some(ref a2a_cfg) = self.config.a2a_server {
426            if a2a_cfg.enabled {
427                let port = a2a_cfg.port;
428                tokio::spawn(crate::a2a_server::start_a2a_server(port));
429            }
430        }
431        #[cfg(not(feature = "a2a-server"))]
432        if let Some(a2a_cfg) = &self.config.a2a_server {
433            if a2a_cfg.enabled {
434                eprintln!("  ⚠️  A2A server configured but 'a2a-server' feature not enabled");
435            }
436        }
437
438        for test in tests {
439            eprintln!("\n╔══════════════════════════════");
440            eprintln!("║  Test: {}", test.name);
441            eprintln!("╚══════════════════════════════");
442
443            self.usage.reset_per_test();
444
445            let test_result = self.run_test(test, &tab);
446            self.usage.commit_test(&test.name);
447
448            if test_result.failed == 0 && test_result.total > 0 {
449                report.tests_passed += 1;
450                eprintln!("  Test ✅ Passed");
451            } else if test_result.total > 0 {
452                report.tests_failed += 1;
453                eprintln!("  Test ❌ Failed");
454            }
455
456            report.passed += test_result.passed;
457            report.failed += test_result.failed;
458            report.skipped += test_result.skipped;
459            report.details.extend(test_result.details);
460        }
461
462        Ok(report)
463    }
464
465    #[allow(clippy::too_many_lines)]
466    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
467        let base_url = test
468            .base_url
469            .clone()
470            .or_else(|| self.config.base_url.clone())
471            .unwrap_or_else(crate::base_url);
472
473        // Per-test viewport override: switch the browser via CDP
474        // device-metrics emulation before this test runs.
475        let vw = test.viewport_width.unwrap_or(self.viewport_width);
476        let vh = test.viewport_height.unwrap_or(self.viewport_height);
477        if self.applied_viewport.get() != (vw, vh) {
478            self.apply_viewport(tab, vw, vh);
479            self.applied_viewport.set((vw, vh));
480        }
481
482        // Per-test isolation: every test starts from its own start_url
483        // (unless auto_navigate is disabled), so a test never inherits the
484        // previous test's page state.
485        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
486
487        let start_url = test
488            .start_url
489            .clone()
490            .or_else(|| self.config.start_url.clone())
491            .unwrap_or_else(|| "/dashboard".to_owned());
492
493        if auto_navigate {
494            let full_url = resolve_url(&start_url, &base_url);
495            eprintln!("  → auto-navigate: {full_url}");
496            let _ = tab.navigate_to(&full_url);
497            let _ = tab.wait_until_navigated();
498            std::thread::sleep(Duration::from_secs(4));
499        }
500
501        let mut result = TestRunResult::default();
502
503        for (step_index, step) in test.steps.iter().enumerate() {
504            result.total += 1;
505
506            let wait_ms = match step {
507                TestStep::Navigate { wait_after_ms, .. }
508                | TestStep::Click { wait_after_ms, .. }
509                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
510                _ => None,
511            };
512
513            let mut step_result = match step {
514                TestStep::Navigate { url, .. } => {
515                    let full_url = resolve_url(url, &base_url);
516                    run_navigate_step(&full_url, tab)
517                }
518                TestStep::Click {
519                    target,
520                    selector,
521                    endpoint,
522                    idempotent,
523                    ..
524                } => self.run_click(
525                    target,
526                    selector.as_deref(),
527                    endpoint.as_deref(),
528                    test.endpoint.as_deref(),
529                    *idempotent,
530                    tab,
531                ),
532                TestStep::Type {
533                    target,
534                    text,
535                    selector,
536                    endpoint,
537                    idempotent,
538                    ..
539                } => self.run_type(
540                    target,
541                    text,
542                    selector.as_deref(),
543                    endpoint.as_deref(),
544                    test.endpoint.as_deref(),
545                    *idempotent,
546                    tab,
547                ),
548                TestStep::Wait {
549                    target,
550                    selector,
551                    text,
552                    timeout_ms,
553                    endpoint,
554                    idempotent,
555                } => self.run_wait(
556                    target,
557                    selector.as_deref(),
558                    text.as_deref(),
559                    *timeout_ms,
560                    endpoint.as_deref(),
561                    test.endpoint.as_deref(),
562                    *idempotent,
563                    tab,
564                ),
565                TestStep::Assert {
566                    definition,
567                    preset,
568                    prompt,
569                    assert_text,
570                    endpoint,
571                    screenshot,
572                } => self.run_assert(
573                    definition.as_deref(),
574                    preset.as_deref(),
575                    prompt.as_deref(),
576                    assert_text.as_deref(),
577                    *screenshot,
578                    endpoint.as_deref(),
579                    test.endpoint.as_deref(),
580                    tab,
581                ),
582                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
583                TestStep::Agent {
584                    agent,
585                    task,
586                    definition,
587                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
588                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
589            };
590
591            // Failure diagnostics: capture the page state and a screenshot so
592            // CI logs say WHAT the page looked like when the step failed,
593            // instead of a bare "timed out: The event waited for never came".
594            if step_result.status == StepStatus::Failed {
595                let state = diagnostics::capture(tab);
596                let screenshot = diagnostics::save_screenshot(
597                    tab,
598                    &self.artifacts_dir,
599                    &test.name,
600                    &test.name,
601                    step_index,
602                    step_kind_label(step),
603                );
604                step_result.message = format!(
605                    "{base} — {excerpt}",
606                    base = step_result.message,
607                    excerpt = diagnostics::inline_excerpt(&state),
608                );
609                eprintln!(
610                    "{}",
611                    diagnostics::full_context(&state, screenshot.as_deref())
612                );
613            }
614
615            eprintln!(
616                "    {} {} — {}",
617                if step_result.status == StepStatus::Passed {
618                    "✅"
619                } else if step_result.status == StepStatus::Failed {
620                    "❌"
621                } else {
622                    "⏭️"
623                },
624                step_result.name,
625                step_result.message,
626            );
627
628            match step_result.status {
629                StepStatus::Passed => result.passed += 1,
630                StepStatus::Failed => result.failed += 1,
631                StepStatus::Skipped => result.skipped += 1,
632            }
633
634            // Fail fast: the first failed step ends the test and the
635            // remaining steps are reported as skipped (no LLM budget is
636            // burned asserting against a page that is already known broken).
637            if step_result.status == StepStatus::Failed
638                && !self.config.continue_on_failure
639                && step_index + 1 < test.steps.len()
640            {
641                eprintln!(
642                    "      ⏭️  failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
643                    test.steps.len() - step_index - 1
644                );
645                for skipped in &test.steps[step_index + 1..] {
646                    result.total += 1;
647                    result.skipped += 1;
648                    eprintln!(
649                        "    ⏭️  {} — skipped: previous step failed",
650                        step_label(skipped)
651                    );
652                    result.details.push(StepResult {
653                        name: step_label(skipped),
654                        status: StepStatus::Skipped,
655                        message: "skipped: previous step failed".into(),
656                    });
657                }
658                result.details.push(step_result);
659                return result;
660            }
661
662            // Check per-test budget after each step
663            let test_usage = self.usage.current_test_snapshot();
664            let global_usage = self.usage.global_snapshot();
665            let budget_status = self.budgets.check_all(
666                &test.name,
667                &test_usage,
668                &global_usage,
669                test.budget.as_ref(),
670            );
671            match budget_status {
672                BudgetStatus::HardExceeded { message, .. } => {
673                    crate::reporting::print_budget_error(&message);
674                    result.details.push(StepResult {
675                        name: "[budget]".into(),
676                        status: StepStatus::Failed,
677                        message,
678                    });
679                    result.failed += 1;
680                    return result;
681                }
682                BudgetStatus::SoftExceeded { message, .. } => {
683                    crate::reporting::print_budget_warning(&message);
684                }
685                BudgetStatus::Ok => {}
686            }
687
688            if let Some(ms) = wait_ms {
689                std::thread::sleep(Duration::from_millis(ms));
690            }
691
692            result.details.push(step_result);
693        }
694
695        result
696    }
697
698    /// Applies a viewport size to the current tab via CDP
699    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
700    /// overrides and the viewport matrix. The initial window size set at
701    /// browser launch is replaced by emulation; failures are logged but
702    /// do not fail the test (a mismatched viewport only weakens coverage).
703    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
704        use headless_chrome::protocol::cdp::Emulation;
705        let _ = self;
706        let params = Emulation::SetDeviceMetricsOverride {
707            width,
708            height,
709            device_scale_factor: 1.0,
710            mobile: false,
711            scale: None,
712            screen_width: Some(width),
713            screen_height: Some(height),
714            position_x: None,
715            position_y: None,
716            dont_set_visible_size: None,
717            screen_orientation: None,
718            viewport: None,
719            display_feature: None,
720            device_posture: None,
721        };
722        eprintln!("      ↻ viewport: {width}x{height}");
723        if let Err(e) = tab.call_method(params) {
724            eprintln!("      ⚠️  viewport switch to {width}x{height} failed: {e}");
725        }
726    }
727
728    // ── step handlers ───────────────────────────────────────────────────
729
730    #[allow(clippy::too_many_lines)]
731    fn run_click(
732        &self,
733        target: &str,
734        selector_override: Option<&str>,
735        step_endpoint: Option<&str>,
736        test_endpoint: Option<&str>,
737        idempotent: bool,
738        tab: &Tab,
739    ) -> StepResult {
740        let name = format!("[click] {target}");
741        let selector = match self.resolve_selector(
742            selector_override,
743            target,
744            step_endpoint,
745            test_endpoint,
746            tab,
747        ) {
748            Ok(s) => s,
749            Err(msg) => {
750                if idempotent {
751                    return StepResult {
752                        name,
753                        status: StepStatus::Skipped,
754                        message: format!("skipped (idempotent): no target found — {msg}"),
755                    };
756                }
757                return StepResult {
758                    name,
759                    status: StepStatus::Failed,
760                    message: msg,
761                };
762            }
763        };
764
765        // Idempotent steps probe briefly: a missing target means the
766        // action was already done / not applicable (e.g. an
767        // already-authenticated session), and skipping is the success
768        // path, not a failure.
769        let probe_secs = if idempotent { 5 } else { 10 };
770        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
771            Ok(element) => match element.click() {
772                Ok(_) => StepResult {
773                    name,
774                    status: StepStatus::Passed,
775                    message: format!("clicked {selector}"),
776                },
777                Err(e) => StepResult {
778                    name,
779                    status: StepStatus::Failed,
780                    message: format!("click failed on {selector}: {e}"),
781                },
782            },
783            Err(e) if idempotent => StepResult {
784                name,
785                status: StepStatus::Skipped,
786                message: format!("skipped (idempotent): element {selector} not present — {e}"),
787            },
788            Err(e) => StepResult {
789                name,
790                status: StepStatus::Failed,
791                message: format!("element {selector} not found: {e}"),
792            },
793        }
794    }
795
796    #[allow(clippy::too_many_arguments)]
797    fn run_type(
798        &self,
799        target: &str,
800        text: &str,
801        selector_override: Option<&str>,
802        step_endpoint: Option<&str>,
803        test_endpoint: Option<&str>,
804        idempotent: bool,
805        tab: &Tab,
806    ) -> StepResult {
807        let name = format!("[type] {target}");
808        let selector = match self.resolve_selector(
809            selector_override,
810            target,
811            step_endpoint,
812            test_endpoint,
813            tab,
814        ) {
815            Ok(s) => s,
816            Err(msg) => {
817                if idempotent {
818                    return StepResult {
819                        name,
820                        status: StepStatus::Skipped,
821                        message: format!("skipped (idempotent): no target found — {msg}"),
822                    };
823                }
824                return StepResult {
825                    name,
826                    status: StepStatus::Failed,
827                    message: msg,
828                };
829            }
830        };
831
832        let probe_secs = if idempotent { 5 } else { 10 };
833        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
834            Ok(element) => {
835                if let Err(e) = element.click() {
836                    return StepResult {
837                        name,
838                        status: StepStatus::Failed,
839                        message: format!("click to focus {selector} failed: {e}"),
840                    };
841                }
842
843                let js = format!(
844                    "document.querySelector('{}').value = '';",
845                    selector.replace('\'', "\\'")
846                );
847                let _ = tab.evaluate(&js, false);
848
849                match element.type_into(text) {
850                    Ok(_) => StepResult {
851                        name,
852                        status: StepStatus::Passed,
853                        message: format!("typed {text:?} into {selector}"),
854                    },
855                    Err(e) => StepResult {
856                        name,
857                        status: StepStatus::Failed,
858                        message: format!("type into {selector} failed: {e}"),
859                    },
860                }
861            }
862            Err(e) if idempotent => StepResult {
863                name,
864                status: StepStatus::Skipped,
865                message: format!("skipped (idempotent): element {selector} not present — {e}"),
866            },
867            Err(e) => StepResult {
868                name,
869                status: StepStatus::Failed,
870                message: format!("element {selector} not found: {e}"),
871            },
872        }
873    }
874
875    #[allow(clippy::too_many_arguments)]
876    #[allow(clippy::too_many_lines)]
877    fn run_wait(
878        &self,
879        target: &str,
880        selector_override: Option<&str>,
881        text: Option<&str>,
882        timeout_ms: Option<u64>,
883        step_endpoint: Option<&str>,
884        test_endpoint: Option<&str>,
885        idempotent: bool,
886        tab: &Tab,
887    ) -> StepResult {
888        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
889        let step_name = format!("[wait] {target}");
890
891        // Resolve an explicit selector only (text-only waits are LLM-free).
892        let selector = match selector_override {
893            Some(s) => Some(s.to_owned()),
894            None if text.is_some() => None,
895            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
896                Ok(s) => Some(s),
897                Err(msg) => {
898                    if idempotent {
899                        return StepResult {
900                            name: step_name,
901                            status: StepStatus::Skipped,
902                            message: format!("skipped (idempotent): no target found — {msg}"),
903                        };
904                    }
905                    return StepResult {
906                        name: step_name,
907                        status: StepStatus::Failed,
908                        message: msg,
909                    };
910                }
911            },
912        };
913
914        if text.is_some() {
915            let sel_js = selector
916                .as_deref()
917                .map(crate::selectors::selector_matches_js);
918            let text_js = text.map(|t| {
919                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
920                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
921            });
922
923            let deadline = Instant::now() + timeout;
924            loop {
925                let sel_ok = sel_js
926                    .as_ref()
927                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
928                let text_ok = text_js
929                    .as_ref()
930                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
931                if sel_ok && text_ok {
932                    let mut what = Vec::new();
933                    if let Some(sel) = &selector {
934                        what.push(format!("found {sel}"));
935                    }
936                    if let Some(t) = text {
937                        what.push(format!("text {t:?} visible"));
938                    }
939                    return StepResult {
940                        name: step_name,
941                        status: StepStatus::Passed,
942                        message: what.join(" and "),
943                    };
944                }
945                if Instant::now() >= deadline {
946                    let mut what = Vec::new();
947                    if let Some(sel) = &selector {
948                        what.push(sel.clone());
949                    }
950                    if let Some(t) = text {
951                        what.push(format!("text {t:?}"));
952                    }
953                    let message = format!(
954                        "wait for {} timed out after {}ms: the event waited for never came",
955                        what.join(" / "),
956                        timeout.as_millis(),
957                    );
958                    if idempotent {
959                        return StepResult {
960                            name: step_name,
961                            status: StepStatus::Skipped,
962                            message: format!("skipped (idempotent): {message}"),
963                        };
964                    }
965                    return StepResult {
966                        name: step_name,
967                        status: StepStatus::Failed,
968                        message,
969                    };
970                }
971                std::thread::sleep(Duration::from_millis(250));
972            }
973        }
974
975        match selector.as_deref() {
976            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
977                Ok(_) => StepResult {
978                    name: step_name,
979                    status: StepStatus::Passed,
980                    message: format!("found {sel}"),
981                },
982                Err(e) if idempotent => StepResult {
983                    name: step_name,
984                    status: StepStatus::Skipped,
985                    message: format!(
986                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
987                        timeout.as_millis()
988                    ),
989                },
990                Err(e) => StepResult {
991                    name: step_name,
992                    status: StepStatus::Failed,
993                    message: format!(
994                        "wait for {sel} timed out after {}ms: {e}",
995                        timeout.as_millis()
996                    ),
997                },
998            },
999            None => StepResult {
1000                name: step_name,
1001                status: StepStatus::Failed,
1002                message: "wait step has neither selector nor text".into(),
1003            },
1004        }
1005    }
1006
1007    #[allow(clippy::too_many_arguments)]
1008    fn run_assert(
1009        &self,
1010        definition: Option<&str>,
1011        preset: Option<&str>,
1012        prompt: Option<&str>,
1013        assert_text: Option<&str>,
1014        screenshot: bool,
1015        step_endpoint: Option<&str>,
1016        test_endpoint: Option<&str>,
1017        tab: &Tab,
1018    ) -> StepResult {
1019        std::thread::sleep(Duration::from_millis(500));
1020
1021        let page_content = get_page_text(tab);
1022
1023        // Vision attach: capture the viewport once per assert step and hand
1024        // the JPEG data URL to the preset/prompt evaluation below.
1025        let image = if screenshot {
1026            let endpoint = self
1027                .endpoints
1028                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1029            if !endpoint.vision {
1030                return StepResult {
1031                    name: "[assert]".into(),
1032                    status: StepStatus::Failed,
1033                    message: format!(
1034                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1035                        name = endpoint.name
1036                    ),
1037                };
1038            }
1039            match crate::vision::capture_screenshot_data_url(
1040                tab,
1041                self.config
1042                    .screenshot_max_dimension
1043                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1044            ) {
1045                Ok(data_url) => Some(data_url),
1046                Err(e) => {
1047                    return StepResult {
1048                        name: "[assert]".into(),
1049                        status: StepStatus::Failed,
1050                        message: format!("screenshot capture failed: {e}"),
1051                    };
1052                }
1053            }
1054        } else {
1055            None
1056        };
1057
1058        if let Some(def_name) = definition {
1059            if let Some(def) = self.definitions.get(def_name) {
1060                return self.run_assert_def(
1061                    def,
1062                    &page_content,
1063                    image.as_deref(),
1064                    step_endpoint,
1065                    test_endpoint,
1066                    tab,
1067                );
1068            }
1069            return StepResult {
1070                name: format!("[assert] {def_name}"),
1071                status: StepStatus::Failed,
1072                message: format!("definition '{def_name}' not found"),
1073            };
1074        }
1075
1076        if let Some(preset_name) = preset {
1077            // Deterministic DOM layout scan — runs JS in the browser and
1078            // never calls the LLM (free, fast, no pixel budget).
1079            if preset_name == "layout_no_issues" {
1080                return self.run_layout_preset(tab);
1081            }
1082            return self.run_preset(
1083                preset_name,
1084                assert_text,
1085                &page_content,
1086                image.as_deref(),
1087                step_endpoint,
1088                test_endpoint,
1089            );
1090        }
1091
1092        if let Some(prompt_text) = prompt {
1093            return self.run_custom(
1094                prompt_text,
1095                &page_content,
1096                image.as_deref(),
1097                step_endpoint,
1098                test_endpoint,
1099            );
1100        }
1101
1102        StepResult {
1103            name: "[assert]".into(),
1104            status: StepStatus::Skipped,
1105            message: "no definition, preset, or prompt specified".into(),
1106        }
1107    }
1108
1109    fn run_assert_def(
1110        &self,
1111        def: &AssertDefinition,
1112        page_content: &PageContent,
1113        image: Option<&str>,
1114        step_endpoint: Option<&str>,
1115        test_endpoint: Option<&str>,
1116        tab: &Tab,
1117    ) -> StepResult {
1118        // Agent-based definition: delegate to an A2A agent
1119        if let Some(ref agent) = def.agent {
1120            if image.is_some() {
1121                return StepResult {
1122                    name: format!("[assert] {}", def.name),
1123                    status: StepStatus::Failed,
1124                    message: "agent-backed assertions do not support screenshots".into(),
1125                };
1126            }
1127            let task = def
1128                .task_template
1129                .as_deref()
1130                .unwrap_or("Evaluate the assertion")
1131                .replace("{url}", &page_content.url)
1132                .replace("{title}", &page_content.title)
1133                .replace("{content}", &page_content.body_text)
1134                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1135
1136            return self.run_agent_step(agent, &task, &def.name);
1137        }
1138
1139        // Custom preset: system + user_template provided in the definition
1140        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1141            return self.run_custom_preset(
1142                &def.name,
1143                system,
1144                template,
1145                def.assert_text.as_deref(),
1146                page_content,
1147                image,
1148                step_endpoint,
1149                test_endpoint,
1150            );
1151        }
1152
1153        def.preset.as_ref().map_or_else(
1154            || {
1155                def.prompt.as_ref().map_or_else(
1156                    || StepResult {
1157                        name: format!("[assert] {}", def.name),
1158                        status: StepStatus::Failed,
1159                        message: "definition has no preset, prompt, or system+user_template".into(),
1160                    },
1161                    |prompt| {
1162                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1163                    },
1164                )
1165            },
1166            |preset_name| {
1167                if preset_name == "layout_no_issues" {
1168                    return self.run_layout_preset(tab);
1169                }
1170                self.run_preset(
1171                    preset_name,
1172                    def.assert_text.as_deref(),
1173                    page_content,
1174                    image,
1175                    step_endpoint,
1176                    test_endpoint,
1177                )
1178            },
1179        )
1180    }
1181
1182    #[allow(clippy::too_many_arguments)]
1183    fn run_custom_preset(
1184        &self,
1185        name: &str,
1186        system: &str,
1187        template: &str,
1188        assert_text: Option<&str>,
1189        page_content: &PageContent,
1190        image: Option<&str>,
1191        step_endpoint: Option<&str>,
1192        test_endpoint: Option<&str>,
1193    ) -> StepResult {
1194        let user_prompt = template
1195            .replace("{url}", &page_content.url)
1196            .replace("{title}", &page_content.title)
1197            .replace("{content}", &page_content.body_text)
1198            .replace("{expected_text}", assert_text.unwrap_or(""))
1199            .replace("{description}", "");
1200
1201        // Custom preset definitions frequently forget the {content}
1202        // placeholder — without it the LLM has no page to evaluate and
1203        // answers "I can't determine that without seeing the page". Always
1204        // append the page context unless the template already references it.
1205        let user_prompt = if template.contains("{content}") {
1206            user_prompt
1207        } else {
1208            format!(
1209                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1210                url = page_content.url,
1211                title = page_content.title,
1212                content = page_content.body_text,
1213            )
1214        };
1215
1216        eprintln!("      assert: {name} (custom preset)");
1217
1218        let endpoint = self
1219            .endpoints
1220            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1221        let llm = self.build_llm_for_endpoint(endpoint);
1222        let usage = Arc::clone(&self.usage);
1223        let endpoint_name = endpoint.name.clone();
1224        let sys = system.to_owned();
1225        let image = image.map(str::to_owned);
1226
1227        let response = std::thread::spawn(move || {
1228            let rt = tokio::runtime::Builder::new_current_thread()
1229                .enable_all()
1230                .build()
1231                .unwrap();
1232            let call = async {
1233                match image.as_deref() {
1234                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user_prompt, img).await,
1235                    None => llm_chat_with_usage(&llm, &sys, &user_prompt).await,
1236                }
1237            };
1238            rt.block_on(call)
1239        })
1240        .join()
1241        .unwrap();
1242
1243        response.map_or_else(
1244            |e| StepResult {
1245                name: format!("[assert] {name}"),
1246                status: StepStatus::Failed,
1247                message: format!("LLM assertion call failed: {e}"),
1248            },
1249            |lr| {
1250                usage.record_llm_call(
1251                    &endpoint_name,
1252                    endpoint,
1253                    lr.usage.prompt_tokens,
1254                    lr.usage.completion_tokens,
1255                );
1256                let content_lower = lr.content.to_lowercase().trim().to_owned();
1257                if content_lower.starts_with("pass") {
1258                    StepResult {
1259                        name: format!("[assert] {name}"),
1260                        status: StepStatus::Passed,
1261                        message: "PASS".into(),
1262                    }
1263                } else {
1264                    StepResult {
1265                        name: format!("[assert] {name}"),
1266                        status: StepStatus::Failed,
1267                        message: lr.content,
1268                    }
1269                }
1270            },
1271        )
1272    }
1273
1274    fn run_preset(
1275        &self,
1276        preset_name: &str,
1277        assert_text: Option<&str>,
1278        page_content: &PageContent,
1279        image: Option<&str>,
1280        step_endpoint: Option<&str>,
1281        test_endpoint: Option<&str>,
1282    ) -> StepResult {
1283        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1284            return StepResult {
1285                name: format!("[assert] {preset_name}"),
1286                status: StepStatus::Failed,
1287                message: format!("unknown assertion preset: {preset_name}"),
1288            };
1289        };
1290        if preset_name.starts_with("visual_") && image.is_none() {
1291            return StepResult {
1292                name: format!("[assert] {preset_name}"),
1293                status: StepStatus::Failed,
1294                message: format!(
1295                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1296                ),
1297            };
1298        }
1299
1300        let user_prompt = preset
1301            .user_template
1302            .replace("{url}", &page_content.url)
1303            .replace("{title}", &page_content.title)
1304            .replace("{content}", &page_content.body_text)
1305            .replace("{expected_text}", assert_text.unwrap_or(""))
1306            .replace("{description}", "");
1307
1308        // Same safety net as custom presets: never let the LLM answer with
1309        // no page context at all.
1310        let user_prompt = if preset.user_template.contains("{content}") {
1311            user_prompt
1312        } else {
1313            format!(
1314                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1315                url = page_content.url,
1316                title = page_content.title,
1317                content = page_content.body_text,
1318            )
1319        };
1320
1321        eprintln!("      assert: {preset_name}");
1322
1323        let endpoint = self
1324            .endpoints
1325            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1326        let llm = self.build_llm_for_endpoint(endpoint);
1327        let usage = Arc::clone(&self.usage);
1328        let endpoint_name = endpoint.name.clone();
1329        let sys = preset.system.to_owned();
1330        let image = image.map(str::to_owned);
1331
1332        let response = std::thread::spawn(move || {
1333            let rt = tokio::runtime::Builder::new_current_thread()
1334                .enable_all()
1335                .build()
1336                .unwrap();
1337            let call = async {
1338                match image.as_deref() {
1339                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user_prompt, img).await,
1340                    None => llm_chat_with_usage(&llm, &sys, &user_prompt).await,
1341                }
1342            };
1343            rt.block_on(call)
1344        })
1345        .join()
1346        .unwrap();
1347
1348        response.map_or_else(
1349            |e| StepResult {
1350                name: format!("[assert] {preset_name}"),
1351                status: StepStatus::Failed,
1352                message: format!("LLM assertion call failed: {e}"),
1353            },
1354            |lr| {
1355                usage.record_llm_call(
1356                    &endpoint_name,
1357                    endpoint,
1358                    lr.usage.prompt_tokens,
1359                    lr.usage.completion_tokens,
1360                );
1361                let content_lower = lr.content.to_lowercase().trim().to_owned();
1362                if content_lower.starts_with("pass") {
1363                    StepResult {
1364                        name: format!("[assert] {preset_name}"),
1365                        status: StepStatus::Passed,
1366                        message: "PASS".into(),
1367                    }
1368                } else {
1369                    StepResult {
1370                        name: format!("[assert] {preset_name}"),
1371                        status: StepStatus::Failed,
1372                        message: lr.content,
1373                    }
1374                }
1375            },
1376        )
1377    }
1378
1379    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1380    ///
1381    /// Evaluates the layout-scan JS in the page and fails with the list of
1382    /// detected issues: horizontal page overflow, visible elements sticking
1383    /// out of the viewport, text clipped by `overflow: hidden` containers,
1384    /// and interactive elements covered by other elements. No LLM call —
1385    /// checks are geometry-based so the check is free, deterministic, and
1386    /// safe to run on every page × viewport variant.
1387    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1388        let name = "[assert] layout_no_issues".to_owned();
1389        eprintln!("      assert: layout_no_issues (DOM layout scan)");
1390        let js = LAYOUT_SCAN_JS.replace(
1391            "__IGNORE_CLASSES__",
1392            &serde_json::to_string(&self.config.layout_ignore_classes)
1393                .unwrap_or_else(|_| "[]".to_owned()),
1394        );
1395        let result = tab.evaluate(&js, false);
1396        let json_str = match result {
1397            Ok(r) => r
1398                .value
1399                .as_ref()
1400                .and_then(|v| v.as_str().map(String::from))
1401                .unwrap_or_else(|| "[]".to_owned()),
1402            Err(e) => {
1403                return StepResult {
1404                    name,
1405                    status: StepStatus::Failed,
1406                    message: format!("layout scan JS failed: {e}"),
1407                };
1408            }
1409        };
1410        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1411        if issues.is_empty() {
1412            return StepResult {
1413                name,
1414                status: StepStatus::Passed,
1415                message: "PASS — no layout defects detected".into(),
1416            };
1417        }
1418        let mut lines: Vec<String> = issues
1419            .iter()
1420            .take(10)
1421            .map(|i| {
1422                format!(
1423                    "- [{type_}] {element}: {detail}",
1424                    type_ = i.issue_type,
1425                    element = i.element,
1426                    detail = i.detail
1427                )
1428            })
1429            .collect();
1430        if issues.len() > 10 {
1431            lines.push(format!("- … and {} more", issues.len() - 10));
1432        }
1433        StepResult {
1434            name,
1435            status: StepStatus::Failed,
1436            message: format!(
1437                "FAIL — {} layout defect(s) detected:\n{}",
1438                issues.len(),
1439                lines.join("\n")
1440            ),
1441        }
1442    }
1443
1444    fn run_custom(
1445        &self,
1446        prompt: &str,
1447        page_content: &PageContent,
1448        image: Option<&str>,
1449        step_endpoint: Option<&str>,
1450        test_endpoint: Option<&str>,
1451    ) -> StepResult {
1452        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1453
1454        let mut user = format!(
1455            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1456            url = page_content.url,
1457            title = page_content.title,
1458            content = page_content.body_text,
1459        );
1460        if image.is_some() {
1461            user.push_str(
1462                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1463            );
1464        }
1465
1466        eprintln!("      custom assert");
1467
1468        let endpoint = self
1469            .endpoints
1470            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1471        let llm = self.build_llm_for_endpoint(endpoint);
1472        let usage = Arc::clone(&self.usage);
1473        let endpoint_name = endpoint.name.clone();
1474        let sys = system.to_owned();
1475        let image = image.map(str::to_owned);
1476
1477        let response = std::thread::spawn(move || {
1478            let rt = tokio::runtime::Builder::new_current_thread()
1479                .enable_all()
1480                .build()
1481                .unwrap();
1482            let call = async {
1483                match image.as_deref() {
1484                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user, img).await,
1485                    None => llm_chat_with_usage(&llm, &sys, &user).await,
1486                }
1487            };
1488            rt.block_on(call)
1489        })
1490        .join()
1491        .unwrap();
1492
1493        response.map_or_else(
1494            |e| StepResult {
1495                name: "[assert] custom".into(),
1496                status: StepStatus::Failed,
1497                message: format!("LLM assertion call failed: {e}"),
1498            },
1499            |lr| {
1500                usage.record_llm_call(
1501                    &endpoint_name,
1502                    endpoint,
1503                    lr.usage.prompt_tokens,
1504                    lr.usage.completion_tokens,
1505                );
1506                let content_lower = lr.content.to_lowercase().trim().to_owned();
1507                if content_lower.starts_with("pass") {
1508                    StepResult {
1509                        name: "[assert] custom".into(),
1510                        status: StepStatus::Passed,
1511                        message: "PASS".into(),
1512                    }
1513                } else {
1514                    StepResult {
1515                        name: "[assert] custom".into(),
1516                        status: StepStatus::Failed,
1517                        message: lr.content,
1518                    }
1519                }
1520            },
1521        )
1522    }
1523
1524    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1525        let path = path.unwrap_or("screenshot.png");
1526
1527        match tab.capture_screenshot(
1528            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1529            None,
1530            None,
1531            true,
1532        ) {
1533            Ok(data) => {
1534                if let Err(e) = std::fs::write(path, &data) {
1535                    return StepResult {
1536                        name: format!("[screenshot] {path}"),
1537                        status: StepStatus::Failed,
1538                        message: format!("failed to write screenshot: {e}"),
1539                    };
1540                }
1541                StepResult {
1542                    name: format!("[screenshot] {path}"),
1543                    status: StepStatus::Passed,
1544                    message: format!("saved to {path}"),
1545                }
1546            }
1547            Err(e) => StepResult {
1548                name: format!("[screenshot] {path}"),
1549                status: StepStatus::Failed,
1550                message: format!("screenshot failed: {e}"),
1551            },
1552        }
1553    }
1554
1555    /// Runs an A2A agent step.
1556    #[allow(clippy::literal_string_with_formatting_args)]
1557    fn run_agent(
1558        &self,
1559        agent_name: &str,
1560        task: &str,
1561        definition: Option<&str>,
1562        _test_endpoint: Option<&str>,
1563    ) -> StepResult {
1564        // If a definition is specified, look up the task template
1565        let resolved_task = if let Some(def_name) = definition {
1566            if let Some(def) = self.definitions.get(def_name) {
1567                let tmpl = def.task_template.as_deref().unwrap_or(task);
1568                tmpl.replace("{task}", task)
1569            } else {
1570                return StepResult {
1571                    name: format!("[agent] {def_name}"),
1572                    status: StepStatus::Failed,
1573                    message: format!("definition '{def_name}' not found"),
1574                };
1575            }
1576        } else {
1577            task.to_owned()
1578        };
1579
1580        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1581    }
1582
1583    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1584        let Some(ep) = self.endpoints.get(agent_name) else {
1585            return StepResult {
1586                name: format!("[agent] {display_name}"),
1587                status: StepStatus::Failed,
1588                message: format!("agent endpoint '{agent_name}' not found"),
1589            };
1590        };
1591
1592        if ep.url.is_empty() {
1593            return StepResult {
1594                name: format!("[agent] {display_name}"),
1595                status: StepStatus::Failed,
1596                message: format!("agent endpoint '{agent_name}' has no URL"),
1597            };
1598        }
1599
1600        eprintln!("      → agent {agent_name}: {task}");
1601
1602        let url = ep.url.clone();
1603        let client = A2aClient::new(&url, self.timeout);
1604        let task_clone = task.to_owned();
1605
1606        let response = std::thread::spawn(move || {
1607            let rt = tokio::runtime::Builder::new_current_thread()
1608                .enable_all()
1609                .build()
1610                .unwrap();
1611            rt.block_on(client.send_task(&task_clone))
1612        })
1613        .join()
1614        .unwrap();
1615
1616        // Record the flat-cost call
1617        self.usage.record_flat_call(agent_name, ep);
1618
1619        match response {
1620            Ok(text) => {
1621                let clean = text.trim().to_owned();
1622                let lower = clean.to_lowercase();
1623                if lower.starts_with("pass") {
1624                    StepResult {
1625                        name: format!("[agent] {display_name}"),
1626                        status: StepStatus::Passed,
1627                        message: format!("PASS: {clean}"),
1628                    }
1629                } else if lower.starts_with("fail") {
1630                    StepResult {
1631                        name: format!("[agent] {display_name}"),
1632                        status: StepStatus::Failed,
1633                        message: clean,
1634                    }
1635                } else {
1636                    StepResult {
1637                        name: format!("[agent] {display_name}"),
1638                        status: StepStatus::Passed,
1639                        message: format!("response: {clean}"),
1640                    }
1641                }
1642            }
1643            Err(e) => StepResult {
1644                name: format!("[agent] {display_name}"),
1645                status: StepStatus::Failed,
1646                message: format!("agent call failed: {e}"),
1647            },
1648        }
1649    }
1650
1651    /// Runs an MCP tool call step.
1652    fn run_mcp(
1653        &self,
1654        server_name: &str,
1655        tool_name: &str,
1656        args: Option<&serde_json::Value>,
1657    ) -> StepResult {
1658        let Some(ep) = self.endpoints.get(server_name) else {
1659            return StepResult {
1660                name: format!("[mcp] {server_name}:{tool_name}"),
1661                status: StepStatus::Failed,
1662                message: format!("MCP server endpoint '{server_name}' not found"),
1663            };
1664        };
1665
1666        let cmd = ep.command.as_deref().unwrap_or("");
1667        if cmd.is_empty() {
1668            return StepResult {
1669                name: format!("[mcp] {server_name}:{tool_name}"),
1670                status: StepStatus::Failed,
1671                message: format!("MCP server '{server_name}' has no command configured"),
1672            };
1673        }
1674
1675        eprintln!("      → mcp {server_name} {tool_name}");
1676
1677        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1678
1679        let command = cmd.to_owned();
1680        let args_vec = ep.args.clone();
1681        let tool = tool_name.to_owned();
1682
1683        let response = std::thread::spawn(move || {
1684            let mut mcp_client =
1685                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1686            mcp_client
1687                .call_tool(&tool, &args_val)
1688                .map_err(|e| e.to_string())
1689        })
1690        .join()
1691        .unwrap();
1692
1693        // Record the flat-cost call
1694        self.usage.record_flat_call(server_name, ep);
1695
1696        match response {
1697            Ok(result) => {
1698                if result.isError {
1699                    StepResult {
1700                        name: format!("[mcp] {server_name}:{tool_name}"),
1701                        status: StepStatus::Failed,
1702                        message: result.to_string(),
1703                    }
1704                } else {
1705                    StepResult {
1706                        name: format!("[mcp] {server_name}:{tool_name}"),
1707                        status: StepStatus::Passed,
1708                        message: result.to_string(),
1709                    }
1710                }
1711            }
1712            Err(e) => StepResult {
1713                name: format!("[mcp] {server_name}:{tool_name}"),
1714                status: StepStatus::Failed,
1715                message: format!("MCP call failed: {e}"),
1716            },
1717        }
1718    }
1719
1720    // ── helpers ──────────────────────────────────────────────────────────
1721
1722    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1723    /// the runner's default LLM config for any unset fields.
1724    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1725        LlmConfig {
1726            url: if endpoint.url.is_empty() {
1727                self.llm.url.clone()
1728            } else {
1729                endpoint.url.clone()
1730            },
1731            model: endpoint
1732                .model
1733                .clone()
1734                .unwrap_or_else(|| self.llm.model.clone()),
1735            api_key: endpoint
1736                .api_key
1737                .clone()
1738                .or_else(|| self.llm.api_key.clone()),
1739            headers: if endpoint.headers.is_empty() {
1740                self.llm.headers.clone()
1741            } else {
1742                endpoint.headers.clone()
1743            },
1744            timeout: self.llm.timeout,
1745            temperature: self.llm.temperature,
1746            thinking: self.llm.thinking,
1747            model_params: self.llm.model_params.clone(),
1748        }
1749    }
1750
1751    /// Resolves a CSS selector for the target element. Uses the explicit
1752    /// `selector` if provided, otherwise asks the LLM to find the element
1753    /// from the natural language `target` description and page DOM.
1754    ///
1755    /// LLM responses are sanitized and verified against the live page: a
1756    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
1757    /// immediately with the raw LLM output, and a selector that matches
1758    /// nothing triggers one retry with feedback before failing.
1759    #[allow(clippy::too_many_lines)]
1760    fn resolve_selector(
1761        &self,
1762        css_override: Option<&str>,
1763        target: &str,
1764        step_endpoint: Option<&str>,
1765        test_endpoint: Option<&str>,
1766        tab: &Tab,
1767    ) -> Result<String, String> {
1768        if let Some(explicit) = css_override {
1769            return Ok(explicit.to_owned());
1770        }
1771
1772        let dom_info = extract_dom_info(tab)?;
1773        let page_content = get_page_text(tab);
1774
1775        let system = concat!(
1776            "You are a browser automation selector generator. ",
1777            "Given a web page's content and interactive elements, ",
1778            "return ONLY the best CSS selector for the described element. ",
1779            "Output nothing except the CSS selector. ",
1780            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
1781            "[name=\"...\"], tag.class, tag. ",
1782            "Never output explanations, markdown, or extra text."
1783        );
1784
1785        let user = format!(
1786            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
1787            page_content.url,
1788            page_content.title,
1789            truncate(&page_content.body_text, 4000),
1790            dom_info,
1791            target,
1792        );
1793
1794        let retry_user = format!(
1795            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
1796            "The selector must match at least one element currently present on the page.",
1797            page_content.url,
1798            page_content.title,
1799            truncate(&page_content.body_text, 4000),
1800            dom_info,
1801            target,
1802        );
1803
1804        eprintln!("      LLM targeting: {target}");
1805
1806        let endpoint = self
1807            .endpoints
1808            .resolve(step_endpoint.or(test_endpoint), TaskType::Targeting);
1809        let llm = self.build_llm_for_endpoint(endpoint);
1810        let usage = Arc::clone(&self.usage);
1811        let endpoint_name = endpoint.name.clone();
1812        let endpoint_clone = endpoint.clone();
1813        let sys = system.to_owned();
1814
1815        let call_llm = |prompt: &str| {
1816            let llm = llm.clone();
1817            let sys = sys.clone();
1818            let prompt = prompt.to_owned();
1819            std::thread::spawn(move || {
1820                let rt = tokio::runtime::Builder::new_current_thread()
1821                    .enable_all()
1822                    .build()
1823                    .unwrap();
1824                rt.block_on(llm_chat_with_usage(&llm, &sys, &prompt))
1825            })
1826            .join()
1827            .unwrap()
1828        };
1829
1830        let first = call_llm(&user);
1831        let lr = match first {
1832            Ok(lr) => lr,
1833            Err(e) => {
1834                return Err(format!("LLM element targeting failed: {e}"));
1835            }
1836        };
1837        usage.record_llm_call(
1838            &endpoint_name,
1839            &endpoint_clone,
1840            lr.usage.prompt_tokens,
1841            lr.usage.completion_tokens,
1842        );
1843        let clean = sanitize_selector(&lr.content);
1844        eprintln!("      resolved selector: {clean}");
1845
1846        if selector_is_useless(&clean) {
1847            return Err(format!(
1848                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
1849                raw = lr.content.trim(),
1850            ));
1851        }
1852        if let Err(reason) = validate_selector(&clean) {
1853            return Err(format!(
1854                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
1855                raw = lr.content.trim(),
1856            ));
1857        }
1858        if !selector_matches(tab, &clean).unwrap_or(false) {
1859            // One retry with feedback: flaky models occasionally invent a
1860            // selector that does not exist on the page.
1861            eprintln!(
1862                "      selector {clean} matches nothing — retrying LLM targeting with feedback"
1863            );
1864            let second = call_llm(&retry_user);
1865            let lr2 = match second {
1866                Ok(lr2) => lr2,
1867                Err(e) => {
1868                    return Err(format!(
1869                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
1870                    ));
1871                }
1872            };
1873            usage.record_llm_call(
1874                &endpoint_name,
1875                &endpoint_clone,
1876                lr2.usage.prompt_tokens,
1877                lr2.usage.completion_tokens,
1878            );
1879            let clean2 = sanitize_selector(&lr2.content);
1880            eprintln!("      resolved selector (retry): {clean2}");
1881            if selector_is_useless(&clean2) {
1882                return Err(format!(
1883                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
1884                    raw = lr2.content.trim(),
1885                    excerpt = truncate(&page_content.body_text, 300),
1886                ));
1887            }
1888            if !selector_matches(tab, &clean2).unwrap_or(false) {
1889                return Err(format!(
1890                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
1891                ));
1892            }
1893            return Ok(clean2);
1894        }
1895
1896        Ok(clean)
1897    }
1898}
1899
1900/// Evaluates a JS expression that is expected to return a boolean.
1901fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
1902    tab.evaluate(js, false)
1903        .map_err(|e| format!("evaluate failed: {e}"))?
1904        .value
1905        .and_then(|v| v.as_bool())
1906        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
1907}
1908
1909/// Checks whether a CSS selector matches at least one current element.
1910fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
1911    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
1912}
1913
1914// ── Free helper functions ──────────────────────────────────────────────
1915
1916fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
1917    let name = format!("[navigate] {full_url}");
1918    match tab.navigate_to(full_url) {
1919        Ok(_) => {
1920            let _ = tab.wait_until_navigated();
1921            StepResult {
1922                name,
1923                status: StepStatus::Passed,
1924                message: format!("navigated to {full_url}"),
1925            }
1926        }
1927        Err(e) => StepResult {
1928            name,
1929            status: StepStatus::Failed,
1930            message: format!("navigation failed: {e}"),
1931        },
1932    }
1933}
1934
1935fn extract_dom_info(tab: &Tab) -> Result<String, String> {
1936    let result = tab
1937        .evaluate(DOM_EXTRACT_JS, false)
1938        .map_err(|e| format!("DOM extraction failed: {e}"))?;
1939
1940    let json_str = result
1941        .value
1942        .as_ref()
1943        .and_then(|v| v.as_str())
1944        .unwrap_or("[]");
1945
1946    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
1947
1948    if elements.is_empty() {
1949        return Ok("(no interactive elements found)".to_owned());
1950    }
1951
1952    Ok(elements.join("\n"))
1953}
1954
1955fn get_page_text(tab: &Tab) -> PageContent {
1956    let url = tab.get_url();
1957
1958    let title = tab
1959        .evaluate("document.title", false)
1960        .ok()
1961        .and_then(|r| r.value)
1962        .and_then(|v| v.as_str().map(String::from))
1963        .unwrap_or_else(|| "unknown".to_owned());
1964
1965    let body_text = tab
1966        .evaluate(
1967            "document.body ? document.body.innerText : document.documentElement.innerText",
1968            false,
1969        )
1970        .ok()
1971        .and_then(|r| r.value)
1972        .and_then(|v| v.as_str().map(String::from))
1973        .unwrap_or_default();
1974
1975    PageContent {
1976        url,
1977        title,
1978        body_text: truncate(&body_text, 8000),
1979    }
1980}
1981
1982fn resolve_url(url: &str, base_url: &str) -> String {
1983    if url.starts_with("http://") || url.starts_with("https://") {
1984        return url.to_owned();
1985    }
1986    let base = base_url.trim_end_matches('/');
1987    if url.starts_with('/') {
1988        format!("{base}{url}")
1989    } else {
1990        format!("{base}/{url}")
1991    }
1992}
1993
1994/// Human-readable label for a step, used when steps are skipped after an
1995/// earlier failure.
1996fn step_label(step: &TestStep) -> String {
1997    match step {
1998        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
1999        TestStep::Click { target, .. } => format!("[click] {target}"),
2000        TestStep::Type { target, .. } => format!("[type] {target}"),
2001        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2002        TestStep::Assert {
2003            definition,
2004            preset,
2005            prompt,
2006            ..
2007        } => definition.as_ref().map_or_else(
2008            || {
2009                preset.as_ref().map_or_else(
2010                    || {
2011                        prompt.as_ref().map_or_else(
2012                            || "[assert]".to_owned(),
2013                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2014                        )
2015                    },
2016                    |p| format!("[assert] {p}"),
2017                )
2018            },
2019            |d| format!("[assert] {d}"),
2020        ),
2021        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2022        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2023        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2024    }
2025}
2026
2027/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2028#[must_use]
2029const fn step_kind_label(step: &TestStep) -> &'static str {
2030    match step {
2031        TestStep::Navigate { .. } => "navigate",
2032        TestStep::Click { .. } => "click",
2033        TestStep::Type { .. } => "type",
2034        TestStep::Wait { .. } => "wait",
2035        TestStep::Assert { .. } => "assert",
2036        TestStep::Screenshot { .. } => "screenshot",
2037        TestStep::Agent { .. } => "agent",
2038        TestStep::Mcp { .. } => "mcp",
2039    }
2040}
2041
2042// ── Support types ──────────────────────────────────────────────────────
2043
2044#[derive(Default)]
2045struct TestRunResult {
2046    passed: u32,
2047    failed: u32,
2048    skipped: u32,
2049    total: u32,
2050    details: Vec<StepResult>,
2051}
2052
2053struct PageContent {
2054    url: String,
2055    title: String,
2056    body_text: String,
2057}