Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// One detected layout defect (`layout_no_issues` preset).
26#[derive(Debug, serde::Deserialize)]
27struct LayoutIssue {
28    #[serde(rename = "type")]
29    issue_type: String,
30    element: String,
31    detail: String,
32}
33
34/// In-browser DOM layout scan for `layout_no_issues`.
35///
36/// Geometry-only checks (no LLM, no pixels):
37/// 1. page horizontal overflow (`scrollWidth` > viewport width);
38/// 2. elements outside the viewport that scrolling cannot reveal
39///    (fixed elements off-screen, left/negative overflow, right-edge
40///    overflow beyond the horizontally scrollable area, and bottom
41///    overflow on a page that cannot scroll down) — below-the-fold
42///    content on a tall scrollable page is normal flow, NOT a defect;
43/// 3. text clipped by `overflow: hidden` containers whose content
44///    is measurably larger than the box;
45/// 4. interactive elements (buttons/links/inputs) whose center point
46///    is covered by a different element that would intercept the click.
47///
48/// Elements whose class matches a configured ignore prefix (default:
49/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
50/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
51/// skipped — they are intentionally 1x1 / off-screen. The prefix list
52/// is injected at `__IGNORE_CLASSES__` from
53/// [`ScenarioConfig::layout_ignore_classes`].
54///
55/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
56/// sticky headers) is excluded by the position/relation filters.
57const LAYOUT_SCAN_JS: &str = r#"
58(() => {
59  const issues = [];
60  const push = (type, el, detail) => {
61    if (issues.length >= 30) return;
62    let element = el.tagName.toLowerCase();
63    if (el.id) element += '#' + el.id;
64    else if (typeof el.className === 'string' && el.className.trim())
65      element += '.' + el.className.trim().split(/\s+/).join('.');
66    issues.push({ type, element, detail: String(detail).slice(0, 220) });
67  };
68  const vw = document.documentElement.clientWidth || window.innerWidth;
69  const vh = document.documentElement.clientHeight || window.innerHeight;
70  if (!vw || !vh) return JSON.stringify(issues);
71  const de = document.documentElement;
72  // 1. Page-level horizontal overflow.
73  if (de.scrollWidth > vw + 2)
74    push('page-overflow-x', de,
75      'page scrollWidth ' + de.scrollWidth + ' exceeds viewport width ' + vw);
76  // Class-name prefixes to skip (injected; default = Angular CDK
77  // screen-reader helpers, which are intentionally 1x1 / off-screen).
78  const ignorePrefixes = __IGNORE_CLASSES__;
79  const isIgnored = (el) => {
80    if (typeof el.className !== 'string' || !el.className.trim()) return false;
81    const classes = el.className.trim().split(/\s+/);
82    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
83  };
84  const vScrollable = de.scrollHeight > vh + 2;
85  const hScrollable = de.scrollWidth > vw + 2;
86  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
87  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
88  const hasContent = (el) =>
89    ((el.textContent || '').trim().length > 0) ||
90    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
91  // 2. Elements outside the viewport that scrolling cannot reveal.
92  for (const el of all) {
93    const cs = getComputedStyle(el);
94    if (!visible(cs)) continue;
95    if (isIgnored(el)) continue;
96    const r = el.getBoundingClientRect();
97    if (r.width < 2 || r.height < 2) continue;
98    if (!hasContent(el) && el.children.length === 0) continue;
99    if (cs.position === 'fixed') {
100      // Fixed elements never move with the scroll: any edge outside the
101      // viewport is unreachable content and therefore a defect.
102      const overTop = -r.top;
103      const overLeft = -r.left;
104      const overRight = r.right - vw;
105      const overBottom = r.bottom - vh;
106      if (overTop > 2 || overLeft > 2 || overRight > 2 || overBottom > 2) {
107        let where = '';
108        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
109        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
110        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
111        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
112        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
113        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
114        push('element-out-of-viewport', el,
115          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
116      }
117      continue;
118    }
119    if (cs.position === 'sticky') continue;
120    // Negative left overflow cannot be reached by scrolling (scrollLeft
121    // never goes below 0).
122    if (r.left < -2) {
123      push('element-out-of-viewport', el,
124        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
125      continue;
126    }
127    // Negative top with the page at the top means the element sits above
128    // the document origin — also unreachable.
129    if (r.top < -2 && de.scrollTop <= 2) {
130      push('element-out-of-viewport', el,
131        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
132      continue;
133    }
134    const overRight = r.right - vw;
135    // Right-edge overflow is only a defect when the page cannot scroll
136    // horizontally to reveal it (or the element sticks out past the
137    // scrollable content width itself).
138    if (overRight > 2 && (!hScrollable || r.right > de.scrollWidth + 2)) {
139      push('element-out-of-viewport', el,
140        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
141      continue;
142    }
143    // Below-the-fold content on a scrollable page is normal (tall landing
144    // pages); only flag bottom overflow the user can never scroll to.
145    const overBottom = r.bottom - vh;
146    if (overBottom > 2 && (!vScrollable || r.bottom > de.scrollHeight + 2)) {
147      push('element-out-of-viewport', el,
148        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
149    }
150  }
151  // 3. Text clipped by overflow:hidden containers.
152  for (const el of all) {
153    if (isIgnored(el)) continue;
154    const cs = getComputedStyle(el);
155    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
156    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
157    if (!(el.textContent || '').trim()) continue;
158    push('text-clipped', el,
159      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
160      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
161  }
162  // 4. Interactive elements covered by a different element.
163  const interactive =
164    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
165  const targets = document.querySelectorAll(interactive);
166  for (const el of targets) {
167    if (isIgnored(el)) continue;
168    const r = el.getBoundingClientRect();
169    if (r.width < 6 || r.height < 6) continue;
170    const cs = getComputedStyle(el);
171    if (!visible(cs)) continue;
172    const cx = r.left + r.width / 2;
173    const cy = r.top + r.height / 2;
174    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
175    const top = document.elementFromPoint(cx, cy);
176    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
177    if (isIgnored(top)) continue;
178    const tcs = getComputedStyle(top);
179    if (!visible(tcs)) continue;
180    if (tcs.pointerEvents === 'none') continue;
181    const tr = top.getBoundingClientRect();
182    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
183    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
184      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
185    push('element-overlap', el,
186      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
187      ' is covered by <' + tname + '>');
188  }
189  return JSON.stringify(issues);
190})()
191"#;
192
193/// How long the CDP connection stays open after the browser goes quiet.
194///
195/// `headless_chrome` ships a 30s default and tears down the entire connection
196/// when no traffic arrives for that long; a run must own its connection for
197/// its full duration instead.
198const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
199
200/// Executes a [`Scenario`] against a real browser with optional LLM
201/// assistance for element targeting and assertions.
202pub struct ScenarioRunner {
203    config: ScenarioConfig,
204    definitions: HashMap<String, AssertDefinition>,
205    llm: LlmConfig,
206    timeout: Duration,
207    viewport_width: u32,
208    viewport_height: u32,
209    /// The viewport currently applied in the browser (CDP emulation).
210    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
211    applied_viewport: std::cell::Cell<(u32, u32)>,
212    endpoints: EndpointRegistry,
213    usage: Arc<UsageTracker>,
214    budgets: BudgetTracker,
215    /// Directory for failure artifacts (screenshots).
216    artifacts_dir: PathBuf,
217    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
218    reporter: Arc<Reporter>,
219    /// The test + step index currently executing, for LLM-call events.
220    current_step: std::cell::RefCell<Option<(String, u32)>>,
221}
222
223/// Aggregated results from a scenario run.
224#[derive(Debug, Default)]
225pub struct RunReport {
226    /// Number of tests that passed.
227    pub tests_passed: u32,
228    /// Number of tests that failed.
229    pub tests_failed: u32,
230    /// Number of steps that passed.
231    pub passed: u32,
232    /// Number of steps that failed.
233    pub failed: u32,
234    /// Number of steps that were skipped.
235    pub skipped: u32,
236    /// Per-step details.
237    pub details: Vec<StepResult>,
238}
239
240/// Result of a single step execution.
241#[derive(Debug)]
242pub struct StepResult {
243    /// The step name.
244    pub name: String,
245    /// Whether the step passed, failed, or was skipped.
246    pub status: StepStatus,
247    /// Human-readable result message.
248    pub message: String,
249}
250
251/// Outcome for a single step.
252pub use crate::events::StepStatus;
253
254/// Predefined assertion preset definition.
255struct AssertPreset {
256    name: &'static str,
257    system: &'static str,
258    user_template: &'static str,
259}
260
261/// Built-in assertion presets.
262#[allow(clippy::literal_string_with_formatting_args)]
263const ASSERTION_PRESETS: &[AssertPreset] = &[
264    AssertPreset {
265        name: "no_error_on_page",
266        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
267        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
268    },
269    AssertPreset {
270        name: "text_visible",
271        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
272        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
273    },
274    AssertPreset {
275        name: "element_exists",
276        system: "You are a QA tester. Check if a described UI element exists on a web page.",
277        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
278    },
279    AssertPreset {
280        name: "visual_no_issues",
281        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
282        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
283    },
284    AssertPreset {
285        name: "visual_no_overlaps",
286        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
287        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
288    },
289    AssertPreset {
290        name: "visual_text_visible",
291        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
292        user_template: "Inspect the attached screenshot and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
293    },
294    AssertPreset {
295        name: "layout_no_issues",
296        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
297        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
298    },
299];
300
301impl ScenarioRunner {
302    /// Creates a new runner with the given scenario configuration and
303    /// assertion definitions.
304    #[must_use]
305    #[allow(clippy::needless_pass_by_value)]
306    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
307        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
308    }
309
310    /// Creates a runner that reports run events through the given reporter.
311    #[must_use]
312    #[allow(clippy::needless_pass_by_value)]
313    pub fn with_reporter(
314        scenario_config: ScenarioConfig,
315        definitions: Vec<AssertDefinition>,
316        reporter: Arc<Reporter>,
317    ) -> Self {
318        let llm = LlmConfig {
319            url: scenario_config
320                .llm_url
321                .clone()
322                .unwrap_or_else(crate::llm_base_url),
323            model: scenario_config
324                .llm_model
325                .clone()
326                .unwrap_or_else(crate::llm_model),
327            api_key: scenario_config
328                .llm_api_key
329                .clone()
330                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
331            headers: if scenario_config.llm_headers.is_empty() {
332                crate::parse_headers_env()
333            } else {
334                scenario_config.llm_headers.clone()
335            },
336            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
337            temperature: scenario_config.temperature,
338            thinking: scenario_config.thinking,
339            model_params: scenario_config.model_params.clone(),
340            max_attempts: crate::default_llm_attempts(),
341            provider: crate::scenario::Provider::Openai,
342            deployment: None,
343            api_version: None,
344            auth: crate::scenario::AuthConfig::default(),
345            header_commands: std::collections::HashMap::new(),
346            aws: crate::scenario::AwsConfig::default(),
347        };
348        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
349        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
350        let defs_map: HashMap<String, AssertDefinition> = definitions
351            .into_iter()
352            .map(|d| (d.name.clone(), d))
353            .collect();
354
355        Self {
356            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
357            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
358            viewport_height: scenario_config.viewport_height.unwrap_or(720),
359            applied_viewport: std::cell::Cell::new((0, 0)),
360            config: scenario_config.clone(),
361            definitions: defs_map,
362            llm,
363            endpoints,
364            usage: Arc::new(UsageTracker::new()),
365            budgets,
366            artifacts_dir: PathBuf::from(
367                scenario_config
368                    .artifacts_dir
369                    .unwrap_or_else(|| "artifacts".to_owned()),
370            ),
371            reporter,
372            current_step: std::cell::RefCell::new(None),
373        }
374    }
375
376    /// Emits an event; a sink failure degrades to a console warning so a
377    /// broken log file can never mask the run itself.
378    fn emit_event(&self, event: &TestEvent) {
379        if let Err(err) = self.reporter.emit(event) {
380            use std::io::Write as _;
381            let mut out = std::io::stderr().lock();
382            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
383        }
384    }
385
386    /// Returns a clone of the [`UsageTracker`] for reporting.
387    #[must_use]
388    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
389        Arc::clone(&self.usage)
390    }
391
392    /// Returns a reference to the [`BudgetTracker`].
393    #[must_use]
394    pub const fn budget_tracker(&self) -> &BudgetTracker {
395        &self.budgets
396    }
397
398    /// Executes all test groups in the scenario and returns a report.
399    ///
400    /// # Errors
401    ///
402    /// Returns an error if the browser fails to launch.
403    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
404    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
405        let mut report = RunReport::default();
406
407        if tests.is_empty() {
408            self.reporter.warn("No tests defined in scenario.");
409            self.emit_event(&TestEvent::RunFinished {
410                tests_passed: 0,
411                tests_failed: 0,
412                steps_passed: 0,
413                steps_failed: 0,
414                steps_skipped: 0,
415                total_cost: 0.0,
416                total_tokens: 0,
417                total_calls: 0,
418            });
419            return Ok(report);
420        }
421
422        self.emit_event(&TestEvent::RunStarted {
423            total_tests: tests.len() as u32,
424        });
425
426        let browser_headless = self.config.browser_headless.unwrap_or(true);
427
428        let launch_opts = LaunchOptions {
429            headless: browser_headless,
430            window_size: Some((self.viewport_width, self.viewport_height)),
431            sandbox: false,
432            // headless_chrome defaults this to 30s and shuts down the whole CDP
433            // connection when no messages arrive for that long. A scenario can
434            // easily exceed 30s of browser silence (slow LLM targeting/assertion
435            // calls, page waits, budget checks between steps), after which every
436            // remaining step fails with "Unable to make method calls because
437            // underlying connection is closed" — one quiet gap kills the run.
438            // Open-ended scenarios must own the connection for their full
439            // duration, so keep it alive for 6 hours.
440            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
441            ..LaunchOptions::default()
442        };
443
444        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
445        let tab = browser.new_tab().context("failed to open browser tab")?;
446        let _ = tab.set_default_timeout(self.timeout);
447
448        // Start MCP server if configured
449        #[cfg(feature = "mcp-server")]
450        if let Some(ref mcp_cfg) = self.config.mcp_server {
451            if mcp_cfg.enabled {
452                let port = mcp_cfg.port;
453                std::thread::spawn(move || {
454                    let _ = crate::mcp_server::start_mcp_server(port);
455                });
456            }
457        }
458        #[cfg(not(feature = "mcp-server"))]
459        if let Some(mcp_cfg) = &self.config.mcp_server {
460            if mcp_cfg.enabled {
461                self.reporter
462                    .warn("MCP server configured but 'mcp-server' feature not enabled");
463            }
464        }
465
466        // Start A2A agent server if configured
467        #[cfg(feature = "a2a-server")]
468        if let Some(ref a2a_cfg) = self.config.a2a_server {
469            if a2a_cfg.enabled {
470                let port = a2a_cfg.port;
471                tokio::spawn(crate::a2a_server::start_a2a_server(port));
472            }
473        }
474        #[cfg(not(feature = "a2a-server"))]
475        if let Some(a2a_cfg) = &self.config.a2a_server {
476            if a2a_cfg.enabled {
477                self.reporter
478                    .warn("A2A server configured but 'a2a-server' feature not enabled");
479            }
480        }
481
482        for test in tests {
483            self.emit_event(&TestEvent::TestStarted {
484                test: test.name.clone(),
485            });
486
487            self.usage.reset_per_test();
488
489            let test_started = Instant::now();
490            let test_result = self.run_test(test, &tab);
491            let duration_ms = test_started.elapsed().as_millis() as u64;
492            let usage = self.usage.current_test_snapshot();
493            self.usage.commit_test(&test.name);
494
495            self.emit_event(&TestEvent::TestFinished {
496                test: test.name.clone(),
497                passed: test_result.passed,
498                failed: test_result.failed,
499                skipped: test_result.skipped,
500                duration_ms,
501                cost: usage.total_cost,
502                tokens: usage.total_tokens,
503                calls: usage.total_calls,
504            });
505
506            if test_result.failed == 0 && test_result.total > 0 {
507                report.tests_passed += 1;
508            } else if test_result.total > 0 {
509                report.tests_failed += 1;
510            }
511
512            report.passed += test_result.passed;
513            report.failed += test_result.failed;
514            report.skipped += test_result.skipped;
515            report.details.extend(test_result.details);
516        }
517
518        let global = self.usage.global_snapshot();
519        self.emit_event(&TestEvent::RunFinished {
520            tests_passed: report.tests_passed,
521            tests_failed: report.tests_failed,
522            steps_passed: report.passed,
523            steps_failed: report.failed,
524            steps_skipped: report.skipped,
525            total_cost: global.total_cost,
526            total_tokens: global.total_tokens,
527            total_calls: global.total_calls,
528        });
529
530        Ok(report)
531    }
532
533    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
534    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
535        let base_url = test
536            .base_url
537            .clone()
538            .or_else(|| self.config.base_url.clone())
539            .unwrap_or_else(crate::base_url);
540
541        // Per-test viewport override: switch the browser via CDP
542        // device-metrics emulation before this test runs.
543        let vw = test.viewport_width.unwrap_or(self.viewport_width);
544        let vh = test.viewport_height.unwrap_or(self.viewport_height);
545        if self.applied_viewport.get() != (vw, vh) {
546            self.apply_viewport(tab, vw, vh);
547            self.applied_viewport.set((vw, vh));
548        }
549
550        // Per-test isolation: every test starts from its own start_url
551        // (unless auto_navigate is disabled), so a test never inherits the
552        // previous test's page state.
553        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
554
555        let start_url = test
556            .start_url
557            .clone()
558            .or_else(|| self.config.start_url.clone())
559            .unwrap_or_else(|| "/dashboard".to_owned());
560
561        if auto_navigate {
562            let full_url = resolve_url(&start_url, &base_url);
563            self.reporter.debug(format!("auto-navigate: {full_url}"));
564            let _ = tab.navigate_to(&full_url);
565            let _ = tab.wait_until_navigated();
566            std::thread::sleep(Duration::from_secs(4));
567        }
568
569        let mut result = TestRunResult::default();
570
571        for (step_index, step) in test.steps.iter().enumerate() {
572            result.total += 1;
573
574            let wait_ms = match step {
575                TestStep::Navigate { wait_after_ms, .. }
576                | TestStep::Click { wait_after_ms, .. }
577                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
578                _ => None,
579            };
580
581            self.current_step
582                .replace(Some((test.name.clone(), step_index as u32)));
583            self.emit_event(&TestEvent::StepStarted {
584                test: test.name.clone(),
585                index: step_index as u32,
586                label: step_label(step),
587            });
588            let step_started = Instant::now();
589
590            let mut step_result = match step {
591                TestStep::Navigate { url, .. } => {
592                    let full_url = resolve_url(url, &base_url);
593                    run_navigate_step(&full_url, tab)
594                }
595                TestStep::Click {
596                    target,
597                    selector,
598                    endpoint,
599                    idempotent,
600                    ..
601                } => self.run_click(
602                    target,
603                    selector.as_deref(),
604                    endpoint.as_deref(),
605                    test.endpoint.as_deref(),
606                    *idempotent,
607                    tab,
608                ),
609                TestStep::Type {
610                    target,
611                    text,
612                    selector,
613                    endpoint,
614                    idempotent,
615                    ..
616                } => self.run_type(
617                    target,
618                    text,
619                    selector.as_deref(),
620                    endpoint.as_deref(),
621                    test.endpoint.as_deref(),
622                    *idempotent,
623                    tab,
624                ),
625                TestStep::Wait {
626                    target,
627                    selector,
628                    text,
629                    timeout_ms,
630                    endpoint,
631                    idempotent,
632                } => self.run_wait(
633                    target,
634                    selector.as_deref(),
635                    text.as_deref(),
636                    *timeout_ms,
637                    endpoint.as_deref(),
638                    test.endpoint.as_deref(),
639                    *idempotent,
640                    tab,
641                ),
642                TestStep::Assert {
643                    definition,
644                    preset,
645                    prompt,
646                    assert_text,
647                    endpoint,
648                    screenshot,
649                } => self.run_assert(
650                    definition.as_deref(),
651                    preset.as_deref(),
652                    prompt.as_deref(),
653                    assert_text.as_deref(),
654                    *screenshot,
655                    endpoint.as_deref(),
656                    test.endpoint.as_deref(),
657                    tab,
658                ),
659                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
660                TestStep::Agent {
661                    agent,
662                    task,
663                    definition,
664                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
665                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
666            };
667
668            // Failure diagnostics: capture the page state and a screenshot so
669            // CI logs say WHAT the page looked like when the step failed,
670            // instead of a bare "timed out: The event waited for never came".
671            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
672                let state = diagnostics::capture(tab);
673                let screenshot = diagnostics::save_screenshot(
674                    tab,
675                    &self.artifacts_dir,
676                    &test.name,
677                    &test.name,
678                    step_index,
679                    step_kind_label(step),
680                );
681                step_result.message = format!(
682                    "{base} — {excerpt}",
683                    base = step_result.message,
684                    excerpt = diagnostics::inline_excerpt(&state),
685                );
686                (Some(diagnostics::full_context(&state)), screenshot)
687            } else {
688                (None, None)
689            };
690
691            let duration_ms = step_started.elapsed().as_millis() as u64;
692            self.emit_event(&TestEvent::StepFinished {
693                test: test.name.clone(),
694                index: step_index as u32,
695                label: step_result.name.clone(),
696                status: step_result.status,
697                duration_ms,
698                message: step_result.message.clone(),
699                diagnostics: diagnostics_block,
700                screenshot: screenshot_path,
701            });
702            self.current_step.replace(None);
703
704            match step_result.status {
705                StepStatus::Passed => result.passed += 1,
706                StepStatus::Failed => result.failed += 1,
707                StepStatus::Skipped => result.skipped += 1,
708            }
709
710            // Fail fast: the first failed step ends the test and the
711            // remaining steps are reported as skipped (no LLM budget is
712            // burned asserting against a page that is already known broken).
713            if step_result.status == StepStatus::Failed
714                && !self.config.continue_on_failure
715                && step_index + 1 < test.steps.len()
716            {
717                self.emit_event(&TestEvent::Warning {
718                    message: format!(
719                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
720                        test.steps.len() - step_index - 1
721                    ),
722                });
723                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
724                    let skipped_index = step_index + 1 + offset;
725                    let label = step_label(skipped);
726                    result.total += 1;
727                    result.skipped += 1;
728                    self.emit_event(&TestEvent::StepStarted {
729                        test: test.name.clone(),
730                        index: skipped_index as u32,
731                        label: label.clone(),
732                    });
733                    self.emit_event(&TestEvent::StepFinished {
734                        test: test.name.clone(),
735                        index: skipped_index as u32,
736                        label,
737                        status: StepStatus::Skipped,
738                        duration_ms: 0,
739                        message: "skipped: previous step failed".into(),
740                        diagnostics: None,
741                        screenshot: None,
742                    });
743                    result.details.push(StepResult {
744                        name: step_label(skipped),
745                        status: StepStatus::Skipped,
746                        message: "skipped: previous step failed".into(),
747                    });
748                }
749                result.details.push(step_result);
750                return result;
751            }
752
753            // Check per-test budget after each step
754            let test_usage = self.usage.current_test_snapshot();
755            let global_usage = self.usage.global_snapshot();
756            let budget_status = self.budgets.check_all(
757                &test.name,
758                &test_usage,
759                &global_usage,
760                test.budget.as_ref(),
761            );
762            match budget_status {
763                BudgetStatus::HardExceeded { message, .. } => {
764                    self.emit_event(&TestEvent::Warning {
765                        message: format!("budget exceeded: {message}"),
766                    });
767                    result.details.push(StepResult {
768                        name: "[budget]".into(),
769                        status: StepStatus::Failed,
770                        message,
771                    });
772                    result.failed += 1;
773                    return result;
774                }
775                BudgetStatus::SoftExceeded { message, .. } => {
776                    self.emit_event(&TestEvent::Warning {
777                        message: format!("budget warning: {message}"),
778                    });
779                }
780                BudgetStatus::Ok => {}
781            }
782
783            if let Some(ms) = wait_ms {
784                std::thread::sleep(Duration::from_millis(ms));
785            }
786
787            result.details.push(step_result);
788        }
789
790        result
791    }
792
793    /// Applies a viewport size to the current tab via CDP
794    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
795    /// overrides and the viewport matrix. The initial window size set at
796    /// browser launch is replaced by emulation; failures are logged but
797    /// do not fail the test (a mismatched viewport only weakens coverage).
798    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
799        use headless_chrome::protocol::cdp::Emulation;
800        let _ = self;
801        let params = Emulation::SetDeviceMetricsOverride {
802            width,
803            height,
804            device_scale_factor: 1.0,
805            mobile: false,
806            scale: None,
807            screen_width: Some(width),
808            screen_height: Some(height),
809            position_x: None,
810            position_y: None,
811            dont_set_visible_size: None,
812            screen_orientation: None,
813            viewport: None,
814            display_feature: None,
815            device_posture: None,
816        };
817        self.reporter.debug(format!("viewport: {width}x{height}"));
818        if let Err(e) = tab.call_method(params) {
819            self.reporter
820                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
821        }
822    }
823
824    // ── step handlers ───────────────────────────────────────────────────
825
826    #[allow(clippy::too_many_lines)]
827    fn run_click(
828        &self,
829        target: &str,
830        selector_override: Option<&str>,
831        step_endpoint: Option<&str>,
832        test_endpoint: Option<&str>,
833        idempotent: bool,
834        tab: &Tab,
835    ) -> StepResult {
836        let name = format!("[click] {target}");
837        let selector = match self.resolve_selector(
838            selector_override,
839            target,
840            step_endpoint,
841            test_endpoint,
842            tab,
843        ) {
844            Ok(s) => s,
845            Err(msg) => {
846                if idempotent {
847                    return StepResult {
848                        name,
849                        status: StepStatus::Skipped,
850                        message: format!("skipped (idempotent): no target found — {msg}"),
851                    };
852                }
853                return StepResult {
854                    name,
855                    status: StepStatus::Failed,
856                    message: msg,
857                };
858            }
859        };
860
861        // Idempotent steps probe briefly: a missing target means the
862        // action was already done / not applicable (e.g. an
863        // already-authenticated session), and skipping is the success
864        // path, not a failure.
865        let probe_secs = if idempotent { 5 } else { 10 };
866        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
867            Ok(element) => match element.click() {
868                Ok(_) => StepResult {
869                    name,
870                    status: StepStatus::Passed,
871                    message: format!("clicked {selector}"),
872                },
873                Err(e) => StepResult {
874                    name,
875                    status: StepStatus::Failed,
876                    message: format!("click failed on {selector}: {e}"),
877                },
878            },
879            Err(e) if idempotent => StepResult {
880                name,
881                status: StepStatus::Skipped,
882                message: format!("skipped (idempotent): element {selector} not present — {e}"),
883            },
884            Err(e) => StepResult {
885                name,
886                status: StepStatus::Failed,
887                message: format!("element {selector} not found: {e}"),
888            },
889        }
890    }
891
892    #[allow(clippy::too_many_arguments)]
893    fn run_type(
894        &self,
895        target: &str,
896        text: &str,
897        selector_override: Option<&str>,
898        step_endpoint: Option<&str>,
899        test_endpoint: Option<&str>,
900        idempotent: bool,
901        tab: &Tab,
902    ) -> StepResult {
903        let name = format!("[type] {target}");
904        let selector = match self.resolve_selector(
905            selector_override,
906            target,
907            step_endpoint,
908            test_endpoint,
909            tab,
910        ) {
911            Ok(s) => s,
912            Err(msg) => {
913                if idempotent {
914                    return StepResult {
915                        name,
916                        status: StepStatus::Skipped,
917                        message: format!("skipped (idempotent): no target found — {msg}"),
918                    };
919                }
920                return StepResult {
921                    name,
922                    status: StepStatus::Failed,
923                    message: msg,
924                };
925            }
926        };
927
928        let probe_secs = if idempotent { 5 } else { 10 };
929        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
930            Ok(element) => {
931                if let Err(e) = element.click() {
932                    return StepResult {
933                        name,
934                        status: StepStatus::Failed,
935                        message: format!("click to focus {selector} failed: {e}"),
936                    };
937                }
938
939                let js = format!(
940                    "document.querySelector('{}').value = '';",
941                    selector.replace('\'', "\\'")
942                );
943                let _ = tab.evaluate(&js, false);
944
945                match element.type_into(text) {
946                    Ok(_) => StepResult {
947                        name,
948                        status: StepStatus::Passed,
949                        message: format!("typed {text:?} into {selector}"),
950                    },
951                    Err(e) => StepResult {
952                        name,
953                        status: StepStatus::Failed,
954                        message: format!("type into {selector} failed: {e}"),
955                    },
956                }
957            }
958            Err(e) if idempotent => StepResult {
959                name,
960                status: StepStatus::Skipped,
961                message: format!("skipped (idempotent): element {selector} not present — {e}"),
962            },
963            Err(e) => StepResult {
964                name,
965                status: StepStatus::Failed,
966                message: format!("element {selector} not found: {e}"),
967            },
968        }
969    }
970
971    #[allow(clippy::too_many_arguments)]
972    #[allow(clippy::too_many_lines)]
973    fn run_wait(
974        &self,
975        target: &str,
976        selector_override: Option<&str>,
977        text: Option<&str>,
978        timeout_ms: Option<u64>,
979        step_endpoint: Option<&str>,
980        test_endpoint: Option<&str>,
981        idempotent: bool,
982        tab: &Tab,
983    ) -> StepResult {
984        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
985        let step_name = format!("[wait] {target}");
986
987        // Resolve an explicit selector only (text-only waits are LLM-free).
988        let selector = match selector_override {
989            Some(s) => Some(s.to_owned()),
990            None if text.is_some() => None,
991            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
992                Ok(s) => Some(s),
993                Err(msg) => {
994                    if idempotent {
995                        return StepResult {
996                            name: step_name,
997                            status: StepStatus::Skipped,
998                            message: format!("skipped (idempotent): no target found — {msg}"),
999                        };
1000                    }
1001                    return StepResult {
1002                        name: step_name,
1003                        status: StepStatus::Failed,
1004                        message: msg,
1005                    };
1006                }
1007            },
1008        };
1009
1010        if text.is_some() {
1011            let sel_js = selector
1012                .as_deref()
1013                .map(crate::selectors::selector_matches_js);
1014            let text_js = text.map(|t| {
1015                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1016                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1017            });
1018
1019            let deadline = Instant::now() + timeout;
1020            loop {
1021                let sel_ok = sel_js
1022                    .as_ref()
1023                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1024                let text_ok = text_js
1025                    .as_ref()
1026                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1027                if sel_ok && text_ok {
1028                    let mut what = Vec::new();
1029                    if let Some(sel) = &selector {
1030                        what.push(format!("found {sel}"));
1031                    }
1032                    if let Some(t) = text {
1033                        what.push(format!("text {t:?} visible"));
1034                    }
1035                    return StepResult {
1036                        name: step_name,
1037                        status: StepStatus::Passed,
1038                        message: what.join(" and "),
1039                    };
1040                }
1041                if Instant::now() >= deadline {
1042                    let mut what = Vec::new();
1043                    if let Some(sel) = &selector {
1044                        what.push(sel.clone());
1045                    }
1046                    if let Some(t) = text {
1047                        what.push(format!("text {t:?}"));
1048                    }
1049                    let message = format!(
1050                        "wait for {} timed out after {}ms: the event waited for never came",
1051                        what.join(" / "),
1052                        timeout.as_millis(),
1053                    );
1054                    if idempotent {
1055                        return StepResult {
1056                            name: step_name,
1057                            status: StepStatus::Skipped,
1058                            message: format!("skipped (idempotent): {message}"),
1059                        };
1060                    }
1061                    return StepResult {
1062                        name: step_name,
1063                        status: StepStatus::Failed,
1064                        message,
1065                    };
1066                }
1067                std::thread::sleep(Duration::from_millis(250));
1068            }
1069        }
1070
1071        match selector.as_deref() {
1072            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1073                Ok(_) => StepResult {
1074                    name: step_name,
1075                    status: StepStatus::Passed,
1076                    message: format!("found {sel}"),
1077                },
1078                Err(e) if idempotent => StepResult {
1079                    name: step_name,
1080                    status: StepStatus::Skipped,
1081                    message: format!(
1082                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1083                        timeout.as_millis()
1084                    ),
1085                },
1086                Err(e) => StepResult {
1087                    name: step_name,
1088                    status: StepStatus::Failed,
1089                    message: format!(
1090                        "wait for {sel} timed out after {}ms: {e}",
1091                        timeout.as_millis()
1092                    ),
1093                },
1094            },
1095            None => StepResult {
1096                name: step_name,
1097                status: StepStatus::Failed,
1098                message: "wait step has neither selector nor text".into(),
1099            },
1100        }
1101    }
1102
1103    #[allow(clippy::too_many_arguments)]
1104    fn run_assert(
1105        &self,
1106        definition: Option<&str>,
1107        preset: Option<&str>,
1108        prompt: Option<&str>,
1109        assert_text: Option<&str>,
1110        screenshot: bool,
1111        step_endpoint: Option<&str>,
1112        test_endpoint: Option<&str>,
1113        tab: &Tab,
1114    ) -> StepResult {
1115        std::thread::sleep(Duration::from_millis(500));
1116
1117        let page_content = get_page_text(tab);
1118
1119        // Vision attach: capture the viewport once per assert step and hand
1120        // the JPEG data URL to the preset/prompt evaluation below.
1121        let image = if screenshot {
1122            let endpoint = self
1123                .endpoints
1124                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1125            if !endpoint.vision {
1126                return StepResult {
1127                    name: "[assert]".into(),
1128                    status: StepStatus::Failed,
1129                    message: format!(
1130                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1131                        name = endpoint.name
1132                    ),
1133                };
1134            }
1135            match crate::vision::capture_screenshot_data_url(
1136                tab,
1137                self.config
1138                    .screenshot_max_dimension
1139                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1140            ) {
1141                Ok(data_url) => Some(data_url),
1142                Err(e) => {
1143                    return StepResult {
1144                        name: "[assert]".into(),
1145                        status: StepStatus::Failed,
1146                        message: format!("screenshot capture failed: {e}"),
1147                    };
1148                }
1149            }
1150        } else {
1151            None
1152        };
1153
1154        if let Some(def_name) = definition {
1155            if let Some(def) = self.definitions.get(def_name) {
1156                return self.run_assert_def(
1157                    def,
1158                    &page_content,
1159                    image.as_deref(),
1160                    step_endpoint,
1161                    test_endpoint,
1162                    tab,
1163                );
1164            }
1165            return StepResult {
1166                name: format!("[assert] {def_name}"),
1167                status: StepStatus::Failed,
1168                message: format!("definition '{def_name}' not found"),
1169            };
1170        }
1171
1172        if let Some(preset_name) = preset {
1173            // Deterministic DOM layout scan — runs JS in the browser and
1174            // never calls the LLM (free, fast, no pixel budget).
1175            if preset_name == "layout_no_issues" {
1176                return self.run_layout_preset(tab);
1177            }
1178            return self.run_preset(
1179                preset_name,
1180                assert_text,
1181                &page_content,
1182                image.as_deref(),
1183                step_endpoint,
1184                test_endpoint,
1185            );
1186        }
1187
1188        if let Some(prompt_text) = prompt {
1189            return self.run_custom(
1190                prompt_text,
1191                &page_content,
1192                image.as_deref(),
1193                step_endpoint,
1194                test_endpoint,
1195            );
1196        }
1197
1198        StepResult {
1199            name: "[assert]".into(),
1200            status: StepStatus::Skipped,
1201            message: "no definition, preset, or prompt specified".into(),
1202        }
1203    }
1204
1205    fn run_assert_def(
1206        &self,
1207        def: &AssertDefinition,
1208        page_content: &PageContent,
1209        image: Option<&str>,
1210        step_endpoint: Option<&str>,
1211        test_endpoint: Option<&str>,
1212        tab: &Tab,
1213    ) -> StepResult {
1214        // Agent-based definition: delegate to an A2A agent
1215        if let Some(ref agent) = def.agent {
1216            if image.is_some() {
1217                return StepResult {
1218                    name: format!("[assert] {}", def.name),
1219                    status: StepStatus::Failed,
1220                    message: "agent-backed assertions do not support screenshots".into(),
1221                };
1222            }
1223            let task = def
1224                .task_template
1225                .as_deref()
1226                .unwrap_or("Evaluate the assertion")
1227                .replace("{url}", &page_content.url)
1228                .replace("{title}", &page_content.title)
1229                .replace("{content}", &page_content.body_text)
1230                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1231
1232            return self.run_agent_step(agent, &task, &def.name);
1233        }
1234
1235        // Custom preset: system + user_template provided in the definition
1236        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1237            return self.run_custom_preset(
1238                &def.name,
1239                system,
1240                template,
1241                def.assert_text.as_deref(),
1242                page_content,
1243                image,
1244                step_endpoint,
1245                test_endpoint,
1246            );
1247        }
1248
1249        def.preset.as_ref().map_or_else(
1250            || {
1251                def.prompt.as_ref().map_or_else(
1252                    || StepResult {
1253                        name: format!("[assert] {}", def.name),
1254                        status: StepStatus::Failed,
1255                        message: "definition has no preset, prompt, or system+user_template".into(),
1256                    },
1257                    |prompt| {
1258                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1259                    },
1260                )
1261            },
1262            |preset_name| {
1263                if preset_name == "layout_no_issues" {
1264                    return self.run_layout_preset(tab);
1265                }
1266                self.run_preset(
1267                    preset_name,
1268                    def.assert_text.as_deref(),
1269                    page_content,
1270                    image,
1271                    step_endpoint,
1272                    test_endpoint,
1273                )
1274            },
1275        )
1276    }
1277
1278    #[allow(clippy::too_many_arguments)]
1279    fn run_custom_preset(
1280        &self,
1281        name: &str,
1282        system: &str,
1283        template: &str,
1284        assert_text: Option<&str>,
1285        page_content: &PageContent,
1286        image: Option<&str>,
1287        step_endpoint: Option<&str>,
1288        test_endpoint: Option<&str>,
1289    ) -> StepResult {
1290        let user_prompt = template
1291            .replace("{url}", &page_content.url)
1292            .replace("{title}", &page_content.title)
1293            .replace("{content}", &page_content.body_text)
1294            .replace("{expected_text}", assert_text.unwrap_or(""))
1295            .replace("{description}", "");
1296
1297        // Custom preset definitions frequently forget the {content}
1298        // placeholder — without it the LLM has no page to evaluate and
1299        // answers "I can't determine that without seeing the page". Always
1300        // append the page context unless the template already references it.
1301        let user_prompt = if template.contains("{content}") {
1302            user_prompt
1303        } else {
1304            format!(
1305                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1306                url = page_content.url,
1307                title = page_content.title,
1308                content = page_content.body_text,
1309            )
1310        };
1311
1312        self.reporter
1313            .debug(format!("assert: {name} (custom preset)"));
1314
1315        let chain = self
1316            .endpoints
1317            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1318        let sys = system.to_owned();
1319
1320        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1321
1322        response.map_or_else(
1323            |e| StepResult {
1324                name: format!("[assert] {name}"),
1325                status: StepStatus::Failed,
1326                message: format!("LLM assertion call failed: {e}"),
1327            },
1328            |(lr, idx)| {
1329                self.usage.record_llm_call(
1330                    &chain[idx].name,
1331                    chain[idx],
1332                    lr.usage.prompt_tokens,
1333                    lr.usage.completion_tokens,
1334                );
1335                let content_lower = lr.content.to_lowercase().trim().to_owned();
1336                if content_lower.starts_with("pass") {
1337                    StepResult {
1338                        name: format!("[assert] {name}"),
1339                        status: StepStatus::Passed,
1340                        message: "PASS".into(),
1341                    }
1342                } else {
1343                    StepResult {
1344                        name: format!("[assert] {name}"),
1345                        status: StepStatus::Failed,
1346                        message: lr.content,
1347                    }
1348                }
1349            },
1350        )
1351    }
1352
1353    fn run_preset(
1354        &self,
1355        preset_name: &str,
1356        assert_text: Option<&str>,
1357        page_content: &PageContent,
1358        image: Option<&str>,
1359        step_endpoint: Option<&str>,
1360        test_endpoint: Option<&str>,
1361    ) -> StepResult {
1362        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1363            return StepResult {
1364                name: format!("[assert] {preset_name}"),
1365                status: StepStatus::Failed,
1366                message: format!("unknown assertion preset: {preset_name}"),
1367            };
1368        };
1369        if preset_name.starts_with("visual_") && image.is_none() {
1370            return StepResult {
1371                name: format!("[assert] {preset_name}"),
1372                status: StepStatus::Failed,
1373                message: format!(
1374                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1375                ),
1376            };
1377        }
1378
1379        let user_prompt = preset
1380            .user_template
1381            .replace("{url}", &page_content.url)
1382            .replace("{title}", &page_content.title)
1383            .replace("{content}", &page_content.body_text)
1384            .replace("{expected_text}", assert_text.unwrap_or(""))
1385            .replace("{description}", "");
1386
1387        // Same safety net as custom presets: never let the LLM answer with
1388        // no page context at all.
1389        let user_prompt = if preset.user_template.contains("{content}") {
1390            user_prompt
1391        } else {
1392            format!(
1393                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1394                url = page_content.url,
1395                title = page_content.title,
1396                content = page_content.body_text,
1397            )
1398        };
1399
1400        self.reporter.debug(format!("assert: {preset_name}"));
1401
1402        let chain = self
1403            .endpoints
1404            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1405        let sys = preset.system.to_owned();
1406
1407        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1408
1409        response.map_or_else(
1410            |e| StepResult {
1411                name: format!("[assert] {preset_name}"),
1412                status: StepStatus::Failed,
1413                message: format!("LLM assertion call failed: {e}"),
1414            },
1415            |(lr, idx)| {
1416                self.usage.record_llm_call(
1417                    &chain[idx].name,
1418                    chain[idx],
1419                    lr.usage.prompt_tokens,
1420                    lr.usage.completion_tokens,
1421                );
1422                let content_lower = lr.content.to_lowercase().trim().to_owned();
1423                if content_lower.starts_with("pass") {
1424                    StepResult {
1425                        name: format!("[assert] {preset_name}"),
1426                        status: StepStatus::Passed,
1427                        message: "PASS".into(),
1428                    }
1429                } else {
1430                    StepResult {
1431                        name: format!("[assert] {preset_name}"),
1432                        status: StepStatus::Failed,
1433                        message: lr.content,
1434                    }
1435                }
1436            },
1437        )
1438    }
1439
1440    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1441    ///
1442    /// Evaluates the layout-scan JS in the page and fails with the list of
1443    /// detected issues: horizontal page overflow, visible elements sticking
1444    /// out of the viewport, text clipped by `overflow: hidden` containers,
1445    /// and interactive elements covered by other elements. No LLM call —
1446    /// checks are geometry-based so the check is free, deterministic, and
1447    /// safe to run on every page × viewport variant.
1448    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1449        let name = "[assert] layout_no_issues".to_owned();
1450        self.reporter
1451            .debug("assert: layout_no_issues (DOM layout scan)");
1452        let js = LAYOUT_SCAN_JS.replace(
1453            "__IGNORE_CLASSES__",
1454            &serde_json::to_string(&self.config.layout_ignore_classes)
1455                .unwrap_or_else(|_| "[]".to_owned()),
1456        );
1457        let result = tab.evaluate(&js, false);
1458        let json_str = match result {
1459            Ok(r) => r
1460                .value
1461                .as_ref()
1462                .and_then(|v| v.as_str().map(String::from))
1463                .unwrap_or_else(|| "[]".to_owned()),
1464            Err(e) => {
1465                return StepResult {
1466                    name,
1467                    status: StepStatus::Failed,
1468                    message: format!("layout scan JS failed: {e}"),
1469                };
1470            }
1471        };
1472        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1473        if issues.is_empty() {
1474            return StepResult {
1475                name,
1476                status: StepStatus::Passed,
1477                message: "PASS — no layout defects detected".into(),
1478            };
1479        }
1480        let mut lines: Vec<String> = issues
1481            .iter()
1482            .take(10)
1483            .map(|i| {
1484                format!(
1485                    "- [{type_}] {element}: {detail}",
1486                    type_ = i.issue_type,
1487                    element = i.element,
1488                    detail = i.detail
1489                )
1490            })
1491            .collect();
1492        if issues.len() > 10 {
1493            lines.push(format!("- … and {} more", issues.len() - 10));
1494        }
1495        StepResult {
1496            name,
1497            status: StepStatus::Failed,
1498            message: format!(
1499                "FAIL — {} layout defect(s) detected:\n{}",
1500                issues.len(),
1501                lines.join("\n")
1502            ),
1503        }
1504    }
1505
1506    fn run_custom(
1507        &self,
1508        prompt: &str,
1509        page_content: &PageContent,
1510        image: Option<&str>,
1511        step_endpoint: Option<&str>,
1512        test_endpoint: Option<&str>,
1513    ) -> StepResult {
1514        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1515
1516        let mut user = format!(
1517            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1518            url = page_content.url,
1519            title = page_content.title,
1520            content = page_content.body_text,
1521        );
1522        if image.is_some() {
1523            user.push_str(
1524                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1525            );
1526        }
1527
1528        self.reporter.debug("custom assert");
1529
1530        let chain = self
1531            .endpoints
1532            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1533        let sys = system.to_owned();
1534
1535        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1536
1537        response.map_or_else(
1538            |e| StepResult {
1539                name: "[assert] custom".into(),
1540                status: StepStatus::Failed,
1541                message: format!("LLM assertion call failed: {e}"),
1542            },
1543            |(lr, idx)| {
1544                self.usage.record_llm_call(
1545                    &chain[idx].name,
1546                    chain[idx],
1547                    lr.usage.prompt_tokens,
1548                    lr.usage.completion_tokens,
1549                );
1550                let content_lower = lr.content.to_lowercase().trim().to_owned();
1551                if content_lower.starts_with("pass") {
1552                    StepResult {
1553                        name: "[assert] custom".into(),
1554                        status: StepStatus::Passed,
1555                        message: "PASS".into(),
1556                    }
1557                } else {
1558                    StepResult {
1559                        name: "[assert] custom".into(),
1560                        status: StepStatus::Failed,
1561                        message: lr.content,
1562                    }
1563                }
1564            },
1565        )
1566    }
1567
1568    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1569        let path = path.unwrap_or("screenshot.png");
1570
1571        match tab.capture_screenshot(
1572            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1573            None,
1574            None,
1575            true,
1576        ) {
1577            Ok(data) => {
1578                if let Err(e) = std::fs::write(path, &data) {
1579                    return StepResult {
1580                        name: format!("[screenshot] {path}"),
1581                        status: StepStatus::Failed,
1582                        message: format!("failed to write screenshot: {e}"),
1583                    };
1584                }
1585                StepResult {
1586                    name: format!("[screenshot] {path}"),
1587                    status: StepStatus::Passed,
1588                    message: format!("saved to {path}"),
1589                }
1590            }
1591            Err(e) => StepResult {
1592                name: format!("[screenshot] {path}"),
1593                status: StepStatus::Failed,
1594                message: format!("screenshot failed: {e}"),
1595            },
1596        }
1597    }
1598
1599    /// Runs an A2A agent step.
1600    #[allow(clippy::literal_string_with_formatting_args)]
1601    fn run_agent(
1602        &self,
1603        agent_name: &str,
1604        task: &str,
1605        definition: Option<&str>,
1606        _test_endpoint: Option<&str>,
1607    ) -> StepResult {
1608        // If a definition is specified, look up the task template
1609        let resolved_task = if let Some(def_name) = definition {
1610            if let Some(def) = self.definitions.get(def_name) {
1611                let tmpl = def.task_template.as_deref().unwrap_or(task);
1612                tmpl.replace("{task}", task)
1613            } else {
1614                return StepResult {
1615                    name: format!("[agent] {def_name}"),
1616                    status: StepStatus::Failed,
1617                    message: format!("definition '{def_name}' not found"),
1618                };
1619            }
1620        } else {
1621            task.to_owned()
1622        };
1623
1624        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1625    }
1626
1627    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1628        let Some(ep) = self.endpoints.get(agent_name) else {
1629            return StepResult {
1630                name: format!("[agent] {display_name}"),
1631                status: StepStatus::Failed,
1632                message: format!("agent endpoint '{agent_name}' not found"),
1633            };
1634        };
1635
1636        if ep.url.is_empty() {
1637            return StepResult {
1638                name: format!("[agent] {display_name}"),
1639                status: StepStatus::Failed,
1640                message: format!("agent endpoint '{agent_name}' has no URL"),
1641            };
1642        }
1643
1644        self.reporter.debug(format!("agent {agent_name}: {task}"));
1645
1646        let url = ep.url.clone();
1647        let client = A2aClient::new(&url, self.timeout);
1648        let task_clone = task.to_owned();
1649
1650        let response = std::thread::spawn(move || {
1651            let rt = tokio::runtime::Builder::new_current_thread()
1652                .enable_all()
1653                .build()
1654                .unwrap();
1655            rt.block_on(client.send_task(&task_clone))
1656        })
1657        .join()
1658        .unwrap();
1659
1660        // Record the flat-cost call
1661        self.usage.record_flat_call(agent_name, ep);
1662
1663        match response {
1664            Ok(text) => {
1665                let clean = text.trim().to_owned();
1666                let lower = clean.to_lowercase();
1667                if lower.starts_with("pass") {
1668                    StepResult {
1669                        name: format!("[agent] {display_name}"),
1670                        status: StepStatus::Passed,
1671                        message: format!("PASS: {clean}"),
1672                    }
1673                } else if lower.starts_with("fail") {
1674                    StepResult {
1675                        name: format!("[agent] {display_name}"),
1676                        status: StepStatus::Failed,
1677                        message: clean,
1678                    }
1679                } else {
1680                    StepResult {
1681                        name: format!("[agent] {display_name}"),
1682                        status: StepStatus::Passed,
1683                        message: format!("response: {clean}"),
1684                    }
1685                }
1686            }
1687            Err(e) => StepResult {
1688                name: format!("[agent] {display_name}"),
1689                status: StepStatus::Failed,
1690                message: format!("agent call failed: {e}"),
1691            },
1692        }
1693    }
1694
1695    /// Runs an MCP tool call step.
1696    fn run_mcp(
1697        &self,
1698        server_name: &str,
1699        tool_name: &str,
1700        args: Option<&serde_json::Value>,
1701    ) -> StepResult {
1702        let Some(ep) = self.endpoints.get(server_name) else {
1703            return StepResult {
1704                name: format!("[mcp] {server_name}:{tool_name}"),
1705                status: StepStatus::Failed,
1706                message: format!("MCP server endpoint '{server_name}' not found"),
1707            };
1708        };
1709
1710        let cmd = ep.command.as_deref().unwrap_or("");
1711        if cmd.is_empty() {
1712            return StepResult {
1713                name: format!("[mcp] {server_name}:{tool_name}"),
1714                status: StepStatus::Failed,
1715                message: format!("MCP server '{server_name}' has no command configured"),
1716            };
1717        }
1718
1719        self.reporter
1720            .debug(format!("mcp {server_name} {tool_name}"));
1721
1722        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1723
1724        let command = cmd.to_owned();
1725        let args_vec = ep.args.clone();
1726        let tool = tool_name.to_owned();
1727
1728        let response = std::thread::spawn(move || {
1729            let mut mcp_client =
1730                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1731            mcp_client
1732                .call_tool(&tool, &args_val)
1733                .map_err(|e| e.to_string())
1734        })
1735        .join()
1736        .unwrap();
1737
1738        // Record the flat-cost call
1739        self.usage.record_flat_call(server_name, ep);
1740
1741        match response {
1742            Ok(result) => {
1743                if result.isError {
1744                    StepResult {
1745                        name: format!("[mcp] {server_name}:{tool_name}"),
1746                        status: StepStatus::Failed,
1747                        message: result.to_string(),
1748                    }
1749                } else {
1750                    StepResult {
1751                        name: format!("[mcp] {server_name}:{tool_name}"),
1752                        status: StepStatus::Passed,
1753                        message: result.to_string(),
1754                    }
1755                }
1756            }
1757            Err(e) => StepResult {
1758                name: format!("[mcp] {server_name}:{tool_name}"),
1759                status: StepStatus::Failed,
1760                message: format!("MCP call failed: {e}"),
1761            },
1762        }
1763    }
1764
1765    // ── helpers ──────────────────────────────────────────────────────────
1766
1767    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1768    /// the runner's default LLM config for any unset fields.
1769    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1770        LlmConfig {
1771            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
1772                // Bedrock builds its endpoint from the resolved AWS region
1773                // when no URL is given — never inherit the default LLM URL.
1774                endpoint.url.clone()
1775            } else if endpoint.url.is_empty() {
1776                self.llm.url.clone()
1777            } else {
1778                endpoint.url.clone()
1779            },
1780            model: endpoint
1781                .model
1782                .clone()
1783                .unwrap_or_else(|| self.llm.model.clone()),
1784            api_key: endpoint
1785                .api_key
1786                .clone()
1787                .or_else(|| self.llm.api_key.clone()),
1788            headers: if endpoint.headers.is_empty() {
1789                self.llm.headers.clone()
1790            } else {
1791                endpoint.headers.clone()
1792            },
1793            timeout: self.llm.timeout,
1794            temperature: self.llm.temperature,
1795            thinking: self.llm.thinking,
1796            model_params: self.llm.model_params.clone(),
1797            max_attempts: endpoint.max_attempts.max(1),
1798            provider: endpoint.provider,
1799            deployment: endpoint.deployment.clone(),
1800            api_version: endpoint.api_version.clone(),
1801            auth: endpoint.auth.clone(),
1802            header_commands: endpoint.header_commands.clone(),
1803            aws: endpoint.aws.clone(),
1804        }
1805    }
1806
1807    /// Runs a single LLM call against an ordered endpoint chain (primary +
1808    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
1809    /// the first endpoint that answers wins. Returns the response together
1810    /// with the chain index of the answering endpoint (0 = primary) so the
1811    /// caller can attribute usage to the correct endpoint.
1812    ///
1813    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
1814    /// shows duration, tokens, cost and the answering endpoint per call.
1815    #[allow(clippy::cast_possible_truncation)]
1816    fn llm_call_chain(
1817        &self,
1818        chain: &[&ResolvedEndpoint],
1819        system: &str,
1820        user: &str,
1821        image: Option<&str>,
1822        purpose: &str,
1823    ) -> Result<(crate::costs::LlmResponse, usize), String> {
1824        if chain.is_empty() {
1825            return Err("empty LLM endpoint chain".into());
1826        }
1827        let primary = self.build_llm_for_endpoint(chain[0]);
1828        let fallbacks: Vec<LlmConfig> = chain[1..]
1829            .iter()
1830            .map(|e| self.build_llm_for_endpoint(e))
1831            .collect();
1832
1833        let (test, index) = self
1834            .current_step
1835            .borrow()
1836            .as_ref()
1837            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
1838        let primary_endpoint = chain[0].name.clone();
1839        let primary_model = primary.model.clone();
1840        self.emit_event(&TestEvent::LlmCallStarted {
1841            test: test.clone(),
1842            index,
1843            endpoint: primary_endpoint.clone(),
1844            model: primary_model.clone(),
1845            purpose: purpose.to_owned(),
1846        });
1847
1848        let started = Instant::now();
1849        let sys = system.to_owned();
1850        let user = user.to_owned();
1851        let image = image.map(str::to_owned);
1852
1853        let result = std::thread::spawn(move || {
1854            let rt = tokio::runtime::Builder::new_current_thread()
1855                .enable_all()
1856                .build()
1857                .unwrap();
1858            let call = async {
1859                match image.as_deref() {
1860                    Some(img) => {
1861                        llm_chat_vision_with_usage_chain(&primary, &fallbacks, &sys, &user, img)
1862                            .await
1863                    }
1864                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
1865                }
1866            };
1867            rt.block_on(call)
1868        })
1869        .join()
1870        .unwrap();
1871
1872        let duration_ms = started.elapsed().as_millis() as u64;
1873        match result {
1874            Ok((lr, idx)) => {
1875                let cost = calculate_llm_cost(
1876                    chain[idx],
1877                    lr.usage.prompt_tokens,
1878                    lr.usage.completion_tokens,
1879                );
1880                let answering = chain[idx].name.clone();
1881                let model = chain[idx]
1882                    .model
1883                    .clone()
1884                    .unwrap_or_else(|| primary_model.clone());
1885                self.emit_event(&TestEvent::LlmCallFinished {
1886                    test,
1887                    index,
1888                    endpoint: answering,
1889                    model,
1890                    purpose: purpose.to_owned(),
1891                    ok: true,
1892                    duration_ms,
1893                    input_tokens: lr.usage.prompt_tokens,
1894                    output_tokens: lr.usage.completion_tokens,
1895                    cost,
1896                    error: None,
1897                });
1898                Ok((lr, idx))
1899            }
1900            Err(e) => {
1901                self.emit_event(&TestEvent::LlmCallFinished {
1902                    test,
1903                    index,
1904                    endpoint: primary_endpoint,
1905                    model: primary_model,
1906                    purpose: purpose.to_owned(),
1907                    ok: false,
1908                    duration_ms,
1909                    input_tokens: 0,
1910                    output_tokens: 0,
1911                    cost: 0.0,
1912                    error: Some(e.clone()),
1913                });
1914                Err(e)
1915            }
1916        }
1917    }
1918
1919    /// Resolves a CSS selector for the target element. Uses the explicit
1920    /// `selector` if provided, otherwise asks the LLM to find the element
1921    /// from the natural language `target` description and page DOM.
1922    ///
1923    /// LLM responses are sanitized and verified against the live page: a
1924    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
1925    /// immediately with the raw LLM output, and a selector that matches
1926    /// nothing triggers one retry with feedback before failing.
1927    #[allow(clippy::too_many_lines)]
1928    fn resolve_selector(
1929        &self,
1930        css_override: Option<&str>,
1931        target: &str,
1932        step_endpoint: Option<&str>,
1933        test_endpoint: Option<&str>,
1934        tab: &Tab,
1935    ) -> Result<String, String> {
1936        if let Some(explicit) = css_override {
1937            return Ok(explicit.to_owned());
1938        }
1939
1940        let dom_info = extract_dom_info(tab)?;
1941        let page_content = get_page_text(tab);
1942
1943        let system = concat!(
1944            "You are a browser automation selector generator. ",
1945            "Given a web page's content and interactive elements, ",
1946            "return ONLY the best CSS selector for the described element. ",
1947            "Output nothing except the CSS selector. ",
1948            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
1949            "[name=\"...\"], tag.class, tag. ",
1950            "Never output explanations, markdown, or extra text."
1951        );
1952
1953        let user = format!(
1954            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
1955            page_content.url,
1956            page_content.title,
1957            truncate(&page_content.body_text, 4000),
1958            dom_info,
1959            target,
1960        );
1961
1962        let retry_user = format!(
1963            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
1964            "The selector must match at least one element currently present on the page.",
1965            page_content.url,
1966            page_content.title,
1967            truncate(&page_content.body_text, 4000),
1968            dom_info,
1969            target,
1970        );
1971
1972        self.reporter.debug(format!("LLM targeting: {target}"));
1973
1974        let chain = self
1975            .endpoints
1976            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
1977        let sys = system.to_owned();
1978
1979        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
1980
1981        let first = call_llm(&user);
1982        let (lr, idx) = match first {
1983            Ok(lr) => lr,
1984            Err(e) => {
1985                return Err(format!("LLM element targeting failed: {e}"));
1986            }
1987        };
1988        self.usage.record_llm_call(
1989            &chain[idx].name,
1990            chain[idx],
1991            lr.usage.prompt_tokens,
1992            lr.usage.completion_tokens,
1993        );
1994        let clean = sanitize_selector(&lr.content);
1995        self.reporter.debug(format!("resolved selector: {clean}"));
1996
1997        if selector_is_useless(&clean) {
1998            return Err(format!(
1999                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2000                raw = lr.content.trim(),
2001            ));
2002        }
2003        if let Err(reason) = validate_selector(&clean) {
2004            return Err(format!(
2005                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2006                raw = lr.content.trim(),
2007            ));
2008        }
2009        if !selector_matches(tab, &clean).unwrap_or(false) {
2010            // One retry with feedback: flaky models occasionally invent a
2011            // selector that does not exist on the page.
2012            self.reporter.warn(format!(
2013                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2014            ));
2015            let second = call_llm(&retry_user);
2016            let (lr2, idx2) = match second {
2017                Ok(lr2) => lr2,
2018                Err(e) => {
2019                    return Err(format!(
2020                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2021                    ));
2022                }
2023            };
2024            self.usage.record_llm_call(
2025                &chain[idx2].name,
2026                chain[idx2],
2027                lr2.usage.prompt_tokens,
2028                lr2.usage.completion_tokens,
2029            );
2030            let clean2 = sanitize_selector(&lr2.content);
2031            self.reporter
2032                .debug(format!("resolved selector (retry): {clean2}"));
2033            if selector_is_useless(&clean2) {
2034                return Err(format!(
2035                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2036                    raw = lr2.content.trim(),
2037                    excerpt = truncate(&page_content.body_text, 300),
2038                ));
2039            }
2040            if !selector_matches(tab, &clean2).unwrap_or(false) {
2041                return Err(format!(
2042                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2043                ));
2044            }
2045            return Ok(clean2);
2046        }
2047
2048        Ok(clean)
2049    }
2050}
2051
2052/// Evaluates a JS expression that is expected to return a boolean.
2053fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2054    tab.evaluate(js, false)
2055        .map_err(|e| format!("evaluate failed: {e}"))?
2056        .value
2057        .and_then(|v| v.as_bool())
2058        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2059}
2060
2061/// Checks whether a CSS selector matches at least one current element.
2062fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2063    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2064}
2065
2066// ── Free helper functions ──────────────────────────────────────────────
2067
2068fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2069    let name = format!("[navigate] {full_url}");
2070    match tab.navigate_to(full_url) {
2071        Ok(_) => {
2072            let _ = tab.wait_until_navigated();
2073            StepResult {
2074                name,
2075                status: StepStatus::Passed,
2076                message: format!("navigated to {full_url}"),
2077            }
2078        }
2079        Err(e) => StepResult {
2080            name,
2081            status: StepStatus::Failed,
2082            message: format!("navigation failed: {e}"),
2083        },
2084    }
2085}
2086
2087fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2088    let result = tab
2089        .evaluate(DOM_EXTRACT_JS, false)
2090        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2091
2092    let json_str = result
2093        .value
2094        .as_ref()
2095        .and_then(|v| v.as_str())
2096        .unwrap_or("[]");
2097
2098    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2099
2100    if elements.is_empty() {
2101        return Ok("(no interactive elements found)".to_owned());
2102    }
2103
2104    Ok(elements.join("\n"))
2105}
2106
2107fn get_page_text(tab: &Tab) -> PageContent {
2108    let url = tab.get_url();
2109
2110    let title = tab
2111        .evaluate("document.title", false)
2112        .ok()
2113        .and_then(|r| r.value)
2114        .and_then(|v| v.as_str().map(String::from))
2115        .unwrap_or_else(|| "unknown".to_owned());
2116
2117    let body_text = tab
2118        .evaluate(
2119            "document.body ? document.body.innerText : document.documentElement.innerText",
2120            false,
2121        )
2122        .ok()
2123        .and_then(|r| r.value)
2124        .and_then(|v| v.as_str().map(String::from))
2125        .unwrap_or_default();
2126
2127    PageContent {
2128        url,
2129        title,
2130        body_text: truncate(&body_text, 8000),
2131    }
2132}
2133
2134fn resolve_url(url: &str, base_url: &str) -> String {
2135    if url.starts_with("http://") || url.starts_with("https://") {
2136        return url.to_owned();
2137    }
2138    let base = base_url.trim_end_matches('/');
2139    if url.starts_with('/') {
2140        format!("{base}{url}")
2141    } else {
2142        format!("{base}/{url}")
2143    }
2144}
2145
2146/// Human-readable label for a step, used when steps are skipped after an
2147/// earlier failure.
2148fn step_label(step: &TestStep) -> String {
2149    match step {
2150        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2151        TestStep::Click { target, .. } => format!("[click] {target}"),
2152        TestStep::Type { target, .. } => format!("[type] {target}"),
2153        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2154        TestStep::Assert {
2155            definition,
2156            preset,
2157            prompt,
2158            ..
2159        } => definition.as_ref().map_or_else(
2160            || {
2161                preset.as_ref().map_or_else(
2162                    || {
2163                        prompt.as_ref().map_or_else(
2164                            || "[assert]".to_owned(),
2165                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2166                        )
2167                    },
2168                    |p| format!("[assert] {p}"),
2169                )
2170            },
2171            |d| format!("[assert] {d}"),
2172        ),
2173        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2174        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2175        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2176    }
2177}
2178
2179/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2180#[must_use]
2181const fn step_kind_label(step: &TestStep) -> &'static str {
2182    match step {
2183        TestStep::Navigate { .. } => "navigate",
2184        TestStep::Click { .. } => "click",
2185        TestStep::Type { .. } => "type",
2186        TestStep::Wait { .. } => "wait",
2187        TestStep::Assert { .. } => "assert",
2188        TestStep::Screenshot { .. } => "screenshot",
2189        TestStep::Agent { .. } => "agent",
2190        TestStep::Mcp { .. } => "mcp",
2191    }
2192}
2193
2194// ── Support types ──────────────────────────────────────────────────────
2195
2196#[derive(Default)]
2197struct TestRunResult {
2198    passed: u32,
2199    failed: u32,
2200    skipped: u32,
2201    total: u32,
2202    details: Vec<StepResult>,
2203}
2204
2205struct PageContent {
2206    url: String,
2207    title: String,
2208    body_text: String,
2209}