Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// One detected layout defect (`layout_no_issues` preset).
26#[derive(Debug, serde::Deserialize)]
27struct LayoutIssue {
28    #[serde(rename = "type")]
29    issue_type: String,
30    element: String,
31    detail: String,
32}
33
34/// In-browser DOM layout scan for `layout_no_issues`.
35///
36/// Geometry-only checks (no LLM, no pixels):
37/// 1. page horizontal overflow (`scrollWidth` > viewport width);
38/// 2. elements outside the viewport that scrolling cannot reveal
39///    (fixed elements off-screen, left/negative overflow, right-edge
40///    overflow beyond the horizontally scrollable area, and bottom
41///    overflow on a page that cannot scroll down) — below-the-fold
42///    content on a tall scrollable page is normal flow, NOT a defect;
43/// 3. text clipped by `overflow: hidden` containers whose content
44///    is measurably larger than the box;
45/// 4. interactive elements (buttons/links/inputs) whose center point
46///    is covered by a different element that would intercept the click.
47///
48/// Elements whose class matches a configured ignore prefix (default:
49/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
50/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
51/// skipped — they are intentionally 1x1 / off-screen. The prefix list
52/// is injected at `__IGNORE_CLASSES__` from
53/// [`ScenarioConfig::layout_ignore_classes`].
54///
55/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
56/// sticky headers) is excluded by the position/relation filters.
57const LAYOUT_SCAN_JS: &str = r#"
58(() => {
59  const issues = [];
60  const push = (type, el, detail) => {
61    if (issues.length >= 30) return;
62    let element = el.tagName.toLowerCase();
63    if (el.id) element += '#' + el.id;
64    else if (typeof el.className === 'string' && el.className.trim())
65      element += '.' + el.className.trim().split(/\s+/).join('.');
66    issues.push({ type, element, detail: String(detail).slice(0, 220) });
67  };
68  const vw = document.documentElement.clientWidth || window.innerWidth;
69  const vh = document.documentElement.clientHeight || window.innerHeight;
70  if (!vw || !vh) return JSON.stringify(issues);
71  const de = document.documentElement;
72  // 1. Page-level horizontal overflow.
73  if (de.scrollWidth > vw + 2)
74    push('page-overflow-x', de,
75      'page scrollWidth ' + de.scrollWidth + ' exceeds viewport width ' + vw);
76  // Class-name prefixes to skip (injected; default = Angular CDK
77  // screen-reader helpers, which are intentionally 1x1 / off-screen).
78  const ignorePrefixes = __IGNORE_CLASSES__;
79  const isIgnored = (el) => {
80    if (typeof el.className !== 'string' || !el.className.trim()) return false;
81    const classes = el.className.trim().split(/\s+/);
82    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
83  };
84  const vScrollable = de.scrollHeight > vh + 2;
85  const hScrollable = de.scrollWidth > vw + 2;
86  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
87  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
88  const hasContent = (el) =>
89    ((el.textContent || '').trim().length > 0) ||
90    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
91  // 2. Elements outside the viewport that scrolling cannot reveal.
92  for (const el of all) {
93    const cs = getComputedStyle(el);
94    if (!visible(cs)) continue;
95    if (isIgnored(el)) continue;
96    const r = el.getBoundingClientRect();
97    if (r.width < 2 || r.height < 2) continue;
98    if (!hasContent(el) && el.children.length === 0) continue;
99    if (cs.position === 'fixed') {
100      // Fixed elements never move with the scroll: any edge outside the
101      // viewport is unreachable content and therefore a defect.
102      const overTop = -r.top;
103      const overLeft = -r.left;
104      const overRight = r.right - vw;
105      const overBottom = r.bottom - vh;
106      if (overTop > 2 || overLeft > 2 || overRight > 2 || overBottom > 2) {
107        let where = '';
108        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
109        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
110        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
111        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
112        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
113        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
114        push('element-out-of-viewport', el,
115          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
116      }
117      continue;
118    }
119    if (cs.position === 'sticky') continue;
120    // Negative left overflow cannot be reached by scrolling (scrollLeft
121    // never goes below 0).
122    if (r.left < -2) {
123      push('element-out-of-viewport', el,
124        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
125      continue;
126    }
127    // Negative top with the page at the top means the element sits above
128    // the document origin — also unreachable.
129    if (r.top < -2 && de.scrollTop <= 2) {
130      push('element-out-of-viewport', el,
131        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
132      continue;
133    }
134    const overRight = r.right - vw;
135    // Right-edge overflow is only a defect when the page cannot scroll
136    // horizontally to reveal it (or the element sticks out past the
137    // scrollable content width itself).
138    if (overRight > 2 && (!hScrollable || r.right > de.scrollWidth + 2)) {
139      push('element-out-of-viewport', el,
140        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
141      continue;
142    }
143    // Below-the-fold content on a scrollable page is normal (tall landing
144    // pages); only flag bottom overflow the user can never scroll to.
145    const overBottom = r.bottom - vh;
146    if (overBottom > 2 && (!vScrollable || r.bottom > de.scrollHeight + 2)) {
147      push('element-out-of-viewport', el,
148        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
149    }
150  }
151  // 3. Text clipped by overflow:hidden containers.
152  for (const el of all) {
153    if (isIgnored(el)) continue;
154    const cs = getComputedStyle(el);
155    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
156    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
157    if (!(el.textContent || '').trim()) continue;
158    push('text-clipped', el,
159      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
160      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
161  }
162  // 4. Interactive elements covered by a different element.
163  const interactive =
164    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
165  const targets = document.querySelectorAll(interactive);
166  for (const el of targets) {
167    if (isIgnored(el)) continue;
168    const r = el.getBoundingClientRect();
169    if (r.width < 6 || r.height < 6) continue;
170    const cs = getComputedStyle(el);
171    if (!visible(cs)) continue;
172    const cx = r.left + r.width / 2;
173    const cy = r.top + r.height / 2;
174    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
175    const top = document.elementFromPoint(cx, cy);
176    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
177    if (isIgnored(top)) continue;
178    const tcs = getComputedStyle(top);
179    if (!visible(tcs)) continue;
180    if (tcs.pointerEvents === 'none') continue;
181    const tr = top.getBoundingClientRect();
182    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
183    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
184      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
185    push('element-overlap', el,
186      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
187      ' is covered by <' + tname + '>');
188  }
189  return JSON.stringify(issues);
190})()
191"#;
192
193/// How long the CDP connection stays open after the browser goes quiet.
194///
195/// `headless_chrome` ships a 30s default and tears down the entire connection
196/// when no traffic arrives for that long; a run must own its connection for
197/// its full duration instead.
198const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
199
200/// Executes a [`Scenario`] against a real browser with optional LLM
201/// assistance for element targeting and assertions.
202pub struct ScenarioRunner {
203    config: ScenarioConfig,
204    definitions: HashMap<String, AssertDefinition>,
205    llm: LlmConfig,
206    timeout: Duration,
207    viewport_width: u32,
208    viewport_height: u32,
209    /// The viewport currently applied in the browser (CDP emulation).
210    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
211    applied_viewport: std::cell::Cell<(u32, u32)>,
212    endpoints: EndpointRegistry,
213    usage: Arc<UsageTracker>,
214    budgets: BudgetTracker,
215    /// Directory for failure artifacts (screenshots).
216    artifacts_dir: PathBuf,
217    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
218    reporter: Arc<Reporter>,
219    /// The test + step index currently executing, for LLM-call events.
220    current_step: std::cell::RefCell<Option<(String, u32)>>,
221}
222
223/// Aggregated results from a scenario run.
224#[derive(Debug, Default)]
225pub struct RunReport {
226    /// Number of tests that passed.
227    pub tests_passed: u32,
228    /// Number of tests that failed.
229    pub tests_failed: u32,
230    /// Number of steps that passed.
231    pub passed: u32,
232    /// Number of steps that failed.
233    pub failed: u32,
234    /// Number of steps that were skipped.
235    pub skipped: u32,
236    /// Per-step details.
237    pub details: Vec<StepResult>,
238}
239
240/// Result of a single step execution.
241#[derive(Debug)]
242pub struct StepResult {
243    /// The step name.
244    pub name: String,
245    /// Whether the step passed, failed, or was skipped.
246    pub status: StepStatus,
247    /// Human-readable result message.
248    pub message: String,
249}
250
251/// Outcome for a single step.
252pub use crate::events::StepStatus;
253
254/// Predefined assertion preset definition.
255struct AssertPreset {
256    name: &'static str,
257    system: &'static str,
258    user_template: &'static str,
259}
260
261/// Built-in assertion presets.
262#[allow(clippy::literal_string_with_formatting_args)]
263const ASSERTION_PRESETS: &[AssertPreset] = &[
264    AssertPreset {
265        name: "no_error_on_page",
266        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
267        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
268    },
269    AssertPreset {
270        name: "text_visible",
271        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
272        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
273    },
274    AssertPreset {
275        name: "element_exists",
276        system: "You are a QA tester. Check if a described UI element exists on a web page.",
277        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
278    },
279    AssertPreset {
280        name: "visual_no_issues",
281        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
282        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
283    },
284    AssertPreset {
285        name: "visual_no_overlaps",
286        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
287        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
288    },
289    AssertPreset {
290        name: "visual_text_visible",
291        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
292        user_template: "Inspect the attached screenshot and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
293    },
294    AssertPreset {
295        name: "layout_no_issues",
296        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
297        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
298    },
299];
300
301impl ScenarioRunner {
302    /// Creates a new runner with the given scenario configuration and
303    /// assertion definitions.
304    #[must_use]
305    #[allow(clippy::needless_pass_by_value)]
306    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
307        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
308    }
309
310    /// Creates a runner that reports run events through the given reporter.
311    #[must_use]
312    #[allow(clippy::needless_pass_by_value)]
313    pub fn with_reporter(
314        scenario_config: ScenarioConfig,
315        definitions: Vec<AssertDefinition>,
316        reporter: Arc<Reporter>,
317    ) -> Self {
318        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
319            &scenario_config,
320        ));
321        let llm = LlmConfig {
322            url: scenario_config
323                .llm_url
324                .clone()
325                .unwrap_or_else(crate::llm_base_url),
326            model: scenario_config
327                .llm_model
328                .clone()
329                .unwrap_or_else(crate::llm_model),
330            api_key: scenario_config
331                .llm_api_key
332                .clone()
333                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
334            headers: if scenario_config.llm_headers.is_empty() {
335                crate::parse_headers_env()
336            } else {
337                scenario_config.llm_headers.clone()
338            },
339            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
340            temperature: scenario_config.temperature,
341            thinking: scenario_config.thinking,
342            model_params: scenario_config.model_params.clone(),
343            max_attempts: crate::default_llm_attempts(),
344            provider: crate::scenario::Provider::Openai,
345            deployment: None,
346            api_version: None,
347            auth: crate::scenario::AuthConfig::default(),
348            header_commands: std::collections::HashMap::new(),
349            aws: crate::scenario::AwsConfig::default(),
350        };
351        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
352        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
353        let defs_map: HashMap<String, AssertDefinition> = definitions
354            .into_iter()
355            .map(|d| (d.name.clone(), d))
356            .collect();
357
358        Self {
359            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
360            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
361            viewport_height: scenario_config.viewport_height.unwrap_or(720),
362            applied_viewport: std::cell::Cell::new((0, 0)),
363            config: scenario_config.clone(),
364            definitions: defs_map,
365            llm,
366            endpoints,
367            usage: Arc::new(UsageTracker::new()),
368            budgets,
369            artifacts_dir: PathBuf::from(
370                scenario_config
371                    .artifacts_dir
372                    .unwrap_or_else(|| "artifacts".to_owned()),
373            ),
374            reporter,
375            current_step: std::cell::RefCell::new(None),
376        }
377    }
378
379    /// Emits an event; a sink failure degrades to a console warning so a
380    /// broken log file can never mask the run itself.
381    fn emit_event(&self, event: &TestEvent) {
382        if let Err(err) = self.reporter.emit(event) {
383            use std::io::Write as _;
384            let mut out = std::io::stderr().lock();
385            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
386        }
387    }
388
389    /// Returns a clone of the [`UsageTracker`] for reporting.
390    #[must_use]
391    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
392        Arc::clone(&self.usage)
393    }
394
395    /// Returns a reference to the [`BudgetTracker`].
396    #[must_use]
397    pub const fn budget_tracker(&self) -> &BudgetTracker {
398        &self.budgets
399    }
400
401    /// Executes all test groups in the scenario and returns a report.
402    ///
403    /// # Errors
404    ///
405    /// Returns an error if the browser fails to launch.
406    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
407    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
408        let mut report = RunReport::default();
409
410        if tests.is_empty() {
411            self.reporter.warn("No tests defined in scenario.");
412            self.emit_event(&TestEvent::RunFinished {
413                tests_passed: 0,
414                tests_failed: 0,
415                steps_passed: 0,
416                steps_failed: 0,
417                steps_skipped: 0,
418                total_cost: 0.0,
419                total_tokens: 0,
420                total_calls: 0,
421            });
422            return Ok(report);
423        }
424
425        self.emit_event(&TestEvent::RunStarted {
426            total_tests: tests.len() as u32,
427        });
428
429        let browser_headless = self.config.browser_headless.unwrap_or(true);
430
431        let launch_opts = LaunchOptions {
432            headless: browser_headless,
433            window_size: Some((self.viewport_width, self.viewport_height)),
434            sandbox: false,
435            // headless_chrome defaults this to 30s and shuts down the whole CDP
436            // connection when no messages arrive for that long. A scenario can
437            // easily exceed 30s of browser silence (slow LLM targeting/assertion
438            // calls, page waits, budget checks between steps), after which every
439            // remaining step fails with "Unable to make method calls because
440            // underlying connection is closed" — one quiet gap kills the run.
441            // Open-ended scenarios must own the connection for their full
442            // duration, so keep it alive for 6 hours.
443            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
444            ..LaunchOptions::default()
445        };
446
447        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
448        let tab = browser.new_tab().context("failed to open browser tab")?;
449        let _ = tab.set_default_timeout(self.timeout);
450
451        // Start MCP server if configured
452        #[cfg(feature = "mcp-server")]
453        if let Some(ref mcp_cfg) = self.config.mcp_server {
454            if mcp_cfg.enabled {
455                let port = mcp_cfg.port;
456                std::thread::spawn(move || {
457                    let _ = crate::mcp_server::start_mcp_server(port);
458                });
459            }
460        }
461        #[cfg(not(feature = "mcp-server"))]
462        if let Some(mcp_cfg) = &self.config.mcp_server {
463            if mcp_cfg.enabled {
464                self.reporter
465                    .warn("MCP server configured but 'mcp-server' feature not enabled");
466            }
467        }
468
469        // Start A2A agent server if configured
470        #[cfg(feature = "a2a-server")]
471        if let Some(ref a2a_cfg) = self.config.a2a_server {
472            if a2a_cfg.enabled {
473                let port = a2a_cfg.port;
474                tokio::spawn(crate::a2a_server::start_a2a_server(port));
475            }
476        }
477        #[cfg(not(feature = "a2a-server"))]
478        if let Some(a2a_cfg) = &self.config.a2a_server {
479            if a2a_cfg.enabled {
480                self.reporter
481                    .warn("A2A server configured but 'a2a-server' feature not enabled");
482            }
483        }
484
485        for test in tests {
486            self.emit_event(&TestEvent::TestStarted {
487                test: test.name.clone(),
488            });
489
490            self.usage.reset_per_test();
491
492            let test_started = Instant::now();
493            let test_result = self.run_test(test, &tab);
494            let duration_ms = test_started.elapsed().as_millis() as u64;
495            let usage = self.usage.current_test_snapshot();
496            self.usage.commit_test(&test.name);
497
498            self.emit_event(&TestEvent::TestFinished {
499                test: test.name.clone(),
500                passed: test_result.passed,
501                failed: test_result.failed,
502                skipped: test_result.skipped,
503                duration_ms,
504                cost: usage.total_cost,
505                tokens: usage.total_tokens,
506                calls: usage.total_calls,
507            });
508
509            if test_result.failed == 0 && test_result.total > 0 {
510                report.tests_passed += 1;
511            } else if test_result.total > 0 {
512                report.tests_failed += 1;
513            }
514
515            report.passed += test_result.passed;
516            report.failed += test_result.failed;
517            report.skipped += test_result.skipped;
518            report.details.extend(test_result.details);
519        }
520
521        let global = self.usage.global_snapshot();
522        self.emit_event(&TestEvent::RunFinished {
523            tests_passed: report.tests_passed,
524            tests_failed: report.tests_failed,
525            steps_passed: report.passed,
526            steps_failed: report.failed,
527            steps_skipped: report.skipped,
528            total_cost: global.total_cost,
529            total_tokens: global.total_tokens,
530            total_calls: global.total_calls,
531        });
532
533        Ok(report)
534    }
535
536    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
537    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
538        let base_url = test
539            .base_url
540            .clone()
541            .or_else(|| self.config.base_url.clone())
542            .unwrap_or_else(crate::base_url);
543
544        // Per-test viewport override: switch the browser via CDP
545        // device-metrics emulation before this test runs.
546        let vw = test.viewport_width.unwrap_or(self.viewport_width);
547        let vh = test.viewport_height.unwrap_or(self.viewport_height);
548        if self.applied_viewport.get() != (vw, vh) {
549            self.apply_viewport(tab, vw, vh);
550            self.applied_viewport.set((vw, vh));
551        }
552
553        // Per-test isolation: every test starts from its own start_url
554        // (unless auto_navigate is disabled), so a test never inherits the
555        // previous test's page state.
556        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
557
558        let start_url = test
559            .start_url
560            .clone()
561            .or_else(|| self.config.start_url.clone())
562            .unwrap_or_else(|| "/dashboard".to_owned());
563
564        if auto_navigate {
565            let full_url = resolve_url(&start_url, &base_url);
566            self.reporter.debug(format!("auto-navigate: {full_url}"));
567            let _ = tab.navigate_to(&full_url);
568            let _ = tab.wait_until_navigated();
569            std::thread::sleep(Duration::from_secs(4));
570        }
571
572        let mut result = TestRunResult::default();
573
574        for (step_index, step) in test.steps.iter().enumerate() {
575            result.total += 1;
576
577            let wait_ms = match step {
578                TestStep::Navigate { wait_after_ms, .. }
579                | TestStep::Click { wait_after_ms, .. }
580                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
581                _ => None,
582            };
583
584            self.current_step
585                .replace(Some((test.name.clone(), step_index as u32)));
586            self.emit_event(&TestEvent::StepStarted {
587                test: test.name.clone(),
588                index: step_index as u32,
589                label: step_label(step),
590            });
591            let step_started = Instant::now();
592
593            let mut step_result = match step {
594                TestStep::Navigate { url, .. } => {
595                    let full_url = resolve_url(url, &base_url);
596                    run_navigate_step(&full_url, tab)
597                }
598                TestStep::Click {
599                    target,
600                    selector,
601                    endpoint,
602                    idempotent,
603                    ..
604                } => self.run_click(
605                    target,
606                    selector.as_deref(),
607                    endpoint.as_deref(),
608                    test.endpoint.as_deref(),
609                    *idempotent,
610                    tab,
611                ),
612                TestStep::Type {
613                    target,
614                    text,
615                    selector,
616                    endpoint,
617                    idempotent,
618                    ..
619                } => self.run_type(
620                    target,
621                    text,
622                    selector.as_deref(),
623                    endpoint.as_deref(),
624                    test.endpoint.as_deref(),
625                    *idempotent,
626                    tab,
627                ),
628                TestStep::Wait {
629                    target,
630                    selector,
631                    text,
632                    timeout_ms,
633                    endpoint,
634                    idempotent,
635                } => self.run_wait(
636                    target,
637                    selector.as_deref(),
638                    text.as_deref(),
639                    *timeout_ms,
640                    endpoint.as_deref(),
641                    test.endpoint.as_deref(),
642                    *idempotent,
643                    tab,
644                ),
645                TestStep::Assert {
646                    definition,
647                    preset,
648                    prompt,
649                    assert_text,
650                    endpoint,
651                    screenshot,
652                } => self.run_assert(
653                    definition.as_deref(),
654                    preset.as_deref(),
655                    prompt.as_deref(),
656                    assert_text.as_deref(),
657                    *screenshot,
658                    endpoint.as_deref(),
659                    test.endpoint.as_deref(),
660                    tab,
661                ),
662                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
663                TestStep::Agent {
664                    agent,
665                    task,
666                    definition,
667                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
668                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
669            };
670
671            // Failure diagnostics: capture the page state and a screenshot so
672            // CI logs say WHAT the page looked like when the step failed,
673            // instead of a bare "timed out: The event waited for never came".
674            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
675                let state = diagnostics::capture(tab);
676                let screenshot = diagnostics::save_screenshot(
677                    tab,
678                    &self.artifacts_dir,
679                    &test.name,
680                    &test.name,
681                    step_index,
682                    step_kind_label(step),
683                );
684                step_result.message = format!(
685                    "{base} — {excerpt}",
686                    base = step_result.message,
687                    excerpt = diagnostics::inline_excerpt(&state),
688                );
689                (Some(diagnostics::full_context(&state)), screenshot)
690            } else {
691                (None, None)
692            };
693
694            let duration_ms = step_started.elapsed().as_millis() as u64;
695            self.emit_event(&TestEvent::StepFinished {
696                test: test.name.clone(),
697                index: step_index as u32,
698                label: step_result.name.clone(),
699                status: step_result.status,
700                duration_ms,
701                message: step_result.message.clone(),
702                diagnostics: diagnostics_block,
703                screenshot: screenshot_path,
704            });
705            self.current_step.replace(None);
706
707            match step_result.status {
708                StepStatus::Passed => result.passed += 1,
709                StepStatus::Failed => result.failed += 1,
710                StepStatus::Skipped => result.skipped += 1,
711            }
712
713            // Fail fast: the first failed step ends the test and the
714            // remaining steps are reported as skipped (no LLM budget is
715            // burned asserting against a page that is already known broken).
716            if step_result.status == StepStatus::Failed
717                && !self.config.continue_on_failure
718                && step_index + 1 < test.steps.len()
719            {
720                self.emit_event(&TestEvent::Warning {
721                    message: format!(
722                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
723                        test.steps.len() - step_index - 1
724                    ),
725                });
726                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
727                    let skipped_index = step_index + 1 + offset;
728                    let label = step_label(skipped);
729                    result.total += 1;
730                    result.skipped += 1;
731                    self.emit_event(&TestEvent::StepStarted {
732                        test: test.name.clone(),
733                        index: skipped_index as u32,
734                        label: label.clone(),
735                    });
736                    self.emit_event(&TestEvent::StepFinished {
737                        test: test.name.clone(),
738                        index: skipped_index as u32,
739                        label,
740                        status: StepStatus::Skipped,
741                        duration_ms: 0,
742                        message: "skipped: previous step failed".into(),
743                        diagnostics: None,
744                        screenshot: None,
745                    });
746                    result.details.push(StepResult {
747                        name: step_label(skipped),
748                        status: StepStatus::Skipped,
749                        message: "skipped: previous step failed".into(),
750                    });
751                }
752                result.details.push(step_result);
753                return result;
754            }
755
756            // Check per-test budget after each step
757            let test_usage = self.usage.current_test_snapshot();
758            let global_usage = self.usage.global_snapshot();
759            let budget_status = self.budgets.check_all(
760                &test.name,
761                &test_usage,
762                &global_usage,
763                test.budget.as_ref(),
764            );
765            match budget_status {
766                BudgetStatus::HardExceeded { message, .. } => {
767                    self.emit_event(&TestEvent::Warning {
768                        message: format!("budget exceeded: {message}"),
769                    });
770                    result.details.push(StepResult {
771                        name: "[budget]".into(),
772                        status: StepStatus::Failed,
773                        message,
774                    });
775                    result.failed += 1;
776                    return result;
777                }
778                BudgetStatus::SoftExceeded { message, .. } => {
779                    self.emit_event(&TestEvent::Warning {
780                        message: format!("budget warning: {message}"),
781                    });
782                }
783                BudgetStatus::Ok => {}
784            }
785
786            if let Some(ms) = wait_ms {
787                std::thread::sleep(Duration::from_millis(ms));
788            }
789
790            result.details.push(step_result);
791        }
792
793        result
794    }
795
796    /// Applies a viewport size to the current tab via CDP
797    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
798    /// overrides and the viewport matrix. The initial window size set at
799    /// browser launch is replaced by emulation; failures are logged but
800    /// do not fail the test (a mismatched viewport only weakens coverage).
801    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
802        use headless_chrome::protocol::cdp::Emulation;
803        let _ = self;
804        let params = Emulation::SetDeviceMetricsOverride {
805            width,
806            height,
807            device_scale_factor: 1.0,
808            mobile: false,
809            scale: None,
810            screen_width: Some(width),
811            screen_height: Some(height),
812            position_x: None,
813            position_y: None,
814            dont_set_visible_size: None,
815            screen_orientation: None,
816            viewport: None,
817            display_feature: None,
818            device_posture: None,
819        };
820        self.reporter.debug(format!("viewport: {width}x{height}"));
821        if let Err(e) = tab.call_method(params) {
822            self.reporter
823                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
824        }
825    }
826
827    // ── step handlers ───────────────────────────────────────────────────
828
829    #[allow(clippy::too_many_lines)]
830    fn run_click(
831        &self,
832        target: &str,
833        selector_override: Option<&str>,
834        step_endpoint: Option<&str>,
835        test_endpoint: Option<&str>,
836        idempotent: bool,
837        tab: &Tab,
838    ) -> StepResult {
839        let name = format!("[click] {target}");
840        let selector = match self.resolve_selector(
841            selector_override,
842            target,
843            step_endpoint,
844            test_endpoint,
845            tab,
846        ) {
847            Ok(s) => s,
848            Err(msg) => {
849                if idempotent {
850                    return StepResult {
851                        name,
852                        status: StepStatus::Skipped,
853                        message: format!("skipped (idempotent): no target found — {msg}"),
854                    };
855                }
856                return StepResult {
857                    name,
858                    status: StepStatus::Failed,
859                    message: msg,
860                };
861            }
862        };
863
864        // Idempotent steps probe briefly: a missing target means the
865        // action was already done / not applicable (e.g. an
866        // already-authenticated session), and skipping is the success
867        // path, not a failure.
868        let probe_secs = if idempotent { 5 } else { 10 };
869        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
870            Ok(element) => match element.click() {
871                Ok(_) => StepResult {
872                    name,
873                    status: StepStatus::Passed,
874                    message: format!("clicked {selector}"),
875                },
876                Err(e) => StepResult {
877                    name,
878                    status: StepStatus::Failed,
879                    message: format!("click failed on {selector}: {e}"),
880                },
881            },
882            Err(e) if idempotent => StepResult {
883                name,
884                status: StepStatus::Skipped,
885                message: format!("skipped (idempotent): element {selector} not present — {e}"),
886            },
887            Err(e) => StepResult {
888                name,
889                status: StepStatus::Failed,
890                message: format!("element {selector} not found: {e}"),
891            },
892        }
893    }
894
895    #[allow(clippy::too_many_arguments)]
896    fn run_type(
897        &self,
898        target: &str,
899        text: &str,
900        selector_override: Option<&str>,
901        step_endpoint: Option<&str>,
902        test_endpoint: Option<&str>,
903        idempotent: bool,
904        tab: &Tab,
905    ) -> StepResult {
906        let name = format!("[type] {target}");
907        let selector = match self.resolve_selector(
908            selector_override,
909            target,
910            step_endpoint,
911            test_endpoint,
912            tab,
913        ) {
914            Ok(s) => s,
915            Err(msg) => {
916                if idempotent {
917                    return StepResult {
918                        name,
919                        status: StepStatus::Skipped,
920                        message: format!("skipped (idempotent): no target found — {msg}"),
921                    };
922                }
923                return StepResult {
924                    name,
925                    status: StepStatus::Failed,
926                    message: msg,
927                };
928            }
929        };
930
931        let probe_secs = if idempotent { 5 } else { 10 };
932        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
933            Ok(element) => {
934                if let Err(e) = element.click() {
935                    return StepResult {
936                        name,
937                        status: StepStatus::Failed,
938                        message: format!("click to focus {selector} failed: {e}"),
939                    };
940                }
941
942                let js = format!(
943                    "document.querySelector('{}').value = '';",
944                    selector.replace('\'', "\\'")
945                );
946                let _ = tab.evaluate(&js, false);
947
948                match element.type_into(text) {
949                    Ok(_) => StepResult {
950                        name,
951                        status: StepStatus::Passed,
952                        message: format!("typed {text:?} into {selector}"),
953                    },
954                    Err(e) => StepResult {
955                        name,
956                        status: StepStatus::Failed,
957                        message: format!("type into {selector} failed: {e}"),
958                    },
959                }
960            }
961            Err(e) if idempotent => StepResult {
962                name,
963                status: StepStatus::Skipped,
964                message: format!("skipped (idempotent): element {selector} not present — {e}"),
965            },
966            Err(e) => StepResult {
967                name,
968                status: StepStatus::Failed,
969                message: format!("element {selector} not found: {e}"),
970            },
971        }
972    }
973
974    #[allow(clippy::too_many_arguments)]
975    #[allow(clippy::too_many_lines)]
976    fn run_wait(
977        &self,
978        target: &str,
979        selector_override: Option<&str>,
980        text: Option<&str>,
981        timeout_ms: Option<u64>,
982        step_endpoint: Option<&str>,
983        test_endpoint: Option<&str>,
984        idempotent: bool,
985        tab: &Tab,
986    ) -> StepResult {
987        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
988        let step_name = format!("[wait] {target}");
989
990        // Resolve an explicit selector only (text-only waits are LLM-free).
991        let selector = match selector_override {
992            Some(s) => Some(s.to_owned()),
993            None if text.is_some() => None,
994            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
995                Ok(s) => Some(s),
996                Err(msg) => {
997                    if idempotent {
998                        return StepResult {
999                            name: step_name,
1000                            status: StepStatus::Skipped,
1001                            message: format!("skipped (idempotent): no target found — {msg}"),
1002                        };
1003                    }
1004                    return StepResult {
1005                        name: step_name,
1006                        status: StepStatus::Failed,
1007                        message: msg,
1008                    };
1009                }
1010            },
1011        };
1012
1013        if text.is_some() {
1014            let sel_js = selector
1015                .as_deref()
1016                .map(crate::selectors::selector_matches_js);
1017            let text_js = text.map(|t| {
1018                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1019                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1020            });
1021
1022            let deadline = Instant::now() + timeout;
1023            loop {
1024                let sel_ok = sel_js
1025                    .as_ref()
1026                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1027                let text_ok = text_js
1028                    .as_ref()
1029                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1030                if sel_ok && text_ok {
1031                    let mut what = Vec::new();
1032                    if let Some(sel) = &selector {
1033                        what.push(format!("found {sel}"));
1034                    }
1035                    if let Some(t) = text {
1036                        what.push(format!("text {t:?} visible"));
1037                    }
1038                    return StepResult {
1039                        name: step_name,
1040                        status: StepStatus::Passed,
1041                        message: what.join(" and "),
1042                    };
1043                }
1044                if Instant::now() >= deadline {
1045                    let mut what = Vec::new();
1046                    if let Some(sel) = &selector {
1047                        what.push(sel.clone());
1048                    }
1049                    if let Some(t) = text {
1050                        what.push(format!("text {t:?}"));
1051                    }
1052                    let message = format!(
1053                        "wait for {} timed out after {}ms: the event waited for never came",
1054                        what.join(" / "),
1055                        timeout.as_millis(),
1056                    );
1057                    if idempotent {
1058                        return StepResult {
1059                            name: step_name,
1060                            status: StepStatus::Skipped,
1061                            message: format!("skipped (idempotent): {message}"),
1062                        };
1063                    }
1064                    return StepResult {
1065                        name: step_name,
1066                        status: StepStatus::Failed,
1067                        message,
1068                    };
1069                }
1070                std::thread::sleep(Duration::from_millis(250));
1071            }
1072        }
1073
1074        match selector.as_deref() {
1075            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1076                Ok(_) => StepResult {
1077                    name: step_name,
1078                    status: StepStatus::Passed,
1079                    message: format!("found {sel}"),
1080                },
1081                Err(e) if idempotent => StepResult {
1082                    name: step_name,
1083                    status: StepStatus::Skipped,
1084                    message: format!(
1085                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1086                        timeout.as_millis()
1087                    ),
1088                },
1089                Err(e) => StepResult {
1090                    name: step_name,
1091                    status: StepStatus::Failed,
1092                    message: format!(
1093                        "wait for {sel} timed out after {}ms: {e}",
1094                        timeout.as_millis()
1095                    ),
1096                },
1097            },
1098            None => StepResult {
1099                name: step_name,
1100                status: StepStatus::Failed,
1101                message: "wait step has neither selector nor text".into(),
1102            },
1103        }
1104    }
1105
1106    #[allow(clippy::too_many_arguments)]
1107    fn run_assert(
1108        &self,
1109        definition: Option<&str>,
1110        preset: Option<&str>,
1111        prompt: Option<&str>,
1112        assert_text: Option<&str>,
1113        screenshot: bool,
1114        step_endpoint: Option<&str>,
1115        test_endpoint: Option<&str>,
1116        tab: &Tab,
1117    ) -> StepResult {
1118        std::thread::sleep(Duration::from_millis(500));
1119
1120        let page_content = get_page_text(tab);
1121
1122        // Vision attach: capture the viewport once per assert step and hand
1123        // the JPEG data URL to the preset/prompt evaluation below.
1124        let image = if screenshot {
1125            let endpoint = self
1126                .endpoints
1127                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1128            if !endpoint.vision {
1129                return StepResult {
1130                    name: "[assert]".into(),
1131                    status: StepStatus::Failed,
1132                    message: format!(
1133                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1134                        name = endpoint.name
1135                    ),
1136                };
1137            }
1138            match crate::vision::capture_screenshot_data_url(
1139                tab,
1140                self.config
1141                    .screenshot_max_dimension
1142                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1143            ) {
1144                Ok(data_url) => Some(data_url),
1145                Err(e) => {
1146                    return StepResult {
1147                        name: "[assert]".into(),
1148                        status: StepStatus::Failed,
1149                        message: format!("screenshot capture failed: {e}"),
1150                    };
1151                }
1152            }
1153        } else {
1154            None
1155        };
1156
1157        if let Some(def_name) = definition {
1158            if let Some(def) = self.definitions.get(def_name) {
1159                return self.run_assert_def(
1160                    def,
1161                    &page_content,
1162                    image.as_deref(),
1163                    step_endpoint,
1164                    test_endpoint,
1165                    tab,
1166                );
1167            }
1168            return StepResult {
1169                name: format!("[assert] {def_name}"),
1170                status: StepStatus::Failed,
1171                message: format!("definition '{def_name}' not found"),
1172            };
1173        }
1174
1175        if let Some(preset_name) = preset {
1176            // Deterministic DOM layout scan — runs JS in the browser and
1177            // never calls the LLM (free, fast, no pixel budget).
1178            if preset_name == "layout_no_issues" {
1179                return self.run_layout_preset(tab);
1180            }
1181            return self.run_preset(
1182                preset_name,
1183                assert_text,
1184                &page_content,
1185                image.as_deref(),
1186                step_endpoint,
1187                test_endpoint,
1188            );
1189        }
1190
1191        if let Some(prompt_text) = prompt {
1192            return self.run_custom(
1193                prompt_text,
1194                &page_content,
1195                image.as_deref(),
1196                step_endpoint,
1197                test_endpoint,
1198            );
1199        }
1200
1201        StepResult {
1202            name: "[assert]".into(),
1203            status: StepStatus::Skipped,
1204            message: "no definition, preset, or prompt specified".into(),
1205        }
1206    }
1207
1208    fn run_assert_def(
1209        &self,
1210        def: &AssertDefinition,
1211        page_content: &PageContent,
1212        image: Option<&str>,
1213        step_endpoint: Option<&str>,
1214        test_endpoint: Option<&str>,
1215        tab: &Tab,
1216    ) -> StepResult {
1217        // Agent-based definition: delegate to an A2A agent
1218        if let Some(ref agent) = def.agent {
1219            if image.is_some() {
1220                return StepResult {
1221                    name: format!("[assert] {}", def.name),
1222                    status: StepStatus::Failed,
1223                    message: "agent-backed assertions do not support screenshots".into(),
1224                };
1225            }
1226            let task = def
1227                .task_template
1228                .as_deref()
1229                .unwrap_or("Evaluate the assertion")
1230                .replace("{url}", &page_content.url)
1231                .replace("{title}", &page_content.title)
1232                .replace("{content}", &page_content.body_text)
1233                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1234
1235            return self.run_agent_step(agent, &task, &def.name);
1236        }
1237
1238        // Custom preset: system + user_template provided in the definition
1239        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1240            return self.run_custom_preset(
1241                &def.name,
1242                system,
1243                template,
1244                def.assert_text.as_deref(),
1245                page_content,
1246                image,
1247                step_endpoint,
1248                test_endpoint,
1249            );
1250        }
1251
1252        def.preset.as_ref().map_or_else(
1253            || {
1254                def.prompt.as_ref().map_or_else(
1255                    || StepResult {
1256                        name: format!("[assert] {}", def.name),
1257                        status: StepStatus::Failed,
1258                        message: "definition has no preset, prompt, or system+user_template".into(),
1259                    },
1260                    |prompt| {
1261                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1262                    },
1263                )
1264            },
1265            |preset_name| {
1266                if preset_name == "layout_no_issues" {
1267                    return self.run_layout_preset(tab);
1268                }
1269                self.run_preset(
1270                    preset_name,
1271                    def.assert_text.as_deref(),
1272                    page_content,
1273                    image,
1274                    step_endpoint,
1275                    test_endpoint,
1276                )
1277            },
1278        )
1279    }
1280
1281    #[allow(clippy::too_many_arguments)]
1282    fn run_custom_preset(
1283        &self,
1284        name: &str,
1285        system: &str,
1286        template: &str,
1287        assert_text: Option<&str>,
1288        page_content: &PageContent,
1289        image: Option<&str>,
1290        step_endpoint: Option<&str>,
1291        test_endpoint: Option<&str>,
1292    ) -> StepResult {
1293        let user_prompt = template
1294            .replace("{url}", &page_content.url)
1295            .replace("{title}", &page_content.title)
1296            .replace("{content}", &page_content.body_text)
1297            .replace("{expected_text}", assert_text.unwrap_or(""))
1298            .replace("{description}", "");
1299
1300        // Custom preset definitions frequently forget the {content}
1301        // placeholder — without it the LLM has no page to evaluate and
1302        // answers "I can't determine that without seeing the page". Always
1303        // append the page context unless the template already references it.
1304        let user_prompt = if template.contains("{content}") {
1305            user_prompt
1306        } else {
1307            format!(
1308                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1309                url = page_content.url,
1310                title = page_content.title,
1311                content = page_content.body_text,
1312            )
1313        };
1314
1315        self.reporter
1316            .debug(format!("assert: {name} (custom preset)"));
1317
1318        let chain = self
1319            .endpoints
1320            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1321        let sys = system.to_owned();
1322
1323        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1324
1325        response.map_or_else(
1326            |e| StepResult {
1327                name: format!("[assert] {name}"),
1328                status: StepStatus::Failed,
1329                message: format!("LLM assertion call failed: {e}"),
1330            },
1331            |(lr, idx)| {
1332                self.usage.record_llm_call(
1333                    &chain[idx].name,
1334                    chain[idx],
1335                    lr.usage.prompt_tokens,
1336                    lr.usage.completion_tokens,
1337                );
1338                let content_lower = lr.content.to_lowercase().trim().to_owned();
1339                if content_lower.starts_with("pass") {
1340                    StepResult {
1341                        name: format!("[assert] {name}"),
1342                        status: StepStatus::Passed,
1343                        message: "PASS".into(),
1344                    }
1345                } else {
1346                    StepResult {
1347                        name: format!("[assert] {name}"),
1348                        status: StepStatus::Failed,
1349                        message: lr.content,
1350                    }
1351                }
1352            },
1353        )
1354    }
1355
1356    fn run_preset(
1357        &self,
1358        preset_name: &str,
1359        assert_text: Option<&str>,
1360        page_content: &PageContent,
1361        image: Option<&str>,
1362        step_endpoint: Option<&str>,
1363        test_endpoint: Option<&str>,
1364    ) -> StepResult {
1365        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1366            return StepResult {
1367                name: format!("[assert] {preset_name}"),
1368                status: StepStatus::Failed,
1369                message: format!("unknown assertion preset: {preset_name}"),
1370            };
1371        };
1372        if preset_name.starts_with("visual_") && image.is_none() {
1373            return StepResult {
1374                name: format!("[assert] {preset_name}"),
1375                status: StepStatus::Failed,
1376                message: format!(
1377                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1378                ),
1379            };
1380        }
1381
1382        let user_prompt = preset
1383            .user_template
1384            .replace("{url}", &page_content.url)
1385            .replace("{title}", &page_content.title)
1386            .replace("{content}", &page_content.body_text)
1387            .replace("{expected_text}", assert_text.unwrap_or(""))
1388            .replace("{description}", "");
1389
1390        // Same safety net as custom presets: never let the LLM answer with
1391        // no page context at all.
1392        let user_prompt = if preset.user_template.contains("{content}") {
1393            user_prompt
1394        } else {
1395            format!(
1396                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1397                url = page_content.url,
1398                title = page_content.title,
1399                content = page_content.body_text,
1400            )
1401        };
1402
1403        self.reporter.debug(format!("assert: {preset_name}"));
1404
1405        let chain = self
1406            .endpoints
1407            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1408        let sys = preset.system.to_owned();
1409
1410        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1411
1412        response.map_or_else(
1413            |e| StepResult {
1414                name: format!("[assert] {preset_name}"),
1415                status: StepStatus::Failed,
1416                message: format!("LLM assertion call failed: {e}"),
1417            },
1418            |(lr, idx)| {
1419                self.usage.record_llm_call(
1420                    &chain[idx].name,
1421                    chain[idx],
1422                    lr.usage.prompt_tokens,
1423                    lr.usage.completion_tokens,
1424                );
1425                let content_lower = lr.content.to_lowercase().trim().to_owned();
1426                if content_lower.starts_with("pass") {
1427                    StepResult {
1428                        name: format!("[assert] {preset_name}"),
1429                        status: StepStatus::Passed,
1430                        message: "PASS".into(),
1431                    }
1432                } else {
1433                    StepResult {
1434                        name: format!("[assert] {preset_name}"),
1435                        status: StepStatus::Failed,
1436                        message: lr.content,
1437                    }
1438                }
1439            },
1440        )
1441    }
1442
1443    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1444    ///
1445    /// Evaluates the layout-scan JS in the page and fails with the list of
1446    /// detected issues: horizontal page overflow, visible elements sticking
1447    /// out of the viewport, text clipped by `overflow: hidden` containers,
1448    /// and interactive elements covered by other elements. No LLM call —
1449    /// checks are geometry-based so the check is free, deterministic, and
1450    /// safe to run on every page × viewport variant.
1451    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1452        let name = "[assert] layout_no_issues".to_owned();
1453        self.reporter
1454            .debug("assert: layout_no_issues (DOM layout scan)");
1455        let js = LAYOUT_SCAN_JS.replace(
1456            "__IGNORE_CLASSES__",
1457            &serde_json::to_string(&self.config.layout_ignore_classes)
1458                .unwrap_or_else(|_| "[]".to_owned()),
1459        );
1460        let result = tab.evaluate(&js, false);
1461        let json_str = match result {
1462            Ok(r) => r
1463                .value
1464                .as_ref()
1465                .and_then(|v| v.as_str().map(String::from))
1466                .unwrap_or_else(|| "[]".to_owned()),
1467            Err(e) => {
1468                return StepResult {
1469                    name,
1470                    status: StepStatus::Failed,
1471                    message: format!("layout scan JS failed: {e}"),
1472                };
1473            }
1474        };
1475        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1476        if issues.is_empty() {
1477            return StepResult {
1478                name,
1479                status: StepStatus::Passed,
1480                message: "PASS — no layout defects detected".into(),
1481            };
1482        }
1483        let mut lines: Vec<String> = issues
1484            .iter()
1485            .take(10)
1486            .map(|i| {
1487                format!(
1488                    "- [{type_}] {element}: {detail}",
1489                    type_ = i.issue_type,
1490                    element = i.element,
1491                    detail = i.detail
1492                )
1493            })
1494            .collect();
1495        if issues.len() > 10 {
1496            lines.push(format!("- … and {} more", issues.len() - 10));
1497        }
1498        StepResult {
1499            name,
1500            status: StepStatus::Failed,
1501            message: format!(
1502                "FAIL — {} layout defect(s) detected:\n{}",
1503                issues.len(),
1504                lines.join("\n")
1505            ),
1506        }
1507    }
1508
1509    fn run_custom(
1510        &self,
1511        prompt: &str,
1512        page_content: &PageContent,
1513        image: Option<&str>,
1514        step_endpoint: Option<&str>,
1515        test_endpoint: Option<&str>,
1516    ) -> StepResult {
1517        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1518
1519        let mut user = format!(
1520            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1521            url = page_content.url,
1522            title = page_content.title,
1523            content = page_content.body_text,
1524        );
1525        if image.is_some() {
1526            user.push_str(
1527                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1528            );
1529        }
1530
1531        self.reporter.debug("custom assert");
1532
1533        let chain = self
1534            .endpoints
1535            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1536        let sys = system.to_owned();
1537
1538        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1539
1540        response.map_or_else(
1541            |e| StepResult {
1542                name: "[assert] custom".into(),
1543                status: StepStatus::Failed,
1544                message: format!("LLM assertion call failed: {e}"),
1545            },
1546            |(lr, idx)| {
1547                self.usage.record_llm_call(
1548                    &chain[idx].name,
1549                    chain[idx],
1550                    lr.usage.prompt_tokens,
1551                    lr.usage.completion_tokens,
1552                );
1553                let content_lower = lr.content.to_lowercase().trim().to_owned();
1554                if content_lower.starts_with("pass") {
1555                    StepResult {
1556                        name: "[assert] custom".into(),
1557                        status: StepStatus::Passed,
1558                        message: "PASS".into(),
1559                    }
1560                } else {
1561                    StepResult {
1562                        name: "[assert] custom".into(),
1563                        status: StepStatus::Failed,
1564                        message: lr.content,
1565                    }
1566                }
1567            },
1568        )
1569    }
1570
1571    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1572        let path = path.unwrap_or("screenshot.png");
1573
1574        match tab.capture_screenshot(
1575            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1576            None,
1577            None,
1578            true,
1579        ) {
1580            Ok(data) => {
1581                if let Err(e) = std::fs::write(path, &data) {
1582                    return StepResult {
1583                        name: format!("[screenshot] {path}"),
1584                        status: StepStatus::Failed,
1585                        message: format!("failed to write screenshot: {e}"),
1586                    };
1587                }
1588                StepResult {
1589                    name: format!("[screenshot] {path}"),
1590                    status: StepStatus::Passed,
1591                    message: format!("saved to {path}"),
1592                }
1593            }
1594            Err(e) => StepResult {
1595                name: format!("[screenshot] {path}"),
1596                status: StepStatus::Failed,
1597                message: format!("screenshot failed: {e}"),
1598            },
1599        }
1600    }
1601
1602    /// Runs an A2A agent step.
1603    #[allow(clippy::literal_string_with_formatting_args)]
1604    fn run_agent(
1605        &self,
1606        agent_name: &str,
1607        task: &str,
1608        definition: Option<&str>,
1609        _test_endpoint: Option<&str>,
1610    ) -> StepResult {
1611        // If a definition is specified, look up the task template
1612        let resolved_task = if let Some(def_name) = definition {
1613            if let Some(def) = self.definitions.get(def_name) {
1614                let tmpl = def.task_template.as_deref().unwrap_or(task);
1615                tmpl.replace("{task}", task)
1616            } else {
1617                return StepResult {
1618                    name: format!("[agent] {def_name}"),
1619                    status: StepStatus::Failed,
1620                    message: format!("definition '{def_name}' not found"),
1621                };
1622            }
1623        } else {
1624            task.to_owned()
1625        };
1626
1627        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1628    }
1629
1630    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1631        let Some(ep) = self.endpoints.get(agent_name) else {
1632            return StepResult {
1633                name: format!("[agent] {display_name}"),
1634                status: StepStatus::Failed,
1635                message: format!("agent endpoint '{agent_name}' not found"),
1636            };
1637        };
1638
1639        if ep.url.is_empty() {
1640            return StepResult {
1641                name: format!("[agent] {display_name}"),
1642                status: StepStatus::Failed,
1643                message: format!("agent endpoint '{agent_name}' has no URL"),
1644            };
1645        }
1646
1647        self.reporter.debug(format!("agent {agent_name}: {task}"));
1648
1649        let url = ep.url.clone();
1650        let client = A2aClient::new(&url, self.timeout);
1651        let task_clone = task.to_owned();
1652
1653        let response = std::thread::spawn(move || {
1654            let rt = tokio::runtime::Builder::new_current_thread()
1655                .enable_all()
1656                .build()
1657                .unwrap();
1658            rt.block_on(client.send_task(&task_clone))
1659        })
1660        .join()
1661        .unwrap();
1662
1663        // Record the flat-cost call
1664        self.usage.record_flat_call(agent_name, ep);
1665
1666        match response {
1667            Ok(text) => {
1668                let clean = text.trim().to_owned();
1669                let lower = clean.to_lowercase();
1670                if lower.starts_with("pass") {
1671                    StepResult {
1672                        name: format!("[agent] {display_name}"),
1673                        status: StepStatus::Passed,
1674                        message: format!("PASS: {clean}"),
1675                    }
1676                } else if lower.starts_with("fail") {
1677                    StepResult {
1678                        name: format!("[agent] {display_name}"),
1679                        status: StepStatus::Failed,
1680                        message: clean,
1681                    }
1682                } else {
1683                    StepResult {
1684                        name: format!("[agent] {display_name}"),
1685                        status: StepStatus::Passed,
1686                        message: format!("response: {clean}"),
1687                    }
1688                }
1689            }
1690            Err(e) => StepResult {
1691                name: format!("[agent] {display_name}"),
1692                status: StepStatus::Failed,
1693                message: format!("agent call failed: {e}"),
1694            },
1695        }
1696    }
1697
1698    /// Runs an MCP tool call step.
1699    fn run_mcp(
1700        &self,
1701        server_name: &str,
1702        tool_name: &str,
1703        args: Option<&serde_json::Value>,
1704    ) -> StepResult {
1705        let Some(ep) = self.endpoints.get(server_name) else {
1706            return StepResult {
1707                name: format!("[mcp] {server_name}:{tool_name}"),
1708                status: StepStatus::Failed,
1709                message: format!("MCP server endpoint '{server_name}' not found"),
1710            };
1711        };
1712
1713        let cmd = ep.command.as_deref().unwrap_or("");
1714        if cmd.is_empty() {
1715            return StepResult {
1716                name: format!("[mcp] {server_name}:{tool_name}"),
1717                status: StepStatus::Failed,
1718                message: format!("MCP server '{server_name}' has no command configured"),
1719            };
1720        }
1721
1722        self.reporter
1723            .debug(format!("mcp {server_name} {tool_name}"));
1724
1725        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1726
1727        let command = cmd.to_owned();
1728        let args_vec = ep.args.clone();
1729        let tool = tool_name.to_owned();
1730
1731        let response = std::thread::spawn(move || {
1732            let mut mcp_client =
1733                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1734            mcp_client
1735                .call_tool(&tool, &args_val)
1736                .map_err(|e| e.to_string())
1737        })
1738        .join()
1739        .unwrap();
1740
1741        // Record the flat-cost call
1742        self.usage.record_flat_call(server_name, ep);
1743
1744        match response {
1745            Ok(result) => {
1746                if result.isError {
1747                    StepResult {
1748                        name: format!("[mcp] {server_name}:{tool_name}"),
1749                        status: StepStatus::Failed,
1750                        message: result.to_string(),
1751                    }
1752                } else {
1753                    StepResult {
1754                        name: format!("[mcp] {server_name}:{tool_name}"),
1755                        status: StepStatus::Passed,
1756                        message: result.to_string(),
1757                    }
1758                }
1759            }
1760            Err(e) => StepResult {
1761                name: format!("[mcp] {server_name}:{tool_name}"),
1762                status: StepStatus::Failed,
1763                message: format!("MCP call failed: {e}"),
1764            },
1765        }
1766    }
1767
1768    // ── helpers ──────────────────────────────────────────────────────────
1769
1770    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1771    /// the runner's default LLM config for any unset fields.
1772    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1773        LlmConfig {
1774            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
1775                // Bedrock builds its endpoint from the resolved AWS region
1776                // when no URL is given — never inherit the default LLM URL.
1777                endpoint.url.clone()
1778            } else if endpoint.url.is_empty() {
1779                self.llm.url.clone()
1780            } else {
1781                endpoint.url.clone()
1782            },
1783            model: endpoint
1784                .model
1785                .clone()
1786                .unwrap_or_else(|| self.llm.model.clone()),
1787            api_key: endpoint
1788                .api_key
1789                .clone()
1790                .or_else(|| self.llm.api_key.clone()),
1791            headers: if endpoint.headers.is_empty() {
1792                self.llm.headers.clone()
1793            } else {
1794                endpoint.headers.clone()
1795            },
1796            timeout: self.llm.timeout,
1797            temperature: self.llm.temperature,
1798            thinking: self.llm.thinking,
1799            model_params: self.llm.model_params.clone(),
1800            max_attempts: endpoint.max_attempts.max(1),
1801            provider: endpoint.provider,
1802            deployment: endpoint.deployment.clone(),
1803            api_version: endpoint.api_version.clone(),
1804            auth: endpoint.auth.clone(),
1805            header_commands: endpoint.header_commands.clone(),
1806            aws: endpoint.aws.clone(),
1807        }
1808    }
1809
1810    /// Runs a single LLM call against an ordered endpoint chain (primary +
1811    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
1812    /// the first endpoint that answers wins. Returns the response together
1813    /// with the chain index of the answering endpoint (0 = primary) so the
1814    /// caller can attribute usage to the correct endpoint.
1815    ///
1816    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
1817    /// shows duration, tokens, cost and the answering endpoint per call.
1818    #[allow(clippy::cast_possible_truncation)]
1819    fn llm_call_chain(
1820        &self,
1821        chain: &[&ResolvedEndpoint],
1822        system: &str,
1823        user: &str,
1824        image: Option<&str>,
1825        purpose: &str,
1826    ) -> Result<(crate::costs::LlmResponse, usize), String> {
1827        if chain.is_empty() {
1828            return Err("empty LLM endpoint chain".into());
1829        }
1830        let primary = self.build_llm_for_endpoint(chain[0]);
1831        let fallbacks: Vec<LlmConfig> = chain[1..]
1832            .iter()
1833            .map(|e| self.build_llm_for_endpoint(e))
1834            .collect();
1835
1836        let (test, index) = self
1837            .current_step
1838            .borrow()
1839            .as_ref()
1840            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
1841        let primary_endpoint = chain[0].name.clone();
1842        let primary_model = primary.model.clone();
1843        self.emit_event(&TestEvent::LlmCallStarted {
1844            test: test.clone(),
1845            index,
1846            endpoint: primary_endpoint.clone(),
1847            model: primary_model.clone(),
1848            purpose: purpose.to_owned(),
1849        });
1850
1851        let started = Instant::now();
1852        let sys = system.to_owned();
1853        let user = user.to_owned();
1854        let image = image.map(str::to_owned);
1855
1856        let result = std::thread::spawn(move || {
1857            let rt = tokio::runtime::Builder::new_current_thread()
1858                .enable_all()
1859                .build()
1860                .unwrap();
1861            let call = async {
1862                match image.as_deref() {
1863                    Some(img) => {
1864                        llm_chat_vision_with_usage_chain(&primary, &fallbacks, &sys, &user, img)
1865                            .await
1866                    }
1867                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
1868                }
1869            };
1870            rt.block_on(call)
1871        })
1872        .join()
1873        .unwrap();
1874
1875        let duration_ms = started.elapsed().as_millis() as u64;
1876        match result {
1877            Ok((lr, idx)) => {
1878                let cost = calculate_llm_cost(
1879                    chain[idx],
1880                    lr.usage.prompt_tokens,
1881                    lr.usage.completion_tokens,
1882                );
1883                let answering = chain[idx].name.clone();
1884                let model = chain[idx]
1885                    .model
1886                    .clone()
1887                    .unwrap_or_else(|| primary_model.clone());
1888                self.emit_event(&TestEvent::LlmCallFinished {
1889                    test,
1890                    index,
1891                    endpoint: answering,
1892                    model,
1893                    purpose: purpose.to_owned(),
1894                    ok: true,
1895                    duration_ms,
1896                    input_tokens: lr.usage.prompt_tokens,
1897                    output_tokens: lr.usage.completion_tokens,
1898                    cost,
1899                    error: None,
1900                });
1901                Ok((lr, idx))
1902            }
1903            Err(e) => {
1904                self.emit_event(&TestEvent::LlmCallFinished {
1905                    test,
1906                    index,
1907                    endpoint: primary_endpoint,
1908                    model: primary_model,
1909                    purpose: purpose.to_owned(),
1910                    ok: false,
1911                    duration_ms,
1912                    input_tokens: 0,
1913                    output_tokens: 0,
1914                    cost: 0.0,
1915                    error: Some(e.clone()),
1916                });
1917                Err(e)
1918            }
1919        }
1920    }
1921
1922    /// Resolves a CSS selector for the target element. Uses the explicit
1923    /// `selector` if provided, otherwise asks the LLM to find the element
1924    /// from the natural language `target` description and page DOM.
1925    ///
1926    /// LLM responses are sanitized and verified against the live page: a
1927    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
1928    /// immediately with the raw LLM output, and a selector that matches
1929    /// nothing triggers one retry with feedback before failing.
1930    #[allow(clippy::too_many_lines)]
1931    fn resolve_selector(
1932        &self,
1933        css_override: Option<&str>,
1934        target: &str,
1935        step_endpoint: Option<&str>,
1936        test_endpoint: Option<&str>,
1937        tab: &Tab,
1938    ) -> Result<String, String> {
1939        if let Some(explicit) = css_override {
1940            return Ok(explicit.to_owned());
1941        }
1942
1943        let dom_info = extract_dom_info(tab)?;
1944        let page_content = get_page_text(tab);
1945
1946        let system = concat!(
1947            "You are a browser automation selector generator. ",
1948            "Given a web page's content and interactive elements, ",
1949            "return ONLY the best CSS selector for the described element. ",
1950            "Output nothing except the CSS selector. ",
1951            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
1952            "[name=\"...\"], tag.class, tag. ",
1953            "Never output explanations, markdown, or extra text."
1954        );
1955
1956        let user = format!(
1957            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
1958            page_content.url,
1959            page_content.title,
1960            truncate(&page_content.body_text, 4000),
1961            dom_info,
1962            target,
1963        );
1964
1965        let retry_user = format!(
1966            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
1967            "The selector must match at least one element currently present on the page.",
1968            page_content.url,
1969            page_content.title,
1970            truncate(&page_content.body_text, 4000),
1971            dom_info,
1972            target,
1973        );
1974
1975        self.reporter.debug(format!("LLM targeting: {target}"));
1976
1977        let chain = self
1978            .endpoints
1979            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
1980        let sys = system.to_owned();
1981
1982        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
1983
1984        let first = call_llm(&user);
1985        let (lr, idx) = match first {
1986            Ok(lr) => lr,
1987            Err(e) => {
1988                return Err(format!("LLM element targeting failed: {e}"));
1989            }
1990        };
1991        self.usage.record_llm_call(
1992            &chain[idx].name,
1993            chain[idx],
1994            lr.usage.prompt_tokens,
1995            lr.usage.completion_tokens,
1996        );
1997        let clean = sanitize_selector(&lr.content);
1998        self.reporter.debug(format!("resolved selector: {clean}"));
1999
2000        if selector_is_useless(&clean) {
2001            return Err(format!(
2002                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2003                raw = lr.content.trim(),
2004            ));
2005        }
2006        if let Err(reason) = validate_selector(&clean) {
2007            return Err(format!(
2008                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2009                raw = lr.content.trim(),
2010            ));
2011        }
2012        if !selector_matches(tab, &clean).unwrap_or(false) {
2013            // One retry with feedback: flaky models occasionally invent a
2014            // selector that does not exist on the page.
2015            self.reporter.warn(format!(
2016                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2017            ));
2018            let second = call_llm(&retry_user);
2019            let (lr2, idx2) = match second {
2020                Ok(lr2) => lr2,
2021                Err(e) => {
2022                    return Err(format!(
2023                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2024                    ));
2025                }
2026            };
2027            self.usage.record_llm_call(
2028                &chain[idx2].name,
2029                chain[idx2],
2030                lr2.usage.prompt_tokens,
2031                lr2.usage.completion_tokens,
2032            );
2033            let clean2 = sanitize_selector(&lr2.content);
2034            self.reporter
2035                .debug(format!("resolved selector (retry): {clean2}"));
2036            if selector_is_useless(&clean2) {
2037                return Err(format!(
2038                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2039                    raw = lr2.content.trim(),
2040                    excerpt = truncate(&page_content.body_text, 300),
2041                ));
2042            }
2043            if !selector_matches(tab, &clean2).unwrap_or(false) {
2044                return Err(format!(
2045                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2046                ));
2047            }
2048            return Ok(clean2);
2049        }
2050
2051        Ok(clean)
2052    }
2053}
2054
2055/// Evaluates a JS expression that is expected to return a boolean.
2056fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2057    tab.evaluate(js, false)
2058        .map_err(|e| format!("evaluate failed: {e}"))?
2059        .value
2060        .and_then(|v| v.as_bool())
2061        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2062}
2063
2064/// Checks whether a CSS selector matches at least one current element.
2065fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2066    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2067}
2068
2069// ── Free helper functions ──────────────────────────────────────────────
2070
2071fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2072    let name = format!("[navigate] {full_url}");
2073    match tab.navigate_to(full_url) {
2074        Ok(_) => {
2075            let _ = tab.wait_until_navigated();
2076            StepResult {
2077                name,
2078                status: StepStatus::Passed,
2079                message: format!("navigated to {full_url}"),
2080            }
2081        }
2082        Err(e) => StepResult {
2083            name,
2084            status: StepStatus::Failed,
2085            message: format!("navigation failed: {e}"),
2086        },
2087    }
2088}
2089
2090fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2091    let result = tab
2092        .evaluate(DOM_EXTRACT_JS, false)
2093        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2094
2095    let json_str = result
2096        .value
2097        .as_ref()
2098        .and_then(|v| v.as_str())
2099        .unwrap_or("[]");
2100
2101    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2102
2103    if elements.is_empty() {
2104        return Ok("(no interactive elements found)".to_owned());
2105    }
2106
2107    Ok(elements.join("\n"))
2108}
2109
2110fn get_page_text(tab: &Tab) -> PageContent {
2111    let url = tab.get_url();
2112
2113    let title = tab
2114        .evaluate("document.title", false)
2115        .ok()
2116        .and_then(|r| r.value)
2117        .and_then(|v| v.as_str().map(String::from))
2118        .unwrap_or_else(|| "unknown".to_owned());
2119
2120    let body_text = tab
2121        .evaluate(
2122            "document.body ? document.body.innerText : document.documentElement.innerText",
2123            false,
2124        )
2125        .ok()
2126        .and_then(|r| r.value)
2127        .and_then(|v| v.as_str().map(String::from))
2128        .unwrap_or_default();
2129
2130    PageContent {
2131        url,
2132        title,
2133        body_text: truncate(&body_text, 8000),
2134    }
2135}
2136
2137fn resolve_url(url: &str, base_url: &str) -> String {
2138    if url.starts_with("http://") || url.starts_with("https://") {
2139        return url.to_owned();
2140    }
2141    let base = base_url.trim_end_matches('/');
2142    if url.starts_with('/') {
2143        format!("{base}{url}")
2144    } else {
2145        format!("{base}/{url}")
2146    }
2147}
2148
2149/// Human-readable label for a step, used when steps are skipped after an
2150/// earlier failure.
2151fn step_label(step: &TestStep) -> String {
2152    match step {
2153        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2154        TestStep::Click { target, .. } => format!("[click] {target}"),
2155        TestStep::Type { target, .. } => format!("[type] {target}"),
2156        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2157        TestStep::Assert {
2158            definition,
2159            preset,
2160            prompt,
2161            ..
2162        } => definition.as_ref().map_or_else(
2163            || {
2164                preset.as_ref().map_or_else(
2165                    || {
2166                        prompt.as_ref().map_or_else(
2167                            || "[assert]".to_owned(),
2168                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2169                        )
2170                    },
2171                    |p| format!("[assert] {p}"),
2172                )
2173            },
2174            |d| format!("[assert] {d}"),
2175        ),
2176        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2177        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2178        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2179    }
2180}
2181
2182/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2183#[must_use]
2184const fn step_kind_label(step: &TestStep) -> &'static str {
2185    match step {
2186        TestStep::Navigate { .. } => "navigate",
2187        TestStep::Click { .. } => "click",
2188        TestStep::Type { .. } => "type",
2189        TestStep::Wait { .. } => "wait",
2190        TestStep::Assert { .. } => "assert",
2191        TestStep::Screenshot { .. } => "screenshot",
2192        TestStep::Agent { .. } => "agent",
2193        TestStep::Mcp { .. } => "mcp",
2194    }
2195}
2196
2197// ── Support types ──────────────────────────────────────────────────────
2198
2199#[derive(Default)]
2200struct TestRunResult {
2201    passed: u32,
2202    failed: u32,
2203    skipped: u32,
2204    total: u32,
2205    details: Vec<StepResult>,
2206}
2207
2208struct PageContent {
2209    url: String,
2210    title: String,
2211    body_text: String,
2212}