Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// One detected layout defect (`layout_no_issues` preset).
26#[derive(Debug, serde::Deserialize)]
27struct LayoutIssue {
28    #[serde(rename = "type")]
29    issue_type: String,
30    element: String,
31    detail: String,
32}
33
34/// In-browser DOM layout scan for `layout_no_issues`.
35///
36/// Geometry-only checks (no LLM, no pixels):
37/// 1. page horizontal overflow (`scrollWidth` > viewport width);
38/// 2. elements outside the viewport that scrolling cannot reveal
39///    (fixed elements off-screen, left/negative overflow, right-edge
40///    overflow beyond the horizontally scrollable area, and bottom
41///    overflow on a page that cannot scroll down) — below-the-fold
42///    content on a tall scrollable page is normal flow, NOT a defect;
43/// 3. text clipped by `overflow: hidden` containers whose content
44///    is measurably larger than the box;
45/// 4. interactive elements (buttons/links/inputs) whose center point
46///    is covered by a different element that would intercept the click.
47///
48/// Elements whose class matches a configured ignore prefix (default:
49/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
50/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
51/// skipped — they are intentionally 1x1 / off-screen. The prefix list
52/// is injected at `__IGNORE_CLASSES__` from
53/// [`ScenarioConfig::layout_ignore_classes`].
54///
55/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
56/// sticky headers) is excluded by the position/relation filters.
57const LAYOUT_SCAN_JS: &str = r#"
58(() => {
59  const issues = [];
60  const push = (type, el, detail) => {
61    if (issues.length >= 30) return;
62    let element = el.tagName.toLowerCase();
63    if (el.id) element += '#' + el.id;
64    else if (typeof el.className === 'string' && el.className.trim())
65      element += '.' + el.className.trim().split(/\s+/).join('.');
66    issues.push({ type, element, detail: String(detail).slice(0, 220) });
67  };
68  const vw = document.documentElement.clientWidth || window.innerWidth;
69  const vh = document.documentElement.clientHeight || window.innerHeight;
70  if (!vw || !vh) return JSON.stringify(issues);
71  const de = document.documentElement;
72  // 1. Page-level horizontal overflow.
73  if (de.scrollWidth > vw + 2)
74    push('page-overflow-x', de,
75      'page scrollWidth ' + de.scrollWidth + ' exceeds viewport width ' + vw);
76  // Class-name prefixes to skip (injected; default = Angular CDK
77  // screen-reader helpers, which are intentionally 1x1 / off-screen).
78  const ignorePrefixes = __IGNORE_CLASSES__;
79  const isIgnored = (el) => {
80    if (typeof el.className !== 'string' || !el.className.trim()) return false;
81    const classes = el.className.trim().split(/\s+/);
82    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
83  };
84  const vScrollable = de.scrollHeight > vh + 2;
85  const hScrollable = de.scrollWidth > vw + 2;
86  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
87  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
88  const hasContent = (el) =>
89    ((el.textContent || '').trim().length > 0) ||
90    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
91  // 2. Elements outside the viewport that scrolling cannot reveal.
92  for (const el of all) {
93    const cs = getComputedStyle(el);
94    if (!visible(cs)) continue;
95    if (isIgnored(el)) continue;
96    const r = el.getBoundingClientRect();
97    if (r.width < 2 || r.height < 2) continue;
98    if (!hasContent(el) && el.children.length === 0) continue;
99    if (cs.position === 'fixed') {
100      // Fixed elements never move with the scroll: any edge outside the
101      // viewport is unreachable content and therefore a defect.
102      const overTop = -r.top;
103      const overLeft = -r.left;
104      const overRight = r.right - vw;
105      const overBottom = r.bottom - vh;
106      if (overTop > 2 || overLeft > 2 || overRight > 2 || overBottom > 2) {
107        let where = '';
108        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
109        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
110        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
111        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
112        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
113        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
114        push('element-out-of-viewport', el,
115          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
116      }
117      continue;
118    }
119    if (cs.position === 'sticky') continue;
120    // Negative left overflow cannot be reached by scrolling (scrollLeft
121    // never goes below 0).
122    if (r.left < -2) {
123      push('element-out-of-viewport', el,
124        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
125      continue;
126    }
127    // Negative top with the page at the top means the element sits above
128    // the document origin — also unreachable.
129    if (r.top < -2 && de.scrollTop <= 2) {
130      push('element-out-of-viewport', el,
131        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
132      continue;
133    }
134    const overRight = r.right - vw;
135    // Right-edge overflow is only a defect when the page cannot scroll
136    // horizontally to reveal it (or the element sticks out past the
137    // scrollable content width itself).
138    if (overRight > 2 && (!hScrollable || r.right > de.scrollWidth + 2)) {
139      push('element-out-of-viewport', el,
140        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
141      continue;
142    }
143    // Below-the-fold content on a scrollable page is normal (tall landing
144    // pages); only flag bottom overflow the user can never scroll to.
145    const overBottom = r.bottom - vh;
146    if (overBottom > 2 && (!vScrollable || r.bottom > de.scrollHeight + 2)) {
147      push('element-out-of-viewport', el,
148        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
149    }
150  }
151  // 3. Text clipped by overflow:hidden containers.
152  for (const el of all) {
153    if (isIgnored(el)) continue;
154    const cs = getComputedStyle(el);
155    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
156    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
157    if (!(el.textContent || '').trim()) continue;
158    push('text-clipped', el,
159      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
160      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
161  }
162  // 4. Interactive elements covered by a different element.
163  const interactive =
164    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
165  const targets = document.querySelectorAll(interactive);
166  for (const el of targets) {
167    if (isIgnored(el)) continue;
168    const r = el.getBoundingClientRect();
169    if (r.width < 6 || r.height < 6) continue;
170    const cs = getComputedStyle(el);
171    if (!visible(cs)) continue;
172    const cx = r.left + r.width / 2;
173    const cy = r.top + r.height / 2;
174    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
175    const top = document.elementFromPoint(cx, cy);
176    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
177    if (isIgnored(top)) continue;
178    const tcs = getComputedStyle(top);
179    if (!visible(tcs)) continue;
180    if (tcs.pointerEvents === 'none') continue;
181    const tr = top.getBoundingClientRect();
182    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
183    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
184      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
185    push('element-overlap', el,
186      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
187      ' is covered by <' + tname + '>');
188  }
189  return JSON.stringify(issues);
190})()
191"#;
192
193/// How long the CDP connection stays open after the browser goes quiet.
194///
195/// `headless_chrome` ships a 30s default and tears down the entire connection
196/// when no traffic arrives for that long; a run must own its connection for
197/// its full duration instead.
198const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
199
200/// Executes a [`Scenario`] against a real browser with optional LLM
201/// assistance for element targeting and assertions.
202pub struct ScenarioRunner {
203    config: ScenarioConfig,
204    definitions: HashMap<String, AssertDefinition>,
205    llm: LlmConfig,
206    timeout: Duration,
207    viewport_width: u32,
208    viewport_height: u32,
209    /// The viewport currently applied in the browser (CDP emulation).
210    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
211    applied_viewport: std::cell::Cell<(u32, u32)>,
212    endpoints: EndpointRegistry,
213    usage: Arc<UsageTracker>,
214    budgets: BudgetTracker,
215    /// Directory for failure artifacts (screenshots).
216    artifacts_dir: PathBuf,
217    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
218    reporter: Arc<Reporter>,
219    /// The test + step index currently executing, for LLM-call events.
220    current_step: std::cell::RefCell<Option<(String, u32)>>,
221}
222
223/// Aggregated results from a scenario run.
224#[derive(Debug, Default)]
225pub struct RunReport {
226    /// Number of tests that passed.
227    pub tests_passed: u32,
228    /// Number of tests that failed.
229    pub tests_failed: u32,
230    /// Number of steps that passed.
231    pub passed: u32,
232    /// Number of steps that failed.
233    pub failed: u32,
234    /// Number of steps that were skipped.
235    pub skipped: u32,
236    /// Per-step details.
237    pub details: Vec<StepResult>,
238}
239
240/// Result of a single step execution.
241#[derive(Debug)]
242pub struct StepResult {
243    /// The step name.
244    pub name: String,
245    /// Whether the step passed, failed, or was skipped.
246    pub status: StepStatus,
247    /// Human-readable result message.
248    pub message: String,
249}
250
251/// Outcome for a single step.
252pub use crate::events::StepStatus;
253
254/// Predefined assertion preset definition.
255struct AssertPreset {
256    name: &'static str,
257    system: &'static str,
258    user_template: &'static str,
259}
260
261/// Built-in assertion presets.
262#[allow(clippy::literal_string_with_formatting_args)]
263const ASSERTION_PRESETS: &[AssertPreset] = &[
264    AssertPreset {
265        name: "no_error_on_page",
266        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
267        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
268    },
269    AssertPreset {
270        name: "text_visible",
271        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
272        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
273    },
274    AssertPreset {
275        name: "element_exists",
276        system: "You are a QA tester. Check if a described UI element exists on a web page.",
277        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
278    },
279    AssertPreset {
280        name: "visual_no_issues",
281        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
282        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
283    },
284    AssertPreset {
285        name: "visual_no_overlaps",
286        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
287        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
288    },
289    AssertPreset {
290        name: "visual_text_visible",
291        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
292        user_template: "Inspect the attached screenshot and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
293    },
294    AssertPreset {
295        name: "layout_no_issues",
296        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
297        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
298    },
299];
300
301impl ScenarioRunner {
302    /// Creates a new runner with the given scenario configuration and
303    /// assertion definitions.
304    #[must_use]
305    #[allow(clippy::needless_pass_by_value)]
306    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
307        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
308    }
309
310    /// Creates a runner that reports run events through the given reporter.
311    #[must_use]
312    #[allow(clippy::needless_pass_by_value)]
313    pub fn with_reporter(
314        scenario_config: ScenarioConfig,
315        definitions: Vec<AssertDefinition>,
316        reporter: Arc<Reporter>,
317    ) -> Self {
318        let llm = LlmConfig {
319            url: scenario_config
320                .llm_url
321                .clone()
322                .unwrap_or_else(crate::llm_base_url),
323            model: scenario_config
324                .llm_model
325                .clone()
326                .unwrap_or_else(crate::llm_model),
327            api_key: scenario_config
328                .llm_api_key
329                .clone()
330                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
331            headers: if scenario_config.llm_headers.is_empty() {
332                crate::parse_headers_env()
333            } else {
334                scenario_config.llm_headers.clone()
335            },
336            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
337            temperature: scenario_config.temperature,
338            thinking: scenario_config.thinking,
339            model_params: scenario_config.model_params.clone(),
340            max_attempts: crate::default_llm_attempts(),
341        };
342        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
343        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
344        let defs_map: HashMap<String, AssertDefinition> = definitions
345            .into_iter()
346            .map(|d| (d.name.clone(), d))
347            .collect();
348
349        Self {
350            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
351            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
352            viewport_height: scenario_config.viewport_height.unwrap_or(720),
353            applied_viewport: std::cell::Cell::new((0, 0)),
354            config: scenario_config.clone(),
355            definitions: defs_map,
356            llm,
357            endpoints,
358            usage: Arc::new(UsageTracker::new()),
359            budgets,
360            artifacts_dir: PathBuf::from(
361                scenario_config
362                    .artifacts_dir
363                    .unwrap_or_else(|| "artifacts".to_owned()),
364            ),
365            reporter,
366            current_step: std::cell::RefCell::new(None),
367        }
368    }
369
370    /// Emits an event; a sink failure degrades to a console warning so a
371    /// broken log file can never mask the run itself.
372    fn emit_event(&self, event: &TestEvent) {
373        if let Err(err) = self.reporter.emit(event) {
374            use std::io::Write as _;
375            let mut out = std::io::stderr().lock();
376            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
377        }
378    }
379
380    /// Returns a clone of the [`UsageTracker`] for reporting.
381    #[must_use]
382    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
383        Arc::clone(&self.usage)
384    }
385
386    /// Returns a reference to the [`BudgetTracker`].
387    #[must_use]
388    pub const fn budget_tracker(&self) -> &BudgetTracker {
389        &self.budgets
390    }
391
392    /// Executes all test groups in the scenario and returns a report.
393    ///
394    /// # Errors
395    ///
396    /// Returns an error if the browser fails to launch.
397    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
398    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
399        let mut report = RunReport::default();
400
401        if tests.is_empty() {
402            self.reporter.warn("No tests defined in scenario.");
403            self.emit_event(&TestEvent::RunFinished {
404                tests_passed: 0,
405                tests_failed: 0,
406                steps_passed: 0,
407                steps_failed: 0,
408                steps_skipped: 0,
409                total_cost: 0.0,
410                total_tokens: 0,
411                total_calls: 0,
412            });
413            return Ok(report);
414        }
415
416        self.emit_event(&TestEvent::RunStarted {
417            total_tests: tests.len() as u32,
418        });
419
420        let browser_headless = self.config.browser_headless.unwrap_or(true);
421
422        let launch_opts = LaunchOptions {
423            headless: browser_headless,
424            window_size: Some((self.viewport_width, self.viewport_height)),
425            sandbox: false,
426            // headless_chrome defaults this to 30s and shuts down the whole CDP
427            // connection when no messages arrive for that long. A scenario can
428            // easily exceed 30s of browser silence (slow LLM targeting/assertion
429            // calls, page waits, budget checks between steps), after which every
430            // remaining step fails with "Unable to make method calls because
431            // underlying connection is closed" — one quiet gap kills the run.
432            // Open-ended scenarios must own the connection for their full
433            // duration, so keep it alive for 6 hours.
434            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
435            ..LaunchOptions::default()
436        };
437
438        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
439        let tab = browser.new_tab().context("failed to open browser tab")?;
440        let _ = tab.set_default_timeout(self.timeout);
441
442        // Start MCP server if configured
443        #[cfg(feature = "mcp-server")]
444        if let Some(ref mcp_cfg) = self.config.mcp_server {
445            if mcp_cfg.enabled {
446                let port = mcp_cfg.port;
447                std::thread::spawn(move || {
448                    let _ = crate::mcp_server::start_mcp_server(port);
449                });
450            }
451        }
452        #[cfg(not(feature = "mcp-server"))]
453        if let Some(mcp_cfg) = &self.config.mcp_server {
454            if mcp_cfg.enabled {
455                self.reporter
456                    .warn("MCP server configured but 'mcp-server' feature not enabled");
457            }
458        }
459
460        // Start A2A agent server if configured
461        #[cfg(feature = "a2a-server")]
462        if let Some(ref a2a_cfg) = self.config.a2a_server {
463            if a2a_cfg.enabled {
464                let port = a2a_cfg.port;
465                tokio::spawn(crate::a2a_server::start_a2a_server(port));
466            }
467        }
468        #[cfg(not(feature = "a2a-server"))]
469        if let Some(a2a_cfg) = &self.config.a2a_server {
470            if a2a_cfg.enabled {
471                self.reporter
472                    .warn("A2A server configured but 'a2a-server' feature not enabled");
473            }
474        }
475
476        for test in tests {
477            self.emit_event(&TestEvent::TestStarted {
478                test: test.name.clone(),
479            });
480
481            self.usage.reset_per_test();
482
483            let test_started = Instant::now();
484            let test_result = self.run_test(test, &tab);
485            let duration_ms = test_started.elapsed().as_millis() as u64;
486            let usage = self.usage.current_test_snapshot();
487            self.usage.commit_test(&test.name);
488
489            self.emit_event(&TestEvent::TestFinished {
490                test: test.name.clone(),
491                passed: test_result.passed,
492                failed: test_result.failed,
493                skipped: test_result.skipped,
494                duration_ms,
495                cost: usage.total_cost,
496                tokens: usage.total_tokens,
497                calls: usage.total_calls,
498            });
499
500            if test_result.failed == 0 && test_result.total > 0 {
501                report.tests_passed += 1;
502            } else if test_result.total > 0 {
503                report.tests_failed += 1;
504            }
505
506            report.passed += test_result.passed;
507            report.failed += test_result.failed;
508            report.skipped += test_result.skipped;
509            report.details.extend(test_result.details);
510        }
511
512        let global = self.usage.global_snapshot();
513        self.emit_event(&TestEvent::RunFinished {
514            tests_passed: report.tests_passed,
515            tests_failed: report.tests_failed,
516            steps_passed: report.passed,
517            steps_failed: report.failed,
518            steps_skipped: report.skipped,
519            total_cost: global.total_cost,
520            total_tokens: global.total_tokens,
521            total_calls: global.total_calls,
522        });
523
524        Ok(report)
525    }
526
527    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
528    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
529        let base_url = test
530            .base_url
531            .clone()
532            .or_else(|| self.config.base_url.clone())
533            .unwrap_or_else(crate::base_url);
534
535        // Per-test viewport override: switch the browser via CDP
536        // device-metrics emulation before this test runs.
537        let vw = test.viewport_width.unwrap_or(self.viewport_width);
538        let vh = test.viewport_height.unwrap_or(self.viewport_height);
539        if self.applied_viewport.get() != (vw, vh) {
540            self.apply_viewport(tab, vw, vh);
541            self.applied_viewport.set((vw, vh));
542        }
543
544        // Per-test isolation: every test starts from its own start_url
545        // (unless auto_navigate is disabled), so a test never inherits the
546        // previous test's page state.
547        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
548
549        let start_url = test
550            .start_url
551            .clone()
552            .or_else(|| self.config.start_url.clone())
553            .unwrap_or_else(|| "/dashboard".to_owned());
554
555        if auto_navigate {
556            let full_url = resolve_url(&start_url, &base_url);
557            self.reporter.debug(format!("auto-navigate: {full_url}"));
558            let _ = tab.navigate_to(&full_url);
559            let _ = tab.wait_until_navigated();
560            std::thread::sleep(Duration::from_secs(4));
561        }
562
563        let mut result = TestRunResult::default();
564
565        for (step_index, step) in test.steps.iter().enumerate() {
566            result.total += 1;
567
568            let wait_ms = match step {
569                TestStep::Navigate { wait_after_ms, .. }
570                | TestStep::Click { wait_after_ms, .. }
571                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
572                _ => None,
573            };
574
575            self.current_step
576                .replace(Some((test.name.clone(), step_index as u32)));
577            self.emit_event(&TestEvent::StepStarted {
578                test: test.name.clone(),
579                index: step_index as u32,
580                label: step_label(step),
581            });
582            let step_started = Instant::now();
583
584            let mut step_result = match step {
585                TestStep::Navigate { url, .. } => {
586                    let full_url = resolve_url(url, &base_url);
587                    run_navigate_step(&full_url, tab)
588                }
589                TestStep::Click {
590                    target,
591                    selector,
592                    endpoint,
593                    idempotent,
594                    ..
595                } => self.run_click(
596                    target,
597                    selector.as_deref(),
598                    endpoint.as_deref(),
599                    test.endpoint.as_deref(),
600                    *idempotent,
601                    tab,
602                ),
603                TestStep::Type {
604                    target,
605                    text,
606                    selector,
607                    endpoint,
608                    idempotent,
609                    ..
610                } => self.run_type(
611                    target,
612                    text,
613                    selector.as_deref(),
614                    endpoint.as_deref(),
615                    test.endpoint.as_deref(),
616                    *idempotent,
617                    tab,
618                ),
619                TestStep::Wait {
620                    target,
621                    selector,
622                    text,
623                    timeout_ms,
624                    endpoint,
625                    idempotent,
626                } => self.run_wait(
627                    target,
628                    selector.as_deref(),
629                    text.as_deref(),
630                    *timeout_ms,
631                    endpoint.as_deref(),
632                    test.endpoint.as_deref(),
633                    *idempotent,
634                    tab,
635                ),
636                TestStep::Assert {
637                    definition,
638                    preset,
639                    prompt,
640                    assert_text,
641                    endpoint,
642                    screenshot,
643                } => self.run_assert(
644                    definition.as_deref(),
645                    preset.as_deref(),
646                    prompt.as_deref(),
647                    assert_text.as_deref(),
648                    *screenshot,
649                    endpoint.as_deref(),
650                    test.endpoint.as_deref(),
651                    tab,
652                ),
653                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
654                TestStep::Agent {
655                    agent,
656                    task,
657                    definition,
658                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
659                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
660            };
661
662            // Failure diagnostics: capture the page state and a screenshot so
663            // CI logs say WHAT the page looked like when the step failed,
664            // instead of a bare "timed out: The event waited for never came".
665            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
666                let state = diagnostics::capture(tab);
667                let screenshot = diagnostics::save_screenshot(
668                    tab,
669                    &self.artifacts_dir,
670                    &test.name,
671                    &test.name,
672                    step_index,
673                    step_kind_label(step),
674                );
675                step_result.message = format!(
676                    "{base} — {excerpt}",
677                    base = step_result.message,
678                    excerpt = diagnostics::inline_excerpt(&state),
679                );
680                (Some(diagnostics::full_context(&state)), screenshot)
681            } else {
682                (None, None)
683            };
684
685            let duration_ms = step_started.elapsed().as_millis() as u64;
686            self.emit_event(&TestEvent::StepFinished {
687                test: test.name.clone(),
688                index: step_index as u32,
689                label: step_result.name.clone(),
690                status: step_result.status,
691                duration_ms,
692                message: step_result.message.clone(),
693                diagnostics: diagnostics_block,
694                screenshot: screenshot_path,
695            });
696            self.current_step.replace(None);
697
698            match step_result.status {
699                StepStatus::Passed => result.passed += 1,
700                StepStatus::Failed => result.failed += 1,
701                StepStatus::Skipped => result.skipped += 1,
702            }
703
704            // Fail fast: the first failed step ends the test and the
705            // remaining steps are reported as skipped (no LLM budget is
706            // burned asserting against a page that is already known broken).
707            if step_result.status == StepStatus::Failed
708                && !self.config.continue_on_failure
709                && step_index + 1 < test.steps.len()
710            {
711                self.emit_event(&TestEvent::Warning {
712                    message: format!(
713                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
714                        test.steps.len() - step_index - 1
715                    ),
716                });
717                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
718                    let skipped_index = step_index + 1 + offset;
719                    let label = step_label(skipped);
720                    result.total += 1;
721                    result.skipped += 1;
722                    self.emit_event(&TestEvent::StepStarted {
723                        test: test.name.clone(),
724                        index: skipped_index as u32,
725                        label: label.clone(),
726                    });
727                    self.emit_event(&TestEvent::StepFinished {
728                        test: test.name.clone(),
729                        index: skipped_index as u32,
730                        label,
731                        status: StepStatus::Skipped,
732                        duration_ms: 0,
733                        message: "skipped: previous step failed".into(),
734                        diagnostics: None,
735                        screenshot: None,
736                    });
737                    result.details.push(StepResult {
738                        name: step_label(skipped),
739                        status: StepStatus::Skipped,
740                        message: "skipped: previous step failed".into(),
741                    });
742                }
743                result.details.push(step_result);
744                return result;
745            }
746
747            // Check per-test budget after each step
748            let test_usage = self.usage.current_test_snapshot();
749            let global_usage = self.usage.global_snapshot();
750            let budget_status = self.budgets.check_all(
751                &test.name,
752                &test_usage,
753                &global_usage,
754                test.budget.as_ref(),
755            );
756            match budget_status {
757                BudgetStatus::HardExceeded { message, .. } => {
758                    self.emit_event(&TestEvent::Warning {
759                        message: format!("budget exceeded: {message}"),
760                    });
761                    result.details.push(StepResult {
762                        name: "[budget]".into(),
763                        status: StepStatus::Failed,
764                        message,
765                    });
766                    result.failed += 1;
767                    return result;
768                }
769                BudgetStatus::SoftExceeded { message, .. } => {
770                    self.emit_event(&TestEvent::Warning {
771                        message: format!("budget warning: {message}"),
772                    });
773                }
774                BudgetStatus::Ok => {}
775            }
776
777            if let Some(ms) = wait_ms {
778                std::thread::sleep(Duration::from_millis(ms));
779            }
780
781            result.details.push(step_result);
782        }
783
784        result
785    }
786
787    /// Applies a viewport size to the current tab via CDP
788    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
789    /// overrides and the viewport matrix. The initial window size set at
790    /// browser launch is replaced by emulation; failures are logged but
791    /// do not fail the test (a mismatched viewport only weakens coverage).
792    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
793        use headless_chrome::protocol::cdp::Emulation;
794        let _ = self;
795        let params = Emulation::SetDeviceMetricsOverride {
796            width,
797            height,
798            device_scale_factor: 1.0,
799            mobile: false,
800            scale: None,
801            screen_width: Some(width),
802            screen_height: Some(height),
803            position_x: None,
804            position_y: None,
805            dont_set_visible_size: None,
806            screen_orientation: None,
807            viewport: None,
808            display_feature: None,
809            device_posture: None,
810        };
811        self.reporter.debug(format!("viewport: {width}x{height}"));
812        if let Err(e) = tab.call_method(params) {
813            self.reporter
814                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
815        }
816    }
817
818    // ── step handlers ───────────────────────────────────────────────────
819
820    #[allow(clippy::too_many_lines)]
821    fn run_click(
822        &self,
823        target: &str,
824        selector_override: Option<&str>,
825        step_endpoint: Option<&str>,
826        test_endpoint: Option<&str>,
827        idempotent: bool,
828        tab: &Tab,
829    ) -> StepResult {
830        let name = format!("[click] {target}");
831        let selector = match self.resolve_selector(
832            selector_override,
833            target,
834            step_endpoint,
835            test_endpoint,
836            tab,
837        ) {
838            Ok(s) => s,
839            Err(msg) => {
840                if idempotent {
841                    return StepResult {
842                        name,
843                        status: StepStatus::Skipped,
844                        message: format!("skipped (idempotent): no target found — {msg}"),
845                    };
846                }
847                return StepResult {
848                    name,
849                    status: StepStatus::Failed,
850                    message: msg,
851                };
852            }
853        };
854
855        // Idempotent steps probe briefly: a missing target means the
856        // action was already done / not applicable (e.g. an
857        // already-authenticated session), and skipping is the success
858        // path, not a failure.
859        let probe_secs = if idempotent { 5 } else { 10 };
860        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
861            Ok(element) => match element.click() {
862                Ok(_) => StepResult {
863                    name,
864                    status: StepStatus::Passed,
865                    message: format!("clicked {selector}"),
866                },
867                Err(e) => StepResult {
868                    name,
869                    status: StepStatus::Failed,
870                    message: format!("click failed on {selector}: {e}"),
871                },
872            },
873            Err(e) if idempotent => StepResult {
874                name,
875                status: StepStatus::Skipped,
876                message: format!("skipped (idempotent): element {selector} not present — {e}"),
877            },
878            Err(e) => StepResult {
879                name,
880                status: StepStatus::Failed,
881                message: format!("element {selector} not found: {e}"),
882            },
883        }
884    }
885
886    #[allow(clippy::too_many_arguments)]
887    fn run_type(
888        &self,
889        target: &str,
890        text: &str,
891        selector_override: Option<&str>,
892        step_endpoint: Option<&str>,
893        test_endpoint: Option<&str>,
894        idempotent: bool,
895        tab: &Tab,
896    ) -> StepResult {
897        let name = format!("[type] {target}");
898        let selector = match self.resolve_selector(
899            selector_override,
900            target,
901            step_endpoint,
902            test_endpoint,
903            tab,
904        ) {
905            Ok(s) => s,
906            Err(msg) => {
907                if idempotent {
908                    return StepResult {
909                        name,
910                        status: StepStatus::Skipped,
911                        message: format!("skipped (idempotent): no target found — {msg}"),
912                    };
913                }
914                return StepResult {
915                    name,
916                    status: StepStatus::Failed,
917                    message: msg,
918                };
919            }
920        };
921
922        let probe_secs = if idempotent { 5 } else { 10 };
923        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
924            Ok(element) => {
925                if let Err(e) = element.click() {
926                    return StepResult {
927                        name,
928                        status: StepStatus::Failed,
929                        message: format!("click to focus {selector} failed: {e}"),
930                    };
931                }
932
933                let js = format!(
934                    "document.querySelector('{}').value = '';",
935                    selector.replace('\'', "\\'")
936                );
937                let _ = tab.evaluate(&js, false);
938
939                match element.type_into(text) {
940                    Ok(_) => StepResult {
941                        name,
942                        status: StepStatus::Passed,
943                        message: format!("typed {text:?} into {selector}"),
944                    },
945                    Err(e) => StepResult {
946                        name,
947                        status: StepStatus::Failed,
948                        message: format!("type into {selector} failed: {e}"),
949                    },
950                }
951            }
952            Err(e) if idempotent => StepResult {
953                name,
954                status: StepStatus::Skipped,
955                message: format!("skipped (idempotent): element {selector} not present — {e}"),
956            },
957            Err(e) => StepResult {
958                name,
959                status: StepStatus::Failed,
960                message: format!("element {selector} not found: {e}"),
961            },
962        }
963    }
964
965    #[allow(clippy::too_many_arguments)]
966    #[allow(clippy::too_many_lines)]
967    fn run_wait(
968        &self,
969        target: &str,
970        selector_override: Option<&str>,
971        text: Option<&str>,
972        timeout_ms: Option<u64>,
973        step_endpoint: Option<&str>,
974        test_endpoint: Option<&str>,
975        idempotent: bool,
976        tab: &Tab,
977    ) -> StepResult {
978        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
979        let step_name = format!("[wait] {target}");
980
981        // Resolve an explicit selector only (text-only waits are LLM-free).
982        let selector = match selector_override {
983            Some(s) => Some(s.to_owned()),
984            None if text.is_some() => None,
985            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
986                Ok(s) => Some(s),
987                Err(msg) => {
988                    if idempotent {
989                        return StepResult {
990                            name: step_name,
991                            status: StepStatus::Skipped,
992                            message: format!("skipped (idempotent): no target found — {msg}"),
993                        };
994                    }
995                    return StepResult {
996                        name: step_name,
997                        status: StepStatus::Failed,
998                        message: msg,
999                    };
1000                }
1001            },
1002        };
1003
1004        if text.is_some() {
1005            let sel_js = selector
1006                .as_deref()
1007                .map(crate::selectors::selector_matches_js);
1008            let text_js = text.map(|t| {
1009                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1010                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1011            });
1012
1013            let deadline = Instant::now() + timeout;
1014            loop {
1015                let sel_ok = sel_js
1016                    .as_ref()
1017                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1018                let text_ok = text_js
1019                    .as_ref()
1020                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1021                if sel_ok && text_ok {
1022                    let mut what = Vec::new();
1023                    if let Some(sel) = &selector {
1024                        what.push(format!("found {sel}"));
1025                    }
1026                    if let Some(t) = text {
1027                        what.push(format!("text {t:?} visible"));
1028                    }
1029                    return StepResult {
1030                        name: step_name,
1031                        status: StepStatus::Passed,
1032                        message: what.join(" and "),
1033                    };
1034                }
1035                if Instant::now() >= deadline {
1036                    let mut what = Vec::new();
1037                    if let Some(sel) = &selector {
1038                        what.push(sel.clone());
1039                    }
1040                    if let Some(t) = text {
1041                        what.push(format!("text {t:?}"));
1042                    }
1043                    let message = format!(
1044                        "wait for {} timed out after {}ms: the event waited for never came",
1045                        what.join(" / "),
1046                        timeout.as_millis(),
1047                    );
1048                    if idempotent {
1049                        return StepResult {
1050                            name: step_name,
1051                            status: StepStatus::Skipped,
1052                            message: format!("skipped (idempotent): {message}"),
1053                        };
1054                    }
1055                    return StepResult {
1056                        name: step_name,
1057                        status: StepStatus::Failed,
1058                        message,
1059                    };
1060                }
1061                std::thread::sleep(Duration::from_millis(250));
1062            }
1063        }
1064
1065        match selector.as_deref() {
1066            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1067                Ok(_) => StepResult {
1068                    name: step_name,
1069                    status: StepStatus::Passed,
1070                    message: format!("found {sel}"),
1071                },
1072                Err(e) if idempotent => StepResult {
1073                    name: step_name,
1074                    status: StepStatus::Skipped,
1075                    message: format!(
1076                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1077                        timeout.as_millis()
1078                    ),
1079                },
1080                Err(e) => StepResult {
1081                    name: step_name,
1082                    status: StepStatus::Failed,
1083                    message: format!(
1084                        "wait for {sel} timed out after {}ms: {e}",
1085                        timeout.as_millis()
1086                    ),
1087                },
1088            },
1089            None => StepResult {
1090                name: step_name,
1091                status: StepStatus::Failed,
1092                message: "wait step has neither selector nor text".into(),
1093            },
1094        }
1095    }
1096
1097    #[allow(clippy::too_many_arguments)]
1098    fn run_assert(
1099        &self,
1100        definition: Option<&str>,
1101        preset: Option<&str>,
1102        prompt: Option<&str>,
1103        assert_text: Option<&str>,
1104        screenshot: bool,
1105        step_endpoint: Option<&str>,
1106        test_endpoint: Option<&str>,
1107        tab: &Tab,
1108    ) -> StepResult {
1109        std::thread::sleep(Duration::from_millis(500));
1110
1111        let page_content = get_page_text(tab);
1112
1113        // Vision attach: capture the viewport once per assert step and hand
1114        // the JPEG data URL to the preset/prompt evaluation below.
1115        let image = if screenshot {
1116            let endpoint = self
1117                .endpoints
1118                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1119            if !endpoint.vision {
1120                return StepResult {
1121                    name: "[assert]".into(),
1122                    status: StepStatus::Failed,
1123                    message: format!(
1124                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1125                        name = endpoint.name
1126                    ),
1127                };
1128            }
1129            match crate::vision::capture_screenshot_data_url(
1130                tab,
1131                self.config
1132                    .screenshot_max_dimension
1133                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1134            ) {
1135                Ok(data_url) => Some(data_url),
1136                Err(e) => {
1137                    return StepResult {
1138                        name: "[assert]".into(),
1139                        status: StepStatus::Failed,
1140                        message: format!("screenshot capture failed: {e}"),
1141                    };
1142                }
1143            }
1144        } else {
1145            None
1146        };
1147
1148        if let Some(def_name) = definition {
1149            if let Some(def) = self.definitions.get(def_name) {
1150                return self.run_assert_def(
1151                    def,
1152                    &page_content,
1153                    image.as_deref(),
1154                    step_endpoint,
1155                    test_endpoint,
1156                    tab,
1157                );
1158            }
1159            return StepResult {
1160                name: format!("[assert] {def_name}"),
1161                status: StepStatus::Failed,
1162                message: format!("definition '{def_name}' not found"),
1163            };
1164        }
1165
1166        if let Some(preset_name) = preset {
1167            // Deterministic DOM layout scan — runs JS in the browser and
1168            // never calls the LLM (free, fast, no pixel budget).
1169            if preset_name == "layout_no_issues" {
1170                return self.run_layout_preset(tab);
1171            }
1172            return self.run_preset(
1173                preset_name,
1174                assert_text,
1175                &page_content,
1176                image.as_deref(),
1177                step_endpoint,
1178                test_endpoint,
1179            );
1180        }
1181
1182        if let Some(prompt_text) = prompt {
1183            return self.run_custom(
1184                prompt_text,
1185                &page_content,
1186                image.as_deref(),
1187                step_endpoint,
1188                test_endpoint,
1189            );
1190        }
1191
1192        StepResult {
1193            name: "[assert]".into(),
1194            status: StepStatus::Skipped,
1195            message: "no definition, preset, or prompt specified".into(),
1196        }
1197    }
1198
1199    fn run_assert_def(
1200        &self,
1201        def: &AssertDefinition,
1202        page_content: &PageContent,
1203        image: Option<&str>,
1204        step_endpoint: Option<&str>,
1205        test_endpoint: Option<&str>,
1206        tab: &Tab,
1207    ) -> StepResult {
1208        // Agent-based definition: delegate to an A2A agent
1209        if let Some(ref agent) = def.agent {
1210            if image.is_some() {
1211                return StepResult {
1212                    name: format!("[assert] {}", def.name),
1213                    status: StepStatus::Failed,
1214                    message: "agent-backed assertions do not support screenshots".into(),
1215                };
1216            }
1217            let task = def
1218                .task_template
1219                .as_deref()
1220                .unwrap_or("Evaluate the assertion")
1221                .replace("{url}", &page_content.url)
1222                .replace("{title}", &page_content.title)
1223                .replace("{content}", &page_content.body_text)
1224                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1225
1226            return self.run_agent_step(agent, &task, &def.name);
1227        }
1228
1229        // Custom preset: system + user_template provided in the definition
1230        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1231            return self.run_custom_preset(
1232                &def.name,
1233                system,
1234                template,
1235                def.assert_text.as_deref(),
1236                page_content,
1237                image,
1238                step_endpoint,
1239                test_endpoint,
1240            );
1241        }
1242
1243        def.preset.as_ref().map_or_else(
1244            || {
1245                def.prompt.as_ref().map_or_else(
1246                    || StepResult {
1247                        name: format!("[assert] {}", def.name),
1248                        status: StepStatus::Failed,
1249                        message: "definition has no preset, prompt, or system+user_template".into(),
1250                    },
1251                    |prompt| {
1252                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1253                    },
1254                )
1255            },
1256            |preset_name| {
1257                if preset_name == "layout_no_issues" {
1258                    return self.run_layout_preset(tab);
1259                }
1260                self.run_preset(
1261                    preset_name,
1262                    def.assert_text.as_deref(),
1263                    page_content,
1264                    image,
1265                    step_endpoint,
1266                    test_endpoint,
1267                )
1268            },
1269        )
1270    }
1271
1272    #[allow(clippy::too_many_arguments)]
1273    fn run_custom_preset(
1274        &self,
1275        name: &str,
1276        system: &str,
1277        template: &str,
1278        assert_text: Option<&str>,
1279        page_content: &PageContent,
1280        image: Option<&str>,
1281        step_endpoint: Option<&str>,
1282        test_endpoint: Option<&str>,
1283    ) -> StepResult {
1284        let user_prompt = template
1285            .replace("{url}", &page_content.url)
1286            .replace("{title}", &page_content.title)
1287            .replace("{content}", &page_content.body_text)
1288            .replace("{expected_text}", assert_text.unwrap_or(""))
1289            .replace("{description}", "");
1290
1291        // Custom preset definitions frequently forget the {content}
1292        // placeholder — without it the LLM has no page to evaluate and
1293        // answers "I can't determine that without seeing the page". Always
1294        // append the page context unless the template already references it.
1295        let user_prompt = if template.contains("{content}") {
1296            user_prompt
1297        } else {
1298            format!(
1299                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1300                url = page_content.url,
1301                title = page_content.title,
1302                content = page_content.body_text,
1303            )
1304        };
1305
1306        self.reporter
1307            .debug(format!("assert: {name} (custom preset)"));
1308
1309        let chain = self
1310            .endpoints
1311            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1312        let sys = system.to_owned();
1313
1314        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1315
1316        response.map_or_else(
1317            |e| StepResult {
1318                name: format!("[assert] {name}"),
1319                status: StepStatus::Failed,
1320                message: format!("LLM assertion call failed: {e}"),
1321            },
1322            |(lr, idx)| {
1323                self.usage.record_llm_call(
1324                    &chain[idx].name,
1325                    chain[idx],
1326                    lr.usage.prompt_tokens,
1327                    lr.usage.completion_tokens,
1328                );
1329                let content_lower = lr.content.to_lowercase().trim().to_owned();
1330                if content_lower.starts_with("pass") {
1331                    StepResult {
1332                        name: format!("[assert] {name}"),
1333                        status: StepStatus::Passed,
1334                        message: "PASS".into(),
1335                    }
1336                } else {
1337                    StepResult {
1338                        name: format!("[assert] {name}"),
1339                        status: StepStatus::Failed,
1340                        message: lr.content,
1341                    }
1342                }
1343            },
1344        )
1345    }
1346
1347    fn run_preset(
1348        &self,
1349        preset_name: &str,
1350        assert_text: Option<&str>,
1351        page_content: &PageContent,
1352        image: Option<&str>,
1353        step_endpoint: Option<&str>,
1354        test_endpoint: Option<&str>,
1355    ) -> StepResult {
1356        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1357            return StepResult {
1358                name: format!("[assert] {preset_name}"),
1359                status: StepStatus::Failed,
1360                message: format!("unknown assertion preset: {preset_name}"),
1361            };
1362        };
1363        if preset_name.starts_with("visual_") && image.is_none() {
1364            return StepResult {
1365                name: format!("[assert] {preset_name}"),
1366                status: StepStatus::Failed,
1367                message: format!(
1368                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1369                ),
1370            };
1371        }
1372
1373        let user_prompt = preset
1374            .user_template
1375            .replace("{url}", &page_content.url)
1376            .replace("{title}", &page_content.title)
1377            .replace("{content}", &page_content.body_text)
1378            .replace("{expected_text}", assert_text.unwrap_or(""))
1379            .replace("{description}", "");
1380
1381        // Same safety net as custom presets: never let the LLM answer with
1382        // no page context at all.
1383        let user_prompt = if preset.user_template.contains("{content}") {
1384            user_prompt
1385        } else {
1386            format!(
1387                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1388                url = page_content.url,
1389                title = page_content.title,
1390                content = page_content.body_text,
1391            )
1392        };
1393
1394        self.reporter.debug(format!("assert: {preset_name}"));
1395
1396        let chain = self
1397            .endpoints
1398            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1399        let sys = preset.system.to_owned();
1400
1401        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1402
1403        response.map_or_else(
1404            |e| StepResult {
1405                name: format!("[assert] {preset_name}"),
1406                status: StepStatus::Failed,
1407                message: format!("LLM assertion call failed: {e}"),
1408            },
1409            |(lr, idx)| {
1410                self.usage.record_llm_call(
1411                    &chain[idx].name,
1412                    chain[idx],
1413                    lr.usage.prompt_tokens,
1414                    lr.usage.completion_tokens,
1415                );
1416                let content_lower = lr.content.to_lowercase().trim().to_owned();
1417                if content_lower.starts_with("pass") {
1418                    StepResult {
1419                        name: format!("[assert] {preset_name}"),
1420                        status: StepStatus::Passed,
1421                        message: "PASS".into(),
1422                    }
1423                } else {
1424                    StepResult {
1425                        name: format!("[assert] {preset_name}"),
1426                        status: StepStatus::Failed,
1427                        message: lr.content,
1428                    }
1429                }
1430            },
1431        )
1432    }
1433
1434    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1435    ///
1436    /// Evaluates the layout-scan JS in the page and fails with the list of
1437    /// detected issues: horizontal page overflow, visible elements sticking
1438    /// out of the viewport, text clipped by `overflow: hidden` containers,
1439    /// and interactive elements covered by other elements. No LLM call —
1440    /// checks are geometry-based so the check is free, deterministic, and
1441    /// safe to run on every page × viewport variant.
1442    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1443        let name = "[assert] layout_no_issues".to_owned();
1444        self.reporter
1445            .debug("assert: layout_no_issues (DOM layout scan)");
1446        let js = LAYOUT_SCAN_JS.replace(
1447            "__IGNORE_CLASSES__",
1448            &serde_json::to_string(&self.config.layout_ignore_classes)
1449                .unwrap_or_else(|_| "[]".to_owned()),
1450        );
1451        let result = tab.evaluate(&js, false);
1452        let json_str = match result {
1453            Ok(r) => r
1454                .value
1455                .as_ref()
1456                .and_then(|v| v.as_str().map(String::from))
1457                .unwrap_or_else(|| "[]".to_owned()),
1458            Err(e) => {
1459                return StepResult {
1460                    name,
1461                    status: StepStatus::Failed,
1462                    message: format!("layout scan JS failed: {e}"),
1463                };
1464            }
1465        };
1466        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1467        if issues.is_empty() {
1468            return StepResult {
1469                name,
1470                status: StepStatus::Passed,
1471                message: "PASS — no layout defects detected".into(),
1472            };
1473        }
1474        let mut lines: Vec<String> = issues
1475            .iter()
1476            .take(10)
1477            .map(|i| {
1478                format!(
1479                    "- [{type_}] {element}: {detail}",
1480                    type_ = i.issue_type,
1481                    element = i.element,
1482                    detail = i.detail
1483                )
1484            })
1485            .collect();
1486        if issues.len() > 10 {
1487            lines.push(format!("- … and {} more", issues.len() - 10));
1488        }
1489        StepResult {
1490            name,
1491            status: StepStatus::Failed,
1492            message: format!(
1493                "FAIL — {} layout defect(s) detected:\n{}",
1494                issues.len(),
1495                lines.join("\n")
1496            ),
1497        }
1498    }
1499
1500    fn run_custom(
1501        &self,
1502        prompt: &str,
1503        page_content: &PageContent,
1504        image: Option<&str>,
1505        step_endpoint: Option<&str>,
1506        test_endpoint: Option<&str>,
1507    ) -> StepResult {
1508        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1509
1510        let mut user = format!(
1511            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1512            url = page_content.url,
1513            title = page_content.title,
1514            content = page_content.body_text,
1515        );
1516        if image.is_some() {
1517            user.push_str(
1518                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1519            );
1520        }
1521
1522        self.reporter.debug("custom assert");
1523
1524        let chain = self
1525            .endpoints
1526            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1527        let sys = system.to_owned();
1528
1529        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1530
1531        response.map_or_else(
1532            |e| StepResult {
1533                name: "[assert] custom".into(),
1534                status: StepStatus::Failed,
1535                message: format!("LLM assertion call failed: {e}"),
1536            },
1537            |(lr, idx)| {
1538                self.usage.record_llm_call(
1539                    &chain[idx].name,
1540                    chain[idx],
1541                    lr.usage.prompt_tokens,
1542                    lr.usage.completion_tokens,
1543                );
1544                let content_lower = lr.content.to_lowercase().trim().to_owned();
1545                if content_lower.starts_with("pass") {
1546                    StepResult {
1547                        name: "[assert] custom".into(),
1548                        status: StepStatus::Passed,
1549                        message: "PASS".into(),
1550                    }
1551                } else {
1552                    StepResult {
1553                        name: "[assert] custom".into(),
1554                        status: StepStatus::Failed,
1555                        message: lr.content,
1556                    }
1557                }
1558            },
1559        )
1560    }
1561
1562    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1563        let path = path.unwrap_or("screenshot.png");
1564
1565        match tab.capture_screenshot(
1566            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1567            None,
1568            None,
1569            true,
1570        ) {
1571            Ok(data) => {
1572                if let Err(e) = std::fs::write(path, &data) {
1573                    return StepResult {
1574                        name: format!("[screenshot] {path}"),
1575                        status: StepStatus::Failed,
1576                        message: format!("failed to write screenshot: {e}"),
1577                    };
1578                }
1579                StepResult {
1580                    name: format!("[screenshot] {path}"),
1581                    status: StepStatus::Passed,
1582                    message: format!("saved to {path}"),
1583                }
1584            }
1585            Err(e) => StepResult {
1586                name: format!("[screenshot] {path}"),
1587                status: StepStatus::Failed,
1588                message: format!("screenshot failed: {e}"),
1589            },
1590        }
1591    }
1592
1593    /// Runs an A2A agent step.
1594    #[allow(clippy::literal_string_with_formatting_args)]
1595    fn run_agent(
1596        &self,
1597        agent_name: &str,
1598        task: &str,
1599        definition: Option<&str>,
1600        _test_endpoint: Option<&str>,
1601    ) -> StepResult {
1602        // If a definition is specified, look up the task template
1603        let resolved_task = if let Some(def_name) = definition {
1604            if let Some(def) = self.definitions.get(def_name) {
1605                let tmpl = def.task_template.as_deref().unwrap_or(task);
1606                tmpl.replace("{task}", task)
1607            } else {
1608                return StepResult {
1609                    name: format!("[agent] {def_name}"),
1610                    status: StepStatus::Failed,
1611                    message: format!("definition '{def_name}' not found"),
1612                };
1613            }
1614        } else {
1615            task.to_owned()
1616        };
1617
1618        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1619    }
1620
1621    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1622        let Some(ep) = self.endpoints.get(agent_name) else {
1623            return StepResult {
1624                name: format!("[agent] {display_name}"),
1625                status: StepStatus::Failed,
1626                message: format!("agent endpoint '{agent_name}' not found"),
1627            };
1628        };
1629
1630        if ep.url.is_empty() {
1631            return StepResult {
1632                name: format!("[agent] {display_name}"),
1633                status: StepStatus::Failed,
1634                message: format!("agent endpoint '{agent_name}' has no URL"),
1635            };
1636        }
1637
1638        self.reporter.debug(format!("agent {agent_name}: {task}"));
1639
1640        let url = ep.url.clone();
1641        let client = A2aClient::new(&url, self.timeout);
1642        let task_clone = task.to_owned();
1643
1644        let response = std::thread::spawn(move || {
1645            let rt = tokio::runtime::Builder::new_current_thread()
1646                .enable_all()
1647                .build()
1648                .unwrap();
1649            rt.block_on(client.send_task(&task_clone))
1650        })
1651        .join()
1652        .unwrap();
1653
1654        // Record the flat-cost call
1655        self.usage.record_flat_call(agent_name, ep);
1656
1657        match response {
1658            Ok(text) => {
1659                let clean = text.trim().to_owned();
1660                let lower = clean.to_lowercase();
1661                if lower.starts_with("pass") {
1662                    StepResult {
1663                        name: format!("[agent] {display_name}"),
1664                        status: StepStatus::Passed,
1665                        message: format!("PASS: {clean}"),
1666                    }
1667                } else if lower.starts_with("fail") {
1668                    StepResult {
1669                        name: format!("[agent] {display_name}"),
1670                        status: StepStatus::Failed,
1671                        message: clean,
1672                    }
1673                } else {
1674                    StepResult {
1675                        name: format!("[agent] {display_name}"),
1676                        status: StepStatus::Passed,
1677                        message: format!("response: {clean}"),
1678                    }
1679                }
1680            }
1681            Err(e) => StepResult {
1682                name: format!("[agent] {display_name}"),
1683                status: StepStatus::Failed,
1684                message: format!("agent call failed: {e}"),
1685            },
1686        }
1687    }
1688
1689    /// Runs an MCP tool call step.
1690    fn run_mcp(
1691        &self,
1692        server_name: &str,
1693        tool_name: &str,
1694        args: Option<&serde_json::Value>,
1695    ) -> StepResult {
1696        let Some(ep) = self.endpoints.get(server_name) else {
1697            return StepResult {
1698                name: format!("[mcp] {server_name}:{tool_name}"),
1699                status: StepStatus::Failed,
1700                message: format!("MCP server endpoint '{server_name}' not found"),
1701            };
1702        };
1703
1704        let cmd = ep.command.as_deref().unwrap_or("");
1705        if cmd.is_empty() {
1706            return StepResult {
1707                name: format!("[mcp] {server_name}:{tool_name}"),
1708                status: StepStatus::Failed,
1709                message: format!("MCP server '{server_name}' has no command configured"),
1710            };
1711        }
1712
1713        self.reporter
1714            .debug(format!("mcp {server_name} {tool_name}"));
1715
1716        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1717
1718        let command = cmd.to_owned();
1719        let args_vec = ep.args.clone();
1720        let tool = tool_name.to_owned();
1721
1722        let response = std::thread::spawn(move || {
1723            let mut mcp_client =
1724                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1725            mcp_client
1726                .call_tool(&tool, &args_val)
1727                .map_err(|e| e.to_string())
1728        })
1729        .join()
1730        .unwrap();
1731
1732        // Record the flat-cost call
1733        self.usage.record_flat_call(server_name, ep);
1734
1735        match response {
1736            Ok(result) => {
1737                if result.isError {
1738                    StepResult {
1739                        name: format!("[mcp] {server_name}:{tool_name}"),
1740                        status: StepStatus::Failed,
1741                        message: result.to_string(),
1742                    }
1743                } else {
1744                    StepResult {
1745                        name: format!("[mcp] {server_name}:{tool_name}"),
1746                        status: StepStatus::Passed,
1747                        message: result.to_string(),
1748                    }
1749                }
1750            }
1751            Err(e) => StepResult {
1752                name: format!("[mcp] {server_name}:{tool_name}"),
1753                status: StepStatus::Failed,
1754                message: format!("MCP call failed: {e}"),
1755            },
1756        }
1757    }
1758
1759    // ── helpers ──────────────────────────────────────────────────────────
1760
1761    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1762    /// the runner's default LLM config for any unset fields.
1763    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1764        LlmConfig {
1765            url: if endpoint.url.is_empty() {
1766                self.llm.url.clone()
1767            } else {
1768                endpoint.url.clone()
1769            },
1770            model: endpoint
1771                .model
1772                .clone()
1773                .unwrap_or_else(|| self.llm.model.clone()),
1774            api_key: endpoint
1775                .api_key
1776                .clone()
1777                .or_else(|| self.llm.api_key.clone()),
1778            headers: if endpoint.headers.is_empty() {
1779                self.llm.headers.clone()
1780            } else {
1781                endpoint.headers.clone()
1782            },
1783            timeout: self.llm.timeout,
1784            temperature: self.llm.temperature,
1785            thinking: self.llm.thinking,
1786            model_params: self.llm.model_params.clone(),
1787            max_attempts: endpoint.max_attempts.max(1),
1788        }
1789    }
1790
1791    /// Runs a single LLM call against an ordered endpoint chain (primary +
1792    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
1793    /// the first endpoint that answers wins. Returns the response together
1794    /// with the chain index of the answering endpoint (0 = primary) so the
1795    /// caller can attribute usage to the correct endpoint.
1796    ///
1797    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
1798    /// shows duration, tokens, cost and the answering endpoint per call.
1799    #[allow(clippy::cast_possible_truncation)]
1800    fn llm_call_chain(
1801        &self,
1802        chain: &[&ResolvedEndpoint],
1803        system: &str,
1804        user: &str,
1805        image: Option<&str>,
1806        purpose: &str,
1807    ) -> Result<(crate::costs::LlmResponse, usize), String> {
1808        if chain.is_empty() {
1809            return Err("empty LLM endpoint chain".into());
1810        }
1811        let primary = self.build_llm_for_endpoint(chain[0]);
1812        let fallbacks: Vec<LlmConfig> = chain[1..]
1813            .iter()
1814            .map(|e| self.build_llm_for_endpoint(e))
1815            .collect();
1816
1817        let (test, index) = self
1818            .current_step
1819            .borrow()
1820            .as_ref()
1821            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
1822        let primary_endpoint = chain[0].name.clone();
1823        let primary_model = primary.model.clone();
1824        self.emit_event(&TestEvent::LlmCallStarted {
1825            test: test.clone(),
1826            index,
1827            endpoint: primary_endpoint.clone(),
1828            model: primary_model.clone(),
1829            purpose: purpose.to_owned(),
1830        });
1831
1832        let started = Instant::now();
1833        let sys = system.to_owned();
1834        let user = user.to_owned();
1835        let image = image.map(str::to_owned);
1836
1837        let result = std::thread::spawn(move || {
1838            let rt = tokio::runtime::Builder::new_current_thread()
1839                .enable_all()
1840                .build()
1841                .unwrap();
1842            let call = async {
1843                match image.as_deref() {
1844                    Some(img) => {
1845                        llm_chat_vision_with_usage_chain(&primary, &fallbacks, &sys, &user, img)
1846                            .await
1847                    }
1848                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
1849                }
1850            };
1851            rt.block_on(call)
1852        })
1853        .join()
1854        .unwrap();
1855
1856        let duration_ms = started.elapsed().as_millis() as u64;
1857        match result {
1858            Ok((lr, idx)) => {
1859                let cost = calculate_llm_cost(
1860                    chain[idx],
1861                    lr.usage.prompt_tokens,
1862                    lr.usage.completion_tokens,
1863                );
1864                let answering = chain[idx].name.clone();
1865                let model = chain[idx]
1866                    .model
1867                    .clone()
1868                    .unwrap_or_else(|| primary_model.clone());
1869                self.emit_event(&TestEvent::LlmCallFinished {
1870                    test,
1871                    index,
1872                    endpoint: answering,
1873                    model,
1874                    purpose: purpose.to_owned(),
1875                    ok: true,
1876                    duration_ms,
1877                    input_tokens: lr.usage.prompt_tokens,
1878                    output_tokens: lr.usage.completion_tokens,
1879                    cost,
1880                    error: None,
1881                });
1882                Ok((lr, idx))
1883            }
1884            Err(e) => {
1885                self.emit_event(&TestEvent::LlmCallFinished {
1886                    test,
1887                    index,
1888                    endpoint: primary_endpoint,
1889                    model: primary_model,
1890                    purpose: purpose.to_owned(),
1891                    ok: false,
1892                    duration_ms,
1893                    input_tokens: 0,
1894                    output_tokens: 0,
1895                    cost: 0.0,
1896                    error: Some(e.clone()),
1897                });
1898                Err(e)
1899            }
1900        }
1901    }
1902
1903    /// Resolves a CSS selector for the target element. Uses the explicit
1904    /// `selector` if provided, otherwise asks the LLM to find the element
1905    /// from the natural language `target` description and page DOM.
1906    ///
1907    /// LLM responses are sanitized and verified against the live page: a
1908    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
1909    /// immediately with the raw LLM output, and a selector that matches
1910    /// nothing triggers one retry with feedback before failing.
1911    #[allow(clippy::too_many_lines)]
1912    fn resolve_selector(
1913        &self,
1914        css_override: Option<&str>,
1915        target: &str,
1916        step_endpoint: Option<&str>,
1917        test_endpoint: Option<&str>,
1918        tab: &Tab,
1919    ) -> Result<String, String> {
1920        if let Some(explicit) = css_override {
1921            return Ok(explicit.to_owned());
1922        }
1923
1924        let dom_info = extract_dom_info(tab)?;
1925        let page_content = get_page_text(tab);
1926
1927        let system = concat!(
1928            "You are a browser automation selector generator. ",
1929            "Given a web page's content and interactive elements, ",
1930            "return ONLY the best CSS selector for the described element. ",
1931            "Output nothing except the CSS selector. ",
1932            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
1933            "[name=\"...\"], tag.class, tag. ",
1934            "Never output explanations, markdown, or extra text."
1935        );
1936
1937        let user = format!(
1938            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
1939            page_content.url,
1940            page_content.title,
1941            truncate(&page_content.body_text, 4000),
1942            dom_info,
1943            target,
1944        );
1945
1946        let retry_user = format!(
1947            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
1948            "The selector must match at least one element currently present on the page.",
1949            page_content.url,
1950            page_content.title,
1951            truncate(&page_content.body_text, 4000),
1952            dom_info,
1953            target,
1954        );
1955
1956        self.reporter.debug(format!("LLM targeting: {target}"));
1957
1958        let chain = self
1959            .endpoints
1960            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
1961        let sys = system.to_owned();
1962
1963        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
1964
1965        let first = call_llm(&user);
1966        let (lr, idx) = match first {
1967            Ok(lr) => lr,
1968            Err(e) => {
1969                return Err(format!("LLM element targeting failed: {e}"));
1970            }
1971        };
1972        self.usage.record_llm_call(
1973            &chain[idx].name,
1974            chain[idx],
1975            lr.usage.prompt_tokens,
1976            lr.usage.completion_tokens,
1977        );
1978        let clean = sanitize_selector(&lr.content);
1979        self.reporter.debug(format!("resolved selector: {clean}"));
1980
1981        if selector_is_useless(&clean) {
1982            return Err(format!(
1983                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
1984                raw = lr.content.trim(),
1985            ));
1986        }
1987        if let Err(reason) = validate_selector(&clean) {
1988            return Err(format!(
1989                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
1990                raw = lr.content.trim(),
1991            ));
1992        }
1993        if !selector_matches(tab, &clean).unwrap_or(false) {
1994            // One retry with feedback: flaky models occasionally invent a
1995            // selector that does not exist on the page.
1996            self.reporter.warn(format!(
1997                "selector {clean} matches nothing — retrying LLM targeting with feedback"
1998            ));
1999            let second = call_llm(&retry_user);
2000            let (lr2, idx2) = match second {
2001                Ok(lr2) => lr2,
2002                Err(e) => {
2003                    return Err(format!(
2004                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2005                    ));
2006                }
2007            };
2008            self.usage.record_llm_call(
2009                &chain[idx2].name,
2010                chain[idx2],
2011                lr2.usage.prompt_tokens,
2012                lr2.usage.completion_tokens,
2013            );
2014            let clean2 = sanitize_selector(&lr2.content);
2015            self.reporter
2016                .debug(format!("resolved selector (retry): {clean2}"));
2017            if selector_is_useless(&clean2) {
2018                return Err(format!(
2019                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2020                    raw = lr2.content.trim(),
2021                    excerpt = truncate(&page_content.body_text, 300),
2022                ));
2023            }
2024            if !selector_matches(tab, &clean2).unwrap_or(false) {
2025                return Err(format!(
2026                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2027                ));
2028            }
2029            return Ok(clean2);
2030        }
2031
2032        Ok(clean)
2033    }
2034}
2035
2036/// Evaluates a JS expression that is expected to return a boolean.
2037fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2038    tab.evaluate(js, false)
2039        .map_err(|e| format!("evaluate failed: {e}"))?
2040        .value
2041        .and_then(|v| v.as_bool())
2042        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2043}
2044
2045/// Checks whether a CSS selector matches at least one current element.
2046fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2047    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2048}
2049
2050// ── Free helper functions ──────────────────────────────────────────────
2051
2052fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2053    let name = format!("[navigate] {full_url}");
2054    match tab.navigate_to(full_url) {
2055        Ok(_) => {
2056            let _ = tab.wait_until_navigated();
2057            StepResult {
2058                name,
2059                status: StepStatus::Passed,
2060                message: format!("navigated to {full_url}"),
2061            }
2062        }
2063        Err(e) => StepResult {
2064            name,
2065            status: StepStatus::Failed,
2066            message: format!("navigation failed: {e}"),
2067        },
2068    }
2069}
2070
2071fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2072    let result = tab
2073        .evaluate(DOM_EXTRACT_JS, false)
2074        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2075
2076    let json_str = result
2077        .value
2078        .as_ref()
2079        .and_then(|v| v.as_str())
2080        .unwrap_or("[]");
2081
2082    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2083
2084    if elements.is_empty() {
2085        return Ok("(no interactive elements found)".to_owned());
2086    }
2087
2088    Ok(elements.join("\n"))
2089}
2090
2091fn get_page_text(tab: &Tab) -> PageContent {
2092    let url = tab.get_url();
2093
2094    let title = tab
2095        .evaluate("document.title", false)
2096        .ok()
2097        .and_then(|r| r.value)
2098        .and_then(|v| v.as_str().map(String::from))
2099        .unwrap_or_else(|| "unknown".to_owned());
2100
2101    let body_text = tab
2102        .evaluate(
2103            "document.body ? document.body.innerText : document.documentElement.innerText",
2104            false,
2105        )
2106        .ok()
2107        .and_then(|r| r.value)
2108        .and_then(|v| v.as_str().map(String::from))
2109        .unwrap_or_default();
2110
2111    PageContent {
2112        url,
2113        title,
2114        body_text: truncate(&body_text, 8000),
2115    }
2116}
2117
2118fn resolve_url(url: &str, base_url: &str) -> String {
2119    if url.starts_with("http://") || url.starts_with("https://") {
2120        return url.to_owned();
2121    }
2122    let base = base_url.trim_end_matches('/');
2123    if url.starts_with('/') {
2124        format!("{base}{url}")
2125    } else {
2126        format!("{base}/{url}")
2127    }
2128}
2129
2130/// Human-readable label for a step, used when steps are skipped after an
2131/// earlier failure.
2132fn step_label(step: &TestStep) -> String {
2133    match step {
2134        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2135        TestStep::Click { target, .. } => format!("[click] {target}"),
2136        TestStep::Type { target, .. } => format!("[type] {target}"),
2137        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2138        TestStep::Assert {
2139            definition,
2140            preset,
2141            prompt,
2142            ..
2143        } => definition.as_ref().map_or_else(
2144            || {
2145                preset.as_ref().map_or_else(
2146                    || {
2147                        prompt.as_ref().map_or_else(
2148                            || "[assert]".to_owned(),
2149                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2150                        )
2151                    },
2152                    |p| format!("[assert] {p}"),
2153                )
2154            },
2155            |d| format!("[assert] {d}"),
2156        ),
2157        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2158        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2159        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2160    }
2161}
2162
2163/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2164#[must_use]
2165const fn step_kind_label(step: &TestStep) -> &'static str {
2166    match step {
2167        TestStep::Navigate { .. } => "navigate",
2168        TestStep::Click { .. } => "click",
2169        TestStep::Type { .. } => "type",
2170        TestStep::Wait { .. } => "wait",
2171        TestStep::Assert { .. } => "assert",
2172        TestStep::Screenshot { .. } => "screenshot",
2173        TestStep::Agent { .. } => "agent",
2174        TestStep::Mcp { .. } => "mcp",
2175    }
2176}
2177
2178// ── Support types ──────────────────────────────────────────────────────
2179
2180#[derive(Default)]
2181struct TestRunResult {
2182    passed: u32,
2183    failed: u32,
2184    skipped: u32,
2185    total: u32,
2186    details: Vec<StepResult>,
2187}
2188
2189struct PageContent {
2190    url: String,
2191    title: String,
2192    body_text: String,
2193}