Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// One detected layout defect (`layout_no_issues` preset).
26#[derive(Debug, serde::Deserialize)]
27struct LayoutIssue {
28    #[serde(rename = "type")]
29    issue_type: String,
30    element: String,
31    detail: String,
32}
33
34/// In-browser DOM layout scan for `layout_no_issues`.
35///
36/// Geometry-only checks (no LLM, no pixels):
37/// 1. page horizontal overflow (`scrollWidth` > viewport width);
38/// 2. elements outside the viewport that scrolling cannot reveal
39///    (fixed elements off-screen, left/negative overflow, right-edge
40///    overflow beyond the horizontally scrollable area, and bottom
41///    overflow on a page that cannot scroll down) — below-the-fold
42///    content on a tall scrollable page is normal flow, NOT a defect;
43/// 3. text clipped by `overflow: hidden` containers whose content
44///    is measurably larger than the box;
45/// 4. interactive elements (buttons/links/inputs) whose center point
46///    is covered by a different element that would intercept the click.
47///
48/// Elements whose class matches a configured ignore prefix (default:
49/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
50/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
51/// skipped — they are intentionally 1x1 / off-screen. The prefix list
52/// is injected at `__IGNORE_CLASSES__` from
53/// [`ScenarioConfig::layout_ignore_classes`].
54///
55/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
56/// sticky headers) is excluded by the position/relation filters.
57const LAYOUT_SCAN_JS: &str = r#"
58(() => {
59  const issues = [];
60  const push = (type, el, detail) => {
61    if (issues.length >= 30) return;
62    let element = el.tagName.toLowerCase();
63    if (el.id) element += '#' + el.id;
64    else if (typeof el.className === 'string' && el.className.trim())
65      element += '.' + el.className.trim().split(/\s+/).join('.');
66    issues.push({ type, element, detail: String(detail).slice(0, 220) });
67  };
68  const vw = document.documentElement.clientWidth || window.innerWidth;
69  const vh = document.documentElement.clientHeight || window.innerHeight;
70  if (!vw || !vh) return JSON.stringify(issues);
71  const de = document.documentElement;
72  // 1. Page-level horizontal overflow.
73  if (maxSW > vw + 2)
74    push('page-overflow-x', de,
75      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
76  // Class-name prefixes to skip (injected; default = Angular CDK
77  // screen-reader helpers, which are intentionally 1x1 / off-screen).
78  const ignorePrefixes = __IGNORE_CLASSES__;
79  const isIgnored = (el) => {
80    if (typeof el.className !== 'string' || !el.className.trim()) return false;
81    const classes = el.className.trim().split(/\s+/);
82    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
83  };
84  const body = document.body;
85  // The document element is not always the scroll container: the app may
86  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
87  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
88  // "not scrollable" — measure the scrollable content height/width of the
89  // document element AND the body, and take the max.
90  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
91  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
92  const vScrollable = maxSH > vh + 2;
93  const hScrollable = maxSW > vw + 2;
94  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
95  // this element into view on the given axis — i.e. the element is inside a
96  // scrolling region, so lying beyond the viewport cut is not a defect.
97  const reachableByScroller = (el, axis) => {
98    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
99    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
100    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
101    let p = el.parentElement;
102    while (p && p !== body) {
103      const pcs = getComputedStyle(p);
104      const o = pcs[ovProp];
105      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
106          p[dim] > p[dimClient] + 2) return true;
107      p = p.parentElement;
108    }
109    return false;
110  };
111  const selfOverflowing = (el, cs, axis) => {
112    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
113    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
114    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
115    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
116      el[dim] > el[dimClient] + 2;
117  };
118  // True when an ancestor clips this axis with overflow:hidden/clip — the
119  // element's overhang is not visible, so treat it as reachable.
120  const clippedByAncestor = (el, axis) => {
121    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
122    let p = el.parentElement;
123    while (p && p !== body) {
124      const o = getComputedStyle(p)[ovProp];
125      if (o === 'hidden' || o === 'clip') return true;
126      p = p.parentElement;
127    }
128    return false;
129  };
130  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
131  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
132  const hasContent = (el) =>
133    ((el.textContent || '').trim().length > 0) ||
134    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
135  // 2. Elements outside the viewport that scrolling cannot reveal.
136  for (const el of all) {
137    const cs = getComputedStyle(el);
138    if (!visible(cs)) continue;
139    if (isIgnored(el)) continue;
140    const r = el.getBoundingClientRect();
141    if (r.width < 2 || r.height < 2) continue;
142    if (!hasContent(el) && el.children.length === 0) continue;
143    if (cs.position === 'fixed') {
144      // Fixed elements never move with the scroll: any edge outside the
145      // viewport is unreachable content and therefore a defect.
146      const overTop = -r.top;
147      const overLeft = -r.left;
148      const overRight = r.right - vw;
149      const overBottom = r.bottom - vh;
150      // Top/left overshoot is never reachable; bottom/right overshoot is
151      // only a defect when the fixed element cannot scroll that content
152      // into view itself (e.g. an opened Material drawer whose inner
153      // container scrolls is not a "cut-off" defect).
154      if (overTop > 2 || overLeft > 2 ||
155          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
156          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
157        let where = '';
158        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
159        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
160        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
161        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
162        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
163        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
164        push('element-out-of-viewport', el,
165          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
166      }
167      continue;
168    }
169    if (cs.position === 'sticky') continue;
170    // Negative left overflow cannot be reached by scrolling (scrollLeft
171    // never goes below 0).
172    if (r.left < -2) {
173      push('element-out-of-viewport', el,
174        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
175      continue;
176    }
177    // Negative top with the page at the top means the element sits above
178    // the document origin — also unreachable.
179    if (r.top < -2 && de.scrollTop <= 2) {
180      push('element-out-of-viewport', el,
181        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
182      continue;
183    }
184    const overRight = r.right - vw;
185    // Right-edge overflow is only a defect when the page cannot scroll
186    // horizontally to reveal it (or the element sticks out past the
187    // scrollable content width itself).
188    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
189      push('element-out-of-viewport', el,
190        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
191      continue;
192    }
193    // Below-the-fold content on a scrollable page is normal (tall landing
194    // pages); only flag bottom overflow the user can never scroll to.
195    const overBottom = r.bottom - vh;
196    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
197      push('element-out-of-viewport', el,
198        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
199    }
200  }
201  // 3. Text clipped by overflow:hidden containers.
202  for (const el of all) {
203    if (isIgnored(el)) continue;
204    const cs = getComputedStyle(el);
205    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
206    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
207    if (!(el.textContent || '').trim()) continue;
208    // Single-line truncation with an ellipsis is an intentional design
209    // pattern (Tailwind .truncate etc.), not a clipping defect.
210    if (cs.textOverflow === 'ellipsis') continue;
211    push('text-clipped', el,
212      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
213      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
214  }
215  // 4. Interactive elements covered by a different element.
216  const interactive =
217    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
218  const targets = document.querySelectorAll(interactive);
219  for (const el of targets) {
220    if (isIgnored(el)) continue;
221    const r = el.getBoundingClientRect();
222    if (r.width < 6 || r.height < 6) continue;
223    const cs = getComputedStyle(el);
224    if (!visible(cs)) continue;
225    const cx = r.left + r.width / 2;
226    const cy = r.top + r.height / 2;
227    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
228    const top = document.elementFromPoint(cx, cy);
229    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
230    if (isIgnored(top)) continue;
231    const tcs = getComputedStyle(top);
232    if (!visible(tcs)) continue;
233    if (tcs.pointerEvents === 'none') continue;
234    const tr = top.getBoundingClientRect();
235    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
236    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
237      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
238    push('element-overlap', el,
239      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
240      ' is covered by <' + tname + '>');
241  }
242  return JSON.stringify(issues);
243})()
244"#;
245
246/// How long the CDP connection stays open after the browser goes quiet.
247///
248/// `headless_chrome` ships a 30s default and tears down the entire connection
249/// when no traffic arrives for that long; a run must own its connection for
250/// its full duration instead.
251const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
252
253/// Executes a [`Scenario`] against a real browser with optional LLM
254/// assistance for element targeting and assertions.
255pub struct ScenarioRunner {
256    config: ScenarioConfig,
257    definitions: HashMap<String, AssertDefinition>,
258    llm: LlmConfig,
259    timeout: Duration,
260    viewport_width: u32,
261    viewport_height: u32,
262    /// The viewport currently applied in the browser (CDP emulation).
263    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
264    applied_viewport: std::cell::Cell<(u32, u32)>,
265    endpoints: EndpointRegistry,
266    usage: Arc<UsageTracker>,
267    budgets: BudgetTracker,
268    /// Directory for failure artifacts (screenshots).
269    artifacts_dir: PathBuf,
270    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
271    reporter: Arc<Reporter>,
272    /// The test + step index currently executing, for LLM-call events.
273    current_step: std::cell::RefCell<Option<(String, u32)>>,
274}
275
276/// Aggregated results from a scenario run.
277#[derive(Debug, Default)]
278pub struct RunReport {
279    /// Number of tests that passed.
280    pub tests_passed: u32,
281    /// Number of tests that failed.
282    pub tests_failed: u32,
283    /// Number of steps that passed.
284    pub passed: u32,
285    /// Number of steps that failed.
286    pub failed: u32,
287    /// Number of steps that were skipped.
288    pub skipped: u32,
289    /// Per-step details.
290    pub details: Vec<StepResult>,
291}
292
293/// Result of a single step execution.
294#[derive(Debug)]
295pub struct StepResult {
296    /// The step name.
297    pub name: String,
298    /// Whether the step passed, failed, or was skipped.
299    pub status: StepStatus,
300    /// Human-readable result message.
301    pub message: String,
302}
303
304/// Outcome for a single step.
305pub use crate::events::StepStatus;
306
307/// Predefined assertion preset definition.
308struct AssertPreset {
309    name: &'static str,
310    system: &'static str,
311    user_template: &'static str,
312}
313
314/// Built-in assertion presets.
315#[allow(clippy::literal_string_with_formatting_args)]
316const ASSERTION_PRESETS: &[AssertPreset] = &[
317    AssertPreset {
318        name: "no_error_on_page",
319        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
320        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
321    },
322    AssertPreset {
323        name: "text_visible",
324        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
325        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
326    },
327    AssertPreset {
328        name: "element_exists",
329        system: "You are a QA tester. Check if a described UI element exists on a web page.",
330        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
331    },
332    AssertPreset {
333        name: "visual_no_issues",
334        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
335        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
336    },
337    AssertPreset {
338        name: "visual_no_overlaps",
339        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
340        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
341    },
342    AssertPreset {
343        name: "visual_text_visible",
344        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
345        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
346    },
347    AssertPreset {
348        name: "layout_no_issues",
349        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
350        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
351    },
352];
353
354impl ScenarioRunner {
355    /// Creates a new runner with the given scenario configuration and
356    /// assertion definitions.
357    #[must_use]
358    #[allow(clippy::needless_pass_by_value)]
359    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
360        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
361    }
362
363    /// Creates a runner that reports run events through the given reporter.
364    #[must_use]
365    #[allow(clippy::needless_pass_by_value)]
366    pub fn with_reporter(
367        scenario_config: ScenarioConfig,
368        definitions: Vec<AssertDefinition>,
369        reporter: Arc<Reporter>,
370    ) -> Self {
371        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
372            &scenario_config,
373        ));
374        let llm = LlmConfig {
375            url: scenario_config
376                .llm_url
377                .clone()
378                .unwrap_or_else(crate::llm_base_url),
379            model: scenario_config
380                .llm_model
381                .clone()
382                .unwrap_or_else(crate::llm_model),
383            api_key: scenario_config
384                .llm_api_key
385                .clone()
386                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
387            headers: if scenario_config.llm_headers.is_empty() {
388                crate::parse_headers_env()
389            } else {
390                scenario_config.llm_headers.clone()
391            },
392            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
393            temperature: scenario_config.temperature,
394            thinking: scenario_config.thinking,
395            model_params: scenario_config.model_params.clone(),
396            max_attempts: crate::default_llm_attempts(),
397            provider: crate::scenario::Provider::Openai,
398            deployment: None,
399            api_version: None,
400            auth: crate::scenario::AuthConfig::default(),
401            header_commands: std::collections::HashMap::new(),
402            aws: crate::scenario::AwsConfig::default(),
403        };
404        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
405        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
406        let defs_map: HashMap<String, AssertDefinition> = definitions
407            .into_iter()
408            .map(|d| (d.name.clone(), d))
409            .collect();
410
411        Self {
412            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
413            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
414            viewport_height: scenario_config.viewport_height.unwrap_or(720),
415            applied_viewport: std::cell::Cell::new((0, 0)),
416            config: scenario_config.clone(),
417            definitions: defs_map,
418            llm,
419            endpoints,
420            usage: Arc::new(UsageTracker::new()),
421            budgets,
422            artifacts_dir: PathBuf::from(
423                scenario_config
424                    .artifacts_dir
425                    .unwrap_or_else(|| "artifacts".to_owned()),
426            ),
427            reporter,
428            current_step: std::cell::RefCell::new(None),
429        }
430    }
431
432    /// Emits an event; a sink failure degrades to a console warning so a
433    /// broken log file can never mask the run itself.
434    fn emit_event(&self, event: &TestEvent) {
435        if let Err(err) = self.reporter.emit(event) {
436            use std::io::Write as _;
437            let mut out = std::io::stderr().lock();
438            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
439        }
440    }
441
442    /// Returns a clone of the [`UsageTracker`] for reporting.
443    #[must_use]
444    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
445        Arc::clone(&self.usage)
446    }
447
448    /// Returns a reference to the [`BudgetTracker`].
449    #[must_use]
450    pub const fn budget_tracker(&self) -> &BudgetTracker {
451        &self.budgets
452    }
453
454    /// Executes all test groups in the scenario and returns a report.
455    ///
456    /// # Errors
457    ///
458    /// Returns an error if the browser fails to launch.
459    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
460    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
461        let mut report = RunReport::default();
462
463        if tests.is_empty() {
464            self.reporter.warn("No tests defined in scenario.");
465            self.emit_event(&TestEvent::RunFinished {
466                tests_passed: 0,
467                tests_failed: 0,
468                steps_passed: 0,
469                steps_failed: 0,
470                steps_skipped: 0,
471                total_cost: 0.0,
472                total_tokens: 0,
473                total_calls: 0,
474            });
475            return Ok(report);
476        }
477
478        self.emit_event(&TestEvent::RunStarted {
479            total_tests: tests.len() as u32,
480        });
481
482        let browser_headless = self.config.browser_headless.unwrap_or(true);
483
484        let launch_opts = LaunchOptions {
485            headless: browser_headless,
486            window_size: Some((self.viewport_width, self.viewport_height)),
487            sandbox: false,
488            // headless_chrome defaults this to 30s and shuts down the whole CDP
489            // connection when no messages arrive for that long. A scenario can
490            // easily exceed 30s of browser silence (slow LLM targeting/assertion
491            // calls, page waits, budget checks between steps), after which every
492            // remaining step fails with "Unable to make method calls because
493            // underlying connection is closed" — one quiet gap kills the run.
494            // Open-ended scenarios must own the connection for their full
495            // duration, so keep it alive for 6 hours.
496            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
497            ..LaunchOptions::default()
498        };
499
500        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
501        let tab = browser.new_tab().context("failed to open browser tab")?;
502        let _ = tab.set_default_timeout(self.timeout);
503
504        // Start MCP server if configured
505        #[cfg(feature = "mcp-server")]
506        if let Some(ref mcp_cfg) = self.config.mcp_server {
507            if mcp_cfg.enabled {
508                let port = mcp_cfg.port;
509                std::thread::spawn(move || {
510                    let _ = crate::mcp_server::start_mcp_server(port);
511                });
512            }
513        }
514        #[cfg(not(feature = "mcp-server"))]
515        if let Some(mcp_cfg) = &self.config.mcp_server {
516            if mcp_cfg.enabled {
517                self.reporter
518                    .warn("MCP server configured but 'mcp-server' feature not enabled");
519            }
520        }
521
522        // Start A2A agent server if configured
523        #[cfg(feature = "a2a-server")]
524        if let Some(ref a2a_cfg) = self.config.a2a_server {
525            if a2a_cfg.enabled {
526                let port = a2a_cfg.port;
527                tokio::spawn(crate::a2a_server::start_a2a_server(port));
528            }
529        }
530        #[cfg(not(feature = "a2a-server"))]
531        if let Some(a2a_cfg) = &self.config.a2a_server {
532            if a2a_cfg.enabled {
533                self.reporter
534                    .warn("A2A server configured but 'a2a-server' feature not enabled");
535            }
536        }
537
538        for test in tests {
539            self.emit_event(&TestEvent::TestStarted {
540                test: test.name.clone(),
541            });
542
543            self.usage.reset_per_test();
544
545            let test_started = Instant::now();
546            let test_result = self.run_test(test, &tab);
547            let duration_ms = test_started.elapsed().as_millis() as u64;
548            let usage = self.usage.current_test_snapshot();
549            self.usage.commit_test(&test.name);
550
551            self.emit_event(&TestEvent::TestFinished {
552                test: test.name.clone(),
553                passed: test_result.passed,
554                failed: test_result.failed,
555                skipped: test_result.skipped,
556                duration_ms,
557                cost: usage.total_cost,
558                tokens: usage.total_tokens,
559                calls: usage.total_calls,
560            });
561
562            if test_result.failed == 0 && test_result.total > 0 {
563                report.tests_passed += 1;
564            } else if test_result.total > 0 {
565                report.tests_failed += 1;
566            }
567
568            report.passed += test_result.passed;
569            report.failed += test_result.failed;
570            report.skipped += test_result.skipped;
571            report.details.extend(test_result.details);
572        }
573
574        let global = self.usage.global_snapshot();
575        self.emit_event(&TestEvent::RunFinished {
576            tests_passed: report.tests_passed,
577            tests_failed: report.tests_failed,
578            steps_passed: report.passed,
579            steps_failed: report.failed,
580            steps_skipped: report.skipped,
581            total_cost: global.total_cost,
582            total_tokens: global.total_tokens,
583            total_calls: global.total_calls,
584        });
585
586        Ok(report)
587    }
588
589    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
590    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
591        let base_url = test
592            .base_url
593            .clone()
594            .or_else(|| self.config.base_url.clone())
595            .unwrap_or_else(crate::base_url);
596
597        // Per-test viewport override: switch the browser via CDP
598        // device-metrics emulation before this test runs.
599        let vw = test.viewport_width.unwrap_or(self.viewport_width);
600        let vh = test.viewport_height.unwrap_or(self.viewport_height);
601        if self.applied_viewport.get() != (vw, vh) {
602            self.apply_viewport(tab, vw, vh);
603            self.applied_viewport.set((vw, vh));
604        }
605
606        // Per-test isolation: every test starts from its own start_url
607        // (unless auto_navigate is disabled), so a test never inherits the
608        // previous test's page state.
609        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
610
611        let start_url = test
612            .start_url
613            .clone()
614            .or_else(|| self.config.start_url.clone())
615            .unwrap_or_else(|| "/dashboard".to_owned());
616
617        if auto_navigate {
618            let full_url = resolve_url(&start_url, &base_url);
619            self.reporter.debug(format!("auto-navigate: {full_url}"));
620            let _ = tab.navigate_to(&full_url);
621            let _ = tab.wait_until_navigated();
622            std::thread::sleep(Duration::from_secs(4));
623        }
624
625        let mut result = TestRunResult::default();
626
627        for (step_index, step) in test.steps.iter().enumerate() {
628            result.total += 1;
629
630            let wait_ms = match step {
631                TestStep::Navigate { wait_after_ms, .. }
632                | TestStep::Click { wait_after_ms, .. }
633                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
634                _ => None,
635            };
636
637            self.current_step
638                .replace(Some((test.name.clone(), step_index as u32)));
639            self.emit_event(&TestEvent::StepStarted {
640                test: test.name.clone(),
641                index: step_index as u32,
642                label: step_label(step),
643            });
644            let step_started = Instant::now();
645
646            let mut step_result = match step {
647                TestStep::Navigate { url, .. } => {
648                    let full_url = resolve_url(url, &base_url);
649                    run_navigate_step(&full_url, tab)
650                }
651                TestStep::Click {
652                    target,
653                    selector,
654                    endpoint,
655                    idempotent,
656                    ..
657                } => self.run_click(
658                    target,
659                    selector.as_deref(),
660                    endpoint.as_deref(),
661                    test.endpoint.as_deref(),
662                    *idempotent,
663                    tab,
664                ),
665                TestStep::Type {
666                    target,
667                    text,
668                    selector,
669                    endpoint,
670                    idempotent,
671                    ..
672                } => self.run_type(
673                    target,
674                    text,
675                    selector.as_deref(),
676                    endpoint.as_deref(),
677                    test.endpoint.as_deref(),
678                    *idempotent,
679                    tab,
680                ),
681                TestStep::Wait {
682                    target,
683                    selector,
684                    text,
685                    timeout_ms,
686                    endpoint,
687                    idempotent,
688                } => self.run_wait(
689                    target,
690                    selector.as_deref(),
691                    text.as_deref(),
692                    *timeout_ms,
693                    endpoint.as_deref(),
694                    test.endpoint.as_deref(),
695                    *idempotent,
696                    tab,
697                ),
698                TestStep::Assert {
699                    definition,
700                    preset,
701                    prompt,
702                    assert_text,
703                    endpoint,
704                    screenshot,
705                } => self.run_assert(
706                    definition.as_deref(),
707                    preset.as_deref(),
708                    prompt.as_deref(),
709                    assert_text.as_deref(),
710                    *screenshot,
711                    endpoint.as_deref(),
712                    test.endpoint.as_deref(),
713                    tab,
714                ),
715                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
716                TestStep::Agent {
717                    agent,
718                    task,
719                    definition,
720                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
721                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
722            };
723
724            // Failure diagnostics: capture the page state and a screenshot so
725            // CI logs say WHAT the page looked like when the step failed,
726            // instead of a bare "timed out: The event waited for never came".
727            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
728                let state = diagnostics::capture(tab);
729                let screenshot = diagnostics::save_screenshot(
730                    tab,
731                    &self.artifacts_dir,
732                    &test.name,
733                    &test.name,
734                    step_index,
735                    step_kind_label(step),
736                );
737                step_result.message = format!(
738                    "{base} — {excerpt}",
739                    base = step_result.message,
740                    excerpt = diagnostics::inline_excerpt(&state),
741                );
742                (Some(diagnostics::full_context(&state)), screenshot)
743            } else {
744                (None, None)
745            };
746
747            let duration_ms = step_started.elapsed().as_millis() as u64;
748            self.emit_event(&TestEvent::StepFinished {
749                test: test.name.clone(),
750                index: step_index as u32,
751                label: step_result.name.clone(),
752                status: step_result.status,
753                duration_ms,
754                message: step_result.message.clone(),
755                diagnostics: diagnostics_block,
756                screenshot: screenshot_path,
757            });
758            self.current_step.replace(None);
759
760            match step_result.status {
761                StepStatus::Passed => result.passed += 1,
762                StepStatus::Failed => result.failed += 1,
763                StepStatus::Skipped => result.skipped += 1,
764            }
765
766            // Fail fast: the first failed step ends the test and the
767            // remaining steps are reported as skipped (no LLM budget is
768            // burned asserting against a page that is already known broken).
769            if step_result.status == StepStatus::Failed
770                && !self.config.continue_on_failure
771                && step_index + 1 < test.steps.len()
772            {
773                self.emit_event(&TestEvent::Warning {
774                    message: format!(
775                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
776                        test.steps.len() - step_index - 1
777                    ),
778                });
779                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
780                    let skipped_index = step_index + 1 + offset;
781                    let label = step_label(skipped);
782                    result.total += 1;
783                    result.skipped += 1;
784                    self.emit_event(&TestEvent::StepStarted {
785                        test: test.name.clone(),
786                        index: skipped_index as u32,
787                        label: label.clone(),
788                    });
789                    self.emit_event(&TestEvent::StepFinished {
790                        test: test.name.clone(),
791                        index: skipped_index as u32,
792                        label,
793                        status: StepStatus::Skipped,
794                        duration_ms: 0,
795                        message: "skipped: previous step failed".into(),
796                        diagnostics: None,
797                        screenshot: None,
798                    });
799                    result.details.push(StepResult {
800                        name: step_label(skipped),
801                        status: StepStatus::Skipped,
802                        message: "skipped: previous step failed".into(),
803                    });
804                }
805                result.details.push(step_result);
806                return result;
807            }
808
809            // Check per-test budget after each step
810            let test_usage = self.usage.current_test_snapshot();
811            let global_usage = self.usage.global_snapshot();
812            let budget_status = self.budgets.check_all(
813                &test.name,
814                &test_usage,
815                &global_usage,
816                test.budget.as_ref(),
817            );
818            match budget_status {
819                BudgetStatus::HardExceeded { message, .. } => {
820                    self.emit_event(&TestEvent::Warning {
821                        message: format!("budget exceeded: {message}"),
822                    });
823                    result.details.push(StepResult {
824                        name: "[budget]".into(),
825                        status: StepStatus::Failed,
826                        message,
827                    });
828                    result.failed += 1;
829                    return result;
830                }
831                BudgetStatus::SoftExceeded { message, .. } => {
832                    self.emit_event(&TestEvent::Warning {
833                        message: format!("budget warning: {message}"),
834                    });
835                }
836                BudgetStatus::Ok => {}
837            }
838
839            if let Some(ms) = wait_ms {
840                std::thread::sleep(Duration::from_millis(ms));
841            }
842
843            result.details.push(step_result);
844        }
845
846        result
847    }
848
849    /// Applies a viewport size to the current tab via CDP
850    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
851    /// overrides and the viewport matrix. The initial window size set at
852    /// browser launch is replaced by emulation; failures are logged but
853    /// do not fail the test (a mismatched viewport only weakens coverage).
854    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
855        use headless_chrome::protocol::cdp::Emulation;
856        let _ = self;
857        let params = Emulation::SetDeviceMetricsOverride {
858            width,
859            height,
860            device_scale_factor: 1.0,
861            mobile: false,
862            scale: None,
863            screen_width: Some(width),
864            screen_height: Some(height),
865            position_x: None,
866            position_y: None,
867            dont_set_visible_size: None,
868            screen_orientation: None,
869            viewport: None,
870            display_feature: None,
871            device_posture: None,
872        };
873        self.reporter.debug(format!("viewport: {width}x{height}"));
874        if let Err(e) = tab.call_method(params) {
875            self.reporter
876                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
877        }
878    }
879
880    /// Height (px) covered by assert-step screenshots: the configured
881    /// `screenshot_max_height` (absolute px or viewport multiple, default
882    /// `"20x"`) resolved against the currently applied viewport, raised to
883    /// at least the viewport height so the visible screen is always fully
884    /// included. The capture is split into viewport-tall tiles, so this
885    /// value bounds total coverage (and hence the number of image parts).
886    #[must_use]
887    fn screenshot_height_cap(&self) -> u32 {
888        let viewport_height = self.current_viewport_height();
889        let cap = self
890            .config
891            .screenshot_max_height
892            .as_ref()
893            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
894        cap.max(viewport_height)
895    }
896
897    /// Height of the viewport currently emulated in the browser (falling
898    /// back to the configured default before any emulation was applied).
899    #[must_use]
900    const fn current_viewport_height(&self) -> u32 {
901        let (_, height) = self.applied_viewport.get();
902        if height > 0 {
903            height
904        } else {
905            self.viewport_height
906        }
907    }
908
909    // ── step handlers ───────────────────────────────────────────────────
910
911    #[allow(clippy::too_many_lines)]
912    fn run_click(
913        &self,
914        target: &str,
915        selector_override: Option<&str>,
916        step_endpoint: Option<&str>,
917        test_endpoint: Option<&str>,
918        idempotent: bool,
919        tab: &Tab,
920    ) -> StepResult {
921        let name = format!("[click] {target}");
922        let selector = match self.resolve_selector(
923            selector_override,
924            target,
925            step_endpoint,
926            test_endpoint,
927            tab,
928        ) {
929            Ok(s) => s,
930            Err(msg) => {
931                if idempotent {
932                    return StepResult {
933                        name,
934                        status: StepStatus::Skipped,
935                        message: format!("skipped (idempotent): no target found — {msg}"),
936                    };
937                }
938                return StepResult {
939                    name,
940                    status: StepStatus::Failed,
941                    message: msg,
942                };
943            }
944        };
945
946        // Idempotent steps probe briefly: a missing target means the
947        // action was already done / not applicable (e.g. an
948        // already-authenticated session), and skipping is the success
949        // path, not a failure.
950        let probe_secs = if idempotent { 5 } else { 10 };
951        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
952            Ok(element) => match element.click() {
953                Ok(_) => StepResult {
954                    name,
955                    status: StepStatus::Passed,
956                    message: format!("clicked {selector}"),
957                },
958                Err(e) => StepResult {
959                    name,
960                    status: StepStatus::Failed,
961                    message: format!("click failed on {selector}: {e}"),
962                },
963            },
964            Err(e) if idempotent => StepResult {
965                name,
966                status: StepStatus::Skipped,
967                message: format!("skipped (idempotent): element {selector} not present — {e}"),
968            },
969            Err(e) => StepResult {
970                name,
971                status: StepStatus::Failed,
972                message: format!("element {selector} not found: {e}"),
973            },
974        }
975    }
976
977    #[allow(clippy::too_many_arguments)]
978    fn run_type(
979        &self,
980        target: &str,
981        text: &str,
982        selector_override: Option<&str>,
983        step_endpoint: Option<&str>,
984        test_endpoint: Option<&str>,
985        idempotent: bool,
986        tab: &Tab,
987    ) -> StepResult {
988        let name = format!("[type] {target}");
989        let selector = match self.resolve_selector(
990            selector_override,
991            target,
992            step_endpoint,
993            test_endpoint,
994            tab,
995        ) {
996            Ok(s) => s,
997            Err(msg) => {
998                if idempotent {
999                    return StepResult {
1000                        name,
1001                        status: StepStatus::Skipped,
1002                        message: format!("skipped (idempotent): no target found — {msg}"),
1003                    };
1004                }
1005                return StepResult {
1006                    name,
1007                    status: StepStatus::Failed,
1008                    message: msg,
1009                };
1010            }
1011        };
1012
1013        let probe_secs = if idempotent { 5 } else { 10 };
1014        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1015            Ok(element) => {
1016                if let Err(e) = element.click() {
1017                    return StepResult {
1018                        name,
1019                        status: StepStatus::Failed,
1020                        message: format!("click to focus {selector} failed: {e}"),
1021                    };
1022                }
1023
1024                let js = format!(
1025                    "document.querySelector('{}').value = '';",
1026                    selector.replace('\'', "\\'")
1027                );
1028                let _ = tab.evaluate(&js, false);
1029
1030                match element.type_into(text) {
1031                    Ok(_) => StepResult {
1032                        name,
1033                        status: StepStatus::Passed,
1034                        message: format!("typed {text:?} into {selector}"),
1035                    },
1036                    Err(e) => StepResult {
1037                        name,
1038                        status: StepStatus::Failed,
1039                        message: format!("type into {selector} failed: {e}"),
1040                    },
1041                }
1042            }
1043            Err(e) if idempotent => StepResult {
1044                name,
1045                status: StepStatus::Skipped,
1046                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1047            },
1048            Err(e) => StepResult {
1049                name,
1050                status: StepStatus::Failed,
1051                message: format!("element {selector} not found: {e}"),
1052            },
1053        }
1054    }
1055
1056    #[allow(clippy::too_many_arguments)]
1057    #[allow(clippy::too_many_lines)]
1058    fn run_wait(
1059        &self,
1060        target: &str,
1061        selector_override: Option<&str>,
1062        text: Option<&str>,
1063        timeout_ms: Option<u64>,
1064        step_endpoint: Option<&str>,
1065        test_endpoint: Option<&str>,
1066        idempotent: bool,
1067        tab: &Tab,
1068    ) -> StepResult {
1069        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1070        let step_name = format!("[wait] {target}");
1071
1072        // Resolve an explicit selector only (text-only waits are LLM-free).
1073        let selector = match selector_override {
1074            Some(s) => Some(s.to_owned()),
1075            None if text.is_some() => None,
1076            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1077                Ok(s) => Some(s),
1078                Err(msg) => {
1079                    if idempotent {
1080                        return StepResult {
1081                            name: step_name,
1082                            status: StepStatus::Skipped,
1083                            message: format!("skipped (idempotent): no target found — {msg}"),
1084                        };
1085                    }
1086                    return StepResult {
1087                        name: step_name,
1088                        status: StepStatus::Failed,
1089                        message: msg,
1090                    };
1091                }
1092            },
1093        };
1094
1095        if text.is_some() {
1096            let sel_js = selector
1097                .as_deref()
1098                .map(crate::selectors::selector_matches_js);
1099            let text_js = text.map(|t| {
1100                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1101                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1102            });
1103
1104            let deadline = Instant::now() + timeout;
1105            loop {
1106                let sel_ok = sel_js
1107                    .as_ref()
1108                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1109                let text_ok = text_js
1110                    .as_ref()
1111                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1112                if sel_ok && text_ok {
1113                    let mut what = Vec::new();
1114                    if let Some(sel) = &selector {
1115                        what.push(format!("found {sel}"));
1116                    }
1117                    if let Some(t) = text {
1118                        what.push(format!("text {t:?} visible"));
1119                    }
1120                    return StepResult {
1121                        name: step_name,
1122                        status: StepStatus::Passed,
1123                        message: what.join(" and "),
1124                    };
1125                }
1126                if Instant::now() >= deadline {
1127                    let mut what = Vec::new();
1128                    if let Some(sel) = &selector {
1129                        what.push(sel.clone());
1130                    }
1131                    if let Some(t) = text {
1132                        what.push(format!("text {t:?}"));
1133                    }
1134                    let message = format!(
1135                        "wait for {} timed out after {}ms: the event waited for never came",
1136                        what.join(" / "),
1137                        timeout.as_millis(),
1138                    );
1139                    if idempotent {
1140                        return StepResult {
1141                            name: step_name,
1142                            status: StepStatus::Skipped,
1143                            message: format!("skipped (idempotent): {message}"),
1144                        };
1145                    }
1146                    return StepResult {
1147                        name: step_name,
1148                        status: StepStatus::Failed,
1149                        message,
1150                    };
1151                }
1152                std::thread::sleep(Duration::from_millis(250));
1153            }
1154        }
1155
1156        match selector.as_deref() {
1157            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1158                Ok(_) => StepResult {
1159                    name: step_name,
1160                    status: StepStatus::Passed,
1161                    message: format!("found {sel}"),
1162                },
1163                Err(e) if idempotent => StepResult {
1164                    name: step_name,
1165                    status: StepStatus::Skipped,
1166                    message: format!(
1167                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1168                        timeout.as_millis()
1169                    ),
1170                },
1171                Err(e) => StepResult {
1172                    name: step_name,
1173                    status: StepStatus::Failed,
1174                    message: format!(
1175                        "wait for {sel} timed out after {}ms: {e}",
1176                        timeout.as_millis()
1177                    ),
1178                },
1179            },
1180            None => StepResult {
1181                name: step_name,
1182                status: StepStatus::Failed,
1183                message: "wait step has neither selector nor text".into(),
1184            },
1185        }
1186    }
1187
1188    #[allow(clippy::too_many_arguments)]
1189    fn run_assert(
1190        &self,
1191        definition: Option<&str>,
1192        preset: Option<&str>,
1193        prompt: Option<&str>,
1194        assert_text: Option<&str>,
1195        screenshot: bool,
1196        step_endpoint: Option<&str>,
1197        test_endpoint: Option<&str>,
1198        tab: &Tab,
1199    ) -> StepResult {
1200        std::thread::sleep(Duration::from_millis(500));
1201
1202        let page_content = get_page_text(tab);
1203
1204        // Vision attach: capture the full page once per assert step and
1205        // split it into viewport-tall tiles (the total coverage is bounded
1206        // by the configured height cap so vision tokens stay sane). All
1207        // tile data URLs are handed to the preset/prompt evaluation below.
1208        let image: Option<Vec<String>> = if screenshot {
1209            let endpoint = self
1210                .endpoints
1211                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1212            if !endpoint.vision {
1213                return StepResult {
1214                    name: "[assert]".into(),
1215                    status: StepStatus::Failed,
1216                    message: format!(
1217                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1218                        name = endpoint.name
1219                    ),
1220                };
1221            }
1222            match crate::vision::capture_screenshot_data_urls(
1223                tab,
1224                self.config
1225                    .screenshot_max_dimension
1226                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1227                self.screenshot_height_cap(),
1228                self.current_viewport_height(),
1229            ) {
1230                Ok(urls) => Some(urls),
1231                Err(e) => {
1232                    return StepResult {
1233                        name: "[assert]".into(),
1234                        status: StepStatus::Failed,
1235                        message: format!("screenshot capture failed: {e}"),
1236                    };
1237                }
1238            }
1239        } else {
1240            None
1241        };
1242
1243        if let Some(def_name) = definition {
1244            if let Some(def) = self.definitions.get(def_name) {
1245                return self.run_assert_def(
1246                    def,
1247                    &page_content,
1248                    image.as_deref(),
1249                    step_endpoint,
1250                    test_endpoint,
1251                    tab,
1252                );
1253            }
1254            return StepResult {
1255                name: format!("[assert] {def_name}"),
1256                status: StepStatus::Failed,
1257                message: format!("definition '{def_name}' not found"),
1258            };
1259        }
1260
1261        if let Some(preset_name) = preset {
1262            // Deterministic DOM layout scan — runs JS in the browser and
1263            // never calls the LLM (free, fast, no pixel budget).
1264            if preset_name == "layout_no_issues" {
1265                return self.run_layout_preset(tab);
1266            }
1267            return self.run_preset(
1268                preset_name,
1269                assert_text,
1270                &page_content,
1271                image.as_deref(),
1272                step_endpoint,
1273                test_endpoint,
1274            );
1275        }
1276
1277        if let Some(prompt_text) = prompt {
1278            return self.run_custom(
1279                prompt_text,
1280                &page_content,
1281                image.as_deref(),
1282                step_endpoint,
1283                test_endpoint,
1284            );
1285        }
1286
1287        StepResult {
1288            name: "[assert]".into(),
1289            status: StepStatus::Skipped,
1290            message: "no definition, preset, or prompt specified".into(),
1291        }
1292    }
1293
1294    fn run_assert_def(
1295        &self,
1296        def: &AssertDefinition,
1297        page_content: &PageContent,
1298        image: Option<&[String]>,
1299        step_endpoint: Option<&str>,
1300        test_endpoint: Option<&str>,
1301        tab: &Tab,
1302    ) -> StepResult {
1303        // Agent-based definition: delegate to an A2A agent
1304        if let Some(ref agent) = def.agent {
1305            if image.is_some() {
1306                return StepResult {
1307                    name: format!("[assert] {}", def.name),
1308                    status: StepStatus::Failed,
1309                    message: "agent-backed assertions do not support screenshots".into(),
1310                };
1311            }
1312            let task = def
1313                .task_template
1314                .as_deref()
1315                .unwrap_or("Evaluate the assertion")
1316                .replace("{url}", &page_content.url)
1317                .replace("{title}", &page_content.title)
1318                .replace("{content}", &page_content.body_text)
1319                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1320
1321            return self.run_agent_step(agent, &task, &def.name);
1322        }
1323
1324        // Custom preset: system + user_template provided in the definition
1325        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1326            return self.run_custom_preset(
1327                &def.name,
1328                system,
1329                template,
1330                def.assert_text.as_deref(),
1331                page_content,
1332                image,
1333                step_endpoint,
1334                test_endpoint,
1335            );
1336        }
1337
1338        def.preset.as_ref().map_or_else(
1339            || {
1340                def.prompt.as_ref().map_or_else(
1341                    || StepResult {
1342                        name: format!("[assert] {}", def.name),
1343                        status: StepStatus::Failed,
1344                        message: "definition has no preset, prompt, or system+user_template".into(),
1345                    },
1346                    |prompt| {
1347                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1348                    },
1349                )
1350            },
1351            |preset_name| {
1352                if preset_name == "layout_no_issues" {
1353                    return self.run_layout_preset(tab);
1354                }
1355                self.run_preset(
1356                    preset_name,
1357                    def.assert_text.as_deref(),
1358                    page_content,
1359                    image,
1360                    step_endpoint,
1361                    test_endpoint,
1362                )
1363            },
1364        )
1365    }
1366
1367    #[allow(clippy::too_many_arguments)]
1368    fn run_custom_preset(
1369        &self,
1370        name: &str,
1371        system: &str,
1372        template: &str,
1373        assert_text: Option<&str>,
1374        page_content: &PageContent,
1375        image: Option<&[String]>,
1376        step_endpoint: Option<&str>,
1377        test_endpoint: Option<&str>,
1378    ) -> StepResult {
1379        let user_prompt = template
1380            .replace("{url}", &page_content.url)
1381            .replace("{title}", &page_content.title)
1382            .replace("{content}", &page_content.body_text)
1383            .replace("{expected_text}", assert_text.unwrap_or(""))
1384            .replace("{description}", "");
1385
1386        // Custom preset definitions frequently forget the {content}
1387        // placeholder — without it the LLM has no page to evaluate and
1388        // answers "I can't determine that without seeing the page". Always
1389        // append the page context unless the template already references it.
1390        let user_prompt = if template.contains("{content}") {
1391            user_prompt
1392        } else {
1393            format!(
1394                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1395                url = page_content.url,
1396                title = page_content.title,
1397                content = page_content.body_text,
1398            )
1399        };
1400
1401        self.reporter
1402            .debug(format!("assert: {name} (custom preset)"));
1403
1404        let chain = self
1405            .endpoints
1406            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1407        let sys = system.to_owned();
1408
1409        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1410
1411        response.map_or_else(
1412            |e| StepResult {
1413                name: format!("[assert] {name}"),
1414                status: StepStatus::Failed,
1415                message: format!("LLM assertion call failed: {e}"),
1416            },
1417            |(lr, idx)| {
1418                self.usage.record_llm_call(
1419                    &chain[idx].name,
1420                    chain[idx],
1421                    lr.usage.prompt_tokens,
1422                    lr.usage.completion_tokens,
1423                );
1424                let content_lower = lr.content.to_lowercase().trim().to_owned();
1425                if content_lower.starts_with("pass") {
1426                    StepResult {
1427                        name: format!("[assert] {name}"),
1428                        status: StepStatus::Passed,
1429                        message: "PASS".into(),
1430                    }
1431                } else {
1432                    StepResult {
1433                        name: format!("[assert] {name}"),
1434                        status: StepStatus::Failed,
1435                        message: lr.content,
1436                    }
1437                }
1438            },
1439        )
1440    }
1441
1442    fn run_preset(
1443        &self,
1444        preset_name: &str,
1445        assert_text: Option<&str>,
1446        page_content: &PageContent,
1447        image: Option<&[String]>,
1448        step_endpoint: Option<&str>,
1449        test_endpoint: Option<&str>,
1450    ) -> StepResult {
1451        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1452            return StepResult {
1453                name: format!("[assert] {preset_name}"),
1454                status: StepStatus::Failed,
1455                message: format!("unknown assertion preset: {preset_name}"),
1456            };
1457        };
1458        if preset_name.starts_with("visual_") && image.is_none() {
1459            return StepResult {
1460                name: format!("[assert] {preset_name}"),
1461                status: StepStatus::Failed,
1462                message: format!(
1463                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1464                ),
1465            };
1466        }
1467
1468        let user_prompt = preset
1469            .user_template
1470            .replace("{url}", &page_content.url)
1471            .replace("{title}", &page_content.title)
1472            .replace("{content}", &page_content.body_text)
1473            .replace("{expected_text}", assert_text.unwrap_or(""))
1474            .replace("{description}", "");
1475
1476        // Same safety net as custom presets: never let the LLM answer with
1477        // no page context at all.
1478        let user_prompt = if preset.user_template.contains("{content}") {
1479            user_prompt
1480        } else {
1481            format!(
1482                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1483                url = page_content.url,
1484                title = page_content.title,
1485                content = page_content.body_text,
1486            )
1487        };
1488
1489        self.reporter.debug(format!("assert: {preset_name}"));
1490
1491        let chain = self
1492            .endpoints
1493            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1494        let sys = preset.system.to_owned();
1495
1496        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1497
1498        response.map_or_else(
1499            |e| StepResult {
1500                name: format!("[assert] {preset_name}"),
1501                status: StepStatus::Failed,
1502                message: format!("LLM assertion call failed: {e}"),
1503            },
1504            |(lr, idx)| {
1505                self.usage.record_llm_call(
1506                    &chain[idx].name,
1507                    chain[idx],
1508                    lr.usage.prompt_tokens,
1509                    lr.usage.completion_tokens,
1510                );
1511                let content_lower = lr.content.to_lowercase().trim().to_owned();
1512                if content_lower.starts_with("pass") {
1513                    StepResult {
1514                        name: format!("[assert] {preset_name}"),
1515                        status: StepStatus::Passed,
1516                        message: "PASS".into(),
1517                    }
1518                } else {
1519                    StepResult {
1520                        name: format!("[assert] {preset_name}"),
1521                        status: StepStatus::Failed,
1522                        message: lr.content,
1523                    }
1524                }
1525            },
1526        )
1527    }
1528
1529    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1530    ///
1531    /// Evaluates the layout-scan JS in the page and fails with the list of
1532    /// detected issues: horizontal page overflow, visible elements sticking
1533    /// out of the viewport, text clipped by `overflow: hidden` containers,
1534    /// and interactive elements covered by other elements. No LLM call —
1535    /// checks are geometry-based so the check is free, deterministic, and
1536    /// safe to run on every page × viewport variant.
1537    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1538        let name = "[assert] layout_no_issues".to_owned();
1539        self.reporter
1540            .debug("assert: layout_no_issues (DOM layout scan)");
1541        let js = LAYOUT_SCAN_JS.replace(
1542            "__IGNORE_CLASSES__",
1543            &serde_json::to_string(&self.config.layout_ignore_classes)
1544                .unwrap_or_else(|_| "[]".to_owned()),
1545        );
1546        let result = tab.evaluate(&js, false);
1547        let json_str = match result {
1548            Ok(r) => r
1549                .value
1550                .as_ref()
1551                .and_then(|v| v.as_str().map(String::from))
1552                .unwrap_or_else(|| "[]".to_owned()),
1553            Err(e) => {
1554                return StepResult {
1555                    name,
1556                    status: StepStatus::Failed,
1557                    message: format!("layout scan JS failed: {e}"),
1558                };
1559            }
1560        };
1561        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1562        if issues.is_empty() {
1563            return StepResult {
1564                name,
1565                status: StepStatus::Passed,
1566                message: "PASS — no layout defects detected".into(),
1567            };
1568        }
1569        let mut lines: Vec<String> = issues
1570            .iter()
1571            .take(10)
1572            .map(|i| {
1573                format!(
1574                    "- [{type_}] {element}: {detail}",
1575                    type_ = i.issue_type,
1576                    element = i.element,
1577                    detail = i.detail
1578                )
1579            })
1580            .collect();
1581        if issues.len() > 10 {
1582            lines.push(format!("- … and {} more", issues.len() - 10));
1583        }
1584        StepResult {
1585            name,
1586            status: StepStatus::Failed,
1587            message: format!(
1588                "FAIL — {} layout defect(s) detected:\n{}",
1589                issues.len(),
1590                lines.join("\n")
1591            ),
1592        }
1593    }
1594
1595    fn run_custom(
1596        &self,
1597        prompt: &str,
1598        page_content: &PageContent,
1599        image: Option<&[String]>,
1600        step_endpoint: Option<&str>,
1601        test_endpoint: Option<&str>,
1602    ) -> StepResult {
1603        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1604
1605        let mut user = format!(
1606            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1607            url = page_content.url,
1608            title = page_content.title,
1609            content = page_content.body_text,
1610        );
1611        if image.is_some() {
1612            user.push_str(
1613                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1614            );
1615        }
1616
1617        self.reporter.debug("custom assert");
1618
1619        let chain = self
1620            .endpoints
1621            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1622        let sys = system.to_owned();
1623
1624        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1625
1626        response.map_or_else(
1627            |e| StepResult {
1628                name: "[assert] custom".into(),
1629                status: StepStatus::Failed,
1630                message: format!("LLM assertion call failed: {e}"),
1631            },
1632            |(lr, idx)| {
1633                self.usage.record_llm_call(
1634                    &chain[idx].name,
1635                    chain[idx],
1636                    lr.usage.prompt_tokens,
1637                    lr.usage.completion_tokens,
1638                );
1639                let content_lower = lr.content.to_lowercase().trim().to_owned();
1640                if content_lower.starts_with("pass") {
1641                    StepResult {
1642                        name: "[assert] custom".into(),
1643                        status: StepStatus::Passed,
1644                        message: "PASS".into(),
1645                    }
1646                } else {
1647                    StepResult {
1648                        name: "[assert] custom".into(),
1649                        status: StepStatus::Failed,
1650                        message: lr.content,
1651                    }
1652                }
1653            },
1654        )
1655    }
1656
1657    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1658        let path = path.unwrap_or("screenshot.png");
1659
1660        match tab.capture_screenshot(
1661            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1662            None,
1663            None,
1664            true,
1665        ) {
1666            Ok(data) => {
1667                if let Err(e) = std::fs::write(path, &data) {
1668                    return StepResult {
1669                        name: format!("[screenshot] {path}"),
1670                        status: StepStatus::Failed,
1671                        message: format!("failed to write screenshot: {e}"),
1672                    };
1673                }
1674                StepResult {
1675                    name: format!("[screenshot] {path}"),
1676                    status: StepStatus::Passed,
1677                    message: format!("saved to {path}"),
1678                }
1679            }
1680            Err(e) => StepResult {
1681                name: format!("[screenshot] {path}"),
1682                status: StepStatus::Failed,
1683                message: format!("screenshot failed: {e}"),
1684            },
1685        }
1686    }
1687
1688    /// Runs an A2A agent step.
1689    #[allow(clippy::literal_string_with_formatting_args)]
1690    fn run_agent(
1691        &self,
1692        agent_name: &str,
1693        task: &str,
1694        definition: Option<&str>,
1695        _test_endpoint: Option<&str>,
1696    ) -> StepResult {
1697        // If a definition is specified, look up the task template
1698        let resolved_task = if let Some(def_name) = definition {
1699            if let Some(def) = self.definitions.get(def_name) {
1700                let tmpl = def.task_template.as_deref().unwrap_or(task);
1701                tmpl.replace("{task}", task)
1702            } else {
1703                return StepResult {
1704                    name: format!("[agent] {def_name}"),
1705                    status: StepStatus::Failed,
1706                    message: format!("definition '{def_name}' not found"),
1707                };
1708            }
1709        } else {
1710            task.to_owned()
1711        };
1712
1713        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1714    }
1715
1716    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1717        let Some(ep) = self.endpoints.get(agent_name) else {
1718            return StepResult {
1719                name: format!("[agent] {display_name}"),
1720                status: StepStatus::Failed,
1721                message: format!("agent endpoint '{agent_name}' not found"),
1722            };
1723        };
1724
1725        if ep.url.is_empty() {
1726            return StepResult {
1727                name: format!("[agent] {display_name}"),
1728                status: StepStatus::Failed,
1729                message: format!("agent endpoint '{agent_name}' has no URL"),
1730            };
1731        }
1732
1733        self.reporter.debug(format!("agent {agent_name}: {task}"));
1734
1735        let url = ep.url.clone();
1736        let client = A2aClient::new(&url, self.timeout);
1737        let task_clone = task.to_owned();
1738
1739        let response = std::thread::spawn(move || {
1740            let rt = tokio::runtime::Builder::new_current_thread()
1741                .enable_all()
1742                .build()
1743                .unwrap();
1744            rt.block_on(client.send_task(&task_clone))
1745        })
1746        .join()
1747        .unwrap();
1748
1749        // Record the flat-cost call
1750        self.usage.record_flat_call(agent_name, ep);
1751
1752        match response {
1753            Ok(text) => {
1754                let clean = text.trim().to_owned();
1755                let lower = clean.to_lowercase();
1756                if lower.starts_with("pass") {
1757                    StepResult {
1758                        name: format!("[agent] {display_name}"),
1759                        status: StepStatus::Passed,
1760                        message: format!("PASS: {clean}"),
1761                    }
1762                } else if lower.starts_with("fail") {
1763                    StepResult {
1764                        name: format!("[agent] {display_name}"),
1765                        status: StepStatus::Failed,
1766                        message: clean,
1767                    }
1768                } else {
1769                    StepResult {
1770                        name: format!("[agent] {display_name}"),
1771                        status: StepStatus::Passed,
1772                        message: format!("response: {clean}"),
1773                    }
1774                }
1775            }
1776            Err(e) => StepResult {
1777                name: format!("[agent] {display_name}"),
1778                status: StepStatus::Failed,
1779                message: format!("agent call failed: {e}"),
1780            },
1781        }
1782    }
1783
1784    /// Runs an MCP tool call step.
1785    fn run_mcp(
1786        &self,
1787        server_name: &str,
1788        tool_name: &str,
1789        args: Option<&serde_json::Value>,
1790    ) -> StepResult {
1791        let Some(ep) = self.endpoints.get(server_name) else {
1792            return StepResult {
1793                name: format!("[mcp] {server_name}:{tool_name}"),
1794                status: StepStatus::Failed,
1795                message: format!("MCP server endpoint '{server_name}' not found"),
1796            };
1797        };
1798
1799        let cmd = ep.command.as_deref().unwrap_or("");
1800        if cmd.is_empty() {
1801            return StepResult {
1802                name: format!("[mcp] {server_name}:{tool_name}"),
1803                status: StepStatus::Failed,
1804                message: format!("MCP server '{server_name}' has no command configured"),
1805            };
1806        }
1807
1808        self.reporter
1809            .debug(format!("mcp {server_name} {tool_name}"));
1810
1811        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1812
1813        let command = cmd.to_owned();
1814        let args_vec = ep.args.clone();
1815        let tool = tool_name.to_owned();
1816
1817        let response = std::thread::spawn(move || {
1818            let mut mcp_client =
1819                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1820            mcp_client
1821                .call_tool(&tool, &args_val)
1822                .map_err(|e| e.to_string())
1823        })
1824        .join()
1825        .unwrap();
1826
1827        // Record the flat-cost call
1828        self.usage.record_flat_call(server_name, ep);
1829
1830        match response {
1831            Ok(result) => {
1832                if result.isError {
1833                    StepResult {
1834                        name: format!("[mcp] {server_name}:{tool_name}"),
1835                        status: StepStatus::Failed,
1836                        message: result.to_string(),
1837                    }
1838                } else {
1839                    StepResult {
1840                        name: format!("[mcp] {server_name}:{tool_name}"),
1841                        status: StepStatus::Passed,
1842                        message: result.to_string(),
1843                    }
1844                }
1845            }
1846            Err(e) => StepResult {
1847                name: format!("[mcp] {server_name}:{tool_name}"),
1848                status: StepStatus::Failed,
1849                message: format!("MCP call failed: {e}"),
1850            },
1851        }
1852    }
1853
1854    // ── helpers ──────────────────────────────────────────────────────────
1855
1856    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1857    /// the runner's default LLM config for any unset fields.
1858    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1859        LlmConfig {
1860            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
1861                // Bedrock builds its endpoint from the resolved AWS region
1862                // when no URL is given — never inherit the default LLM URL.
1863                endpoint.url.clone()
1864            } else if endpoint.url.is_empty() {
1865                self.llm.url.clone()
1866            } else {
1867                endpoint.url.clone()
1868            },
1869            model: endpoint
1870                .model
1871                .clone()
1872                .unwrap_or_else(|| self.llm.model.clone()),
1873            api_key: endpoint
1874                .api_key
1875                .clone()
1876                .or_else(|| self.llm.api_key.clone()),
1877            headers: if endpoint.headers.is_empty() {
1878                self.llm.headers.clone()
1879            } else {
1880                endpoint.headers.clone()
1881            },
1882            timeout: self.llm.timeout,
1883            temperature: self.llm.temperature,
1884            thinking: self.llm.thinking,
1885            model_params: self.llm.model_params.clone(),
1886            max_attempts: endpoint.max_attempts.max(1),
1887            provider: endpoint.provider,
1888            deployment: endpoint.deployment.clone(),
1889            api_version: endpoint.api_version.clone(),
1890            auth: endpoint.auth.clone(),
1891            header_commands: endpoint.header_commands.clone(),
1892            aws: endpoint.aws.clone(),
1893        }
1894    }
1895
1896    /// Runs a single LLM call against an ordered endpoint chain (primary +
1897    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
1898    /// the first endpoint that answers wins. Returns the response together
1899    /// with the chain index of the answering endpoint (0 = primary) so the
1900    /// caller can attribute usage to the correct endpoint.
1901    ///
1902    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
1903    /// shows duration, tokens, cost and the answering endpoint per call.
1904    #[allow(clippy::cast_possible_truncation)]
1905    fn llm_call_chain(
1906        &self,
1907        chain: &[&ResolvedEndpoint],
1908        system: &str,
1909        user: &str,
1910        image: Option<&[String]>,
1911        purpose: &str,
1912    ) -> Result<(crate::costs::LlmResponse, usize), String> {
1913        if chain.is_empty() {
1914            return Err("empty LLM endpoint chain".into());
1915        }
1916        let primary = self.build_llm_for_endpoint(chain[0]);
1917        let fallbacks: Vec<LlmConfig> = chain[1..]
1918            .iter()
1919            .map(|e| self.build_llm_for_endpoint(e))
1920            .collect();
1921
1922        let (test, index) = self
1923            .current_step
1924            .borrow()
1925            .as_ref()
1926            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
1927        let primary_endpoint = chain[0].name.clone();
1928        let primary_model = primary.model.clone();
1929        self.emit_event(&TestEvent::LlmCallStarted {
1930            test: test.clone(),
1931            index,
1932            endpoint: primary_endpoint.clone(),
1933            model: primary_model.clone(),
1934            purpose: purpose.to_owned(),
1935        });
1936
1937        let started = Instant::now();
1938        let sys = system.to_owned();
1939        let user = user.to_owned();
1940        let image = image.map(<[String]>::to_vec);
1941
1942        let result = std::thread::spawn(move || {
1943            let rt = tokio::runtime::Builder::new_current_thread()
1944                .enable_all()
1945                .build()
1946                .unwrap();
1947            let call = async {
1948                match image.as_deref() {
1949                    Some(img) => {
1950                        llm_chat_vision_with_usage_chain(
1951                            &primary,
1952                            &fallbacks,
1953                            &sys,
1954                            &user,
1955                            Some(img),
1956                        )
1957                        .await
1958                    }
1959                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
1960                }
1961            };
1962            rt.block_on(call)
1963        })
1964        .join()
1965        .unwrap();
1966
1967        let duration_ms = started.elapsed().as_millis() as u64;
1968        match result {
1969            Ok((lr, idx)) => {
1970                let cost = calculate_llm_cost(
1971                    chain[idx],
1972                    lr.usage.prompt_tokens,
1973                    lr.usage.completion_tokens,
1974                );
1975                let answering = chain[idx].name.clone();
1976                let model = chain[idx]
1977                    .model
1978                    .clone()
1979                    .unwrap_or_else(|| primary_model.clone());
1980                self.emit_event(&TestEvent::LlmCallFinished {
1981                    test,
1982                    index,
1983                    endpoint: answering,
1984                    model,
1985                    purpose: purpose.to_owned(),
1986                    ok: true,
1987                    duration_ms,
1988                    input_tokens: lr.usage.prompt_tokens,
1989                    output_tokens: lr.usage.completion_tokens,
1990                    cost,
1991                    error: None,
1992                });
1993                Ok((lr, idx))
1994            }
1995            Err(e) => {
1996                self.emit_event(&TestEvent::LlmCallFinished {
1997                    test,
1998                    index,
1999                    endpoint: primary_endpoint,
2000                    model: primary_model,
2001                    purpose: purpose.to_owned(),
2002                    ok: false,
2003                    duration_ms,
2004                    input_tokens: 0,
2005                    output_tokens: 0,
2006                    cost: 0.0,
2007                    error: Some(e.clone()),
2008                });
2009                Err(e)
2010            }
2011        }
2012    }
2013
2014    /// Resolves a CSS selector for the target element. Uses the explicit
2015    /// `selector` if provided, otherwise asks the LLM to find the element
2016    /// from the natural language `target` description and page DOM.
2017    ///
2018    /// LLM responses are sanitized and verified against the live page: a
2019    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2020    /// immediately with the raw LLM output, and a selector that matches
2021    /// nothing triggers one retry with feedback before failing.
2022    #[allow(clippy::too_many_lines)]
2023    fn resolve_selector(
2024        &self,
2025        css_override: Option<&str>,
2026        target: &str,
2027        step_endpoint: Option<&str>,
2028        test_endpoint: Option<&str>,
2029        tab: &Tab,
2030    ) -> Result<String, String> {
2031        if let Some(explicit) = css_override {
2032            return Ok(explicit.to_owned());
2033        }
2034
2035        let dom_info = extract_dom_info(tab)?;
2036        let page_content = get_page_text(tab);
2037
2038        let system = concat!(
2039            "You are a browser automation selector generator. ",
2040            "Given a web page's content and interactive elements, ",
2041            "return ONLY the best CSS selector for the described element. ",
2042            "Output nothing except the CSS selector. ",
2043            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2044            "[name=\"...\"], tag.class, tag. ",
2045            "Never output explanations, markdown, or extra text."
2046        );
2047
2048        let user = format!(
2049            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2050            page_content.url,
2051            page_content.title,
2052            truncate(&page_content.body_text, 4000),
2053            dom_info,
2054            target,
2055        );
2056
2057        let retry_user = format!(
2058            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2059            "The selector must match at least one element currently present on the page.",
2060            page_content.url,
2061            page_content.title,
2062            truncate(&page_content.body_text, 4000),
2063            dom_info,
2064            target,
2065        );
2066
2067        self.reporter.debug(format!("LLM targeting: {target}"));
2068
2069        let chain = self
2070            .endpoints
2071            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2072        let sys = system.to_owned();
2073
2074        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2075
2076        let first = call_llm(&user);
2077        let (lr, idx) = match first {
2078            Ok(lr) => lr,
2079            Err(e) => {
2080                return Err(format!("LLM element targeting failed: {e}"));
2081            }
2082        };
2083        self.usage.record_llm_call(
2084            &chain[idx].name,
2085            chain[idx],
2086            lr.usage.prompt_tokens,
2087            lr.usage.completion_tokens,
2088        );
2089        let clean = sanitize_selector(&lr.content);
2090        self.reporter.debug(format!("resolved selector: {clean}"));
2091
2092        if selector_is_useless(&clean) {
2093            return Err(format!(
2094                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2095                raw = lr.content.trim(),
2096            ));
2097        }
2098        if let Err(reason) = validate_selector(&clean) {
2099            return Err(format!(
2100                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2101                raw = lr.content.trim(),
2102            ));
2103        }
2104        if !selector_matches(tab, &clean).unwrap_or(false) {
2105            // One retry with feedback: flaky models occasionally invent a
2106            // selector that does not exist on the page.
2107            self.reporter.warn(format!(
2108                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2109            ));
2110            let second = call_llm(&retry_user);
2111            let (lr2, idx2) = match second {
2112                Ok(lr2) => lr2,
2113                Err(e) => {
2114                    return Err(format!(
2115                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2116                    ));
2117                }
2118            };
2119            self.usage.record_llm_call(
2120                &chain[idx2].name,
2121                chain[idx2],
2122                lr2.usage.prompt_tokens,
2123                lr2.usage.completion_tokens,
2124            );
2125            let clean2 = sanitize_selector(&lr2.content);
2126            self.reporter
2127                .debug(format!("resolved selector (retry): {clean2}"));
2128            if selector_is_useless(&clean2) {
2129                return Err(format!(
2130                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2131                    raw = lr2.content.trim(),
2132                    excerpt = truncate(&page_content.body_text, 300),
2133                ));
2134            }
2135            if !selector_matches(tab, &clean2).unwrap_or(false) {
2136                return Err(format!(
2137                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2138                ));
2139            }
2140            return Ok(clean2);
2141        }
2142
2143        Ok(clean)
2144    }
2145}
2146
2147/// Evaluates a JS expression that is expected to return a boolean.
2148fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2149    tab.evaluate(js, false)
2150        .map_err(|e| format!("evaluate failed: {e}"))?
2151        .value
2152        .and_then(|v| v.as_bool())
2153        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2154}
2155
2156/// Checks whether a CSS selector matches at least one current element.
2157fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2158    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2159}
2160
2161// ── Free helper functions ──────────────────────────────────────────────
2162
2163fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2164    let name = format!("[navigate] {full_url}");
2165    match tab.navigate_to(full_url) {
2166        Ok(_) => {
2167            let _ = tab.wait_until_navigated();
2168            StepResult {
2169                name,
2170                status: StepStatus::Passed,
2171                message: format!("navigated to {full_url}"),
2172            }
2173        }
2174        Err(e) => StepResult {
2175            name,
2176            status: StepStatus::Failed,
2177            message: format!("navigation failed: {e}"),
2178        },
2179    }
2180}
2181
2182fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2183    let result = tab
2184        .evaluate(DOM_EXTRACT_JS, false)
2185        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2186
2187    let json_str = result
2188        .value
2189        .as_ref()
2190        .and_then(|v| v.as_str())
2191        .unwrap_or("[]");
2192
2193    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2194
2195    if elements.is_empty() {
2196        return Ok("(no interactive elements found)".to_owned());
2197    }
2198
2199    Ok(elements.join("\n"))
2200}
2201
2202fn get_page_text(tab: &Tab) -> PageContent {
2203    let url = tab.get_url();
2204
2205    let title = tab
2206        .evaluate("document.title", false)
2207        .ok()
2208        .and_then(|r| r.value)
2209        .and_then(|v| v.as_str().map(String::from))
2210        .unwrap_or_else(|| "unknown".to_owned());
2211
2212    let body_text = tab
2213        .evaluate(
2214            "document.body ? document.body.innerText : document.documentElement.innerText",
2215            false,
2216        )
2217        .ok()
2218        .and_then(|r| r.value)
2219        .and_then(|v| v.as_str().map(String::from))
2220        .unwrap_or_default();
2221
2222    PageContent {
2223        url,
2224        title,
2225        body_text: truncate(&body_text, 8000),
2226    }
2227}
2228
2229fn resolve_url(url: &str, base_url: &str) -> String {
2230    if url.starts_with("http://") || url.starts_with("https://") {
2231        return url.to_owned();
2232    }
2233    let base = base_url.trim_end_matches('/');
2234    if url.starts_with('/') {
2235        format!("{base}{url}")
2236    } else {
2237        format!("{base}/{url}")
2238    }
2239}
2240
2241/// Human-readable label for a step, used when steps are skipped after an
2242/// earlier failure.
2243fn step_label(step: &TestStep) -> String {
2244    match step {
2245        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2246        TestStep::Click { target, .. } => format!("[click] {target}"),
2247        TestStep::Type { target, .. } => format!("[type] {target}"),
2248        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2249        TestStep::Assert {
2250            definition,
2251            preset,
2252            prompt,
2253            ..
2254        } => definition.as_ref().map_or_else(
2255            || {
2256                preset.as_ref().map_or_else(
2257                    || {
2258                        prompt.as_ref().map_or_else(
2259                            || "[assert]".to_owned(),
2260                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2261                        )
2262                    },
2263                    |p| format!("[assert] {p}"),
2264                )
2265            },
2266            |d| format!("[assert] {d}"),
2267        ),
2268        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2269        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2270        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2271    }
2272}
2273
2274/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2275#[must_use]
2276const fn step_kind_label(step: &TestStep) -> &'static str {
2277    match step {
2278        TestStep::Navigate { .. } => "navigate",
2279        TestStep::Click { .. } => "click",
2280        TestStep::Type { .. } => "type",
2281        TestStep::Wait { .. } => "wait",
2282        TestStep::Assert { .. } => "assert",
2283        TestStep::Screenshot { .. } => "screenshot",
2284        TestStep::Agent { .. } => "agent",
2285        TestStep::Mcp { .. } => "mcp",
2286    }
2287}
2288
2289// ── Support types ──────────────────────────────────────────────────────
2290
2291#[derive(Default)]
2292struct TestRunResult {
2293    passed: u32,
2294    failed: u32,
2295    skipped: u32,
2296    total: u32,
2297    details: Vec<StepResult>,
2298}
2299
2300struct PageContent {
2301    url: String,
2302    title: String,
2303    body_text: String,
2304}