Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// Formats a UNIX epoch timestamp as an RFC 3339 UTC string
26/// (`2026-09-19T16:04:00Z`) without pulling in a date dependency.
27/// Civil-from-days conversion after Howard Hinnant.
28#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31    let secs = i64::try_from(secs).unwrap_or(0);
32    let days = secs.div_euclid(86_400);
33    let rem = secs.rem_euclid(86_400);
34    let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35    let z = days + 719_468;
36    let era = z.div_euclid(146_097);
37    let doe = z.rem_euclid(146_097);
38    let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40    let mp = (5 * doy + 2) / 153;
41    let d = doy - (153 * mp + 2) / 5 + 1;
42    let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43    let y = yoe + era * 400 + i64::from(mth <= 2);
44    format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47/// One detected layout defect (`layout_no_issues` preset).
48#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50    #[serde(rename = "type")]
51    issue_type: String,
52    element: String,
53    detail: String,
54}
55
56/// In-browser DOM layout scan for `layout_no_issues`.
57///
58/// Geometry-only checks (no LLM, no pixels):
59/// 1. page horizontal overflow (`scrollWidth` > viewport width);
60/// 2. elements outside the viewport that scrolling cannot reveal
61///    (fixed elements off-screen, left/negative overflow, right-edge
62///    overflow beyond the horizontally scrollable area, and bottom
63///    overflow on a page that cannot scroll down) — below-the-fold
64///    content on a tall scrollable page is normal flow, NOT a defect;
65/// 3. text clipped by `overflow: hidden` containers whose content
66///    is measurably larger than the box;
67/// 4. interactive elements (buttons/links/inputs) whose center point
68///    is covered by a different element that would intercept the click.
69///
70/// Elements whose class matches a configured ignore prefix (default:
71/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
72/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
73/// skipped — they are intentionally 1x1 / off-screen. The prefix list
74/// is injected at `__IGNORE_CLASSES__` from
75/// [`ScenarioConfig::layout_ignore_classes`].
76///
77/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
78/// sticky headers) is excluded by the position/relation filters.
79const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81  const issues = [];
82  const push = (type, el, detail) => {
83    if (issues.length >= 30) return;
84    let element = el.tagName.toLowerCase();
85    if (el.id) element += '#' + el.id;
86    else if (typeof el.className === 'string' && el.className.trim())
87      element += '.' + el.className.trim().split(/\s+/).join('.');
88    issues.push({ type, element, detail: String(detail).slice(0, 220) });
89  };
90  const vw = document.documentElement.clientWidth || window.innerWidth;
91  const vh = document.documentElement.clientHeight || window.innerHeight;
92  if (!vw || !vh) return JSON.stringify(issues);
93  const de = document.documentElement;
94  // 1. Page-level horizontal overflow.
95  if (maxSW > vw + 2)
96    push('page-overflow-x', de,
97      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98  // Class-name prefixes to skip (injected; default = Angular CDK
99  // screen-reader helpers, which are intentionally 1x1 / off-screen).
100  const ignorePrefixes = __IGNORE_CLASSES__;
101  const isIgnored = (el) => {
102    if (typeof el.className !== 'string' || !el.className.trim()) return false;
103    const classes = el.className.trim().split(/\s+/);
104    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105  };
106  const body = document.body;
107  // The document element is not always the scroll container: the app may
108  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110  // "not scrollable" — measure the scrollable content height/width of the
111  // document element AND the body, and take the max.
112  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114  const vScrollable = maxSH > vh + 2;
115  const hScrollable = maxSW > vw + 2;
116  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117  // this element into view on the given axis — i.e. the element is inside a
118  // scrolling region, so lying beyond the viewport cut is not a defect.
119  const reachableByScroller = (el, axis) => {
120    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123    let p = el.parentElement;
124    while (p && p !== body) {
125      const pcs = getComputedStyle(p);
126      const o = pcs[ovProp];
127      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128          p[dim] > p[dimClient] + 2) return true;
129      p = p.parentElement;
130    }
131    return false;
132  };
133  const selfOverflowing = (el, cs, axis) => {
134    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138      el[dim] > el[dimClient] + 2;
139  };
140  // True when an ancestor clips this axis with overflow:hidden/clip — the
141  // element's overhang is not visible, so treat it as reachable.
142  const clippedByAncestor = (el, axis) => {
143    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144    let p = el.parentElement;
145    while (p && p !== body) {
146      const o = getComputedStyle(p)[ovProp];
147      if (o === 'hidden' || o === 'clip') return true;
148      p = p.parentElement;
149    }
150    return false;
151  };
152  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154  const hasContent = (el) =>
155    ((el.textContent || '').trim().length > 0) ||
156    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157  // 2. Elements outside the viewport that scrolling cannot reveal.
158  for (const el of all) {
159    const cs = getComputedStyle(el);
160    if (!visible(cs)) continue;
161    if (isIgnored(el)) continue;
162    const r = el.getBoundingClientRect();
163    if (r.width < 2 || r.height < 2) continue;
164    if (!hasContent(el) && el.children.length === 0) continue;
165    if (cs.position === 'fixed') {
166      // Fixed elements never move with the scroll: any edge outside the
167      // viewport is unreachable content and therefore a defect.
168      const overTop = -r.top;
169      const overLeft = -r.left;
170      const overRight = r.right - vw;
171      const overBottom = r.bottom - vh;
172      // Top/left overshoot is never reachable; bottom/right overshoot is
173      // only a defect when the fixed element cannot scroll that content
174      // into view itself (e.g. an opened Material drawer whose inner
175      // container scrolls is not a "cut-off" defect).
176      if (overTop > 2 || overLeft > 2 ||
177          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179        let where = '';
180        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186        push('element-out-of-viewport', el,
187          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188      }
189      continue;
190    }
191    if (cs.position === 'sticky') continue;
192    // Negative left overflow cannot be reached by scrolling (scrollLeft
193    // never goes below 0).
194    if (r.left < -2) {
195      push('element-out-of-viewport', el,
196        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197      continue;
198    }
199    // Negative top with the page at the top means the element sits above
200    // the document origin — also unreachable.
201    if (r.top < -2 && de.scrollTop <= 2) {
202      push('element-out-of-viewport', el,
203        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204      continue;
205    }
206    const overRight = r.right - vw;
207    // Right-edge overflow is only a defect when the page cannot scroll
208    // horizontally to reveal it (or the element sticks out past the
209    // scrollable content width itself).
210    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211      push('element-out-of-viewport', el,
212        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213      continue;
214    }
215    // Below-the-fold content on a scrollable page is normal (tall landing
216    // pages); only flag bottom overflow the user can never scroll to.
217    const overBottom = r.bottom - vh;
218    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219      push('element-out-of-viewport', el,
220        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221    }
222  }
223  // 3. Text clipped by overflow:hidden containers.
224  for (const el of all) {
225    if (isIgnored(el)) continue;
226    const cs = getComputedStyle(el);
227    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229    if (!(el.textContent || '').trim()) continue;
230    // Single-line truncation with an ellipsis is an intentional design
231    // pattern (Tailwind .truncate etc.), not a clipping defect.
232    if (cs.textOverflow === 'ellipsis') continue;
233    push('text-clipped', el,
234      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236  }
237  // 4. Interactive elements covered by a different element.
238  const interactive =
239    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240  const targets = document.querySelectorAll(interactive);
241  for (const el of targets) {
242    if (isIgnored(el)) continue;
243    const r = el.getBoundingClientRect();
244    if (r.width < 6 || r.height < 6) continue;
245    const cs = getComputedStyle(el);
246    if (!visible(cs)) continue;
247    const cx = r.left + r.width / 2;
248    const cy = r.top + r.height / 2;
249    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250    const top = document.elementFromPoint(cx, cy);
251    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252    if (isIgnored(top)) continue;
253    const tcs = getComputedStyle(top);
254    if (!visible(tcs)) continue;
255    if (tcs.pointerEvents === 'none') continue;
256    const tr = top.getBoundingClientRect();
257    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260    push('element-overlap', el,
261      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262      ' is covered by <' + tname + '>');
263  }
264  return JSON.stringify(issues);
265})()
266"#;
267
268/// How long the CDP connection stays open after the browser goes quiet.
269///
270/// `headless_chrome` ships a 30s default and tears down the entire connection
271/// when no traffic arrives for that long; a run must own its connection for
272/// its full duration instead.
273const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275/// Executes a [`Scenario`] against a real browser with optional LLM
276/// assistance for element targeting and assertions.
277pub struct ScenarioRunner {
278    config: ScenarioConfig,
279    definitions: HashMap<String, AssertDefinition>,
280    llm: LlmConfig,
281    timeout: Duration,
282    viewport_width: u32,
283    viewport_height: u32,
284    /// The viewport currently applied in the browser (CDP emulation).
285    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
286    applied_viewport: std::cell::Cell<(u32, u32)>,
287    endpoints: EndpointRegistry,
288    usage: Arc<UsageTracker>,
289    budgets: BudgetTracker,
290    /// Directory for failure artifacts (screenshots).
291    artifacts_dir: PathBuf,
292    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
293    reporter: Arc<Reporter>,
294    /// The test + step index currently executing, for LLM-call events.
295    current_step: std::cell::RefCell<Option<(String, u32)>>,
296    /// Epoch seconds of runner creation — constant for the whole run so
297    /// the LLM context preamble below is cacheable by upstream providers.
298    run_started: u64,
299}
300
301/// Aggregated results from a scenario run.
302#[derive(Debug, Default)]
303pub struct RunReport {
304    /// Number of tests that passed.
305    pub tests_passed: u32,
306    /// Number of tests that failed.
307    pub tests_failed: u32,
308    /// Number of steps that passed.
309    pub passed: u32,
310    /// Number of steps that failed.
311    pub failed: u32,
312    /// Number of steps that were skipped.
313    pub skipped: u32,
314    /// Per-step details.
315    pub details: Vec<StepResult>,
316}
317
318/// Result of a single step execution.
319#[derive(Debug)]
320pub struct StepResult {
321    /// The step name.
322    pub name: String,
323    /// Whether the step passed, failed, or was skipped.
324    pub status: StepStatus,
325    /// Human-readable result message.
326    pub message: String,
327}
328
329/// Outcome for a single step.
330pub use crate::events::StepStatus;
331
332/// Predefined assertion preset definition.
333struct AssertPreset {
334    name: &'static str,
335    system: &'static str,
336    user_template: &'static str,
337}
338
339/// Built-in assertion presets.
340#[allow(clippy::literal_string_with_formatting_args)]
341const ASSERTION_PRESETS: &[AssertPreset] = &[
342    AssertPreset {
343        name: "no_error_on_page",
344        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
345        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
346    },
347    AssertPreset {
348        name: "text_visible",
349        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
350        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
351    },
352    AssertPreset {
353        name: "element_exists",
354        system: "You are a QA tester. Check if a described UI element exists on a web page.",
355        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
356    },
357    AssertPreset {
358        name: "visual_no_issues",
359        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
360        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
361    },
362    AssertPreset {
363        name: "visual_no_overlaps",
364        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
365        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
366    },
367    AssertPreset {
368        name: "visual_text_visible",
369        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
370        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
371    },
372    AssertPreset {
373        name: "layout_no_issues",
374        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
375        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
376    },
377];
378
379impl ScenarioRunner {
380    /// Creates a new runner with the given scenario configuration and
381    /// assertion definitions.
382    #[must_use]
383    #[allow(clippy::needless_pass_by_value)]
384    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
385        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
386    }
387
388    /// Creates a runner that reports run events through the given reporter.
389    #[must_use]
390    #[allow(clippy::needless_pass_by_value)]
391    pub fn with_reporter(
392        scenario_config: ScenarioConfig,
393        definitions: Vec<AssertDefinition>,
394        reporter: Arc<Reporter>,
395    ) -> Self {
396        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
397            &scenario_config,
398        ));
399        let llm = LlmConfig {
400            url: scenario_config
401                .llm_url
402                .clone()
403                .unwrap_or_else(crate::llm_base_url),
404            model: scenario_config
405                .llm_model
406                .clone()
407                .unwrap_or_else(crate::llm_model),
408            api_key: scenario_config
409                .llm_api_key
410                .clone()
411                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
412            headers: if scenario_config.llm_headers.is_empty() {
413                crate::parse_headers_env()
414            } else {
415                scenario_config.llm_headers.clone()
416            },
417            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
418            temperature: scenario_config.temperature,
419            thinking: scenario_config.thinking,
420            model_params: scenario_config.model_params.clone(),
421            max_attempts: crate::default_llm_attempts(),
422            provider: crate::scenario::Provider::Openai,
423            deployment: None,
424            api_version: None,
425            auth: crate::scenario::AuthConfig::default(),
426            header_commands: std::collections::HashMap::new(),
427            aws: crate::scenario::AwsConfig::default(),
428        };
429        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
430        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
431        let defs_map: HashMap<String, AssertDefinition> = definitions
432            .into_iter()
433            .map(|d| (d.name.clone(), d))
434            .collect();
435
436        Self {
437            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
438            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
439            viewport_height: scenario_config.viewport_height.unwrap_or(720),
440            applied_viewport: std::cell::Cell::new((0, 0)),
441            config: scenario_config.clone(),
442            definitions: defs_map,
443            llm,
444            endpoints,
445            usage: Arc::new(UsageTracker::new()),
446            budgets,
447            artifacts_dir: PathBuf::from(
448                scenario_config
449                    .artifacts_dir
450                    .unwrap_or_else(|| "artifacts".to_owned()),
451            ),
452            reporter,
453            current_step: std::cell::RefCell::new(None),
454            run_started: SystemTime::now()
455                .duration_since(UNIX_EPOCH)
456                .map_or(0, |d| d.as_secs()),
457        }
458    }
459
460    /// Emits an event; a sink failure degrades to a console warning so a
461    /// broken log file can never mask the run itself.
462    fn emit_event(&self, event: &TestEvent) {
463        if let Err(err) = self.reporter.emit(event) {
464            use std::io::Write as _;
465            let mut out = std::io::stderr().lock();
466            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
467        }
468    }
469
470    /// Returns a clone of the [`UsageTracker`] for reporting.
471    #[must_use]
472    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
473        Arc::clone(&self.usage)
474    }
475
476    /// Returns a reference to the [`BudgetTracker`].
477    #[must_use]
478    pub const fn budget_tracker(&self) -> &BudgetTracker {
479        &self.budgets
480    }
481
482    /// Executes all test groups in the scenario and returns a report.
483    ///
484    /// # Errors
485    ///
486    /// Returns an error if the browser fails to launch.
487    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
488    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
489        let mut report = RunReport::default();
490
491        if tests.is_empty() {
492            self.reporter.warn("No tests defined in scenario.");
493            self.emit_event(&TestEvent::RunFinished {
494                tests_passed: 0,
495                tests_failed: 0,
496                steps_passed: 0,
497                steps_failed: 0,
498                steps_skipped: 0,
499                total_cost: 0.0,
500                total_tokens: 0,
501                total_calls: 0,
502            });
503            return Ok(report);
504        }
505
506        self.emit_event(&TestEvent::RunStarted {
507            total_tests: tests.len() as u32,
508        });
509
510        let browser_headless = self.config.browser_headless.unwrap_or(true);
511
512        let launch_opts = LaunchOptions {
513            headless: browser_headless,
514            window_size: Some((self.viewport_width, self.viewport_height)),
515            sandbox: false,
516            // headless_chrome defaults this to 30s and shuts down the whole CDP
517            // connection when no messages arrive for that long. A scenario can
518            // easily exceed 30s of browser silence (slow LLM targeting/assertion
519            // calls, page waits, budget checks between steps), after which every
520            // remaining step fails with "Unable to make method calls because
521            // underlying connection is closed" — one quiet gap kills the run.
522            // Open-ended scenarios must own the connection for their full
523            // duration, so keep it alive for 6 hours.
524            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
525            ..LaunchOptions::default()
526        };
527
528        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
529        let tab = browser.new_tab().context("failed to open browser tab")?;
530        let _ = tab.set_default_timeout(self.timeout);
531
532        // Start MCP server if configured
533        #[cfg(feature = "mcp-server")]
534        if let Some(ref mcp_cfg) = self.config.mcp_server {
535            if mcp_cfg.enabled {
536                let port = mcp_cfg.port;
537                std::thread::spawn(move || {
538                    let _ = crate::mcp_server::start_mcp_server(port);
539                });
540            }
541        }
542        #[cfg(not(feature = "mcp-server"))]
543        if let Some(mcp_cfg) = &self.config.mcp_server {
544            if mcp_cfg.enabled {
545                self.reporter
546                    .warn("MCP server configured but 'mcp-server' feature not enabled");
547            }
548        }
549
550        // Start A2A agent server if configured
551        #[cfg(feature = "a2a-server")]
552        if let Some(ref a2a_cfg) = self.config.a2a_server {
553            if a2a_cfg.enabled {
554                let port = a2a_cfg.port;
555                tokio::spawn(crate::a2a_server::start_a2a_server(port));
556            }
557        }
558        #[cfg(not(feature = "a2a-server"))]
559        if let Some(a2a_cfg) = &self.config.a2a_server {
560            if a2a_cfg.enabled {
561                self.reporter
562                    .warn("A2A server configured but 'a2a-server' feature not enabled");
563            }
564        }
565
566        for test in tests {
567            self.emit_event(&TestEvent::TestStarted {
568                test: test.name.clone(),
569            });
570
571            self.usage.reset_per_test();
572
573            let test_started = Instant::now();
574            let test_result = self.run_test(test, &tab);
575            let duration_ms = test_started.elapsed().as_millis() as u64;
576            let usage = self.usage.current_test_snapshot();
577            self.usage.commit_test(&test.name);
578
579            self.emit_event(&TestEvent::TestFinished {
580                test: test.name.clone(),
581                passed: test_result.passed,
582                failed: test_result.failed,
583                skipped: test_result.skipped,
584                duration_ms,
585                cost: usage.total_cost,
586                tokens: usage.total_tokens,
587                calls: usage.total_calls,
588            });
589
590            if test_result.failed == 0 && test_result.total > 0 {
591                report.tests_passed += 1;
592            } else if test_result.total > 0 {
593                report.tests_failed += 1;
594            }
595
596            report.passed += test_result.passed;
597            report.failed += test_result.failed;
598            report.skipped += test_result.skipped;
599            report.details.extend(test_result.details);
600        }
601
602        let global = self.usage.global_snapshot();
603        self.emit_event(&TestEvent::RunFinished {
604            tests_passed: report.tests_passed,
605            tests_failed: report.tests_failed,
606            steps_passed: report.passed,
607            steps_failed: report.failed,
608            steps_skipped: report.skipped,
609            total_cost: global.total_cost,
610            total_tokens: global.total_tokens,
611            total_calls: global.total_calls,
612        });
613
614        Ok(report)
615    }
616
617    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
618    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
619        let base_url = test
620            .base_url
621            .clone()
622            .or_else(|| self.config.base_url.clone())
623            .unwrap_or_else(crate::base_url);
624
625        // Per-test viewport override: switch the browser via CDP
626        // device-metrics emulation before this test runs.
627        let vw = test.viewport_width.unwrap_or(self.viewport_width);
628        let vh = test.viewport_height.unwrap_or(self.viewport_height);
629        if self.applied_viewport.get() != (vw, vh) {
630            self.apply_viewport(tab, vw, vh);
631            self.applied_viewport.set((vw, vh));
632        }
633
634        // Per-test isolation: every test starts from its own start_url
635        // (unless auto_navigate is disabled), so a test never inherits the
636        // previous test's page state.
637        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
638
639        let start_url = test
640            .start_url
641            .clone()
642            .or_else(|| self.config.start_url.clone())
643            .unwrap_or_else(|| "/dashboard".to_owned());
644
645        if auto_navigate {
646            let full_url = resolve_url(&start_url, &base_url);
647            self.reporter.debug(format!("auto-navigate: {full_url}"));
648            let _ = tab.navigate_to(&full_url);
649            let _ = tab.wait_until_navigated();
650            std::thread::sleep(Duration::from_secs(4));
651        }
652
653        let mut result = TestRunResult::default();
654
655        for (step_index, step) in test.steps.iter().enumerate() {
656            result.total += 1;
657
658            let wait_ms = match step {
659                TestStep::Navigate { wait_after_ms, .. }
660                | TestStep::Click { wait_after_ms, .. }
661                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
662                _ => None,
663            };
664
665            self.current_step
666                .replace(Some((test.name.clone(), step_index as u32)));
667            self.emit_event(&TestEvent::StepStarted {
668                test: test.name.clone(),
669                index: step_index as u32,
670                label: step_label(step),
671            });
672            let step_started = Instant::now();
673
674            let mut step_result = match step {
675                TestStep::Navigate { url, .. } => {
676                    let full_url = resolve_url(url, &base_url);
677                    run_navigate_step(&full_url, tab)
678                }
679                TestStep::Click {
680                    target,
681                    selector,
682                    endpoint,
683                    idempotent,
684                    ..
685                } => self.run_click(
686                    target,
687                    selector.as_deref(),
688                    endpoint.as_deref(),
689                    test.endpoint.as_deref(),
690                    *idempotent,
691                    tab,
692                ),
693                TestStep::Type {
694                    target,
695                    text,
696                    selector,
697                    endpoint,
698                    idempotent,
699                    ..
700                } => self.run_type(
701                    target,
702                    text,
703                    selector.as_deref(),
704                    endpoint.as_deref(),
705                    test.endpoint.as_deref(),
706                    *idempotent,
707                    tab,
708                ),
709                TestStep::Wait {
710                    target,
711                    selector,
712                    text,
713                    timeout_ms,
714                    endpoint,
715                    idempotent,
716                } => self.run_wait(
717                    target,
718                    selector.as_deref(),
719                    text.as_deref(),
720                    *timeout_ms,
721                    endpoint.as_deref(),
722                    test.endpoint.as_deref(),
723                    *idempotent,
724                    tab,
725                ),
726                TestStep::Assert {
727                    definition,
728                    preset,
729                    prompt,
730                    assert_text,
731                    endpoint,
732                    screenshot,
733                } => self.run_assert(
734                    definition.as_deref(),
735                    preset.as_deref(),
736                    prompt.as_deref(),
737                    assert_text.as_deref(),
738                    *screenshot,
739                    endpoint.as_deref(),
740                    test.endpoint.as_deref(),
741                    tab,
742                ),
743                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
744                TestStep::Agent {
745                    agent,
746                    task,
747                    definition,
748                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
749                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
750            };
751
752            // Failure diagnostics: capture the page state and a screenshot so
753            // CI logs say WHAT the page looked like when the step failed,
754            // instead of a bare "timed out: The event waited for never came".
755            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
756                let state = diagnostics::capture(tab);
757                let screenshot = diagnostics::save_screenshot(
758                    tab,
759                    &self.artifacts_dir,
760                    &test.name,
761                    &test.name,
762                    step_index,
763                    step_kind_label(step),
764                );
765                step_result.message = format!(
766                    "{base} — {excerpt}",
767                    base = step_result.message,
768                    excerpt = diagnostics::inline_excerpt(&state),
769                );
770                (Some(diagnostics::full_context(&state)), screenshot)
771            } else {
772                (None, None)
773            };
774
775            let duration_ms = step_started.elapsed().as_millis() as u64;
776            self.emit_event(&TestEvent::StepFinished {
777                test: test.name.clone(),
778                index: step_index as u32,
779                label: step_result.name.clone(),
780                status: step_result.status,
781                duration_ms,
782                message: step_result.message.clone(),
783                diagnostics: diagnostics_block,
784                screenshot: screenshot_path,
785            });
786            self.current_step.replace(None);
787
788            match step_result.status {
789                StepStatus::Passed => result.passed += 1,
790                StepStatus::Failed => result.failed += 1,
791                StepStatus::Skipped => result.skipped += 1,
792            }
793
794            // Fail fast: the first failed step ends the test and the
795            // remaining steps are reported as skipped (no LLM budget is
796            // burned asserting against a page that is already known broken).
797            if step_result.status == StepStatus::Failed
798                && !self.config.continue_on_failure
799                && step_index + 1 < test.steps.len()
800            {
801                self.emit_event(&TestEvent::Warning {
802                    message: format!(
803                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
804                        test.steps.len() - step_index - 1
805                    ),
806                });
807                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
808                    let skipped_index = step_index + 1 + offset;
809                    let label = step_label(skipped);
810                    result.total += 1;
811                    result.skipped += 1;
812                    self.emit_event(&TestEvent::StepStarted {
813                        test: test.name.clone(),
814                        index: skipped_index as u32,
815                        label: label.clone(),
816                    });
817                    self.emit_event(&TestEvent::StepFinished {
818                        test: test.name.clone(),
819                        index: skipped_index as u32,
820                        label,
821                        status: StepStatus::Skipped,
822                        duration_ms: 0,
823                        message: "skipped: previous step failed".into(),
824                        diagnostics: None,
825                        screenshot: None,
826                    });
827                    result.details.push(StepResult {
828                        name: step_label(skipped),
829                        status: StepStatus::Skipped,
830                        message: "skipped: previous step failed".into(),
831                    });
832                }
833                result.details.push(step_result);
834                return result;
835            }
836
837            // Check per-test budget after each step
838            let test_usage = self.usage.current_test_snapshot();
839            let global_usage = self.usage.global_snapshot();
840            let budget_status = self.budgets.check_all(
841                &test.name,
842                &test_usage,
843                &global_usage,
844                test.budget.as_ref(),
845            );
846            match budget_status {
847                BudgetStatus::HardExceeded { message, .. } => {
848                    self.emit_event(&TestEvent::Warning {
849                        message: format!("budget exceeded: {message}"),
850                    });
851                    result.details.push(StepResult {
852                        name: "[budget]".into(),
853                        status: StepStatus::Failed,
854                        message,
855                    });
856                    result.failed += 1;
857                    return result;
858                }
859                BudgetStatus::SoftExceeded { message, .. } => {
860                    self.emit_event(&TestEvent::Warning {
861                        message: format!("budget warning: {message}"),
862                    });
863                }
864                BudgetStatus::Ok => {}
865            }
866
867            if let Some(ms) = wait_ms {
868                std::thread::sleep(Duration::from_millis(ms));
869            }
870
871            result.details.push(step_result);
872        }
873
874        result
875    }
876
877    /// Applies a viewport size to the current tab via CDP
878    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
879    /// overrides and the viewport matrix. The initial window size set at
880    /// browser launch is replaced by emulation; failures are logged but
881    /// do not fail the test (a mismatched viewport only weakens coverage).
882    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
883        use headless_chrome::protocol::cdp::Emulation;
884        let _ = self;
885        let params = Emulation::SetDeviceMetricsOverride {
886            width,
887            height,
888            device_scale_factor: 1.0,
889            mobile: false,
890            scale: None,
891            screen_width: Some(width),
892            screen_height: Some(height),
893            position_x: None,
894            position_y: None,
895            dont_set_visible_size: None,
896            screen_orientation: None,
897            viewport: None,
898            display_feature: None,
899            device_posture: None,
900        };
901        self.reporter.debug(format!("viewport: {width}x{height}"));
902        if let Err(e) = tab.call_method(params) {
903            self.reporter
904                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
905        }
906    }
907
908    /// Height (px) covered by assert-step screenshots: the configured
909    /// `screenshot_max_height` (absolute px or viewport multiple, default
910    /// `"20x"`) resolved against the currently applied viewport, raised to
911    /// at least the viewport height so the visible screen is always fully
912    /// included. The capture is split into viewport-tall tiles, so this
913    /// value bounds total coverage (and hence the number of image parts).
914    #[must_use]
915    fn screenshot_height_cap(&self) -> u32 {
916        let viewport_height = self.current_viewport_height();
917        let cap = self
918            .config
919            .screenshot_max_height
920            .as_ref()
921            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
922        cap.max(viewport_height)
923    }
924
925    /// Height of the viewport currently emulated in the browser (falling
926    /// back to the configured default before any emulation was applied).
927    #[must_use]
928    const fn current_viewport_height(&self) -> u32 {
929        let (_, height) = self.applied_viewport.get();
930        if height > 0 {
931            height
932        } else {
933            self.viewport_height
934        }
935    }
936
937    // ── step handlers ───────────────────────────────────────────────────
938
939    #[allow(clippy::too_many_lines)]
940    fn run_click(
941        &self,
942        target: &str,
943        selector_override: Option<&str>,
944        step_endpoint: Option<&str>,
945        test_endpoint: Option<&str>,
946        idempotent: bool,
947        tab: &Tab,
948    ) -> StepResult {
949        let name = format!("[click] {target}");
950        let selector = match self.resolve_selector(
951            selector_override,
952            target,
953            step_endpoint,
954            test_endpoint,
955            tab,
956        ) {
957            Ok(s) => s,
958            Err(msg) => {
959                if idempotent {
960                    return StepResult {
961                        name,
962                        status: StepStatus::Skipped,
963                        message: format!("skipped (idempotent): no target found — {msg}"),
964                    };
965                }
966                return StepResult {
967                    name,
968                    status: StepStatus::Failed,
969                    message: msg,
970                };
971            }
972        };
973
974        // Idempotent steps probe briefly: a missing target means the
975        // action was already done / not applicable (e.g. an
976        // already-authenticated session), and skipping is the success
977        // path, not a failure.
978        let probe_secs = if idempotent { 5 } else { 10 };
979        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
980            Ok(element) => match element.click() {
981                Ok(_) => StepResult {
982                    name,
983                    status: StepStatus::Passed,
984                    message: format!("clicked {selector}"),
985                },
986                Err(e) => StepResult {
987                    name,
988                    status: StepStatus::Failed,
989                    message: format!("click failed on {selector}: {e}"),
990                },
991            },
992            Err(e) if idempotent => StepResult {
993                name,
994                status: StepStatus::Skipped,
995                message: format!("skipped (idempotent): element {selector} not present — {e}"),
996            },
997            Err(e) => StepResult {
998                name,
999                status: StepStatus::Failed,
1000                message: format!("element {selector} not found: {e}"),
1001            },
1002        }
1003    }
1004
1005    #[allow(clippy::too_many_arguments)]
1006    fn run_type(
1007        &self,
1008        target: &str,
1009        text: &str,
1010        selector_override: Option<&str>,
1011        step_endpoint: Option<&str>,
1012        test_endpoint: Option<&str>,
1013        idempotent: bool,
1014        tab: &Tab,
1015    ) -> StepResult {
1016        let name = format!("[type] {target}");
1017        let selector = match self.resolve_selector(
1018            selector_override,
1019            target,
1020            step_endpoint,
1021            test_endpoint,
1022            tab,
1023        ) {
1024            Ok(s) => s,
1025            Err(msg) => {
1026                if idempotent {
1027                    return StepResult {
1028                        name,
1029                        status: StepStatus::Skipped,
1030                        message: format!("skipped (idempotent): no target found — {msg}"),
1031                    };
1032                }
1033                return StepResult {
1034                    name,
1035                    status: StepStatus::Failed,
1036                    message: msg,
1037                };
1038            }
1039        };
1040
1041        let probe_secs = if idempotent { 5 } else { 10 };
1042        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1043            Ok(element) => {
1044                if let Err(e) = element.click() {
1045                    return StepResult {
1046                        name,
1047                        status: StepStatus::Failed,
1048                        message: format!("click to focus {selector} failed: {e}"),
1049                    };
1050                }
1051
1052                let js = format!(
1053                    "document.querySelector('{}').value = '';",
1054                    selector.replace('\'', "\\'")
1055                );
1056                let _ = tab.evaluate(&js, false);
1057
1058                match element.type_into(text) {
1059                    Ok(_) => StepResult {
1060                        name,
1061                        status: StepStatus::Passed,
1062                        message: format!("typed {text:?} into {selector}"),
1063                    },
1064                    Err(e) => StepResult {
1065                        name,
1066                        status: StepStatus::Failed,
1067                        message: format!("type into {selector} failed: {e}"),
1068                    },
1069                }
1070            }
1071            Err(e) if idempotent => StepResult {
1072                name,
1073                status: StepStatus::Skipped,
1074                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1075            },
1076            Err(e) => StepResult {
1077                name,
1078                status: StepStatus::Failed,
1079                message: format!("element {selector} not found: {e}"),
1080            },
1081        }
1082    }
1083
1084    #[allow(clippy::too_many_arguments)]
1085    #[allow(clippy::too_many_lines)]
1086    fn run_wait(
1087        &self,
1088        target: &str,
1089        selector_override: Option<&str>,
1090        text: Option<&str>,
1091        timeout_ms: Option<u64>,
1092        step_endpoint: Option<&str>,
1093        test_endpoint: Option<&str>,
1094        idempotent: bool,
1095        tab: &Tab,
1096    ) -> StepResult {
1097        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1098        let step_name = format!("[wait] {target}");
1099
1100        // Resolve an explicit selector only (text-only waits are LLM-free).
1101        let selector = match selector_override {
1102            Some(s) => Some(s.to_owned()),
1103            None if text.is_some() => None,
1104            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1105                Ok(s) => Some(s),
1106                Err(msg) => {
1107                    if idempotent {
1108                        return StepResult {
1109                            name: step_name,
1110                            status: StepStatus::Skipped,
1111                            message: format!("skipped (idempotent): no target found — {msg}"),
1112                        };
1113                    }
1114                    return StepResult {
1115                        name: step_name,
1116                        status: StepStatus::Failed,
1117                        message: msg,
1118                    };
1119                }
1120            },
1121        };
1122
1123        if text.is_some() {
1124            let sel_js = selector
1125                .as_deref()
1126                .map(crate::selectors::selector_matches_js);
1127            let text_js = text.map(|t| {
1128                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1129                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1130            });
1131
1132            let deadline = Instant::now() + timeout;
1133            loop {
1134                let sel_ok = sel_js
1135                    .as_ref()
1136                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1137                let text_ok = text_js
1138                    .as_ref()
1139                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1140                if sel_ok && text_ok {
1141                    let mut what = Vec::new();
1142                    if let Some(sel) = &selector {
1143                        what.push(format!("found {sel}"));
1144                    }
1145                    if let Some(t) = text {
1146                        what.push(format!("text {t:?} visible"));
1147                    }
1148                    return StepResult {
1149                        name: step_name,
1150                        status: StepStatus::Passed,
1151                        message: what.join(" and "),
1152                    };
1153                }
1154                if Instant::now() >= deadline {
1155                    let mut what = Vec::new();
1156                    if let Some(sel) = &selector {
1157                        what.push(sel.clone());
1158                    }
1159                    if let Some(t) = text {
1160                        what.push(format!("text {t:?}"));
1161                    }
1162                    let message = format!(
1163                        "wait for {} timed out after {}ms: the event waited for never came",
1164                        what.join(" / "),
1165                        timeout.as_millis(),
1166                    );
1167                    if idempotent {
1168                        return StepResult {
1169                            name: step_name,
1170                            status: StepStatus::Skipped,
1171                            message: format!("skipped (idempotent): {message}"),
1172                        };
1173                    }
1174                    return StepResult {
1175                        name: step_name,
1176                        status: StepStatus::Failed,
1177                        message,
1178                    };
1179                }
1180                std::thread::sleep(Duration::from_millis(250));
1181            }
1182        }
1183
1184        match selector.as_deref() {
1185            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1186                Ok(_) => StepResult {
1187                    name: step_name,
1188                    status: StepStatus::Passed,
1189                    message: format!("found {sel}"),
1190                },
1191                Err(e) if idempotent => StepResult {
1192                    name: step_name,
1193                    status: StepStatus::Skipped,
1194                    message: format!(
1195                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1196                        timeout.as_millis()
1197                    ),
1198                },
1199                Err(e) => StepResult {
1200                    name: step_name,
1201                    status: StepStatus::Failed,
1202                    message: format!(
1203                        "wait for {sel} timed out after {}ms: {e}",
1204                        timeout.as_millis()
1205                    ),
1206                },
1207            },
1208            None => StepResult {
1209                name: step_name,
1210                status: StepStatus::Failed,
1211                message: "wait step has neither selector nor text".into(),
1212            },
1213        }
1214    }
1215
1216    #[allow(clippy::too_many_arguments)]
1217    fn run_assert(
1218        &self,
1219        definition: Option<&str>,
1220        preset: Option<&str>,
1221        prompt: Option<&str>,
1222        assert_text: Option<&str>,
1223        screenshot: bool,
1224        step_endpoint: Option<&str>,
1225        test_endpoint: Option<&str>,
1226        tab: &Tab,
1227    ) -> StepResult {
1228        std::thread::sleep(Duration::from_millis(500));
1229
1230        let page_content = get_page_text(tab);
1231
1232        // Vision attach: capture the full page once per assert step and
1233        // split it into viewport-tall tiles (the total coverage is bounded
1234        // by the configured height cap so vision tokens stay sane). All
1235        // tile data URLs are handed to the preset/prompt evaluation below.
1236        let image: Option<Vec<String>> = if screenshot {
1237            let endpoint = self
1238                .endpoints
1239                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1240            if !endpoint.vision {
1241                return StepResult {
1242                    name: "[assert]".into(),
1243                    status: StepStatus::Failed,
1244                    message: format!(
1245                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1246                        name = endpoint.name
1247                    ),
1248                };
1249            }
1250            match crate::vision::capture_screenshot_data_urls(
1251                tab,
1252                self.config
1253                    .screenshot_max_dimension
1254                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1255                self.screenshot_height_cap(),
1256                self.current_viewport_height(),
1257            ) {
1258                Ok(urls) => Some(urls),
1259                Err(e) => {
1260                    return StepResult {
1261                        name: "[assert]".into(),
1262                        status: StepStatus::Failed,
1263                        message: format!("screenshot capture failed: {e}"),
1264                    };
1265                }
1266            }
1267        } else {
1268            None
1269        };
1270
1271        if let Some(def_name) = definition {
1272            if let Some(def) = self.definitions.get(def_name) {
1273                return self.run_assert_def(
1274                    def,
1275                    &page_content,
1276                    image.as_deref(),
1277                    step_endpoint,
1278                    test_endpoint,
1279                    tab,
1280                );
1281            }
1282            return StepResult {
1283                name: format!("[assert] {def_name}"),
1284                status: StepStatus::Failed,
1285                message: format!("definition '{def_name}' not found"),
1286            };
1287        }
1288
1289        if let Some(preset_name) = preset {
1290            // Deterministic DOM layout scan — runs JS in the browser and
1291            // never calls the LLM (free, fast, no pixel budget).
1292            if preset_name == "layout_no_issues" {
1293                return self.run_layout_preset(tab);
1294            }
1295            return self.run_preset(
1296                preset_name,
1297                assert_text,
1298                &page_content,
1299                image.as_deref(),
1300                step_endpoint,
1301                test_endpoint,
1302            );
1303        }
1304
1305        if let Some(prompt_text) = prompt {
1306            return self.run_custom(
1307                prompt_text,
1308                &page_content,
1309                image.as_deref(),
1310                step_endpoint,
1311                test_endpoint,
1312            );
1313        }
1314
1315        StepResult {
1316            name: "[assert]".into(),
1317            status: StepStatus::Skipped,
1318            message: "no definition, preset, or prompt specified".into(),
1319        }
1320    }
1321
1322    fn run_assert_def(
1323        &self,
1324        def: &AssertDefinition,
1325        page_content: &PageContent,
1326        image: Option<&[String]>,
1327        step_endpoint: Option<&str>,
1328        test_endpoint: Option<&str>,
1329        tab: &Tab,
1330    ) -> StepResult {
1331        // Agent-based definition: delegate to an A2A agent
1332        if let Some(ref agent) = def.agent {
1333            if image.is_some() {
1334                return StepResult {
1335                    name: format!("[assert] {}", def.name),
1336                    status: StepStatus::Failed,
1337                    message: "agent-backed assertions do not support screenshots".into(),
1338                };
1339            }
1340            let task = def
1341                .task_template
1342                .as_deref()
1343                .unwrap_or("Evaluate the assertion")
1344                .replace("{url}", &page_content.url)
1345                .replace("{title}", &page_content.title)
1346                .replace("{content}", &page_content.body_text)
1347                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1348
1349            return self.run_agent_step(agent, &task, &def.name);
1350        }
1351
1352        // Custom preset: system + user_template provided in the definition
1353        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1354            return self.run_custom_preset(
1355                &def.name,
1356                system,
1357                template,
1358                def.assert_text.as_deref(),
1359                page_content,
1360                image,
1361                step_endpoint,
1362                test_endpoint,
1363            );
1364        }
1365
1366        def.preset.as_ref().map_or_else(
1367            || {
1368                def.prompt.as_ref().map_or_else(
1369                    || StepResult {
1370                        name: format!("[assert] {}", def.name),
1371                        status: StepStatus::Failed,
1372                        message: "definition has no preset, prompt, or system+user_template".into(),
1373                    },
1374                    |prompt| {
1375                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1376                    },
1377                )
1378            },
1379            |preset_name| {
1380                if preset_name == "layout_no_issues" {
1381                    return self.run_layout_preset(tab);
1382                }
1383                self.run_preset(
1384                    preset_name,
1385                    def.assert_text.as_deref(),
1386                    page_content,
1387                    image,
1388                    step_endpoint,
1389                    test_endpoint,
1390                )
1391            },
1392        )
1393    }
1394
1395    #[allow(clippy::too_many_arguments)]
1396    fn run_custom_preset(
1397        &self,
1398        name: &str,
1399        system: &str,
1400        template: &str,
1401        assert_text: Option<&str>,
1402        page_content: &PageContent,
1403        image: Option<&[String]>,
1404        step_endpoint: Option<&str>,
1405        test_endpoint: Option<&str>,
1406    ) -> StepResult {
1407        let user_prompt = template
1408            .replace("{url}", &page_content.url)
1409            .replace("{title}", &page_content.title)
1410            .replace("{content}", &page_content.body_text)
1411            .replace("{expected_text}", assert_text.unwrap_or(""))
1412            .replace("{description}", "");
1413
1414        // Custom preset definitions frequently forget the {content}
1415        // placeholder — without it the LLM has no page to evaluate and
1416        // answers "I can't determine that without seeing the page". Always
1417        // append the page context unless the template already references it.
1418        let user_prompt = if template.contains("{content}") {
1419            user_prompt
1420        } else {
1421            format!(
1422                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1423                url = page_content.url,
1424                title = page_content.title,
1425                content = page_content.body_text,
1426            )
1427        };
1428
1429        self.reporter
1430            .debug(format!("assert: {name} (custom preset)"));
1431
1432        let chain = self
1433            .endpoints
1434            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1435        let sys = system.to_owned();
1436
1437        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1438
1439        response.map_or_else(
1440            |e| StepResult {
1441                name: format!("[assert] {name}"),
1442                status: StepStatus::Failed,
1443                message: format!("LLM assertion call failed: {e}"),
1444            },
1445            |(lr, idx)| {
1446                self.usage.record_llm_call(
1447                    &chain[idx].name,
1448                    chain[idx],
1449                    lr.usage.prompt_tokens,
1450                    lr.usage.completion_tokens,
1451                );
1452                let content_lower = lr.content.to_lowercase().trim().to_owned();
1453                if content_lower.starts_with("pass") {
1454                    StepResult {
1455                        name: format!("[assert] {name}"),
1456                        status: StepStatus::Passed,
1457                        message: "PASS".into(),
1458                    }
1459                } else {
1460                    StepResult {
1461                        name: format!("[assert] {name}"),
1462                        status: StepStatus::Failed,
1463                        message: lr.content,
1464                    }
1465                }
1466            },
1467        )
1468    }
1469
1470    fn run_preset(
1471        &self,
1472        preset_name: &str,
1473        assert_text: Option<&str>,
1474        page_content: &PageContent,
1475        image: Option<&[String]>,
1476        step_endpoint: Option<&str>,
1477        test_endpoint: Option<&str>,
1478    ) -> StepResult {
1479        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1480            return StepResult {
1481                name: format!("[assert] {preset_name}"),
1482                status: StepStatus::Failed,
1483                message: format!("unknown assertion preset: {preset_name}"),
1484            };
1485        };
1486        if preset_name.starts_with("visual_") && image.is_none() {
1487            return StepResult {
1488                name: format!("[assert] {preset_name}"),
1489                status: StepStatus::Failed,
1490                message: format!(
1491                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1492                ),
1493            };
1494        }
1495
1496        let user_prompt = preset
1497            .user_template
1498            .replace("{url}", &page_content.url)
1499            .replace("{title}", &page_content.title)
1500            .replace("{content}", &page_content.body_text)
1501            .replace("{expected_text}", assert_text.unwrap_or(""))
1502            .replace("{description}", "");
1503
1504        // Same safety net as custom presets: never let the LLM answer with
1505        // no page context at all.
1506        let user_prompt = if preset.user_template.contains("{content}") {
1507            user_prompt
1508        } else {
1509            format!(
1510                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1511                url = page_content.url,
1512                title = page_content.title,
1513                content = page_content.body_text,
1514            )
1515        };
1516
1517        self.reporter.debug(format!("assert: {preset_name}"));
1518
1519        let chain = self
1520            .endpoints
1521            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1522        let sys = preset.system.to_owned();
1523
1524        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1525
1526        response.map_or_else(
1527            |e| StepResult {
1528                name: format!("[assert] {preset_name}"),
1529                status: StepStatus::Failed,
1530                message: format!("LLM assertion call failed: {e}"),
1531            },
1532            |(lr, idx)| {
1533                self.usage.record_llm_call(
1534                    &chain[idx].name,
1535                    chain[idx],
1536                    lr.usage.prompt_tokens,
1537                    lr.usage.completion_tokens,
1538                );
1539                let content_lower = lr.content.to_lowercase().trim().to_owned();
1540                if content_lower.starts_with("pass") {
1541                    StepResult {
1542                        name: format!("[assert] {preset_name}"),
1543                        status: StepStatus::Passed,
1544                        message: "PASS".into(),
1545                    }
1546                } else {
1547                    StepResult {
1548                        name: format!("[assert] {preset_name}"),
1549                        status: StepStatus::Failed,
1550                        message: lr.content,
1551                    }
1552                }
1553            },
1554        )
1555    }
1556
1557    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1558    ///
1559    /// Evaluates the layout-scan JS in the page and fails with the list of
1560    /// detected issues: horizontal page overflow, visible elements sticking
1561    /// out of the viewport, text clipped by `overflow: hidden` containers,
1562    /// and interactive elements covered by other elements. No LLM call —
1563    /// checks are geometry-based so the check is free, deterministic, and
1564    /// safe to run on every page × viewport variant.
1565    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1566        let name = "[assert] layout_no_issues".to_owned();
1567        self.reporter
1568            .debug("assert: layout_no_issues (DOM layout scan)");
1569        let js = LAYOUT_SCAN_JS.replace(
1570            "__IGNORE_CLASSES__",
1571            &serde_json::to_string(&self.config.layout_ignore_classes)
1572                .unwrap_or_else(|_| "[]".to_owned()),
1573        );
1574        let result = tab.evaluate(&js, false);
1575        let json_str = match result {
1576            Ok(r) => r
1577                .value
1578                .as_ref()
1579                .and_then(|v| v.as_str().map(String::from))
1580                .unwrap_or_else(|| "[]".to_owned()),
1581            Err(e) => {
1582                return StepResult {
1583                    name,
1584                    status: StepStatus::Failed,
1585                    message: format!("layout scan JS failed: {e}"),
1586                };
1587            }
1588        };
1589        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1590        if issues.is_empty() {
1591            return StepResult {
1592                name,
1593                status: StepStatus::Passed,
1594                message: "PASS — no layout defects detected".into(),
1595            };
1596        }
1597        let mut lines: Vec<String> = issues
1598            .iter()
1599            .take(10)
1600            .map(|i| {
1601                format!(
1602                    "- [{type_}] {element}: {detail}",
1603                    type_ = i.issue_type,
1604                    element = i.element,
1605                    detail = i.detail
1606                )
1607            })
1608            .collect();
1609        if issues.len() > 10 {
1610            lines.push(format!("- … and {} more", issues.len() - 10));
1611        }
1612        StepResult {
1613            name,
1614            status: StepStatus::Failed,
1615            message: format!(
1616                "FAIL — {} layout defect(s) detected:\n{}",
1617                issues.len(),
1618                lines.join("\n")
1619            ),
1620        }
1621    }
1622
1623    fn run_custom(
1624        &self,
1625        prompt: &str,
1626        page_content: &PageContent,
1627        image: Option<&[String]>,
1628        step_endpoint: Option<&str>,
1629        test_endpoint: Option<&str>,
1630    ) -> StepResult {
1631        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1632
1633        let mut user = format!(
1634            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1635            url = page_content.url,
1636            title = page_content.title,
1637            content = page_content.body_text,
1638        );
1639        if image.is_some() {
1640            user.push_str(
1641                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1642            );
1643        }
1644
1645        self.reporter.debug("custom assert");
1646
1647        let chain = self
1648            .endpoints
1649            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1650        let sys = system.to_owned();
1651
1652        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1653
1654        response.map_or_else(
1655            |e| StepResult {
1656                name: "[assert] custom".into(),
1657                status: StepStatus::Failed,
1658                message: format!("LLM assertion call failed: {e}"),
1659            },
1660            |(lr, idx)| {
1661                self.usage.record_llm_call(
1662                    &chain[idx].name,
1663                    chain[idx],
1664                    lr.usage.prompt_tokens,
1665                    lr.usage.completion_tokens,
1666                );
1667                let content_lower = lr.content.to_lowercase().trim().to_owned();
1668                if content_lower.starts_with("pass") {
1669                    StepResult {
1670                        name: "[assert] custom".into(),
1671                        status: StepStatus::Passed,
1672                        message: "PASS".into(),
1673                    }
1674                } else {
1675                    StepResult {
1676                        name: "[assert] custom".into(),
1677                        status: StepStatus::Failed,
1678                        message: lr.content,
1679                    }
1680                }
1681            },
1682        )
1683    }
1684
1685    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1686        let path = path.unwrap_or("screenshot.png");
1687
1688        match tab.capture_screenshot(
1689            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1690            None,
1691            None,
1692            true,
1693        ) {
1694            Ok(data) => {
1695                if let Err(e) = std::fs::write(path, &data) {
1696                    return StepResult {
1697                        name: format!("[screenshot] {path}"),
1698                        status: StepStatus::Failed,
1699                        message: format!("failed to write screenshot: {e}"),
1700                    };
1701                }
1702                StepResult {
1703                    name: format!("[screenshot] {path}"),
1704                    status: StepStatus::Passed,
1705                    message: format!("saved to {path}"),
1706                }
1707            }
1708            Err(e) => StepResult {
1709                name: format!("[screenshot] {path}"),
1710                status: StepStatus::Failed,
1711                message: format!("screenshot failed: {e}"),
1712            },
1713        }
1714    }
1715
1716    /// Runs an A2A agent step.
1717    #[allow(clippy::literal_string_with_formatting_args)]
1718    fn run_agent(
1719        &self,
1720        agent_name: &str,
1721        task: &str,
1722        definition: Option<&str>,
1723        _test_endpoint: Option<&str>,
1724    ) -> StepResult {
1725        // If a definition is specified, look up the task template
1726        let resolved_task = if let Some(def_name) = definition {
1727            if let Some(def) = self.definitions.get(def_name) {
1728                let tmpl = def.task_template.as_deref().unwrap_or(task);
1729                tmpl.replace("{task}", task)
1730            } else {
1731                return StepResult {
1732                    name: format!("[agent] {def_name}"),
1733                    status: StepStatus::Failed,
1734                    message: format!("definition '{def_name}' not found"),
1735                };
1736            }
1737        } else {
1738            task.to_owned()
1739        };
1740
1741        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1742    }
1743
1744    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1745        let Some(ep) = self.endpoints.get(agent_name) else {
1746            return StepResult {
1747                name: format!("[agent] {display_name}"),
1748                status: StepStatus::Failed,
1749                message: format!("agent endpoint '{agent_name}' not found"),
1750            };
1751        };
1752
1753        if ep.url.is_empty() {
1754            return StepResult {
1755                name: format!("[agent] {display_name}"),
1756                status: StepStatus::Failed,
1757                message: format!("agent endpoint '{agent_name}' has no URL"),
1758            };
1759        }
1760
1761        self.reporter.debug(format!("agent {agent_name}: {task}"));
1762
1763        let url = ep.url.clone();
1764        let client = A2aClient::new(&url, self.timeout);
1765        let task_clone = task.to_owned();
1766
1767        let response = std::thread::spawn(move || {
1768            let rt = tokio::runtime::Builder::new_current_thread()
1769                .enable_all()
1770                .build()
1771                .unwrap();
1772            rt.block_on(client.send_task(&task_clone))
1773        })
1774        .join()
1775        .unwrap();
1776
1777        // Record the flat-cost call
1778        self.usage.record_flat_call(agent_name, ep);
1779
1780        match response {
1781            Ok(text) => {
1782                let clean = text.trim().to_owned();
1783                let lower = clean.to_lowercase();
1784                if lower.starts_with("pass") {
1785                    StepResult {
1786                        name: format!("[agent] {display_name}"),
1787                        status: StepStatus::Passed,
1788                        message: format!("PASS: {clean}"),
1789                    }
1790                } else if lower.starts_with("fail") {
1791                    StepResult {
1792                        name: format!("[agent] {display_name}"),
1793                        status: StepStatus::Failed,
1794                        message: clean,
1795                    }
1796                } else {
1797                    StepResult {
1798                        name: format!("[agent] {display_name}"),
1799                        status: StepStatus::Passed,
1800                        message: format!("response: {clean}"),
1801                    }
1802                }
1803            }
1804            Err(e) => StepResult {
1805                name: format!("[agent] {display_name}"),
1806                status: StepStatus::Failed,
1807                message: format!("agent call failed: {e}"),
1808            },
1809        }
1810    }
1811
1812    /// Runs an MCP tool call step.
1813    fn run_mcp(
1814        &self,
1815        server_name: &str,
1816        tool_name: &str,
1817        args: Option<&serde_json::Value>,
1818    ) -> StepResult {
1819        let Some(ep) = self.endpoints.get(server_name) else {
1820            return StepResult {
1821                name: format!("[mcp] {server_name}:{tool_name}"),
1822                status: StepStatus::Failed,
1823                message: format!("MCP server endpoint '{server_name}' not found"),
1824            };
1825        };
1826
1827        let cmd = ep.command.as_deref().unwrap_or("");
1828        if cmd.is_empty() {
1829            return StepResult {
1830                name: format!("[mcp] {server_name}:{tool_name}"),
1831                status: StepStatus::Failed,
1832                message: format!("MCP server '{server_name}' has no command configured"),
1833            };
1834        }
1835
1836        self.reporter
1837            .debug(format!("mcp {server_name} {tool_name}"));
1838
1839        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1840
1841        let command = cmd.to_owned();
1842        let args_vec = ep.args.clone();
1843        let tool = tool_name.to_owned();
1844
1845        let response = std::thread::spawn(move || {
1846            let mut mcp_client =
1847                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1848            mcp_client
1849                .call_tool(&tool, &args_val)
1850                .map_err(|e| e.to_string())
1851        })
1852        .join()
1853        .unwrap();
1854
1855        // Record the flat-cost call
1856        self.usage.record_flat_call(server_name, ep);
1857
1858        match response {
1859            Ok(result) => {
1860                if result.isError {
1861                    StepResult {
1862                        name: format!("[mcp] {server_name}:{tool_name}"),
1863                        status: StepStatus::Failed,
1864                        message: result.to_string(),
1865                    }
1866                } else {
1867                    StepResult {
1868                        name: format!("[mcp] {server_name}:{tool_name}"),
1869                        status: StepStatus::Passed,
1870                        message: result.to_string(),
1871                    }
1872                }
1873            }
1874            Err(e) => StepResult {
1875                name: format!("[mcp] {server_name}:{tool_name}"),
1876                status: StepStatus::Failed,
1877                message: format!("MCP call failed: {e}"),
1878            },
1879        }
1880    }
1881
1882    // ── helpers ──────────────────────────────────────────────────────────
1883
1884    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1885    /// the runner's default LLM config for any unset fields.
1886    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1887        LlmConfig {
1888            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
1889                // Bedrock builds its endpoint from the resolved AWS region
1890                // when no URL is given — never inherit the default LLM URL.
1891                endpoint.url.clone()
1892            } else if endpoint.url.is_empty() {
1893                self.llm.url.clone()
1894            } else {
1895                endpoint.url.clone()
1896            },
1897            model: endpoint
1898                .model
1899                .clone()
1900                .unwrap_or_else(|| self.llm.model.clone()),
1901            api_key: endpoint
1902                .api_key
1903                .clone()
1904                .or_else(|| self.llm.api_key.clone()),
1905            headers: if endpoint.headers.is_empty() {
1906                self.llm.headers.clone()
1907            } else {
1908                endpoint.headers.clone()
1909            },
1910            timeout: self.llm.timeout,
1911            temperature: self.llm.temperature,
1912            thinking: self.llm.thinking,
1913            model_params: self.llm.model_params.clone(),
1914            max_attempts: endpoint.max_attempts.max(1),
1915            provider: endpoint.provider,
1916            deployment: endpoint.deployment.clone(),
1917            api_version: endpoint.api_version.clone(),
1918            auth: endpoint.auth.clone(),
1919            header_commands: endpoint.header_commands.clone(),
1920            aws: endpoint.aws.clone(),
1921        }
1922    }
1923
1924    /// Runs a single LLM call against an ordered endpoint chain (primary +
1925    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
1926    /// the first endpoint that answers wins. Returns the response together
1927    /// with the chain index of the answering endpoint (0 = primary) so the
1928    /// caller can attribute usage to the correct endpoint.
1929    ///
1930    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
1931    /// shows duration, tokens, cost and the answering endpoint per call.
1932    /// Run-level context handed to every LLM call as the FIRST block of the
1933    /// user message. Contents that are stable for the whole run ("run
1934    /// started", "target site") come first so upstream provider prefix
1935    /// caching stays effective; the current time is the last line because
1936    /// it changes on every call.
1937    fn run_context(&self) -> String {
1938        let mut parts = vec![
1939            "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
1940            "================================================================".into(),
1941            format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
1942        ];
1943        if let Some(base) = self.config.base_url.as_deref() {
1944            parts.push(format!("Target site: {base}"));
1945        }
1946        let now = SystemTime::now()
1947            .duration_since(UNIX_EPOCH)
1948            .map_or(0, |d| d.as_secs());
1949        parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
1950        parts.join("\n")
1951    }
1952
1953    #[allow(clippy::cast_possible_truncation)]
1954    fn llm_call_chain(
1955        &self,
1956        chain: &[&ResolvedEndpoint],
1957        system: &str,
1958        user: &str,
1959        image: Option<&[String]>,
1960        purpose: &str,
1961    ) -> Result<(crate::costs::LlmResponse, usize), String> {
1962        if chain.is_empty() {
1963            return Err("empty LLM endpoint chain".into());
1964        }
1965        let primary = self.build_llm_for_endpoint(chain[0]);
1966        let fallbacks: Vec<LlmConfig> = chain[1..]
1967            .iter()
1968            .map(|e| self.build_llm_for_endpoint(e))
1969            .collect();
1970
1971        let (test, index) = self
1972            .current_step
1973            .borrow()
1974            .as_ref()
1975            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
1976        let primary_endpoint = chain[0].name.clone();
1977        let primary_model = primary.model.clone();
1978        self.emit_event(&TestEvent::LlmCallStarted {
1979            test: test.clone(),
1980            index,
1981            endpoint: primary_endpoint.clone(),
1982            model: primary_model.clone(),
1983            purpose: purpose.to_owned(),
1984        });
1985
1986        let started = Instant::now();
1987        let sys = system.to_owned();
1988        let context = self.run_context();
1989        let user = if context.is_empty() {
1990            user.to_owned()
1991        } else {
1992            format!("{context}\n\n{user}")
1993        };
1994        let image = image.map(<[String]>::to_vec);
1995
1996        let result = std::thread::spawn(move || {
1997            let rt = tokio::runtime::Builder::new_current_thread()
1998                .enable_all()
1999                .build()
2000                .unwrap();
2001            let call = async {
2002                match image.as_deref() {
2003                    Some(img) => {
2004                        llm_chat_vision_with_usage_chain(
2005                            &primary,
2006                            &fallbacks,
2007                            &sys,
2008                            &user,
2009                            Some(img),
2010                        )
2011                        .await
2012                    }
2013                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2014                }
2015            };
2016            rt.block_on(call)
2017        })
2018        .join()
2019        .unwrap();
2020
2021        let duration_ms = started.elapsed().as_millis() as u64;
2022        match result {
2023            Ok((lr, idx)) => {
2024                let cost = calculate_llm_cost(
2025                    chain[idx],
2026                    lr.usage.prompt_tokens,
2027                    lr.usage.completion_tokens,
2028                );
2029                let answering = chain[idx].name.clone();
2030                let model = chain[idx]
2031                    .model
2032                    .clone()
2033                    .unwrap_or_else(|| primary_model.clone());
2034                self.emit_event(&TestEvent::LlmCallFinished {
2035                    test,
2036                    index,
2037                    endpoint: answering,
2038                    model,
2039                    purpose: purpose.to_owned(),
2040                    ok: true,
2041                    duration_ms,
2042                    input_tokens: lr.usage.prompt_tokens,
2043                    output_tokens: lr.usage.completion_tokens,
2044                    cost,
2045                    error: None,
2046                });
2047                Ok((lr, idx))
2048            }
2049            Err(e) => {
2050                self.emit_event(&TestEvent::LlmCallFinished {
2051                    test,
2052                    index,
2053                    endpoint: primary_endpoint,
2054                    model: primary_model,
2055                    purpose: purpose.to_owned(),
2056                    ok: false,
2057                    duration_ms,
2058                    input_tokens: 0,
2059                    output_tokens: 0,
2060                    cost: 0.0,
2061                    error: Some(e.clone()),
2062                });
2063                Err(e)
2064            }
2065        }
2066    }
2067
2068    /// Resolves a CSS selector for the target element. Uses the explicit
2069    /// `selector` if provided, otherwise asks the LLM to find the element
2070    /// from the natural language `target` description and page DOM.
2071    ///
2072    /// LLM responses are sanitized and verified against the live page: a
2073    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2074    /// immediately with the raw LLM output, and a selector that matches
2075    /// nothing triggers one retry with feedback before failing.
2076    #[allow(clippy::too_many_lines)]
2077    fn resolve_selector(
2078        &self,
2079        css_override: Option<&str>,
2080        target: &str,
2081        step_endpoint: Option<&str>,
2082        test_endpoint: Option<&str>,
2083        tab: &Tab,
2084    ) -> Result<String, String> {
2085        if let Some(explicit) = css_override {
2086            return Ok(explicit.to_owned());
2087        }
2088
2089        let dom_info = extract_dom_info(tab)?;
2090        let page_content = get_page_text(tab);
2091
2092        let system = concat!(
2093            "You are a browser automation selector generator. ",
2094            "Given a web page's content and interactive elements, ",
2095            "return ONLY the best CSS selector for the described element. ",
2096            "Output nothing except the CSS selector. ",
2097            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2098            "[name=\"...\"], tag.class, tag. ",
2099            "Never output explanations, markdown, or extra text."
2100        );
2101
2102        let user = format!(
2103            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2104            page_content.url,
2105            page_content.title,
2106            truncate(&page_content.body_text, 4000),
2107            dom_info,
2108            target,
2109        );
2110
2111        let retry_user = format!(
2112            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2113            "The selector must match at least one element currently present on the page.",
2114            page_content.url,
2115            page_content.title,
2116            truncate(&page_content.body_text, 4000),
2117            dom_info,
2118            target,
2119        );
2120
2121        self.reporter.debug(format!("LLM targeting: {target}"));
2122
2123        let chain = self
2124            .endpoints
2125            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2126        let sys = system.to_owned();
2127
2128        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2129
2130        let first = call_llm(&user);
2131        let (lr, idx) = match first {
2132            Ok(lr) => lr,
2133            Err(e) => {
2134                return Err(format!("LLM element targeting failed: {e}"));
2135            }
2136        };
2137        self.usage.record_llm_call(
2138            &chain[idx].name,
2139            chain[idx],
2140            lr.usage.prompt_tokens,
2141            lr.usage.completion_tokens,
2142        );
2143        let clean = sanitize_selector(&lr.content);
2144        self.reporter.debug(format!("resolved selector: {clean}"));
2145
2146        if selector_is_useless(&clean) {
2147            return Err(format!(
2148                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2149                raw = lr.content.trim(),
2150            ));
2151        }
2152        if let Err(reason) = validate_selector(&clean) {
2153            return Err(format!(
2154                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2155                raw = lr.content.trim(),
2156            ));
2157        }
2158        if !selector_matches(tab, &clean).unwrap_or(false) {
2159            // One retry with feedback: flaky models occasionally invent a
2160            // selector that does not exist on the page.
2161            self.reporter.warn(format!(
2162                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2163            ));
2164            let second = call_llm(&retry_user);
2165            let (lr2, idx2) = match second {
2166                Ok(lr2) => lr2,
2167                Err(e) => {
2168                    return Err(format!(
2169                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2170                    ));
2171                }
2172            };
2173            self.usage.record_llm_call(
2174                &chain[idx2].name,
2175                chain[idx2],
2176                lr2.usage.prompt_tokens,
2177                lr2.usage.completion_tokens,
2178            );
2179            let clean2 = sanitize_selector(&lr2.content);
2180            self.reporter
2181                .debug(format!("resolved selector (retry): {clean2}"));
2182            if selector_is_useless(&clean2) {
2183                return Err(format!(
2184                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2185                    raw = lr2.content.trim(),
2186                    excerpt = truncate(&page_content.body_text, 300),
2187                ));
2188            }
2189            if !selector_matches(tab, &clean2).unwrap_or(false) {
2190                return Err(format!(
2191                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2192                ));
2193            }
2194            return Ok(clean2);
2195        }
2196
2197        Ok(clean)
2198    }
2199}
2200
2201/// Evaluates a JS expression that is expected to return a boolean.
2202fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2203    tab.evaluate(js, false)
2204        .map_err(|e| format!("evaluate failed: {e}"))?
2205        .value
2206        .and_then(|v| v.as_bool())
2207        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2208}
2209
2210/// Checks whether a CSS selector matches at least one current element.
2211fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2212    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2213}
2214
2215// ── Free helper functions ──────────────────────────────────────────────
2216
2217fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2218    let name = format!("[navigate] {full_url}");
2219    match tab.navigate_to(full_url) {
2220        Ok(_) => {
2221            let _ = tab.wait_until_navigated();
2222            StepResult {
2223                name,
2224                status: StepStatus::Passed,
2225                message: format!("navigated to {full_url}"),
2226            }
2227        }
2228        Err(e) => StepResult {
2229            name,
2230            status: StepStatus::Failed,
2231            message: format!("navigation failed: {e}"),
2232        },
2233    }
2234}
2235
2236fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2237    let result = tab
2238        .evaluate(DOM_EXTRACT_JS, false)
2239        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2240
2241    let json_str = result
2242        .value
2243        .as_ref()
2244        .and_then(|v| v.as_str())
2245        .unwrap_or("[]");
2246
2247    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2248
2249    if elements.is_empty() {
2250        return Ok("(no interactive elements found)".to_owned());
2251    }
2252
2253    Ok(elements.join("\n"))
2254}
2255
2256fn get_page_text(tab: &Tab) -> PageContent {
2257    let url = tab.get_url();
2258
2259    let title = tab
2260        .evaluate("document.title", false)
2261        .ok()
2262        .and_then(|r| r.value)
2263        .and_then(|v| v.as_str().map(String::from))
2264        .unwrap_or_else(|| "unknown".to_owned());
2265
2266    let body_text = tab
2267        .evaluate(
2268            "document.body ? document.body.innerText : document.documentElement.innerText",
2269            false,
2270        )
2271        .ok()
2272        .and_then(|r| r.value)
2273        .and_then(|v| v.as_str().map(String::from))
2274        .unwrap_or_default();
2275
2276    PageContent {
2277        url,
2278        title,
2279        body_text: truncate(&body_text, 8000),
2280    }
2281}
2282
2283fn resolve_url(url: &str, base_url: &str) -> String {
2284    if url.starts_with("http://") || url.starts_with("https://") {
2285        return url.to_owned();
2286    }
2287    let base = base_url.trim_end_matches('/');
2288    if url.starts_with('/') {
2289        format!("{base}{url}")
2290    } else {
2291        format!("{base}/{url}")
2292    }
2293}
2294
2295/// Human-readable label for a step, used when steps are skipped after an
2296/// earlier failure.
2297fn step_label(step: &TestStep) -> String {
2298    match step {
2299        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2300        TestStep::Click { target, .. } => format!("[click] {target}"),
2301        TestStep::Type { target, .. } => format!("[type] {target}"),
2302        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2303        TestStep::Assert {
2304            definition,
2305            preset,
2306            prompt,
2307            ..
2308        } => definition.as_ref().map_or_else(
2309            || {
2310                preset.as_ref().map_or_else(
2311                    || {
2312                        prompt.as_ref().map_or_else(
2313                            || "[assert]".to_owned(),
2314                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2315                        )
2316                    },
2317                    |p| format!("[assert] {p}"),
2318                )
2319            },
2320            |d| format!("[assert] {d}"),
2321        ),
2322        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2323        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2324        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2325    }
2326}
2327
2328/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2329#[must_use]
2330const fn step_kind_label(step: &TestStep) -> &'static str {
2331    match step {
2332        TestStep::Navigate { .. } => "navigate",
2333        TestStep::Click { .. } => "click",
2334        TestStep::Type { .. } => "type",
2335        TestStep::Wait { .. } => "wait",
2336        TestStep::Assert { .. } => "assert",
2337        TestStep::Screenshot { .. } => "screenshot",
2338        TestStep::Agent { .. } => "agent",
2339        TestStep::Mcp { .. } => "mcp",
2340    }
2341}
2342
2343// ── Support types ──────────────────────────────────────────────────────
2344
2345#[derive(Default)]
2346struct TestRunResult {
2347    passed: u32,
2348    failed: u32,
2349    skipped: u32,
2350    total: u32,
2351    details: Vec<StepResult>,
2352}
2353
2354struct PageContent {
2355    url: String,
2356    title: String,
2357    body_text: String,
2358}
2359
2360#[cfg(test)]
2361mod tests {
2362    use super::unix_to_rfc3339;
2363
2364    #[test]
2365    fn rfc3339_epoch_and_reference_dates() {
2366        assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2367        assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2368        assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2369        assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2370    }
2371
2372    #[test]
2373    fn rfc3339_handles_leap_years() {
2374        assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2375        assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2376    }
2377}