Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// Formats a UNIX epoch timestamp as an RFC 3339 UTC string
26/// (`2026-09-19T16:04:00Z`) without pulling in a date dependency.
27/// Civil-from-days conversion after Howard Hinnant.
28#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31    let secs = i64::try_from(secs).unwrap_or(0);
32    let days = secs.div_euclid(86_400);
33    let rem = secs.rem_euclid(86_400);
34    let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35    let z = days + 719_468;
36    let era = z.div_euclid(146_097);
37    let doe = z.rem_euclid(146_097);
38    let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40    let mp = (5 * doy + 2) / 153;
41    let d = doy - (153 * mp + 2) / 5 + 1;
42    let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43    let y = yoe + era * 400 + i64::from(mth <= 2);
44    format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47/// One detected layout defect (`layout_no_issues` preset).
48#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50    #[serde(rename = "type")]
51    issue_type: String,
52    element: String,
53    detail: String,
54}
55
56/// In-browser DOM layout scan for `layout_no_issues`.
57///
58/// Geometry-only checks (no LLM, no pixels):
59/// 1. page horizontal overflow (`scrollWidth` > viewport width);
60/// 2. elements outside the viewport that scrolling cannot reveal
61///    (fixed elements off-screen, left/negative overflow, right-edge
62///    overflow beyond the horizontally scrollable area, and bottom
63///    overflow on a page that cannot scroll down) — below-the-fold
64///    content on a tall scrollable page is normal flow, NOT a defect;
65/// 3. text clipped by `overflow: hidden` containers whose content
66///    is measurably larger than the box;
67/// 4. interactive elements (buttons/links/inputs) whose center point
68///    is covered by a different element that would intercept the click.
69///
70/// Elements whose class matches a configured ignore prefix (default:
71/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
72/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
73/// skipped — they are intentionally 1x1 / off-screen. The prefix list
74/// is injected at `__IGNORE_CLASSES__` from
75/// [`ScenarioConfig::layout_ignore_classes`].
76///
77/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
78/// sticky headers) is excluded by the position/relation filters.
79const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81  const issues = [];
82  const push = (type, el, detail) => {
83    if (issues.length >= 30) return;
84    let element = el.tagName.toLowerCase();
85    if (el.id) element += '#' + el.id;
86    else if (typeof el.className === 'string' && el.className.trim())
87      element += '.' + el.className.trim().split(/\s+/).join('.');
88    issues.push({ type, element, detail: String(detail).slice(0, 220) });
89  };
90  const vw = document.documentElement.clientWidth || window.innerWidth;
91  const vh = document.documentElement.clientHeight || window.innerHeight;
92  if (!vw || !vh) return JSON.stringify(issues);
93  const de = document.documentElement;
94  // 1. Page-level horizontal overflow.
95  if (maxSW > vw + 2)
96    push('page-overflow-x', de,
97      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98  // Class-name prefixes to skip (injected; default = Angular CDK
99  // screen-reader helpers, which are intentionally 1x1 / off-screen).
100  const ignorePrefixes = __IGNORE_CLASSES__;
101  const isIgnored = (el) => {
102    if (typeof el.className !== 'string' || !el.className.trim()) return false;
103    const classes = el.className.trim().split(/\s+/);
104    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105  };
106  const body = document.body;
107  // The document element is not always the scroll container: the app may
108  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110  // "not scrollable" — measure the scrollable content height/width of the
111  // document element AND the body, and take the max.
112  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114  const vScrollable = maxSH > vh + 2;
115  const hScrollable = maxSW > vw + 2;
116  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117  // this element into view on the given axis — i.e. the element is inside a
118  // scrolling region, so lying beyond the viewport cut is not a defect.
119  const reachableByScroller = (el, axis) => {
120    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123    let p = el.parentElement;
124    while (p && p !== body) {
125      const pcs = getComputedStyle(p);
126      const o = pcs[ovProp];
127      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128          p[dim] > p[dimClient] + 2) return true;
129      p = p.parentElement;
130    }
131    return false;
132  };
133  const selfOverflowing = (el, cs, axis) => {
134    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138      el[dim] > el[dimClient] + 2;
139  };
140  // True when an ancestor clips this axis with overflow:hidden/clip — the
141  // element's overhang is not visible, so treat it as reachable.
142  const clippedByAncestor = (el, axis) => {
143    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144    let p = el.parentElement;
145    while (p && p !== body) {
146      const o = getComputedStyle(p)[ovProp];
147      if (o === 'hidden' || o === 'clip') return true;
148      p = p.parentElement;
149    }
150    return false;
151  };
152  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154  const hasContent = (el) =>
155    ((el.textContent || '').trim().length > 0) ||
156    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157  // 2. Elements outside the viewport that scrolling cannot reveal.
158  for (const el of all) {
159    const cs = getComputedStyle(el);
160    if (!visible(cs)) continue;
161    if (isIgnored(el)) continue;
162    const r = el.getBoundingClientRect();
163    if (r.width < 2 || r.height < 2) continue;
164    if (!hasContent(el) && el.children.length === 0) continue;
165    if (cs.position === 'fixed') {
166      // Fixed elements never move with the scroll: any edge outside the
167      // viewport is unreachable content and therefore a defect.
168      const overTop = -r.top;
169      const overLeft = -r.left;
170      const overRight = r.right - vw;
171      const overBottom = r.bottom - vh;
172      // Top/left overshoot is never reachable; bottom/right overshoot is
173      // only a defect when the fixed element cannot scroll that content
174      // into view itself (e.g. an opened Material drawer whose inner
175      // container scrolls is not a "cut-off" defect).
176      if (overTop > 2 || overLeft > 2 ||
177          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179        let where = '';
180        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186        push('element-out-of-viewport', el,
187          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188      }
189      continue;
190    }
191    if (cs.position === 'sticky') continue;
192    // Negative left overflow cannot be reached by scrolling (scrollLeft
193    // never goes below 0).
194    if (r.left < -2) {
195      push('element-out-of-viewport', el,
196        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197      continue;
198    }
199    // Negative top with the page at the top means the element sits above
200    // the document origin — also unreachable.
201    if (r.top < -2 && de.scrollTop <= 2) {
202      push('element-out-of-viewport', el,
203        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204      continue;
205    }
206    const overRight = r.right - vw;
207    // Right-edge overflow is only a defect when the page cannot scroll
208    // horizontally to reveal it (or the element sticks out past the
209    // scrollable content width itself).
210    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211      push('element-out-of-viewport', el,
212        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213      continue;
214    }
215    // Below-the-fold content on a scrollable page is normal (tall landing
216    // pages); only flag bottom overflow the user can never scroll to.
217    const overBottom = r.bottom - vh;
218    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219      push('element-out-of-viewport', el,
220        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221    }
222  }
223  // 3. Text clipped by overflow:hidden containers.
224  for (const el of all) {
225    if (isIgnored(el)) continue;
226    const cs = getComputedStyle(el);
227    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229    if (!(el.textContent || '').trim()) continue;
230    // Single-line truncation with an ellipsis is an intentional design
231    // pattern (Tailwind .truncate etc.), not a clipping defect.
232    if (cs.textOverflow === 'ellipsis') continue;
233    push('text-clipped', el,
234      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236  }
237  // 4. Interactive elements covered by a different element.
238  const interactive =
239    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240  const targets = document.querySelectorAll(interactive);
241  for (const el of targets) {
242    if (isIgnored(el)) continue;
243    const r = el.getBoundingClientRect();
244    if (r.width < 6 || r.height < 6) continue;
245    const cs = getComputedStyle(el);
246    if (!visible(cs)) continue;
247    const cx = r.left + r.width / 2;
248    const cy = r.top + r.height / 2;
249    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250    const top = document.elementFromPoint(cx, cy);
251    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252    if (isIgnored(top)) continue;
253    const tcs = getComputedStyle(top);
254    if (!visible(tcs)) continue;
255    if (tcs.pointerEvents === 'none') continue;
256    const tr = top.getBoundingClientRect();
257    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260    push('element-overlap', el,
261      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262      ' is covered by <' + tname + '>');
263  }
264  return JSON.stringify(issues);
265})()
266"#;
267
268/// How long the CDP connection stays open after the browser goes quiet.
269///
270/// `headless_chrome` ships a 30s default and tears down the entire connection
271/// when no traffic arrives for that long; a run must own its connection for
272/// its full duration instead.
273const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275/// Executes a [`Scenario`] against a real browser with optional LLM
276/// assistance for element targeting and assertions.
277pub struct ScenarioRunner {
278    config: ScenarioConfig,
279    definitions: HashMap<String, AssertDefinition>,
280    llm: LlmConfig,
281    timeout: Duration,
282    viewport_width: u32,
283    viewport_height: u32,
284    /// The viewport currently applied in the browser (CDP emulation).
285    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
286    applied_viewport: std::cell::Cell<(u32, u32)>,
287    endpoints: EndpointRegistry,
288    usage: Arc<UsageTracker>,
289    budgets: BudgetTracker,
290    /// Directory for failure artifacts (screenshots).
291    artifacts_dir: PathBuf,
292    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
293    reporter: Arc<Reporter>,
294    /// The test + step index currently executing, for LLM-call events.
295    current_step: std::cell::RefCell<Option<(String, u32)>>,
296    /// Epoch seconds of runner creation — constant for the whole run so
297    /// the LLM context preamble below is cacheable by upstream providers.
298    run_started: u64,
299    /// Whether this runner emits the `RunStarted`/`RunFinished` events.
300    /// When several scenario files are run concurrently by
301    /// [`crate::parallel::run_scenarios`], the orchestrator owns those two
302    /// events (one per batch), so each per-file runner suppresses its own.
303    emit_run_events: bool,
304}
305
306/// Aggregated results from a scenario run.
307#[derive(Debug, Default)]
308pub struct RunReport {
309    /// Number of tests that passed.
310    pub tests_passed: u32,
311    /// Number of tests that failed.
312    pub tests_failed: u32,
313    /// Number of steps that passed.
314    pub passed: u32,
315    /// Number of steps that failed.
316    pub failed: u32,
317    /// Number of steps that were skipped.
318    pub skipped: u32,
319    /// Per-step details.
320    pub details: Vec<StepResult>,
321}
322
323/// Result of a single step execution.
324#[derive(Debug, Clone)]
325pub struct StepResult {
326    /// The step name.
327    pub name: String,
328    /// Whether the step passed, failed, or was skipped.
329    pub status: StepStatus,
330    /// Human-readable result message.
331    pub message: String,
332}
333
334/// Outcome for a single step.
335pub use crate::events::StepStatus;
336
337/// Predefined assertion preset definition.
338struct AssertPreset {
339    name: &'static str,
340    system: &'static str,
341    user_template: &'static str,
342}
343
344/// Built-in assertion presets.
345#[allow(clippy::literal_string_with_formatting_args)]
346const ASSERTION_PRESETS: &[AssertPreset] = &[
347    AssertPreset {
348        name: "no_error_on_page",
349        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
350        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
351    },
352    AssertPreset {
353        name: "text_visible",
354        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
355        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
356    },
357    AssertPreset {
358        name: "element_exists",
359        system: "You are a QA tester. Check if a described UI element exists on a web page.",
360        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
361    },
362    AssertPreset {
363        name: "visual_no_issues",
364        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
365        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
366    },
367    AssertPreset {
368        name: "visual_no_overlaps",
369        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
370        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
371    },
372    AssertPreset {
373        name: "visual_text_visible",
374        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
375        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
376    },
377    AssertPreset {
378        name: "layout_no_issues",
379        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
380        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
381    },
382];
383
384/// Whether an LLM verdict should be read as PASS.
385///
386/// Models routinely wrap the verdict in markdown (`**PASS** …`) or prefix it
387/// with punctuation, which a bare `starts_with("pass")` misreads as a failure.
388/// Strip any leading non-alphanumerics, then match the leading word; this also
389/// accepts the natural-language forms `pass` / `passes`.
390fn verdict_is_pass(content: &str) -> bool {
391    content
392        .trim()
393        .trim_start_matches(|c: char| !c.is_alphanumeric())
394        .to_lowercase()
395        .starts_with("pass")
396}
397
398/// Whether an LLM verdict should be read as FAIL (see [`verdict_is_pass`]).
399fn verdict_is_fail(content: &str) -> bool {
400    content
401        .trim()
402        .trim_start_matches(|c: char| !c.is_alphanumeric())
403        .to_lowercase()
404        .starts_with("fail")
405}
406
407impl ScenarioRunner {
408    /// Creates a new runner with the given scenario configuration and
409    /// assertion definitions.
410    #[must_use]
411    #[allow(clippy::needless_pass_by_value)]
412    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
413        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
414    }
415
416    /// Creates a runner that reports run events through the given reporter.
417    #[must_use]
418    #[allow(clippy::needless_pass_by_value)]
419    pub fn with_reporter(
420        scenario_config: ScenarioConfig,
421        definitions: Vec<AssertDefinition>,
422        reporter: Arc<Reporter>,
423    ) -> Self {
424        Self::with_reporter_mode(scenario_config, definitions, reporter, true)
425    }
426
427    /// Creates a runner that reports run events through the given reporter
428    /// but suppresses the `RunStarted`/`RunFinished` events. Used by the
429    /// parallel orchestrator, which emits those once per batch.
430    #[must_use]
431    #[allow(clippy::needless_pass_by_value)]
432    pub fn with_reporter_parallel(
433        scenario_config: ScenarioConfig,
434        definitions: Vec<AssertDefinition>,
435        reporter: Arc<Reporter>,
436    ) -> Self {
437        Self::with_reporter_mode(scenario_config, definitions, reporter, false)
438    }
439
440    #[must_use]
441    #[allow(clippy::needless_pass_by_value)]
442    fn with_reporter_mode(
443        scenario_config: ScenarioConfig,
444        definitions: Vec<AssertDefinition>,
445        reporter: Arc<Reporter>,
446        emit_run_events: bool,
447    ) -> Self {
448        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
449            &scenario_config,
450        ));
451        let llm = LlmConfig {
452            url: scenario_config
453                .llm_url
454                .clone()
455                .unwrap_or_else(crate::llm_base_url),
456            model: scenario_config
457                .llm_model
458                .clone()
459                .unwrap_or_else(crate::llm_model),
460            api_key: scenario_config
461                .llm_api_key
462                .clone()
463                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
464            headers: if scenario_config.llm_headers.is_empty() {
465                crate::parse_headers_env()
466            } else {
467                scenario_config.llm_headers.clone()
468            },
469            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
470            temperature: scenario_config.temperature,
471            thinking: scenario_config.thinking,
472            model_params: scenario_config.model_params.clone(),
473            cache: scenario_config.cache.unwrap_or(true),
474            max_attempts: crate::default_llm_attempts(),
475            provider: crate::scenario::Provider::Openai,
476            deployment: None,
477            api_version: None,
478            auth: crate::scenario::AuthConfig::default(),
479            header_commands: std::collections::HashMap::new(),
480            aws: crate::scenario::AwsConfig::default(),
481        };
482        let endpoints = EndpointRegistry::from_config(
483            &scenario_config.endpoints,
484            Some(&llm),
485            crate::endpoints::EndpointDefaults {
486                cache: scenario_config.cache,
487                cache_pricing: scenario_config.cache_pricing,
488            },
489        );
490        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
491        let defs_map: HashMap<String, AssertDefinition> = definitions
492            .into_iter()
493            .map(|d| (d.name.clone(), d))
494            .collect();
495
496        Self {
497            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
498            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
499            viewport_height: scenario_config.viewport_height.unwrap_or(720),
500            applied_viewport: std::cell::Cell::new((0, 0)),
501            config: scenario_config.clone(),
502            definitions: defs_map,
503            llm,
504            endpoints,
505            usage: Arc::new(UsageTracker::new()),
506            budgets,
507            artifacts_dir: PathBuf::from(
508                scenario_config
509                    .artifacts_dir
510                    .unwrap_or_else(|| "artifacts".to_owned()),
511            ),
512            reporter,
513            current_step: std::cell::RefCell::new(None),
514            run_started: SystemTime::now()
515                .duration_since(UNIX_EPOCH)
516                .map_or(0, |d| d.as_secs()),
517            emit_run_events,
518        }
519    }
520
521    /// Emits an event; a sink failure degrades to a console warning so a
522    /// broken log file can never mask the run itself.
523    fn emit_event(&self, event: &TestEvent) {
524        if let Err(err) = self.reporter.emit(event) {
525            use std::io::Write as _;
526            let mut out = std::io::stderr().lock();
527            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
528        }
529    }
530
531    /// Returns a clone of the [`UsageTracker`] for reporting.
532    #[must_use]
533    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
534        Arc::clone(&self.usage)
535    }
536
537    /// Returns a reference to the [`BudgetTracker`].
538    #[must_use]
539    pub const fn budget_tracker(&self) -> &BudgetTracker {
540        &self.budgets
541    }
542
543    /// Executes all test groups in the scenario and returns a report.
544    ///
545    /// # Errors
546    ///
547    /// Returns an error if the browser fails to launch.
548    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
549    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
550        let mut report = RunReport::default();
551
552        if tests.is_empty() {
553            self.reporter.warn("No tests defined in scenario.");
554            if self.emit_run_events {
555                self.emit_event(&TestEvent::RunFinished {
556                    tests_passed: 0,
557                    tests_failed: 0,
558                    steps_passed: 0,
559                    steps_failed: 0,
560                    steps_skipped: 0,
561                    total_cost: 0.0,
562                    total_tokens: 0,
563                    total_input_tokens: 0,
564                    total_output_tokens: 0,
565                    total_cached_input_tokens: 0,
566                    total_cache_creation_input_tokens: 0,
567                    models: Vec::new(),
568                    total_calls: 0,
569                });
570            }
571            return Ok(report);
572        }
573
574        if self.emit_run_events {
575            self.emit_event(&TestEvent::RunStarted {
576                total_tests: tests.len() as u32,
577            });
578        }
579
580        let browser_headless = self.config.browser_headless.unwrap_or(true);
581
582        let launch_opts = LaunchOptions {
583            headless: browser_headless,
584            window_size: Some((self.viewport_width, self.viewport_height)),
585            sandbox: false,
586            // headless_chrome defaults this to 30s and shuts down the whole CDP
587            // connection when no messages arrive for that long. A scenario can
588            // easily exceed 30s of browser silence (slow LLM targeting/assertion
589            // calls, page waits, budget checks between steps), after which every
590            // remaining step fails with "Unable to make method calls because
591            // underlying connection is closed" — one quiet gap kills the run.
592            // Open-ended scenarios must own the connection for their full
593            // duration, so keep it alive for 6 hours.
594            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
595            ..LaunchOptions::default()
596        };
597
598        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
599        let tab = browser.new_tab().context("failed to open browser tab")?;
600        let _ = tab.set_default_timeout(self.timeout);
601
602        // Start MCP server if configured
603        #[cfg(feature = "mcp-server")]
604        if let Some(ref mcp_cfg) = self.config.mcp_server {
605            if mcp_cfg.enabled {
606                let port = mcp_cfg.port;
607                std::thread::spawn(move || {
608                    let _ = crate::mcp_server::start_mcp_server(port);
609                });
610            }
611        }
612        #[cfg(not(feature = "mcp-server"))]
613        if let Some(mcp_cfg) = &self.config.mcp_server {
614            if mcp_cfg.enabled {
615                self.reporter
616                    .warn("MCP server configured but 'mcp-server' feature not enabled");
617            }
618        }
619
620        // Start A2A agent server if configured
621        #[cfg(feature = "a2a-server")]
622        if let Some(ref a2a_cfg) = self.config.a2a_server {
623            if a2a_cfg.enabled {
624                let port = a2a_cfg.port;
625                tokio::spawn(crate::a2a_server::start_a2a_server(port));
626            }
627        }
628        #[cfg(not(feature = "a2a-server"))]
629        if let Some(a2a_cfg) = &self.config.a2a_server {
630            if a2a_cfg.enabled {
631                self.reporter
632                    .warn("A2A server configured but 'a2a-server' feature not enabled");
633            }
634        }
635
636        for test in tests {
637            self.emit_event(&TestEvent::TestStarted {
638                test: test.name.clone(),
639            });
640
641            self.usage.reset_per_test();
642
643            let test_started = Instant::now();
644            let test_result = self.run_test(test, &tab);
645            let duration_ms = test_started.elapsed().as_millis() as u64;
646            let usage = self.usage.current_test_snapshot();
647            self.usage.commit_test(&test.name);
648
649            self.emit_event(&TestEvent::TestFinished {
650                test: test.name.clone(),
651                passed: test_result.passed,
652                failed: test_result.failed,
653                skipped: test_result.skipped,
654                duration_ms,
655                cost: usage.total_cost,
656                tokens: usage.total_tokens,
657                input_tokens: usage.total_input_tokens,
658                output_tokens: usage.total_output_tokens,
659                cached_input_tokens: usage.total_cached_input_tokens,
660                cache_creation_input_tokens: usage.total_cache_creation_input_tokens,
661                models: usage.models.clone(),
662                calls: usage.total_calls,
663            });
664
665            if test_result.failed == 0 && test_result.total > 0 {
666                report.tests_passed += 1;
667            } else if test_result.total > 0 {
668                report.tests_failed += 1;
669            }
670
671            report.passed += test_result.passed;
672            report.failed += test_result.failed;
673            report.skipped += test_result.skipped;
674            report.details.extend(test_result.details);
675        }
676
677        let global = self.usage.global_snapshot();
678        if self.emit_run_events {
679            self.emit_event(&TestEvent::RunFinished {
680                tests_passed: report.tests_passed,
681                tests_failed: report.tests_failed,
682                steps_passed: report.passed,
683                steps_failed: report.failed,
684                steps_skipped: report.skipped,
685                total_cost: global.total_cost,
686                total_tokens: global.total_tokens,
687                total_input_tokens: global.total_input_tokens,
688                total_output_tokens: global.total_output_tokens,
689                total_cached_input_tokens: global.total_cached_input_tokens,
690                total_cache_creation_input_tokens: global.total_cache_creation_input_tokens,
691                models: global.models.clone(),
692                total_calls: global.total_calls,
693            });
694        }
695
696        Ok(report)
697    }
698
699    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
700    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
701        let base_url = test
702            .base_url
703            .clone()
704            .or_else(|| self.config.base_url.clone())
705            .unwrap_or_else(crate::base_url);
706
707        // Per-test viewport override: switch the browser via CDP
708        // device-metrics emulation before this test runs.
709        let vw = test.viewport_width.unwrap_or(self.viewport_width);
710        let vh = test.viewport_height.unwrap_or(self.viewport_height);
711        if self.applied_viewport.get() != (vw, vh) {
712            self.apply_viewport(tab, vw, vh);
713            self.applied_viewport.set((vw, vh));
714        }
715
716        // Per-test isolation: every test starts from its own start_url
717        // (unless auto_navigate is disabled), so a test never inherits the
718        // previous test's page state.
719        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
720
721        let start_url = test
722            .start_url
723            .clone()
724            .or_else(|| self.config.start_url.clone())
725            .unwrap_or_else(|| "/dashboard".to_owned());
726
727        if auto_navigate {
728            let full_url = resolve_url(&start_url, &base_url);
729            self.reporter.debug(format!("auto-navigate: {full_url}"));
730            let _ = tab.navigate_to(&full_url);
731            let _ = tab.wait_until_navigated();
732            std::thread::sleep(Duration::from_secs(4));
733        }
734
735        let mut result = TestRunResult::default();
736
737        for (step_index, step) in test.steps.iter().enumerate() {
738            result.total += 1;
739
740            let wait_ms = match step {
741                TestStep::Navigate { wait_after_ms, .. }
742                | TestStep::Click { wait_after_ms, .. }
743                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
744                _ => None,
745            };
746
747            self.current_step
748                .replace(Some((test.name.clone(), step_index as u32)));
749            self.emit_event(&TestEvent::StepStarted {
750                test: test.name.clone(),
751                index: step_index as u32,
752                label: step_label(step),
753            });
754            let step_started = Instant::now();
755
756            let mut step_result = match step {
757                TestStep::Navigate { url, .. } => {
758                    let full_url = resolve_url(url, &base_url);
759                    run_navigate_step(&full_url, tab)
760                }
761                TestStep::Click {
762                    target,
763                    selector,
764                    endpoint,
765                    idempotent,
766                    ..
767                } => self.run_click(
768                    target,
769                    selector.as_deref(),
770                    endpoint.as_deref(),
771                    test.endpoint.as_deref(),
772                    *idempotent,
773                    tab,
774                ),
775                TestStep::Type {
776                    target,
777                    text,
778                    selector,
779                    endpoint,
780                    idempotent,
781                    ..
782                } => self.run_type(
783                    target,
784                    text,
785                    selector.as_deref(),
786                    endpoint.as_deref(),
787                    test.endpoint.as_deref(),
788                    *idempotent,
789                    tab,
790                ),
791                TestStep::Wait {
792                    target,
793                    selector,
794                    text,
795                    timeout_ms,
796                    endpoint,
797                    idempotent,
798                } => self.run_wait(
799                    target,
800                    selector.as_deref(),
801                    text.as_deref(),
802                    *timeout_ms,
803                    endpoint.as_deref(),
804                    test.endpoint.as_deref(),
805                    *idempotent,
806                    tab,
807                ),
808                TestStep::Assert {
809                    definition,
810                    preset,
811                    prompt,
812                    assert_text,
813                    endpoint,
814                    screenshot,
815                } => self.run_assert(
816                    definition.as_deref(),
817                    preset.as_deref(),
818                    prompt.as_deref(),
819                    assert_text.as_deref(),
820                    *screenshot,
821                    endpoint.as_deref(),
822                    test.endpoint.as_deref(),
823                    tab,
824                ),
825                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
826                TestStep::Agent {
827                    agent,
828                    task,
829                    definition,
830                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
831                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
832            };
833
834            // Failure diagnostics: capture the page state and a screenshot so
835            // CI logs say WHAT the page looked like when the step failed,
836            // instead of a bare "timed out: The event waited for never came".
837            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
838                let state = diagnostics::capture(tab);
839                let screenshot = diagnostics::save_screenshot(
840                    tab,
841                    &self.artifacts_dir,
842                    &test.name,
843                    &test.name,
844                    step_index,
845                    step_kind_label(step),
846                );
847                step_result.message = format!(
848                    "{base} — {excerpt}",
849                    base = step_result.message,
850                    excerpt = diagnostics::inline_excerpt(&state),
851                );
852                (Some(diagnostics::full_context(&state)), screenshot)
853            } else {
854                (None, None)
855            };
856
857            let duration_ms = step_started.elapsed().as_millis() as u64;
858            self.emit_event(&TestEvent::StepFinished {
859                test: test.name.clone(),
860                index: step_index as u32,
861                label: step_result.name.clone(),
862                status: step_result.status,
863                duration_ms,
864                message: step_result.message.clone(),
865                diagnostics: diagnostics_block,
866                screenshot: screenshot_path,
867            });
868            self.current_step.replace(None);
869
870            match step_result.status {
871                StepStatus::Passed => result.passed += 1,
872                StepStatus::Failed => result.failed += 1,
873                StepStatus::Skipped => result.skipped += 1,
874            }
875
876            // Fail fast: the first failed step ends the test and the
877            // remaining steps are reported as skipped (no LLM budget is
878            // burned asserting against a page that is already known broken).
879            if step_result.status == StepStatus::Failed
880                && !self.config.continue_on_failure
881                && step_index + 1 < test.steps.len()
882            {
883                self.emit_event(&TestEvent::Warning {
884                    message: format!(
885                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
886                        test.steps.len() - step_index - 1
887                    ),
888                });
889                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
890                    let skipped_index = step_index + 1 + offset;
891                    let label = step_label(skipped);
892                    result.total += 1;
893                    result.skipped += 1;
894                    self.emit_event(&TestEvent::StepStarted {
895                        test: test.name.clone(),
896                        index: skipped_index as u32,
897                        label: label.clone(),
898                    });
899                    self.emit_event(&TestEvent::StepFinished {
900                        test: test.name.clone(),
901                        index: skipped_index as u32,
902                        label,
903                        status: StepStatus::Skipped,
904                        duration_ms: 0,
905                        message: "skipped: previous step failed".into(),
906                        diagnostics: None,
907                        screenshot: None,
908                    });
909                    result.details.push(StepResult {
910                        name: step_label(skipped),
911                        status: StepStatus::Skipped,
912                        message: "skipped: previous step failed".into(),
913                    });
914                }
915                result.details.push(step_result);
916                return result;
917            }
918
919            // Check per-test budget after each step
920            let test_usage = self.usage.current_test_snapshot();
921            let global_usage = self.usage.global_snapshot();
922            let budget_status = self.budgets.check_all(
923                &test.name,
924                &test_usage,
925                &global_usage,
926                test.budget.as_ref(),
927            );
928            match budget_status {
929                BudgetStatus::HardExceeded { message, .. } => {
930                    self.emit_event(&TestEvent::Warning {
931                        message: format!("budget exceeded: {message}"),
932                    });
933                    result.details.push(StepResult {
934                        name: "[budget]".into(),
935                        status: StepStatus::Failed,
936                        message,
937                    });
938                    result.failed += 1;
939                    return result;
940                }
941                BudgetStatus::SoftExceeded { message, .. } => {
942                    self.emit_event(&TestEvent::Warning {
943                        message: format!("budget warning: {message}"),
944                    });
945                }
946                BudgetStatus::Ok => {}
947            }
948
949            if let Some(ms) = wait_ms {
950                std::thread::sleep(Duration::from_millis(ms));
951            }
952
953            result.details.push(step_result);
954        }
955
956        result
957    }
958
959    /// Applies a viewport size to the current tab via CDP
960    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
961    /// overrides and the viewport matrix. The initial window size set at
962    /// browser launch is replaced by emulation; failures are logged but
963    /// do not fail the test (a mismatched viewport only weakens coverage).
964    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
965        use headless_chrome::protocol::cdp::Emulation;
966        let _ = self;
967        let params = Emulation::SetDeviceMetricsOverride {
968            width,
969            height,
970            device_scale_factor: 1.0,
971            mobile: false,
972            scale: None,
973            screen_width: Some(width),
974            screen_height: Some(height),
975            position_x: None,
976            position_y: None,
977            dont_set_visible_size: None,
978            screen_orientation: None,
979            viewport: None,
980            display_feature: None,
981            device_posture: None,
982        };
983        self.reporter.debug(format!("viewport: {width}x{height}"));
984        if let Err(e) = tab.call_method(params) {
985            self.reporter
986                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
987        }
988    }
989
990    /// Height (px) covered by assert-step screenshots: the configured
991    /// `screenshot_max_height` (absolute px or viewport multiple, default
992    /// `"20x"`) resolved against the currently applied viewport, raised to
993    /// at least the viewport height so the visible screen is always fully
994    /// included. The capture is split into viewport-tall tiles, so this
995    /// value bounds total coverage (and hence the number of image parts).
996    #[must_use]
997    fn screenshot_height_cap(&self) -> u32 {
998        let viewport_height = self.current_viewport_height();
999        let cap = self
1000            .config
1001            .screenshot_max_height
1002            .as_ref()
1003            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
1004        cap.max(viewport_height)
1005    }
1006
1007    /// Height of the viewport currently emulated in the browser (falling
1008    /// back to the configured default before any emulation was applied).
1009    #[must_use]
1010    const fn current_viewport_height(&self) -> u32 {
1011        let (_, height) = self.applied_viewport.get();
1012        if height > 0 {
1013            height
1014        } else {
1015            self.viewport_height
1016        }
1017    }
1018
1019    // ── step handlers ───────────────────────────────────────────────────
1020
1021    #[allow(clippy::too_many_lines)]
1022    fn run_click(
1023        &self,
1024        target: &str,
1025        selector_override: Option<&str>,
1026        step_endpoint: Option<&str>,
1027        test_endpoint: Option<&str>,
1028        idempotent: bool,
1029        tab: &Tab,
1030    ) -> StepResult {
1031        let name = format!("[click] {target}");
1032        let selector = match self.resolve_selector(
1033            selector_override,
1034            target,
1035            step_endpoint,
1036            test_endpoint,
1037            tab,
1038        ) {
1039            Ok(s) => s,
1040            Err(msg) => {
1041                if idempotent {
1042                    return StepResult {
1043                        name,
1044                        status: StepStatus::Skipped,
1045                        message: format!("skipped (idempotent): no target found — {msg}"),
1046                    };
1047                }
1048                return StepResult {
1049                    name,
1050                    status: StepStatus::Failed,
1051                    message: msg,
1052                };
1053            }
1054        };
1055
1056        // Idempotent steps probe briefly: a missing target means the
1057        // action was already done / not applicable (e.g. an
1058        // already-authenticated session), and skipping is the success
1059        // path, not a failure.
1060        let probe_secs = if idempotent { 5 } else { 10 };
1061        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1062            Ok(element) => match element.click() {
1063                Ok(_) => StepResult {
1064                    name,
1065                    status: StepStatus::Passed,
1066                    message: format!("clicked {selector}"),
1067                },
1068                Err(e) => StepResult {
1069                    name,
1070                    status: StepStatus::Failed,
1071                    message: format!("click failed on {selector}: {e}"),
1072                },
1073            },
1074            Err(e) if idempotent => StepResult {
1075                name,
1076                status: StepStatus::Skipped,
1077                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1078            },
1079            Err(e) => StepResult {
1080                name,
1081                status: StepStatus::Failed,
1082                message: format!("element {selector} not found: {e}"),
1083            },
1084        }
1085    }
1086
1087    #[allow(clippy::too_many_arguments)]
1088    fn run_type(
1089        &self,
1090        target: &str,
1091        text: &str,
1092        selector_override: Option<&str>,
1093        step_endpoint: Option<&str>,
1094        test_endpoint: Option<&str>,
1095        idempotent: bool,
1096        tab: &Tab,
1097    ) -> StepResult {
1098        let name = format!("[type] {target}");
1099        let selector = match self.resolve_selector(
1100            selector_override,
1101            target,
1102            step_endpoint,
1103            test_endpoint,
1104            tab,
1105        ) {
1106            Ok(s) => s,
1107            Err(msg) => {
1108                if idempotent {
1109                    return StepResult {
1110                        name,
1111                        status: StepStatus::Skipped,
1112                        message: format!("skipped (idempotent): no target found — {msg}"),
1113                    };
1114                }
1115                return StepResult {
1116                    name,
1117                    status: StepStatus::Failed,
1118                    message: msg,
1119                };
1120            }
1121        };
1122
1123        let probe_secs = if idempotent { 5 } else { 10 };
1124        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1125            Ok(element) => {
1126                if let Err(e) = element.click() {
1127                    return StepResult {
1128                        name,
1129                        status: StepStatus::Failed,
1130                        message: format!("click to focus {selector} failed: {e}"),
1131                    };
1132                }
1133
1134                let js = format!(
1135                    "document.querySelector('{}').value = '';",
1136                    selector.replace('\'', "\\'")
1137                );
1138                let _ = tab.evaluate(&js, false);
1139
1140                match element.type_into(text) {
1141                    Ok(_) => StepResult {
1142                        name,
1143                        status: StepStatus::Passed,
1144                        message: format!("typed {text:?} into {selector}"),
1145                    },
1146                    Err(e) => StepResult {
1147                        name,
1148                        status: StepStatus::Failed,
1149                        message: format!("type into {selector} failed: {e}"),
1150                    },
1151                }
1152            }
1153            Err(e) if idempotent => StepResult {
1154                name,
1155                status: StepStatus::Skipped,
1156                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1157            },
1158            Err(e) => StepResult {
1159                name,
1160                status: StepStatus::Failed,
1161                message: format!("element {selector} not found: {e}"),
1162            },
1163        }
1164    }
1165
1166    #[allow(clippy::too_many_arguments)]
1167    #[allow(clippy::too_many_lines)]
1168    fn run_wait(
1169        &self,
1170        target: &str,
1171        selector_override: Option<&str>,
1172        text: Option<&str>,
1173        timeout_ms: Option<u64>,
1174        step_endpoint: Option<&str>,
1175        test_endpoint: Option<&str>,
1176        idempotent: bool,
1177        tab: &Tab,
1178    ) -> StepResult {
1179        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1180        let step_name = format!("[wait] {target}");
1181
1182        // Resolve an explicit selector only (text-only waits are LLM-free).
1183        let selector = match selector_override {
1184            Some(s) => Some(s.to_owned()),
1185            None if text.is_some() => None,
1186            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1187                Ok(s) => Some(s),
1188                Err(msg) => {
1189                    if idempotent {
1190                        return StepResult {
1191                            name: step_name,
1192                            status: StepStatus::Skipped,
1193                            message: format!("skipped (idempotent): no target found — {msg}"),
1194                        };
1195                    }
1196                    return StepResult {
1197                        name: step_name,
1198                        status: StepStatus::Failed,
1199                        message: msg,
1200                    };
1201                }
1202            },
1203        };
1204
1205        if text.is_some() {
1206            let sel_js = selector
1207                .as_deref()
1208                .map(crate::selectors::selector_matches_js);
1209            let text_js = text.map(|t| {
1210                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1211                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1212            });
1213
1214            let deadline = Instant::now() + timeout;
1215            loop {
1216                let sel_ok = sel_js
1217                    .as_ref()
1218                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1219                let text_ok = text_js
1220                    .as_ref()
1221                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1222                if sel_ok && text_ok {
1223                    let mut what = Vec::new();
1224                    if let Some(sel) = &selector {
1225                        what.push(format!("found {sel}"));
1226                    }
1227                    if let Some(t) = text {
1228                        what.push(format!("text {t:?} visible"));
1229                    }
1230                    return StepResult {
1231                        name: step_name,
1232                        status: StepStatus::Passed,
1233                        message: what.join(" and "),
1234                    };
1235                }
1236                if Instant::now() >= deadline {
1237                    let mut what = Vec::new();
1238                    if let Some(sel) = &selector {
1239                        what.push(sel.clone());
1240                    }
1241                    if let Some(t) = text {
1242                        what.push(format!("text {t:?}"));
1243                    }
1244                    let message = format!(
1245                        "wait for {} timed out after {}ms: the event waited for never came",
1246                        what.join(" / "),
1247                        timeout.as_millis(),
1248                    );
1249                    if idempotent {
1250                        return StepResult {
1251                            name: step_name,
1252                            status: StepStatus::Skipped,
1253                            message: format!("skipped (idempotent): {message}"),
1254                        };
1255                    }
1256                    return StepResult {
1257                        name: step_name,
1258                        status: StepStatus::Failed,
1259                        message,
1260                    };
1261                }
1262                std::thread::sleep(Duration::from_millis(250));
1263            }
1264        }
1265
1266        match selector.as_deref() {
1267            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1268                Ok(_) => StepResult {
1269                    name: step_name,
1270                    status: StepStatus::Passed,
1271                    message: format!("found {sel}"),
1272                },
1273                Err(e) if idempotent => StepResult {
1274                    name: step_name,
1275                    status: StepStatus::Skipped,
1276                    message: format!(
1277                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1278                        timeout.as_millis()
1279                    ),
1280                },
1281                Err(e) => StepResult {
1282                    name: step_name,
1283                    status: StepStatus::Failed,
1284                    message: format!(
1285                        "wait for {sel} timed out after {}ms: {e}",
1286                        timeout.as_millis()
1287                    ),
1288                },
1289            },
1290            None => StepResult {
1291                name: step_name,
1292                status: StepStatus::Failed,
1293                message: "wait step has neither selector nor text".into(),
1294            },
1295        }
1296    }
1297
1298    #[allow(clippy::too_many_arguments)]
1299    fn run_assert(
1300        &self,
1301        definition: Option<&str>,
1302        preset: Option<&str>,
1303        prompt: Option<&str>,
1304        assert_text: Option<&str>,
1305        screenshot: bool,
1306        step_endpoint: Option<&str>,
1307        test_endpoint: Option<&str>,
1308        tab: &Tab,
1309    ) -> StepResult {
1310        std::thread::sleep(Duration::from_millis(500));
1311
1312        let page_content = get_page_text(tab);
1313
1314        // Vision attach: capture the full page once per assert step and
1315        // split it into viewport-tall tiles (the total coverage is bounded
1316        // by the configured height cap so vision tokens stay sane). All
1317        // tile data URLs are handed to the preset/prompt evaluation below.
1318        let image: Option<Vec<String>> = if screenshot {
1319            let endpoint = self
1320                .endpoints
1321                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1322            if !endpoint.vision {
1323                return StepResult {
1324                    name: "[assert]".into(),
1325                    status: StepStatus::Failed,
1326                    message: format!(
1327                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1328                        name = endpoint.name
1329                    ),
1330                };
1331            }
1332            match crate::vision::capture_screenshot_data_urls(
1333                tab,
1334                self.config
1335                    .screenshot_max_dimension
1336                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1337                self.screenshot_height_cap(),
1338                self.current_viewport_height(),
1339            ) {
1340                Ok(urls) => Some(urls),
1341                Err(e) => {
1342                    return StepResult {
1343                        name: "[assert]".into(),
1344                        status: StepStatus::Failed,
1345                        message: format!("screenshot capture failed: {e}"),
1346                    };
1347                }
1348            }
1349        } else {
1350            None
1351        };
1352
1353        if let Some(def_name) = definition {
1354            if let Some(def) = self.definitions.get(def_name) {
1355                return self.run_assert_def(
1356                    def,
1357                    &page_content,
1358                    image.as_deref(),
1359                    step_endpoint,
1360                    test_endpoint,
1361                    tab,
1362                );
1363            }
1364            return StepResult {
1365                name: format!("[assert] {def_name}"),
1366                status: StepStatus::Failed,
1367                message: format!("definition '{def_name}' not found"),
1368            };
1369        }
1370
1371        if let Some(preset_name) = preset {
1372            // Deterministic DOM layout scan — runs JS in the browser and
1373            // never calls the LLM (free, fast, no pixel budget).
1374            if preset_name == "layout_no_issues" {
1375                return self.run_layout_preset(tab);
1376            }
1377            return self.run_preset(
1378                preset_name,
1379                assert_text,
1380                &page_content,
1381                image.as_deref(),
1382                step_endpoint,
1383                test_endpoint,
1384            );
1385        }
1386
1387        if let Some(prompt_text) = prompt {
1388            return self.run_custom(
1389                prompt_text,
1390                &page_content,
1391                image.as_deref(),
1392                step_endpoint,
1393                test_endpoint,
1394            );
1395        }
1396
1397        StepResult {
1398            name: "[assert]".into(),
1399            status: StepStatus::Skipped,
1400            message: "no definition, preset, or prompt specified".into(),
1401        }
1402    }
1403
1404    fn run_assert_def(
1405        &self,
1406        def: &AssertDefinition,
1407        page_content: &PageContent,
1408        image: Option<&[String]>,
1409        step_endpoint: Option<&str>,
1410        test_endpoint: Option<&str>,
1411        tab: &Tab,
1412    ) -> StepResult {
1413        // Agent-based definition: delegate to an A2A agent
1414        if let Some(ref agent) = def.agent {
1415            if image.is_some() {
1416                return StepResult {
1417                    name: format!("[assert] {}", def.name),
1418                    status: StepStatus::Failed,
1419                    message: "agent-backed assertions do not support screenshots".into(),
1420                };
1421            }
1422            let task = def
1423                .task_template
1424                .as_deref()
1425                .unwrap_or("Evaluate the assertion")
1426                .replace("{url}", &page_content.url)
1427                .replace("{title}", &page_content.title)
1428                .replace("{content}", &page_content.body_text)
1429                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1430
1431            return self.run_agent_step(agent, &task, &def.name);
1432        }
1433
1434        // Custom preset: system + user_template provided in the definition
1435        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1436            return self.run_custom_preset(
1437                &def.name,
1438                system,
1439                template,
1440                def.assert_text.as_deref(),
1441                page_content,
1442                image,
1443                step_endpoint,
1444                test_endpoint,
1445            );
1446        }
1447
1448        def.preset.as_ref().map_or_else(
1449            || {
1450                def.prompt.as_ref().map_or_else(
1451                    || StepResult {
1452                        name: format!("[assert] {}", def.name),
1453                        status: StepStatus::Failed,
1454                        message: "definition has no preset, prompt, or system+user_template".into(),
1455                    },
1456                    |prompt| {
1457                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1458                    },
1459                )
1460            },
1461            |preset_name| {
1462                if preset_name == "layout_no_issues" {
1463                    return self.run_layout_preset(tab);
1464                }
1465                self.run_preset(
1466                    preset_name,
1467                    def.assert_text.as_deref(),
1468                    page_content,
1469                    image,
1470                    step_endpoint,
1471                    test_endpoint,
1472                )
1473            },
1474        )
1475    }
1476
1477    #[allow(clippy::too_many_arguments)]
1478    fn run_custom_preset(
1479        &self,
1480        name: &str,
1481        system: &str,
1482        template: &str,
1483        assert_text: Option<&str>,
1484        page_content: &PageContent,
1485        image: Option<&[String]>,
1486        step_endpoint: Option<&str>,
1487        test_endpoint: Option<&str>,
1488    ) -> StepResult {
1489        let user_prompt = template
1490            .replace("{url}", &page_content.url)
1491            .replace("{title}", &page_content.title)
1492            .replace("{content}", &page_content.body_text)
1493            .replace("{expected_text}", assert_text.unwrap_or(""))
1494            .replace("{description}", "");
1495
1496        // Custom preset definitions frequently forget the {content}
1497        // placeholder — without it the LLM has no page to evaluate and
1498        // answers "I can't determine that without seeing the page". Always
1499        // append the page context unless the template already references it.
1500        let user_prompt = if template.contains("{content}") {
1501            user_prompt
1502        } else {
1503            format!(
1504                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1505                url = page_content.url,
1506                title = page_content.title,
1507                content = page_content.body_text,
1508            )
1509        };
1510
1511        self.reporter
1512            .debug(format!("assert: {name} (custom preset)"));
1513
1514        let chain = self
1515            .endpoints
1516            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1517        let sys = system.to_owned();
1518
1519        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1520
1521        response.map_or_else(
1522            |e| StepResult {
1523                name: format!("[assert] {name}"),
1524                status: StepStatus::Failed,
1525                message: format!("LLM assertion call failed: {e}"),
1526            },
1527            |(lr, _idx)| {
1528                if verdict_is_pass(&lr.content) {
1529                    StepResult {
1530                        name: format!("[assert] {name}"),
1531                        status: StepStatus::Passed,
1532                        message: "PASS".into(),
1533                    }
1534                } else {
1535                    StepResult {
1536                        name: format!("[assert] {name}"),
1537                        status: StepStatus::Failed,
1538                        message: lr.content,
1539                    }
1540                }
1541            },
1542        )
1543    }
1544
1545    fn run_preset(
1546        &self,
1547        preset_name: &str,
1548        assert_text: Option<&str>,
1549        page_content: &PageContent,
1550        image: Option<&[String]>,
1551        step_endpoint: Option<&str>,
1552        test_endpoint: Option<&str>,
1553    ) -> StepResult {
1554        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1555            return StepResult {
1556                name: format!("[assert] {preset_name}"),
1557                status: StepStatus::Failed,
1558                message: format!("unknown assertion preset: {preset_name}"),
1559            };
1560        };
1561        if preset_name.starts_with("visual_") && image.is_none() {
1562            return StepResult {
1563                name: format!("[assert] {preset_name}"),
1564                status: StepStatus::Failed,
1565                message: format!(
1566                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1567                ),
1568            };
1569        }
1570
1571        let user_prompt = preset
1572            .user_template
1573            .replace("{url}", &page_content.url)
1574            .replace("{title}", &page_content.title)
1575            .replace("{content}", &page_content.body_text)
1576            .replace("{expected_text}", assert_text.unwrap_or(""))
1577            .replace("{description}", "");
1578
1579        // Same safety net as custom presets: never let the LLM answer with
1580        // no page context at all.
1581        let user_prompt = if preset.user_template.contains("{content}") {
1582            user_prompt
1583        } else {
1584            format!(
1585                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1586                url = page_content.url,
1587                title = page_content.title,
1588                content = page_content.body_text,
1589            )
1590        };
1591
1592        self.reporter.debug(format!("assert: {preset_name}"));
1593
1594        let chain = self
1595            .endpoints
1596            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1597        let sys = preset.system.to_owned();
1598
1599        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1600
1601        response.map_or_else(
1602            |e| StepResult {
1603                name: format!("[assert] {preset_name}"),
1604                status: StepStatus::Failed,
1605                message: format!("LLM assertion call failed: {e}"),
1606            },
1607            |(lr, _idx)| {
1608                if verdict_is_pass(&lr.content) {
1609                    StepResult {
1610                        name: format!("[assert] {preset_name}"),
1611                        status: StepStatus::Passed,
1612                        message: "PASS".into(),
1613                    }
1614                } else {
1615                    StepResult {
1616                        name: format!("[assert] {preset_name}"),
1617                        status: StepStatus::Failed,
1618                        message: lr.content,
1619                    }
1620                }
1621            },
1622        )
1623    }
1624
1625    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1626    ///
1627    /// Evaluates the layout-scan JS in the page and fails with the list of
1628    /// detected issues: horizontal page overflow, visible elements sticking
1629    /// out of the viewport, text clipped by `overflow: hidden` containers,
1630    /// and interactive elements covered by other elements. No LLM call —
1631    /// checks are geometry-based so the check is free, deterministic, and
1632    /// safe to run on every page × viewport variant.
1633    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1634        let name = "[assert] layout_no_issues".to_owned();
1635        self.reporter
1636            .debug("assert: layout_no_issues (DOM layout scan)");
1637        let js = LAYOUT_SCAN_JS.replace(
1638            "__IGNORE_CLASSES__",
1639            &serde_json::to_string(&self.config.layout_ignore_classes)
1640                .unwrap_or_else(|_| "[]".to_owned()),
1641        );
1642        let result = tab.evaluate(&js, false);
1643        let json_str = match result {
1644            Ok(r) => r
1645                .value
1646                .as_ref()
1647                .and_then(|v| v.as_str().map(String::from))
1648                .unwrap_or_else(|| "[]".to_owned()),
1649            Err(e) => {
1650                return StepResult {
1651                    name,
1652                    status: StepStatus::Failed,
1653                    message: format!("layout scan JS failed: {e}"),
1654                };
1655            }
1656        };
1657        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1658        if issues.is_empty() {
1659            return StepResult {
1660                name,
1661                status: StepStatus::Passed,
1662                message: "PASS — no layout defects detected".into(),
1663            };
1664        }
1665        let mut lines: Vec<String> = issues
1666            .iter()
1667            .take(10)
1668            .map(|i| {
1669                format!(
1670                    "- [{type_}] {element}: {detail}",
1671                    type_ = i.issue_type,
1672                    element = i.element,
1673                    detail = i.detail
1674                )
1675            })
1676            .collect();
1677        if issues.len() > 10 {
1678            lines.push(format!("- … and {} more", issues.len() - 10));
1679        }
1680        StepResult {
1681            name,
1682            status: StepStatus::Failed,
1683            message: format!(
1684                "FAIL — {} layout defect(s) detected:\n{}",
1685                issues.len(),
1686                lines.join("\n")
1687            ),
1688        }
1689    }
1690
1691    fn run_custom(
1692        &self,
1693        prompt: &str,
1694        page_content: &PageContent,
1695        image: Option<&[String]>,
1696        step_endpoint: Option<&str>,
1697        test_endpoint: Option<&str>,
1698    ) -> StepResult {
1699        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1700
1701        let mut user = format!(
1702            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1703            url = page_content.url,
1704            title = page_content.title,
1705            content = page_content.body_text,
1706        );
1707        if image.is_some() {
1708            user.push_str(
1709                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1710            );
1711        }
1712
1713        self.reporter.debug("custom assert");
1714
1715        let chain = self
1716            .endpoints
1717            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1718        let sys = system.to_owned();
1719
1720        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1721
1722        response.map_or_else(
1723            |e| StepResult {
1724                name: "[assert] custom".into(),
1725                status: StepStatus::Failed,
1726                message: format!("LLM assertion call failed: {e}"),
1727            },
1728            |(lr, _idx)| {
1729                if verdict_is_pass(&lr.content) {
1730                    StepResult {
1731                        name: "[assert] custom".into(),
1732                        status: StepStatus::Passed,
1733                        message: "PASS".into(),
1734                    }
1735                } else {
1736                    StepResult {
1737                        name: "[assert] custom".into(),
1738                        status: StepStatus::Failed,
1739                        message: lr.content,
1740                    }
1741                }
1742            },
1743        )
1744    }
1745
1746    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1747        let path = path.unwrap_or("screenshot.png");
1748
1749        match tab.capture_screenshot(
1750            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1751            None,
1752            None,
1753            true,
1754        ) {
1755            Ok(data) => {
1756                if let Err(e) = std::fs::write(path, &data) {
1757                    return StepResult {
1758                        name: format!("[screenshot] {path}"),
1759                        status: StepStatus::Failed,
1760                        message: format!("failed to write screenshot: {e}"),
1761                    };
1762                }
1763                StepResult {
1764                    name: format!("[screenshot] {path}"),
1765                    status: StepStatus::Passed,
1766                    message: format!("saved to {path}"),
1767                }
1768            }
1769            Err(e) => StepResult {
1770                name: format!("[screenshot] {path}"),
1771                status: StepStatus::Failed,
1772                message: format!("screenshot failed: {e}"),
1773            },
1774        }
1775    }
1776
1777    /// Runs an A2A agent step.
1778    #[allow(clippy::literal_string_with_formatting_args)]
1779    fn run_agent(
1780        &self,
1781        agent_name: &str,
1782        task: &str,
1783        definition: Option<&str>,
1784        _test_endpoint: Option<&str>,
1785    ) -> StepResult {
1786        // If a definition is specified, look up the task template
1787        let resolved_task = if let Some(def_name) = definition {
1788            if let Some(def) = self.definitions.get(def_name) {
1789                let tmpl = def.task_template.as_deref().unwrap_or(task);
1790                tmpl.replace("{task}", task)
1791            } else {
1792                return StepResult {
1793                    name: format!("[agent] {def_name}"),
1794                    status: StepStatus::Failed,
1795                    message: format!("definition '{def_name}' not found"),
1796                };
1797            }
1798        } else {
1799            task.to_owned()
1800        };
1801
1802        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1803    }
1804
1805    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1806        let Some(ep) = self.endpoints.get(agent_name) else {
1807            return StepResult {
1808                name: format!("[agent] {display_name}"),
1809                status: StepStatus::Failed,
1810                message: format!("agent endpoint '{agent_name}' not found"),
1811            };
1812        };
1813
1814        if ep.url.is_empty() {
1815            return StepResult {
1816                name: format!("[agent] {display_name}"),
1817                status: StepStatus::Failed,
1818                message: format!("agent endpoint '{agent_name}' has no URL"),
1819            };
1820        }
1821
1822        self.reporter.debug(format!("agent {agent_name}: {task}"));
1823
1824        let url = ep.url.clone();
1825        let client = A2aClient::new(&url, self.timeout);
1826        let task_clone = task.to_owned();
1827
1828        let response = std::thread::spawn(move || {
1829            let rt = tokio::runtime::Builder::new_current_thread()
1830                .enable_all()
1831                .build()
1832                .unwrap();
1833            rt.block_on(client.send_task(&task_clone))
1834        })
1835        .join()
1836        .unwrap();
1837
1838        // Record the flat-cost call
1839        self.usage.record_flat_call(agent_name, ep);
1840
1841        match response {
1842            Ok(text) => {
1843                let clean = text.trim().to_owned();
1844                if verdict_is_pass(&clean) {
1845                    StepResult {
1846                        name: format!("[agent] {display_name}"),
1847                        status: StepStatus::Passed,
1848                        message: format!("PASS: {clean}"),
1849                    }
1850                } else if verdict_is_fail(&clean) {
1851                    StepResult {
1852                        name: format!("[agent] {display_name}"),
1853                        status: StepStatus::Failed,
1854                        message: clean,
1855                    }
1856                } else {
1857                    StepResult {
1858                        name: format!("[agent] {display_name}"),
1859                        status: StepStatus::Passed,
1860                        message: format!("response: {clean}"),
1861                    }
1862                }
1863            }
1864            Err(e) => StepResult {
1865                name: format!("[agent] {display_name}"),
1866                status: StepStatus::Failed,
1867                message: format!("agent call failed: {e}"),
1868            },
1869        }
1870    }
1871
1872    /// Runs an MCP tool call step.
1873    fn run_mcp(
1874        &self,
1875        server_name: &str,
1876        tool_name: &str,
1877        args: Option<&serde_json::Value>,
1878    ) -> StepResult {
1879        let Some(ep) = self.endpoints.get(server_name) else {
1880            return StepResult {
1881                name: format!("[mcp] {server_name}:{tool_name}"),
1882                status: StepStatus::Failed,
1883                message: format!("MCP server endpoint '{server_name}' not found"),
1884            };
1885        };
1886
1887        let cmd = ep.command.as_deref().unwrap_or("");
1888        if cmd.is_empty() {
1889            return StepResult {
1890                name: format!("[mcp] {server_name}:{tool_name}"),
1891                status: StepStatus::Failed,
1892                message: format!("MCP server '{server_name}' has no command configured"),
1893            };
1894        }
1895
1896        self.reporter
1897            .debug(format!("mcp {server_name} {tool_name}"));
1898
1899        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1900
1901        let command = cmd.to_owned();
1902        let args_vec = ep.args.clone();
1903        let tool = tool_name.to_owned();
1904
1905        let response = std::thread::spawn(move || {
1906            let mut mcp_client =
1907                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1908            mcp_client
1909                .call_tool(&tool, &args_val)
1910                .map_err(|e| e.to_string())
1911        })
1912        .join()
1913        .unwrap();
1914
1915        // Record the flat-cost call
1916        self.usage.record_flat_call(server_name, ep);
1917
1918        match response {
1919            Ok(result) => {
1920                if result.isError {
1921                    StepResult {
1922                        name: format!("[mcp] {server_name}:{tool_name}"),
1923                        status: StepStatus::Failed,
1924                        message: result.to_string(),
1925                    }
1926                } else {
1927                    StepResult {
1928                        name: format!("[mcp] {server_name}:{tool_name}"),
1929                        status: StepStatus::Passed,
1930                        message: result.to_string(),
1931                    }
1932                }
1933            }
1934            Err(e) => StepResult {
1935                name: format!("[mcp] {server_name}:{tool_name}"),
1936                status: StepStatus::Failed,
1937                message: format!("MCP call failed: {e}"),
1938            },
1939        }
1940    }
1941
1942    // ── helpers ──────────────────────────────────────────────────────────
1943
1944    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1945    /// the runner's default LLM config for any unset fields.
1946    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1947        LlmConfig {
1948            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
1949                // Bedrock builds its endpoint from the resolved AWS region
1950                // when no URL is given — never inherit the default LLM URL.
1951                endpoint.url.clone()
1952            } else if endpoint.url.is_empty() {
1953                self.llm.url.clone()
1954            } else {
1955                endpoint.url.clone()
1956            },
1957            model: endpoint
1958                .model
1959                .clone()
1960                .unwrap_or_else(|| self.llm.model.clone()),
1961            api_key: endpoint
1962                .api_key
1963                .clone()
1964                .or_else(|| self.llm.api_key.clone()),
1965            headers: if endpoint.headers.is_empty() {
1966                self.llm.headers.clone()
1967            } else {
1968                endpoint.headers.clone()
1969            },
1970            timeout: self.llm.timeout,
1971            temperature: self.llm.temperature,
1972            thinking: self.llm.thinking,
1973            model_params: self.llm.model_params.clone(),
1974            cache: endpoint.cache_markers,
1975            max_attempts: endpoint.max_attempts.max(1),
1976            provider: endpoint.provider,
1977            deployment: endpoint.deployment.clone(),
1978            api_version: endpoint.api_version.clone(),
1979            auth: endpoint.auth.clone(),
1980            header_commands: endpoint.header_commands.clone(),
1981            aws: endpoint.aws.clone(),
1982        }
1983    }
1984
1985    /// Runs a single LLM call against an ordered endpoint chain (primary +
1986    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
1987    /// the first endpoint that answers wins. Returns the response together
1988    /// with the chain index of the answering endpoint (0 = primary) so the
1989    /// caller can attribute usage to the correct endpoint.
1990    ///
1991    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
1992    /// shows duration, tokens, cost and the answering endpoint per call.
1993    /// Run-level context handed to every LLM call as the FIRST block of the
1994    /// user message. Contents that are stable for the whole run ("run
1995    /// started", "target site") come first so upstream provider prefix
1996    /// caching stays effective; the current time is the last line because
1997    /// it changes on every call.
1998    fn run_context(&self) -> String {
1999        let mut parts = vec![
2000            "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
2001            "================================================================".into(),
2002            format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
2003        ];
2004        if let Some(base) = self.config.base_url.as_deref() {
2005            parts.push(format!("Target site: {base}"));
2006        }
2007        let now = SystemTime::now()
2008            .duration_since(UNIX_EPOCH)
2009            .map_or(0, |d| d.as_secs());
2010        parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
2011        parts.join("\n")
2012    }
2013
2014    #[allow(clippy::cast_possible_truncation, clippy::too_many_lines)]
2015    fn llm_call_chain(
2016        &self,
2017        chain: &[&ResolvedEndpoint],
2018        system: &str,
2019        user: &str,
2020        image: Option<&[String]>,
2021        purpose: &str,
2022    ) -> Result<(crate::costs::LlmResponse, usize), String> {
2023        if chain.is_empty() {
2024            return Err("empty LLM endpoint chain".into());
2025        }
2026        let primary = self.build_llm_for_endpoint(chain[0]);
2027        let fallbacks: Vec<LlmConfig> = chain[1..]
2028            .iter()
2029            .map(|e| self.build_llm_for_endpoint(e))
2030            .collect();
2031
2032        let (test, index) = self
2033            .current_step
2034            .borrow()
2035            .as_ref()
2036            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
2037        let primary_endpoint = chain[0].name.clone();
2038        let primary_model = primary.model.clone();
2039        self.emit_event(&TestEvent::LlmCallStarted {
2040            test: test.clone(),
2041            index,
2042            endpoint: primary_endpoint.clone(),
2043            model: primary_model.clone(),
2044            purpose: purpose.to_owned(),
2045        });
2046
2047        let started = Instant::now();
2048        let sys = system.to_owned();
2049        let context = self.run_context();
2050        let user = if context.is_empty() {
2051            user.to_owned()
2052        } else {
2053            format!("{context}\n\n{user}")
2054        };
2055        let image = image.map(<[String]>::to_vec);
2056
2057        let result = std::thread::spawn(move || {
2058            let rt = tokio::runtime::Builder::new_current_thread()
2059                .enable_all()
2060                .build()
2061                .unwrap();
2062            let call = async {
2063                match image.as_deref() {
2064                    Some(img) => {
2065                        llm_chat_vision_with_usage_chain(
2066                            &primary,
2067                            &fallbacks,
2068                            &sys,
2069                            &user,
2070                            Some(img),
2071                        )
2072                        .await
2073                    }
2074                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2075                }
2076            };
2077            rt.block_on(call)
2078        })
2079        .join()
2080        .unwrap();
2081
2082        let duration_ms = started.elapsed().as_millis() as u64;
2083        match result {
2084            Ok((lr, idx)) => {
2085                let cost = calculate_llm_cost(chain[idx], &lr.usage);
2086                let answering = chain[idx].name.clone();
2087                let model = chain[idx]
2088                    .model
2089                    .clone()
2090                    .unwrap_or_else(|| primary_model.clone());
2091                self.usage
2092                    .record_llm_call(&answering, chain[idx], &model, &lr.usage);
2093                self.emit_event(&TestEvent::LlmCallFinished {
2094                    test,
2095                    index,
2096                    endpoint: answering,
2097                    model,
2098                    purpose: purpose.to_owned(),
2099                    ok: true,
2100                    duration_ms,
2101                    input_tokens: lr.usage.prompt_tokens,
2102                    output_tokens: lr.usage.completion_tokens,
2103                    cached_input_tokens: lr.usage.cached_input_tokens,
2104                    cache_creation_input_tokens: lr.usage.cache_creation_input_tokens,
2105                    cost,
2106                    error: None,
2107                });
2108                Ok((lr, idx))
2109            }
2110            Err(e) => {
2111                self.emit_event(&TestEvent::LlmCallFinished {
2112                    test,
2113                    index,
2114                    endpoint: primary_endpoint,
2115                    model: primary_model,
2116                    purpose: purpose.to_owned(),
2117                    ok: false,
2118                    duration_ms,
2119                    input_tokens: 0,
2120                    output_tokens: 0,
2121                    cached_input_tokens: 0,
2122                    cache_creation_input_tokens: 0,
2123                    cost: 0.0,
2124                    error: Some(e.clone()),
2125                });
2126                Err(e)
2127            }
2128        }
2129    }
2130
2131    /// Resolves a CSS selector for the target element. Uses the explicit
2132    /// `selector` if provided, otherwise asks the LLM to find the element
2133    /// from the natural language `target` description and page DOM.
2134    ///
2135    /// LLM responses are sanitized and verified against the live page: a
2136    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2137    /// immediately with the raw LLM output, and a selector that matches
2138    /// nothing triggers one retry with feedback before failing.
2139    #[allow(clippy::too_many_lines)]
2140    fn resolve_selector(
2141        &self,
2142        css_override: Option<&str>,
2143        target: &str,
2144        step_endpoint: Option<&str>,
2145        test_endpoint: Option<&str>,
2146        tab: &Tab,
2147    ) -> Result<String, String> {
2148        if let Some(explicit) = css_override {
2149            return Ok(explicit.to_owned());
2150        }
2151
2152        let dom_info = extract_dom_info(tab)?;
2153        let page_content = get_page_text(tab);
2154
2155        let system = concat!(
2156            "You are a browser automation selector generator. ",
2157            "Given a web page's content and interactive elements, ",
2158            "return ONLY the best CSS selector for the described element. ",
2159            "Output nothing except the CSS selector. ",
2160            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2161            "[name=\"...\"], tag.class, tag. ",
2162            "Never output explanations, markdown, or extra text."
2163        );
2164
2165        let user = format!(
2166            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2167            page_content.url,
2168            page_content.title,
2169            truncate(&page_content.body_text, 4000),
2170            dom_info,
2171            target,
2172        );
2173
2174        let retry_user = format!(
2175            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2176            "The selector must match at least one element currently present on the page.",
2177            page_content.url,
2178            page_content.title,
2179            truncate(&page_content.body_text, 4000),
2180            dom_info,
2181            target,
2182        );
2183
2184        self.reporter.debug(format!("LLM targeting: {target}"));
2185
2186        let chain = self
2187            .endpoints
2188            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2189        let sys = system.to_owned();
2190
2191        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2192
2193        let first = call_llm(&user);
2194        let (lr, _idx) = match first {
2195            Ok(lr) => lr,
2196            Err(e) => {
2197                return Err(format!("LLM element targeting failed: {e}"));
2198            }
2199        };
2200        let clean = sanitize_selector(&lr.content);
2201        self.reporter.debug(format!("resolved selector: {clean}"));
2202
2203        if selector_is_useless(&clean) {
2204            return Err(format!(
2205                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2206                raw = lr.content.trim(),
2207            ));
2208        }
2209        if let Err(reason) = validate_selector(&clean) {
2210            return Err(format!(
2211                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2212                raw = lr.content.trim(),
2213            ));
2214        }
2215        if !selector_matches(tab, &clean).unwrap_or(false) {
2216            // One retry with feedback: flaky models occasionally invent a
2217            // selector that does not exist on the page.
2218            self.reporter.warn(format!(
2219                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2220            ));
2221            let second = call_llm(&retry_user);
2222            let (lr2, _idx2) = match second {
2223                Ok(lr2) => lr2,
2224                Err(e) => {
2225                    return Err(format!(
2226                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2227                    ));
2228                }
2229            };
2230            let clean2 = sanitize_selector(&lr2.content);
2231            self.reporter
2232                .debug(format!("resolved selector (retry): {clean2}"));
2233            if selector_is_useless(&clean2) {
2234                return Err(format!(
2235                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2236                    raw = lr2.content.trim(),
2237                    excerpt = truncate(&page_content.body_text, 300),
2238                ));
2239            }
2240            if !selector_matches(tab, &clean2).unwrap_or(false) {
2241                return Err(format!(
2242                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2243                ));
2244            }
2245            return Ok(clean2);
2246        }
2247
2248        Ok(clean)
2249    }
2250}
2251
2252/// Evaluates a JS expression that is expected to return a boolean.
2253fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2254    tab.evaluate(js, false)
2255        .map_err(|e| format!("evaluate failed: {e}"))?
2256        .value
2257        .and_then(|v| v.as_bool())
2258        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2259}
2260
2261/// Checks whether a CSS selector matches at least one current element.
2262fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2263    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2264}
2265
2266// ── Free helper functions ──────────────────────────────────────────────
2267
2268fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2269    let name = format!("[navigate] {full_url}");
2270    match tab.navigate_to(full_url) {
2271        Ok(_) => {
2272            let _ = tab.wait_until_navigated();
2273            StepResult {
2274                name,
2275                status: StepStatus::Passed,
2276                message: format!("navigated to {full_url}"),
2277            }
2278        }
2279        Err(e) => StepResult {
2280            name,
2281            status: StepStatus::Failed,
2282            message: format!("navigation failed: {e}"),
2283        },
2284    }
2285}
2286
2287fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2288    let result = tab
2289        .evaluate(DOM_EXTRACT_JS, false)
2290        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2291
2292    let json_str = result
2293        .value
2294        .as_ref()
2295        .and_then(|v| v.as_str())
2296        .unwrap_or("[]");
2297
2298    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2299
2300    if elements.is_empty() {
2301        return Ok("(no interactive elements found)".to_owned());
2302    }
2303
2304    Ok(elements.join("\n"))
2305}
2306
2307fn get_page_text(tab: &Tab) -> PageContent {
2308    let url = tab.get_url();
2309
2310    let title = tab
2311        .evaluate("document.title", false)
2312        .ok()
2313        .and_then(|r| r.value)
2314        .and_then(|v| v.as_str().map(String::from))
2315        .unwrap_or_else(|| "unknown".to_owned());
2316
2317    let body_text = tab
2318        .evaluate(
2319            "document.body ? document.body.innerText : document.documentElement.innerText",
2320            false,
2321        )
2322        .ok()
2323        .and_then(|r| r.value)
2324        .and_then(|v| v.as_str().map(String::from))
2325        .unwrap_or_default();
2326
2327    PageContent {
2328        url,
2329        title,
2330        body_text: truncate(&body_text, 8000),
2331    }
2332}
2333
2334fn resolve_url(url: &str, base_url: &str) -> String {
2335    if url.starts_with("http://") || url.starts_with("https://") {
2336        return url.to_owned();
2337    }
2338    let base = base_url.trim_end_matches('/');
2339    if url.starts_with('/') {
2340        format!("{base}{url}")
2341    } else {
2342        format!("{base}/{url}")
2343    }
2344}
2345
2346/// Human-readable label for a step, used when steps are skipped after an
2347/// earlier failure.
2348fn step_label(step: &TestStep) -> String {
2349    match step {
2350        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2351        TestStep::Click { target, .. } => format!("[click] {target}"),
2352        TestStep::Type { target, .. } => format!("[type] {target}"),
2353        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2354        TestStep::Assert {
2355            definition,
2356            preset,
2357            prompt,
2358            ..
2359        } => definition.as_ref().map_or_else(
2360            || {
2361                preset.as_ref().map_or_else(
2362                    || {
2363                        prompt.as_ref().map_or_else(
2364                            || "[assert]".to_owned(),
2365                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2366                        )
2367                    },
2368                    |p| format!("[assert] {p}"),
2369                )
2370            },
2371            |d| format!("[assert] {d}"),
2372        ),
2373        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2374        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2375        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2376    }
2377}
2378
2379/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2380#[must_use]
2381const fn step_kind_label(step: &TestStep) -> &'static str {
2382    match step {
2383        TestStep::Navigate { .. } => "navigate",
2384        TestStep::Click { .. } => "click",
2385        TestStep::Type { .. } => "type",
2386        TestStep::Wait { .. } => "wait",
2387        TestStep::Assert { .. } => "assert",
2388        TestStep::Screenshot { .. } => "screenshot",
2389        TestStep::Agent { .. } => "agent",
2390        TestStep::Mcp { .. } => "mcp",
2391    }
2392}
2393
2394// ── Support types ──────────────────────────────────────────────────────
2395
2396#[derive(Default)]
2397struct TestRunResult {
2398    passed: u32,
2399    failed: u32,
2400    skipped: u32,
2401    total: u32,
2402    details: Vec<StepResult>,
2403}
2404
2405struct PageContent {
2406    url: String,
2407    title: String,
2408    body_text: String,
2409}
2410
2411#[cfg(test)]
2412mod tests {
2413    use super::unix_to_rfc3339;
2414
2415    #[test]
2416    fn rfc3339_epoch_and_reference_dates() {
2417        assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2418        assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2419        assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2420        assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2421    }
2422
2423    #[test]
2424    fn rfc3339_handles_leap_years() {
2425        assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2426        assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2427    }
2428}
2429
2430#[cfg(test)]
2431mod verdict_parse_tests {
2432    use super::{verdict_is_fail, verdict_is_pass};
2433
2434    #[test]
2435    fn tolerates_markdown_punctuation_and_natural_language() {
2436        assert!(verdict_is_pass("PASS"));
2437        assert!(verdict_is_pass("pass"));
2438        assert!(verdict_is_pass("**PASS**\n\nThe page is clean."));
2439        assert!(verdict_is_pass("passes - no explicit error visible"));
2440        assert!(verdict_is_pass("  \"pass\""));
2441        assert!(!verdict_is_pass("FAIL: something broke"));
2442        assert!(!verdict_is_pass("**FAIL** broken"));
2443
2444        assert!(verdict_is_fail("**FAIL** broken"));
2445        assert!(verdict_is_fail("fails - error toast shown"));
2446        assert!(!verdict_is_fail("passes - ok"));
2447    }
2448}