Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// Formats a UNIX epoch timestamp as an RFC 3339 UTC string
26/// (`2026-09-19T16:04:00Z`) without pulling in a date dependency.
27/// Civil-from-days conversion after Howard Hinnant.
28#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31    let secs = i64::try_from(secs).unwrap_or(0);
32    let days = secs.div_euclid(86_400);
33    let rem = secs.rem_euclid(86_400);
34    let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35    let z = days + 719_468;
36    let era = z.div_euclid(146_097);
37    let doe = z.rem_euclid(146_097);
38    let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40    let mp = (5 * doy + 2) / 153;
41    let d = doy - (153 * mp + 2) / 5 + 1;
42    let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43    let y = yoe + era * 400 + i64::from(mth <= 2);
44    format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47/// One detected layout defect (`layout_no_issues` preset).
48#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50    #[serde(rename = "type")]
51    issue_type: String,
52    element: String,
53    detail: String,
54}
55
56/// In-browser DOM layout scan for `layout_no_issues`.
57///
58/// Geometry-only checks (no LLM, no pixels):
59/// 1. page horizontal overflow (`scrollWidth` > viewport width);
60/// 2. elements outside the viewport that scrolling cannot reveal
61///    (fixed elements off-screen, left/negative overflow, right-edge
62///    overflow beyond the horizontally scrollable area, and bottom
63///    overflow on a page that cannot scroll down) — below-the-fold
64///    content on a tall scrollable page is normal flow, NOT a defect;
65/// 3. text clipped by `overflow: hidden` containers whose content
66///    is measurably larger than the box;
67/// 4. interactive elements (buttons/links/inputs) whose center point
68///    is covered by a different element that would intercept the click.
69///
70/// Elements whose class matches a configured ignore prefix (default:
71/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
72/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
73/// skipped — they are intentionally 1x1 / off-screen. The prefix list
74/// is injected at `__IGNORE_CLASSES__` from
75/// [`ScenarioConfig::layout_ignore_classes`].
76///
77/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
78/// sticky headers) is excluded by the position/relation filters.
79const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81  const issues = [];
82  const push = (type, el, detail) => {
83    if (issues.length >= 30) return;
84    let element = el.tagName.toLowerCase();
85    if (el.id) element += '#' + el.id;
86    else if (typeof el.className === 'string' && el.className.trim())
87      element += '.' + el.className.trim().split(/\s+/).join('.');
88    issues.push({ type, element, detail: String(detail).slice(0, 220) });
89  };
90  const vw = document.documentElement.clientWidth || window.innerWidth;
91  const vh = document.documentElement.clientHeight || window.innerHeight;
92  if (!vw || !vh) return JSON.stringify(issues);
93  const de = document.documentElement;
94  // 1. Page-level horizontal overflow.
95  if (maxSW > vw + 2)
96    push('page-overflow-x', de,
97      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98  // Class-name prefixes to skip (injected; default = Angular CDK
99  // screen-reader helpers, which are intentionally 1x1 / off-screen).
100  const ignorePrefixes = __IGNORE_CLASSES__;
101  const isIgnored = (el) => {
102    if (typeof el.className !== 'string' || !el.className.trim()) return false;
103    const classes = el.className.trim().split(/\s+/);
104    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105  };
106  const body = document.body;
107  // The document element is not always the scroll container: the app may
108  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110  // "not scrollable" — measure the scrollable content height/width of the
111  // document element AND the body, and take the max.
112  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114  const vScrollable = maxSH > vh + 2;
115  const hScrollable = maxSW > vw + 2;
116  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117  // this element into view on the given axis — i.e. the element is inside a
118  // scrolling region, so lying beyond the viewport cut is not a defect.
119  const reachableByScroller = (el, axis) => {
120    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123    let p = el.parentElement;
124    while (p && p !== body) {
125      const pcs = getComputedStyle(p);
126      const o = pcs[ovProp];
127      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128          p[dim] > p[dimClient] + 2) return true;
129      p = p.parentElement;
130    }
131    return false;
132  };
133  const selfOverflowing = (el, cs, axis) => {
134    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138      el[dim] > el[dimClient] + 2;
139  };
140  // True when an ancestor clips this axis with overflow:hidden/clip — the
141  // element's overhang is not visible, so treat it as reachable.
142  const clippedByAncestor = (el, axis) => {
143    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144    let p = el.parentElement;
145    while (p && p !== body) {
146      const o = getComputedStyle(p)[ovProp];
147      if (o === 'hidden' || o === 'clip') return true;
148      p = p.parentElement;
149    }
150    return false;
151  };
152  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154  const hasContent = (el) =>
155    ((el.textContent || '').trim().length > 0) ||
156    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157  // 2. Elements outside the viewport that scrolling cannot reveal.
158  for (const el of all) {
159    const cs = getComputedStyle(el);
160    if (!visible(cs)) continue;
161    if (isIgnored(el)) continue;
162    const r = el.getBoundingClientRect();
163    if (r.width < 2 || r.height < 2) continue;
164    if (!hasContent(el) && el.children.length === 0) continue;
165    if (cs.position === 'fixed') {
166      // Fixed elements never move with the scroll: any edge outside the
167      // viewport is unreachable content and therefore a defect.
168      const overTop = -r.top;
169      const overLeft = -r.left;
170      const overRight = r.right - vw;
171      const overBottom = r.bottom - vh;
172      // Top/left overshoot is never reachable; bottom/right overshoot is
173      // only a defect when the fixed element cannot scroll that content
174      // into view itself (e.g. an opened Material drawer whose inner
175      // container scrolls is not a "cut-off" defect).
176      if (overTop > 2 || overLeft > 2 ||
177          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179        let where = '';
180        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186        push('element-out-of-viewport', el,
187          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188      }
189      continue;
190    }
191    if (cs.position === 'sticky') continue;
192    // Negative left overflow cannot be reached by scrolling (scrollLeft
193    // never goes below 0).
194    if (r.left < -2) {
195      push('element-out-of-viewport', el,
196        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197      continue;
198    }
199    // Negative top with the page at the top means the element sits above
200    // the document origin — also unreachable.
201    if (r.top < -2 && de.scrollTop <= 2) {
202      push('element-out-of-viewport', el,
203        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204      continue;
205    }
206    const overRight = r.right - vw;
207    // Right-edge overflow is only a defect when the page cannot scroll
208    // horizontally to reveal it (or the element sticks out past the
209    // scrollable content width itself).
210    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211      push('element-out-of-viewport', el,
212        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213      continue;
214    }
215    // Below-the-fold content on a scrollable page is normal (tall landing
216    // pages); only flag bottom overflow the user can never scroll to.
217    const overBottom = r.bottom - vh;
218    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219      push('element-out-of-viewport', el,
220        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221    }
222  }
223  // 3. Text clipped by overflow:hidden containers.
224  for (const el of all) {
225    if (isIgnored(el)) continue;
226    const cs = getComputedStyle(el);
227    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229    if (!(el.textContent || '').trim()) continue;
230    // Single-line truncation with an ellipsis is an intentional design
231    // pattern (Tailwind .truncate etc.), not a clipping defect.
232    if (cs.textOverflow === 'ellipsis') continue;
233    push('text-clipped', el,
234      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236  }
237  // 4. Interactive elements covered by a different element.
238  const interactive =
239    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240  const targets = document.querySelectorAll(interactive);
241  for (const el of targets) {
242    if (isIgnored(el)) continue;
243    const r = el.getBoundingClientRect();
244    if (r.width < 6 || r.height < 6) continue;
245    const cs = getComputedStyle(el);
246    if (!visible(cs)) continue;
247    const cx = r.left + r.width / 2;
248    const cy = r.top + r.height / 2;
249    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250    const top = document.elementFromPoint(cx, cy);
251    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252    if (isIgnored(top)) continue;
253    const tcs = getComputedStyle(top);
254    if (!visible(tcs)) continue;
255    if (tcs.pointerEvents === 'none') continue;
256    const tr = top.getBoundingClientRect();
257    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260    push('element-overlap', el,
261      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262      ' is covered by <' + tname + '>');
263  }
264  return JSON.stringify(issues);
265})()
266"#;
267
268/// How long the CDP connection stays open after the browser goes quiet.
269///
270/// `headless_chrome` ships a 30s default and tears down the entire connection
271/// when no traffic arrives for that long; a run must own its connection for
272/// its full duration instead.
273const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275/// Executes a [`Scenario`] against a real browser with optional LLM
276/// assistance for element targeting and assertions.
277pub struct ScenarioRunner {
278    config: ScenarioConfig,
279    definitions: HashMap<String, AssertDefinition>,
280    llm: LlmConfig,
281    timeout: Duration,
282    viewport_width: u32,
283    viewport_height: u32,
284    /// The viewport currently applied in the browser (CDP emulation).
285    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
286    applied_viewport: std::cell::Cell<(u32, u32)>,
287    endpoints: EndpointRegistry,
288    usage: Arc<UsageTracker>,
289    budgets: BudgetTracker,
290    /// Directory for failure artifacts (screenshots).
291    artifacts_dir: PathBuf,
292    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
293    reporter: Arc<Reporter>,
294    /// The test + step index currently executing, for LLM-call events.
295    current_step: std::cell::RefCell<Option<(String, u32)>>,
296    /// Epoch seconds of runner creation — constant for the whole run so
297    /// the LLM context preamble below is cacheable by upstream providers.
298    run_started: u64,
299    /// Whether this runner emits the `RunStarted`/`RunFinished` events.
300    /// When several scenario files are run concurrently by
301    /// [`crate::parallel::run_scenarios`], the orchestrator owns those two
302    /// events (one per batch), so each per-file runner suppresses its own.
303    emit_run_events: bool,
304}
305
306/// Aggregated results from a scenario run.
307#[derive(Debug, Default)]
308pub struct RunReport {
309    /// Number of tests that passed.
310    pub tests_passed: u32,
311    /// Number of tests that failed.
312    pub tests_failed: u32,
313    /// Number of steps that passed.
314    pub passed: u32,
315    /// Number of steps that failed.
316    pub failed: u32,
317    /// Number of steps that were skipped.
318    pub skipped: u32,
319    /// Per-step details.
320    pub details: Vec<StepResult>,
321}
322
323/// Result of a single step execution.
324#[derive(Debug, Clone)]
325pub struct StepResult {
326    /// The step name.
327    pub name: String,
328    /// Whether the step passed, failed, or was skipped.
329    pub status: StepStatus,
330    /// Human-readable result message.
331    pub message: String,
332}
333
334/// Outcome for a single step.
335pub use crate::events::StepStatus;
336
337/// Predefined assertion preset definition.
338struct AssertPreset {
339    name: &'static str,
340    system: &'static str,
341    user_template: &'static str,
342}
343
344/// Built-in assertion presets.
345#[allow(clippy::literal_string_with_formatting_args)]
346const ASSERTION_PRESETS: &[AssertPreset] = &[
347    AssertPreset {
348        name: "no_error_on_page",
349        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
350        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
351    },
352    AssertPreset {
353        name: "text_visible",
354        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
355        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
356    },
357    AssertPreset {
358        name: "element_exists",
359        system: "You are a QA tester. Check if a described UI element exists on a web page.",
360        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
361    },
362    AssertPreset {
363        name: "visual_no_issues",
364        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
365        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
366    },
367    AssertPreset {
368        name: "visual_no_overlaps",
369        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
370        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
371    },
372    AssertPreset {
373        name: "visual_text_visible",
374        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
375        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
376    },
377    AssertPreset {
378        name: "layout_no_issues",
379        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
380        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
381    },
382];
383
384/// Whether an LLM verdict should be read as PASS.
385///
386/// Models routinely wrap the verdict in markdown (`**PASS** …`) or prefix it
387/// with punctuation, which a bare `starts_with("pass")` misreads as a failure.
388/// Strip any leading non-alphanumerics, then match the leading word; this also
389/// accepts the natural-language forms `pass` / `passes`.
390fn verdict_is_pass(content: &str) -> bool {
391    content
392        .trim()
393        .trim_start_matches(|c: char| !c.is_alphanumeric())
394        .to_lowercase()
395        .starts_with("pass")
396}
397
398/// Whether an LLM verdict should be read as FAIL (see [`verdict_is_pass`]).
399fn verdict_is_fail(content: &str) -> bool {
400    content
401        .trim()
402        .trim_start_matches(|c: char| !c.is_alphanumeric())
403        .to_lowercase()
404        .starts_with("fail")
405}
406
407impl ScenarioRunner {
408    /// Creates a new runner with the given scenario configuration and
409    /// assertion definitions.
410    #[must_use]
411    #[allow(clippy::needless_pass_by_value)]
412    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
413        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
414    }
415
416    /// Creates a runner that reports run events through the given reporter.
417    #[must_use]
418    #[allow(clippy::needless_pass_by_value)]
419    pub fn with_reporter(
420        scenario_config: ScenarioConfig,
421        definitions: Vec<AssertDefinition>,
422        reporter: Arc<Reporter>,
423    ) -> Self {
424        Self::with_reporter_mode(scenario_config, definitions, reporter, true)
425    }
426
427    /// Creates a runner that reports run events through the given reporter
428    /// but suppresses the `RunStarted`/`RunFinished` events. Used by the
429    /// parallel orchestrator, which emits those once per batch.
430    #[must_use]
431    #[allow(clippy::needless_pass_by_value)]
432    pub fn with_reporter_parallel(
433        scenario_config: ScenarioConfig,
434        definitions: Vec<AssertDefinition>,
435        reporter: Arc<Reporter>,
436    ) -> Self {
437        Self::with_reporter_mode(scenario_config, definitions, reporter, false)
438    }
439
440    #[must_use]
441    #[allow(clippy::needless_pass_by_value)]
442    fn with_reporter_mode(
443        scenario_config: ScenarioConfig,
444        definitions: Vec<AssertDefinition>,
445        reporter: Arc<Reporter>,
446        emit_run_events: bool,
447    ) -> Self {
448        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
449            &scenario_config,
450        ));
451        let llm = LlmConfig {
452            url: scenario_config
453                .llm_url
454                .clone()
455                .unwrap_or_else(crate::llm_base_url),
456            model: scenario_config
457                .llm_model
458                .clone()
459                .unwrap_or_else(crate::llm_model),
460            api_key: scenario_config
461                .llm_api_key
462                .clone()
463                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
464            headers: if scenario_config.llm_headers.is_empty() {
465                crate::parse_headers_env()
466            } else {
467                scenario_config.llm_headers.clone()
468            },
469            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
470            temperature: scenario_config.temperature,
471            thinking: scenario_config.thinking,
472            model_params: scenario_config.model_params.clone(),
473            cache: scenario_config.cache.unwrap_or(true),
474            max_attempts: crate::default_llm_attempts(),
475            provider: crate::scenario::Provider::Openai,
476            deployment: None,
477            api_version: None,
478            auth: crate::scenario::AuthConfig::default(),
479            header_commands: std::collections::HashMap::new(),
480            aws: crate::scenario::AwsConfig::default(),
481        };
482        let endpoints = EndpointRegistry::from_config(
483            &scenario_config.endpoints,
484            Some(&llm),
485            crate::endpoints::EndpointDefaults {
486                cache: scenario_config.cache,
487                cache_pricing: scenario_config.cache_pricing,
488            },
489        );
490        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
491        let defs_map: HashMap<String, AssertDefinition> = definitions
492            .into_iter()
493            .map(|d| (d.name.clone(), d))
494            .collect();
495
496        Self {
497            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
498            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
499            viewport_height: scenario_config.viewport_height.unwrap_or(720),
500            applied_viewport: std::cell::Cell::new((0, 0)),
501            config: scenario_config.clone(),
502            definitions: defs_map,
503            llm,
504            endpoints,
505            usage: Arc::new(UsageTracker::new()),
506            budgets,
507            artifacts_dir: PathBuf::from(
508                scenario_config
509                    .artifacts_dir
510                    .unwrap_or_else(|| "artifacts".to_owned()),
511            ),
512            reporter,
513            current_step: std::cell::RefCell::new(None),
514            run_started: SystemTime::now()
515                .duration_since(UNIX_EPOCH)
516                .map_or(0, |d| d.as_secs()),
517            emit_run_events,
518        }
519    }
520
521    /// Emits an event; a sink failure degrades to a console warning so a
522    /// broken log file can never mask the run itself.
523    fn emit_event(&self, event: &TestEvent) {
524        if let Err(err) = self.reporter.emit(event) {
525            use std::io::Write as _;
526            let mut out = std::io::stderr().lock();
527            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
528        }
529    }
530
531    /// Returns a clone of the [`UsageTracker`] for reporting.
532    #[must_use]
533    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
534        Arc::clone(&self.usage)
535    }
536
537    /// Returns a reference to the [`BudgetTracker`].
538    #[must_use]
539    pub const fn budget_tracker(&self) -> &BudgetTracker {
540        &self.budgets
541    }
542
543    /// Executes all test groups in the scenario and returns a report.
544    ///
545    /// # Errors
546    ///
547    /// Returns an error if the browser fails to launch.
548    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
549    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
550        let mut report = RunReport::default();
551
552        if tests.is_empty() {
553            self.reporter.warn("No tests defined in scenario.");
554            if self.emit_run_events {
555                self.emit_event(&TestEvent::RunFinished {
556                    tests_passed: 0,
557                    tests_failed: 0,
558                    steps_passed: 0,
559                    steps_failed: 0,
560                    steps_skipped: 0,
561                    total_cost: 0.0,
562                    total_tokens: 0,
563                    total_input_tokens: 0,
564                    total_output_tokens: 0,
565                    total_cached_input_tokens: 0,
566                    total_cache_creation_input_tokens: 0,
567                    models: Vec::new(),
568                    total_calls: 0,
569                });
570            }
571            return Ok(report);
572        }
573
574        if self.emit_run_events {
575            self.emit_event(&TestEvent::RunStarted {
576                total_tests: tests.len() as u32,
577            });
578        }
579
580        let browser_headless = self.config.browser_headless.unwrap_or(true);
581
582        let launch_opts = LaunchOptions {
583            headless: browser_headless,
584            window_size: Some((self.viewport_width, self.viewport_height)),
585            sandbox: false,
586            // headless_chrome defaults this to 30s and shuts down the whole CDP
587            // connection when no messages arrive for that long. A scenario can
588            // easily exceed 30s of browser silence (slow LLM targeting/assertion
589            // calls, page waits, budget checks between steps), after which every
590            // remaining step fails with "Unable to make method calls because
591            // underlying connection is closed" — one quiet gap kills the run.
592            // Open-ended scenarios must own the connection for their full
593            // duration, so keep it alive for 6 hours.
594            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
595            ..LaunchOptions::default()
596        };
597
598        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
599        let tab = browser.new_tab().context("failed to open browser tab")?;
600        match (
601            self.config.browser_basic_auth_user.clone(),
602            self.config.browser_basic_auth_password.clone(),
603        ) {
604            (Some(username), Some(password)) if !username.is_empty() && !password.is_empty() => {
605                tab.authenticate(Some(username), Some(password))
606                    .context("failed to configure browser HTTP Basic Auth")?;
607                tab.enable_fetch(None, Some(true))
608                    .context("failed to enable browser HTTP authentication")?;
609            }
610            (None, None) => {}
611            _ => anyhow::bail!("browser Basic Auth requires both username and password"),
612        }
613        let _ = tab.set_default_timeout(self.timeout);
614
615        // Start MCP server if configured
616        #[cfg(feature = "mcp-server")]
617        if let Some(ref mcp_cfg) = self.config.mcp_server {
618            if mcp_cfg.enabled {
619                let port = mcp_cfg.port;
620                std::thread::spawn(move || {
621                    let _ = crate::mcp_server::start_mcp_server(port);
622                });
623            }
624        }
625        #[cfg(not(feature = "mcp-server"))]
626        if let Some(mcp_cfg) = &self.config.mcp_server {
627            if mcp_cfg.enabled {
628                self.reporter
629                    .warn("MCP server configured but 'mcp-server' feature not enabled");
630            }
631        }
632
633        // Start A2A agent server if configured
634        #[cfg(feature = "a2a-server")]
635        if let Some(ref a2a_cfg) = self.config.a2a_server {
636            if a2a_cfg.enabled {
637                let port = a2a_cfg.port;
638                tokio::spawn(crate::a2a_server::start_a2a_server(port));
639            }
640        }
641        #[cfg(not(feature = "a2a-server"))]
642        if let Some(a2a_cfg) = &self.config.a2a_server {
643            if a2a_cfg.enabled {
644                self.reporter
645                    .warn("A2A server configured but 'a2a-server' feature not enabled");
646            }
647        }
648
649        for test in tests {
650            self.emit_event(&TestEvent::TestStarted {
651                test: test.name.clone(),
652            });
653
654            self.usage.reset_per_test();
655
656            let test_started = Instant::now();
657            let test_result = self.run_test(test, &tab);
658            let duration_ms = test_started.elapsed().as_millis() as u64;
659            let usage = self.usage.current_test_snapshot();
660            self.usage.commit_test(&test.name);
661
662            self.emit_event(&TestEvent::TestFinished {
663                test: test.name.clone(),
664                passed: test_result.passed,
665                failed: test_result.failed,
666                skipped: test_result.skipped,
667                duration_ms,
668                cost: usage.total_cost,
669                tokens: usage.total_tokens,
670                input_tokens: usage.total_input_tokens,
671                output_tokens: usage.total_output_tokens,
672                cached_input_tokens: usage.total_cached_input_tokens,
673                cache_creation_input_tokens: usage.total_cache_creation_input_tokens,
674                models: usage.models.clone(),
675                calls: usage.total_calls,
676            });
677
678            if test_result.failed == 0 && test_result.total > 0 {
679                report.tests_passed += 1;
680            } else if test_result.total > 0 {
681                report.tests_failed += 1;
682            }
683
684            report.passed += test_result.passed;
685            report.failed += test_result.failed;
686            report.skipped += test_result.skipped;
687            report.details.extend(test_result.details);
688        }
689
690        let global = self.usage.global_snapshot();
691        if self.emit_run_events {
692            self.emit_event(&TestEvent::RunFinished {
693                tests_passed: report.tests_passed,
694                tests_failed: report.tests_failed,
695                steps_passed: report.passed,
696                steps_failed: report.failed,
697                steps_skipped: report.skipped,
698                total_cost: global.total_cost,
699                total_tokens: global.total_tokens,
700                total_input_tokens: global.total_input_tokens,
701                total_output_tokens: global.total_output_tokens,
702                total_cached_input_tokens: global.total_cached_input_tokens,
703                total_cache_creation_input_tokens: global.total_cache_creation_input_tokens,
704                models: global.models.clone(),
705                total_calls: global.total_calls,
706            });
707        }
708
709        Ok(report)
710    }
711
712    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
713    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
714        let base_url = test
715            .base_url
716            .clone()
717            .or_else(|| self.config.base_url.clone())
718            .unwrap_or_else(crate::base_url);
719
720        // Per-test viewport override: switch the browser via CDP
721        // device-metrics emulation before this test runs.
722        let vw = test.viewport_width.unwrap_or(self.viewport_width);
723        let vh = test.viewport_height.unwrap_or(self.viewport_height);
724        if self.applied_viewport.get() != (vw, vh) {
725            self.apply_viewport(tab, vw, vh);
726            self.applied_viewport.set((vw, vh));
727        }
728
729        // Per-test isolation: every test starts from its own start_url
730        // (unless auto_navigate is disabled), so a test never inherits the
731        // previous test's page state.
732        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
733
734        let start_url = test
735            .start_url
736            .clone()
737            .or_else(|| self.config.start_url.clone())
738            .unwrap_or_else(|| "/dashboard".to_owned());
739
740        if auto_navigate {
741            let full_url = resolve_url(&start_url, &base_url);
742            self.reporter.debug(format!("auto-navigate: {full_url}"));
743            let _ = tab.navigate_to(&full_url);
744            let _ = tab.wait_until_navigated();
745            std::thread::sleep(Duration::from_secs(4));
746        }
747
748        let mut result = TestRunResult::default();
749
750        for (step_index, step) in test.steps.iter().enumerate() {
751            result.total += 1;
752
753            let wait_ms = match step {
754                TestStep::Navigate { wait_after_ms, .. }
755                | TestStep::Click { wait_after_ms, .. }
756                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
757                _ => None,
758            };
759
760            self.current_step
761                .replace(Some((test.name.clone(), step_index as u32)));
762            self.emit_event(&TestEvent::StepStarted {
763                test: test.name.clone(),
764                index: step_index as u32,
765                label: step_label(step),
766            });
767            let step_started = Instant::now();
768
769            let mut step_result = match step {
770                TestStep::Navigate { url, .. } => {
771                    let full_url = resolve_url(url, &base_url);
772                    run_navigate_step(&full_url, tab)
773                }
774                TestStep::Click {
775                    target,
776                    selector,
777                    endpoint,
778                    idempotent,
779                    ..
780                } => self.run_click(
781                    target,
782                    selector.as_deref(),
783                    endpoint.as_deref(),
784                    test.endpoint.as_deref(),
785                    *idempotent,
786                    tab,
787                ),
788                TestStep::Type {
789                    target,
790                    text,
791                    selector,
792                    endpoint,
793                    idempotent,
794                    ..
795                } => self.run_type(
796                    target,
797                    text,
798                    selector.as_deref(),
799                    endpoint.as_deref(),
800                    test.endpoint.as_deref(),
801                    *idempotent,
802                    tab,
803                ),
804                TestStep::Wait {
805                    target,
806                    selector,
807                    text,
808                    timeout_ms,
809                    endpoint,
810                    idempotent,
811                } => self.run_wait(
812                    target,
813                    selector.as_deref(),
814                    text.as_deref(),
815                    *timeout_ms,
816                    endpoint.as_deref(),
817                    test.endpoint.as_deref(),
818                    *idempotent,
819                    tab,
820                ),
821                TestStep::Assert {
822                    definition,
823                    preset,
824                    prompt,
825                    assert_text,
826                    endpoint,
827                    screenshot,
828                } => self.run_assert(
829                    definition.as_deref(),
830                    preset.as_deref(),
831                    prompt.as_deref(),
832                    assert_text.as_deref(),
833                    *screenshot,
834                    endpoint.as_deref(),
835                    test.endpoint.as_deref(),
836                    tab,
837                ),
838                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
839                TestStep::Agent {
840                    agent,
841                    task,
842                    definition,
843                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
844                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
845            };
846
847            // Failure diagnostics: capture the page state and a screenshot so
848            // CI logs say WHAT the page looked like when the step failed,
849            // instead of a bare "timed out: The event waited for never came".
850            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
851                let state = diagnostics::capture(tab);
852                let screenshot = diagnostics::save_screenshot(
853                    tab,
854                    &self.artifacts_dir,
855                    &test.name,
856                    &test.name,
857                    step_index,
858                    step_kind_label(step),
859                );
860                step_result.message = format!(
861                    "{base} — {excerpt}",
862                    base = step_result.message,
863                    excerpt = diagnostics::inline_excerpt(&state),
864                );
865                (Some(diagnostics::full_context(&state)), screenshot)
866            } else {
867                (None, None)
868            };
869
870            let duration_ms = step_started.elapsed().as_millis() as u64;
871            self.emit_event(&TestEvent::StepFinished {
872                test: test.name.clone(),
873                index: step_index as u32,
874                label: step_result.name.clone(),
875                status: step_result.status,
876                duration_ms,
877                message: step_result.message.clone(),
878                diagnostics: diagnostics_block,
879                screenshot: screenshot_path,
880            });
881            self.current_step.replace(None);
882
883            match step_result.status {
884                StepStatus::Passed => result.passed += 1,
885                StepStatus::Failed => result.failed += 1,
886                StepStatus::Skipped => result.skipped += 1,
887            }
888
889            // Fail fast: the first failed step ends the test and the
890            // remaining steps are reported as skipped (no LLM budget is
891            // burned asserting against a page that is already known broken).
892            if step_result.status == StepStatus::Failed
893                && !self.config.continue_on_failure
894                && step_index + 1 < test.steps.len()
895            {
896                self.emit_event(&TestEvent::Warning {
897                    message: format!(
898                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
899                        test.steps.len() - step_index - 1
900                    ),
901                });
902                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
903                    let skipped_index = step_index + 1 + offset;
904                    let label = step_label(skipped);
905                    result.total += 1;
906                    result.skipped += 1;
907                    self.emit_event(&TestEvent::StepStarted {
908                        test: test.name.clone(),
909                        index: skipped_index as u32,
910                        label: label.clone(),
911                    });
912                    self.emit_event(&TestEvent::StepFinished {
913                        test: test.name.clone(),
914                        index: skipped_index as u32,
915                        label,
916                        status: StepStatus::Skipped,
917                        duration_ms: 0,
918                        message: "skipped: previous step failed".into(),
919                        diagnostics: None,
920                        screenshot: None,
921                    });
922                    result.details.push(StepResult {
923                        name: step_label(skipped),
924                        status: StepStatus::Skipped,
925                        message: "skipped: previous step failed".into(),
926                    });
927                }
928                result.details.push(step_result);
929                return result;
930            }
931
932            // Check per-test budget after each step
933            let test_usage = self.usage.current_test_snapshot();
934            let global_usage = self.usage.global_snapshot();
935            let budget_status = self.budgets.check_all(
936                &test.name,
937                &test_usage,
938                &global_usage,
939                test.budget.as_ref(),
940            );
941            match budget_status {
942                BudgetStatus::HardExceeded { message, .. } => {
943                    self.emit_event(&TestEvent::Warning {
944                        message: format!("budget exceeded: {message}"),
945                    });
946                    result.details.push(StepResult {
947                        name: "[budget]".into(),
948                        status: StepStatus::Failed,
949                        message,
950                    });
951                    result.failed += 1;
952                    return result;
953                }
954                BudgetStatus::SoftExceeded { message, .. } => {
955                    self.emit_event(&TestEvent::Warning {
956                        message: format!("budget warning: {message}"),
957                    });
958                }
959                BudgetStatus::Ok => {}
960            }
961
962            if let Some(ms) = wait_ms {
963                std::thread::sleep(Duration::from_millis(ms));
964            }
965
966            result.details.push(step_result);
967        }
968
969        result
970    }
971
972    /// Applies a viewport size to the current tab via CDP
973    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
974    /// overrides and the viewport matrix. The initial window size set at
975    /// browser launch is replaced by emulation; failures are logged but
976    /// do not fail the test (a mismatched viewport only weakens coverage).
977    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
978        use headless_chrome::protocol::cdp::Emulation;
979        let _ = self;
980        let params = Emulation::SetDeviceMetricsOverride {
981            width,
982            height,
983            device_scale_factor: 1.0,
984            mobile: false,
985            scale: None,
986            screen_width: Some(width),
987            screen_height: Some(height),
988            position_x: None,
989            position_y: None,
990            dont_set_visible_size: None,
991            screen_orientation: None,
992            viewport: None,
993            display_feature: None,
994            device_posture: None,
995        };
996        self.reporter.debug(format!("viewport: {width}x{height}"));
997        if let Err(e) = tab.call_method(params) {
998            self.reporter
999                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
1000        }
1001    }
1002
1003    /// Height (px) covered by assert-step screenshots: the configured
1004    /// `screenshot_max_height` (absolute px or viewport multiple, default
1005    /// `"20x"`) resolved against the currently applied viewport, raised to
1006    /// at least the viewport height so the visible screen is always fully
1007    /// included. The capture is split into viewport-tall tiles, so this
1008    /// value bounds total coverage (and hence the number of image parts).
1009    #[must_use]
1010    fn screenshot_height_cap(&self) -> u32 {
1011        let viewport_height = self.current_viewport_height();
1012        let cap = self
1013            .config
1014            .screenshot_max_height
1015            .as_ref()
1016            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
1017        cap.max(viewport_height)
1018    }
1019
1020    /// Height of the viewport currently emulated in the browser (falling
1021    /// back to the configured default before any emulation was applied).
1022    #[must_use]
1023    const fn current_viewport_height(&self) -> u32 {
1024        let (_, height) = self.applied_viewport.get();
1025        if height > 0 {
1026            height
1027        } else {
1028            self.viewport_height
1029        }
1030    }
1031
1032    // ── step handlers ───────────────────────────────────────────────────
1033
1034    #[allow(clippy::too_many_lines)]
1035    fn run_click(
1036        &self,
1037        target: &str,
1038        selector_override: Option<&str>,
1039        step_endpoint: Option<&str>,
1040        test_endpoint: Option<&str>,
1041        idempotent: bool,
1042        tab: &Tab,
1043    ) -> StepResult {
1044        let name = format!("[click] {target}");
1045        let selector = match self.resolve_selector(
1046            selector_override,
1047            target,
1048            step_endpoint,
1049            test_endpoint,
1050            tab,
1051        ) {
1052            Ok(s) => s,
1053            Err(msg) => {
1054                if idempotent {
1055                    return StepResult {
1056                        name,
1057                        status: StepStatus::Skipped,
1058                        message: format!("skipped (idempotent): no target found — {msg}"),
1059                    };
1060                }
1061                return StepResult {
1062                    name,
1063                    status: StepStatus::Failed,
1064                    message: msg,
1065                };
1066            }
1067        };
1068
1069        // Idempotent steps probe briefly: a missing target means the
1070        // action was already done / not applicable (e.g. an
1071        // already-authenticated session), and skipping is the success
1072        // path, not a failure.
1073        let probe_secs = if idempotent { 5 } else { 10 };
1074        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1075            Ok(element) => match element.click() {
1076                Ok(_) => StepResult {
1077                    name,
1078                    status: StepStatus::Passed,
1079                    message: format!("clicked {selector}"),
1080                },
1081                Err(e) => StepResult {
1082                    name,
1083                    status: StepStatus::Failed,
1084                    message: format!("click failed on {selector}: {e}"),
1085                },
1086            },
1087            Err(e) if idempotent => StepResult {
1088                name,
1089                status: StepStatus::Skipped,
1090                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1091            },
1092            Err(e) => StepResult {
1093                name,
1094                status: StepStatus::Failed,
1095                message: format!("element {selector} not found: {e}"),
1096            },
1097        }
1098    }
1099
1100    #[allow(clippy::too_many_arguments)]
1101    fn run_type(
1102        &self,
1103        target: &str,
1104        text: &str,
1105        selector_override: Option<&str>,
1106        step_endpoint: Option<&str>,
1107        test_endpoint: Option<&str>,
1108        idempotent: bool,
1109        tab: &Tab,
1110    ) -> StepResult {
1111        let name = format!("[type] {target}");
1112        let selector = match self.resolve_selector(
1113            selector_override,
1114            target,
1115            step_endpoint,
1116            test_endpoint,
1117            tab,
1118        ) {
1119            Ok(s) => s,
1120            Err(msg) => {
1121                if idempotent {
1122                    return StepResult {
1123                        name,
1124                        status: StepStatus::Skipped,
1125                        message: format!("skipped (idempotent): no target found — {msg}"),
1126                    };
1127                }
1128                return StepResult {
1129                    name,
1130                    status: StepStatus::Failed,
1131                    message: msg,
1132                };
1133            }
1134        };
1135
1136        let probe_secs = if idempotent { 5 } else { 10 };
1137        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1138            Ok(element) => {
1139                if let Err(e) = element.click() {
1140                    return StepResult {
1141                        name,
1142                        status: StepStatus::Failed,
1143                        message: format!("click to focus {selector} failed: {e}"),
1144                    };
1145                }
1146
1147                let js = format!(
1148                    "document.querySelector('{}').value = '';",
1149                    selector.replace('\'', "\\'")
1150                );
1151                let _ = tab.evaluate(&js, false);
1152
1153                match element.type_into(text) {
1154                    Ok(_) => StepResult {
1155                        name,
1156                        status: StepStatus::Passed,
1157                        message: format!("typed {text:?} into {selector}"),
1158                    },
1159                    Err(e) => StepResult {
1160                        name,
1161                        status: StepStatus::Failed,
1162                        message: format!("type into {selector} failed: {e}"),
1163                    },
1164                }
1165            }
1166            Err(e) if idempotent => StepResult {
1167                name,
1168                status: StepStatus::Skipped,
1169                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1170            },
1171            Err(e) => StepResult {
1172                name,
1173                status: StepStatus::Failed,
1174                message: format!("element {selector} not found: {e}"),
1175            },
1176        }
1177    }
1178
1179    #[allow(clippy::too_many_arguments)]
1180    #[allow(clippy::too_many_lines)]
1181    fn run_wait(
1182        &self,
1183        target: &str,
1184        selector_override: Option<&str>,
1185        text: Option<&str>,
1186        timeout_ms: Option<u64>,
1187        step_endpoint: Option<&str>,
1188        test_endpoint: Option<&str>,
1189        idempotent: bool,
1190        tab: &Tab,
1191    ) -> StepResult {
1192        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1193        let step_name = format!("[wait] {target}");
1194
1195        // Resolve an explicit selector only (text-only waits are LLM-free).
1196        let selector = match selector_override {
1197            Some(s) => Some(s.to_owned()),
1198            None if text.is_some() => None,
1199            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1200                Ok(s) => Some(s),
1201                Err(msg) => {
1202                    if idempotent {
1203                        return StepResult {
1204                            name: step_name,
1205                            status: StepStatus::Skipped,
1206                            message: format!("skipped (idempotent): no target found — {msg}"),
1207                        };
1208                    }
1209                    return StepResult {
1210                        name: step_name,
1211                        status: StepStatus::Failed,
1212                        message: msg,
1213                    };
1214                }
1215            },
1216        };
1217
1218        if text.is_some() {
1219            let sel_js = selector
1220                .as_deref()
1221                .map(crate::selectors::selector_matches_js);
1222            let text_js = text.map(|t| {
1223                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1224                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1225            });
1226
1227            let deadline = Instant::now() + timeout;
1228            loop {
1229                let sel_ok = sel_js
1230                    .as_ref()
1231                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1232                let text_ok = text_js
1233                    .as_ref()
1234                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1235                if sel_ok && text_ok {
1236                    let mut what = Vec::new();
1237                    if let Some(sel) = &selector {
1238                        what.push(format!("found {sel}"));
1239                    }
1240                    if let Some(t) = text {
1241                        what.push(format!("text {t:?} visible"));
1242                    }
1243                    return StepResult {
1244                        name: step_name,
1245                        status: StepStatus::Passed,
1246                        message: what.join(" and "),
1247                    };
1248                }
1249                if Instant::now() >= deadline {
1250                    let mut what = Vec::new();
1251                    if let Some(sel) = &selector {
1252                        what.push(sel.clone());
1253                    }
1254                    if let Some(t) = text {
1255                        what.push(format!("text {t:?}"));
1256                    }
1257                    let message = format!(
1258                        "wait for {} timed out after {}ms: the event waited for never came",
1259                        what.join(" / "),
1260                        timeout.as_millis(),
1261                    );
1262                    if idempotent {
1263                        return StepResult {
1264                            name: step_name,
1265                            status: StepStatus::Skipped,
1266                            message: format!("skipped (idempotent): {message}"),
1267                        };
1268                    }
1269                    return StepResult {
1270                        name: step_name,
1271                        status: StepStatus::Failed,
1272                        message,
1273                    };
1274                }
1275                std::thread::sleep(Duration::from_millis(250));
1276            }
1277        }
1278
1279        match selector.as_deref() {
1280            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1281                Ok(_) => StepResult {
1282                    name: step_name,
1283                    status: StepStatus::Passed,
1284                    message: format!("found {sel}"),
1285                },
1286                Err(e) if idempotent => StepResult {
1287                    name: step_name,
1288                    status: StepStatus::Skipped,
1289                    message: format!(
1290                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1291                        timeout.as_millis()
1292                    ),
1293                },
1294                Err(e) => StepResult {
1295                    name: step_name,
1296                    status: StepStatus::Failed,
1297                    message: format!(
1298                        "wait for {sel} timed out after {}ms: {e}",
1299                        timeout.as_millis()
1300                    ),
1301                },
1302            },
1303            None => StepResult {
1304                name: step_name,
1305                status: StepStatus::Failed,
1306                message: "wait step has neither selector nor text".into(),
1307            },
1308        }
1309    }
1310
1311    #[allow(clippy::too_many_arguments)]
1312    fn run_assert(
1313        &self,
1314        definition: Option<&str>,
1315        preset: Option<&str>,
1316        prompt: Option<&str>,
1317        assert_text: Option<&str>,
1318        screenshot: bool,
1319        step_endpoint: Option<&str>,
1320        test_endpoint: Option<&str>,
1321        tab: &Tab,
1322    ) -> StepResult {
1323        std::thread::sleep(Duration::from_millis(500));
1324
1325        let page_content = get_page_text(tab);
1326
1327        // Vision attach: capture the full page once per assert step and
1328        // split it into viewport-tall tiles (the total coverage is bounded
1329        // by the configured height cap so vision tokens stay sane). All
1330        // tile data URLs are handed to the preset/prompt evaluation below.
1331        let image: Option<Vec<String>> = if screenshot {
1332            let endpoint = self
1333                .endpoints
1334                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1335            if !endpoint.vision {
1336                return StepResult {
1337                    name: "[assert]".into(),
1338                    status: StepStatus::Failed,
1339                    message: format!(
1340                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1341                        name = endpoint.name
1342                    ),
1343                };
1344            }
1345            match crate::vision::capture_screenshot_data_urls(
1346                tab,
1347                self.config
1348                    .screenshot_max_dimension
1349                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1350                self.screenshot_height_cap(),
1351                self.current_viewport_height(),
1352            ) {
1353                Ok(urls) => Some(urls),
1354                Err(e) => {
1355                    return StepResult {
1356                        name: "[assert]".into(),
1357                        status: StepStatus::Failed,
1358                        message: format!("screenshot capture failed: {e}"),
1359                    };
1360                }
1361            }
1362        } else {
1363            None
1364        };
1365
1366        if let Some(def_name) = definition {
1367            if let Some(def) = self.definitions.get(def_name) {
1368                return self.run_assert_def(
1369                    def,
1370                    &page_content,
1371                    image.as_deref(),
1372                    step_endpoint,
1373                    test_endpoint,
1374                    tab,
1375                );
1376            }
1377            return StepResult {
1378                name: format!("[assert] {def_name}"),
1379                status: StepStatus::Failed,
1380                message: format!("definition '{def_name}' not found"),
1381            };
1382        }
1383
1384        if let Some(preset_name) = preset {
1385            // Deterministic DOM layout scan — runs JS in the browser and
1386            // never calls the LLM (free, fast, no pixel budget).
1387            if preset_name == "layout_no_issues" {
1388                return self.run_layout_preset(tab);
1389            }
1390            return self.run_preset(
1391                preset_name,
1392                assert_text,
1393                &page_content,
1394                image.as_deref(),
1395                step_endpoint,
1396                test_endpoint,
1397            );
1398        }
1399
1400        if let Some(prompt_text) = prompt {
1401            return self.run_custom(
1402                prompt_text,
1403                &page_content,
1404                image.as_deref(),
1405                step_endpoint,
1406                test_endpoint,
1407            );
1408        }
1409
1410        StepResult {
1411            name: "[assert]".into(),
1412            status: StepStatus::Skipped,
1413            message: "no definition, preset, or prompt specified".into(),
1414        }
1415    }
1416
1417    fn run_assert_def(
1418        &self,
1419        def: &AssertDefinition,
1420        page_content: &PageContent,
1421        image: Option<&[String]>,
1422        step_endpoint: Option<&str>,
1423        test_endpoint: Option<&str>,
1424        tab: &Tab,
1425    ) -> StepResult {
1426        // Agent-based definition: delegate to an A2A agent
1427        if let Some(ref agent) = def.agent {
1428            if image.is_some() {
1429                return StepResult {
1430                    name: format!("[assert] {}", def.name),
1431                    status: StepStatus::Failed,
1432                    message: "agent-backed assertions do not support screenshots".into(),
1433                };
1434            }
1435            let task = def
1436                .task_template
1437                .as_deref()
1438                .unwrap_or("Evaluate the assertion")
1439                .replace("{url}", &page_content.url)
1440                .replace("{title}", &page_content.title)
1441                .replace("{content}", &page_content.body_text)
1442                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1443
1444            return self.run_agent_step(agent, &task, &def.name);
1445        }
1446
1447        // Custom preset: system + user_template provided in the definition
1448        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1449            return self.run_custom_preset(
1450                &def.name,
1451                system,
1452                template,
1453                def.assert_text.as_deref(),
1454                page_content,
1455                image,
1456                step_endpoint,
1457                test_endpoint,
1458            );
1459        }
1460
1461        def.preset.as_ref().map_or_else(
1462            || {
1463                def.prompt.as_ref().map_or_else(
1464                    || StepResult {
1465                        name: format!("[assert] {}", def.name),
1466                        status: StepStatus::Failed,
1467                        message: "definition has no preset, prompt, or system+user_template".into(),
1468                    },
1469                    |prompt| {
1470                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1471                    },
1472                )
1473            },
1474            |preset_name| {
1475                if preset_name == "layout_no_issues" {
1476                    return self.run_layout_preset(tab);
1477                }
1478                self.run_preset(
1479                    preset_name,
1480                    def.assert_text.as_deref(),
1481                    page_content,
1482                    image,
1483                    step_endpoint,
1484                    test_endpoint,
1485                )
1486            },
1487        )
1488    }
1489
1490    #[allow(clippy::too_many_arguments)]
1491    fn run_custom_preset(
1492        &self,
1493        name: &str,
1494        system: &str,
1495        template: &str,
1496        assert_text: Option<&str>,
1497        page_content: &PageContent,
1498        image: Option<&[String]>,
1499        step_endpoint: Option<&str>,
1500        test_endpoint: Option<&str>,
1501    ) -> StepResult {
1502        let user_prompt = template
1503            .replace("{url}", &page_content.url)
1504            .replace("{title}", &page_content.title)
1505            .replace("{content}", &page_content.body_text)
1506            .replace("{expected_text}", assert_text.unwrap_or(""))
1507            .replace("{description}", "");
1508
1509        // Custom preset definitions frequently forget the {content}
1510        // placeholder — without it the LLM has no page to evaluate and
1511        // answers "I can't determine that without seeing the page". Always
1512        // append the page context unless the template already references it.
1513        let user_prompt = if template.contains("{content}") {
1514            user_prompt
1515        } else {
1516            format!(
1517                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1518                url = page_content.url,
1519                title = page_content.title,
1520                content = page_content.body_text,
1521            )
1522        };
1523
1524        self.reporter
1525            .debug(format!("assert: {name} (custom preset)"));
1526
1527        let chain = self
1528            .endpoints
1529            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1530        let sys = system.to_owned();
1531
1532        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1533
1534        response.map_or_else(
1535            |e| StepResult {
1536                name: format!("[assert] {name}"),
1537                status: StepStatus::Failed,
1538                message: format!("LLM assertion call failed: {e}"),
1539            },
1540            |(lr, _idx)| {
1541                if verdict_is_pass(&lr.content) {
1542                    StepResult {
1543                        name: format!("[assert] {name}"),
1544                        status: StepStatus::Passed,
1545                        message: "PASS".into(),
1546                    }
1547                } else {
1548                    StepResult {
1549                        name: format!("[assert] {name}"),
1550                        status: StepStatus::Failed,
1551                        message: lr.content,
1552                    }
1553                }
1554            },
1555        )
1556    }
1557
1558    fn run_preset(
1559        &self,
1560        preset_name: &str,
1561        assert_text: Option<&str>,
1562        page_content: &PageContent,
1563        image: Option<&[String]>,
1564        step_endpoint: Option<&str>,
1565        test_endpoint: Option<&str>,
1566    ) -> StepResult {
1567        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1568            return StepResult {
1569                name: format!("[assert] {preset_name}"),
1570                status: StepStatus::Failed,
1571                message: format!("unknown assertion preset: {preset_name}"),
1572            };
1573        };
1574        if preset_name.starts_with("visual_") && image.is_none() {
1575            return StepResult {
1576                name: format!("[assert] {preset_name}"),
1577                status: StepStatus::Failed,
1578                message: format!(
1579                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1580                ),
1581            };
1582        }
1583
1584        let user_prompt = preset
1585            .user_template
1586            .replace("{url}", &page_content.url)
1587            .replace("{title}", &page_content.title)
1588            .replace("{content}", &page_content.body_text)
1589            .replace("{expected_text}", assert_text.unwrap_or(""))
1590            .replace("{description}", "");
1591
1592        // Same safety net as custom presets: never let the LLM answer with
1593        // no page context at all.
1594        let user_prompt = if preset.user_template.contains("{content}") {
1595            user_prompt
1596        } else {
1597            format!(
1598                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1599                url = page_content.url,
1600                title = page_content.title,
1601                content = page_content.body_text,
1602            )
1603        };
1604
1605        self.reporter.debug(format!("assert: {preset_name}"));
1606
1607        let chain = self
1608            .endpoints
1609            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1610        let sys = preset.system.to_owned();
1611
1612        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1613
1614        response.map_or_else(
1615            |e| StepResult {
1616                name: format!("[assert] {preset_name}"),
1617                status: StepStatus::Failed,
1618                message: format!("LLM assertion call failed: {e}"),
1619            },
1620            |(lr, _idx)| {
1621                if verdict_is_pass(&lr.content) {
1622                    StepResult {
1623                        name: format!("[assert] {preset_name}"),
1624                        status: StepStatus::Passed,
1625                        message: "PASS".into(),
1626                    }
1627                } else {
1628                    StepResult {
1629                        name: format!("[assert] {preset_name}"),
1630                        status: StepStatus::Failed,
1631                        message: lr.content,
1632                    }
1633                }
1634            },
1635        )
1636    }
1637
1638    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1639    ///
1640    /// Evaluates the layout-scan JS in the page and fails with the list of
1641    /// detected issues: horizontal page overflow, visible elements sticking
1642    /// out of the viewport, text clipped by `overflow: hidden` containers,
1643    /// and interactive elements covered by other elements. No LLM call —
1644    /// checks are geometry-based so the check is free, deterministic, and
1645    /// safe to run on every page × viewport variant.
1646    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1647        let name = "[assert] layout_no_issues".to_owned();
1648        self.reporter
1649            .debug("assert: layout_no_issues (DOM layout scan)");
1650        let js = LAYOUT_SCAN_JS.replace(
1651            "__IGNORE_CLASSES__",
1652            &serde_json::to_string(&self.config.layout_ignore_classes)
1653                .unwrap_or_else(|_| "[]".to_owned()),
1654        );
1655        let result = tab.evaluate(&js, false);
1656        let json_str = match result {
1657            Ok(r) => r
1658                .value
1659                .as_ref()
1660                .and_then(|v| v.as_str().map(String::from))
1661                .unwrap_or_else(|| "[]".to_owned()),
1662            Err(e) => {
1663                return StepResult {
1664                    name,
1665                    status: StepStatus::Failed,
1666                    message: format!("layout scan JS failed: {e}"),
1667                };
1668            }
1669        };
1670        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1671        if issues.is_empty() {
1672            return StepResult {
1673                name,
1674                status: StepStatus::Passed,
1675                message: "PASS — no layout defects detected".into(),
1676            };
1677        }
1678        let mut lines: Vec<String> = issues
1679            .iter()
1680            .take(10)
1681            .map(|i| {
1682                format!(
1683                    "- [{type_}] {element}: {detail}",
1684                    type_ = i.issue_type,
1685                    element = i.element,
1686                    detail = i.detail
1687                )
1688            })
1689            .collect();
1690        if issues.len() > 10 {
1691            lines.push(format!("- … and {} more", issues.len() - 10));
1692        }
1693        StepResult {
1694            name,
1695            status: StepStatus::Failed,
1696            message: format!(
1697                "FAIL — {} layout defect(s) detected:\n{}",
1698                issues.len(),
1699                lines.join("\n")
1700            ),
1701        }
1702    }
1703
1704    fn run_custom(
1705        &self,
1706        prompt: &str,
1707        page_content: &PageContent,
1708        image: Option<&[String]>,
1709        step_endpoint: Option<&str>,
1710        test_endpoint: Option<&str>,
1711    ) -> StepResult {
1712        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1713
1714        let mut user = format!(
1715            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1716            url = page_content.url,
1717            title = page_content.title,
1718            content = page_content.body_text,
1719        );
1720        if image.is_some() {
1721            user.push_str(
1722                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1723            );
1724        }
1725
1726        self.reporter.debug("custom assert");
1727
1728        let chain = self
1729            .endpoints
1730            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1731        let sys = system.to_owned();
1732
1733        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1734
1735        response.map_or_else(
1736            |e| StepResult {
1737                name: "[assert] custom".into(),
1738                status: StepStatus::Failed,
1739                message: format!("LLM assertion call failed: {e}"),
1740            },
1741            |(lr, _idx)| {
1742                if verdict_is_pass(&lr.content) {
1743                    StepResult {
1744                        name: "[assert] custom".into(),
1745                        status: StepStatus::Passed,
1746                        message: "PASS".into(),
1747                    }
1748                } else {
1749                    StepResult {
1750                        name: "[assert] custom".into(),
1751                        status: StepStatus::Failed,
1752                        message: lr.content,
1753                    }
1754                }
1755            },
1756        )
1757    }
1758
1759    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1760        let path = path.unwrap_or("screenshot.png");
1761
1762        match tab.capture_screenshot(
1763            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1764            None,
1765            None,
1766            true,
1767        ) {
1768            Ok(data) => {
1769                if let Err(e) = std::fs::write(path, &data) {
1770                    return StepResult {
1771                        name: format!("[screenshot] {path}"),
1772                        status: StepStatus::Failed,
1773                        message: format!("failed to write screenshot: {e}"),
1774                    };
1775                }
1776                StepResult {
1777                    name: format!("[screenshot] {path}"),
1778                    status: StepStatus::Passed,
1779                    message: format!("saved to {path}"),
1780                }
1781            }
1782            Err(e) => StepResult {
1783                name: format!("[screenshot] {path}"),
1784                status: StepStatus::Failed,
1785                message: format!("screenshot failed: {e}"),
1786            },
1787        }
1788    }
1789
1790    /// Runs an A2A agent step.
1791    #[allow(clippy::literal_string_with_formatting_args)]
1792    fn run_agent(
1793        &self,
1794        agent_name: &str,
1795        task: &str,
1796        definition: Option<&str>,
1797        _test_endpoint: Option<&str>,
1798    ) -> StepResult {
1799        // If a definition is specified, look up the task template
1800        let resolved_task = if let Some(def_name) = definition {
1801            if let Some(def) = self.definitions.get(def_name) {
1802                let tmpl = def.task_template.as_deref().unwrap_or(task);
1803                tmpl.replace("{task}", task)
1804            } else {
1805                return StepResult {
1806                    name: format!("[agent] {def_name}"),
1807                    status: StepStatus::Failed,
1808                    message: format!("definition '{def_name}' not found"),
1809                };
1810            }
1811        } else {
1812            task.to_owned()
1813        };
1814
1815        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1816    }
1817
1818    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1819        let Some(ep) = self.endpoints.get(agent_name) else {
1820            return StepResult {
1821                name: format!("[agent] {display_name}"),
1822                status: StepStatus::Failed,
1823                message: format!("agent endpoint '{agent_name}' not found"),
1824            };
1825        };
1826
1827        if ep.url.is_empty() {
1828            return StepResult {
1829                name: format!("[agent] {display_name}"),
1830                status: StepStatus::Failed,
1831                message: format!("agent endpoint '{agent_name}' has no URL"),
1832            };
1833        }
1834
1835        self.reporter.debug(format!("agent {agent_name}: {task}"));
1836
1837        let url = ep.url.clone();
1838        let client = A2aClient::new(&url, self.timeout);
1839        let task_clone = task.to_owned();
1840
1841        let response = std::thread::spawn(move || {
1842            let rt = tokio::runtime::Builder::new_current_thread()
1843                .enable_all()
1844                .build()
1845                .unwrap();
1846            rt.block_on(client.send_task(&task_clone))
1847        })
1848        .join()
1849        .unwrap();
1850
1851        // Record the flat-cost call
1852        self.usage.record_flat_call(agent_name, ep);
1853
1854        match response {
1855            Ok(text) => {
1856                let clean = text.trim().to_owned();
1857                if verdict_is_pass(&clean) {
1858                    StepResult {
1859                        name: format!("[agent] {display_name}"),
1860                        status: StepStatus::Passed,
1861                        message: format!("PASS: {clean}"),
1862                    }
1863                } else if verdict_is_fail(&clean) {
1864                    StepResult {
1865                        name: format!("[agent] {display_name}"),
1866                        status: StepStatus::Failed,
1867                        message: clean,
1868                    }
1869                } else {
1870                    StepResult {
1871                        name: format!("[agent] {display_name}"),
1872                        status: StepStatus::Passed,
1873                        message: format!("response: {clean}"),
1874                    }
1875                }
1876            }
1877            Err(e) => StepResult {
1878                name: format!("[agent] {display_name}"),
1879                status: StepStatus::Failed,
1880                message: format!("agent call failed: {e}"),
1881            },
1882        }
1883    }
1884
1885    /// Runs an MCP tool call step.
1886    fn run_mcp(
1887        &self,
1888        server_name: &str,
1889        tool_name: &str,
1890        args: Option<&serde_json::Value>,
1891    ) -> StepResult {
1892        let Some(ep) = self.endpoints.get(server_name) else {
1893            return StepResult {
1894                name: format!("[mcp] {server_name}:{tool_name}"),
1895                status: StepStatus::Failed,
1896                message: format!("MCP server endpoint '{server_name}' not found"),
1897            };
1898        };
1899
1900        let cmd = ep.command.as_deref().unwrap_or("");
1901        if cmd.is_empty() {
1902            return StepResult {
1903                name: format!("[mcp] {server_name}:{tool_name}"),
1904                status: StepStatus::Failed,
1905                message: format!("MCP server '{server_name}' has no command configured"),
1906            };
1907        }
1908
1909        self.reporter
1910            .debug(format!("mcp {server_name} {tool_name}"));
1911
1912        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1913
1914        let command = cmd.to_owned();
1915        let args_vec = ep.args.clone();
1916        let tool = tool_name.to_owned();
1917
1918        let response = std::thread::spawn(move || {
1919            let mut mcp_client =
1920                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1921            mcp_client
1922                .call_tool(&tool, &args_val)
1923                .map_err(|e| e.to_string())
1924        })
1925        .join()
1926        .unwrap();
1927
1928        // Record the flat-cost call
1929        self.usage.record_flat_call(server_name, ep);
1930
1931        match response {
1932            Ok(result) => {
1933                if result.isError {
1934                    StepResult {
1935                        name: format!("[mcp] {server_name}:{tool_name}"),
1936                        status: StepStatus::Failed,
1937                        message: result.to_string(),
1938                    }
1939                } else {
1940                    StepResult {
1941                        name: format!("[mcp] {server_name}:{tool_name}"),
1942                        status: StepStatus::Passed,
1943                        message: result.to_string(),
1944                    }
1945                }
1946            }
1947            Err(e) => StepResult {
1948                name: format!("[mcp] {server_name}:{tool_name}"),
1949                status: StepStatus::Failed,
1950                message: format!("MCP call failed: {e}"),
1951            },
1952        }
1953    }
1954
1955    // ── helpers ──────────────────────────────────────────────────────────
1956
1957    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1958    /// the runner's default LLM config for any unset fields.
1959    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1960        LlmConfig {
1961            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
1962                // Bedrock builds its endpoint from the resolved AWS region
1963                // when no URL is given — never inherit the default LLM URL.
1964                endpoint.url.clone()
1965            } else if endpoint.url.is_empty() {
1966                self.llm.url.clone()
1967            } else {
1968                endpoint.url.clone()
1969            },
1970            model: endpoint
1971                .model
1972                .clone()
1973                .unwrap_or_else(|| self.llm.model.clone()),
1974            api_key: endpoint
1975                .api_key
1976                .clone()
1977                .or_else(|| self.llm.api_key.clone()),
1978            headers: if endpoint.headers.is_empty() {
1979                self.llm.headers.clone()
1980            } else {
1981                endpoint.headers.clone()
1982            },
1983            timeout: self.llm.timeout,
1984            temperature: self.llm.temperature,
1985            thinking: self.llm.thinking,
1986            model_params: self.llm.model_params.clone(),
1987            cache: endpoint.cache_markers,
1988            max_attempts: endpoint.max_attempts.max(1),
1989            provider: endpoint.provider,
1990            deployment: endpoint.deployment.clone(),
1991            api_version: endpoint.api_version.clone(),
1992            auth: endpoint.auth.clone(),
1993            header_commands: endpoint.header_commands.clone(),
1994            aws: endpoint.aws.clone(),
1995        }
1996    }
1997
1998    /// Runs a single LLM call against an ordered endpoint chain (primary +
1999    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
2000    /// the first endpoint that answers wins. Returns the response together
2001    /// with the chain index of the answering endpoint (0 = primary) so the
2002    /// caller can attribute usage to the correct endpoint.
2003    ///
2004    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
2005    /// shows duration, tokens, cost and the answering endpoint per call.
2006    /// Run-level context handed to every LLM call as the FIRST block of the
2007    /// user message. Contents that are stable for the whole run ("run
2008    /// started", "target site") come first so upstream provider prefix
2009    /// caching stays effective; the current time is the last line because
2010    /// it changes on every call.
2011    fn run_context(&self) -> String {
2012        let mut parts = vec![
2013            "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
2014            "================================================================".into(),
2015            format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
2016        ];
2017        if let Some(base) = self.config.base_url.as_deref() {
2018            parts.push(format!("Target site: {base}"));
2019        }
2020        let now = SystemTime::now()
2021            .duration_since(UNIX_EPOCH)
2022            .map_or(0, |d| d.as_secs());
2023        parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
2024        parts.join("\n")
2025    }
2026
2027    #[allow(clippy::cast_possible_truncation, clippy::too_many_lines)]
2028    fn llm_call_chain(
2029        &self,
2030        chain: &[&ResolvedEndpoint],
2031        system: &str,
2032        user: &str,
2033        image: Option<&[String]>,
2034        purpose: &str,
2035    ) -> Result<(crate::costs::LlmResponse, usize), String> {
2036        if chain.is_empty() {
2037            return Err("empty LLM endpoint chain".into());
2038        }
2039        let primary = self.build_llm_for_endpoint(chain[0]);
2040        let fallbacks: Vec<LlmConfig> = chain[1..]
2041            .iter()
2042            .map(|e| self.build_llm_for_endpoint(e))
2043            .collect();
2044
2045        let (test, index) = self
2046            .current_step
2047            .borrow()
2048            .as_ref()
2049            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
2050        let primary_endpoint = chain[0].name.clone();
2051        let primary_model = primary.model.clone();
2052        self.emit_event(&TestEvent::LlmCallStarted {
2053            test: test.clone(),
2054            index,
2055            endpoint: primary_endpoint.clone(),
2056            model: primary_model.clone(),
2057            purpose: purpose.to_owned(),
2058        });
2059
2060        let started = Instant::now();
2061        let sys = system.to_owned();
2062        let context = self.run_context();
2063        let user = if context.is_empty() {
2064            user.to_owned()
2065        } else {
2066            format!("{context}\n\n{user}")
2067        };
2068        let image = image.map(<[String]>::to_vec);
2069
2070        let result = std::thread::spawn(move || {
2071            let rt = tokio::runtime::Builder::new_current_thread()
2072                .enable_all()
2073                .build()
2074                .unwrap();
2075            let call = async {
2076                match image.as_deref() {
2077                    Some(img) => {
2078                        llm_chat_vision_with_usage_chain(
2079                            &primary,
2080                            &fallbacks,
2081                            &sys,
2082                            &user,
2083                            Some(img),
2084                        )
2085                        .await
2086                    }
2087                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2088                }
2089            };
2090            rt.block_on(call)
2091        })
2092        .join()
2093        .unwrap();
2094
2095        let duration_ms = started.elapsed().as_millis() as u64;
2096        match result {
2097            Ok((lr, idx)) => {
2098                let cost = calculate_llm_cost(chain[idx], &lr.usage);
2099                let answering = chain[idx].name.clone();
2100                let model = chain[idx]
2101                    .model
2102                    .clone()
2103                    .unwrap_or_else(|| primary_model.clone());
2104                self.usage
2105                    .record_llm_call(&answering, chain[idx], &model, &lr.usage);
2106                self.emit_event(&TestEvent::LlmCallFinished {
2107                    test,
2108                    index,
2109                    endpoint: answering,
2110                    model,
2111                    purpose: purpose.to_owned(),
2112                    ok: true,
2113                    duration_ms,
2114                    input_tokens: lr.usage.prompt_tokens,
2115                    output_tokens: lr.usage.completion_tokens,
2116                    cached_input_tokens: lr.usage.cached_input_tokens,
2117                    cache_creation_input_tokens: lr.usage.cache_creation_input_tokens,
2118                    cost,
2119                    error: None,
2120                });
2121                Ok((lr, idx))
2122            }
2123            Err(e) => {
2124                self.emit_event(&TestEvent::LlmCallFinished {
2125                    test,
2126                    index,
2127                    endpoint: primary_endpoint,
2128                    model: primary_model,
2129                    purpose: purpose.to_owned(),
2130                    ok: false,
2131                    duration_ms,
2132                    input_tokens: 0,
2133                    output_tokens: 0,
2134                    cached_input_tokens: 0,
2135                    cache_creation_input_tokens: 0,
2136                    cost: 0.0,
2137                    error: Some(e.clone()),
2138                });
2139                Err(e)
2140            }
2141        }
2142    }
2143
2144    /// Resolves a CSS selector for the target element. Uses the explicit
2145    /// `selector` if provided, otherwise asks the LLM to find the element
2146    /// from the natural language `target` description and page DOM.
2147    ///
2148    /// LLM responses are sanitized and verified against the live page: a
2149    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2150    /// immediately with the raw LLM output, and a selector that matches
2151    /// nothing triggers one retry with feedback before failing.
2152    #[allow(clippy::too_many_lines)]
2153    fn resolve_selector(
2154        &self,
2155        css_override: Option<&str>,
2156        target: &str,
2157        step_endpoint: Option<&str>,
2158        test_endpoint: Option<&str>,
2159        tab: &Tab,
2160    ) -> Result<String, String> {
2161        if let Some(explicit) = css_override {
2162            return Ok(explicit.to_owned());
2163        }
2164
2165        let dom_info = extract_dom_info(tab)?;
2166        let page_content = get_page_text(tab);
2167
2168        let system = concat!(
2169            "You are a browser automation selector generator. ",
2170            "Given a web page's content and interactive elements, ",
2171            "return ONLY the best CSS selector for the described element. ",
2172            "Output nothing except the CSS selector. ",
2173            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2174            "[name=\"...\"], tag.class, tag. ",
2175            "Never output explanations, markdown, or extra text."
2176        );
2177
2178        let user = format!(
2179            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2180            page_content.url,
2181            page_content.title,
2182            truncate(&page_content.body_text, 4000),
2183            dom_info,
2184            target,
2185        );
2186
2187        let retry_user = format!(
2188            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2189            "The selector must match at least one element currently present on the page.",
2190            page_content.url,
2191            page_content.title,
2192            truncate(&page_content.body_text, 4000),
2193            dom_info,
2194            target,
2195        );
2196
2197        self.reporter.debug(format!("LLM targeting: {target}"));
2198
2199        let chain = self
2200            .endpoints
2201            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2202        let sys = system.to_owned();
2203
2204        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2205
2206        let first = call_llm(&user);
2207        let (lr, _idx) = match first {
2208            Ok(lr) => lr,
2209            Err(e) => {
2210                return Err(format!("LLM element targeting failed: {e}"));
2211            }
2212        };
2213        let clean = sanitize_selector(&lr.content);
2214        self.reporter.debug(format!("resolved selector: {clean}"));
2215
2216        if selector_is_useless(&clean) {
2217            return Err(format!(
2218                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2219                raw = lr.content.trim(),
2220            ));
2221        }
2222        if let Err(reason) = validate_selector(&clean) {
2223            return Err(format!(
2224                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2225                raw = lr.content.trim(),
2226            ));
2227        }
2228        if !selector_matches(tab, &clean).unwrap_or(false) {
2229            // One retry with feedback: flaky models occasionally invent a
2230            // selector that does not exist on the page.
2231            self.reporter.warn(format!(
2232                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2233            ));
2234            let second = call_llm(&retry_user);
2235            let (lr2, _idx2) = match second {
2236                Ok(lr2) => lr2,
2237                Err(e) => {
2238                    return Err(format!(
2239                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2240                    ));
2241                }
2242            };
2243            let clean2 = sanitize_selector(&lr2.content);
2244            self.reporter
2245                .debug(format!("resolved selector (retry): {clean2}"));
2246            if selector_is_useless(&clean2) {
2247                return Err(format!(
2248                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2249                    raw = lr2.content.trim(),
2250                    excerpt = truncate(&page_content.body_text, 300),
2251                ));
2252            }
2253            if !selector_matches(tab, &clean2).unwrap_or(false) {
2254                return Err(format!(
2255                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2256                ));
2257            }
2258            return Ok(clean2);
2259        }
2260
2261        Ok(clean)
2262    }
2263}
2264
2265/// Evaluates a JS expression that is expected to return a boolean.
2266fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2267    tab.evaluate(js, false)
2268        .map_err(|e| format!("evaluate failed: {e}"))?
2269        .value
2270        .and_then(|v| v.as_bool())
2271        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2272}
2273
2274/// Checks whether a CSS selector matches at least one current element.
2275fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2276    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2277}
2278
2279// ── Free helper functions ──────────────────────────────────────────────
2280
2281fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2282    let name = format!("[navigate] {full_url}");
2283    match tab.navigate_to(full_url) {
2284        Ok(_) => {
2285            let _ = tab.wait_until_navigated();
2286            StepResult {
2287                name,
2288                status: StepStatus::Passed,
2289                message: format!("navigated to {full_url}"),
2290            }
2291        }
2292        Err(e) => StepResult {
2293            name,
2294            status: StepStatus::Failed,
2295            message: format!("navigation failed: {e}"),
2296        },
2297    }
2298}
2299
2300fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2301    let result = tab
2302        .evaluate(DOM_EXTRACT_JS, false)
2303        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2304
2305    let json_str = result
2306        .value
2307        .as_ref()
2308        .and_then(|v| v.as_str())
2309        .unwrap_or("[]");
2310
2311    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2312
2313    if elements.is_empty() {
2314        return Ok("(no interactive elements found)".to_owned());
2315    }
2316
2317    Ok(elements.join("\n"))
2318}
2319
2320fn get_page_text(tab: &Tab) -> PageContent {
2321    let url = tab.get_url();
2322
2323    let title = tab
2324        .evaluate("document.title", false)
2325        .ok()
2326        .and_then(|r| r.value)
2327        .and_then(|v| v.as_str().map(String::from))
2328        .unwrap_or_else(|| "unknown".to_owned());
2329
2330    let body_text = tab
2331        .evaluate(
2332            "document.body ? document.body.innerText : document.documentElement.innerText",
2333            false,
2334        )
2335        .ok()
2336        .and_then(|r| r.value)
2337        .and_then(|v| v.as_str().map(String::from))
2338        .unwrap_or_default();
2339
2340    PageContent {
2341        url,
2342        title,
2343        body_text: truncate(&body_text, 8000),
2344    }
2345}
2346
2347fn resolve_url(url: &str, base_url: &str) -> String {
2348    if url.starts_with("http://") || url.starts_with("https://") {
2349        return url.to_owned();
2350    }
2351    let base = base_url.trim_end_matches('/');
2352    if url.starts_with('/') {
2353        format!("{base}{url}")
2354    } else {
2355        format!("{base}/{url}")
2356    }
2357}
2358
2359/// Human-readable label for a step, used when steps are skipped after an
2360/// earlier failure.
2361fn step_label(step: &TestStep) -> String {
2362    match step {
2363        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2364        TestStep::Click { target, .. } => format!("[click] {target}"),
2365        TestStep::Type { target, .. } => format!("[type] {target}"),
2366        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2367        TestStep::Assert {
2368            definition,
2369            preset,
2370            prompt,
2371            ..
2372        } => definition.as_ref().map_or_else(
2373            || {
2374                preset.as_ref().map_or_else(
2375                    || {
2376                        prompt.as_ref().map_or_else(
2377                            || "[assert]".to_owned(),
2378                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2379                        )
2380                    },
2381                    |p| format!("[assert] {p}"),
2382                )
2383            },
2384            |d| format!("[assert] {d}"),
2385        ),
2386        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2387        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2388        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2389    }
2390}
2391
2392/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2393#[must_use]
2394const fn step_kind_label(step: &TestStep) -> &'static str {
2395    match step {
2396        TestStep::Navigate { .. } => "navigate",
2397        TestStep::Click { .. } => "click",
2398        TestStep::Type { .. } => "type",
2399        TestStep::Wait { .. } => "wait",
2400        TestStep::Assert { .. } => "assert",
2401        TestStep::Screenshot { .. } => "screenshot",
2402        TestStep::Agent { .. } => "agent",
2403        TestStep::Mcp { .. } => "mcp",
2404    }
2405}
2406
2407// ── Support types ──────────────────────────────────────────────────────
2408
2409#[derive(Default)]
2410struct TestRunResult {
2411    passed: u32,
2412    failed: u32,
2413    skipped: u32,
2414    total: u32,
2415    details: Vec<StepResult>,
2416}
2417
2418struct PageContent {
2419    url: String,
2420    title: String,
2421    body_text: String,
2422}
2423
2424#[cfg(test)]
2425mod tests {
2426    use super::unix_to_rfc3339;
2427
2428    #[test]
2429    fn rfc3339_epoch_and_reference_dates() {
2430        assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2431        assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2432        assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2433        assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2434    }
2435
2436    #[test]
2437    fn rfc3339_handles_leap_years() {
2438        assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2439        assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2440    }
2441}
2442
2443#[cfg(test)]
2444mod verdict_parse_tests {
2445    use super::{verdict_is_fail, verdict_is_pass};
2446
2447    #[test]
2448    fn tolerates_markdown_punctuation_and_natural_language() {
2449        assert!(verdict_is_pass("PASS"));
2450        assert!(verdict_is_pass("pass"));
2451        assert!(verdict_is_pass("**PASS**\n\nThe page is clean."));
2452        assert!(verdict_is_pass("passes - no explicit error visible"));
2453        assert!(verdict_is_pass("  \"pass\""));
2454        assert!(!verdict_is_pass("FAIL: something broke"));
2455        assert!(!verdict_is_pass("**FAIL** broken"));
2456
2457        assert!(verdict_is_fail("**FAIL** broken"));
2458        assert!(verdict_is_fail("fails - error toast shown"));
2459        assert!(!verdict_is_fail("passes - ok"));
2460    }
2461}