Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// Formats a UNIX epoch timestamp as an RFC 3339 UTC string
26/// (`2026-09-19T16:04:00Z`) without pulling in a date dependency.
27/// Civil-from-days conversion after Howard Hinnant.
28#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31    let secs = i64::try_from(secs).unwrap_or(0);
32    let days = secs.div_euclid(86_400);
33    let rem = secs.rem_euclid(86_400);
34    let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35    let z = days + 719_468;
36    let era = z.div_euclid(146_097);
37    let doe = z.rem_euclid(146_097);
38    let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40    let mp = (5 * doy + 2) / 153;
41    let d = doy - (153 * mp + 2) / 5 + 1;
42    let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43    let y = yoe + era * 400 + i64::from(mth <= 2);
44    format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47/// One detected layout defect (`layout_no_issues` preset).
48#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50    #[serde(rename = "type")]
51    issue_type: String,
52    element: String,
53    detail: String,
54}
55
56/// In-browser DOM layout scan for `layout_no_issues`.
57///
58/// Geometry-only checks (no LLM, no pixels):
59/// 1. page horizontal overflow (`scrollWidth` > viewport width);
60/// 2. elements outside the viewport that scrolling cannot reveal
61///    (fixed elements off-screen, left/negative overflow, right-edge
62///    overflow beyond the horizontally scrollable area, and bottom
63///    overflow on a page that cannot scroll down) — below-the-fold
64///    content on a tall scrollable page is normal flow, NOT a defect;
65/// 3. text clipped by `overflow: hidden` containers whose content
66///    is measurably larger than the box;
67/// 4. interactive elements (buttons/links/inputs) whose center point
68///    is covered by a different element that would intercept the click.
69///
70/// Elements whose class matches a configured ignore prefix (default:
71/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
72/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
73/// skipped — they are intentionally 1x1 / off-screen. The prefix list
74/// is injected at `__IGNORE_CLASSES__` from
75/// [`ScenarioConfig::layout_ignore_classes`].
76///
77/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
78/// sticky headers) is excluded by the position/relation filters.
79const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81  const issues = [];
82  const push = (type, el, detail) => {
83    if (issues.length >= 30) return;
84    let element = el.tagName.toLowerCase();
85    if (el.id) element += '#' + el.id;
86    else if (typeof el.className === 'string' && el.className.trim())
87      element += '.' + el.className.trim().split(/\s+/).join('.');
88    issues.push({ type, element, detail: String(detail).slice(0, 220) });
89  };
90  const vw = document.documentElement.clientWidth || window.innerWidth;
91  const vh = document.documentElement.clientHeight || window.innerHeight;
92  if (!vw || !vh) return JSON.stringify(issues);
93  const de = document.documentElement;
94  // 1. Page-level horizontal overflow.
95  if (maxSW > vw + 2)
96    push('page-overflow-x', de,
97      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98  // Class-name prefixes to skip (injected; default = Angular CDK
99  // screen-reader helpers, which are intentionally 1x1 / off-screen).
100  const ignorePrefixes = __IGNORE_CLASSES__;
101  const isIgnored = (el) => {
102    if (typeof el.className !== 'string' || !el.className.trim()) return false;
103    const classes = el.className.trim().split(/\s+/);
104    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105  };
106  const body = document.body;
107  // The document element is not always the scroll container: the app may
108  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110  // "not scrollable" — measure the scrollable content height/width of the
111  // document element AND the body, and take the max.
112  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114  const vScrollable = maxSH > vh + 2;
115  const hScrollable = maxSW > vw + 2;
116  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117  // this element into view on the given axis — i.e. the element is inside a
118  // scrolling region, so lying beyond the viewport cut is not a defect.
119  const reachableByScroller = (el, axis) => {
120    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123    let p = el.parentElement;
124    while (p && p !== body) {
125      const pcs = getComputedStyle(p);
126      const o = pcs[ovProp];
127      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128          p[dim] > p[dimClient] + 2) return true;
129      p = p.parentElement;
130    }
131    return false;
132  };
133  const selfOverflowing = (el, cs, axis) => {
134    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138      el[dim] > el[dimClient] + 2;
139  };
140  // True when an ancestor clips this axis with overflow:hidden/clip — the
141  // element's overhang is not visible, so treat it as reachable.
142  const clippedByAncestor = (el, axis) => {
143    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144    let p = el.parentElement;
145    while (p && p !== body) {
146      const o = getComputedStyle(p)[ovProp];
147      if (o === 'hidden' || o === 'clip') return true;
148      p = p.parentElement;
149    }
150    return false;
151  };
152  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154  const hasContent = (el) =>
155    ((el.textContent || '').trim().length > 0) ||
156    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157  // 2. Elements outside the viewport that scrolling cannot reveal.
158  for (const el of all) {
159    const cs = getComputedStyle(el);
160    if (!visible(cs)) continue;
161    if (isIgnored(el)) continue;
162    const r = el.getBoundingClientRect();
163    if (r.width < 2 || r.height < 2) continue;
164    if (!hasContent(el) && el.children.length === 0) continue;
165    if (cs.position === 'fixed') {
166      // Fixed elements never move with the scroll: any edge outside the
167      // viewport is unreachable content and therefore a defect.
168      const overTop = -r.top;
169      const overLeft = -r.left;
170      const overRight = r.right - vw;
171      const overBottom = r.bottom - vh;
172      // Top/left overshoot is never reachable; bottom/right overshoot is
173      // only a defect when the fixed element cannot scroll that content
174      // into view itself (e.g. an opened Material drawer whose inner
175      // container scrolls is not a "cut-off" defect).
176      if (overTop > 2 || overLeft > 2 ||
177          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179        let where = '';
180        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186        push('element-out-of-viewport', el,
187          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188      }
189      continue;
190    }
191    if (cs.position === 'sticky') continue;
192    // Negative left overflow cannot be reached by scrolling (scrollLeft
193    // never goes below 0).
194    if (r.left < -2) {
195      push('element-out-of-viewport', el,
196        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197      continue;
198    }
199    // Negative top with the page at the top means the element sits above
200    // the document origin — also unreachable.
201    if (r.top < -2 && de.scrollTop <= 2) {
202      push('element-out-of-viewport', el,
203        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204      continue;
205    }
206    const overRight = r.right - vw;
207    // Right-edge overflow is only a defect when the page cannot scroll
208    // horizontally to reveal it (or the element sticks out past the
209    // scrollable content width itself).
210    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211      push('element-out-of-viewport', el,
212        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213      continue;
214    }
215    // Below-the-fold content on a scrollable page is normal (tall landing
216    // pages); only flag bottom overflow the user can never scroll to.
217    const overBottom = r.bottom - vh;
218    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219      push('element-out-of-viewport', el,
220        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221    }
222  }
223  // 3. Text clipped by overflow:hidden containers.
224  for (const el of all) {
225    if (isIgnored(el)) continue;
226    const cs = getComputedStyle(el);
227    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229    if (!(el.textContent || '').trim()) continue;
230    // Single-line truncation with an ellipsis is an intentional design
231    // pattern (Tailwind .truncate etc.), not a clipping defect.
232    if (cs.textOverflow === 'ellipsis') continue;
233    push('text-clipped', el,
234      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236  }
237  // 4. Interactive elements covered by a different element.
238  const interactive =
239    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240  const targets = document.querySelectorAll(interactive);
241  for (const el of targets) {
242    if (isIgnored(el)) continue;
243    const r = el.getBoundingClientRect();
244    if (r.width < 6 || r.height < 6) continue;
245    const cs = getComputedStyle(el);
246    if (!visible(cs)) continue;
247    const cx = r.left + r.width / 2;
248    const cy = r.top + r.height / 2;
249    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250    const top = document.elementFromPoint(cx, cy);
251    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252    if (isIgnored(top)) continue;
253    const tcs = getComputedStyle(top);
254    if (!visible(tcs)) continue;
255    if (tcs.pointerEvents === 'none') continue;
256    const tr = top.getBoundingClientRect();
257    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260    push('element-overlap', el,
261      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262      ' is covered by <' + tname + '>');
263  }
264  return JSON.stringify(issues);
265})()
266"#;
267
268/// How long the CDP connection stays open after the browser goes quiet.
269///
270/// `headless_chrome` ships a 30s default and tears down the entire connection
271/// when no traffic arrives for that long; a run must own its connection for
272/// its full duration instead.
273const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275/// Executes a [`Scenario`] against a real browser with optional LLM
276/// assistance for element targeting and assertions.
277pub struct ScenarioRunner {
278    config: ScenarioConfig,
279    definitions: HashMap<String, AssertDefinition>,
280    llm: LlmConfig,
281    timeout: Duration,
282    viewport_width: u32,
283    viewport_height: u32,
284    /// The viewport currently applied in the browser (CDP emulation).
285    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
286    applied_viewport: std::cell::Cell<(u32, u32)>,
287    endpoints: EndpointRegistry,
288    usage: Arc<UsageTracker>,
289    budgets: BudgetTracker,
290    /// Directory for failure artifacts (screenshots).
291    artifacts_dir: PathBuf,
292    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
293    reporter: Arc<Reporter>,
294    /// The test + step index currently executing, for LLM-call events.
295    current_step: std::cell::RefCell<Option<(String, u32)>>,
296    /// Epoch seconds of runner creation — constant for the whole run so
297    /// the LLM context preamble below is cacheable by upstream providers.
298    run_started: u64,
299    /// Whether this runner emits the `RunStarted`/`RunFinished` events.
300    /// When several scenario files are run concurrently by
301    /// [`crate::parallel::run_scenarios`], the orchestrator owns those two
302    /// events (one per batch), so each per-file runner suppresses its own.
303    emit_run_events: bool,
304}
305
306/// Aggregated results from a scenario run.
307#[derive(Debug, Default)]
308pub struct RunReport {
309    /// Number of tests that passed.
310    pub tests_passed: u32,
311    /// Number of tests that failed.
312    pub tests_failed: u32,
313    /// Number of steps that passed.
314    pub passed: u32,
315    /// Number of steps that failed.
316    pub failed: u32,
317    /// Number of steps that were skipped.
318    pub skipped: u32,
319    /// Per-step details.
320    pub details: Vec<StepResult>,
321}
322
323/// Result of a single step execution.
324#[derive(Debug, Clone)]
325pub struct StepResult {
326    /// The step name.
327    pub name: String,
328    /// Whether the step passed, failed, or was skipped.
329    pub status: StepStatus,
330    /// Human-readable result message.
331    pub message: String,
332}
333
334/// Outcome for a single step.
335pub use crate::events::StepStatus;
336
337/// Predefined assertion preset definition.
338struct AssertPreset {
339    name: &'static str,
340    system: &'static str,
341    user_template: &'static str,
342}
343
344/// Built-in assertion presets.
345#[allow(clippy::literal_string_with_formatting_args)]
346const ASSERTION_PRESETS: &[AssertPreset] = &[
347    AssertPreset {
348        name: "no_error_on_page",
349        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
350        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
351    },
352    AssertPreset {
353        name: "text_visible",
354        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
355        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
356    },
357    AssertPreset {
358        name: "element_exists",
359        system: "You are a QA tester. Check if a described UI element exists on a web page.",
360        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
361    },
362    AssertPreset {
363        name: "visual_no_issues",
364        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
365        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
366    },
367    AssertPreset {
368        name: "visual_no_overlaps",
369        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
370        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
371    },
372    AssertPreset {
373        name: "visual_fills_viewport",
374        system: "You are a visual QA engineer inspecting a website screenshot for UNDER-FILLED layouts: the rendered content stops at a fraction of the viewport and the remaining area is a large blank region the layout should have filled. Typical defects: page content ending around the halfway mark of the viewport with an empty, background-only area below it; an app shell, dashboard, or main panel collapsing to part of its expected height and leaving dead space; content that fills only the left portion of a wide viewport with unused space to the right. Do NOT fail intentionally short or centered pages: a login or signup card, a landing hero, a short article, a confirmation step, or an empty state that does not fill the viewport is normal design. Only fail when the blank region is clearly a layout defect — the page looks cut off or collapsed, or a content container (calendar, table, chart, map, panel) visibly fails to stretch to the space its own layout provides.",
375        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nDoes the page content fill the viewport as its layout intends — with no large blank region where the page appears to end partway down or partway across? Respond with exactly \"PASS\" if the page fills the viewport or is intentionally short, or \"FAIL: <describe the under-filled region and where it appears>\" if the content clearly stops at a fraction of the available page.",
376    },
377    AssertPreset {
378        name: "visual_text_visible",
379        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
380        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
381    },
382    AssertPreset {
383        name: "layout_no_issues",
384        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
385        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
386    },
387];
388
389/// Whether an LLM verdict should be read as PASS.
390///
391/// Models routinely wrap the verdict in markdown (`**PASS** …`) or prefix it
392/// with punctuation, which a bare `starts_with("pass")` misreads as a failure.
393/// Strip any leading non-alphanumerics, then match the leading word; this also
394/// accepts the natural-language forms `pass` / `passes`.
395fn verdict_is_pass(content: &str) -> bool {
396    content
397        .trim()
398        .trim_start_matches(|c: char| !c.is_alphanumeric())
399        .to_lowercase()
400        .starts_with("pass")
401}
402
403/// Whether an LLM verdict should be read as FAIL (see [`verdict_is_pass`]).
404fn verdict_is_fail(content: &str) -> bool {
405    content
406        .trim()
407        .trim_start_matches(|c: char| !c.is_alphanumeric())
408        .to_lowercase()
409        .starts_with("fail")
410}
411
412impl ScenarioRunner {
413    /// Creates a new runner with the given scenario configuration and
414    /// assertion definitions.
415    #[must_use]
416    #[allow(clippy::needless_pass_by_value)]
417    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
418        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
419    }
420
421    /// Creates a runner that reports run events through the given reporter.
422    #[must_use]
423    #[allow(clippy::needless_pass_by_value)]
424    pub fn with_reporter(
425        scenario_config: ScenarioConfig,
426        definitions: Vec<AssertDefinition>,
427        reporter: Arc<Reporter>,
428    ) -> Self {
429        Self::with_reporter_mode(scenario_config, definitions, reporter, true)
430    }
431
432    /// Creates a runner that reports run events through the given reporter
433    /// but suppresses the `RunStarted`/`RunFinished` events. Used by the
434    /// parallel orchestrator, which emits those once per batch.
435    #[must_use]
436    #[allow(clippy::needless_pass_by_value)]
437    pub fn with_reporter_parallel(
438        scenario_config: ScenarioConfig,
439        definitions: Vec<AssertDefinition>,
440        reporter: Arc<Reporter>,
441    ) -> Self {
442        Self::with_reporter_mode(scenario_config, definitions, reporter, false)
443    }
444
445    #[must_use]
446    #[allow(clippy::needless_pass_by_value)]
447    fn with_reporter_mode(
448        scenario_config: ScenarioConfig,
449        definitions: Vec<AssertDefinition>,
450        reporter: Arc<Reporter>,
451        emit_run_events: bool,
452    ) -> Self {
453        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
454            &scenario_config,
455        ));
456        let llm = LlmConfig {
457            url: scenario_config
458                .llm_url
459                .clone()
460                .unwrap_or_else(crate::llm_base_url),
461            model: scenario_config
462                .llm_model
463                .clone()
464                .unwrap_or_else(crate::llm_model),
465            api_key: scenario_config
466                .llm_api_key
467                .clone()
468                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
469            headers: if scenario_config.llm_headers.is_empty() {
470                crate::parse_headers_env()
471            } else {
472                scenario_config.llm_headers.clone()
473            },
474            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
475            temperature: scenario_config.temperature,
476            thinking: scenario_config.thinking,
477            model_params: scenario_config.model_params.clone(),
478            cache: scenario_config.cache.unwrap_or(true),
479            max_attempts: crate::default_llm_attempts(),
480            provider: crate::scenario::Provider::Openai,
481            deployment: None,
482            api_version: None,
483            auth: crate::scenario::AuthConfig::default(),
484            header_commands: std::collections::HashMap::new(),
485            aws: crate::scenario::AwsConfig::default(),
486        };
487        let endpoints = EndpointRegistry::from_config(
488            &scenario_config.endpoints,
489            Some(&llm),
490            crate::endpoints::EndpointDefaults {
491                cache: scenario_config.cache,
492                cache_pricing: scenario_config.cache_pricing,
493            },
494        );
495        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
496        let defs_map: HashMap<String, AssertDefinition> = definitions
497            .into_iter()
498            .map(|d| (d.name.clone(), d))
499            .collect();
500
501        Self {
502            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
503            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
504            viewport_height: scenario_config.viewport_height.unwrap_or(720),
505            applied_viewport: std::cell::Cell::new((0, 0)),
506            config: scenario_config.clone(),
507            definitions: defs_map,
508            llm,
509            endpoints,
510            usage: Arc::new(UsageTracker::new()),
511            budgets,
512            artifacts_dir: PathBuf::from(
513                scenario_config
514                    .artifacts_dir
515                    .unwrap_or_else(|| "artifacts".to_owned()),
516            ),
517            reporter,
518            current_step: std::cell::RefCell::new(None),
519            run_started: SystemTime::now()
520                .duration_since(UNIX_EPOCH)
521                .map_or(0, |d| d.as_secs()),
522            emit_run_events,
523        }
524    }
525
526    /// Emits an event; a sink failure degrades to a console warning so a
527    /// broken log file can never mask the run itself.
528    fn emit_event(&self, event: &TestEvent) {
529        if let Err(err) = self.reporter.emit(event) {
530            use std::io::Write as _;
531            let mut out = std::io::stderr().lock();
532            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
533        }
534    }
535
536    /// Returns a clone of the [`UsageTracker`] for reporting.
537    #[must_use]
538    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
539        Arc::clone(&self.usage)
540    }
541
542    /// Returns a reference to the [`BudgetTracker`].
543    #[must_use]
544    pub const fn budget_tracker(&self) -> &BudgetTracker {
545        &self.budgets
546    }
547
548    /// Executes all test groups in the scenario and returns a report.
549    ///
550    /// # Errors
551    ///
552    /// Returns an error if the browser fails to launch.
553    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
554    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
555        let mut report = RunReport::default();
556
557        if tests.is_empty() {
558            self.reporter.warn("No tests defined in scenario.");
559            if self.emit_run_events {
560                self.emit_event(&TestEvent::RunFinished {
561                    tests_passed: 0,
562                    tests_failed: 0,
563                    steps_passed: 0,
564                    steps_failed: 0,
565                    steps_skipped: 0,
566                    total_cost: 0.0,
567                    total_tokens: 0,
568                    total_input_tokens: 0,
569                    total_output_tokens: 0,
570                    total_cached_input_tokens: 0,
571                    total_cache_creation_input_tokens: 0,
572                    models: Vec::new(),
573                    total_calls: 0,
574                });
575            }
576            return Ok(report);
577        }
578
579        if self.emit_run_events {
580            self.emit_event(&TestEvent::RunStarted {
581                total_tests: tests.len() as u32,
582            });
583        }
584
585        let browser_headless = self.config.browser_headless.unwrap_or(true);
586
587        let launch_opts = LaunchOptions {
588            headless: browser_headless,
589            window_size: Some((self.viewport_width, self.viewport_height)),
590            sandbox: false,
591            // headless_chrome defaults this to 30s and shuts down the whole CDP
592            // connection when no messages arrive for that long. A scenario can
593            // easily exceed 30s of browser silence (slow LLM targeting/assertion
594            // calls, page waits, budget checks between steps), after which every
595            // remaining step fails with "Unable to make method calls because
596            // underlying connection is closed" — one quiet gap kills the run.
597            // Open-ended scenarios must own the connection for their full
598            // duration, so keep it alive for 6 hours.
599            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
600            ..LaunchOptions::default()
601        };
602
603        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
604        let tab = browser.new_tab().context("failed to open browser tab")?;
605        match (
606            self.config.browser_basic_auth_user.clone(),
607            self.config.browser_basic_auth_password.clone(),
608        ) {
609            (Some(username), Some(password)) if !username.is_empty() && !password.is_empty() => {
610                tab.authenticate(Some(username), Some(password))
611                    .context("failed to configure browser HTTP Basic Auth")?;
612                tab.enable_fetch(None, Some(true))
613                    .context("failed to enable browser HTTP authentication")?;
614            }
615            (None, None) => {}
616            _ => anyhow::bail!("browser Basic Auth requires both username and password"),
617        }
618        let _ = tab.set_default_timeout(self.timeout);
619
620        // Start MCP server if configured
621        #[cfg(feature = "mcp-server")]
622        if let Some(ref mcp_cfg) = self.config.mcp_server {
623            if mcp_cfg.enabled {
624                let port = mcp_cfg.port;
625                std::thread::spawn(move || {
626                    let _ = crate::mcp_server::start_mcp_server(port);
627                });
628            }
629        }
630        #[cfg(not(feature = "mcp-server"))]
631        if let Some(mcp_cfg) = &self.config.mcp_server {
632            if mcp_cfg.enabled {
633                self.reporter
634                    .warn("MCP server configured but 'mcp-server' feature not enabled");
635            }
636        }
637
638        // Start A2A agent server if configured
639        #[cfg(feature = "a2a-server")]
640        if let Some(ref a2a_cfg) = self.config.a2a_server {
641            if a2a_cfg.enabled {
642                let port = a2a_cfg.port;
643                tokio::spawn(crate::a2a_server::start_a2a_server(port));
644            }
645        }
646        #[cfg(not(feature = "a2a-server"))]
647        if let Some(a2a_cfg) = &self.config.a2a_server {
648            if a2a_cfg.enabled {
649                self.reporter
650                    .warn("A2A server configured but 'a2a-server' feature not enabled");
651            }
652        }
653
654        for test in tests {
655            self.emit_event(&TestEvent::TestStarted {
656                test: test.name.clone(),
657            });
658
659            self.usage.reset_per_test();
660
661            let test_started = Instant::now();
662            let test_result = self.run_test(test, &tab);
663            let duration_ms = test_started.elapsed().as_millis() as u64;
664            let usage = self.usage.current_test_snapshot();
665            self.usage.commit_test(&test.name);
666
667            self.emit_event(&TestEvent::TestFinished {
668                test: test.name.clone(),
669                passed: test_result.passed,
670                failed: test_result.failed,
671                skipped: test_result.skipped,
672                duration_ms,
673                cost: usage.total_cost,
674                tokens: usage.total_tokens,
675                input_tokens: usage.total_input_tokens,
676                output_tokens: usage.total_output_tokens,
677                cached_input_tokens: usage.total_cached_input_tokens,
678                cache_creation_input_tokens: usage.total_cache_creation_input_tokens,
679                models: usage.models.clone(),
680                calls: usage.total_calls,
681            });
682
683            if test_result.failed == 0 && test_result.total > 0 {
684                report.tests_passed += 1;
685            } else if test_result.total > 0 {
686                report.tests_failed += 1;
687            }
688
689            report.passed += test_result.passed;
690            report.failed += test_result.failed;
691            report.skipped += test_result.skipped;
692            report.details.extend(test_result.details);
693        }
694
695        let global = self.usage.global_snapshot();
696        if self.emit_run_events {
697            self.emit_event(&TestEvent::RunFinished {
698                tests_passed: report.tests_passed,
699                tests_failed: report.tests_failed,
700                steps_passed: report.passed,
701                steps_failed: report.failed,
702                steps_skipped: report.skipped,
703                total_cost: global.total_cost,
704                total_tokens: global.total_tokens,
705                total_input_tokens: global.total_input_tokens,
706                total_output_tokens: global.total_output_tokens,
707                total_cached_input_tokens: global.total_cached_input_tokens,
708                total_cache_creation_input_tokens: global.total_cache_creation_input_tokens,
709                models: global.models.clone(),
710                total_calls: global.total_calls,
711            });
712        }
713
714        Ok(report)
715    }
716
717    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
718    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
719        let base_url = test
720            .base_url
721            .clone()
722            .or_else(|| self.config.base_url.clone())
723            .unwrap_or_else(crate::base_url);
724
725        // Per-test viewport override: switch the browser via CDP
726        // device-metrics emulation before this test runs.
727        let vw = test.viewport_width.unwrap_or(self.viewport_width);
728        let vh = test.viewport_height.unwrap_or(self.viewport_height);
729        if self.applied_viewport.get() != (vw, vh) {
730            self.apply_viewport(tab, vw, vh);
731            self.applied_viewport.set((vw, vh));
732        }
733
734        // Per-test isolation: every test starts from its own start_url
735        // (unless auto_navigate is disabled), so a test never inherits the
736        // previous test's page state.
737        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
738
739        let start_url = test
740            .start_url
741            .clone()
742            .or_else(|| self.config.start_url.clone())
743            .unwrap_or_else(|| "/dashboard".to_owned());
744
745        if auto_navigate {
746            let full_url = resolve_url(&start_url, &base_url);
747            self.reporter.debug(format!("auto-navigate: {full_url}"));
748            let _ = tab.navigate_to(&full_url);
749            let _ = tab.wait_until_navigated();
750            std::thread::sleep(Duration::from_secs(4));
751        }
752
753        let mut result = TestRunResult::default();
754
755        for (step_index, step) in test.steps.iter().enumerate() {
756            result.total += 1;
757
758            let wait_ms = match step {
759                TestStep::Navigate { wait_after_ms, .. }
760                | TestStep::Click { wait_after_ms, .. }
761                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
762                _ => None,
763            };
764
765            self.current_step
766                .replace(Some((test.name.clone(), step_index as u32)));
767            self.emit_event(&TestEvent::StepStarted {
768                test: test.name.clone(),
769                index: step_index as u32,
770                label: step_label(step),
771            });
772            let step_started = Instant::now();
773
774            let mut step_result = match step {
775                TestStep::Navigate { url, .. } => {
776                    let full_url = resolve_url(url, &base_url);
777                    run_navigate_step(&full_url, tab)
778                }
779                TestStep::Click {
780                    target,
781                    selector,
782                    endpoint,
783                    idempotent,
784                    ..
785                } => self.run_click(
786                    target,
787                    selector.as_deref(),
788                    endpoint.as_deref(),
789                    test.endpoint.as_deref(),
790                    *idempotent,
791                    tab,
792                ),
793                TestStep::Type {
794                    target,
795                    text,
796                    selector,
797                    endpoint,
798                    idempotent,
799                    ..
800                } => self.run_type(
801                    target,
802                    text,
803                    selector.as_deref(),
804                    endpoint.as_deref(),
805                    test.endpoint.as_deref(),
806                    *idempotent,
807                    tab,
808                ),
809                TestStep::Wait {
810                    target,
811                    selector,
812                    text,
813                    timeout_ms,
814                    endpoint,
815                    idempotent,
816                } => self.run_wait(
817                    target,
818                    selector.as_deref(),
819                    text.as_deref(),
820                    *timeout_ms,
821                    endpoint.as_deref(),
822                    test.endpoint.as_deref(),
823                    *idempotent,
824                    tab,
825                ),
826                TestStep::Assert {
827                    definition,
828                    preset,
829                    prompt,
830                    assert_text,
831                    endpoint,
832                    screenshot,
833                } => self.run_assert(
834                    definition.as_deref(),
835                    preset.as_deref(),
836                    prompt.as_deref(),
837                    assert_text.as_deref(),
838                    *screenshot,
839                    endpoint.as_deref(),
840                    test.endpoint.as_deref(),
841                    tab,
842                ),
843                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
844                TestStep::Agent {
845                    agent,
846                    task,
847                    definition,
848                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
849                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
850            };
851
852            // Failure diagnostics: capture the page state and a screenshot so
853            // CI logs say WHAT the page looked like when the step failed,
854            // instead of a bare "timed out: The event waited for never came".
855            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
856                let state = diagnostics::capture(tab);
857                let screenshot = diagnostics::save_screenshot(
858                    tab,
859                    &self.artifacts_dir,
860                    &test.name,
861                    &test.name,
862                    step_index,
863                    step_kind_label(step),
864                );
865                step_result.message = format!(
866                    "{base} — {excerpt}",
867                    base = step_result.message,
868                    excerpt = diagnostics::inline_excerpt(&state),
869                );
870                (Some(diagnostics::full_context(&state)), screenshot)
871            } else {
872                (None, None)
873            };
874
875            let duration_ms = step_started.elapsed().as_millis() as u64;
876            self.emit_event(&TestEvent::StepFinished {
877                test: test.name.clone(),
878                index: step_index as u32,
879                label: step_result.name.clone(),
880                status: step_result.status,
881                duration_ms,
882                message: step_result.message.clone(),
883                diagnostics: diagnostics_block,
884                screenshot: screenshot_path,
885            });
886            self.current_step.replace(None);
887
888            match step_result.status {
889                StepStatus::Passed => result.passed += 1,
890                StepStatus::Failed => result.failed += 1,
891                StepStatus::Skipped => result.skipped += 1,
892            }
893
894            // Fail fast: the first failed step ends the test and the
895            // remaining steps are reported as skipped (no LLM budget is
896            // burned asserting against a page that is already known broken).
897            if step_result.status == StepStatus::Failed
898                && !self.config.continue_on_failure
899                && step_index + 1 < test.steps.len()
900            {
901                self.emit_event(&TestEvent::Warning {
902                    message: format!(
903                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
904                        test.steps.len() - step_index - 1
905                    ),
906                });
907                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
908                    let skipped_index = step_index + 1 + offset;
909                    let label = step_label(skipped);
910                    result.total += 1;
911                    result.skipped += 1;
912                    self.emit_event(&TestEvent::StepStarted {
913                        test: test.name.clone(),
914                        index: skipped_index as u32,
915                        label: label.clone(),
916                    });
917                    self.emit_event(&TestEvent::StepFinished {
918                        test: test.name.clone(),
919                        index: skipped_index as u32,
920                        label,
921                        status: StepStatus::Skipped,
922                        duration_ms: 0,
923                        message: "skipped: previous step failed".into(),
924                        diagnostics: None,
925                        screenshot: None,
926                    });
927                    result.details.push(StepResult {
928                        name: step_label(skipped),
929                        status: StepStatus::Skipped,
930                        message: "skipped: previous step failed".into(),
931                    });
932                }
933                result.details.push(step_result);
934                return result;
935            }
936
937            // Check per-test budget after each step
938            let test_usage = self.usage.current_test_snapshot();
939            let global_usage = self.usage.global_snapshot();
940            let budget_status = self.budgets.check_all(
941                &test.name,
942                &test_usage,
943                &global_usage,
944                test.budget.as_ref(),
945            );
946            match budget_status {
947                BudgetStatus::HardExceeded { message, .. } => {
948                    self.emit_event(&TestEvent::Warning {
949                        message: format!("budget exceeded: {message}"),
950                    });
951                    result.details.push(StepResult {
952                        name: "[budget]".into(),
953                        status: StepStatus::Failed,
954                        message,
955                    });
956                    result.failed += 1;
957                    return result;
958                }
959                BudgetStatus::SoftExceeded { message, .. } => {
960                    self.emit_event(&TestEvent::Warning {
961                        message: format!("budget warning: {message}"),
962                    });
963                }
964                BudgetStatus::Ok => {}
965            }
966
967            if let Some(ms) = wait_ms {
968                std::thread::sleep(Duration::from_millis(ms));
969            }
970
971            result.details.push(step_result);
972        }
973
974        result
975    }
976
977    /// Applies a viewport size to the current tab via CDP
978    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
979    /// overrides and the viewport matrix. The initial window size set at
980    /// browser launch is replaced by emulation; failures are logged but
981    /// do not fail the test (a mismatched viewport only weakens coverage).
982    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
983        use headless_chrome::protocol::cdp::Emulation;
984        let _ = self;
985        let params = Emulation::SetDeviceMetricsOverride {
986            width,
987            height,
988            device_scale_factor: 1.0,
989            mobile: false,
990            scale: None,
991            screen_width: Some(width),
992            screen_height: Some(height),
993            position_x: None,
994            position_y: None,
995            dont_set_visible_size: None,
996            screen_orientation: None,
997            viewport: None,
998            display_feature: None,
999            device_posture: None,
1000        };
1001        self.reporter.debug(format!("viewport: {width}x{height}"));
1002        if let Err(e) = tab.call_method(params) {
1003            self.reporter
1004                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
1005        }
1006    }
1007
1008    /// Height (px) covered by assert-step screenshots: the configured
1009    /// `screenshot_max_height` (absolute px or viewport multiple, default
1010    /// `"20x"`) resolved against the currently applied viewport, raised to
1011    /// at least the viewport height so the visible screen is always fully
1012    /// included. The capture is split into viewport-tall tiles, so this
1013    /// value bounds total coverage (and hence the number of image parts).
1014    #[must_use]
1015    fn screenshot_height_cap(&self) -> u32 {
1016        let viewport_height = self.current_viewport_height();
1017        let cap = self
1018            .config
1019            .screenshot_max_height
1020            .as_ref()
1021            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
1022        cap.max(viewport_height)
1023    }
1024
1025    /// Height of the viewport currently emulated in the browser (falling
1026    /// back to the configured default before any emulation was applied).
1027    #[must_use]
1028    const fn current_viewport_height(&self) -> u32 {
1029        let (_, height) = self.applied_viewport.get();
1030        if height > 0 {
1031            height
1032        } else {
1033            self.viewport_height
1034        }
1035    }
1036
1037    // ── step handlers ───────────────────────────────────────────────────
1038
1039    #[allow(clippy::too_many_lines)]
1040    fn run_click(
1041        &self,
1042        target: &str,
1043        selector_override: Option<&str>,
1044        step_endpoint: Option<&str>,
1045        test_endpoint: Option<&str>,
1046        idempotent: bool,
1047        tab: &Tab,
1048    ) -> StepResult {
1049        let name = format!("[click] {target}");
1050        let selector = match self.resolve_selector(
1051            selector_override,
1052            target,
1053            step_endpoint,
1054            test_endpoint,
1055            tab,
1056        ) {
1057            Ok(s) => s,
1058            Err(msg) => {
1059                if idempotent {
1060                    return StepResult {
1061                        name,
1062                        status: StepStatus::Skipped,
1063                        message: format!("skipped (idempotent): no target found — {msg}"),
1064                    };
1065                }
1066                return StepResult {
1067                    name,
1068                    status: StepStatus::Failed,
1069                    message: msg,
1070                };
1071            }
1072        };
1073
1074        // Idempotent steps probe briefly: a missing target means the
1075        // action was already done / not applicable (e.g. an
1076        // already-authenticated session), and skipping is the success
1077        // path, not a failure.
1078        let probe_secs = if idempotent { 5 } else { 10 };
1079        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1080            Ok(element) => match element.click() {
1081                Ok(_) => StepResult {
1082                    name,
1083                    status: StepStatus::Passed,
1084                    message: format!("clicked {selector}"),
1085                },
1086                Err(e) => StepResult {
1087                    name,
1088                    status: StepStatus::Failed,
1089                    message: format!("click failed on {selector}: {e}"),
1090                },
1091            },
1092            Err(e) if idempotent => StepResult {
1093                name,
1094                status: StepStatus::Skipped,
1095                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1096            },
1097            Err(e) => StepResult {
1098                name,
1099                status: StepStatus::Failed,
1100                message: format!("element {selector} not found: {e}"),
1101            },
1102        }
1103    }
1104
1105    #[allow(clippy::too_many_arguments)]
1106    fn run_type(
1107        &self,
1108        target: &str,
1109        text: &str,
1110        selector_override: Option<&str>,
1111        step_endpoint: Option<&str>,
1112        test_endpoint: Option<&str>,
1113        idempotent: bool,
1114        tab: &Tab,
1115    ) -> StepResult {
1116        let name = format!("[type] {target}");
1117        let selector = match self.resolve_selector(
1118            selector_override,
1119            target,
1120            step_endpoint,
1121            test_endpoint,
1122            tab,
1123        ) {
1124            Ok(s) => s,
1125            Err(msg) => {
1126                if idempotent {
1127                    return StepResult {
1128                        name,
1129                        status: StepStatus::Skipped,
1130                        message: format!("skipped (idempotent): no target found — {msg}"),
1131                    };
1132                }
1133                return StepResult {
1134                    name,
1135                    status: StepStatus::Failed,
1136                    message: msg,
1137                };
1138            }
1139        };
1140
1141        let probe_secs = if idempotent { 5 } else { 10 };
1142        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1143            Ok(element) => {
1144                if let Err(e) = element.click() {
1145                    return StepResult {
1146                        name,
1147                        status: StepStatus::Failed,
1148                        message: format!("click to focus {selector} failed: {e}"),
1149                    };
1150                }
1151
1152                let js = format!(
1153                    "document.querySelector('{}').value = '';",
1154                    selector.replace('\'', "\\'")
1155                );
1156                let _ = tab.evaluate(&js, false);
1157
1158                match element.type_into(text) {
1159                    Ok(_) => StepResult {
1160                        name,
1161                        status: StepStatus::Passed,
1162                        message: format!("typed {text:?} into {selector}"),
1163                    },
1164                    Err(e) => StepResult {
1165                        name,
1166                        status: StepStatus::Failed,
1167                        message: format!("type into {selector} failed: {e}"),
1168                    },
1169                }
1170            }
1171            Err(e) if idempotent => StepResult {
1172                name,
1173                status: StepStatus::Skipped,
1174                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1175            },
1176            Err(e) => StepResult {
1177                name,
1178                status: StepStatus::Failed,
1179                message: format!("element {selector} not found: {e}"),
1180            },
1181        }
1182    }
1183
1184    #[allow(clippy::too_many_arguments)]
1185    #[allow(clippy::too_many_lines)]
1186    fn run_wait(
1187        &self,
1188        target: &str,
1189        selector_override: Option<&str>,
1190        text: Option<&str>,
1191        timeout_ms: Option<u64>,
1192        step_endpoint: Option<&str>,
1193        test_endpoint: Option<&str>,
1194        idempotent: bool,
1195        tab: &Tab,
1196    ) -> StepResult {
1197        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1198        let step_name = format!("[wait] {target}");
1199
1200        // Resolve an explicit selector only (text-only waits are LLM-free).
1201        let selector = match selector_override {
1202            Some(s) => Some(s.to_owned()),
1203            None if text.is_some() => None,
1204            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1205                Ok(s) => Some(s),
1206                Err(msg) => {
1207                    if idempotent {
1208                        return StepResult {
1209                            name: step_name,
1210                            status: StepStatus::Skipped,
1211                            message: format!("skipped (idempotent): no target found — {msg}"),
1212                        };
1213                    }
1214                    return StepResult {
1215                        name: step_name,
1216                        status: StepStatus::Failed,
1217                        message: msg,
1218                    };
1219                }
1220            },
1221        };
1222
1223        if text.is_some() {
1224            let sel_js = selector
1225                .as_deref()
1226                .map(crate::selectors::selector_matches_js);
1227            let text_js = text.map(|t| {
1228                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1229                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1230            });
1231
1232            let deadline = Instant::now() + timeout;
1233            loop {
1234                let sel_ok = sel_js
1235                    .as_ref()
1236                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1237                let text_ok = text_js
1238                    .as_ref()
1239                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1240                if sel_ok && text_ok {
1241                    let mut what = Vec::new();
1242                    if let Some(sel) = &selector {
1243                        what.push(format!("found {sel}"));
1244                    }
1245                    if let Some(t) = text {
1246                        what.push(format!("text {t:?} visible"));
1247                    }
1248                    return StepResult {
1249                        name: step_name,
1250                        status: StepStatus::Passed,
1251                        message: what.join(" and "),
1252                    };
1253                }
1254                if Instant::now() >= deadline {
1255                    let mut what = Vec::new();
1256                    if let Some(sel) = &selector {
1257                        what.push(sel.clone());
1258                    }
1259                    if let Some(t) = text {
1260                        what.push(format!("text {t:?}"));
1261                    }
1262                    let message = format!(
1263                        "wait for {} timed out after {}ms: the event waited for never came",
1264                        what.join(" / "),
1265                        timeout.as_millis(),
1266                    );
1267                    if idempotent {
1268                        return StepResult {
1269                            name: step_name,
1270                            status: StepStatus::Skipped,
1271                            message: format!("skipped (idempotent): {message}"),
1272                        };
1273                    }
1274                    return StepResult {
1275                        name: step_name,
1276                        status: StepStatus::Failed,
1277                        message,
1278                    };
1279                }
1280                std::thread::sleep(Duration::from_millis(250));
1281            }
1282        }
1283
1284        match selector.as_deref() {
1285            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1286                Ok(_) => StepResult {
1287                    name: step_name,
1288                    status: StepStatus::Passed,
1289                    message: format!("found {sel}"),
1290                },
1291                Err(e) if idempotent => StepResult {
1292                    name: step_name,
1293                    status: StepStatus::Skipped,
1294                    message: format!(
1295                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1296                        timeout.as_millis()
1297                    ),
1298                },
1299                Err(e) => StepResult {
1300                    name: step_name,
1301                    status: StepStatus::Failed,
1302                    message: format!(
1303                        "wait for {sel} timed out after {}ms: {e}",
1304                        timeout.as_millis()
1305                    ),
1306                },
1307            },
1308            None => StepResult {
1309                name: step_name,
1310                status: StepStatus::Failed,
1311                message: "wait step has neither selector nor text".into(),
1312            },
1313        }
1314    }
1315
1316    #[allow(clippy::too_many_arguments)]
1317    fn run_assert(
1318        &self,
1319        definition: Option<&str>,
1320        preset: Option<&str>,
1321        prompt: Option<&str>,
1322        assert_text: Option<&str>,
1323        screenshot: bool,
1324        step_endpoint: Option<&str>,
1325        test_endpoint: Option<&str>,
1326        tab: &Tab,
1327    ) -> StepResult {
1328        std::thread::sleep(Duration::from_millis(500));
1329
1330        let page_content = get_page_text(tab);
1331
1332        // Vision attach: capture the full page once per assert step and
1333        // split it into viewport-tall tiles (the total coverage is bounded
1334        // by the configured height cap so vision tokens stay sane). All
1335        // tile data URLs are handed to the preset/prompt evaluation below.
1336        let image: Option<Vec<String>> = if screenshot {
1337            let endpoint = self
1338                .endpoints
1339                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1340            if !endpoint.vision {
1341                return StepResult {
1342                    name: "[assert]".into(),
1343                    status: StepStatus::Failed,
1344                    message: format!(
1345                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1346                        name = endpoint.name
1347                    ),
1348                };
1349            }
1350            match crate::vision::capture_screenshot_data_urls(
1351                tab,
1352                self.config
1353                    .screenshot_max_dimension
1354                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1355                self.screenshot_height_cap(),
1356                self.current_viewport_height(),
1357            ) {
1358                Ok(urls) => Some(urls),
1359                Err(e) => {
1360                    return StepResult {
1361                        name: "[assert]".into(),
1362                        status: StepStatus::Failed,
1363                        message: format!("screenshot capture failed: {e}"),
1364                    };
1365                }
1366            }
1367        } else {
1368            None
1369        };
1370
1371        if let Some(def_name) = definition {
1372            if let Some(def) = self.definitions.get(def_name) {
1373                return self.run_assert_def(
1374                    def,
1375                    &page_content,
1376                    image.as_deref(),
1377                    step_endpoint,
1378                    test_endpoint,
1379                    tab,
1380                );
1381            }
1382            return StepResult {
1383                name: format!("[assert] {def_name}"),
1384                status: StepStatus::Failed,
1385                message: format!("definition '{def_name}' not found"),
1386            };
1387        }
1388
1389        if let Some(preset_name) = preset {
1390            // Deterministic DOM layout scan — runs JS in the browser and
1391            // never calls the LLM (free, fast, no pixel budget).
1392            if preset_name == "layout_no_issues" {
1393                return self.run_layout_preset(tab);
1394            }
1395            return self.run_preset(
1396                preset_name,
1397                assert_text,
1398                &page_content,
1399                image.as_deref(),
1400                step_endpoint,
1401                test_endpoint,
1402            );
1403        }
1404
1405        if let Some(prompt_text) = prompt {
1406            return self.run_custom(
1407                prompt_text,
1408                &page_content,
1409                image.as_deref(),
1410                step_endpoint,
1411                test_endpoint,
1412            );
1413        }
1414
1415        StepResult {
1416            name: "[assert]".into(),
1417            status: StepStatus::Skipped,
1418            message: "no definition, preset, or prompt specified".into(),
1419        }
1420    }
1421
1422    fn run_assert_def(
1423        &self,
1424        def: &AssertDefinition,
1425        page_content: &PageContent,
1426        image: Option<&[String]>,
1427        step_endpoint: Option<&str>,
1428        test_endpoint: Option<&str>,
1429        tab: &Tab,
1430    ) -> StepResult {
1431        // Agent-based definition: delegate to an A2A agent
1432        if let Some(ref agent) = def.agent {
1433            if image.is_some() {
1434                return StepResult {
1435                    name: format!("[assert] {}", def.name),
1436                    status: StepStatus::Failed,
1437                    message: "agent-backed assertions do not support screenshots".into(),
1438                };
1439            }
1440            let task = def
1441                .task_template
1442                .as_deref()
1443                .unwrap_or("Evaluate the assertion")
1444                .replace("{url}", &page_content.url)
1445                .replace("{title}", &page_content.title)
1446                .replace("{content}", &page_content.body_text)
1447                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1448
1449            return self.run_agent_step(agent, &task, &def.name);
1450        }
1451
1452        // Custom preset: system + user_template provided in the definition
1453        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1454            return self.run_custom_preset(
1455                &def.name,
1456                system,
1457                template,
1458                def.assert_text.as_deref(),
1459                page_content,
1460                image,
1461                step_endpoint,
1462                test_endpoint,
1463            );
1464        }
1465
1466        def.preset.as_ref().map_or_else(
1467            || {
1468                def.prompt.as_ref().map_or_else(
1469                    || StepResult {
1470                        name: format!("[assert] {}", def.name),
1471                        status: StepStatus::Failed,
1472                        message: "definition has no preset, prompt, or system+user_template".into(),
1473                    },
1474                    |prompt| {
1475                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1476                    },
1477                )
1478            },
1479            |preset_name| {
1480                if preset_name == "layout_no_issues" {
1481                    return self.run_layout_preset(tab);
1482                }
1483                self.run_preset(
1484                    preset_name,
1485                    def.assert_text.as_deref(),
1486                    page_content,
1487                    image,
1488                    step_endpoint,
1489                    test_endpoint,
1490                )
1491            },
1492        )
1493    }
1494
1495    #[allow(clippy::too_many_arguments)]
1496    fn run_custom_preset(
1497        &self,
1498        name: &str,
1499        system: &str,
1500        template: &str,
1501        assert_text: Option<&str>,
1502        page_content: &PageContent,
1503        image: Option<&[String]>,
1504        step_endpoint: Option<&str>,
1505        test_endpoint: Option<&str>,
1506    ) -> StepResult {
1507        let user_prompt = template
1508            .replace("{url}", &page_content.url)
1509            .replace("{title}", &page_content.title)
1510            .replace("{content}", &page_content.body_text)
1511            .replace("{expected_text}", assert_text.unwrap_or(""))
1512            .replace("{description}", "");
1513
1514        // Custom preset definitions frequently forget the {content}
1515        // placeholder — without it the LLM has no page to evaluate and
1516        // answers "I can't determine that without seeing the page". Always
1517        // append the page context unless the template already references it.
1518        let user_prompt = if template.contains("{content}") {
1519            user_prompt
1520        } else {
1521            format!(
1522                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1523                url = page_content.url,
1524                title = page_content.title,
1525                content = page_content.body_text,
1526            )
1527        };
1528
1529        self.reporter
1530            .debug(format!("assert: {name} (custom preset)"));
1531
1532        let chain = self
1533            .endpoints
1534            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1535        let sys = system.to_owned();
1536
1537        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1538
1539        response.map_or_else(
1540            |e| StepResult {
1541                name: format!("[assert] {name}"),
1542                status: StepStatus::Failed,
1543                message: format!("LLM assertion call failed: {e}"),
1544            },
1545            |(lr, _idx)| {
1546                if verdict_is_pass(&lr.content) {
1547                    StepResult {
1548                        name: format!("[assert] {name}"),
1549                        status: StepStatus::Passed,
1550                        message: "PASS".into(),
1551                    }
1552                } else {
1553                    StepResult {
1554                        name: format!("[assert] {name}"),
1555                        status: StepStatus::Failed,
1556                        message: lr.content,
1557                    }
1558                }
1559            },
1560        )
1561    }
1562
1563    fn run_preset(
1564        &self,
1565        preset_name: &str,
1566        assert_text: Option<&str>,
1567        page_content: &PageContent,
1568        image: Option<&[String]>,
1569        step_endpoint: Option<&str>,
1570        test_endpoint: Option<&str>,
1571    ) -> StepResult {
1572        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1573            return StepResult {
1574                name: format!("[assert] {preset_name}"),
1575                status: StepStatus::Failed,
1576                message: format!("unknown assertion preset: {preset_name}"),
1577            };
1578        };
1579        if preset_name.starts_with("visual_") && image.is_none() {
1580            return StepResult {
1581                name: format!("[assert] {preset_name}"),
1582                status: StepStatus::Failed,
1583                message: format!(
1584                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1585                ),
1586            };
1587        }
1588
1589        let user_prompt = preset
1590            .user_template
1591            .replace("{url}", &page_content.url)
1592            .replace("{title}", &page_content.title)
1593            .replace("{content}", &page_content.body_text)
1594            .replace("{expected_text}", assert_text.unwrap_or(""))
1595            .replace("{description}", "");
1596
1597        // Same safety net as custom presets: never let the LLM answer with
1598        // no page context at all.
1599        let user_prompt = if preset.user_template.contains("{content}") {
1600            user_prompt
1601        } else {
1602            format!(
1603                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1604                url = page_content.url,
1605                title = page_content.title,
1606                content = page_content.body_text,
1607            )
1608        };
1609
1610        self.reporter.debug(format!("assert: {preset_name}"));
1611
1612        let chain = self
1613            .endpoints
1614            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1615        let sys = preset.system.to_owned();
1616
1617        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1618
1619        response.map_or_else(
1620            |e| StepResult {
1621                name: format!("[assert] {preset_name}"),
1622                status: StepStatus::Failed,
1623                message: format!("LLM assertion call failed: {e}"),
1624            },
1625            |(lr, _idx)| {
1626                if verdict_is_pass(&lr.content) {
1627                    StepResult {
1628                        name: format!("[assert] {preset_name}"),
1629                        status: StepStatus::Passed,
1630                        message: "PASS".into(),
1631                    }
1632                } else {
1633                    StepResult {
1634                        name: format!("[assert] {preset_name}"),
1635                        status: StepStatus::Failed,
1636                        message: lr.content,
1637                    }
1638                }
1639            },
1640        )
1641    }
1642
1643    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1644    ///
1645    /// Evaluates the layout-scan JS in the page and fails with the list of
1646    /// detected issues: horizontal page overflow, visible elements sticking
1647    /// out of the viewport, text clipped by `overflow: hidden` containers,
1648    /// and interactive elements covered by other elements. No LLM call —
1649    /// checks are geometry-based so the check is free, deterministic, and
1650    /// safe to run on every page × viewport variant.
1651    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1652        let name = "[assert] layout_no_issues".to_owned();
1653        self.reporter
1654            .debug("assert: layout_no_issues (DOM layout scan)");
1655        let js = LAYOUT_SCAN_JS.replace(
1656            "__IGNORE_CLASSES__",
1657            &serde_json::to_string(&self.config.layout_ignore_classes)
1658                .unwrap_or_else(|_| "[]".to_owned()),
1659        );
1660        let result = tab.evaluate(&js, false);
1661        let json_str = match result {
1662            Ok(r) => r
1663                .value
1664                .as_ref()
1665                .and_then(|v| v.as_str().map(String::from))
1666                .unwrap_or_else(|| "[]".to_owned()),
1667            Err(e) => {
1668                return StepResult {
1669                    name,
1670                    status: StepStatus::Failed,
1671                    message: format!("layout scan JS failed: {e}"),
1672                };
1673            }
1674        };
1675        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1676        if issues.is_empty() {
1677            return StepResult {
1678                name,
1679                status: StepStatus::Passed,
1680                message: "PASS — no layout defects detected".into(),
1681            };
1682        }
1683        let mut lines: Vec<String> = issues
1684            .iter()
1685            .take(10)
1686            .map(|i| {
1687                format!(
1688                    "- [{type_}] {element}: {detail}",
1689                    type_ = i.issue_type,
1690                    element = i.element,
1691                    detail = i.detail
1692                )
1693            })
1694            .collect();
1695        if issues.len() > 10 {
1696            lines.push(format!("- … and {} more", issues.len() - 10));
1697        }
1698        StepResult {
1699            name,
1700            status: StepStatus::Failed,
1701            message: format!(
1702                "FAIL — {} layout defect(s) detected:\n{}",
1703                issues.len(),
1704                lines.join("\n")
1705            ),
1706        }
1707    }
1708
1709    fn run_custom(
1710        &self,
1711        prompt: &str,
1712        page_content: &PageContent,
1713        image: Option<&[String]>,
1714        step_endpoint: Option<&str>,
1715        test_endpoint: Option<&str>,
1716    ) -> StepResult {
1717        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1718
1719        let mut user = format!(
1720            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1721            url = page_content.url,
1722            title = page_content.title,
1723            content = page_content.body_text,
1724        );
1725        if image.is_some() {
1726            user.push_str(
1727                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1728            );
1729        }
1730
1731        self.reporter.debug("custom assert");
1732
1733        let chain = self
1734            .endpoints
1735            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1736        let sys = system.to_owned();
1737
1738        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1739
1740        response.map_or_else(
1741            |e| StepResult {
1742                name: "[assert] custom".into(),
1743                status: StepStatus::Failed,
1744                message: format!("LLM assertion call failed: {e}"),
1745            },
1746            |(lr, _idx)| {
1747                if verdict_is_pass(&lr.content) {
1748                    StepResult {
1749                        name: "[assert] custom".into(),
1750                        status: StepStatus::Passed,
1751                        message: "PASS".into(),
1752                    }
1753                } else {
1754                    StepResult {
1755                        name: "[assert] custom".into(),
1756                        status: StepStatus::Failed,
1757                        message: lr.content,
1758                    }
1759                }
1760            },
1761        )
1762    }
1763
1764    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1765        let path = path.unwrap_or("screenshot.png");
1766
1767        match tab.capture_screenshot(
1768            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1769            None,
1770            None,
1771            true,
1772        ) {
1773            Ok(data) => {
1774                if let Err(e) = std::fs::write(path, &data) {
1775                    return StepResult {
1776                        name: format!("[screenshot] {path}"),
1777                        status: StepStatus::Failed,
1778                        message: format!("failed to write screenshot: {e}"),
1779                    };
1780                }
1781                StepResult {
1782                    name: format!("[screenshot] {path}"),
1783                    status: StepStatus::Passed,
1784                    message: format!("saved to {path}"),
1785                }
1786            }
1787            Err(e) => StepResult {
1788                name: format!("[screenshot] {path}"),
1789                status: StepStatus::Failed,
1790                message: format!("screenshot failed: {e}"),
1791            },
1792        }
1793    }
1794
1795    /// Runs an A2A agent step.
1796    #[allow(clippy::literal_string_with_formatting_args)]
1797    fn run_agent(
1798        &self,
1799        agent_name: &str,
1800        task: &str,
1801        definition: Option<&str>,
1802        _test_endpoint: Option<&str>,
1803    ) -> StepResult {
1804        // If a definition is specified, look up the task template
1805        let resolved_task = if let Some(def_name) = definition {
1806            if let Some(def) = self.definitions.get(def_name) {
1807                let tmpl = def.task_template.as_deref().unwrap_or(task);
1808                tmpl.replace("{task}", task)
1809            } else {
1810                return StepResult {
1811                    name: format!("[agent] {def_name}"),
1812                    status: StepStatus::Failed,
1813                    message: format!("definition '{def_name}' not found"),
1814                };
1815            }
1816        } else {
1817            task.to_owned()
1818        };
1819
1820        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1821    }
1822
1823    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1824        let Some(ep) = self.endpoints.get(agent_name) else {
1825            return StepResult {
1826                name: format!("[agent] {display_name}"),
1827                status: StepStatus::Failed,
1828                message: format!("agent endpoint '{agent_name}' not found"),
1829            };
1830        };
1831
1832        if ep.url.is_empty() {
1833            return StepResult {
1834                name: format!("[agent] {display_name}"),
1835                status: StepStatus::Failed,
1836                message: format!("agent endpoint '{agent_name}' has no URL"),
1837            };
1838        }
1839
1840        self.reporter.debug(format!("agent {agent_name}: {task}"));
1841
1842        let url = ep.url.clone();
1843        let client = A2aClient::new(&url, self.timeout);
1844        let task_clone = task.to_owned();
1845
1846        let response = std::thread::spawn(move || {
1847            let rt = tokio::runtime::Builder::new_current_thread()
1848                .enable_all()
1849                .build()
1850                .unwrap();
1851            rt.block_on(client.send_task(&task_clone))
1852        })
1853        .join()
1854        .unwrap();
1855
1856        // Record the flat-cost call
1857        self.usage.record_flat_call(agent_name, ep);
1858
1859        match response {
1860            Ok(text) => {
1861                let clean = text.trim().to_owned();
1862                if verdict_is_pass(&clean) {
1863                    StepResult {
1864                        name: format!("[agent] {display_name}"),
1865                        status: StepStatus::Passed,
1866                        message: format!("PASS: {clean}"),
1867                    }
1868                } else if verdict_is_fail(&clean) {
1869                    StepResult {
1870                        name: format!("[agent] {display_name}"),
1871                        status: StepStatus::Failed,
1872                        message: clean,
1873                    }
1874                } else {
1875                    StepResult {
1876                        name: format!("[agent] {display_name}"),
1877                        status: StepStatus::Passed,
1878                        message: format!("response: {clean}"),
1879                    }
1880                }
1881            }
1882            Err(e) => StepResult {
1883                name: format!("[agent] {display_name}"),
1884                status: StepStatus::Failed,
1885                message: format!("agent call failed: {e}"),
1886            },
1887        }
1888    }
1889
1890    /// Runs an MCP tool call step.
1891    fn run_mcp(
1892        &self,
1893        server_name: &str,
1894        tool_name: &str,
1895        args: Option<&serde_json::Value>,
1896    ) -> StepResult {
1897        let Some(ep) = self.endpoints.get(server_name) else {
1898            return StepResult {
1899                name: format!("[mcp] {server_name}:{tool_name}"),
1900                status: StepStatus::Failed,
1901                message: format!("MCP server endpoint '{server_name}' not found"),
1902            };
1903        };
1904
1905        let cmd = ep.command.as_deref().unwrap_or("");
1906        if cmd.is_empty() {
1907            return StepResult {
1908                name: format!("[mcp] {server_name}:{tool_name}"),
1909                status: StepStatus::Failed,
1910                message: format!("MCP server '{server_name}' has no command configured"),
1911            };
1912        }
1913
1914        self.reporter
1915            .debug(format!("mcp {server_name} {tool_name}"));
1916
1917        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1918
1919        let command = cmd.to_owned();
1920        let args_vec = ep.args.clone();
1921        let tool = tool_name.to_owned();
1922
1923        let response = std::thread::spawn(move || {
1924            let mut mcp_client =
1925                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1926            mcp_client
1927                .call_tool(&tool, &args_val)
1928                .map_err(|e| e.to_string())
1929        })
1930        .join()
1931        .unwrap();
1932
1933        // Record the flat-cost call
1934        self.usage.record_flat_call(server_name, ep);
1935
1936        match response {
1937            Ok(result) => {
1938                if result.isError {
1939                    StepResult {
1940                        name: format!("[mcp] {server_name}:{tool_name}"),
1941                        status: StepStatus::Failed,
1942                        message: result.to_string(),
1943                    }
1944                } else {
1945                    StepResult {
1946                        name: format!("[mcp] {server_name}:{tool_name}"),
1947                        status: StepStatus::Passed,
1948                        message: result.to_string(),
1949                    }
1950                }
1951            }
1952            Err(e) => StepResult {
1953                name: format!("[mcp] {server_name}:{tool_name}"),
1954                status: StepStatus::Failed,
1955                message: format!("MCP call failed: {e}"),
1956            },
1957        }
1958    }
1959
1960    // ── helpers ──────────────────────────────────────────────────────────
1961
1962    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1963    /// the runner's default LLM config for any unset fields.
1964    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1965        LlmConfig {
1966            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
1967                // Bedrock builds its endpoint from the resolved AWS region
1968                // when no URL is given — never inherit the default LLM URL.
1969                endpoint.url.clone()
1970            } else if endpoint.url.is_empty() {
1971                self.llm.url.clone()
1972            } else {
1973                endpoint.url.clone()
1974            },
1975            model: endpoint
1976                .model
1977                .clone()
1978                .unwrap_or_else(|| self.llm.model.clone()),
1979            api_key: endpoint
1980                .api_key
1981                .clone()
1982                .or_else(|| self.llm.api_key.clone()),
1983            headers: if endpoint.headers.is_empty() {
1984                self.llm.headers.clone()
1985            } else {
1986                endpoint.headers.clone()
1987            },
1988            timeout: self.llm.timeout,
1989            temperature: self.llm.temperature,
1990            thinking: self.llm.thinking,
1991            model_params: self.llm.model_params.clone(),
1992            cache: endpoint.cache_markers,
1993            max_attempts: endpoint.max_attempts.max(1),
1994            provider: endpoint.provider,
1995            deployment: endpoint.deployment.clone(),
1996            api_version: endpoint.api_version.clone(),
1997            auth: endpoint.auth.clone(),
1998            header_commands: endpoint.header_commands.clone(),
1999            aws: endpoint.aws.clone(),
2000        }
2001    }
2002
2003    /// Runs a single LLM call against an ordered endpoint chain (primary +
2004    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
2005    /// the first endpoint that answers wins. Returns the response together
2006    /// with the chain index of the answering endpoint (0 = primary) so the
2007    /// caller can attribute usage to the correct endpoint.
2008    ///
2009    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
2010    /// shows duration, tokens, cost and the answering endpoint per call.
2011    /// Run-level context handed to every LLM call as the FIRST block of the
2012    /// user message. Contents that are stable for the whole run ("run
2013    /// started", "target site") come first so upstream provider prefix
2014    /// caching stays effective; the current time is the last line because
2015    /// it changes on every call.
2016    fn run_context(&self) -> String {
2017        let mut parts = vec![
2018            "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
2019            "================================================================".into(),
2020            format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
2021        ];
2022        if let Some(base) = self.config.base_url.as_deref() {
2023            parts.push(format!("Target site: {base}"));
2024        }
2025        let now = SystemTime::now()
2026            .duration_since(UNIX_EPOCH)
2027            .map_or(0, |d| d.as_secs());
2028        parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
2029        parts.join("\n")
2030    }
2031
2032    #[allow(clippy::cast_possible_truncation, clippy::too_many_lines)]
2033    fn llm_call_chain(
2034        &self,
2035        chain: &[&ResolvedEndpoint],
2036        system: &str,
2037        user: &str,
2038        image: Option<&[String]>,
2039        purpose: &str,
2040    ) -> Result<(crate::costs::LlmResponse, usize), String> {
2041        if chain.is_empty() {
2042            return Err("empty LLM endpoint chain".into());
2043        }
2044        let primary = self.build_llm_for_endpoint(chain[0]);
2045        let fallbacks: Vec<LlmConfig> = chain[1..]
2046            .iter()
2047            .map(|e| self.build_llm_for_endpoint(e))
2048            .collect();
2049
2050        let (test, index) = self
2051            .current_step
2052            .borrow()
2053            .as_ref()
2054            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
2055        let primary_endpoint = chain[0].name.clone();
2056        let primary_model = primary.model.clone();
2057        self.emit_event(&TestEvent::LlmCallStarted {
2058            test: test.clone(),
2059            index,
2060            endpoint: primary_endpoint.clone(),
2061            model: primary_model.clone(),
2062            purpose: purpose.to_owned(),
2063        });
2064
2065        let started = Instant::now();
2066        let sys = system.to_owned();
2067        let context = self.run_context();
2068        let user = if context.is_empty() {
2069            user.to_owned()
2070        } else {
2071            format!("{context}\n\n{user}")
2072        };
2073        let image = image.map(<[String]>::to_vec);
2074
2075        let result = std::thread::spawn(move || {
2076            let rt = tokio::runtime::Builder::new_current_thread()
2077                .enable_all()
2078                .build()
2079                .unwrap();
2080            let call = async {
2081                match image.as_deref() {
2082                    Some(img) => {
2083                        llm_chat_vision_with_usage_chain(
2084                            &primary,
2085                            &fallbacks,
2086                            &sys,
2087                            &user,
2088                            Some(img),
2089                        )
2090                        .await
2091                    }
2092                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2093                }
2094            };
2095            rt.block_on(call)
2096        })
2097        .join()
2098        .unwrap();
2099
2100        let duration_ms = started.elapsed().as_millis() as u64;
2101        match result {
2102            Ok((lr, idx)) => {
2103                let cost = calculate_llm_cost(chain[idx], &lr.usage);
2104                let answering = chain[idx].name.clone();
2105                let model = chain[idx]
2106                    .model
2107                    .clone()
2108                    .unwrap_or_else(|| primary_model.clone());
2109                self.usage
2110                    .record_llm_call(&answering, chain[idx], &model, &lr.usage);
2111                self.emit_event(&TestEvent::LlmCallFinished {
2112                    test,
2113                    index,
2114                    endpoint: answering,
2115                    model,
2116                    purpose: purpose.to_owned(),
2117                    ok: true,
2118                    duration_ms,
2119                    input_tokens: lr.usage.prompt_tokens,
2120                    output_tokens: lr.usage.completion_tokens,
2121                    cached_input_tokens: lr.usage.cached_input_tokens,
2122                    cache_creation_input_tokens: lr.usage.cache_creation_input_tokens,
2123                    cost,
2124                    error: None,
2125                });
2126                Ok((lr, idx))
2127            }
2128            Err(e) => {
2129                self.emit_event(&TestEvent::LlmCallFinished {
2130                    test,
2131                    index,
2132                    endpoint: primary_endpoint,
2133                    model: primary_model,
2134                    purpose: purpose.to_owned(),
2135                    ok: false,
2136                    duration_ms,
2137                    input_tokens: 0,
2138                    output_tokens: 0,
2139                    cached_input_tokens: 0,
2140                    cache_creation_input_tokens: 0,
2141                    cost: 0.0,
2142                    error: Some(e.clone()),
2143                });
2144                Err(e)
2145            }
2146        }
2147    }
2148
2149    /// Resolves a CSS selector for the target element. Uses the explicit
2150    /// `selector` if provided, otherwise asks the LLM to find the element
2151    /// from the natural language `target` description and page DOM.
2152    ///
2153    /// LLM responses are sanitized and verified against the live page: a
2154    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2155    /// immediately with the raw LLM output, and a selector that matches
2156    /// nothing triggers one retry with feedback before failing.
2157    #[allow(clippy::too_many_lines)]
2158    fn resolve_selector(
2159        &self,
2160        css_override: Option<&str>,
2161        target: &str,
2162        step_endpoint: Option<&str>,
2163        test_endpoint: Option<&str>,
2164        tab: &Tab,
2165    ) -> Result<String, String> {
2166        if let Some(explicit) = css_override {
2167            return Ok(explicit.to_owned());
2168        }
2169
2170        let dom_info = extract_dom_info(tab)?;
2171        let page_content = get_page_text(tab);
2172
2173        let system = concat!(
2174            "You are a browser automation selector generator. ",
2175            "Given a web page's content and interactive elements, ",
2176            "return ONLY the best CSS selector for the described element. ",
2177            "Output nothing except the CSS selector. ",
2178            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2179            "[name=\"...\"], tag.class, tag. ",
2180            "Never output explanations, markdown, or extra text."
2181        );
2182
2183        let user = format!(
2184            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2185            page_content.url,
2186            page_content.title,
2187            truncate(&page_content.body_text, 4000),
2188            dom_info,
2189            target,
2190        );
2191
2192        let retry_user = format!(
2193            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2194            "The selector must match at least one element currently present on the page.",
2195            page_content.url,
2196            page_content.title,
2197            truncate(&page_content.body_text, 4000),
2198            dom_info,
2199            target,
2200        );
2201
2202        self.reporter.debug(format!("LLM targeting: {target}"));
2203
2204        let chain = self
2205            .endpoints
2206            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2207        let sys = system.to_owned();
2208
2209        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2210
2211        let first = call_llm(&user);
2212        let (lr, _idx) = match first {
2213            Ok(lr) => lr,
2214            Err(e) => {
2215                return Err(format!("LLM element targeting failed: {e}"));
2216            }
2217        };
2218        let clean = sanitize_selector(&lr.content);
2219        self.reporter.debug(format!("resolved selector: {clean}"));
2220
2221        if selector_is_useless(&clean) {
2222            return Err(format!(
2223                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2224                raw = lr.content.trim(),
2225            ));
2226        }
2227        if let Err(reason) = validate_selector(&clean) {
2228            return Err(format!(
2229                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2230                raw = lr.content.trim(),
2231            ));
2232        }
2233        if !selector_matches(tab, &clean).unwrap_or(false) {
2234            // One retry with feedback: flaky models occasionally invent a
2235            // selector that does not exist on the page.
2236            self.reporter.warn(format!(
2237                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2238            ));
2239            let second = call_llm(&retry_user);
2240            let (lr2, _idx2) = match second {
2241                Ok(lr2) => lr2,
2242                Err(e) => {
2243                    return Err(format!(
2244                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2245                    ));
2246                }
2247            };
2248            let clean2 = sanitize_selector(&lr2.content);
2249            self.reporter
2250                .debug(format!("resolved selector (retry): {clean2}"));
2251            if selector_is_useless(&clean2) {
2252                return Err(format!(
2253                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2254                    raw = lr2.content.trim(),
2255                    excerpt = truncate(&page_content.body_text, 300),
2256                ));
2257            }
2258            if !selector_matches(tab, &clean2).unwrap_or(false) {
2259                return Err(format!(
2260                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2261                ));
2262            }
2263            return Ok(clean2);
2264        }
2265
2266        Ok(clean)
2267    }
2268}
2269
2270/// Evaluates a JS expression that is expected to return a boolean.
2271fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2272    tab.evaluate(js, false)
2273        .map_err(|e| format!("evaluate failed: {e}"))?
2274        .value
2275        .and_then(|v| v.as_bool())
2276        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2277}
2278
2279/// Checks whether a CSS selector matches at least one current element.
2280fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2281    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2282}
2283
2284// ── Free helper functions ──────────────────────────────────────────────
2285
2286fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2287    let name = format!("[navigate] {full_url}");
2288    match tab.navigate_to(full_url) {
2289        Ok(_) => {
2290            let _ = tab.wait_until_navigated();
2291            StepResult {
2292                name,
2293                status: StepStatus::Passed,
2294                message: format!("navigated to {full_url}"),
2295            }
2296        }
2297        Err(e) => StepResult {
2298            name,
2299            status: StepStatus::Failed,
2300            message: format!("navigation failed: {e}"),
2301        },
2302    }
2303}
2304
2305fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2306    let result = tab
2307        .evaluate(DOM_EXTRACT_JS, false)
2308        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2309
2310    let json_str = result
2311        .value
2312        .as_ref()
2313        .and_then(|v| v.as_str())
2314        .unwrap_or("[]");
2315
2316    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2317
2318    if elements.is_empty() {
2319        return Ok("(no interactive elements found)".to_owned());
2320    }
2321
2322    Ok(elements.join("\n"))
2323}
2324
2325fn get_page_text(tab: &Tab) -> PageContent {
2326    let url = tab.get_url();
2327
2328    let title = tab
2329        .evaluate("document.title", false)
2330        .ok()
2331        .and_then(|r| r.value)
2332        .and_then(|v| v.as_str().map(String::from))
2333        .unwrap_or_else(|| "unknown".to_owned());
2334
2335    let body_text = tab
2336        .evaluate(
2337            "document.body ? document.body.innerText : document.documentElement.innerText",
2338            false,
2339        )
2340        .ok()
2341        .and_then(|r| r.value)
2342        .and_then(|v| v.as_str().map(String::from))
2343        .unwrap_or_default();
2344
2345    PageContent {
2346        url,
2347        title,
2348        body_text: truncate(&body_text, 8000),
2349    }
2350}
2351
2352fn resolve_url(url: &str, base_url: &str) -> String {
2353    if url.starts_with("http://") || url.starts_with("https://") {
2354        return url.to_owned();
2355    }
2356    let base = base_url.trim_end_matches('/');
2357    if url.starts_with('/') {
2358        format!("{base}{url}")
2359    } else {
2360        format!("{base}/{url}")
2361    }
2362}
2363
2364/// Human-readable label for a step, used when steps are skipped after an
2365/// earlier failure.
2366fn step_label(step: &TestStep) -> String {
2367    match step {
2368        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2369        TestStep::Click { target, .. } => format!("[click] {target}"),
2370        TestStep::Type { target, .. } => format!("[type] {target}"),
2371        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2372        TestStep::Assert {
2373            definition,
2374            preset,
2375            prompt,
2376            ..
2377        } => definition.as_ref().map_or_else(
2378            || {
2379                preset.as_ref().map_or_else(
2380                    || {
2381                        prompt.as_ref().map_or_else(
2382                            || "[assert]".to_owned(),
2383                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2384                        )
2385                    },
2386                    |p| format!("[assert] {p}"),
2387                )
2388            },
2389            |d| format!("[assert] {d}"),
2390        ),
2391        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2392        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2393        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2394    }
2395}
2396
2397/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2398#[must_use]
2399const fn step_kind_label(step: &TestStep) -> &'static str {
2400    match step {
2401        TestStep::Navigate { .. } => "navigate",
2402        TestStep::Click { .. } => "click",
2403        TestStep::Type { .. } => "type",
2404        TestStep::Wait { .. } => "wait",
2405        TestStep::Assert { .. } => "assert",
2406        TestStep::Screenshot { .. } => "screenshot",
2407        TestStep::Agent { .. } => "agent",
2408        TestStep::Mcp { .. } => "mcp",
2409    }
2410}
2411
2412// ── Support types ──────────────────────────────────────────────────────
2413
2414#[derive(Default)]
2415struct TestRunResult {
2416    passed: u32,
2417    failed: u32,
2418    skipped: u32,
2419    total: u32,
2420    details: Vec<StepResult>,
2421}
2422
2423struct PageContent {
2424    url: String,
2425    title: String,
2426    body_text: String,
2427}
2428
2429#[cfg(test)]
2430mod tests {
2431    use super::unix_to_rfc3339;
2432
2433    #[test]
2434    fn rfc3339_epoch_and_reference_dates() {
2435        assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2436        assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2437        assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2438        assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2439    }
2440
2441    #[test]
2442    fn rfc3339_handles_leap_years() {
2443        assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2444        assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2445    }
2446}
2447
2448#[cfg(test)]
2449mod verdict_parse_tests {
2450    use super::{verdict_is_fail, verdict_is_pass};
2451
2452    #[test]
2453    fn tolerates_markdown_punctuation_and_natural_language() {
2454        assert!(verdict_is_pass("PASS"));
2455        assert!(verdict_is_pass("pass"));
2456        assert!(verdict_is_pass("**PASS**\n\nThe page is clean."));
2457        assert!(verdict_is_pass("passes - no explicit error visible"));
2458        assert!(verdict_is_pass("  \"pass\""));
2459        assert!(!verdict_is_pass("FAIL: something broke"));
2460        assert!(!verdict_is_pass("**FAIL** broken"));
2461
2462        assert!(verdict_is_fail("**FAIL** broken"));
2463        assert!(verdict_is_fail("fails - error toast shown"));
2464        assert!(!verdict_is_fail("passes - ok"));
2465    }
2466}