Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// Formats a UNIX epoch timestamp as an RFC 3339 UTC string
26/// (`2026-09-19T16:04:00Z`) without pulling in a date dependency.
27/// Civil-from-days conversion after Howard Hinnant.
28#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31    let secs = i64::try_from(secs).unwrap_or(0);
32    let days = secs.div_euclid(86_400);
33    let rem = secs.rem_euclid(86_400);
34    let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35    let z = days + 719_468;
36    let era = z.div_euclid(146_097);
37    let doe = z.rem_euclid(146_097);
38    let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40    let mp = (5 * doy + 2) / 153;
41    let d = doy - (153 * mp + 2) / 5 + 1;
42    let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43    let y = yoe + era * 400 + i64::from(mth <= 2);
44    format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47/// One detected layout defect (`layout_no_issues` preset).
48#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50    #[serde(rename = "type")]
51    issue_type: String,
52    element: String,
53    detail: String,
54}
55
56/// In-browser DOM layout scan for `layout_no_issues`.
57///
58/// Geometry-only checks (no LLM, no pixels):
59/// 1. page horizontal overflow (`scrollWidth` > viewport width);
60/// 2. elements outside the viewport that scrolling cannot reveal
61///    (fixed elements off-screen, left/negative overflow, right-edge
62///    overflow beyond the horizontally scrollable area, and bottom
63///    overflow on a page that cannot scroll down) — below-the-fold
64///    content on a tall scrollable page is normal flow, NOT a defect;
65/// 3. text clipped by `overflow: hidden` containers whose content
66///    is measurably larger than the box;
67/// 4. interactive elements (buttons/links/inputs) whose center point
68///    is covered by a different element that would intercept the click.
69///
70/// Elements whose class matches a configured ignore prefix (default:
71/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
72/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
73/// skipped — they are intentionally 1x1 / off-screen. The prefix list
74/// is injected at `__IGNORE_CLASSES__` from
75/// [`ScenarioConfig::layout_ignore_classes`].
76///
77/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
78/// sticky headers) is excluded by the position/relation filters.
79const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81  const issues = [];
82  const push = (type, el, detail) => {
83    if (issues.length >= 30) return;
84    let element = el.tagName.toLowerCase();
85    if (el.id) element += '#' + el.id;
86    else if (typeof el.className === 'string' && el.className.trim())
87      element += '.' + el.className.trim().split(/\s+/).join('.');
88    issues.push({ type, element, detail: String(detail).slice(0, 220) });
89  };
90  const vw = document.documentElement.clientWidth || window.innerWidth;
91  const vh = document.documentElement.clientHeight || window.innerHeight;
92  if (!vw || !vh) return JSON.stringify(issues);
93  const de = document.documentElement;
94  // 1. Page-level horizontal overflow.
95  if (maxSW > vw + 2)
96    push('page-overflow-x', de,
97      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98  // Class-name prefixes to skip (injected; default = Angular CDK
99  // screen-reader helpers, which are intentionally 1x1 / off-screen).
100  const ignorePrefixes = __IGNORE_CLASSES__;
101  const isIgnored = (el) => {
102    if (typeof el.className !== 'string' || !el.className.trim()) return false;
103    const classes = el.className.trim().split(/\s+/);
104    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105  };
106  const body = document.body;
107  // The document element is not always the scroll container: the app may
108  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110  // "not scrollable" — measure the scrollable content height/width of the
111  // document element AND the body, and take the max.
112  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114  const vScrollable = maxSH > vh + 2;
115  const hScrollable = maxSW > vw + 2;
116  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117  // this element into view on the given axis — i.e. the element is inside a
118  // scrolling region, so lying beyond the viewport cut is not a defect.
119  const reachableByScroller = (el, axis) => {
120    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123    let p = el.parentElement;
124    while (p && p !== body) {
125      const pcs = getComputedStyle(p);
126      const o = pcs[ovProp];
127      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128          p[dim] > p[dimClient] + 2) return true;
129      p = p.parentElement;
130    }
131    return false;
132  };
133  const selfOverflowing = (el, cs, axis) => {
134    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138      el[dim] > el[dimClient] + 2;
139  };
140  // True when an ancestor clips this axis with overflow:hidden/clip — the
141  // element's overhang is not visible, so treat it as reachable.
142  const clippedByAncestor = (el, axis) => {
143    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144    let p = el.parentElement;
145    while (p && p !== body) {
146      const o = getComputedStyle(p)[ovProp];
147      if (o === 'hidden' || o === 'clip') return true;
148      p = p.parentElement;
149    }
150    return false;
151  };
152  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154  const hasContent = (el) =>
155    ((el.textContent || '').trim().length > 0) ||
156    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157  // 2. Elements outside the viewport that scrolling cannot reveal.
158  for (const el of all) {
159    const cs = getComputedStyle(el);
160    if (!visible(cs)) continue;
161    if (isIgnored(el)) continue;
162    const r = el.getBoundingClientRect();
163    if (r.width < 2 || r.height < 2) continue;
164    if (!hasContent(el) && el.children.length === 0) continue;
165    if (cs.position === 'fixed') {
166      // Fixed elements never move with the scroll: any edge outside the
167      // viewport is unreachable content and therefore a defect.
168      const overTop = -r.top;
169      const overLeft = -r.left;
170      const overRight = r.right - vw;
171      const overBottom = r.bottom - vh;
172      // Top/left overshoot is never reachable; bottom/right overshoot is
173      // only a defect when the fixed element cannot scroll that content
174      // into view itself (e.g. an opened Material drawer whose inner
175      // container scrolls is not a "cut-off" defect).
176      if (overTop > 2 || overLeft > 2 ||
177          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179        let where = '';
180        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186        push('element-out-of-viewport', el,
187          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188      }
189      continue;
190    }
191    if (cs.position === 'sticky') continue;
192    // Negative left overflow cannot be reached by scrolling (scrollLeft
193    // never goes below 0).
194    if (r.left < -2) {
195      push('element-out-of-viewport', el,
196        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197      continue;
198    }
199    // Negative top with the page at the top means the element sits above
200    // the document origin — also unreachable.
201    if (r.top < -2 && de.scrollTop <= 2) {
202      push('element-out-of-viewport', el,
203        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204      continue;
205    }
206    const overRight = r.right - vw;
207    // Right-edge overflow is only a defect when the page cannot scroll
208    // horizontally to reveal it (or the element sticks out past the
209    // scrollable content width itself).
210    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211      push('element-out-of-viewport', el,
212        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213      continue;
214    }
215    // Below-the-fold content on a scrollable page is normal (tall landing
216    // pages); only flag bottom overflow the user can never scroll to.
217    const overBottom = r.bottom - vh;
218    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219      push('element-out-of-viewport', el,
220        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221    }
222  }
223  // 3. Text clipped by overflow:hidden containers.
224  for (const el of all) {
225    if (isIgnored(el)) continue;
226    const cs = getComputedStyle(el);
227    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229    if (!(el.textContent || '').trim()) continue;
230    // Single-line truncation with an ellipsis is an intentional design
231    // pattern (Tailwind .truncate etc.), not a clipping defect.
232    if (cs.textOverflow === 'ellipsis') continue;
233    push('text-clipped', el,
234      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236  }
237  // 4. Interactive elements covered by a different element.
238  const interactive =
239    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240  const targets = document.querySelectorAll(interactive);
241  for (const el of targets) {
242    if (isIgnored(el)) continue;
243    const r = el.getBoundingClientRect();
244    if (r.width < 6 || r.height < 6) continue;
245    const cs = getComputedStyle(el);
246    if (!visible(cs)) continue;
247    const cx = r.left + r.width / 2;
248    const cy = r.top + r.height / 2;
249    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250    const top = document.elementFromPoint(cx, cy);
251    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252    if (isIgnored(top)) continue;
253    const tcs = getComputedStyle(top);
254    if (!visible(tcs)) continue;
255    if (tcs.pointerEvents === 'none') continue;
256    const tr = top.getBoundingClientRect();
257    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260    push('element-overlap', el,
261      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262      ' is covered by <' + tname + '>');
263  }
264  return JSON.stringify(issues);
265})()
266"#;
267
268/// How long the CDP connection stays open after the browser goes quiet.
269///
270/// `headless_chrome` ships a 30s default and tears down the entire connection
271/// when no traffic arrives for that long; a run must own its connection for
272/// its full duration instead.
273const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275/// Executes a [`Scenario`] against a real browser with optional LLM
276/// assistance for element targeting and assertions.
277pub struct ScenarioRunner {
278    config: ScenarioConfig,
279    definitions: HashMap<String, AssertDefinition>,
280    llm: LlmConfig,
281    timeout: Duration,
282    viewport_width: u32,
283    viewport_height: u32,
284    /// The viewport currently applied in the browser (CDP emulation).
285    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
286    applied_viewport: std::cell::Cell<(u32, u32)>,
287    endpoints: EndpointRegistry,
288    usage: Arc<UsageTracker>,
289    budgets: BudgetTracker,
290    /// Directory for failure artifacts (screenshots).
291    artifacts_dir: PathBuf,
292    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
293    reporter: Arc<Reporter>,
294    /// The test + step index currently executing, for LLM-call events.
295    current_step: std::cell::RefCell<Option<(String, u32)>>,
296    /// Epoch seconds of runner creation — constant for the whole run so
297    /// the LLM context preamble below is cacheable by upstream providers.
298    run_started: u64,
299    /// Whether this runner emits the `RunStarted`/`RunFinished` events.
300    /// When several scenario files are run concurrently by
301    /// [`crate::parallel::run_scenarios`], the orchestrator owns those two
302    /// events (one per batch), so each per-file runner suppresses its own.
303    emit_run_events: bool,
304}
305
306/// Aggregated results from a scenario run.
307#[derive(Debug, Default)]
308pub struct RunReport {
309    /// Number of tests that passed.
310    pub tests_passed: u32,
311    /// Number of tests that failed.
312    pub tests_failed: u32,
313    /// Number of steps that passed.
314    pub passed: u32,
315    /// Number of steps that failed.
316    pub failed: u32,
317    /// Number of steps that were skipped.
318    pub skipped: u32,
319    /// Per-step details.
320    pub details: Vec<StepResult>,
321}
322
323/// Result of a single step execution.
324#[derive(Debug, Clone)]
325pub struct StepResult {
326    /// The step name.
327    pub name: String,
328    /// Whether the step passed, failed, or was skipped.
329    pub status: StepStatus,
330    /// Human-readable result message.
331    pub message: String,
332}
333
334/// Outcome for a single step.
335pub use crate::events::StepStatus;
336
337/// Predefined assertion preset definition.
338struct AssertPreset {
339    name: &'static str,
340    system: &'static str,
341    user_template: &'static str,
342}
343
344/// Built-in assertion presets.
345#[allow(clippy::literal_string_with_formatting_args)]
346const ASSERTION_PRESETS: &[AssertPreset] = &[
347    AssertPreset {
348        name: "no_error_on_page",
349        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
350        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
351    },
352    AssertPreset {
353        name: "text_visible",
354        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
355        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
356    },
357    AssertPreset {
358        name: "element_exists",
359        system: "You are a QA tester. Check if a described UI element exists on a web page.",
360        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
361    },
362    AssertPreset {
363        name: "visual_no_issues",
364        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
365        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
366    },
367    AssertPreset {
368        name: "visual_no_overlaps",
369        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
370        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
371    },
372    AssertPreset {
373        name: "visual_fills_viewport",
374        system: "You are a visual QA engineer inspecting a website screenshot for UNDER-FILLED layouts: the rendered content stops at a fraction of the viewport and the remaining area is a large blank region the layout should have filled. Typical defects: page content ending around the halfway mark of the viewport with an empty, background-only area below it; an app shell, dashboard, or main panel collapsing to part of its expected height and leaving dead space; content that fills only the left portion of a wide viewport with unused space to the right. Do NOT fail intentionally short or centered pages: a login or signup card, a landing hero, a short article, a confirmation step, or an empty state that does not fill the viewport is normal design. Only fail when the blank region is clearly a layout defect — the page looks cut off or collapsed, or a content container (calendar, table, chart, map, panel) visibly fails to stretch to the space its own layout provides.",
375        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nDoes the page content fill the viewport as its layout intends — with no large blank region where the page appears to end partway down or partway across? Respond with exactly \"PASS\" if the page fills the viewport or is intentionally short, or \"FAIL: <describe the under-filled region and where it appears>\" if the content clearly stops at a fraction of the available page.",
376    },
377    AssertPreset {
378        name: "visual_text_visible",
379        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
380        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
381    },
382    AssertPreset {
383        name: "layout_no_issues",
384        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
385        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
386    },
387];
388
389/// Whether an LLM verdict should be read as PASS.
390///
391/// Models routinely wrap the verdict in markdown (`**PASS** …`) or prefix it
392/// with punctuation, which a bare `starts_with("pass")` misreads as a failure.
393/// Strip any leading non-alphanumerics, then match the leading word; this also
394/// accepts the natural-language forms `pass` / `passes`.
395fn verdict_is_pass(content: &str) -> bool {
396    content
397        .trim()
398        .trim_start_matches(|c: char| !c.is_alphanumeric())
399        .to_lowercase()
400        .starts_with("pass")
401}
402
403/// Whether an LLM verdict should be read as FAIL (see [`verdict_is_pass`]).
404fn verdict_is_fail(content: &str) -> bool {
405    content
406        .trim()
407        .trim_start_matches(|c: char| !c.is_alphanumeric())
408        .to_lowercase()
409        .starts_with("fail")
410}
411
412impl ScenarioRunner {
413    /// Creates a new runner with the given scenario configuration and
414    /// assertion definitions.
415    #[must_use]
416    #[allow(clippy::needless_pass_by_value)]
417    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
418        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
419    }
420
421    /// Creates a runner that reports run events through the given reporter.
422    #[must_use]
423    #[allow(clippy::needless_pass_by_value)]
424    pub fn with_reporter(
425        scenario_config: ScenarioConfig,
426        definitions: Vec<AssertDefinition>,
427        reporter: Arc<Reporter>,
428    ) -> Self {
429        Self::with_reporter_mode(scenario_config, definitions, reporter, true)
430    }
431
432    /// Creates a runner that reports run events through the given reporter
433    /// but suppresses the `RunStarted`/`RunFinished` events. Used by the
434    /// parallel orchestrator, which emits those once per batch.
435    #[must_use]
436    #[allow(clippy::needless_pass_by_value)]
437    pub fn with_reporter_parallel(
438        scenario_config: ScenarioConfig,
439        definitions: Vec<AssertDefinition>,
440        reporter: Arc<Reporter>,
441    ) -> Self {
442        Self::with_reporter_mode(scenario_config, definitions, reporter, false)
443    }
444
445    #[must_use]
446    #[allow(clippy::needless_pass_by_value)]
447    fn with_reporter_mode(
448        scenario_config: ScenarioConfig,
449        definitions: Vec<AssertDefinition>,
450        reporter: Arc<Reporter>,
451        emit_run_events: bool,
452    ) -> Self {
453        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
454            &scenario_config,
455        ));
456        let llm = LlmConfig {
457            url: scenario_config
458                .llm_url
459                .clone()
460                .unwrap_or_else(crate::llm_base_url),
461            model: scenario_config
462                .llm_model
463                .clone()
464                .unwrap_or_else(crate::llm_model),
465            api_key: scenario_config
466                .llm_api_key
467                .clone()
468                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
469            headers: if scenario_config.llm_headers.is_empty() {
470                crate::parse_headers_env()
471            } else {
472                scenario_config.llm_headers.clone()
473            },
474            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
475            temperature: scenario_config.temperature,
476            thinking: scenario_config.thinking,
477            model_params: scenario_config.model_params.clone(),
478            cache: scenario_config.cache.unwrap_or(true),
479            max_attempts: crate::default_llm_attempts(),
480            provider: crate::scenario::Provider::Openai,
481            deployment: None,
482            api_version: None,
483            auth: crate::scenario::AuthConfig::default(),
484            header_commands: std::collections::HashMap::new(),
485            aws: crate::scenario::AwsConfig::default(),
486        };
487        let endpoints = EndpointRegistry::from_config(
488            &scenario_config.endpoints,
489            Some(&llm),
490            crate::endpoints::EndpointDefaults {
491                cache: scenario_config.cache,
492                cache_pricing: scenario_config.cache_pricing,
493            },
494        );
495        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
496        let defs_map: HashMap<String, AssertDefinition> = definitions
497            .into_iter()
498            .map(|d| (d.name.clone(), d))
499            .collect();
500
501        Self {
502            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
503            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
504            viewport_height: scenario_config.viewport_height.unwrap_or(720),
505            applied_viewport: std::cell::Cell::new((0, 0)),
506            config: scenario_config.clone(),
507            definitions: defs_map,
508            llm,
509            endpoints,
510            usage: Arc::new(UsageTracker::new()),
511            budgets,
512            artifacts_dir: PathBuf::from(
513                scenario_config
514                    .artifacts_dir
515                    .unwrap_or_else(|| "artifacts".to_owned()),
516            ),
517            reporter,
518            current_step: std::cell::RefCell::new(None),
519            run_started: SystemTime::now()
520                .duration_since(UNIX_EPOCH)
521                .map_or(0, |d| d.as_secs()),
522            emit_run_events,
523        }
524    }
525
526    /// Emits an event; a sink failure degrades to a console warning so a
527    /// broken log file can never mask the run itself.
528    fn emit_event(&self, event: &TestEvent) {
529        if let Err(err) = self.reporter.emit(event) {
530            use std::io::Write as _;
531            let mut out = std::io::stderr().lock();
532            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
533        }
534    }
535
536    /// Returns a clone of the [`UsageTracker`] for reporting.
537    #[must_use]
538    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
539        Arc::clone(&self.usage)
540    }
541
542    /// Returns a reference to the [`BudgetTracker`].
543    #[must_use]
544    pub const fn budget_tracker(&self) -> &BudgetTracker {
545        &self.budgets
546    }
547
548    /// Executes all test groups in the scenario and returns a report.
549    ///
550    /// # Errors
551    ///
552    /// Returns an error if the browser fails to launch.
553    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
554    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
555        let mut report = RunReport::default();
556
557        if tests.is_empty() {
558            self.reporter.warn("No tests defined in scenario.");
559            if self.emit_run_events {
560                self.emit_event(&TestEvent::RunFinished {
561                    tests_passed: 0,
562                    tests_failed: 0,
563                    steps_passed: 0,
564                    steps_failed: 0,
565                    steps_skipped: 0,
566                    total_cost: 0.0,
567                    total_tokens: 0,
568                    total_input_tokens: 0,
569                    total_output_tokens: 0,
570                    total_cached_input_tokens: 0,
571                    total_cache_creation_input_tokens: 0,
572                    models: Vec::new(),
573                    total_calls: 0,
574                });
575            }
576            return Ok(report);
577        }
578
579        if self.emit_run_events {
580            self.emit_event(&TestEvent::RunStarted {
581                total_tests: tests.len() as u32,
582            });
583        }
584
585        let browser_headless = self.config.browser_headless.unwrap_or(true);
586
587        let launch_opts = LaunchOptions {
588            headless: browser_headless,
589            window_size: Some((self.viewport_width, self.viewport_height)),
590            sandbox: false,
591            // headless_chrome defaults this to 30s and shuts down the whole CDP
592            // connection when no messages arrive for that long. A scenario can
593            // easily exceed 30s of browser silence (slow LLM targeting/assertion
594            // calls, page waits, budget checks between steps), after which every
595            // remaining step fails with "Unable to make method calls because
596            // underlying connection is closed" — one quiet gap kills the run.
597            // Open-ended scenarios must own the connection for their full
598            // duration, so keep it alive for 6 hours.
599            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
600            ..LaunchOptions::default()
601        };
602
603        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
604        let tab = browser.new_tab().context("failed to open browser tab")?;
605        match (
606            self.config.browser_basic_auth_user.clone(),
607            self.config.browser_basic_auth_password.clone(),
608        ) {
609            (Some(username), Some(password)) if !username.is_empty() && !password.is_empty() => {
610                tab.authenticate(Some(username), Some(password))
611                    .context("failed to configure browser HTTP Basic Auth")?;
612                tab.enable_fetch(None, Some(true))
613                    .context("failed to enable browser HTTP authentication")?;
614            }
615            (None, None) => {}
616            _ => anyhow::bail!("browser Basic Auth requires both username and password"),
617        }
618        let _ = tab.set_default_timeout(self.timeout);
619
620        // Start MCP server if configured
621        #[cfg(feature = "mcp-server")]
622        if let Some(ref mcp_cfg) = self.config.mcp_server {
623            if mcp_cfg.enabled {
624                let port = mcp_cfg.port;
625                std::thread::spawn(move || {
626                    let _ = crate::mcp_server::start_mcp_server(port);
627                });
628            }
629        }
630        #[cfg(not(feature = "mcp-server"))]
631        if let Some(mcp_cfg) = &self.config.mcp_server {
632            if mcp_cfg.enabled {
633                self.reporter
634                    .warn("MCP server configured but 'mcp-server' feature not enabled");
635            }
636        }
637
638        // Start A2A agent server if configured
639        #[cfg(feature = "a2a-server")]
640        if let Some(ref a2a_cfg) = self.config.a2a_server {
641            if a2a_cfg.enabled {
642                let port = a2a_cfg.port;
643                tokio::spawn(crate::a2a_server::start_a2a_server(port));
644            }
645        }
646        #[cfg(not(feature = "a2a-server"))]
647        if let Some(a2a_cfg) = &self.config.a2a_server {
648            if a2a_cfg.enabled {
649                self.reporter
650                    .warn("A2A server configured but 'a2a-server' feature not enabled");
651            }
652        }
653
654        for test in tests {
655            self.emit_event(&TestEvent::TestStarted {
656                test: test.name.clone(),
657            });
658
659            self.usage.reset_per_test();
660
661            let test_started = Instant::now();
662            let test_result = self.run_test(test, &tab);
663            let duration_ms = test_started.elapsed().as_millis() as u64;
664            let usage = self.usage.current_test_snapshot();
665            self.usage.commit_test(&test.name);
666
667            self.emit_event(&TestEvent::TestFinished {
668                test: test.name.clone(),
669                passed: test_result.passed,
670                failed: test_result.failed,
671                skipped: test_result.skipped,
672                duration_ms,
673                cost: usage.total_cost,
674                tokens: usage.total_tokens,
675                input_tokens: usage.total_input_tokens,
676                output_tokens: usage.total_output_tokens,
677                cached_input_tokens: usage.total_cached_input_tokens,
678                cache_creation_input_tokens: usage.total_cache_creation_input_tokens,
679                models: usage.models.clone(),
680                calls: usage.total_calls,
681            });
682
683            if test_result.failed == 0 && test_result.total > 0 {
684                report.tests_passed += 1;
685            } else if test_result.total > 0 {
686                report.tests_failed += 1;
687            }
688
689            report.passed += test_result.passed;
690            report.failed += test_result.failed;
691            report.skipped += test_result.skipped;
692            report.details.extend(test_result.details);
693        }
694
695        let global = self.usage.global_snapshot();
696        if self.emit_run_events {
697            self.emit_event(&TestEvent::RunFinished {
698                tests_passed: report.tests_passed,
699                tests_failed: report.tests_failed,
700                steps_passed: report.passed,
701                steps_failed: report.failed,
702                steps_skipped: report.skipped,
703                total_cost: global.total_cost,
704                total_tokens: global.total_tokens,
705                total_input_tokens: global.total_input_tokens,
706                total_output_tokens: global.total_output_tokens,
707                total_cached_input_tokens: global.total_cached_input_tokens,
708                total_cache_creation_input_tokens: global.total_cache_creation_input_tokens,
709                models: global.models.clone(),
710                total_calls: global.total_calls,
711            });
712        }
713
714        Ok(report)
715    }
716
717    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
718    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
719        let base_url = test
720            .base_url
721            .clone()
722            .or_else(|| self.config.base_url.clone())
723            .unwrap_or_else(crate::base_url);
724
725        // Per-test viewport override: switch the browser via CDP
726        // device-metrics emulation before this test runs.
727        let vw = test.viewport_width.unwrap_or(self.viewport_width);
728        let vh = test.viewport_height.unwrap_or(self.viewport_height);
729        if self.applied_viewport.get() != (vw, vh) {
730            self.apply_viewport(tab, vw, vh);
731            self.applied_viewport.set((vw, vh));
732        }
733
734        // Per-test isolation: every test starts from its own start_url
735        // (unless auto_navigate is disabled), so a test never inherits the
736        // previous test's page state.
737        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
738
739        let start_url = test
740            .start_url
741            .clone()
742            .or_else(|| self.config.start_url.clone())
743            .unwrap_or_else(|| "/dashboard".to_owned());
744
745        // Per-test state isolation for LOGIN tests: the tab is shared
746        // across the file's tests, and a previous test's login persists
747        // (Cognito tokens in localStorage + the hosted-UI cookies). The
748        // SPA's login pages then auto-continue authenticated visitors on
749        // boot, so a later login test never sees the form and times out
750        // waiting for `#email` (observed on runs #1641/#1642: a shard's
751        // first three logins pass, the fourth onward time out). Tests
752        // that navigate a login route start with cleared cookies + origin
753        // storage; every other test keeps the shared session (most
754        // "page loads" tests rely on it). The HTTP cache is deliberately
755        // NOT cleared — re-downloading the SPA bundle per test would only
756        // add boot time.
757        if auto_navigate && test_targets_login(&start_url, &test.steps) {
758            use headless_chrome::protocol::cdp::{Network, Storage};
759            if let Err(e) = tab.call_method(Network::ClearBrowserCookies(None)) {
760                self.reporter
761                    .warn(format!("per-test cookie clear failed: {e}"));
762            }
763            if let Some(origin) = origin_of(&base_url) {
764                if let Err(e) = tab.call_method(Storage::ClearDataForOrigin {
765                    origin,
766                    storage_Types: "all".to_string(),
767                }) {
768                    self.reporter
769                        .warn(format!("per-test storage clear failed: {e}"));
770                }
771            }
772        }
773        if auto_navigate {
774            let full_url = resolve_url(&start_url, &base_url);
775            self.reporter.debug(format!("auto-navigate: {full_url}"));
776            let _ = tab.navigate_to(&full_url);
777            let _ = tab.wait_until_navigated();
778            std::thread::sleep(Duration::from_secs(4));
779        }
780
781        let mut result = TestRunResult::default();
782
783        for (step_index, step) in test.steps.iter().enumerate() {
784            result.total += 1;
785
786            let wait_ms = match step {
787                TestStep::Navigate { wait_after_ms, .. }
788                | TestStep::Click { wait_after_ms, .. }
789                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
790                _ => None,
791            };
792
793            self.current_step
794                .replace(Some((test.name.clone(), step_index as u32)));
795            self.emit_event(&TestEvent::StepStarted {
796                test: test.name.clone(),
797                index: step_index as u32,
798                label: step_label(step),
799            });
800            let step_started = Instant::now();
801
802            let mut step_result = match step {
803                TestStep::Navigate { url, .. } => {
804                    let full_url = resolve_url(url, &base_url);
805                    run_navigate_step(&full_url, tab)
806                }
807                TestStep::Click {
808                    target,
809                    selector,
810                    endpoint,
811                    idempotent,
812                    ..
813                } => self.run_click(
814                    target,
815                    selector.as_deref(),
816                    endpoint.as_deref(),
817                    test.endpoint.as_deref(),
818                    *idempotent,
819                    tab,
820                ),
821                TestStep::Type {
822                    target,
823                    text,
824                    selector,
825                    endpoint,
826                    idempotent,
827                    ..
828                } => self.run_type(
829                    target,
830                    text,
831                    selector.as_deref(),
832                    endpoint.as_deref(),
833                    test.endpoint.as_deref(),
834                    *idempotent,
835                    tab,
836                ),
837                TestStep::Wait {
838                    target,
839                    selector,
840                    text,
841                    timeout_ms,
842                    endpoint,
843                    idempotent,
844                } => self.run_wait(
845                    target,
846                    selector.as_deref(),
847                    text.as_deref(),
848                    *timeout_ms,
849                    endpoint.as_deref(),
850                    test.endpoint.as_deref(),
851                    *idempotent,
852                    tab,
853                ),
854                TestStep::Assert {
855                    definition,
856                    preset,
857                    prompt,
858                    assert_text,
859                    endpoint,
860                    screenshot,
861                } => self.run_assert(
862                    definition.as_deref(),
863                    preset.as_deref(),
864                    prompt.as_deref(),
865                    assert_text.as_deref(),
866                    *screenshot,
867                    endpoint.as_deref(),
868                    test.endpoint.as_deref(),
869                    tab,
870                ),
871                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
872                TestStep::Agent {
873                    agent,
874                    task,
875                    definition,
876                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
877                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
878            };
879
880            // Failure diagnostics: capture the page state and a screenshot so
881            // CI logs say WHAT the page looked like when the step failed,
882            // instead of a bare "timed out: The event waited for never came".
883            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
884                let state = diagnostics::capture(tab);
885                let screenshot = diagnostics::save_screenshot(
886                    tab,
887                    &self.artifacts_dir,
888                    &test.name,
889                    &test.name,
890                    step_index,
891                    step_kind_label(step),
892                );
893                step_result.message = format!(
894                    "{base} — {excerpt}",
895                    base = step_result.message,
896                    excerpt = diagnostics::inline_excerpt(&state),
897                );
898                (Some(diagnostics::full_context(&state)), screenshot)
899            } else {
900                (None, None)
901            };
902
903            let duration_ms = step_started.elapsed().as_millis() as u64;
904            self.emit_event(&TestEvent::StepFinished {
905                test: test.name.clone(),
906                index: step_index as u32,
907                label: step_result.name.clone(),
908                status: step_result.status,
909                duration_ms,
910                message: step_result.message.clone(),
911                diagnostics: diagnostics_block,
912                screenshot: screenshot_path,
913            });
914            self.current_step.replace(None);
915
916            match step_result.status {
917                StepStatus::Passed => result.passed += 1,
918                StepStatus::Failed => result.failed += 1,
919                StepStatus::Skipped => result.skipped += 1,
920            }
921
922            // Fail fast: the first failed step ends the test and the
923            // remaining steps are reported as skipped (no LLM budget is
924            // burned asserting against a page that is already known broken).
925            if step_result.status == StepStatus::Failed
926                && !self.config.continue_on_failure
927                && step_index + 1 < test.steps.len()
928            {
929                self.emit_event(&TestEvent::Warning {
930                    message: format!(
931                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
932                        test.steps.len() - step_index - 1
933                    ),
934                });
935                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
936                    let skipped_index = step_index + 1 + offset;
937                    let label = step_label(skipped);
938                    result.total += 1;
939                    result.skipped += 1;
940                    self.emit_event(&TestEvent::StepStarted {
941                        test: test.name.clone(),
942                        index: skipped_index as u32,
943                        label: label.clone(),
944                    });
945                    self.emit_event(&TestEvent::StepFinished {
946                        test: test.name.clone(),
947                        index: skipped_index as u32,
948                        label,
949                        status: StepStatus::Skipped,
950                        duration_ms: 0,
951                        message: "skipped: previous step failed".into(),
952                        diagnostics: None,
953                        screenshot: None,
954                    });
955                    result.details.push(StepResult {
956                        name: step_label(skipped),
957                        status: StepStatus::Skipped,
958                        message: "skipped: previous step failed".into(),
959                    });
960                }
961                result.details.push(step_result);
962                return result;
963            }
964
965            // Check per-test budget after each step
966            let test_usage = self.usage.current_test_snapshot();
967            let global_usage = self.usage.global_snapshot();
968            let budget_status = self.budgets.check_all(
969                &test.name,
970                &test_usage,
971                &global_usage,
972                test.budget.as_ref(),
973            );
974            match budget_status {
975                BudgetStatus::HardExceeded { message, .. } => {
976                    self.emit_event(&TestEvent::Warning {
977                        message: format!("budget exceeded: {message}"),
978                    });
979                    result.details.push(StepResult {
980                        name: "[budget]".into(),
981                        status: StepStatus::Failed,
982                        message,
983                    });
984                    result.failed += 1;
985                    return result;
986                }
987                BudgetStatus::SoftExceeded { message, .. } => {
988                    self.emit_event(&TestEvent::Warning {
989                        message: format!("budget warning: {message}"),
990                    });
991                }
992                BudgetStatus::Ok => {}
993            }
994
995            if let Some(ms) = wait_ms {
996                std::thread::sleep(Duration::from_millis(ms));
997            }
998
999            result.details.push(step_result);
1000        }
1001
1002        result
1003    }
1004
1005    /// Applies a viewport size to the current tab via CDP
1006    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
1007    /// overrides and the viewport matrix. The initial window size set at
1008    /// browser launch is replaced by emulation; failures are logged but
1009    /// do not fail the test (a mismatched viewport only weakens coverage).
1010    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
1011        use headless_chrome::protocol::cdp::Emulation;
1012        let _ = self;
1013        let params = Emulation::SetDeviceMetricsOverride {
1014            width,
1015            height,
1016            device_scale_factor: 1.0,
1017            mobile: false,
1018            scale: None,
1019            screen_width: Some(width),
1020            screen_height: Some(height),
1021            position_x: None,
1022            position_y: None,
1023            dont_set_visible_size: None,
1024            screen_orientation: None,
1025            viewport: None,
1026            display_feature: None,
1027            device_posture: None,
1028        };
1029        self.reporter.debug(format!("viewport: {width}x{height}"));
1030        if let Err(e) = tab.call_method(params) {
1031            self.reporter
1032                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
1033        }
1034    }
1035
1036    /// Height (px) covered by assert-step screenshots: the configured
1037    /// `screenshot_max_height` (absolute px or viewport multiple, default
1038    /// `"20x"`) resolved against the currently applied viewport, raised to
1039    /// at least the viewport height so the visible screen is always fully
1040    /// included. The capture is split into viewport-tall tiles, so this
1041    /// value bounds total coverage (and hence the number of image parts).
1042    #[must_use]
1043    fn screenshot_height_cap(&self) -> u32 {
1044        let viewport_height = self.current_viewport_height();
1045        let cap = self
1046            .config
1047            .screenshot_max_height
1048            .as_ref()
1049            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
1050        cap.max(viewport_height)
1051    }
1052
1053    /// Height of the viewport currently emulated in the browser (falling
1054    /// back to the configured default before any emulation was applied).
1055    #[must_use]
1056    const fn current_viewport_height(&self) -> u32 {
1057        let (_, height) = self.applied_viewport.get();
1058        if height > 0 {
1059            height
1060        } else {
1061            self.viewport_height
1062        }
1063    }
1064
1065    // ── step handlers ───────────────────────────────────────────────────
1066
1067    #[allow(clippy::too_many_lines)]
1068    fn run_click(
1069        &self,
1070        target: &str,
1071        selector_override: Option<&str>,
1072        step_endpoint: Option<&str>,
1073        test_endpoint: Option<&str>,
1074        idempotent: bool,
1075        tab: &Tab,
1076    ) -> StepResult {
1077        let name = format!("[click] {target}");
1078        let selector = match self.resolve_selector(
1079            selector_override,
1080            target,
1081            step_endpoint,
1082            test_endpoint,
1083            tab,
1084        ) {
1085            Ok(s) => s,
1086            Err(msg) => {
1087                if idempotent {
1088                    return StepResult {
1089                        name,
1090                        status: StepStatus::Skipped,
1091                        message: format!("skipped (idempotent): no target found — {msg}"),
1092                    };
1093                }
1094                return StepResult {
1095                    name,
1096                    status: StepStatus::Failed,
1097                    message: msg,
1098                };
1099            }
1100        };
1101
1102        // Idempotent steps probe briefly: a missing target means the
1103        // action was already done / not applicable (e.g. an
1104        // already-authenticated session), and skipping is the success
1105        // path, not a failure.
1106        let probe_secs = if idempotent { 5 } else { 10 };
1107        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1108            Ok(element) => match element.click() {
1109                Ok(_) => StepResult {
1110                    name,
1111                    status: StepStatus::Passed,
1112                    message: format!("clicked {selector}"),
1113                },
1114                Err(e) => StepResult {
1115                    name,
1116                    status: StepStatus::Failed,
1117                    message: format!("click failed on {selector}: {e}"),
1118                },
1119            },
1120            Err(e) if idempotent => StepResult {
1121                name,
1122                status: StepStatus::Skipped,
1123                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1124            },
1125            Err(e) => StepResult {
1126                name,
1127                status: StepStatus::Failed,
1128                message: format!("element {selector} not found: {e}"),
1129            },
1130        }
1131    }
1132
1133    #[allow(clippy::too_many_arguments)]
1134    fn run_type(
1135        &self,
1136        target: &str,
1137        text: &str,
1138        selector_override: Option<&str>,
1139        step_endpoint: Option<&str>,
1140        test_endpoint: Option<&str>,
1141        idempotent: bool,
1142        tab: &Tab,
1143    ) -> StepResult {
1144        let name = format!("[type] {target}");
1145        let selector = match self.resolve_selector(
1146            selector_override,
1147            target,
1148            step_endpoint,
1149            test_endpoint,
1150            tab,
1151        ) {
1152            Ok(s) => s,
1153            Err(msg) => {
1154                if idempotent {
1155                    return StepResult {
1156                        name,
1157                        status: StepStatus::Skipped,
1158                        message: format!("skipped (idempotent): no target found — {msg}"),
1159                    };
1160                }
1161                return StepResult {
1162                    name,
1163                    status: StepStatus::Failed,
1164                    message: msg,
1165                };
1166            }
1167        };
1168
1169        let probe_secs = if idempotent { 5 } else { 10 };
1170        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1171            Ok(element) => {
1172                if let Err(e) = element.click() {
1173                    return StepResult {
1174                        name,
1175                        status: StepStatus::Failed,
1176                        message: format!("click to focus {selector} failed: {e}"),
1177                    };
1178                }
1179
1180                let js = format!(
1181                    "document.querySelector('{}').value = '';",
1182                    selector.replace('\'', "\\'")
1183                );
1184                let _ = tab.evaluate(&js, false);
1185
1186                match element.type_into(text) {
1187                    Ok(_) => StepResult {
1188                        name,
1189                        status: StepStatus::Passed,
1190                        message: format!("typed {text:?} into {selector}"),
1191                    },
1192                    Err(e) => StepResult {
1193                        name,
1194                        status: StepStatus::Failed,
1195                        message: format!("type into {selector} failed: {e}"),
1196                    },
1197                }
1198            }
1199            Err(e) if idempotent => StepResult {
1200                name,
1201                status: StepStatus::Skipped,
1202                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1203            },
1204            Err(e) => StepResult {
1205                name,
1206                status: StepStatus::Failed,
1207                message: format!("element {selector} not found: {e}"),
1208            },
1209        }
1210    }
1211
1212    #[allow(clippy::too_many_arguments)]
1213    #[allow(clippy::too_many_lines)]
1214    fn run_wait(
1215        &self,
1216        target: &str,
1217        selector_override: Option<&str>,
1218        text: Option<&str>,
1219        timeout_ms: Option<u64>,
1220        step_endpoint: Option<&str>,
1221        test_endpoint: Option<&str>,
1222        idempotent: bool,
1223        tab: &Tab,
1224    ) -> StepResult {
1225        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1226        let step_name = format!("[wait] {target}");
1227
1228        // Resolve an explicit selector only (text-only waits are LLM-free).
1229        let selector = match selector_override {
1230            Some(s) => Some(s.to_owned()),
1231            None if text.is_some() => None,
1232            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1233                Ok(s) => Some(s),
1234                Err(msg) => {
1235                    if idempotent {
1236                        return StepResult {
1237                            name: step_name,
1238                            status: StepStatus::Skipped,
1239                            message: format!("skipped (idempotent): no target found — {msg}"),
1240                        };
1241                    }
1242                    return StepResult {
1243                        name: step_name,
1244                        status: StepStatus::Failed,
1245                        message: msg,
1246                    };
1247                }
1248            },
1249        };
1250
1251        if text.is_some() {
1252            let sel_js = selector
1253                .as_deref()
1254                .map(crate::selectors::selector_matches_js);
1255            let text_js = text.map(|t| {
1256                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1257                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1258            });
1259
1260            let deadline = Instant::now() + timeout;
1261            loop {
1262                let sel_ok = sel_js
1263                    .as_ref()
1264                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1265                let text_ok = text_js
1266                    .as_ref()
1267                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1268                if sel_ok && text_ok {
1269                    let mut what = Vec::new();
1270                    if let Some(sel) = &selector {
1271                        what.push(format!("found {sel}"));
1272                    }
1273                    if let Some(t) = text {
1274                        what.push(format!("text {t:?} visible"));
1275                    }
1276                    return StepResult {
1277                        name: step_name,
1278                        status: StepStatus::Passed,
1279                        message: what.join(" and "),
1280                    };
1281                }
1282                if Instant::now() >= deadline {
1283                    let mut what = Vec::new();
1284                    if let Some(sel) = &selector {
1285                        what.push(sel.clone());
1286                    }
1287                    if let Some(t) = text {
1288                        what.push(format!("text {t:?}"));
1289                    }
1290                    let message = format!(
1291                        "wait for {} timed out after {}ms: the event waited for never came",
1292                        what.join(" / "),
1293                        timeout.as_millis(),
1294                    );
1295                    if idempotent {
1296                        return StepResult {
1297                            name: step_name,
1298                            status: StepStatus::Skipped,
1299                            message: format!("skipped (idempotent): {message}"),
1300                        };
1301                    }
1302                    return StepResult {
1303                        name: step_name,
1304                        status: StepStatus::Failed,
1305                        message,
1306                    };
1307                }
1308                std::thread::sleep(Duration::from_millis(250));
1309            }
1310        }
1311
1312        match selector.as_deref() {
1313            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1314                Ok(_) => StepResult {
1315                    name: step_name,
1316                    status: StepStatus::Passed,
1317                    message: format!("found {sel}"),
1318                },
1319                Err(e) if idempotent => StepResult {
1320                    name: step_name,
1321                    status: StepStatus::Skipped,
1322                    message: format!(
1323                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1324                        timeout.as_millis()
1325                    ),
1326                },
1327                Err(e) => StepResult {
1328                    name: step_name,
1329                    status: StepStatus::Failed,
1330                    message: format!(
1331                        "wait for {sel} timed out after {}ms: {e}",
1332                        timeout.as_millis()
1333                    ),
1334                },
1335            },
1336            None => StepResult {
1337                name: step_name,
1338                status: StepStatus::Failed,
1339                message: "wait step has neither selector nor text".into(),
1340            },
1341        }
1342    }
1343
1344    #[allow(clippy::too_many_arguments)]
1345    fn run_assert(
1346        &self,
1347        definition: Option<&str>,
1348        preset: Option<&str>,
1349        prompt: Option<&str>,
1350        assert_text: Option<&str>,
1351        screenshot: bool,
1352        step_endpoint: Option<&str>,
1353        test_endpoint: Option<&str>,
1354        tab: &Tab,
1355    ) -> StepResult {
1356        std::thread::sleep(Duration::from_millis(500));
1357
1358        let page_content = get_page_text(tab);
1359
1360        // Vision attach: capture the full page once per assert step and
1361        // split it into viewport-tall tiles (the total coverage is bounded
1362        // by the configured height cap so vision tokens stay sane). All
1363        // tile data URLs are handed to the preset/prompt evaluation below.
1364        let image: Option<Vec<String>> = if screenshot {
1365            let endpoint = self
1366                .endpoints
1367                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1368            if !endpoint.vision {
1369                return StepResult {
1370                    name: "[assert]".into(),
1371                    status: StepStatus::Failed,
1372                    message: format!(
1373                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1374                        name = endpoint.name
1375                    ),
1376                };
1377            }
1378            match crate::vision::capture_screenshot_data_urls(
1379                tab,
1380                self.config
1381                    .screenshot_max_dimension
1382                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1383                self.screenshot_height_cap(),
1384                self.current_viewport_height(),
1385            ) {
1386                Ok(urls) => Some(urls),
1387                Err(e) => {
1388                    return StepResult {
1389                        name: "[assert]".into(),
1390                        status: StepStatus::Failed,
1391                        message: format!("screenshot capture failed: {e}"),
1392                    };
1393                }
1394            }
1395        } else {
1396            None
1397        };
1398
1399        if let Some(def_name) = definition {
1400            if let Some(def) = self.definitions.get(def_name) {
1401                return self.run_assert_def(
1402                    def,
1403                    &page_content,
1404                    image.as_deref(),
1405                    step_endpoint,
1406                    test_endpoint,
1407                    tab,
1408                );
1409            }
1410            return StepResult {
1411                name: format!("[assert] {def_name}"),
1412                status: StepStatus::Failed,
1413                message: format!("definition '{def_name}' not found"),
1414            };
1415        }
1416
1417        if let Some(preset_name) = preset {
1418            // Deterministic DOM layout scan — runs JS in the browser and
1419            // never calls the LLM (free, fast, no pixel budget).
1420            if preset_name == "layout_no_issues" {
1421                return self.run_layout_preset(tab);
1422            }
1423            return self.run_preset(
1424                preset_name,
1425                assert_text,
1426                &page_content,
1427                image.as_deref(),
1428                step_endpoint,
1429                test_endpoint,
1430            );
1431        }
1432
1433        if let Some(prompt_text) = prompt {
1434            return self.run_custom(
1435                prompt_text,
1436                &page_content,
1437                image.as_deref(),
1438                step_endpoint,
1439                test_endpoint,
1440            );
1441        }
1442
1443        StepResult {
1444            name: "[assert]".into(),
1445            status: StepStatus::Skipped,
1446            message: "no definition, preset, or prompt specified".into(),
1447        }
1448    }
1449
1450    fn run_assert_def(
1451        &self,
1452        def: &AssertDefinition,
1453        page_content: &PageContent,
1454        image: Option<&[String]>,
1455        step_endpoint: Option<&str>,
1456        test_endpoint: Option<&str>,
1457        tab: &Tab,
1458    ) -> StepResult {
1459        // Agent-based definition: delegate to an A2A agent
1460        if let Some(ref agent) = def.agent {
1461            if image.is_some() {
1462                return StepResult {
1463                    name: format!("[assert] {}", def.name),
1464                    status: StepStatus::Failed,
1465                    message: "agent-backed assertions do not support screenshots".into(),
1466                };
1467            }
1468            let task = def
1469                .task_template
1470                .as_deref()
1471                .unwrap_or("Evaluate the assertion")
1472                .replace("{url}", &page_content.url)
1473                .replace("{title}", &page_content.title)
1474                .replace("{content}", &page_content.body_text)
1475                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1476
1477            return self.run_agent_step(agent, &task, &def.name);
1478        }
1479
1480        // Custom preset: system + user_template provided in the definition
1481        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1482            return self.run_custom_preset(
1483                &def.name,
1484                system,
1485                template,
1486                def.assert_text.as_deref(),
1487                page_content,
1488                image,
1489                step_endpoint,
1490                test_endpoint,
1491            );
1492        }
1493
1494        def.preset.as_ref().map_or_else(
1495            || {
1496                def.prompt.as_ref().map_or_else(
1497                    || StepResult {
1498                        name: format!("[assert] {}", def.name),
1499                        status: StepStatus::Failed,
1500                        message: "definition has no preset, prompt, or system+user_template".into(),
1501                    },
1502                    |prompt| {
1503                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1504                    },
1505                )
1506            },
1507            |preset_name| {
1508                if preset_name == "layout_no_issues" {
1509                    return self.run_layout_preset(tab);
1510                }
1511                self.run_preset(
1512                    preset_name,
1513                    def.assert_text.as_deref(),
1514                    page_content,
1515                    image,
1516                    step_endpoint,
1517                    test_endpoint,
1518                )
1519            },
1520        )
1521    }
1522
1523    #[allow(clippy::too_many_arguments)]
1524    fn run_custom_preset(
1525        &self,
1526        name: &str,
1527        system: &str,
1528        template: &str,
1529        assert_text: Option<&str>,
1530        page_content: &PageContent,
1531        image: Option<&[String]>,
1532        step_endpoint: Option<&str>,
1533        test_endpoint: Option<&str>,
1534    ) -> StepResult {
1535        let user_prompt = template
1536            .replace("{url}", &page_content.url)
1537            .replace("{title}", &page_content.title)
1538            .replace("{content}", &page_content.body_text)
1539            .replace("{expected_text}", assert_text.unwrap_or(""))
1540            .replace("{description}", "");
1541
1542        // Custom preset definitions frequently forget the {content}
1543        // placeholder — without it the LLM has no page to evaluate and
1544        // answers "I can't determine that without seeing the page". Always
1545        // append the page context unless the template already references it.
1546        let user_prompt = if template.contains("{content}") {
1547            user_prompt
1548        } else {
1549            format!(
1550                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1551                url = page_content.url,
1552                title = page_content.title,
1553                content = page_content.body_text,
1554            )
1555        };
1556
1557        self.reporter
1558            .debug(format!("assert: {name} (custom preset)"));
1559
1560        let chain = self
1561            .endpoints
1562            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1563        let sys = system.to_owned();
1564
1565        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1566
1567        response.map_or_else(
1568            |e| StepResult {
1569                name: format!("[assert] {name}"),
1570                status: StepStatus::Failed,
1571                message: format!("LLM assertion call failed: {e}"),
1572            },
1573            |(lr, _idx)| {
1574                if verdict_is_pass(&lr.content) {
1575                    StepResult {
1576                        name: format!("[assert] {name}"),
1577                        status: StepStatus::Passed,
1578                        message: "PASS".into(),
1579                    }
1580                } else {
1581                    StepResult {
1582                        name: format!("[assert] {name}"),
1583                        status: StepStatus::Failed,
1584                        message: lr.content,
1585                    }
1586                }
1587            },
1588        )
1589    }
1590
1591    fn run_preset(
1592        &self,
1593        preset_name: &str,
1594        assert_text: Option<&str>,
1595        page_content: &PageContent,
1596        image: Option<&[String]>,
1597        step_endpoint: Option<&str>,
1598        test_endpoint: Option<&str>,
1599    ) -> StepResult {
1600        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1601            return StepResult {
1602                name: format!("[assert] {preset_name}"),
1603                status: StepStatus::Failed,
1604                message: format!("unknown assertion preset: {preset_name}"),
1605            };
1606        };
1607        if preset_name.starts_with("visual_") && image.is_none() {
1608            return StepResult {
1609                name: format!("[assert] {preset_name}"),
1610                status: StepStatus::Failed,
1611                message: format!(
1612                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1613                ),
1614            };
1615        }
1616
1617        let user_prompt = preset
1618            .user_template
1619            .replace("{url}", &page_content.url)
1620            .replace("{title}", &page_content.title)
1621            .replace("{content}", &page_content.body_text)
1622            .replace("{expected_text}", assert_text.unwrap_or(""))
1623            .replace("{description}", "");
1624
1625        // Same safety net as custom presets: never let the LLM answer with
1626        // no page context at all.
1627        let user_prompt = if preset.user_template.contains("{content}") {
1628            user_prompt
1629        } else {
1630            format!(
1631                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1632                url = page_content.url,
1633                title = page_content.title,
1634                content = page_content.body_text,
1635            )
1636        };
1637
1638        self.reporter.debug(format!("assert: {preset_name}"));
1639
1640        let chain = self
1641            .endpoints
1642            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1643        let sys = preset.system.to_owned();
1644
1645        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1646
1647        response.map_or_else(
1648            |e| StepResult {
1649                name: format!("[assert] {preset_name}"),
1650                status: StepStatus::Failed,
1651                message: format!("LLM assertion call failed: {e}"),
1652            },
1653            |(lr, _idx)| {
1654                if verdict_is_pass(&lr.content) {
1655                    StepResult {
1656                        name: format!("[assert] {preset_name}"),
1657                        status: StepStatus::Passed,
1658                        message: "PASS".into(),
1659                    }
1660                } else {
1661                    StepResult {
1662                        name: format!("[assert] {preset_name}"),
1663                        status: StepStatus::Failed,
1664                        message: lr.content,
1665                    }
1666                }
1667            },
1668        )
1669    }
1670
1671    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1672    ///
1673    /// Evaluates the layout-scan JS in the page and fails with the list of
1674    /// detected issues: horizontal page overflow, visible elements sticking
1675    /// out of the viewport, text clipped by `overflow: hidden` containers,
1676    /// and interactive elements covered by other elements. No LLM call —
1677    /// checks are geometry-based so the check is free, deterministic, and
1678    /// safe to run on every page × viewport variant.
1679    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1680        let name = "[assert] layout_no_issues".to_owned();
1681        self.reporter
1682            .debug("assert: layout_no_issues (DOM layout scan)");
1683        let js = LAYOUT_SCAN_JS.replace(
1684            "__IGNORE_CLASSES__",
1685            &serde_json::to_string(&self.config.layout_ignore_classes)
1686                .unwrap_or_else(|_| "[]".to_owned()),
1687        );
1688        let result = tab.evaluate(&js, false);
1689        let json_str = match result {
1690            Ok(r) => r
1691                .value
1692                .as_ref()
1693                .and_then(|v| v.as_str().map(String::from))
1694                .unwrap_or_else(|| "[]".to_owned()),
1695            Err(e) => {
1696                return StepResult {
1697                    name,
1698                    status: StepStatus::Failed,
1699                    message: format!("layout scan JS failed: {e}"),
1700                };
1701            }
1702        };
1703        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1704        if issues.is_empty() {
1705            return StepResult {
1706                name,
1707                status: StepStatus::Passed,
1708                message: "PASS — no layout defects detected".into(),
1709            };
1710        }
1711        let mut lines: Vec<String> = issues
1712            .iter()
1713            .take(10)
1714            .map(|i| {
1715                format!(
1716                    "- [{type_}] {element}: {detail}",
1717                    type_ = i.issue_type,
1718                    element = i.element,
1719                    detail = i.detail
1720                )
1721            })
1722            .collect();
1723        if issues.len() > 10 {
1724            lines.push(format!("- … and {} more", issues.len() - 10));
1725        }
1726        StepResult {
1727            name,
1728            status: StepStatus::Failed,
1729            message: format!(
1730                "FAIL — {} layout defect(s) detected:\n{}",
1731                issues.len(),
1732                lines.join("\n")
1733            ),
1734        }
1735    }
1736
1737    fn run_custom(
1738        &self,
1739        prompt: &str,
1740        page_content: &PageContent,
1741        image: Option<&[String]>,
1742        step_endpoint: Option<&str>,
1743        test_endpoint: Option<&str>,
1744    ) -> StepResult {
1745        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1746
1747        let mut user = format!(
1748            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1749            url = page_content.url,
1750            title = page_content.title,
1751            content = page_content.body_text,
1752        );
1753        if image.is_some() {
1754            user.push_str(
1755                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1756            );
1757        }
1758
1759        self.reporter.debug("custom assert");
1760
1761        let chain = self
1762            .endpoints
1763            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1764        let sys = system.to_owned();
1765
1766        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1767
1768        response.map_or_else(
1769            |e| StepResult {
1770                name: "[assert] custom".into(),
1771                status: StepStatus::Failed,
1772                message: format!("LLM assertion call failed: {e}"),
1773            },
1774            |(lr, _idx)| {
1775                if verdict_is_pass(&lr.content) {
1776                    StepResult {
1777                        name: "[assert] custom".into(),
1778                        status: StepStatus::Passed,
1779                        message: "PASS".into(),
1780                    }
1781                } else {
1782                    StepResult {
1783                        name: "[assert] custom".into(),
1784                        status: StepStatus::Failed,
1785                        message: lr.content,
1786                    }
1787                }
1788            },
1789        )
1790    }
1791
1792    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1793        let path = path.unwrap_or("screenshot.png");
1794
1795        match tab.capture_screenshot(
1796            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1797            None,
1798            None,
1799            true,
1800        ) {
1801            Ok(data) => {
1802                if let Err(e) = std::fs::write(path, &data) {
1803                    return StepResult {
1804                        name: format!("[screenshot] {path}"),
1805                        status: StepStatus::Failed,
1806                        message: format!("failed to write screenshot: {e}"),
1807                    };
1808                }
1809                StepResult {
1810                    name: format!("[screenshot] {path}"),
1811                    status: StepStatus::Passed,
1812                    message: format!("saved to {path}"),
1813                }
1814            }
1815            Err(e) => StepResult {
1816                name: format!("[screenshot] {path}"),
1817                status: StepStatus::Failed,
1818                message: format!("screenshot failed: {e}"),
1819            },
1820        }
1821    }
1822
1823    /// Runs an A2A agent step.
1824    #[allow(clippy::literal_string_with_formatting_args)]
1825    fn run_agent(
1826        &self,
1827        agent_name: &str,
1828        task: &str,
1829        definition: Option<&str>,
1830        _test_endpoint: Option<&str>,
1831    ) -> StepResult {
1832        // If a definition is specified, look up the task template
1833        let resolved_task = if let Some(def_name) = definition {
1834            if let Some(def) = self.definitions.get(def_name) {
1835                let tmpl = def.task_template.as_deref().unwrap_or(task);
1836                tmpl.replace("{task}", task)
1837            } else {
1838                return StepResult {
1839                    name: format!("[agent] {def_name}"),
1840                    status: StepStatus::Failed,
1841                    message: format!("definition '{def_name}' not found"),
1842                };
1843            }
1844        } else {
1845            task.to_owned()
1846        };
1847
1848        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1849    }
1850
1851    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1852        let Some(ep) = self.endpoints.get(agent_name) else {
1853            return StepResult {
1854                name: format!("[agent] {display_name}"),
1855                status: StepStatus::Failed,
1856                message: format!("agent endpoint '{agent_name}' not found"),
1857            };
1858        };
1859
1860        if ep.url.is_empty() {
1861            return StepResult {
1862                name: format!("[agent] {display_name}"),
1863                status: StepStatus::Failed,
1864                message: format!("agent endpoint '{agent_name}' has no URL"),
1865            };
1866        }
1867
1868        self.reporter.debug(format!("agent {agent_name}: {task}"));
1869
1870        let url = ep.url.clone();
1871        let client = A2aClient::new(&url, self.timeout);
1872        let task_clone = task.to_owned();
1873
1874        let response = std::thread::spawn(move || {
1875            let rt = tokio::runtime::Builder::new_current_thread()
1876                .enable_all()
1877                .build()
1878                .unwrap();
1879            rt.block_on(client.send_task(&task_clone))
1880        })
1881        .join()
1882        .unwrap();
1883
1884        // Record the flat-cost call
1885        self.usage.record_flat_call(agent_name, ep);
1886
1887        match response {
1888            Ok(text) => {
1889                let clean = text.trim().to_owned();
1890                if verdict_is_pass(&clean) {
1891                    StepResult {
1892                        name: format!("[agent] {display_name}"),
1893                        status: StepStatus::Passed,
1894                        message: format!("PASS: {clean}"),
1895                    }
1896                } else if verdict_is_fail(&clean) {
1897                    StepResult {
1898                        name: format!("[agent] {display_name}"),
1899                        status: StepStatus::Failed,
1900                        message: clean,
1901                    }
1902                } else {
1903                    StepResult {
1904                        name: format!("[agent] {display_name}"),
1905                        status: StepStatus::Passed,
1906                        message: format!("response: {clean}"),
1907                    }
1908                }
1909            }
1910            Err(e) => StepResult {
1911                name: format!("[agent] {display_name}"),
1912                status: StepStatus::Failed,
1913                message: format!("agent call failed: {e}"),
1914            },
1915        }
1916    }
1917
1918    /// Runs an MCP tool call step.
1919    fn run_mcp(
1920        &self,
1921        server_name: &str,
1922        tool_name: &str,
1923        args: Option<&serde_json::Value>,
1924    ) -> StepResult {
1925        let Some(ep) = self.endpoints.get(server_name) else {
1926            return StepResult {
1927                name: format!("[mcp] {server_name}:{tool_name}"),
1928                status: StepStatus::Failed,
1929                message: format!("MCP server endpoint '{server_name}' not found"),
1930            };
1931        };
1932
1933        let cmd = ep.command.as_deref().unwrap_or("");
1934        if cmd.is_empty() {
1935            return StepResult {
1936                name: format!("[mcp] {server_name}:{tool_name}"),
1937                status: StepStatus::Failed,
1938                message: format!("MCP server '{server_name}' has no command configured"),
1939            };
1940        }
1941
1942        self.reporter
1943            .debug(format!("mcp {server_name} {tool_name}"));
1944
1945        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1946
1947        let command = cmd.to_owned();
1948        let args_vec = ep.args.clone();
1949        let tool = tool_name.to_owned();
1950
1951        let response = std::thread::spawn(move || {
1952            let mut mcp_client =
1953                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1954            mcp_client
1955                .call_tool(&tool, &args_val)
1956                .map_err(|e| e.to_string())
1957        })
1958        .join()
1959        .unwrap();
1960
1961        // Record the flat-cost call
1962        self.usage.record_flat_call(server_name, ep);
1963
1964        match response {
1965            Ok(result) => {
1966                if result.isError {
1967                    StepResult {
1968                        name: format!("[mcp] {server_name}:{tool_name}"),
1969                        status: StepStatus::Failed,
1970                        message: result.to_string(),
1971                    }
1972                } else {
1973                    StepResult {
1974                        name: format!("[mcp] {server_name}:{tool_name}"),
1975                        status: StepStatus::Passed,
1976                        message: result.to_string(),
1977                    }
1978                }
1979            }
1980            Err(e) => StepResult {
1981                name: format!("[mcp] {server_name}:{tool_name}"),
1982                status: StepStatus::Failed,
1983                message: format!("MCP call failed: {e}"),
1984            },
1985        }
1986    }
1987
1988    // ── helpers ──────────────────────────────────────────────────────────
1989
1990    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1991    /// the runner's default LLM config for any unset fields.
1992    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1993        LlmConfig {
1994            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
1995                // Bedrock builds its endpoint from the resolved AWS region
1996                // when no URL is given — never inherit the default LLM URL.
1997                endpoint.url.clone()
1998            } else if endpoint.url.is_empty() {
1999                self.llm.url.clone()
2000            } else {
2001                endpoint.url.clone()
2002            },
2003            model: endpoint
2004                .model
2005                .clone()
2006                .unwrap_or_else(|| self.llm.model.clone()),
2007            api_key: endpoint
2008                .api_key
2009                .clone()
2010                .or_else(|| self.llm.api_key.clone()),
2011            headers: if endpoint.headers.is_empty() {
2012                self.llm.headers.clone()
2013            } else {
2014                endpoint.headers.clone()
2015            },
2016            timeout: self.llm.timeout,
2017            temperature: self.llm.temperature,
2018            thinking: self.llm.thinking,
2019            model_params: self.llm.model_params.clone(),
2020            cache: endpoint.cache_markers,
2021            max_attempts: endpoint.max_attempts.max(1),
2022            provider: endpoint.provider,
2023            deployment: endpoint.deployment.clone(),
2024            api_version: endpoint.api_version.clone(),
2025            auth: endpoint.auth.clone(),
2026            header_commands: endpoint.header_commands.clone(),
2027            aws: endpoint.aws.clone(),
2028        }
2029    }
2030
2031    /// Runs a single LLM call against an ordered endpoint chain (primary +
2032    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
2033    /// the first endpoint that answers wins. Returns the response together
2034    /// with the chain index of the answering endpoint (0 = primary) so the
2035    /// caller can attribute usage to the correct endpoint.
2036    ///
2037    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
2038    /// shows duration, tokens, cost and the answering endpoint per call.
2039    /// Run-level context handed to every LLM call as the FIRST block of the
2040    /// user message. Contents that are stable for the whole run ("run
2041    /// started", "target site") come first so upstream provider prefix
2042    /// caching stays effective; the current time is the last line because
2043    /// it changes on every call.
2044    fn run_context(&self) -> String {
2045        let mut parts = vec![
2046            "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
2047            "================================================================".into(),
2048            format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
2049        ];
2050        if let Some(base) = self.config.base_url.as_deref() {
2051            parts.push(format!("Target site: {base}"));
2052        }
2053        let now = SystemTime::now()
2054            .duration_since(UNIX_EPOCH)
2055            .map_or(0, |d| d.as_secs());
2056        parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
2057        parts.join("\n")
2058    }
2059
2060    #[allow(clippy::cast_possible_truncation, clippy::too_many_lines)]
2061    fn llm_call_chain(
2062        &self,
2063        chain: &[&ResolvedEndpoint],
2064        system: &str,
2065        user: &str,
2066        image: Option<&[String]>,
2067        purpose: &str,
2068    ) -> Result<(crate::costs::LlmResponse, usize), String> {
2069        if chain.is_empty() {
2070            return Err("empty LLM endpoint chain".into());
2071        }
2072        let primary = self.build_llm_for_endpoint(chain[0]);
2073        let fallbacks: Vec<LlmConfig> = chain[1..]
2074            .iter()
2075            .map(|e| self.build_llm_for_endpoint(e))
2076            .collect();
2077
2078        let (test, index) = self
2079            .current_step
2080            .borrow()
2081            .as_ref()
2082            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
2083        let primary_endpoint = chain[0].name.clone();
2084        let primary_model = primary.model.clone();
2085        self.emit_event(&TestEvent::LlmCallStarted {
2086            test: test.clone(),
2087            index,
2088            endpoint: primary_endpoint.clone(),
2089            model: primary_model.clone(),
2090            purpose: purpose.to_owned(),
2091        });
2092
2093        let started = Instant::now();
2094        let sys = system.to_owned();
2095        let context = self.run_context();
2096        let user = if context.is_empty() {
2097            user.to_owned()
2098        } else {
2099            format!("{context}\n\n{user}")
2100        };
2101        let image = image.map(<[String]>::to_vec);
2102
2103        let result = std::thread::spawn(move || {
2104            let rt = tokio::runtime::Builder::new_current_thread()
2105                .enable_all()
2106                .build()
2107                .unwrap();
2108            let call = async {
2109                match image.as_deref() {
2110                    Some(img) => {
2111                        llm_chat_vision_with_usage_chain(
2112                            &primary,
2113                            &fallbacks,
2114                            &sys,
2115                            &user,
2116                            Some(img),
2117                        )
2118                        .await
2119                    }
2120                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2121                }
2122            };
2123            rt.block_on(call)
2124        })
2125        .join()
2126        .unwrap();
2127
2128        let duration_ms = started.elapsed().as_millis() as u64;
2129        match result {
2130            Ok((lr, idx)) => {
2131                let cost = calculate_llm_cost(chain[idx], &lr.usage);
2132                let answering = chain[idx].name.clone();
2133                let model = chain[idx]
2134                    .model
2135                    .clone()
2136                    .unwrap_or_else(|| primary_model.clone());
2137                self.usage
2138                    .record_llm_call(&answering, chain[idx], &model, &lr.usage);
2139                self.emit_event(&TestEvent::LlmCallFinished {
2140                    test,
2141                    index,
2142                    endpoint: answering,
2143                    model,
2144                    purpose: purpose.to_owned(),
2145                    ok: true,
2146                    duration_ms,
2147                    input_tokens: lr.usage.prompt_tokens,
2148                    output_tokens: lr.usage.completion_tokens,
2149                    cached_input_tokens: lr.usage.cached_input_tokens,
2150                    cache_creation_input_tokens: lr.usage.cache_creation_input_tokens,
2151                    cost,
2152                    error: None,
2153                });
2154                Ok((lr, idx))
2155            }
2156            Err(e) => {
2157                self.emit_event(&TestEvent::LlmCallFinished {
2158                    test,
2159                    index,
2160                    endpoint: primary_endpoint,
2161                    model: primary_model,
2162                    purpose: purpose.to_owned(),
2163                    ok: false,
2164                    duration_ms,
2165                    input_tokens: 0,
2166                    output_tokens: 0,
2167                    cached_input_tokens: 0,
2168                    cache_creation_input_tokens: 0,
2169                    cost: 0.0,
2170                    error: Some(e.clone()),
2171                });
2172                Err(e)
2173            }
2174        }
2175    }
2176
2177    /// Resolves a CSS selector for the target element. Uses the explicit
2178    /// `selector` if provided, otherwise asks the LLM to find the element
2179    /// from the natural language `target` description and page DOM.
2180    ///
2181    /// LLM responses are sanitized and verified against the live page: a
2182    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2183    /// immediately with the raw LLM output, and a selector that matches
2184    /// nothing triggers one retry with feedback before failing.
2185    #[allow(clippy::too_many_lines)]
2186    fn resolve_selector(
2187        &self,
2188        css_override: Option<&str>,
2189        target: &str,
2190        step_endpoint: Option<&str>,
2191        test_endpoint: Option<&str>,
2192        tab: &Tab,
2193    ) -> Result<String, String> {
2194        if let Some(explicit) = css_override {
2195            return Ok(explicit.to_owned());
2196        }
2197
2198        let dom_info = extract_dom_info(tab)?;
2199        let page_content = get_page_text(tab);
2200
2201        let system = concat!(
2202            "You are a browser automation selector generator. ",
2203            "Given a web page's content and interactive elements, ",
2204            "return ONLY the best CSS selector for the described element. ",
2205            "Output nothing except the CSS selector. ",
2206            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2207            "[name=\"...\"], tag.class, tag. ",
2208            "Never output explanations, markdown, or extra text."
2209        );
2210
2211        let user = format!(
2212            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2213            page_content.url,
2214            page_content.title,
2215            truncate(&page_content.body_text, 4000),
2216            dom_info,
2217            target,
2218        );
2219
2220        let retry_user = format!(
2221            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2222            "The selector must match at least one element currently present on the page.",
2223            page_content.url,
2224            page_content.title,
2225            truncate(&page_content.body_text, 4000),
2226            dom_info,
2227            target,
2228        );
2229
2230        self.reporter.debug(format!("LLM targeting: {target}"));
2231
2232        let chain = self
2233            .endpoints
2234            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2235        let sys = system.to_owned();
2236
2237        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2238
2239        let first = call_llm(&user);
2240        let (lr, _idx) = match first {
2241            Ok(lr) => lr,
2242            Err(e) => {
2243                return Err(format!("LLM element targeting failed: {e}"));
2244            }
2245        };
2246        let clean = sanitize_selector(&lr.content);
2247        self.reporter.debug(format!("resolved selector: {clean}"));
2248
2249        if selector_is_useless(&clean) {
2250            return Err(format!(
2251                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2252                raw = lr.content.trim(),
2253            ));
2254        }
2255        if let Err(reason) = validate_selector(&clean) {
2256            return Err(format!(
2257                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2258                raw = lr.content.trim(),
2259            ));
2260        }
2261        if !selector_matches(tab, &clean).unwrap_or(false) {
2262            // One retry with feedback: flaky models occasionally invent a
2263            // selector that does not exist on the page.
2264            self.reporter.warn(format!(
2265                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2266            ));
2267            let second = call_llm(&retry_user);
2268            let (lr2, _idx2) = match second {
2269                Ok(lr2) => lr2,
2270                Err(e) => {
2271                    return Err(format!(
2272                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2273                    ));
2274                }
2275            };
2276            let clean2 = sanitize_selector(&lr2.content);
2277            self.reporter
2278                .debug(format!("resolved selector (retry): {clean2}"));
2279            if selector_is_useless(&clean2) {
2280                return Err(format!(
2281                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2282                    raw = lr2.content.trim(),
2283                    excerpt = truncate(&page_content.body_text, 300),
2284                ));
2285            }
2286            if !selector_matches(tab, &clean2).unwrap_or(false) {
2287                return Err(format!(
2288                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2289                ));
2290            }
2291            return Ok(clean2);
2292        }
2293
2294        Ok(clean)
2295    }
2296}
2297
2298/// Evaluates a JS expression that is expected to return a boolean.
2299fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2300    tab.evaluate(js, false)
2301        .map_err(|e| format!("evaluate failed: {e}"))?
2302        .value
2303        .and_then(|v| v.as_bool())
2304        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2305}
2306
2307/// Checks whether a CSS selector matches at least one current element.
2308fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2309    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2310}
2311
2312// ── Free helper functions ──────────────────────────────────────────────
2313
2314fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2315    let name = format!("[navigate] {full_url}");
2316    match tab.navigate_to(full_url) {
2317        Ok(_) => {
2318            let _ = tab.wait_until_navigated();
2319            StepResult {
2320                name,
2321                status: StepStatus::Passed,
2322                message: format!("navigated to {full_url}"),
2323            }
2324        }
2325        Err(e) => StepResult {
2326            name,
2327            status: StepStatus::Failed,
2328            message: format!("navigation failed: {e}"),
2329        },
2330    }
2331}
2332
2333fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2334    let result = tab
2335        .evaluate(DOM_EXTRACT_JS, false)
2336        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2337
2338    let json_str = result
2339        .value
2340        .as_ref()
2341        .and_then(|v| v.as_str())
2342        .unwrap_or("[]");
2343
2344    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2345
2346    if elements.is_empty() {
2347        return Ok("(no interactive elements found)".to_owned());
2348    }
2349
2350    Ok(elements.join("\n"))
2351}
2352
2353fn get_page_text(tab: &Tab) -> PageContent {
2354    let url = tab.get_url();
2355
2356    let title = tab
2357        .evaluate("document.title", false)
2358        .ok()
2359        .and_then(|r| r.value)
2360        .and_then(|v| v.as_str().map(String::from))
2361        .unwrap_or_else(|| "unknown".to_owned());
2362
2363    let body_text = tab
2364        .evaluate(
2365            "document.body ? document.body.innerText : document.documentElement.innerText",
2366            false,
2367        )
2368        .ok()
2369        .and_then(|r| r.value)
2370        .and_then(|v| v.as_str().map(String::from))
2371        .unwrap_or_default();
2372
2373    PageContent {
2374        url,
2375        title,
2376        body_text: truncate(&body_text, 8000),
2377    }
2378}
2379
2380fn resolve_url(url: &str, base_url: &str) -> String {
2381    if url.starts_with("http://") || url.starts_with("https://") {
2382        return url.to_owned();
2383    }
2384    let base = base_url.trim_end_matches('/');
2385    if url.starts_with('/') {
2386        format!("{base}{url}")
2387    } else {
2388        format!("{base}/{url}")
2389    }
2390}
2391
2392/// Origin (`scheme://host[:port]`) of a base URL, used to scope the
2393/// per-test `Storage.clearDataForOrigin` call. Returns `None` when the
2394/// URL has no recognizable scheme/host (the clear is skipped).
2395#[must_use]
2396fn origin_of(base_url: &str) -> Option<String> {
2397    let url = if base_url.contains("://") {
2398        base_url.to_owned()
2399    } else {
2400        format!("https://{base_url}")
2401    };
2402    let (scheme, rest) = url.split_once("://")?;
2403    let authority = rest
2404        .split(['/', '?', '#'])
2405        .next()
2406        .filter(|a| !a.is_empty())?;
2407    Some(format!("{scheme}://{authority}"))
2408}
2409
2410/// True when a test performs its own login: its `start_url` or its first
2411/// navigate step targets a login route (org `/auth/login`, tenant
2412/// `/tenant/login`). Such tests need a cleared session — with a live one
2413/// the SPA bounces the login page before the form ever mounts.
2414#[must_use]
2415fn test_targets_login(start_url: &str, steps: &[TestStep]) -> bool {
2416    if start_url.contains("login") {
2417        return true;
2418    }
2419    steps
2420        .iter()
2421        .find_map(|s| match s {
2422            TestStep::Navigate { url, .. } => Some(url.clone()),
2423            _ => None,
2424        })
2425        .is_some_and(|u| u.contains("login"))
2426}
2427
2428/// Human-readable label for a step, used when steps are skipped after an
2429/// earlier failure.
2430fn step_label(step: &TestStep) -> String {
2431    match step {
2432        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2433        TestStep::Click { target, .. } => format!("[click] {target}"),
2434        TestStep::Type { target, .. } => format!("[type] {target}"),
2435        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2436        TestStep::Assert {
2437            definition,
2438            preset,
2439            prompt,
2440            ..
2441        } => definition.as_ref().map_or_else(
2442            || {
2443                preset.as_ref().map_or_else(
2444                    || {
2445                        prompt.as_ref().map_or_else(
2446                            || "[assert]".to_owned(),
2447                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2448                        )
2449                    },
2450                    |p| format!("[assert] {p}"),
2451                )
2452            },
2453            |d| format!("[assert] {d}"),
2454        ),
2455        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2456        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2457        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2458    }
2459}
2460
2461/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2462#[must_use]
2463const fn step_kind_label(step: &TestStep) -> &'static str {
2464    match step {
2465        TestStep::Navigate { .. } => "navigate",
2466        TestStep::Click { .. } => "click",
2467        TestStep::Type { .. } => "type",
2468        TestStep::Wait { .. } => "wait",
2469        TestStep::Assert { .. } => "assert",
2470        TestStep::Screenshot { .. } => "screenshot",
2471        TestStep::Agent { .. } => "agent",
2472        TestStep::Mcp { .. } => "mcp",
2473    }
2474}
2475
2476// ── Support types ──────────────────────────────────────────────────────
2477
2478#[derive(Default)]
2479struct TestRunResult {
2480    passed: u32,
2481    failed: u32,
2482    skipped: u32,
2483    total: u32,
2484    details: Vec<StepResult>,
2485}
2486
2487struct PageContent {
2488    url: String,
2489    title: String,
2490    body_text: String,
2491}
2492
2493#[cfg(test)]
2494mod tests {
2495    use super::unix_to_rfc3339;
2496
2497    #[test]
2498    fn rfc3339_epoch_and_reference_dates() {
2499        assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2500        assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2501        assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2502        assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2503    }
2504
2505    #[test]
2506    fn rfc3339_handles_leap_years() {
2507        assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2508        assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2509    }
2510}
2511
2512#[cfg(test)]
2513mod verdict_parse_tests {
2514    use super::{verdict_is_fail, verdict_is_pass};
2515
2516    #[test]
2517    fn tolerates_markdown_punctuation_and_natural_language() {
2518        assert!(verdict_is_pass("PASS"));
2519        assert!(verdict_is_pass("pass"));
2520        assert!(verdict_is_pass("**PASS**\n\nThe page is clean."));
2521        assert!(verdict_is_pass("passes - no explicit error visible"));
2522        assert!(verdict_is_pass("  \"pass\""));
2523        assert!(!verdict_is_pass("FAIL: something broke"));
2524        assert!(!verdict_is_pass("**FAIL** broken"));
2525
2526        assert!(verdict_is_fail("**FAIL** broken"));
2527        assert!(verdict_is_fail("fails - error toast shown"));
2528        assert!(!verdict_is_fail("passes - ok"));
2529    }
2530}