Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// Formats a UNIX epoch timestamp as an RFC 3339 UTC string
26/// (`2026-09-19T16:04:00Z`) without pulling in a date dependency.
27/// Civil-from-days conversion after Howard Hinnant.
28#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31    let secs = i64::try_from(secs).unwrap_or(0);
32    let days = secs.div_euclid(86_400);
33    let rem = secs.rem_euclid(86_400);
34    let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35    let z = days + 719_468;
36    let era = z.div_euclid(146_097);
37    let doe = z.rem_euclid(146_097);
38    let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40    let mp = (5 * doy + 2) / 153;
41    let d = doy - (153 * mp + 2) / 5 + 1;
42    let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43    let y = yoe + era * 400 + i64::from(mth <= 2);
44    format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47/// One detected layout defect (`layout_no_issues` preset).
48#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50    #[serde(rename = "type")]
51    issue_type: String,
52    element: String,
53    detail: String,
54}
55
56/// In-browser DOM layout scan for `layout_no_issues`.
57///
58/// Geometry-only checks (no LLM, no pixels):
59/// 1. page horizontal overflow (`scrollWidth` > viewport width);
60/// 2. elements outside the viewport that scrolling cannot reveal
61///    (fixed elements off-screen, left/negative overflow, right-edge
62///    overflow beyond the horizontally scrollable area, and bottom
63///    overflow on a page that cannot scroll down) — below-the-fold
64///    content on a tall scrollable page is normal flow, NOT a defect;
65/// 3. text clipped by `overflow: hidden` containers whose content
66///    is measurably larger than the box;
67/// 4. interactive elements (buttons/links/inputs) whose center point
68///    is covered by a different element that would intercept the click.
69///
70/// Elements whose class matches a configured ignore prefix (default:
71/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
72/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
73/// skipped — they are intentionally 1x1 / off-screen. The prefix list
74/// is injected at `__IGNORE_CLASSES__` from
75/// [`ScenarioConfig::layout_ignore_classes`].
76///
77/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
78/// sticky headers) is excluded by the position/relation filters.
79const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81  const issues = [];
82  const push = (type, el, detail) => {
83    if (issues.length >= 30) return;
84    let element = el.tagName.toLowerCase();
85    if (el.id) element += '#' + el.id;
86    else if (typeof el.className === 'string' && el.className.trim())
87      element += '.' + el.className.trim().split(/\s+/).join('.');
88    issues.push({ type, element, detail: String(detail).slice(0, 220) });
89  };
90  const vw = document.documentElement.clientWidth || window.innerWidth;
91  const vh = document.documentElement.clientHeight || window.innerHeight;
92  if (!vw || !vh) return JSON.stringify(issues);
93  const de = document.documentElement;
94  // 1. Page-level horizontal overflow.
95  if (maxSW > vw + 2)
96    push('page-overflow-x', de,
97      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98  // Class-name prefixes to skip (injected; default = Angular CDK
99  // screen-reader helpers, which are intentionally 1x1 / off-screen).
100  const ignorePrefixes = __IGNORE_CLASSES__;
101  const isIgnored = (el) => {
102    if (typeof el.className !== 'string' || !el.className.trim()) return false;
103    const classes = el.className.trim().split(/\s+/);
104    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105  };
106  const body = document.body;
107  // The document element is not always the scroll container: the app may
108  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110  // "not scrollable" — measure the scrollable content height/width of the
111  // document element AND the body, and take the max.
112  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114  const vScrollable = maxSH > vh + 2;
115  const hScrollable = maxSW > vw + 2;
116  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117  // this element into view on the given axis — i.e. the element is inside a
118  // scrolling region, so lying beyond the viewport cut is not a defect.
119  const reachableByScroller = (el, axis) => {
120    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123    let p = el.parentElement;
124    while (p && p !== body) {
125      const pcs = getComputedStyle(p);
126      const o = pcs[ovProp];
127      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128          p[dim] > p[dimClient] + 2) return true;
129      p = p.parentElement;
130    }
131    return false;
132  };
133  const selfOverflowing = (el, cs, axis) => {
134    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138      el[dim] > el[dimClient] + 2;
139  };
140  // True when an ancestor clips this axis with overflow:hidden/clip — the
141  // element's overhang is not visible, so treat it as reachable.
142  const clippedByAncestor = (el, axis) => {
143    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144    let p = el.parentElement;
145    while (p && p !== body) {
146      const o = getComputedStyle(p)[ovProp];
147      if (o === 'hidden' || o === 'clip') return true;
148      p = p.parentElement;
149    }
150    return false;
151  };
152  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154  const hasContent = (el) =>
155    ((el.textContent || '').trim().length > 0) ||
156    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157  // 2. Elements outside the viewport that scrolling cannot reveal.
158  for (const el of all) {
159    const cs = getComputedStyle(el);
160    if (!visible(cs)) continue;
161    if (isIgnored(el)) continue;
162    const r = el.getBoundingClientRect();
163    if (r.width < 2 || r.height < 2) continue;
164    if (!hasContent(el) && el.children.length === 0) continue;
165    if (cs.position === 'fixed') {
166      // Fixed elements never move with the scroll: any edge outside the
167      // viewport is unreachable content and therefore a defect.
168      const overTop = -r.top;
169      const overLeft = -r.left;
170      const overRight = r.right - vw;
171      const overBottom = r.bottom - vh;
172      // Top/left overshoot is never reachable; bottom/right overshoot is
173      // only a defect when the fixed element cannot scroll that content
174      // into view itself (e.g. an opened Material drawer whose inner
175      // container scrolls is not a "cut-off" defect).
176      if (overTop > 2 || overLeft > 2 ||
177          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179        let where = '';
180        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186        push('element-out-of-viewport', el,
187          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188      }
189      continue;
190    }
191    if (cs.position === 'sticky') continue;
192    // Negative left overflow cannot be reached by scrolling (scrollLeft
193    // never goes below 0).
194    if (r.left < -2) {
195      push('element-out-of-viewport', el,
196        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197      continue;
198    }
199    // Negative top with the page at the top means the element sits above
200    // the document origin — also unreachable.
201    if (r.top < -2 && de.scrollTop <= 2) {
202      push('element-out-of-viewport', el,
203        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204      continue;
205    }
206    const overRight = r.right - vw;
207    // Right-edge overflow is only a defect when the page cannot scroll
208    // horizontally to reveal it (or the element sticks out past the
209    // scrollable content width itself).
210    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211      push('element-out-of-viewport', el,
212        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213      continue;
214    }
215    // Below-the-fold content on a scrollable page is normal (tall landing
216    // pages); only flag bottom overflow the user can never scroll to.
217    const overBottom = r.bottom - vh;
218    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219      push('element-out-of-viewport', el,
220        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221    }
222  }
223  // 3. Text clipped by overflow:hidden containers.
224  for (const el of all) {
225    if (isIgnored(el)) continue;
226    const cs = getComputedStyle(el);
227    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229    if (!(el.textContent || '').trim()) continue;
230    // Single-line truncation with an ellipsis is an intentional design
231    // pattern (Tailwind .truncate etc.), not a clipping defect.
232    if (cs.textOverflow === 'ellipsis') continue;
233    push('text-clipped', el,
234      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236  }
237  // 4. Interactive elements covered by a different element.
238  const interactive =
239    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240  const targets = document.querySelectorAll(interactive);
241  for (const el of targets) {
242    if (isIgnored(el)) continue;
243    const r = el.getBoundingClientRect();
244    if (r.width < 6 || r.height < 6) continue;
245    const cs = getComputedStyle(el);
246    if (!visible(cs)) continue;
247    const cx = r.left + r.width / 2;
248    const cy = r.top + r.height / 2;
249    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250    const top = document.elementFromPoint(cx, cy);
251    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252    if (isIgnored(top)) continue;
253    const tcs = getComputedStyle(top);
254    if (!visible(tcs)) continue;
255    if (tcs.pointerEvents === 'none') continue;
256    const tr = top.getBoundingClientRect();
257    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260    push('element-overlap', el,
261      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262      ' is covered by <' + tname + '>');
263  }
264  return JSON.stringify(issues);
265})()
266"#;
267
268/// How long the CDP connection stays open after the browser goes quiet.
269///
270/// `headless_chrome` ships a 30s default and tears down the entire connection
271/// when no traffic arrives for that long; a run must own its connection for
272/// its full duration instead.
273const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275/// Executes a [`Scenario`] against a real browser with optional LLM
276/// assistance for element targeting and assertions.
277pub struct ScenarioRunner {
278    config: ScenarioConfig,
279    definitions: HashMap<String, AssertDefinition>,
280    llm: LlmConfig,
281    timeout: Duration,
282    viewport_width: u32,
283    viewport_height: u32,
284    /// The viewport currently applied in the browser (CDP emulation).
285    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
286    applied_viewport: std::cell::Cell<(u32, u32)>,
287    endpoints: EndpointRegistry,
288    usage: Arc<UsageTracker>,
289    budgets: BudgetTracker,
290    /// Directory for failure artifacts (screenshots).
291    artifacts_dir: PathBuf,
292    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
293    reporter: Arc<Reporter>,
294    /// The test + step index currently executing, for LLM-call events.
295    current_step: std::cell::RefCell<Option<(String, u32)>>,
296    /// Epoch seconds of runner creation — constant for the whole run so
297    /// the LLM context preamble below is cacheable by upstream providers.
298    run_started: u64,
299    /// Whether this runner emits the `RunStarted`/`RunFinished` events.
300    /// When several scenario files are run concurrently by
301    /// [`crate::parallel::run_scenarios`], the orchestrator owns those two
302    /// events (one per batch), so each per-file runner suppresses its own.
303    emit_run_events: bool,
304}
305
306/// Aggregated results from a scenario run.
307#[derive(Debug, Default)]
308pub struct RunReport {
309    /// Number of tests that passed.
310    pub tests_passed: u32,
311    /// Number of tests that failed.
312    pub tests_failed: u32,
313    /// Number of steps that passed.
314    pub passed: u32,
315    /// Number of steps that failed.
316    pub failed: u32,
317    /// Number of steps that were skipped.
318    pub skipped: u32,
319    /// Per-step details.
320    pub details: Vec<StepResult>,
321}
322
323/// Result of a single step execution.
324#[derive(Debug, Clone)]
325pub struct StepResult {
326    /// The step name.
327    pub name: String,
328    /// Whether the step passed, failed, or was skipped.
329    pub status: StepStatus,
330    /// Human-readable result message.
331    pub message: String,
332}
333
334/// Outcome for a single step.
335pub use crate::events::StepStatus;
336
337/// Predefined assertion preset definition.
338struct AssertPreset {
339    name: &'static str,
340    system: &'static str,
341    user_template: &'static str,
342}
343
344/// Built-in assertion presets.
345#[allow(clippy::literal_string_with_formatting_args)]
346const ASSERTION_PRESETS: &[AssertPreset] = &[
347    AssertPreset {
348        name: "no_error_on_page",
349        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
350        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
351    },
352    AssertPreset {
353        name: "text_visible",
354        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
355        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
356    },
357    AssertPreset {
358        name: "element_exists",
359        system: "You are a QA tester. Check if a described UI element exists on a web page.",
360        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
361    },
362    AssertPreset {
363        name: "visual_no_issues",
364        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
365        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
366    },
367    AssertPreset {
368        name: "visual_no_overlaps",
369        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
370        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
371    },
372    AssertPreset {
373        name: "visual_fills_viewport",
374        system: "You are a visual QA engineer inspecting a website screenshot for UNDER-FILLED layouts: the rendered content stops at a fraction of the viewport and the remaining area is a large blank region the layout should have filled. Typical defects: page content ending around the halfway mark of the viewport with an empty, background-only area below it; an app shell, dashboard, or main panel collapsing to part of its expected height and leaving dead space; content that fills only the left portion of a wide viewport with unused space to the right. Do NOT fail intentionally short or centered pages: a login or signup card, a landing hero, a short article, a confirmation step, or an empty state that does not fill the viewport is normal design. Only fail when the blank region is clearly a layout defect — the page looks cut off or collapsed, or a content container (calendar, table, chart, map, panel) visibly fails to stretch to the space its own layout provides.",
375        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nDoes the page content fill the viewport as its layout intends — with no large blank region where the page appears to end partway down or partway across? Respond with exactly \"PASS\" if the page fills the viewport or is intentionally short, or \"FAIL: <describe the under-filled region and where it appears>\" if the content clearly stops at a fraction of the available page.",
376    },
377    AssertPreset {
378        name: "visual_text_visible",
379        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
380        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
381    },
382    AssertPreset {
383        name: "layout_no_issues",
384        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
385        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
386    },
387];
388
389/// Whether an LLM verdict should be read as PASS.
390///
391/// Models routinely wrap the verdict in markdown (`**PASS** …`) or prefix it
392/// with punctuation, which a bare `starts_with("pass")` misreads as a failure.
393/// Strip any leading non-alphanumerics, then match the leading word; this also
394/// accepts the natural-language forms `pass` / `passes`.
395fn verdict_is_pass(content: &str) -> bool {
396    content
397        .trim()
398        .trim_start_matches(|c: char| !c.is_alphanumeric())
399        .to_lowercase()
400        .starts_with("pass")
401}
402
403/// Whether an LLM verdict should be read as FAIL (see [`verdict_is_pass`]).
404fn verdict_is_fail(content: &str) -> bool {
405    content
406        .trim()
407        .trim_start_matches(|c: char| !c.is_alphanumeric())
408        .to_lowercase()
409        .starts_with("fail")
410}
411
412impl ScenarioRunner {
413    /// Creates a new runner with the given scenario configuration and
414    /// assertion definitions.
415    #[must_use]
416    #[allow(clippy::needless_pass_by_value)]
417    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
418        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
419    }
420
421    /// Creates a runner that reports run events through the given reporter.
422    #[must_use]
423    #[allow(clippy::needless_pass_by_value)]
424    pub fn with_reporter(
425        scenario_config: ScenarioConfig,
426        definitions: Vec<AssertDefinition>,
427        reporter: Arc<Reporter>,
428    ) -> Self {
429        Self::with_reporter_mode(scenario_config, definitions, reporter, true)
430    }
431
432    /// Creates a runner that reports run events through the given reporter
433    /// but suppresses the `RunStarted`/`RunFinished` events. Used by the
434    /// parallel orchestrator, which emits those once per batch.
435    #[must_use]
436    #[allow(clippy::needless_pass_by_value)]
437    pub fn with_reporter_parallel(
438        scenario_config: ScenarioConfig,
439        definitions: Vec<AssertDefinition>,
440        reporter: Arc<Reporter>,
441    ) -> Self {
442        Self::with_reporter_mode(scenario_config, definitions, reporter, false)
443    }
444
445    #[must_use]
446    #[allow(clippy::needless_pass_by_value)]
447    fn with_reporter_mode(
448        scenario_config: ScenarioConfig,
449        definitions: Vec<AssertDefinition>,
450        reporter: Arc<Reporter>,
451        emit_run_events: bool,
452    ) -> Self {
453        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
454            &scenario_config,
455        ));
456        let llm = LlmConfig {
457            url: scenario_config
458                .llm_url
459                .clone()
460                .unwrap_or_else(crate::llm_base_url),
461            model: scenario_config
462                .llm_model
463                .clone()
464                .unwrap_or_else(crate::llm_model),
465            api_key: scenario_config
466                .llm_api_key
467                .clone()
468                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
469            headers: if scenario_config.llm_headers.is_empty() {
470                crate::parse_headers_env()
471            } else {
472                scenario_config.llm_headers.clone()
473            },
474            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
475            temperature: scenario_config.temperature,
476            thinking: scenario_config.thinking,
477            model_params: scenario_config.model_params.clone(),
478            cache: scenario_config.cache.unwrap_or(true),
479            max_attempts: crate::default_llm_attempts(),
480            provider: crate::scenario::Provider::Openai,
481            deployment: None,
482            api_version: None,
483            auth: crate::scenario::AuthConfig::default(),
484            header_commands: std::collections::HashMap::new(),
485            aws: crate::scenario::AwsConfig::default(),
486        };
487        let endpoints = EndpointRegistry::from_config(
488            &scenario_config.endpoints,
489            Some(&llm),
490            crate::endpoints::EndpointDefaults {
491                cache: scenario_config.cache,
492                cache_pricing: scenario_config.cache_pricing,
493            },
494        );
495        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
496        let defs_map: HashMap<String, AssertDefinition> = definitions
497            .into_iter()
498            .map(|d| (d.name.clone(), d))
499            .collect();
500
501        Self {
502            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
503            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
504            viewport_height: scenario_config.viewport_height.unwrap_or(720),
505            applied_viewport: std::cell::Cell::new((0, 0)),
506            config: scenario_config.clone(),
507            definitions: defs_map,
508            llm,
509            endpoints,
510            usage: Arc::new(UsageTracker::new()),
511            budgets,
512            artifacts_dir: PathBuf::from(
513                scenario_config
514                    .artifacts_dir
515                    .unwrap_or_else(|| "artifacts".to_owned()),
516            ),
517            reporter,
518            current_step: std::cell::RefCell::new(None),
519            run_started: SystemTime::now()
520                .duration_since(UNIX_EPOCH)
521                .map_or(0, |d| d.as_secs()),
522            emit_run_events,
523        }
524    }
525
526    /// Emits an event; a sink failure degrades to a console warning so a
527    /// broken log file can never mask the run itself.
528    fn emit_event(&self, event: &TestEvent) {
529        if let Err(err) = self.reporter.emit(event) {
530            use std::io::Write as _;
531            let mut out = std::io::stderr().lock();
532            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
533        }
534    }
535
536    /// Returns a clone of the [`UsageTracker`] for reporting.
537    #[must_use]
538    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
539        Arc::clone(&self.usage)
540    }
541
542    /// Returns a reference to the [`BudgetTracker`].
543    #[must_use]
544    pub const fn budget_tracker(&self) -> &BudgetTracker {
545        &self.budgets
546    }
547
548    /// Executes all test groups in the scenario and returns a report.
549    ///
550    /// # Errors
551    ///
552    /// Returns an error if the browser fails to launch.
553    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
554    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
555        let mut report = RunReport::default();
556
557        if tests.is_empty() {
558            self.reporter.warn("No tests defined in scenario.");
559            if self.emit_run_events {
560                self.emit_event(&TestEvent::RunFinished {
561                    tests_passed: 0,
562                    tests_failed: 0,
563                    steps_passed: 0,
564                    steps_failed: 0,
565                    steps_skipped: 0,
566                    total_cost: 0.0,
567                    total_tokens: 0,
568                    total_input_tokens: 0,
569                    total_output_tokens: 0,
570                    total_cached_input_tokens: 0,
571                    total_cache_creation_input_tokens: 0,
572                    models: Vec::new(),
573                    total_calls: 0,
574                });
575            }
576            return Ok(report);
577        }
578
579        if self.emit_run_events {
580            self.emit_event(&TestEvent::RunStarted {
581                total_tests: tests.len() as u32,
582            });
583        }
584
585        let browser_headless = self.config.browser_headless.unwrap_or(true);
586
587        let launch_opts = LaunchOptions {
588            headless: browser_headless,
589            window_size: Some((self.viewport_width, self.viewport_height)),
590            sandbox: false,
591            // headless_chrome defaults this to 30s and shuts down the whole CDP
592            // connection when no messages arrive for that long. A scenario can
593            // easily exceed 30s of browser silence (slow LLM targeting/assertion
594            // calls, page waits, budget checks between steps), after which every
595            // remaining step fails with "Unable to make method calls because
596            // underlying connection is closed" — one quiet gap kills the run.
597            // Open-ended scenarios must own the connection for their full
598            // duration, so keep it alive for 6 hours.
599            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
600            ..LaunchOptions::default()
601        };
602
603        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
604        let tab = browser.new_tab().context("failed to open browser tab")?;
605        match (
606            self.config.browser_basic_auth_user.clone(),
607            self.config.browser_basic_auth_password.clone(),
608        ) {
609            (Some(username), Some(password)) if !username.is_empty() && !password.is_empty() => {
610                tab.authenticate(Some(username), Some(password))
611                    .context("failed to configure browser HTTP Basic Auth")?;
612                tab.enable_fetch(None, Some(true))
613                    .context("failed to enable browser HTTP authentication")?;
614            }
615            (None, None) => {}
616            _ => anyhow::bail!("browser Basic Auth requires both username and password"),
617        }
618        let _ = tab.set_default_timeout(self.timeout);
619
620        // Start MCP server if configured
621        #[cfg(feature = "mcp-server")]
622        if let Some(ref mcp_cfg) = self.config.mcp_server {
623            if mcp_cfg.enabled {
624                let port = mcp_cfg.port;
625                std::thread::spawn(move || {
626                    let _ = crate::mcp_server::start_mcp_server(port);
627                });
628            }
629        }
630        #[cfg(not(feature = "mcp-server"))]
631        if let Some(mcp_cfg) = &self.config.mcp_server {
632            if mcp_cfg.enabled {
633                self.reporter
634                    .warn("MCP server configured but 'mcp-server' feature not enabled");
635            }
636        }
637
638        // Start A2A agent server if configured
639        #[cfg(feature = "a2a-server")]
640        if let Some(ref a2a_cfg) = self.config.a2a_server {
641            if a2a_cfg.enabled {
642                let port = a2a_cfg.port;
643                tokio::spawn(crate::a2a_server::start_a2a_server(port));
644            }
645        }
646        #[cfg(not(feature = "a2a-server"))]
647        if let Some(a2a_cfg) = &self.config.a2a_server {
648            if a2a_cfg.enabled {
649                self.reporter
650                    .warn("A2A server configured but 'a2a-server' feature not enabled");
651            }
652        }
653
654        let retry_budget = self.config.retry_failed_tests.unwrap_or(0);
655
656        for test in tests {
657            // A shared tab on a contended runner makes some page loads
658            // stall (a JS chunk or a GraphQL call hangs mid-flight) and
659            // the test fails on a wait/assert that a fresh run passes.
660            // Re-run the WHOLE test when it failed and the retry budget
661            // allows; the fresh per-test isolation (login clear +
662            // auto-navigate) applies to the retry too. Both attempts'
663            // LLM spend stays in the budget accounting (each attempt
664            // commits its usage); only the final attempt is reported.
665            let details_len = report.details.len();
666            let passed_before = report.passed;
667            let failed_before = report.failed;
668            let skipped_before = report.skipped;
669            #[allow(unused_assignments)]
670            let mut final_result = None;
671            let mut attempt = 0;
672            loop {
673                attempt += 1;
674                self.emit_event(&TestEvent::TestStarted {
675                    test: test.name.clone(),
676                });
677
678                self.usage.reset_per_test();
679
680                let test_started = Instant::now();
681                let test_result = self.run_test(test, &tab);
682                let duration_ms = test_started.elapsed().as_millis() as u64;
683                let usage = self.usage.current_test_snapshot();
684                self.usage.commit_test(&test.name);
685
686                self.emit_event(&TestEvent::TestFinished {
687                    test: test.name.clone(),
688                    passed: test_result.passed,
689                    failed: test_result.failed,
690                    skipped: test_result.skipped,
691                    duration_ms,
692                    cost: usage.total_cost,
693                    tokens: usage.total_tokens,
694                    input_tokens: usage.total_input_tokens,
695                    output_tokens: usage.total_output_tokens,
696                    cached_input_tokens: usage.total_cached_input_tokens,
697                    cache_creation_input_tokens: usage.total_cache_creation_input_tokens,
698                    models: usage.models.clone(),
699                    calls: usage.total_calls,
700                });
701
702                final_result = Some(test_result);
703                let result = final_result.as_ref().expect("just assigned");
704                if result.failed == 0 || result.total == 0 || attempt > retry_budget {
705                    break;
706                }
707                self.reporter.warn(format!(
708                    "! retrying failed test '{}' (attempt {}/{}) — a fresh run passes on \
709                     transient page-load stalls",
710                    test.name,
711                    attempt + 1,
712                    retry_budget + 1
713                ));
714                // Roll this failed attempt out of the report so the
715                // retry's outcome replaces it.
716                report.details.truncate(details_len);
717                report.passed = passed_before;
718                report.failed = failed_before;
719                report.skipped = skipped_before;
720            }
721
722            let test_result = final_result.unwrap_or_default();
723            if test_result.failed == 0 && test_result.total > 0 {
724                report.tests_passed += 1;
725            } else if test_result.total > 0 {
726                report.tests_failed += 1;
727            }
728
729            report.passed += test_result.passed;
730            report.failed += test_result.failed;
731            report.skipped += test_result.skipped;
732            report.details.extend(test_result.details);
733        }
734
735        let global = self.usage.global_snapshot();
736        if self.emit_run_events {
737            self.emit_event(&TestEvent::RunFinished {
738                tests_passed: report.tests_passed,
739                tests_failed: report.tests_failed,
740                steps_passed: report.passed,
741                steps_failed: report.failed,
742                steps_skipped: report.skipped,
743                total_cost: global.total_cost,
744                total_tokens: global.total_tokens,
745                total_input_tokens: global.total_input_tokens,
746                total_output_tokens: global.total_output_tokens,
747                total_cached_input_tokens: global.total_cached_input_tokens,
748                total_cache_creation_input_tokens: global.total_cache_creation_input_tokens,
749                models: global.models.clone(),
750                total_calls: global.total_calls,
751            });
752        }
753
754        Ok(report)
755    }
756
757    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
758    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
759        let base_url = test
760            .base_url
761            .clone()
762            .or_else(|| self.config.base_url.clone())
763            .unwrap_or_else(crate::base_url);
764
765        // Per-test viewport override: switch the browser via CDP
766        // device-metrics emulation before this test runs.
767        let vw = test.viewport_width.unwrap_or(self.viewport_width);
768        let vh = test.viewport_height.unwrap_or(self.viewport_height);
769        if self.applied_viewport.get() != (vw, vh) {
770            self.apply_viewport(tab, vw, vh);
771            self.applied_viewport.set((vw, vh));
772        }
773
774        // Per-test isolation: every test starts from its own start_url
775        // (unless auto_navigate is disabled), so a test never inherits the
776        // previous test's page state.
777        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
778
779        let start_url = test
780            .start_url
781            .clone()
782            .or_else(|| self.config.start_url.clone())
783            .unwrap_or_else(|| "/dashboard".to_owned());
784
785        // Per-test state isolation for LOGIN tests: the tab is shared
786        // across the file's tests, and a previous test's login persists
787        // (Cognito tokens in localStorage + the hosted-UI cookies). The
788        // SPA's login pages then auto-continue authenticated visitors on
789        // boot, so a later login test never sees the form and times out
790        // waiting for `#email` (observed on immosai runs #1641/#1642: a
791        // shard's first three logins pass, the fourth onward time out).
792        // Clear cookies + origin storage ONLY for tests that perform
793        // their own login (their first navigate step targets a login
794        // route, or they have no navigate step and ride the auto-nav
795        // onto a login start_url). The many "page loads" tests that
796        // navigate app pages directly RELY on the shared session — a
797        // blanket clear bounces them off their target (run #1643: every
798        // shard failed with "the current page is the login page, not
799        // the mailboxes page"). The HTTP cache is deliberately NOT
800        // cleared — re-downloading the SPA bundle per test would only
801        // add boot time on a slow runner egress.
802        if auto_navigate && test_targets_login(&start_url, &test.steps) {
803            use headless_chrome::protocol::cdp::{Network, Storage};
804            if let Err(e) = tab.call_method(Network::ClearBrowserCookies(None)) {
805                self.reporter
806                    .warn(format!("per-test cookie clear failed: {e}"));
807            }
808            if let Some(origin) = origin_of(&base_url) {
809                if let Err(e) = tab.call_method(Storage::ClearDataForOrigin {
810                    origin,
811                    storage_Types: "all".to_string(),
812                }) {
813                    self.reporter
814                        .warn(format!("per-test storage clear failed: {e}"));
815                }
816            }
817        }
818        if auto_navigate {
819            let full_url = resolve_url(&start_url, &base_url);
820            self.reporter.debug(format!("auto-navigate: {full_url}"));
821            let _ = tab.navigate_to(&full_url);
822            let _ = tab.wait_until_navigated();
823            std::thread::sleep(Duration::from_secs(4));
824        }
825
826        let mut result = TestRunResult::default();
827
828        for (step_index, step) in test.steps.iter().enumerate() {
829            result.total += 1;
830
831            let wait_ms = match step {
832                TestStep::Navigate { wait_after_ms, .. }
833                | TestStep::Click { wait_after_ms, .. }
834                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
835                _ => None,
836            };
837
838            self.current_step
839                .replace(Some((test.name.clone(), step_index as u32)));
840            self.emit_event(&TestEvent::StepStarted {
841                test: test.name.clone(),
842                index: step_index as u32,
843                label: step_label(step),
844            });
845            let step_started = Instant::now();
846
847            let mut step_result = match step {
848                TestStep::Navigate { url, .. } => {
849                    let full_url = resolve_url(url, &base_url);
850                    run_navigate_step(&full_url, tab)
851                }
852                TestStep::Click {
853                    target,
854                    selector,
855                    endpoint,
856                    idempotent,
857                    ..
858                } => self.run_click(
859                    target,
860                    selector.as_deref(),
861                    endpoint.as_deref(),
862                    test.endpoint.as_deref(),
863                    *idempotent,
864                    tab,
865                ),
866                TestStep::Type {
867                    target,
868                    text,
869                    selector,
870                    endpoint,
871                    idempotent,
872                    ..
873                } => self.run_type(
874                    target,
875                    text,
876                    selector.as_deref(),
877                    endpoint.as_deref(),
878                    test.endpoint.as_deref(),
879                    *idempotent,
880                    tab,
881                ),
882                TestStep::Wait {
883                    target,
884                    selector,
885                    text,
886                    timeout_ms,
887                    endpoint,
888                    idempotent,
889                } => self.run_wait(
890                    target,
891                    selector.as_deref(),
892                    text.as_deref(),
893                    *timeout_ms,
894                    endpoint.as_deref(),
895                    test.endpoint.as_deref(),
896                    *idempotent,
897                    tab,
898                ),
899                TestStep::Assert {
900                    definition,
901                    preset,
902                    prompt,
903                    assert_text,
904                    endpoint,
905                    screenshot,
906                } => self.run_assert(
907                    definition.as_deref(),
908                    preset.as_deref(),
909                    prompt.as_deref(),
910                    assert_text.as_deref(),
911                    *screenshot,
912                    endpoint.as_deref(),
913                    test.endpoint.as_deref(),
914                    tab,
915                ),
916                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
917                TestStep::Agent {
918                    agent,
919                    task,
920                    definition,
921                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
922                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
923            };
924
925            // Failure diagnostics: capture the page state and a screenshot so
926            // CI logs say WHAT the page looked like when the step failed,
927            // instead of a bare "timed out: The event waited for never came".
928            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
929                let state = diagnostics::capture(tab);
930                let screenshot = diagnostics::save_screenshot(
931                    tab,
932                    &self.artifacts_dir,
933                    &test.name,
934                    &test.name,
935                    step_index,
936                    step_kind_label(step),
937                );
938                step_result.message = format!(
939                    "{base} — {excerpt}",
940                    base = step_result.message,
941                    excerpt = diagnostics::inline_excerpt(&state),
942                );
943                (Some(diagnostics::full_context(&state)), screenshot)
944            } else {
945                (None, None)
946            };
947
948            let duration_ms = step_started.elapsed().as_millis() as u64;
949            self.emit_event(&TestEvent::StepFinished {
950                test: test.name.clone(),
951                index: step_index as u32,
952                label: step_result.name.clone(),
953                status: step_result.status,
954                duration_ms,
955                message: step_result.message.clone(),
956                diagnostics: diagnostics_block,
957                screenshot: screenshot_path,
958            });
959            self.current_step.replace(None);
960
961            match step_result.status {
962                StepStatus::Passed => result.passed += 1,
963                StepStatus::Failed => result.failed += 1,
964                StepStatus::Skipped => result.skipped += 1,
965            }
966
967            // Fail fast: the first failed step ends the test and the
968            // remaining steps are reported as skipped (no LLM budget is
969            // burned asserting against a page that is already known broken).
970            if step_result.status == StepStatus::Failed
971                && !self.config.continue_on_failure
972                && step_index + 1 < test.steps.len()
973            {
974                self.emit_event(&TestEvent::Warning {
975                    message: format!(
976                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
977                        test.steps.len() - step_index - 1
978                    ),
979                });
980                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
981                    let skipped_index = step_index + 1 + offset;
982                    let label = step_label(skipped);
983                    result.total += 1;
984                    result.skipped += 1;
985                    self.emit_event(&TestEvent::StepStarted {
986                        test: test.name.clone(),
987                        index: skipped_index as u32,
988                        label: label.clone(),
989                    });
990                    self.emit_event(&TestEvent::StepFinished {
991                        test: test.name.clone(),
992                        index: skipped_index as u32,
993                        label,
994                        status: StepStatus::Skipped,
995                        duration_ms: 0,
996                        message: "skipped: previous step failed".into(),
997                        diagnostics: None,
998                        screenshot: None,
999                    });
1000                    result.details.push(StepResult {
1001                        name: step_label(skipped),
1002                        status: StepStatus::Skipped,
1003                        message: "skipped: previous step failed".into(),
1004                    });
1005                }
1006                result.details.push(step_result);
1007                return result;
1008            }
1009
1010            // Check per-test budget after each step
1011            let test_usage = self.usage.current_test_snapshot();
1012            let global_usage = self.usage.global_snapshot();
1013            let budget_status = self.budgets.check_all(
1014                &test.name,
1015                &test_usage,
1016                &global_usage,
1017                test.budget.as_ref(),
1018            );
1019            match budget_status {
1020                BudgetStatus::HardExceeded { message, .. } => {
1021                    self.emit_event(&TestEvent::Warning {
1022                        message: format!("budget exceeded: {message}"),
1023                    });
1024                    result.details.push(StepResult {
1025                        name: "[budget]".into(),
1026                        status: StepStatus::Failed,
1027                        message,
1028                    });
1029                    result.failed += 1;
1030                    return result;
1031                }
1032                BudgetStatus::SoftExceeded { message, .. } => {
1033                    self.emit_event(&TestEvent::Warning {
1034                        message: format!("budget warning: {message}"),
1035                    });
1036                }
1037                BudgetStatus::Ok => {}
1038            }
1039
1040            if let Some(ms) = wait_ms {
1041                std::thread::sleep(Duration::from_millis(ms));
1042            }
1043
1044            result.details.push(step_result);
1045        }
1046
1047        result
1048    }
1049
1050    /// Applies a viewport size to the current tab via CDP
1051    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
1052    /// overrides and the viewport matrix. The initial window size set at
1053    /// browser launch is replaced by emulation; failures are logged but
1054    /// do not fail the test (a mismatched viewport only weakens coverage).
1055    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
1056        use headless_chrome::protocol::cdp::Emulation;
1057        let _ = self;
1058        let params = Emulation::SetDeviceMetricsOverride {
1059            width,
1060            height,
1061            device_scale_factor: 1.0,
1062            mobile: false,
1063            scale: None,
1064            screen_width: Some(width),
1065            screen_height: Some(height),
1066            position_x: None,
1067            position_y: None,
1068            dont_set_visible_size: None,
1069            screen_orientation: None,
1070            viewport: None,
1071            display_feature: None,
1072            device_posture: None,
1073        };
1074        self.reporter.debug(format!("viewport: {width}x{height}"));
1075        if let Err(e) = tab.call_method(params) {
1076            self.reporter
1077                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
1078        }
1079    }
1080
1081    /// Height (px) covered by assert-step screenshots: the configured
1082    /// `screenshot_max_height` (absolute px or viewport multiple, default
1083    /// `"20x"`) resolved against the currently applied viewport, raised to
1084    /// at least the viewport height so the visible screen is always fully
1085    /// included. The capture is split into viewport-tall tiles, so this
1086    /// value bounds total coverage (and hence the number of image parts).
1087    #[must_use]
1088    fn screenshot_height_cap(&self) -> u32 {
1089        let viewport_height = self.current_viewport_height();
1090        let cap = self
1091            .config
1092            .screenshot_max_height
1093            .as_ref()
1094            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
1095        cap.max(viewport_height)
1096    }
1097
1098    /// Height of the viewport currently emulated in the browser (falling
1099    /// back to the configured default before any emulation was applied).
1100    #[must_use]
1101    const fn current_viewport_height(&self) -> u32 {
1102        let (_, height) = self.applied_viewport.get();
1103        if height > 0 {
1104            height
1105        } else {
1106            self.viewport_height
1107        }
1108    }
1109
1110    // ── step handlers ───────────────────────────────────────────────────
1111
1112    #[allow(clippy::too_many_lines)]
1113    fn run_click(
1114        &self,
1115        target: &str,
1116        selector_override: Option<&str>,
1117        step_endpoint: Option<&str>,
1118        test_endpoint: Option<&str>,
1119        idempotent: bool,
1120        tab: &Tab,
1121    ) -> StepResult {
1122        let name = format!("[click] {target}");
1123        let selector = match self.resolve_selector(
1124            selector_override,
1125            target,
1126            step_endpoint,
1127            test_endpoint,
1128            tab,
1129        ) {
1130            Ok(s) => s,
1131            Err(msg) => {
1132                if idempotent {
1133                    return StepResult {
1134                        name,
1135                        status: StepStatus::Skipped,
1136                        message: format!("skipped (idempotent): no target found — {msg}"),
1137                    };
1138                }
1139                return StepResult {
1140                    name,
1141                    status: StepStatus::Failed,
1142                    message: msg,
1143                };
1144            }
1145        };
1146
1147        // Idempotent steps probe briefly: a missing target means the
1148        // action was already done / not applicable (e.g. an
1149        // already-authenticated session), and skipping is the success
1150        // path, not a failure.
1151        let probe_secs = if idempotent { 5 } else { 10 };
1152        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1153            Ok(element) => match element.click() {
1154                Ok(_) => StepResult {
1155                    name,
1156                    status: StepStatus::Passed,
1157                    message: format!("clicked {selector}"),
1158                },
1159                Err(e) => StepResult {
1160                    name,
1161                    status: StepStatus::Failed,
1162                    message: format!("click failed on {selector}: {e}"),
1163                },
1164            },
1165            Err(e) if idempotent => StepResult {
1166                name,
1167                status: StepStatus::Skipped,
1168                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1169            },
1170            Err(e) => StepResult {
1171                name,
1172                status: StepStatus::Failed,
1173                message: format!("element {selector} not found: {e}"),
1174            },
1175        }
1176    }
1177
1178    #[allow(clippy::too_many_arguments)]
1179    fn run_type(
1180        &self,
1181        target: &str,
1182        text: &str,
1183        selector_override: Option<&str>,
1184        step_endpoint: Option<&str>,
1185        test_endpoint: Option<&str>,
1186        idempotent: bool,
1187        tab: &Tab,
1188    ) -> StepResult {
1189        let name = format!("[type] {target}");
1190        let selector = match self.resolve_selector(
1191            selector_override,
1192            target,
1193            step_endpoint,
1194            test_endpoint,
1195            tab,
1196        ) {
1197            Ok(s) => s,
1198            Err(msg) => {
1199                if idempotent {
1200                    return StepResult {
1201                        name,
1202                        status: StepStatus::Skipped,
1203                        message: format!("skipped (idempotent): no target found — {msg}"),
1204                    };
1205                }
1206                return StepResult {
1207                    name,
1208                    status: StepStatus::Failed,
1209                    message: msg,
1210                };
1211            }
1212        };
1213
1214        let probe_secs = if idempotent { 5 } else { 10 };
1215        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1216            Ok(element) => {
1217                if let Err(e) = element.click() {
1218                    return StepResult {
1219                        name,
1220                        status: StepStatus::Failed,
1221                        message: format!("click to focus {selector} failed: {e}"),
1222                    };
1223                }
1224
1225                let js = format!(
1226                    "document.querySelector('{}').value = '';",
1227                    selector.replace('\'', "\\'")
1228                );
1229                let _ = tab.evaluate(&js, false);
1230
1231                match element.type_into(text) {
1232                    Ok(_) => StepResult {
1233                        name,
1234                        status: StepStatus::Passed,
1235                        message: format!("typed {text:?} into {selector}"),
1236                    },
1237                    Err(e) => StepResult {
1238                        name,
1239                        status: StepStatus::Failed,
1240                        message: format!("type into {selector} failed: {e}"),
1241                    },
1242                }
1243            }
1244            Err(e) if idempotent => StepResult {
1245                name,
1246                status: StepStatus::Skipped,
1247                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1248            },
1249            Err(e) => StepResult {
1250                name,
1251                status: StepStatus::Failed,
1252                message: format!("element {selector} not found: {e}"),
1253            },
1254        }
1255    }
1256
1257    #[allow(clippy::too_many_arguments)]
1258    #[allow(clippy::too_many_lines)]
1259    fn run_wait(
1260        &self,
1261        target: &str,
1262        selector_override: Option<&str>,
1263        text: Option<&str>,
1264        timeout_ms: Option<u64>,
1265        step_endpoint: Option<&str>,
1266        test_endpoint: Option<&str>,
1267        idempotent: bool,
1268        tab: &Tab,
1269    ) -> StepResult {
1270        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1271        let step_name = format!("[wait] {target}");
1272
1273        // Resolve an explicit selector only (text-only waits are LLM-free).
1274        let selector = match selector_override {
1275            Some(s) => Some(s.to_owned()),
1276            None if text.is_some() => None,
1277            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1278                Ok(s) => Some(s),
1279                Err(msg) => {
1280                    if idempotent {
1281                        return StepResult {
1282                            name: step_name,
1283                            status: StepStatus::Skipped,
1284                            message: format!("skipped (idempotent): no target found — {msg}"),
1285                        };
1286                    }
1287                    return StepResult {
1288                        name: step_name,
1289                        status: StepStatus::Failed,
1290                        message: msg,
1291                    };
1292                }
1293            },
1294        };
1295
1296        if text.is_some() {
1297            let sel_js = selector
1298                .as_deref()
1299                .map(crate::selectors::selector_matches_js);
1300            let text_js = text.map(|t| {
1301                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1302                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1303            });
1304
1305            let mut reload_note = String::new();
1306            let mut reloaded = false;
1307            loop {
1308                let deadline = Instant::now() + timeout;
1309                loop {
1310                    let sel_ok = sel_js
1311                        .as_ref()
1312                        .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1313                    let text_ok = text_js
1314                        .as_ref()
1315                        .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1316                    if sel_ok && text_ok {
1317                        let mut what = Vec::new();
1318                        if let Some(sel) = &selector {
1319                            what.push(format!("found {sel}"));
1320                        }
1321                        if let Some(t) = text {
1322                            what.push(format!("text {t:?} visible"));
1323                        }
1324                        return StepResult {
1325                            name: step_name,
1326                            status: StepStatus::Passed,
1327                            message: format!("{}{reload_note}", what.join(" and ")),
1328                        };
1329                    }
1330                    if Instant::now() >= deadline {
1331                        break;
1332                    }
1333                    std::thread::sleep(Duration::from_millis(250));
1334                }
1335                // Blank-page self-heal: a stalled runner egress leaves the
1336                // SPA unbooted (index.html loaded, JS chunks never arrived,
1337                // immosai runs #1646/#1647 — every failure screenshot is an
1338                // empty viewport). A wait against that page can never
1339                // succeed; reload once and re-wait with the full budget.
1340                if !reloaded && page_is_blank(tab) {
1341                    reloaded = true;
1342                    " (blank page — reloaded once and re-waited)".clone_into(&mut reload_note);
1343                    let _ = tab.reload(true, None);
1344                    let _ = tab.wait_until_navigated();
1345                    std::thread::sleep(Duration::from_secs(2));
1346                    continue;
1347                }
1348                let mut what = Vec::new();
1349                if let Some(sel) = &selector {
1350                    what.push(sel.clone());
1351                }
1352                if let Some(t) = text {
1353                    what.push(format!("text {t:?}"));
1354                }
1355                let message = format!(
1356                    "wait for {} timed out after {}ms: the event waited for never came{reload_note}",
1357                    what.join(" / "),
1358                    timeout.as_millis(),
1359                );
1360                if idempotent {
1361                    return StepResult {
1362                        name: step_name,
1363                        status: StepStatus::Skipped,
1364                        message: format!("skipped (idempotent): {message}"),
1365                    };
1366                }
1367                return StepResult {
1368                    name: step_name,
1369                    status: StepStatus::Failed,
1370                    message,
1371                };
1372            }
1373        }
1374
1375        let mut reload_note = String::new();
1376        let mut reloaded = false;
1377        let result = loop {
1378            match selector.as_deref() {
1379                Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1380                    Ok(_) => break Ok((sel.to_owned(), reload_note)),
1381                    Err(e) => {
1382                        // Blank-page self-heal (see the text-wait branch):
1383                        // an unbooted SPA never renders the target; reload
1384                        // once and re-wait with the full budget.
1385                        if !reloaded && page_is_blank(tab) {
1386                            reloaded = true;
1387                            " (blank page — reloaded once and re-waited)"
1388                                .clone_into(&mut reload_note);
1389                            let _ = tab.reload(true, None);
1390                            let _ = tab.wait_until_navigated();
1391                            std::thread::sleep(Duration::from_secs(2));
1392                            continue;
1393                        }
1394                        if idempotent {
1395                            break Err((
1396                                format!(
1397                                    "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1398                                    timeout.as_millis()
1399                                ),
1400                                StepStatus::Skipped,
1401                            ));
1402                        }
1403                        break Err((
1404                            format!(
1405                                "wait for {sel} timed out after {}ms: {e}{reload_note}",
1406                                timeout.as_millis()
1407                            ),
1408                            StepStatus::Failed,
1409                        ));
1410                    }
1411                },
1412                None => break Err((
1413                    "wait step has neither selector nor text".to_owned(),
1414                    StepStatus::Failed,
1415                )),
1416            }
1417        };
1418        match result {
1419            Ok((sel, note)) => StepResult {
1420                name: step_name,
1421                status: StepStatus::Passed,
1422                message: format!("found {sel}{note}"),
1423            },
1424            Err((message, status)) => StepResult {
1425                name: step_name,
1426                status,
1427                message,
1428            },
1429        }
1430    }
1431
1432    #[allow(clippy::too_many_arguments)]
1433    fn run_assert(
1434        &self,
1435        definition: Option<&str>,
1436        preset: Option<&str>,
1437        prompt: Option<&str>,
1438        assert_text: Option<&str>,
1439        screenshot: bool,
1440        step_endpoint: Option<&str>,
1441        test_endpoint: Option<&str>,
1442        tab: &Tab,
1443    ) -> StepResult {
1444        std::thread::sleep(Duration::from_millis(500));
1445
1446        let page_content = get_page_text(tab);
1447
1448        // Vision attach: capture the full page once per assert step and
1449        // split it into viewport-tall tiles (the total coverage is bounded
1450        // by the configured height cap so vision tokens stay sane). All
1451        // tile data URLs are handed to the preset/prompt evaluation below.
1452        let image: Option<Vec<String>> = if screenshot {
1453            let endpoint = self
1454                .endpoints
1455                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1456            if !endpoint.vision {
1457                return StepResult {
1458                    name: "[assert]".into(),
1459                    status: StepStatus::Failed,
1460                    message: format!(
1461                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1462                        name = endpoint.name
1463                    ),
1464                };
1465            }
1466            match crate::vision::capture_screenshot_data_urls(
1467                tab,
1468                self.config
1469                    .screenshot_max_dimension
1470                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1471                self.screenshot_height_cap(),
1472                self.current_viewport_height(),
1473            ) {
1474                Ok(urls) => Some(urls),
1475                Err(e) => {
1476                    return StepResult {
1477                        name: "[assert]".into(),
1478                        status: StepStatus::Failed,
1479                        message: format!("screenshot capture failed: {e}"),
1480                    };
1481                }
1482            }
1483        } else {
1484            None
1485        };
1486
1487        if let Some(def_name) = definition {
1488            if let Some(def) = self.definitions.get(def_name) {
1489                return self.run_assert_def(
1490                    def,
1491                    &page_content,
1492                    image.as_deref(),
1493                    step_endpoint,
1494                    test_endpoint,
1495                    tab,
1496                );
1497            }
1498            return StepResult {
1499                name: format!("[assert] {def_name}"),
1500                status: StepStatus::Failed,
1501                message: format!("definition '{def_name}' not found"),
1502            };
1503        }
1504
1505        if let Some(preset_name) = preset {
1506            // Deterministic DOM layout scan — runs JS in the browser and
1507            // never calls the LLM (free, fast, no pixel budget).
1508            if preset_name == "layout_no_issues" {
1509                return self.run_layout_preset(tab);
1510            }
1511            return self.run_preset(
1512                preset_name,
1513                assert_text,
1514                &page_content,
1515                image.as_deref(),
1516                step_endpoint,
1517                test_endpoint,
1518            );
1519        }
1520
1521        if let Some(prompt_text) = prompt {
1522            return self.run_custom(
1523                prompt_text,
1524                &page_content,
1525                image.as_deref(),
1526                step_endpoint,
1527                test_endpoint,
1528            );
1529        }
1530
1531        StepResult {
1532            name: "[assert]".into(),
1533            status: StepStatus::Skipped,
1534            message: "no definition, preset, or prompt specified".into(),
1535        }
1536    }
1537
1538    fn run_assert_def(
1539        &self,
1540        def: &AssertDefinition,
1541        page_content: &PageContent,
1542        image: Option<&[String]>,
1543        step_endpoint: Option<&str>,
1544        test_endpoint: Option<&str>,
1545        tab: &Tab,
1546    ) -> StepResult {
1547        // Agent-based definition: delegate to an A2A agent
1548        if let Some(ref agent) = def.agent {
1549            if image.is_some() {
1550                return StepResult {
1551                    name: format!("[assert] {}", def.name),
1552                    status: StepStatus::Failed,
1553                    message: "agent-backed assertions do not support screenshots".into(),
1554                };
1555            }
1556            let task = def
1557                .task_template
1558                .as_deref()
1559                .unwrap_or("Evaluate the assertion")
1560                .replace("{url}", &page_content.url)
1561                .replace("{title}", &page_content.title)
1562                .replace("{content}", &page_content.body_text)
1563                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1564
1565            return self.run_agent_step(agent, &task, &def.name);
1566        }
1567
1568        // Custom preset: system + user_template provided in the definition
1569        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1570            return self.run_custom_preset(
1571                &def.name,
1572                system,
1573                template,
1574                def.assert_text.as_deref(),
1575                page_content,
1576                image,
1577                step_endpoint,
1578                test_endpoint,
1579            );
1580        }
1581
1582        def.preset.as_ref().map_or_else(
1583            || {
1584                def.prompt.as_ref().map_or_else(
1585                    || StepResult {
1586                        name: format!("[assert] {}", def.name),
1587                        status: StepStatus::Failed,
1588                        message: "definition has no preset, prompt, or system+user_template".into(),
1589                    },
1590                    |prompt| {
1591                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1592                    },
1593                )
1594            },
1595            |preset_name| {
1596                if preset_name == "layout_no_issues" {
1597                    return self.run_layout_preset(tab);
1598                }
1599                self.run_preset(
1600                    preset_name,
1601                    def.assert_text.as_deref(),
1602                    page_content,
1603                    image,
1604                    step_endpoint,
1605                    test_endpoint,
1606                )
1607            },
1608        )
1609    }
1610
1611    #[allow(clippy::too_many_arguments)]
1612    fn run_custom_preset(
1613        &self,
1614        name: &str,
1615        system: &str,
1616        template: &str,
1617        assert_text: Option<&str>,
1618        page_content: &PageContent,
1619        image: Option<&[String]>,
1620        step_endpoint: Option<&str>,
1621        test_endpoint: Option<&str>,
1622    ) -> StepResult {
1623        let user_prompt = template
1624            .replace("{url}", &page_content.url)
1625            .replace("{title}", &page_content.title)
1626            .replace("{content}", &page_content.body_text)
1627            .replace("{expected_text}", assert_text.unwrap_or(""))
1628            .replace("{description}", "");
1629
1630        // Custom preset definitions frequently forget the {content}
1631        // placeholder — without it the LLM has no page to evaluate and
1632        // answers "I can't determine that without seeing the page". Always
1633        // append the page context unless the template already references it.
1634        let user_prompt = if template.contains("{content}") {
1635            user_prompt
1636        } else {
1637            format!(
1638                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1639                url = page_content.url,
1640                title = page_content.title,
1641                content = page_content.body_text,
1642            )
1643        };
1644
1645        self.reporter
1646            .debug(format!("assert: {name} (custom preset)"));
1647
1648        let chain = self
1649            .endpoints
1650            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1651        let sys = system.to_owned();
1652
1653        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1654
1655        response.map_or_else(
1656            |e| StepResult {
1657                name: format!("[assert] {name}"),
1658                status: StepStatus::Failed,
1659                message: format!("LLM assertion call failed: {e}"),
1660            },
1661            |(lr, _idx)| {
1662                if verdict_is_pass(&lr.content) {
1663                    StepResult {
1664                        name: format!("[assert] {name}"),
1665                        status: StepStatus::Passed,
1666                        message: "PASS".into(),
1667                    }
1668                } else {
1669                    StepResult {
1670                        name: format!("[assert] {name}"),
1671                        status: StepStatus::Failed,
1672                        message: lr.content,
1673                    }
1674                }
1675            },
1676        )
1677    }
1678
1679    fn run_preset(
1680        &self,
1681        preset_name: &str,
1682        assert_text: Option<&str>,
1683        page_content: &PageContent,
1684        image: Option<&[String]>,
1685        step_endpoint: Option<&str>,
1686        test_endpoint: Option<&str>,
1687    ) -> StepResult {
1688        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1689            return StepResult {
1690                name: format!("[assert] {preset_name}"),
1691                status: StepStatus::Failed,
1692                message: format!("unknown assertion preset: {preset_name}"),
1693            };
1694        };
1695        if preset_name.starts_with("visual_") && image.is_none() {
1696            return StepResult {
1697                name: format!("[assert] {preset_name}"),
1698                status: StepStatus::Failed,
1699                message: format!(
1700                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1701                ),
1702            };
1703        }
1704
1705        let user_prompt = preset
1706            .user_template
1707            .replace("{url}", &page_content.url)
1708            .replace("{title}", &page_content.title)
1709            .replace("{content}", &page_content.body_text)
1710            .replace("{expected_text}", assert_text.unwrap_or(""))
1711            .replace("{description}", "");
1712
1713        // Same safety net as custom presets: never let the LLM answer with
1714        // no page context at all.
1715        let user_prompt = if preset.user_template.contains("{content}") {
1716            user_prompt
1717        } else {
1718            format!(
1719                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1720                url = page_content.url,
1721                title = page_content.title,
1722                content = page_content.body_text,
1723            )
1724        };
1725
1726        self.reporter.debug(format!("assert: {preset_name}"));
1727
1728        let chain = self
1729            .endpoints
1730            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1731        let sys = preset.system.to_owned();
1732
1733        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1734
1735        response.map_or_else(
1736            |e| StepResult {
1737                name: format!("[assert] {preset_name}"),
1738                status: StepStatus::Failed,
1739                message: format!("LLM assertion call failed: {e}"),
1740            },
1741            |(lr, _idx)| {
1742                if verdict_is_pass(&lr.content) {
1743                    StepResult {
1744                        name: format!("[assert] {preset_name}"),
1745                        status: StepStatus::Passed,
1746                        message: "PASS".into(),
1747                    }
1748                } else {
1749                    StepResult {
1750                        name: format!("[assert] {preset_name}"),
1751                        status: StepStatus::Failed,
1752                        message: lr.content,
1753                    }
1754                }
1755            },
1756        )
1757    }
1758
1759    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1760    ///
1761    /// Evaluates the layout-scan JS in the page and fails with the list of
1762    /// detected issues: horizontal page overflow, visible elements sticking
1763    /// out of the viewport, text clipped by `overflow: hidden` containers,
1764    /// and interactive elements covered by other elements. No LLM call —
1765    /// checks are geometry-based so the check is free, deterministic, and
1766    /// safe to run on every page × viewport variant.
1767    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1768        let name = "[assert] layout_no_issues".to_owned();
1769        self.reporter
1770            .debug("assert: layout_no_issues (DOM layout scan)");
1771        let js = LAYOUT_SCAN_JS.replace(
1772            "__IGNORE_CLASSES__",
1773            &serde_json::to_string(&self.config.layout_ignore_classes)
1774                .unwrap_or_else(|_| "[]".to_owned()),
1775        );
1776        let result = tab.evaluate(&js, false);
1777        let json_str = match result {
1778            Ok(r) => r
1779                .value
1780                .as_ref()
1781                .and_then(|v| v.as_str().map(String::from))
1782                .unwrap_or_else(|| "[]".to_owned()),
1783            Err(e) => {
1784                return StepResult {
1785                    name,
1786                    status: StepStatus::Failed,
1787                    message: format!("layout scan JS failed: {e}"),
1788                };
1789            }
1790        };
1791        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1792        if issues.is_empty() {
1793            return StepResult {
1794                name,
1795                status: StepStatus::Passed,
1796                message: "PASS — no layout defects detected".into(),
1797            };
1798        }
1799        let mut lines: Vec<String> = issues
1800            .iter()
1801            .take(10)
1802            .map(|i| {
1803                format!(
1804                    "- [{type_}] {element}: {detail}",
1805                    type_ = i.issue_type,
1806                    element = i.element,
1807                    detail = i.detail
1808                )
1809            })
1810            .collect();
1811        if issues.len() > 10 {
1812            lines.push(format!("- … and {} more", issues.len() - 10));
1813        }
1814        StepResult {
1815            name,
1816            status: StepStatus::Failed,
1817            message: format!(
1818                "FAIL — {} layout defect(s) detected:\n{}",
1819                issues.len(),
1820                lines.join("\n")
1821            ),
1822        }
1823    }
1824
1825    fn run_custom(
1826        &self,
1827        prompt: &str,
1828        page_content: &PageContent,
1829        image: Option<&[String]>,
1830        step_endpoint: Option<&str>,
1831        test_endpoint: Option<&str>,
1832    ) -> StepResult {
1833        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1834
1835        let mut user = format!(
1836            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1837            url = page_content.url,
1838            title = page_content.title,
1839            content = page_content.body_text,
1840        );
1841        if image.is_some() {
1842            user.push_str(
1843                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1844            );
1845        }
1846
1847        self.reporter.debug("custom assert");
1848
1849        let chain = self
1850            .endpoints
1851            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1852        let sys = system.to_owned();
1853
1854        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1855
1856        response.map_or_else(
1857            |e| StepResult {
1858                name: "[assert] custom".into(),
1859                status: StepStatus::Failed,
1860                message: format!("LLM assertion call failed: {e}"),
1861            },
1862            |(lr, _idx)| {
1863                if verdict_is_pass(&lr.content) {
1864                    StepResult {
1865                        name: "[assert] custom".into(),
1866                        status: StepStatus::Passed,
1867                        message: "PASS".into(),
1868                    }
1869                } else {
1870                    StepResult {
1871                        name: "[assert] custom".into(),
1872                        status: StepStatus::Failed,
1873                        message: lr.content,
1874                    }
1875                }
1876            },
1877        )
1878    }
1879
1880    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1881        let path = path.unwrap_or("screenshot.png");
1882
1883        match tab.capture_screenshot(
1884            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1885            None,
1886            None,
1887            true,
1888        ) {
1889            Ok(data) => {
1890                if let Err(e) = std::fs::write(path, &data) {
1891                    return StepResult {
1892                        name: format!("[screenshot] {path}"),
1893                        status: StepStatus::Failed,
1894                        message: format!("failed to write screenshot: {e}"),
1895                    };
1896                }
1897                StepResult {
1898                    name: format!("[screenshot] {path}"),
1899                    status: StepStatus::Passed,
1900                    message: format!("saved to {path}"),
1901                }
1902            }
1903            Err(e) => StepResult {
1904                name: format!("[screenshot] {path}"),
1905                status: StepStatus::Failed,
1906                message: format!("screenshot failed: {e}"),
1907            },
1908        }
1909    }
1910
1911    /// Runs an A2A agent step.
1912    #[allow(clippy::literal_string_with_formatting_args)]
1913    fn run_agent(
1914        &self,
1915        agent_name: &str,
1916        task: &str,
1917        definition: Option<&str>,
1918        _test_endpoint: Option<&str>,
1919    ) -> StepResult {
1920        // If a definition is specified, look up the task template
1921        let resolved_task = if let Some(def_name) = definition {
1922            if let Some(def) = self.definitions.get(def_name) {
1923                let tmpl = def.task_template.as_deref().unwrap_or(task);
1924                tmpl.replace("{task}", task)
1925            } else {
1926                return StepResult {
1927                    name: format!("[agent] {def_name}"),
1928                    status: StepStatus::Failed,
1929                    message: format!("definition '{def_name}' not found"),
1930                };
1931            }
1932        } else {
1933            task.to_owned()
1934        };
1935
1936        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1937    }
1938
1939    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1940        let Some(ep) = self.endpoints.get(agent_name) else {
1941            return StepResult {
1942                name: format!("[agent] {display_name}"),
1943                status: StepStatus::Failed,
1944                message: format!("agent endpoint '{agent_name}' not found"),
1945            };
1946        };
1947
1948        if ep.url.is_empty() {
1949            return StepResult {
1950                name: format!("[agent] {display_name}"),
1951                status: StepStatus::Failed,
1952                message: format!("agent endpoint '{agent_name}' has no URL"),
1953            };
1954        }
1955
1956        self.reporter.debug(format!("agent {agent_name}: {task}"));
1957
1958        let url = ep.url.clone();
1959        let client = A2aClient::new(&url, self.timeout);
1960        let task_clone = task.to_owned();
1961
1962        let response = std::thread::spawn(move || {
1963            let rt = tokio::runtime::Builder::new_current_thread()
1964                .enable_all()
1965                .build()
1966                .unwrap();
1967            rt.block_on(client.send_task(&task_clone))
1968        })
1969        .join()
1970        .unwrap();
1971
1972        // Record the flat-cost call
1973        self.usage.record_flat_call(agent_name, ep);
1974
1975        match response {
1976            Ok(text) => {
1977                let clean = text.trim().to_owned();
1978                if verdict_is_pass(&clean) {
1979                    StepResult {
1980                        name: format!("[agent] {display_name}"),
1981                        status: StepStatus::Passed,
1982                        message: format!("PASS: {clean}"),
1983                    }
1984                } else if verdict_is_fail(&clean) {
1985                    StepResult {
1986                        name: format!("[agent] {display_name}"),
1987                        status: StepStatus::Failed,
1988                        message: clean,
1989                    }
1990                } else {
1991                    StepResult {
1992                        name: format!("[agent] {display_name}"),
1993                        status: StepStatus::Passed,
1994                        message: format!("response: {clean}"),
1995                    }
1996                }
1997            }
1998            Err(e) => StepResult {
1999                name: format!("[agent] {display_name}"),
2000                status: StepStatus::Failed,
2001                message: format!("agent call failed: {e}"),
2002            },
2003        }
2004    }
2005
2006    /// Runs an MCP tool call step.
2007    fn run_mcp(
2008        &self,
2009        server_name: &str,
2010        tool_name: &str,
2011        args: Option<&serde_json::Value>,
2012    ) -> StepResult {
2013        let Some(ep) = self.endpoints.get(server_name) else {
2014            return StepResult {
2015                name: format!("[mcp] {server_name}:{tool_name}"),
2016                status: StepStatus::Failed,
2017                message: format!("MCP server endpoint '{server_name}' not found"),
2018            };
2019        };
2020
2021        let cmd = ep.command.as_deref().unwrap_or("");
2022        if cmd.is_empty() {
2023            return StepResult {
2024                name: format!("[mcp] {server_name}:{tool_name}"),
2025                status: StepStatus::Failed,
2026                message: format!("MCP server '{server_name}' has no command configured"),
2027            };
2028        }
2029
2030        self.reporter
2031            .debug(format!("mcp {server_name} {tool_name}"));
2032
2033        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
2034
2035        let command = cmd.to_owned();
2036        let args_vec = ep.args.clone();
2037        let tool = tool_name.to_owned();
2038
2039        let response = std::thread::spawn(move || {
2040            let mut mcp_client =
2041                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
2042            mcp_client
2043                .call_tool(&tool, &args_val)
2044                .map_err(|e| e.to_string())
2045        })
2046        .join()
2047        .unwrap();
2048
2049        // Record the flat-cost call
2050        self.usage.record_flat_call(server_name, ep);
2051
2052        match response {
2053            Ok(result) => {
2054                if result.isError {
2055                    StepResult {
2056                        name: format!("[mcp] {server_name}:{tool_name}"),
2057                        status: StepStatus::Failed,
2058                        message: result.to_string(),
2059                    }
2060                } else {
2061                    StepResult {
2062                        name: format!("[mcp] {server_name}:{tool_name}"),
2063                        status: StepStatus::Passed,
2064                        message: result.to_string(),
2065                    }
2066                }
2067            }
2068            Err(e) => StepResult {
2069                name: format!("[mcp] {server_name}:{tool_name}"),
2070                status: StepStatus::Failed,
2071                message: format!("MCP call failed: {e}"),
2072            },
2073        }
2074    }
2075
2076    // ── helpers ──────────────────────────────────────────────────────────
2077
2078    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
2079    /// the runner's default LLM config for any unset fields.
2080    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
2081        LlmConfig {
2082            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
2083                // Bedrock builds its endpoint from the resolved AWS region
2084                // when no URL is given — never inherit the default LLM URL.
2085                endpoint.url.clone()
2086            } else if endpoint.url.is_empty() {
2087                self.llm.url.clone()
2088            } else {
2089                endpoint.url.clone()
2090            },
2091            model: endpoint
2092                .model
2093                .clone()
2094                .unwrap_or_else(|| self.llm.model.clone()),
2095            api_key: endpoint
2096                .api_key
2097                .clone()
2098                .or_else(|| self.llm.api_key.clone()),
2099            headers: if endpoint.headers.is_empty() {
2100                self.llm.headers.clone()
2101            } else {
2102                endpoint.headers.clone()
2103            },
2104            timeout: self.llm.timeout,
2105            temperature: self.llm.temperature,
2106            thinking: self.llm.thinking,
2107            model_params: self.llm.model_params.clone(),
2108            cache: endpoint.cache_markers,
2109            max_attempts: endpoint.max_attempts.max(1),
2110            provider: endpoint.provider,
2111            deployment: endpoint.deployment.clone(),
2112            api_version: endpoint.api_version.clone(),
2113            auth: endpoint.auth.clone(),
2114            header_commands: endpoint.header_commands.clone(),
2115            aws: endpoint.aws.clone(),
2116        }
2117    }
2118
2119    /// Runs a single LLM call against an ordered endpoint chain (primary +
2120    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
2121    /// the first endpoint that answers wins. Returns the response together
2122    /// with the chain index of the answering endpoint (0 = primary) so the
2123    /// caller can attribute usage to the correct endpoint.
2124    ///
2125    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
2126    /// shows duration, tokens, cost and the answering endpoint per call.
2127    /// Run-level context handed to every LLM call as the FIRST block of the
2128    /// user message. Contents that are stable for the whole run ("run
2129    /// started", "target site") come first so upstream provider prefix
2130    /// caching stays effective; the current time is the last line because
2131    /// it changes on every call.
2132    fn run_context(&self) -> String {
2133        let mut parts = vec![
2134            "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
2135            "================================================================".into(),
2136            format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
2137        ];
2138        if let Some(base) = self.config.base_url.as_deref() {
2139            parts.push(format!("Target site: {base}"));
2140        }
2141        let now = SystemTime::now()
2142            .duration_since(UNIX_EPOCH)
2143            .map_or(0, |d| d.as_secs());
2144        parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
2145        parts.join("\n")
2146    }
2147
2148    #[allow(clippy::cast_possible_truncation, clippy::too_many_lines)]
2149    fn llm_call_chain(
2150        &self,
2151        chain: &[&ResolvedEndpoint],
2152        system: &str,
2153        user: &str,
2154        image: Option<&[String]>,
2155        purpose: &str,
2156    ) -> Result<(crate::costs::LlmResponse, usize), String> {
2157        if chain.is_empty() {
2158            return Err("empty LLM endpoint chain".into());
2159        }
2160        let primary = self.build_llm_for_endpoint(chain[0]);
2161        let fallbacks: Vec<LlmConfig> = chain[1..]
2162            .iter()
2163            .map(|e| self.build_llm_for_endpoint(e))
2164            .collect();
2165
2166        let (test, index) = self
2167            .current_step
2168            .borrow()
2169            .as_ref()
2170            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
2171        let primary_endpoint = chain[0].name.clone();
2172        let primary_model = primary.model.clone();
2173        self.emit_event(&TestEvent::LlmCallStarted {
2174            test: test.clone(),
2175            index,
2176            endpoint: primary_endpoint.clone(),
2177            model: primary_model.clone(),
2178            purpose: purpose.to_owned(),
2179        });
2180
2181        let started = Instant::now();
2182        let sys = system.to_owned();
2183        let context = self.run_context();
2184        let user = if context.is_empty() {
2185            user.to_owned()
2186        } else {
2187            format!("{context}\n\n{user}")
2188        };
2189        let image = image.map(<[String]>::to_vec);
2190
2191        let result = std::thread::spawn(move || {
2192            let rt = tokio::runtime::Builder::new_current_thread()
2193                .enable_all()
2194                .build()
2195                .unwrap();
2196            let call = async {
2197                match image.as_deref() {
2198                    Some(img) => {
2199                        llm_chat_vision_with_usage_chain(
2200                            &primary,
2201                            &fallbacks,
2202                            &sys,
2203                            &user,
2204                            Some(img),
2205                        )
2206                        .await
2207                    }
2208                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2209                }
2210            };
2211            rt.block_on(call)
2212        })
2213        .join()
2214        .unwrap();
2215
2216        let duration_ms = started.elapsed().as_millis() as u64;
2217        match result {
2218            Ok((lr, idx)) => {
2219                let cost = calculate_llm_cost(chain[idx], &lr.usage);
2220                let answering = chain[idx].name.clone();
2221                let model = chain[idx]
2222                    .model
2223                    .clone()
2224                    .unwrap_or_else(|| primary_model.clone());
2225                self.usage
2226                    .record_llm_call(&answering, chain[idx], &model, &lr.usage);
2227                self.emit_event(&TestEvent::LlmCallFinished {
2228                    test,
2229                    index,
2230                    endpoint: answering,
2231                    model,
2232                    purpose: purpose.to_owned(),
2233                    ok: true,
2234                    duration_ms,
2235                    input_tokens: lr.usage.prompt_tokens,
2236                    output_tokens: lr.usage.completion_tokens,
2237                    cached_input_tokens: lr.usage.cached_input_tokens,
2238                    cache_creation_input_tokens: lr.usage.cache_creation_input_tokens,
2239                    cost,
2240                    error: None,
2241                });
2242                Ok((lr, idx))
2243            }
2244            Err(e) => {
2245                self.emit_event(&TestEvent::LlmCallFinished {
2246                    test,
2247                    index,
2248                    endpoint: primary_endpoint,
2249                    model: primary_model,
2250                    purpose: purpose.to_owned(),
2251                    ok: false,
2252                    duration_ms,
2253                    input_tokens: 0,
2254                    output_tokens: 0,
2255                    cached_input_tokens: 0,
2256                    cache_creation_input_tokens: 0,
2257                    cost: 0.0,
2258                    error: Some(e.clone()),
2259                });
2260                Err(e)
2261            }
2262        }
2263    }
2264
2265    /// Resolves a CSS selector for the target element. Uses the explicit
2266    /// `selector` if provided, otherwise asks the LLM to find the element
2267    /// from the natural language `target` description and page DOM.
2268    ///
2269    /// LLM responses are sanitized and verified against the live page: a
2270    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2271    /// immediately with the raw LLM output, and a selector that matches
2272    /// nothing triggers one retry with feedback before failing.
2273    #[allow(clippy::too_many_lines)]
2274    fn resolve_selector(
2275        &self,
2276        css_override: Option<&str>,
2277        target: &str,
2278        step_endpoint: Option<&str>,
2279        test_endpoint: Option<&str>,
2280        tab: &Tab,
2281    ) -> Result<String, String> {
2282        if let Some(explicit) = css_override {
2283            return Ok(explicit.to_owned());
2284        }
2285
2286        let dom_info = extract_dom_info(tab)?;
2287        let page_content = get_page_text(tab);
2288
2289        let system = concat!(
2290            "You are a browser automation selector generator. ",
2291            "Given a web page's content and interactive elements, ",
2292            "return ONLY the best CSS selector for the described element. ",
2293            "Output nothing except the CSS selector. ",
2294            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2295            "[name=\"...\"], tag.class, tag. ",
2296            "Never output explanations, markdown, or extra text."
2297        );
2298
2299        let user = format!(
2300            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2301            page_content.url,
2302            page_content.title,
2303            truncate(&page_content.body_text, 4000),
2304            dom_info,
2305            target,
2306        );
2307
2308        let retry_user = format!(
2309            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2310            "The selector must match at least one element currently present on the page.",
2311            page_content.url,
2312            page_content.title,
2313            truncate(&page_content.body_text, 4000),
2314            dom_info,
2315            target,
2316        );
2317
2318        self.reporter.debug(format!("LLM targeting: {target}"));
2319
2320        let chain = self
2321            .endpoints
2322            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2323        let sys = system.to_owned();
2324
2325        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2326
2327        let first = call_llm(&user);
2328        let (lr, _idx) = match first {
2329            Ok(lr) => lr,
2330            Err(e) => {
2331                return Err(format!("LLM element targeting failed: {e}"));
2332            }
2333        };
2334        let clean = sanitize_selector(&lr.content);
2335        self.reporter.debug(format!("resolved selector: {clean}"));
2336
2337        if selector_is_useless(&clean) {
2338            return Err(format!(
2339                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2340                raw = lr.content.trim(),
2341            ));
2342        }
2343        if let Err(reason) = validate_selector(&clean) {
2344            return Err(format!(
2345                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2346                raw = lr.content.trim(),
2347            ));
2348        }
2349        if !selector_matches(tab, &clean).unwrap_or(false) {
2350            // One retry with feedback: flaky models occasionally invent a
2351            // selector that does not exist on the page.
2352            self.reporter.warn(format!(
2353                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2354            ));
2355            let second = call_llm(&retry_user);
2356            let (lr2, _idx2) = match second {
2357                Ok(lr2) => lr2,
2358                Err(e) => {
2359                    return Err(format!(
2360                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2361                    ));
2362                }
2363            };
2364            let clean2 = sanitize_selector(&lr2.content);
2365            self.reporter
2366                .debug(format!("resolved selector (retry): {clean2}"));
2367            if selector_is_useless(&clean2) {
2368                return Err(format!(
2369                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2370                    raw = lr2.content.trim(),
2371                    excerpt = truncate(&page_content.body_text, 300),
2372                ));
2373            }
2374            if !selector_matches(tab, &clean2).unwrap_or(false) {
2375                return Err(format!(
2376                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2377                ));
2378            }
2379            return Ok(clean2);
2380        }
2381
2382        Ok(clean)
2383    }
2384}
2385
2386/// Evaluates a JS expression that is expected to return a boolean.
2387fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2388    tab.evaluate(js, false)
2389        .map_err(|e| format!("evaluate failed: {e}"))?
2390        .value
2391        .and_then(|v| v.as_bool())
2392        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2393}
2394
2395/// True when the tab rendered nothing meaningful: no body, an empty
2396/// body, or body text that is only whitespace. This is the signature
2397/// of a stalled SPA boot (index.html served, JS chunks never arrived —
2398/// runner egress stall), where any wait can only time out. A real
2399/// error page (e.g. `ERR_CONNECTION_REFUSED`) carries text and is NOT
2400/// blank, so those are left alone.
2401fn page_is_blank(tab: &Tab) -> bool {
2402    const JS: &str = "(() => { if (!document.body) return true; \
2403        const t = (document.body.innerText || '').trim(); \
2404        return document.body.childElementCount === 0 || t.length === 0; })()";
2405    eval_bool(tab, JS).unwrap_or(false)
2406}
2407
2408/// Checks whether a CSS selector matches at least one current element.
2409fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2410    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2411}
2412
2413// ── Free helper functions ──────────────────────────────────────────────
2414
2415fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2416    let name = format!("[navigate] {full_url}");
2417    match tab.navigate_to(full_url) {
2418        Ok(_) => {
2419            let _ = tab.wait_until_navigated();
2420            StepResult {
2421                name,
2422                status: StepStatus::Passed,
2423                message: format!("navigated to {full_url}"),
2424            }
2425        }
2426        Err(e) => StepResult {
2427            name,
2428            status: StepStatus::Failed,
2429            message: format!("navigation failed: {e}"),
2430        },
2431    }
2432}
2433
2434fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2435    let result = tab
2436        .evaluate(DOM_EXTRACT_JS, false)
2437        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2438
2439    let json_str = result
2440        .value
2441        .as_ref()
2442        .and_then(|v| v.as_str())
2443        .unwrap_or("[]");
2444
2445    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2446
2447    if elements.is_empty() {
2448        return Ok("(no interactive elements found)".to_owned());
2449    }
2450
2451    Ok(elements.join("\n"))
2452}
2453
2454fn get_page_text(tab: &Tab) -> PageContent {
2455    let url = tab.get_url();
2456
2457    let title = tab
2458        .evaluate("document.title", false)
2459        .ok()
2460        .and_then(|r| r.value)
2461        .and_then(|v| v.as_str().map(String::from))
2462        .unwrap_or_else(|| "unknown".to_owned());
2463
2464    let body_text = tab
2465        .evaluate(
2466            "document.body ? document.body.innerText : document.documentElement.innerText",
2467            false,
2468        )
2469        .ok()
2470        .and_then(|r| r.value)
2471        .and_then(|v| v.as_str().map(String::from))
2472        .unwrap_or_default();
2473
2474    PageContent {
2475        url,
2476        title,
2477        body_text: truncate(&body_text, 8000),
2478    }
2479}
2480
2481fn resolve_url(url: &str, base_url: &str) -> String {
2482    if url.starts_with("http://") || url.starts_with("https://") {
2483        return url.to_owned();
2484    }
2485    let base = base_url.trim_end_matches('/');
2486    if url.starts_with('/') {
2487        format!("{base}{url}")
2488    } else {
2489        format!("{base}/{url}")
2490    }
2491}
2492
2493/// Origin (`scheme://host[:port]`) of a base URL, used to scope the
2494/// per-test `Storage.clearDataForOrigin` call. Returns `None` when the
2495/// URL has no recognizable scheme/host (the clear is skipped).
2496#[must_use]
2497fn origin_of(base_url: &str) -> Option<String> {
2498    let url = if base_url.contains("://") {
2499        base_url.to_owned()
2500    } else {
2501        format!("https://{base_url}")
2502    };
2503    let (scheme, rest) = url.split_once("://")?;
2504    let authority = rest
2505        .split(['/', '?', '#'])
2506        .next()
2507        .filter(|a| !a.is_empty())?;
2508    Some(format!("{scheme}://{authority}"))
2509}
2510
2511/// True when a test performs its own login: its first navigate step
2512/// targets a login route (org `/auth/login`, tenant `/tenant/login`),
2513/// or it has no navigate step at all and rides the auto-navigate onto a
2514/// login `start_url`. Such tests need a cleared session — with a live one
2515/// the SPA bounces the login page before the form ever mounts. Tests
2516/// whose own first navigate goes to an app page keep the shared session
2517/// (most "page loads" tests rely on it) — a blanket clear broke exactly
2518/// those on immosai run #1643.
2519#[must_use]
2520fn test_targets_login(start_url: &str, steps: &[TestStep]) -> bool {
2521    for step in steps {
2522        if let TestStep::Navigate { url, .. } = step {
2523            return url.contains("login");
2524        }
2525    }
2526    start_url.contains("login")
2527}
2528
2529/// Human-readable label for a step, used when steps are skipped after an
2530/// earlier failure.
2531fn step_label(step: &TestStep) -> String {
2532    match step {
2533        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2534        TestStep::Click { target, .. } => format!("[click] {target}"),
2535        TestStep::Type { target, .. } => format!("[type] {target}"),
2536        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2537        TestStep::Assert {
2538            definition,
2539            preset,
2540            prompt,
2541            ..
2542        } => definition.as_ref().map_or_else(
2543            || {
2544                preset.as_ref().map_or_else(
2545                    || {
2546                        prompt.as_ref().map_or_else(
2547                            || "[assert]".to_owned(),
2548                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2549                        )
2550                    },
2551                    |p| format!("[assert] {p}"),
2552                )
2553            },
2554            |d| format!("[assert] {d}"),
2555        ),
2556        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2557        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2558        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2559    }
2560}
2561
2562/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2563#[must_use]
2564const fn step_kind_label(step: &TestStep) -> &'static str {
2565    match step {
2566        TestStep::Navigate { .. } => "navigate",
2567        TestStep::Click { .. } => "click",
2568        TestStep::Type { .. } => "type",
2569        TestStep::Wait { .. } => "wait",
2570        TestStep::Assert { .. } => "assert",
2571        TestStep::Screenshot { .. } => "screenshot",
2572        TestStep::Agent { .. } => "agent",
2573        TestStep::Mcp { .. } => "mcp",
2574    }
2575}
2576
2577// ── Support types ──────────────────────────────────────────────────────
2578
2579#[derive(Default)]
2580struct TestRunResult {
2581    passed: u32,
2582    failed: u32,
2583    skipped: u32,
2584    total: u32,
2585    details: Vec<StepResult>,
2586}
2587
2588struct PageContent {
2589    url: String,
2590    title: String,
2591    body_text: String,
2592}
2593
2594#[cfg(test)]
2595mod tests {
2596    use super::unix_to_rfc3339;
2597
2598    #[test]
2599    fn rfc3339_epoch_and_reference_dates() {
2600        assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2601        assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2602        assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2603        assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2604    }
2605
2606    #[test]
2607    fn rfc3339_handles_leap_years() {
2608        assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2609        assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2610    }
2611}
2612
2613#[cfg(test)]
2614mod verdict_parse_tests {
2615    use super::{verdict_is_fail, verdict_is_pass};
2616
2617    #[test]
2618    fn tolerates_markdown_punctuation_and_natural_language() {
2619        assert!(verdict_is_pass("PASS"));
2620        assert!(verdict_is_pass("pass"));
2621        assert!(verdict_is_pass("**PASS**\n\nThe page is clean."));
2622        assert!(verdict_is_pass("passes - no explicit error visible"));
2623        assert!(verdict_is_pass("  \"pass\""));
2624        assert!(!verdict_is_pass("FAIL: something broke"));
2625        assert!(!verdict_is_pass("**FAIL** broken"));
2626
2627        assert!(verdict_is_fail("**FAIL** broken"));
2628        assert!(verdict_is_fail("fails - error toast shown"));
2629        assert!(!verdict_is_fail("passes - ok"));
2630    }
2631}