Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// Formats a UNIX epoch timestamp as an RFC 3339 UTC string
26/// (`2026-09-19T16:04:00Z`) without pulling in a date dependency.
27/// Civil-from-days conversion after Howard Hinnant.
28#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31    let secs = i64::try_from(secs).unwrap_or(0);
32    let days = secs.div_euclid(86_400);
33    let rem = secs.rem_euclid(86_400);
34    let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35    let z = days + 719_468;
36    let era = z.div_euclid(146_097);
37    let doe = z.rem_euclid(146_097);
38    let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40    let mp = (5 * doy + 2) / 153;
41    let d = doy - (153 * mp + 2) / 5 + 1;
42    let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43    let y = yoe + era * 400 + i64::from(mth <= 2);
44    format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47/// One detected layout defect (`layout_no_issues` preset).
48#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50    #[serde(rename = "type")]
51    issue_type: String,
52    element: String,
53    detail: String,
54}
55
56/// In-browser DOM layout scan for `layout_no_issues`.
57///
58/// Geometry-only checks (no LLM, no pixels):
59/// 1. page horizontal overflow (`scrollWidth` > viewport width);
60/// 2. elements outside the viewport that scrolling cannot reveal
61///    (fixed elements off-screen, left/negative overflow, right-edge
62///    overflow beyond the horizontally scrollable area, and bottom
63///    overflow on a page that cannot scroll down) — below-the-fold
64///    content on a tall scrollable page is normal flow, NOT a defect;
65/// 3. text clipped by `overflow: hidden` containers whose content
66///    is measurably larger than the box;
67/// 4. interactive elements (buttons/links/inputs) whose center point
68///    is covered by a different element that would intercept the click.
69///
70/// Elements whose class matches a configured ignore prefix (default:
71/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
72/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
73/// skipped — they are intentionally 1x1 / off-screen. The prefix list
74/// is injected at `__IGNORE_CLASSES__` from
75/// [`ScenarioConfig::layout_ignore_classes`].
76///
77/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
78/// sticky headers) is excluded by the position/relation filters.
79const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81  const issues = [];
82  const push = (type, el, detail) => {
83    if (issues.length >= 30) return;
84    let element = el.tagName.toLowerCase();
85    if (el.id) element += '#' + el.id;
86    else if (typeof el.className === 'string' && el.className.trim())
87      element += '.' + el.className.trim().split(/\s+/).join('.');
88    issues.push({ type, element, detail: String(detail).slice(0, 220) });
89  };
90  const vw = document.documentElement.clientWidth || window.innerWidth;
91  const vh = document.documentElement.clientHeight || window.innerHeight;
92  if (!vw || !vh) return JSON.stringify(issues);
93  const de = document.documentElement;
94  // 1. Page-level horizontal overflow.
95  if (maxSW > vw + 2)
96    push('page-overflow-x', de,
97      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98  // Class-name prefixes to skip (injected; default = Angular CDK
99  // screen-reader helpers, which are intentionally 1x1 / off-screen).
100  const ignorePrefixes = __IGNORE_CLASSES__;
101  const isIgnored = (el) => {
102    if (typeof el.className !== 'string' || !el.className.trim()) return false;
103    const classes = el.className.trim().split(/\s+/);
104    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105  };
106  const body = document.body;
107  // The document element is not always the scroll container: the app may
108  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110  // "not scrollable" — measure the scrollable content height/width of the
111  // document element AND the body, and take the max.
112  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114  const vScrollable = maxSH > vh + 2;
115  const hScrollable = maxSW > vw + 2;
116  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117  // this element into view on the given axis — i.e. the element is inside a
118  // scrolling region, so lying beyond the viewport cut is not a defect.
119  const reachableByScroller = (el, axis) => {
120    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123    let p = el.parentElement;
124    while (p && p !== body) {
125      const pcs = getComputedStyle(p);
126      const o = pcs[ovProp];
127      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128          p[dim] > p[dimClient] + 2) return true;
129      p = p.parentElement;
130    }
131    return false;
132  };
133  const selfOverflowing = (el, cs, axis) => {
134    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138      el[dim] > el[dimClient] + 2;
139  };
140  // True when an ancestor clips this axis with overflow:hidden/clip — the
141  // element's overhang is not visible, so treat it as reachable.
142  const clippedByAncestor = (el, axis) => {
143    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144    let p = el.parentElement;
145    while (p && p !== body) {
146      const o = getComputedStyle(p)[ovProp];
147      if (o === 'hidden' || o === 'clip') return true;
148      p = p.parentElement;
149    }
150    return false;
151  };
152  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154  const hasContent = (el) =>
155    ((el.textContent || '').trim().length > 0) ||
156    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157  // 2. Elements outside the viewport that scrolling cannot reveal.
158  for (const el of all) {
159    const cs = getComputedStyle(el);
160    if (!visible(cs)) continue;
161    if (isIgnored(el)) continue;
162    const r = el.getBoundingClientRect();
163    if (r.width < 2 || r.height < 2) continue;
164    if (!hasContent(el) && el.children.length === 0) continue;
165    if (cs.position === 'fixed') {
166      // Fixed elements never move with the scroll: any edge outside the
167      // viewport is unreachable content and therefore a defect.
168      const overTop = -r.top;
169      const overLeft = -r.left;
170      const overRight = r.right - vw;
171      const overBottom = r.bottom - vh;
172      // Top/left overshoot is never reachable; bottom/right overshoot is
173      // only a defect when the fixed element cannot scroll that content
174      // into view itself (e.g. an opened Material drawer whose inner
175      // container scrolls is not a "cut-off" defect).
176      if (overTop > 2 || overLeft > 2 ||
177          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179        let where = '';
180        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186        push('element-out-of-viewport', el,
187          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188      }
189      continue;
190    }
191    if (cs.position === 'sticky') continue;
192    // Negative left overflow cannot be reached by scrolling (scrollLeft
193    // never goes below 0).
194    if (r.left < -2) {
195      push('element-out-of-viewport', el,
196        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197      continue;
198    }
199    // Negative top with the page at the top means the element sits above
200    // the document origin — also unreachable.
201    if (r.top < -2 && de.scrollTop <= 2) {
202      push('element-out-of-viewport', el,
203        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204      continue;
205    }
206    const overRight = r.right - vw;
207    // Right-edge overflow is only a defect when the page cannot scroll
208    // horizontally to reveal it (or the element sticks out past the
209    // scrollable content width itself).
210    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211      push('element-out-of-viewport', el,
212        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213      continue;
214    }
215    // Below-the-fold content on a scrollable page is normal (tall landing
216    // pages); only flag bottom overflow the user can never scroll to.
217    const overBottom = r.bottom - vh;
218    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219      push('element-out-of-viewport', el,
220        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221    }
222  }
223  // 3. Text clipped by overflow:hidden containers.
224  for (const el of all) {
225    if (isIgnored(el)) continue;
226    const cs = getComputedStyle(el);
227    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229    if (!(el.textContent || '').trim()) continue;
230    // Single-line truncation with an ellipsis is an intentional design
231    // pattern (Tailwind .truncate etc.), not a clipping defect.
232    if (cs.textOverflow === 'ellipsis') continue;
233    push('text-clipped', el,
234      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236  }
237  // 4. Interactive elements covered by a different element.
238  const interactive =
239    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240  const targets = document.querySelectorAll(interactive);
241  for (const el of targets) {
242    if (isIgnored(el)) continue;
243    const r = el.getBoundingClientRect();
244    if (r.width < 6 || r.height < 6) continue;
245    const cs = getComputedStyle(el);
246    if (!visible(cs)) continue;
247    const cx = r.left + r.width / 2;
248    const cy = r.top + r.height / 2;
249    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250    const top = document.elementFromPoint(cx, cy);
251    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252    if (isIgnored(top)) continue;
253    const tcs = getComputedStyle(top);
254    if (!visible(tcs)) continue;
255    if (tcs.pointerEvents === 'none') continue;
256    const tr = top.getBoundingClientRect();
257    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260    push('element-overlap', el,
261      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262      ' is covered by <' + tname + '>');
263  }
264  return JSON.stringify(issues);
265})()
266"#;
267
268/// How long the CDP connection stays open after the browser goes quiet.
269///
270/// `headless_chrome` ships a 30s default and tears down the entire connection
271/// when no traffic arrives for that long; a run must own its connection for
272/// its full duration instead.
273const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275/// Executes a [`Scenario`] against a real browser with optional LLM
276/// assistance for element targeting and assertions.
277pub struct ScenarioRunner {
278    config: ScenarioConfig,
279    definitions: HashMap<String, AssertDefinition>,
280    llm: LlmConfig,
281    timeout: Duration,
282    viewport_width: u32,
283    viewport_height: u32,
284    /// The viewport currently applied in the browser (CDP emulation).
285    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
286    applied_viewport: std::cell::Cell<(u32, u32)>,
287    endpoints: EndpointRegistry,
288    usage: Arc<UsageTracker>,
289    budgets: BudgetTracker,
290    /// Directory for failure artifacts (screenshots).
291    artifacts_dir: PathBuf,
292    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
293    reporter: Arc<Reporter>,
294    /// The test + step index currently executing, for LLM-call events.
295    current_step: std::cell::RefCell<Option<(String, u32)>>,
296    /// Epoch seconds of runner creation — constant for the whole run so
297    /// the LLM context preamble below is cacheable by upstream providers.
298    run_started: u64,
299    /// Whether this runner emits the `RunStarted`/`RunFinished` events.
300    /// When several scenario files are run concurrently by
301    /// [`crate::parallel::run_scenarios`], the orchestrator owns those two
302    /// events (one per batch), so each per-file runner suppresses its own.
303    emit_run_events: bool,
304}
305
306/// Aggregated results from a scenario run.
307#[derive(Debug, Default)]
308pub struct RunReport {
309    /// Number of tests that passed.
310    pub tests_passed: u32,
311    /// Number of tests that failed.
312    pub tests_failed: u32,
313    /// Number of steps that passed.
314    pub passed: u32,
315    /// Number of steps that failed.
316    pub failed: u32,
317    /// Number of steps that were skipped.
318    pub skipped: u32,
319    /// Per-step details.
320    pub details: Vec<StepResult>,
321}
322
323/// Result of a single step execution.
324#[derive(Debug, Clone)]
325pub struct StepResult {
326    /// The step name.
327    pub name: String,
328    /// Whether the step passed, failed, or was skipped.
329    pub status: StepStatus,
330    /// Human-readable result message.
331    pub message: String,
332}
333
334/// Outcome for a single step.
335pub use crate::events::StepStatus;
336
337/// Predefined assertion preset definition.
338struct AssertPreset {
339    name: &'static str,
340    system: &'static str,
341    user_template: &'static str,
342}
343
344/// Built-in assertion presets.
345#[allow(clippy::literal_string_with_formatting_args)]
346const ASSERTION_PRESETS: &[AssertPreset] = &[
347    AssertPreset {
348        name: "no_error_on_page",
349        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
350        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
351    },
352    AssertPreset {
353        name: "text_visible",
354        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
355        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
356    },
357    AssertPreset {
358        name: "element_exists",
359        system: "You are a QA tester. Check if a described UI element exists on a web page.",
360        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
361    },
362    AssertPreset {
363        name: "visual_no_issues",
364        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
365        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
366    },
367    AssertPreset {
368        name: "visual_no_overlaps",
369        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
370        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
371    },
372    AssertPreset {
373        name: "visual_fills_viewport",
374        system: "You are a visual QA engineer inspecting a website screenshot for UNDER-FILLED layouts: the rendered content stops at a fraction of the viewport and the remaining area is a large blank region the layout should have filled. Typical defects: page content ending around the halfway mark of the viewport with an empty, background-only area below it; an app shell, dashboard, or main panel collapsing to part of its expected height and leaving dead space; content that fills only the left portion of a wide viewport with unused space to the right. Do NOT fail intentionally short or centered pages: a login or signup card, a landing hero, a short article, a confirmation step, or an empty state that does not fill the viewport is normal design. Only fail when the blank region is clearly a layout defect — the page looks cut off or collapsed, or a content container (calendar, table, chart, map, panel) visibly fails to stretch to the space its own layout provides.",
375        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nDoes the page content fill the viewport as its layout intends — with no large blank region where the page appears to end partway down or partway across? Respond with exactly \"PASS\" if the page fills the viewport or is intentionally short, or \"FAIL: <describe the under-filled region and where it appears>\" if the content clearly stops at a fraction of the available page.",
376    },
377    AssertPreset {
378        name: "visual_text_visible",
379        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
380        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
381    },
382    AssertPreset {
383        name: "layout_no_issues",
384        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
385        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
386    },
387];
388
389/// Whether an LLM verdict should be read as PASS.
390///
391/// Models routinely wrap the verdict in markdown (`**PASS** …`) or prefix it
392/// with punctuation, which a bare `starts_with("pass")` misreads as a failure.
393/// Strip any leading non-alphanumerics, then match the leading word; this also
394/// accepts the natural-language forms `pass` / `passes`.
395fn verdict_is_pass(content: &str) -> bool {
396    content
397        .trim()
398        .trim_start_matches(|c: char| !c.is_alphanumeric())
399        .to_lowercase()
400        .starts_with("pass")
401}
402
403/// Whether an LLM verdict should be read as FAIL (see [`verdict_is_pass`]).
404fn verdict_is_fail(content: &str) -> bool {
405    content
406        .trim()
407        .trim_start_matches(|c: char| !c.is_alphanumeric())
408        .to_lowercase()
409        .starts_with("fail")
410}
411
412impl ScenarioRunner {
413    /// Creates a new runner with the given scenario configuration and
414    /// assertion definitions.
415    #[must_use]
416    #[allow(clippy::needless_pass_by_value)]
417    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
418        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
419    }
420
421    /// Creates a runner that reports run events through the given reporter.
422    #[must_use]
423    #[allow(clippy::needless_pass_by_value)]
424    pub fn with_reporter(
425        scenario_config: ScenarioConfig,
426        definitions: Vec<AssertDefinition>,
427        reporter: Arc<Reporter>,
428    ) -> Self {
429        Self::with_reporter_mode(scenario_config, definitions, reporter, true)
430    }
431
432    /// Creates a runner that reports run events through the given reporter
433    /// but suppresses the `RunStarted`/`RunFinished` events. Used by the
434    /// parallel orchestrator, which emits those once per batch.
435    #[must_use]
436    #[allow(clippy::needless_pass_by_value)]
437    pub fn with_reporter_parallel(
438        scenario_config: ScenarioConfig,
439        definitions: Vec<AssertDefinition>,
440        reporter: Arc<Reporter>,
441    ) -> Self {
442        Self::with_reporter_mode(scenario_config, definitions, reporter, false)
443    }
444
445    #[must_use]
446    #[allow(clippy::needless_pass_by_value)]
447    fn with_reporter_mode(
448        scenario_config: ScenarioConfig,
449        definitions: Vec<AssertDefinition>,
450        reporter: Arc<Reporter>,
451        emit_run_events: bool,
452    ) -> Self {
453        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
454            &scenario_config,
455        ));
456        let llm = LlmConfig {
457            url: scenario_config
458                .llm_url
459                .clone()
460                .unwrap_or_else(crate::llm_base_url),
461            model: scenario_config
462                .llm_model
463                .clone()
464                .unwrap_or_else(crate::llm_model),
465            api_key: scenario_config
466                .llm_api_key
467                .clone()
468                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
469            headers: if scenario_config.llm_headers.is_empty() {
470                crate::parse_headers_env()
471            } else {
472                scenario_config.llm_headers.clone()
473            },
474            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
475            temperature: scenario_config.temperature,
476            thinking: scenario_config.thinking,
477            model_params: scenario_config.model_params.clone(),
478            cache: scenario_config.cache.unwrap_or(true),
479            max_attempts: crate::default_llm_attempts(),
480            provider: crate::scenario::Provider::Openai,
481            deployment: None,
482            api_version: None,
483            auth: crate::scenario::AuthConfig::default(),
484            header_commands: std::collections::HashMap::new(),
485            aws: crate::scenario::AwsConfig::default(),
486        };
487        let endpoints = EndpointRegistry::from_config(
488            &scenario_config.endpoints,
489            Some(&llm),
490            crate::endpoints::EndpointDefaults {
491                cache: scenario_config.cache,
492                cache_pricing: scenario_config.cache_pricing,
493            },
494        );
495        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
496        let defs_map: HashMap<String, AssertDefinition> = definitions
497            .into_iter()
498            .map(|d| (d.name.clone(), d))
499            .collect();
500
501        Self {
502            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
503            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
504            viewport_height: scenario_config.viewport_height.unwrap_or(720),
505            applied_viewport: std::cell::Cell::new((0, 0)),
506            config: scenario_config.clone(),
507            definitions: defs_map,
508            llm,
509            endpoints,
510            usage: Arc::new(UsageTracker::new()),
511            budgets,
512            artifacts_dir: PathBuf::from(
513                scenario_config
514                    .artifacts_dir
515                    .unwrap_or_else(|| "artifacts".to_owned()),
516            ),
517            reporter,
518            current_step: std::cell::RefCell::new(None),
519            run_started: SystemTime::now()
520                .duration_since(UNIX_EPOCH)
521                .map_or(0, |d| d.as_secs()),
522            emit_run_events,
523        }
524    }
525
526    /// Emits an event; a sink failure degrades to a console warning so a
527    /// broken log file can never mask the run itself.
528    fn emit_event(&self, event: &TestEvent) {
529        if let Err(err) = self.reporter.emit(event) {
530            use std::io::Write as _;
531            let mut out = std::io::stderr().lock();
532            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
533        }
534    }
535
536    /// Returns a clone of the [`UsageTracker`] for reporting.
537    #[must_use]
538    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
539        Arc::clone(&self.usage)
540    }
541
542    /// Returns a reference to the [`BudgetTracker`].
543    #[must_use]
544    pub const fn budget_tracker(&self) -> &BudgetTracker {
545        &self.budgets
546    }
547
548    /// Executes all test groups in the scenario and returns a report.
549    ///
550    /// # Errors
551    ///
552    /// Returns an error if the browser fails to launch.
553    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
554    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
555        let mut report = RunReport::default();
556
557        if tests.is_empty() {
558            self.reporter.warn("No tests defined in scenario.");
559            if self.emit_run_events {
560                self.emit_event(&TestEvent::RunFinished {
561                    tests_passed: 0,
562                    tests_failed: 0,
563                    steps_passed: 0,
564                    steps_failed: 0,
565                    steps_skipped: 0,
566                    total_cost: 0.0,
567                    total_tokens: 0,
568                    total_input_tokens: 0,
569                    total_output_tokens: 0,
570                    total_cached_input_tokens: 0,
571                    total_cache_creation_input_tokens: 0,
572                    models: Vec::new(),
573                    total_calls: 0,
574                });
575            }
576            return Ok(report);
577        }
578
579        if self.emit_run_events {
580            self.emit_event(&TestEvent::RunStarted {
581                total_tests: tests.len() as u32,
582            });
583        }
584
585        let browser_headless = self.config.browser_headless.unwrap_or(true);
586
587        let launch_opts = LaunchOptions {
588            headless: browser_headless,
589            window_size: Some((self.viewport_width, self.viewport_height)),
590            sandbox: false,
591            // headless_chrome defaults this to 30s and shuts down the whole CDP
592            // connection when no messages arrive for that long. A scenario can
593            // easily exceed 30s of browser silence (slow LLM targeting/assertion
594            // calls, page waits, budget checks between steps), after which every
595            // remaining step fails with "Unable to make method calls because
596            // underlying connection is closed" — one quiet gap kills the run.
597            // Open-ended scenarios must own the connection for their full
598            // duration, so keep it alive for 6 hours.
599            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
600            ..LaunchOptions::default()
601        };
602
603        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
604        let tab = browser.new_tab().context("failed to open browser tab")?;
605        match (
606            self.config.browser_basic_auth_user.clone(),
607            self.config.browser_basic_auth_password.clone(),
608        ) {
609            (Some(username), Some(password)) if !username.is_empty() && !password.is_empty() => {
610                tab.authenticate(Some(username), Some(password))
611                    .context("failed to configure browser HTTP Basic Auth")?;
612                tab.enable_fetch(None, Some(true))
613                    .context("failed to enable browser HTTP authentication")?;
614            }
615            (None, None) => {}
616            _ => anyhow::bail!("browser Basic Auth requires both username and password"),
617        }
618        let _ = tab.set_default_timeout(self.timeout);
619
620        // Start MCP server if configured
621        #[cfg(feature = "mcp-server")]
622        if let Some(ref mcp_cfg) = self.config.mcp_server {
623            if mcp_cfg.enabled {
624                let port = mcp_cfg.port;
625                std::thread::spawn(move || {
626                    let _ = crate::mcp_server::start_mcp_server(port);
627                });
628            }
629        }
630        #[cfg(not(feature = "mcp-server"))]
631        if let Some(mcp_cfg) = &self.config.mcp_server {
632            if mcp_cfg.enabled {
633                self.reporter
634                    .warn("MCP server configured but 'mcp-server' feature not enabled");
635            }
636        }
637
638        // Start A2A agent server if configured
639        #[cfg(feature = "a2a-server")]
640        if let Some(ref a2a_cfg) = self.config.a2a_server {
641            if a2a_cfg.enabled {
642                let port = a2a_cfg.port;
643                tokio::spawn(crate::a2a_server::start_a2a_server(port));
644            }
645        }
646        #[cfg(not(feature = "a2a-server"))]
647        if let Some(a2a_cfg) = &self.config.a2a_server {
648            if a2a_cfg.enabled {
649                self.reporter
650                    .warn("A2A server configured but 'a2a-server' feature not enabled");
651            }
652        }
653
654        let retry_budget = self.config.retry_failed_tests.unwrap_or(0);
655
656        for test in tests {
657            // A shared tab on a contended runner makes some page loads
658            // stall (a JS chunk or a GraphQL call hangs mid-flight) and
659            // the test fails on a wait/assert that a fresh run passes.
660            // Re-run the WHOLE test when it failed and the retry budget
661            // allows; the fresh per-test isolation (login clear +
662            // auto-navigate) applies to the retry too. Both attempts'
663            // LLM spend stays in the budget accounting (each attempt
664            // commits its usage); only the final attempt is reported.
665            let details_len = report.details.len();
666            let passed_before = report.passed;
667            let failed_before = report.failed;
668            let skipped_before = report.skipped;
669            #[allow(unused_assignments)]
670            let mut final_result = None;
671            let mut attempt = 0;
672            loop {
673                attempt += 1;
674                self.emit_event(&TestEvent::TestStarted {
675                    test: test.name.clone(),
676                });
677
678                self.usage.reset_per_test();
679
680                let test_started = Instant::now();
681                let test_result = self.run_test(test, &tab);
682                let duration_ms = test_started.elapsed().as_millis() as u64;
683                let usage = self.usage.current_test_snapshot();
684                self.usage.commit_test(&test.name);
685
686                self.emit_event(&TestEvent::TestFinished {
687                    test: test.name.clone(),
688                    passed: test_result.passed,
689                    failed: test_result.failed,
690                    skipped: test_result.skipped,
691                    duration_ms,
692                    cost: usage.total_cost,
693                    tokens: usage.total_tokens,
694                    input_tokens: usage.total_input_tokens,
695                    output_tokens: usage.total_output_tokens,
696                    cached_input_tokens: usage.total_cached_input_tokens,
697                    cache_creation_input_tokens: usage.total_cache_creation_input_tokens,
698                    models: usage.models.clone(),
699                    calls: usage.total_calls,
700                });
701
702                final_result = Some(test_result);
703                let result = final_result.as_ref().expect("just assigned");
704                if result.failed == 0 || result.total == 0 || attempt > retry_budget {
705                    break;
706                }
707                self.reporter.warn(format!(
708                    "! retrying failed test '{}' (attempt {}/{}) — a fresh run passes on \
709                     transient page-load stalls",
710                    test.name,
711                    attempt + 1,
712                    retry_budget + 1
713                ));
714                // Roll this failed attempt out of the report so the
715                // retry's outcome replaces it.
716                report.details.truncate(details_len);
717                report.passed = passed_before;
718                report.failed = failed_before;
719                report.skipped = skipped_before;
720            }
721
722            let test_result = final_result.unwrap_or_default();
723            if test_result.failed == 0 && test_result.total > 0 {
724                report.tests_passed += 1;
725            } else if test_result.total > 0 {
726                report.tests_failed += 1;
727            }
728
729            report.passed += test_result.passed;
730            report.failed += test_result.failed;
731            report.skipped += test_result.skipped;
732            report.details.extend(test_result.details);
733        }
734
735        let global = self.usage.global_snapshot();
736        if self.emit_run_events {
737            self.emit_event(&TestEvent::RunFinished {
738                tests_passed: report.tests_passed,
739                tests_failed: report.tests_failed,
740                steps_passed: report.passed,
741                steps_failed: report.failed,
742                steps_skipped: report.skipped,
743                total_cost: global.total_cost,
744                total_tokens: global.total_tokens,
745                total_input_tokens: global.total_input_tokens,
746                total_output_tokens: global.total_output_tokens,
747                total_cached_input_tokens: global.total_cached_input_tokens,
748                total_cache_creation_input_tokens: global.total_cache_creation_input_tokens,
749                models: global.models.clone(),
750                total_calls: global.total_calls,
751            });
752        }
753
754        Ok(report)
755    }
756
757    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
758    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
759        let base_url = test
760            .base_url
761            .clone()
762            .or_else(|| self.config.base_url.clone())
763            .unwrap_or_else(crate::base_url);
764
765        // Per-test viewport override: switch the browser via CDP
766        // device-metrics emulation before this test runs.
767        let vw = test.viewport_width.unwrap_or(self.viewport_width);
768        let vh = test.viewport_height.unwrap_or(self.viewport_height);
769        if self.applied_viewport.get() != (vw, vh) {
770            self.apply_viewport(tab, vw, vh);
771            self.applied_viewport.set((vw, vh));
772        }
773
774        // Per-test isolation: every test starts from its own start_url
775        // (unless auto_navigate is disabled), so a test never inherits the
776        // previous test's page state.
777        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
778
779        let start_url = test
780            .start_url
781            .clone()
782            .or_else(|| self.config.start_url.clone())
783            .unwrap_or_else(|| "/dashboard".to_owned());
784
785        // Per-test state isolation for LOGIN tests: the tab is shared
786        // across the file's tests, and a previous test's login persists
787        // (Cognito tokens in localStorage + the hosted-UI cookies). The
788        // SPA's login pages then auto-continue authenticated visitors on
789        // boot, so a later login test never sees the form and times out
790        // waiting for `#email` (observed on immosai runs #1641/#1642: a
791        // shard's first three logins pass, the fourth onward time out).
792        // Clear cookies + origin storage ONLY for tests that perform
793        // their own login (their first navigate step targets a login
794        // route, or they have no navigate step and ride the auto-nav
795        // onto a login start_url). The many "page loads" tests that
796        // navigate app pages directly RELY on the shared session — a
797        // blanket clear bounces them off their target (run #1643: every
798        // shard failed with "the current page is the login page, not
799        // the mailboxes page"). The HTTP cache is deliberately NOT
800        // cleared — re-downloading the SPA bundle per test would only
801        // add boot time on a slow runner egress.
802        if auto_navigate && test_targets_login(&start_url, &test.steps) {
803            use headless_chrome::protocol::cdp::{Network, Storage};
804            if let Err(e) = tab.call_method(Network::ClearBrowserCookies(None)) {
805                self.reporter
806                    .warn(format!("per-test cookie clear failed: {e}"));
807            }
808            if let Some(origin) = origin_of(&base_url) {
809                if let Err(e) = tab.call_method(Storage::ClearDataForOrigin {
810                    origin,
811                    storage_Types: "all".to_string(),
812                }) {
813                    self.reporter
814                        .warn(format!("per-test storage clear failed: {e}"));
815                }
816            }
817        }
818        if auto_navigate {
819            let full_url = resolve_url(&start_url, &base_url);
820            self.reporter.debug(format!("auto-navigate: {full_url}"));
821            let _ = tab.navigate_to(&full_url);
822            let _ = tab.wait_until_navigated();
823            std::thread::sleep(Duration::from_secs(4));
824        }
825
826        let mut result = TestRunResult::default();
827
828        for (step_index, step) in test.steps.iter().enumerate() {
829            result.total += 1;
830
831            let wait_ms = match step {
832                TestStep::Navigate { wait_after_ms, .. }
833                | TestStep::Click { wait_after_ms, .. }
834                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
835                _ => None,
836            };
837
838            self.current_step
839                .replace(Some((test.name.clone(), step_index as u32)));
840            self.emit_event(&TestEvent::StepStarted {
841                test: test.name.clone(),
842                index: step_index as u32,
843                label: step_label(step),
844            });
845            let step_started = Instant::now();
846
847            let mut step_result = match step {
848                TestStep::Navigate { url, .. } => {
849                    let full_url = resolve_url(url, &base_url);
850                    run_navigate_step(&full_url, tab)
851                }
852                TestStep::Click {
853                    target,
854                    selector,
855                    endpoint,
856                    idempotent,
857                    ..
858                } => self.run_click(
859                    target,
860                    selector.as_deref(),
861                    endpoint.as_deref(),
862                    test.endpoint.as_deref(),
863                    *idempotent,
864                    tab,
865                ),
866                TestStep::Type {
867                    target,
868                    text,
869                    selector,
870                    endpoint,
871                    idempotent,
872                    ..
873                } => self.run_type(
874                    target,
875                    text,
876                    selector.as_deref(),
877                    endpoint.as_deref(),
878                    test.endpoint.as_deref(),
879                    *idempotent,
880                    tab,
881                ),
882                TestStep::Wait {
883                    target,
884                    selector,
885                    text,
886                    timeout_ms,
887                    endpoint,
888                    idempotent,
889                } => self.run_wait(
890                    target,
891                    selector.as_deref(),
892                    text.as_deref(),
893                    *timeout_ms,
894                    endpoint.as_deref(),
895                    test.endpoint.as_deref(),
896                    *idempotent,
897                    tab,
898                ),
899                TestStep::Assert {
900                    definition,
901                    preset,
902                    prompt,
903                    assert_text,
904                    endpoint,
905                    screenshot,
906                } => self.run_assert(
907                    definition.as_deref(),
908                    preset.as_deref(),
909                    prompt.as_deref(),
910                    assert_text.as_deref(),
911                    *screenshot,
912                    endpoint.as_deref(),
913                    test.endpoint.as_deref(),
914                    tab,
915                ),
916                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
917                TestStep::Agent {
918                    agent,
919                    task,
920                    definition,
921                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
922                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
923            };
924
925            // Failure diagnostics: capture the page state and a screenshot so
926            // CI logs say WHAT the page looked like when the step failed,
927            // instead of a bare "timed out: The event waited for never came".
928            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
929                let state = diagnostics::capture(tab);
930                let screenshot = diagnostics::save_screenshot(
931                    tab,
932                    &self.artifacts_dir,
933                    &test.name,
934                    &test.name,
935                    step_index,
936                    step_kind_label(step),
937                );
938                step_result.message = format!(
939                    "{base} — {excerpt}",
940                    base = step_result.message,
941                    excerpt = diagnostics::inline_excerpt(&state),
942                );
943                (Some(diagnostics::full_context(&state)), screenshot)
944            } else {
945                (None, None)
946            };
947
948            let duration_ms = step_started.elapsed().as_millis() as u64;
949            self.emit_event(&TestEvent::StepFinished {
950                test: test.name.clone(),
951                index: step_index as u32,
952                label: step_result.name.clone(),
953                status: step_result.status,
954                duration_ms,
955                message: step_result.message.clone(),
956                diagnostics: diagnostics_block,
957                screenshot: screenshot_path,
958            });
959            self.current_step.replace(None);
960
961            match step_result.status {
962                StepStatus::Passed => result.passed += 1,
963                StepStatus::Failed => result.failed += 1,
964                StepStatus::Skipped => result.skipped += 1,
965            }
966
967            // Fail fast: the first failed step ends the test and the
968            // remaining steps are reported as skipped (no LLM budget is
969            // burned asserting against a page that is already known broken).
970            if step_result.status == StepStatus::Failed
971                && !self.config.continue_on_failure
972                && step_index + 1 < test.steps.len()
973            {
974                self.emit_event(&TestEvent::Warning {
975                    message: format!(
976                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
977                        test.steps.len() - step_index - 1
978                    ),
979                });
980                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
981                    let skipped_index = step_index + 1 + offset;
982                    let label = step_label(skipped);
983                    result.total += 1;
984                    result.skipped += 1;
985                    self.emit_event(&TestEvent::StepStarted {
986                        test: test.name.clone(),
987                        index: skipped_index as u32,
988                        label: label.clone(),
989                    });
990                    self.emit_event(&TestEvent::StepFinished {
991                        test: test.name.clone(),
992                        index: skipped_index as u32,
993                        label,
994                        status: StepStatus::Skipped,
995                        duration_ms: 0,
996                        message: "skipped: previous step failed".into(),
997                        diagnostics: None,
998                        screenshot: None,
999                    });
1000                    result.details.push(StepResult {
1001                        name: step_label(skipped),
1002                        status: StepStatus::Skipped,
1003                        message: "skipped: previous step failed".into(),
1004                    });
1005                }
1006                result.details.push(step_result);
1007                return result;
1008            }
1009
1010            // Check per-test budget after each step
1011            let test_usage = self.usage.current_test_snapshot();
1012            let global_usage = self.usage.global_snapshot();
1013            let budget_status = self.budgets.check_all(
1014                &test.name,
1015                &test_usage,
1016                &global_usage,
1017                test.budget.as_ref(),
1018            );
1019            match budget_status {
1020                BudgetStatus::HardExceeded { message, .. } => {
1021                    self.emit_event(&TestEvent::Warning {
1022                        message: format!("budget exceeded: {message}"),
1023                    });
1024                    result.details.push(StepResult {
1025                        name: "[budget]".into(),
1026                        status: StepStatus::Failed,
1027                        message,
1028                    });
1029                    result.failed += 1;
1030                    return result;
1031                }
1032                BudgetStatus::SoftExceeded { message, .. } => {
1033                    self.emit_event(&TestEvent::Warning {
1034                        message: format!("budget warning: {message}"),
1035                    });
1036                }
1037                BudgetStatus::Ok => {}
1038            }
1039
1040            if let Some(ms) = wait_ms {
1041                std::thread::sleep(Duration::from_millis(ms));
1042            }
1043
1044            result.details.push(step_result);
1045        }
1046
1047        result
1048    }
1049
1050    /// Applies a viewport size to the current tab via CDP
1051    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
1052    /// overrides and the viewport matrix. The initial window size set at
1053    /// browser launch is replaced by emulation; failures are logged but
1054    /// do not fail the test (a mismatched viewport only weakens coverage).
1055    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
1056        use headless_chrome::protocol::cdp::Emulation;
1057        let _ = self;
1058        let params = Emulation::SetDeviceMetricsOverride {
1059            width,
1060            height,
1061            device_scale_factor: 1.0,
1062            mobile: false,
1063            scale: None,
1064            screen_width: Some(width),
1065            screen_height: Some(height),
1066            position_x: None,
1067            position_y: None,
1068            dont_set_visible_size: None,
1069            screen_orientation: None,
1070            viewport: None,
1071            display_feature: None,
1072            device_posture: None,
1073        };
1074        self.reporter.debug(format!("viewport: {width}x{height}"));
1075        if let Err(e) = tab.call_method(params) {
1076            self.reporter
1077                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
1078        }
1079    }
1080
1081    /// Height (px) covered by assert-step screenshots: the configured
1082    /// `screenshot_max_height` (absolute px or viewport multiple, default
1083    /// `"20x"`) resolved against the currently applied viewport, raised to
1084    /// at least the viewport height so the visible screen is always fully
1085    /// included. The capture is split into viewport-tall tiles, so this
1086    /// value bounds total coverage (and hence the number of image parts).
1087    #[must_use]
1088    fn screenshot_height_cap(&self) -> u32 {
1089        let viewport_height = self.current_viewport_height();
1090        let cap = self
1091            .config
1092            .screenshot_max_height
1093            .as_ref()
1094            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
1095        cap.max(viewport_height)
1096    }
1097
1098    /// Height of the viewport currently emulated in the browser (falling
1099    /// back to the configured default before any emulation was applied).
1100    #[must_use]
1101    const fn current_viewport_height(&self) -> u32 {
1102        let (_, height) = self.applied_viewport.get();
1103        if height > 0 {
1104            height
1105        } else {
1106            self.viewport_height
1107        }
1108    }
1109
1110    // ── step handlers ───────────────────────────────────────────────────
1111
1112    #[allow(clippy::too_many_lines)]
1113    fn run_click(
1114        &self,
1115        target: &str,
1116        selector_override: Option<&str>,
1117        step_endpoint: Option<&str>,
1118        test_endpoint: Option<&str>,
1119        idempotent: bool,
1120        tab: &Tab,
1121    ) -> StepResult {
1122        let name = format!("[click] {target}");
1123        let selector = match self.resolve_selector(
1124            selector_override,
1125            target,
1126            step_endpoint,
1127            test_endpoint,
1128            tab,
1129        ) {
1130            Ok(s) => s,
1131            Err(msg) => {
1132                if idempotent {
1133                    return StepResult {
1134                        name,
1135                        status: StepStatus::Skipped,
1136                        message: format!("skipped (idempotent): no target found — {msg}"),
1137                    };
1138                }
1139                return StepResult {
1140                    name,
1141                    status: StepStatus::Failed,
1142                    message: msg,
1143                };
1144            }
1145        };
1146
1147        // Idempotent steps probe briefly: a missing target means the
1148        // action was already done / not applicable (e.g. an
1149        // already-authenticated session), and skipping is the success
1150        // path, not a failure.
1151        let probe_secs = if idempotent { 5 } else { 10 };
1152        // Blank-page self-heal (same as the wait steps): a runner
1153        // egress/resource burst can leave the SPA unbooted when a
1154        // scenario's FIRST interactive step is a click (run #1652
1155        // core-a: contacts.toml opens with navigate → click, the page
1156        // was blank 3/3 attempts, and the click never had the wait
1157        // step's heal). Reload and re-probe with backoff before
1158        // failing.
1159        let mut heal_note = String::new();
1160        let mut heals = 0u32;
1161        let probe = loop {
1162            match tab
1163                .wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs))
1164            {
1165                Ok(element) => break Ok(element),
1166                Err(e) => {
1167                    if !idempotent && heals < MAX_BLANK_HEALS && page_is_blank(tab) {
1168                        heals += 1;
1169                        if heals > 1 {
1170                            std::thread::sleep(Duration::from_secs(BLANK_HEAL_BACKOFF_SECS));
1171                        }
1172                        heal_note = format!(" (blank page — reloaded {heals}x and re-clicked)");
1173                        let _ = tab.reload(true, None);
1174                        let _ = tab.wait_until_navigated();
1175                        std::thread::sleep(Duration::from_secs(2));
1176                        continue;
1177                    }
1178                    break Err(e);
1179                }
1180            }
1181        };
1182        match probe {
1183            Ok(element) => match element.click() {
1184                Ok(_) => StepResult {
1185                    name,
1186                    status: StepStatus::Passed,
1187                    message: format!("clicked {selector}{heal_note}"),
1188                },
1189                Err(e) => StepResult {
1190                    name,
1191                    status: StepStatus::Failed,
1192                    message: format!("click failed on {selector}: {e}"),
1193                },
1194            },
1195            Err(e) if idempotent => StepResult {
1196                name,
1197                status: StepStatus::Skipped,
1198                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1199            },
1200            Err(e) => StepResult {
1201                name,
1202                status: StepStatus::Failed,
1203                message: format!("element {selector} not found: {e}{heal_note}"),
1204            },
1205        }
1206    }
1207
1208    #[allow(clippy::too_many_arguments)]
1209    fn run_type(
1210        &self,
1211        target: &str,
1212        text: &str,
1213        selector_override: Option<&str>,
1214        step_endpoint: Option<&str>,
1215        test_endpoint: Option<&str>,
1216        idempotent: bool,
1217        tab: &Tab,
1218    ) -> StepResult {
1219        let name = format!("[type] {target}");
1220        let selector = match self.resolve_selector(
1221            selector_override,
1222            target,
1223            step_endpoint,
1224            test_endpoint,
1225            tab,
1226        ) {
1227            Ok(s) => s,
1228            Err(msg) => {
1229                if idempotent {
1230                    return StepResult {
1231                        name,
1232                        status: StepStatus::Skipped,
1233                        message: format!("skipped (idempotent): no target found — {msg}"),
1234                    };
1235                }
1236                return StepResult {
1237                    name,
1238                    status: StepStatus::Failed,
1239                    message: msg,
1240                };
1241            }
1242        };
1243
1244        let probe_secs = if idempotent { 5 } else { 10 };
1245        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1246            Ok(element) => {
1247                if let Err(e) = element.click() {
1248                    return StepResult {
1249                        name,
1250                        status: StepStatus::Failed,
1251                        message: format!("click to focus {selector} failed: {e}"),
1252                    };
1253                }
1254
1255                let js = format!(
1256                    "document.querySelector('{}').value = '';",
1257                    selector.replace('\'', "\\'")
1258                );
1259                let _ = tab.evaluate(&js, false);
1260
1261                match element.type_into(text) {
1262                    Ok(_) => StepResult {
1263                        name,
1264                        status: StepStatus::Passed,
1265                        message: format!("typed {text:?} into {selector}"),
1266                    },
1267                    Err(e) => StepResult {
1268                        name,
1269                        status: StepStatus::Failed,
1270                        message: format!("type into {selector} failed: {e}"),
1271                    },
1272                }
1273            }
1274            Err(e) if idempotent => StepResult {
1275                name,
1276                status: StepStatus::Skipped,
1277                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1278            },
1279            Err(e) => StepResult {
1280                name,
1281                status: StepStatus::Failed,
1282                message: format!("element {selector} not found: {e}"),
1283            },
1284        }
1285    }
1286
1287    #[allow(clippy::too_many_arguments)]
1288    #[allow(clippy::too_many_lines)]
1289    fn run_wait(
1290        &self,
1291        target: &str,
1292        selector_override: Option<&str>,
1293        text: Option<&str>,
1294        timeout_ms: Option<u64>,
1295        step_endpoint: Option<&str>,
1296        test_endpoint: Option<&str>,
1297        idempotent: bool,
1298        tab: &Tab,
1299    ) -> StepResult {
1300        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1301        let step_name = format!("[wait] {target}");
1302
1303        // Resolve an explicit selector only (text-only waits are LLM-free).
1304        let selector = match selector_override {
1305            Some(s) => Some(s.to_owned()),
1306            None if text.is_some() => None,
1307            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1308                Ok(s) => Some(s),
1309                Err(msg) => {
1310                    if idempotent {
1311                        return StepResult {
1312                            name: step_name,
1313                            status: StepStatus::Skipped,
1314                            message: format!("skipped (idempotent): no target found — {msg}"),
1315                        };
1316                    }
1317                    return StepResult {
1318                        name: step_name,
1319                        status: StepStatus::Failed,
1320                        message: msg,
1321                    };
1322                }
1323            },
1324        };
1325
1326        if text.is_some() {
1327            let sel_js = selector
1328                .as_deref()
1329                .map(crate::selectors::selector_matches_js);
1330            let text_js = text.map(|t| {
1331                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1332                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1333            });
1334
1335            let mut reload_note = String::new();
1336            let mut reloaded = 0u32;
1337            loop {
1338                let deadline = Instant::now() + timeout;
1339                loop {
1340                    let sel_ok = sel_js
1341                        .as_ref()
1342                        .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1343                    let text_ok = text_js
1344                        .as_ref()
1345                        .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1346                    if sel_ok && text_ok {
1347                        let mut what = Vec::new();
1348                        if let Some(sel) = &selector {
1349                            what.push(format!("found {sel}"));
1350                        }
1351                        if let Some(t) = text {
1352                            what.push(format!("text {t:?} visible"));
1353                        }
1354                        return StepResult {
1355                            name: step_name,
1356                            status: StepStatus::Passed,
1357                            message: format!("{}{reload_note}", what.join(" and ")),
1358                        };
1359                    }
1360                    if Instant::now() >= deadline {
1361                        break;
1362                    }
1363                    std::thread::sleep(Duration::from_millis(250));
1364                }
1365                // Blank-page self-heal: a stalled runner egress leaves the
1366                // SPA unbooted (index.html loaded, JS chunks never arrived,
1367                // immosai runs #1646/#1647 — every failure screenshot is an
1368                // empty viewport). A wait against that page can never
1369                // succeed; reload and re-wait with the full budget. The
1370                // runner's egress/resource bursts last minutes (v0.19.5's
1371                // single reload was not enough when the burst outlived two
1372                // wait budgets — immosai run #1651), so allow up to
1373                // MAX_BLANK_HEALS heal cycles with a backoff sleep between
1374                // them.
1375                if reloaded < MAX_BLANK_HEALS && page_is_blank(tab) {
1376                    reloaded += 1;
1377                    if reloaded > 1 {
1378                        std::thread::sleep(Duration::from_secs(BLANK_HEAL_BACKOFF_SECS));
1379                    }
1380                    reload_note = format!(" (blank page — reloaded {reloaded}x and re-waited)");
1381                    let _ = tab.reload(true, None);
1382                    let _ = tab.wait_until_navigated();
1383                    std::thread::sleep(Duration::from_secs(2));
1384                    continue;
1385                }
1386                let mut what = Vec::new();
1387                if let Some(sel) = &selector {
1388                    what.push(sel.clone());
1389                }
1390                if let Some(t) = text {
1391                    what.push(format!("text {t:?}"));
1392                }
1393                let message = format!(
1394                    "wait for {} timed out after {}ms: the event waited for never came{reload_note}",
1395                    what.join(" / "),
1396                    timeout.as_millis(),
1397                );
1398                if idempotent {
1399                    return StepResult {
1400                        name: step_name,
1401                        status: StepStatus::Skipped,
1402                        message: format!("skipped (idempotent): {message}"),
1403                    };
1404                }
1405                return StepResult {
1406                    name: step_name,
1407                    status: StepStatus::Failed,
1408                    message,
1409                };
1410            }
1411        }
1412
1413        let mut reload_note = String::new();
1414        let mut reloaded = 0u32;
1415        let result = loop {
1416            match selector.as_deref() {
1417                Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1418                    Ok(_) => break Ok((sel.to_owned(), reload_note)),
1419                    Err(e) => {
1420                        // Blank-page self-heal (see the text-wait branch):
1421                        // an unbooted SPA never renders the target; reload
1422                        // and re-wait with the full budget, with a backoff
1423                        // between heal cycles for the runner's multi-minute
1424                        // egress/resource bursts (immosai run #1651).
1425                        if reloaded < MAX_BLANK_HEALS && page_is_blank(tab) {
1426                            reloaded += 1;
1427                            if reloaded > 1 {
1428                                std::thread::sleep(Duration::from_secs(BLANK_HEAL_BACKOFF_SECS));
1429                            }
1430                            format!(" (blank page — reloaded {reloaded}x and re-waited)")
1431                                .clone_into(&mut reload_note);
1432                            let _ = tab.reload(true, None);
1433                            let _ = tab.wait_until_navigated();
1434                            std::thread::sleep(Duration::from_secs(2));
1435                            continue;
1436                        }
1437                        if idempotent {
1438                            break Err((
1439                                format!(
1440                                    "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1441                                    timeout.as_millis()
1442                                ),
1443                                StepStatus::Skipped,
1444                            ));
1445                        }
1446                        break Err((
1447                            format!(
1448                                "wait for {sel} timed out after {}ms: {e}{reload_note}",
1449                                timeout.as_millis()
1450                            ),
1451                            StepStatus::Failed,
1452                        ));
1453                    }
1454                },
1455                None => {
1456                    break Err((
1457                        "wait step has neither selector nor text".to_owned(),
1458                        StepStatus::Failed,
1459                    ))
1460                }
1461            }
1462        };
1463        match result {
1464            Ok((sel, note)) => StepResult {
1465                name: step_name,
1466                status: StepStatus::Passed,
1467                message: format!("found {sel}{note}"),
1468            },
1469            Err((message, status)) => StepResult {
1470                name: step_name,
1471                status,
1472                message,
1473            },
1474        }
1475    }
1476
1477    #[allow(clippy::too_many_arguments)]
1478    fn run_assert(
1479        &self,
1480        definition: Option<&str>,
1481        preset: Option<&str>,
1482        prompt: Option<&str>,
1483        assert_text: Option<&str>,
1484        screenshot: bool,
1485        step_endpoint: Option<&str>,
1486        test_endpoint: Option<&str>,
1487        tab: &Tab,
1488    ) -> StepResult {
1489        std::thread::sleep(Duration::from_millis(500));
1490
1491        let page_content = get_page_text(tab);
1492
1493        // Vision attach: capture the full page once per assert step and
1494        // split it into viewport-tall tiles (the total coverage is bounded
1495        // by the configured height cap so vision tokens stay sane). All
1496        // tile data URLs are handed to the preset/prompt evaluation below.
1497        let image: Option<Vec<String>> = if screenshot {
1498            let endpoint = self
1499                .endpoints
1500                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1501            if !endpoint.vision {
1502                return StepResult {
1503                    name: "[assert]".into(),
1504                    status: StepStatus::Failed,
1505                    message: format!(
1506                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1507                        name = endpoint.name
1508                    ),
1509                };
1510            }
1511            match crate::vision::capture_screenshot_data_urls(
1512                tab,
1513                self.config
1514                    .screenshot_max_dimension
1515                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1516                self.screenshot_height_cap(),
1517                self.current_viewport_height(),
1518            ) {
1519                Ok(urls) => Some(urls),
1520                Err(e) => {
1521                    return StepResult {
1522                        name: "[assert]".into(),
1523                        status: StepStatus::Failed,
1524                        message: format!("screenshot capture failed: {e}"),
1525                    };
1526                }
1527            }
1528        } else {
1529            None
1530        };
1531
1532        if let Some(def_name) = definition {
1533            if let Some(def) = self.definitions.get(def_name) {
1534                return self.run_assert_def(
1535                    def,
1536                    &page_content,
1537                    image.as_deref(),
1538                    step_endpoint,
1539                    test_endpoint,
1540                    tab,
1541                );
1542            }
1543            return StepResult {
1544                name: format!("[assert] {def_name}"),
1545                status: StepStatus::Failed,
1546                message: format!("definition '{def_name}' not found"),
1547            };
1548        }
1549
1550        if let Some(preset_name) = preset {
1551            // Deterministic DOM layout scan — runs JS in the browser and
1552            // never calls the LLM (free, fast, no pixel budget).
1553            if preset_name == "layout_no_issues" {
1554                return self.run_layout_preset(tab);
1555            }
1556            return self.run_preset(
1557                preset_name,
1558                assert_text,
1559                &page_content,
1560                image.as_deref(),
1561                step_endpoint,
1562                test_endpoint,
1563            );
1564        }
1565
1566        if let Some(prompt_text) = prompt {
1567            return self.run_custom(
1568                prompt_text,
1569                &page_content,
1570                image.as_deref(),
1571                step_endpoint,
1572                test_endpoint,
1573            );
1574        }
1575
1576        StepResult {
1577            name: "[assert]".into(),
1578            status: StepStatus::Skipped,
1579            message: "no definition, preset, or prompt specified".into(),
1580        }
1581    }
1582
1583    fn run_assert_def(
1584        &self,
1585        def: &AssertDefinition,
1586        page_content: &PageContent,
1587        image: Option<&[String]>,
1588        step_endpoint: Option<&str>,
1589        test_endpoint: Option<&str>,
1590        tab: &Tab,
1591    ) -> StepResult {
1592        // Agent-based definition: delegate to an A2A agent
1593        if let Some(ref agent) = def.agent {
1594            if image.is_some() {
1595                return StepResult {
1596                    name: format!("[assert] {}", def.name),
1597                    status: StepStatus::Failed,
1598                    message: "agent-backed assertions do not support screenshots".into(),
1599                };
1600            }
1601            let task = def
1602                .task_template
1603                .as_deref()
1604                .unwrap_or("Evaluate the assertion")
1605                .replace("{url}", &page_content.url)
1606                .replace("{title}", &page_content.title)
1607                .replace("{content}", &page_content.body_text)
1608                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1609
1610            return self.run_agent_step(agent, &task, &def.name);
1611        }
1612
1613        // Custom preset: system + user_template provided in the definition
1614        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1615            return self.run_custom_preset(
1616                &def.name,
1617                system,
1618                template,
1619                def.assert_text.as_deref(),
1620                page_content,
1621                image,
1622                step_endpoint,
1623                test_endpoint,
1624            );
1625        }
1626
1627        def.preset.as_ref().map_or_else(
1628            || {
1629                def.prompt.as_ref().map_or_else(
1630                    || StepResult {
1631                        name: format!("[assert] {}", def.name),
1632                        status: StepStatus::Failed,
1633                        message: "definition has no preset, prompt, or system+user_template".into(),
1634                    },
1635                    |prompt| {
1636                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1637                    },
1638                )
1639            },
1640            |preset_name| {
1641                if preset_name == "layout_no_issues" {
1642                    return self.run_layout_preset(tab);
1643                }
1644                self.run_preset(
1645                    preset_name,
1646                    def.assert_text.as_deref(),
1647                    page_content,
1648                    image,
1649                    step_endpoint,
1650                    test_endpoint,
1651                )
1652            },
1653        )
1654    }
1655
1656    #[allow(clippy::too_many_arguments)]
1657    fn run_custom_preset(
1658        &self,
1659        name: &str,
1660        system: &str,
1661        template: &str,
1662        assert_text: Option<&str>,
1663        page_content: &PageContent,
1664        image: Option<&[String]>,
1665        step_endpoint: Option<&str>,
1666        test_endpoint: Option<&str>,
1667    ) -> StepResult {
1668        let user_prompt = template
1669            .replace("{url}", &page_content.url)
1670            .replace("{title}", &page_content.title)
1671            .replace("{content}", &page_content.body_text)
1672            .replace("{expected_text}", assert_text.unwrap_or(""))
1673            .replace("{description}", "");
1674
1675        // Custom preset definitions frequently forget the {content}
1676        // placeholder — without it the LLM has no page to evaluate and
1677        // answers "I can't determine that without seeing the page". Always
1678        // append the page context unless the template already references it.
1679        let user_prompt = if template.contains("{content}") {
1680            user_prompt
1681        } else {
1682            format!(
1683                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1684                url = page_content.url,
1685                title = page_content.title,
1686                content = page_content.body_text,
1687            )
1688        };
1689
1690        self.reporter
1691            .debug(format!("assert: {name} (custom preset)"));
1692
1693        let chain = self
1694            .endpoints
1695            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1696        let sys = system.to_owned();
1697
1698        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1699
1700        response.map_or_else(
1701            |e| StepResult {
1702                name: format!("[assert] {name}"),
1703                status: StepStatus::Failed,
1704                message: format!("LLM assertion call failed: {e}"),
1705            },
1706            |(lr, _idx)| {
1707                if verdict_is_pass(&lr.content) {
1708                    StepResult {
1709                        name: format!("[assert] {name}"),
1710                        status: StepStatus::Passed,
1711                        message: "PASS".into(),
1712                    }
1713                } else {
1714                    StepResult {
1715                        name: format!("[assert] {name}"),
1716                        status: StepStatus::Failed,
1717                        message: lr.content,
1718                    }
1719                }
1720            },
1721        )
1722    }
1723
1724    fn run_preset(
1725        &self,
1726        preset_name: &str,
1727        assert_text: Option<&str>,
1728        page_content: &PageContent,
1729        image: Option<&[String]>,
1730        step_endpoint: Option<&str>,
1731        test_endpoint: Option<&str>,
1732    ) -> StepResult {
1733        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1734            return StepResult {
1735                name: format!("[assert] {preset_name}"),
1736                status: StepStatus::Failed,
1737                message: format!("unknown assertion preset: {preset_name}"),
1738            };
1739        };
1740        if preset_name.starts_with("visual_") && image.is_none() {
1741            return StepResult {
1742                name: format!("[assert] {preset_name}"),
1743                status: StepStatus::Failed,
1744                message: format!(
1745                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1746                ),
1747            };
1748        }
1749
1750        let user_prompt = preset
1751            .user_template
1752            .replace("{url}", &page_content.url)
1753            .replace("{title}", &page_content.title)
1754            .replace("{content}", &page_content.body_text)
1755            .replace("{expected_text}", assert_text.unwrap_or(""))
1756            .replace("{description}", "");
1757
1758        // Same safety net as custom presets: never let the LLM answer with
1759        // no page context at all.
1760        let user_prompt = if preset.user_template.contains("{content}") {
1761            user_prompt
1762        } else {
1763            format!(
1764                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1765                url = page_content.url,
1766                title = page_content.title,
1767                content = page_content.body_text,
1768            )
1769        };
1770
1771        self.reporter.debug(format!("assert: {preset_name}"));
1772
1773        let chain = self
1774            .endpoints
1775            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1776        let sys = preset.system.to_owned();
1777
1778        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1779
1780        response.map_or_else(
1781            |e| StepResult {
1782                name: format!("[assert] {preset_name}"),
1783                status: StepStatus::Failed,
1784                message: format!("LLM assertion call failed: {e}"),
1785            },
1786            |(lr, _idx)| {
1787                if verdict_is_pass(&lr.content) {
1788                    StepResult {
1789                        name: format!("[assert] {preset_name}"),
1790                        status: StepStatus::Passed,
1791                        message: "PASS".into(),
1792                    }
1793                } else {
1794                    StepResult {
1795                        name: format!("[assert] {preset_name}"),
1796                        status: StepStatus::Failed,
1797                        message: lr.content,
1798                    }
1799                }
1800            },
1801        )
1802    }
1803
1804    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1805    ///
1806    /// Evaluates the layout-scan JS in the page and fails with the list of
1807    /// detected issues: horizontal page overflow, visible elements sticking
1808    /// out of the viewport, text clipped by `overflow: hidden` containers,
1809    /// and interactive elements covered by other elements. No LLM call —
1810    /// checks are geometry-based so the check is free, deterministic, and
1811    /// safe to run on every page × viewport variant.
1812    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1813        let name = "[assert] layout_no_issues".to_owned();
1814        self.reporter
1815            .debug("assert: layout_no_issues (DOM layout scan)");
1816        let js = LAYOUT_SCAN_JS.replace(
1817            "__IGNORE_CLASSES__",
1818            &serde_json::to_string(&self.config.layout_ignore_classes)
1819                .unwrap_or_else(|_| "[]".to_owned()),
1820        );
1821        let result = tab.evaluate(&js, false);
1822        let json_str = match result {
1823            Ok(r) => r
1824                .value
1825                .as_ref()
1826                .and_then(|v| v.as_str().map(String::from))
1827                .unwrap_or_else(|| "[]".to_owned()),
1828            Err(e) => {
1829                return StepResult {
1830                    name,
1831                    status: StepStatus::Failed,
1832                    message: format!("layout scan JS failed: {e}"),
1833                };
1834            }
1835        };
1836        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1837        if issues.is_empty() {
1838            return StepResult {
1839                name,
1840                status: StepStatus::Passed,
1841                message: "PASS — no layout defects detected".into(),
1842            };
1843        }
1844        let mut lines: Vec<String> = issues
1845            .iter()
1846            .take(10)
1847            .map(|i| {
1848                format!(
1849                    "- [{type_}] {element}: {detail}",
1850                    type_ = i.issue_type,
1851                    element = i.element,
1852                    detail = i.detail
1853                )
1854            })
1855            .collect();
1856        if issues.len() > 10 {
1857            lines.push(format!("- … and {} more", issues.len() - 10));
1858        }
1859        StepResult {
1860            name,
1861            status: StepStatus::Failed,
1862            message: format!(
1863                "FAIL — {} layout defect(s) detected:\n{}",
1864                issues.len(),
1865                lines.join("\n")
1866            ),
1867        }
1868    }
1869
1870    fn run_custom(
1871        &self,
1872        prompt: &str,
1873        page_content: &PageContent,
1874        image: Option<&[String]>,
1875        step_endpoint: Option<&str>,
1876        test_endpoint: Option<&str>,
1877    ) -> StepResult {
1878        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1879
1880        let mut user = format!(
1881            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1882            url = page_content.url,
1883            title = page_content.title,
1884            content = page_content.body_text,
1885        );
1886        if image.is_some() {
1887            user.push_str(
1888                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1889            );
1890        }
1891
1892        self.reporter.debug("custom assert");
1893
1894        let chain = self
1895            .endpoints
1896            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1897        let sys = system.to_owned();
1898
1899        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1900
1901        response.map_or_else(
1902            |e| StepResult {
1903                name: "[assert] custom".into(),
1904                status: StepStatus::Failed,
1905                message: format!("LLM assertion call failed: {e}"),
1906            },
1907            |(lr, _idx)| {
1908                if verdict_is_pass(&lr.content) {
1909                    StepResult {
1910                        name: "[assert] custom".into(),
1911                        status: StepStatus::Passed,
1912                        message: "PASS".into(),
1913                    }
1914                } else {
1915                    StepResult {
1916                        name: "[assert] custom".into(),
1917                        status: StepStatus::Failed,
1918                        message: lr.content,
1919                    }
1920                }
1921            },
1922        )
1923    }
1924
1925    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1926        let path = path.unwrap_or("screenshot.png");
1927
1928        match tab.capture_screenshot(
1929            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1930            None,
1931            None,
1932            true,
1933        ) {
1934            Ok(data) => {
1935                if let Err(e) = std::fs::write(path, &data) {
1936                    return StepResult {
1937                        name: format!("[screenshot] {path}"),
1938                        status: StepStatus::Failed,
1939                        message: format!("failed to write screenshot: {e}"),
1940                    };
1941                }
1942                StepResult {
1943                    name: format!("[screenshot] {path}"),
1944                    status: StepStatus::Passed,
1945                    message: format!("saved to {path}"),
1946                }
1947            }
1948            Err(e) => StepResult {
1949                name: format!("[screenshot] {path}"),
1950                status: StepStatus::Failed,
1951                message: format!("screenshot failed: {e}"),
1952            },
1953        }
1954    }
1955
1956    /// Runs an A2A agent step.
1957    #[allow(clippy::literal_string_with_formatting_args)]
1958    fn run_agent(
1959        &self,
1960        agent_name: &str,
1961        task: &str,
1962        definition: Option<&str>,
1963        _test_endpoint: Option<&str>,
1964    ) -> StepResult {
1965        // If a definition is specified, look up the task template
1966        let resolved_task = if let Some(def_name) = definition {
1967            if let Some(def) = self.definitions.get(def_name) {
1968                let tmpl = def.task_template.as_deref().unwrap_or(task);
1969                tmpl.replace("{task}", task)
1970            } else {
1971                return StepResult {
1972                    name: format!("[agent] {def_name}"),
1973                    status: StepStatus::Failed,
1974                    message: format!("definition '{def_name}' not found"),
1975                };
1976            }
1977        } else {
1978            task.to_owned()
1979        };
1980
1981        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1982    }
1983
1984    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1985        let Some(ep) = self.endpoints.get(agent_name) else {
1986            return StepResult {
1987                name: format!("[agent] {display_name}"),
1988                status: StepStatus::Failed,
1989                message: format!("agent endpoint '{agent_name}' not found"),
1990            };
1991        };
1992
1993        if ep.url.is_empty() {
1994            return StepResult {
1995                name: format!("[agent] {display_name}"),
1996                status: StepStatus::Failed,
1997                message: format!("agent endpoint '{agent_name}' has no URL"),
1998            };
1999        }
2000
2001        self.reporter.debug(format!("agent {agent_name}: {task}"));
2002
2003        let url = ep.url.clone();
2004        let client = A2aClient::new(&url, self.timeout);
2005        let task_clone = task.to_owned();
2006
2007        let response = std::thread::spawn(move || {
2008            let rt = tokio::runtime::Builder::new_current_thread()
2009                .enable_all()
2010                .build()
2011                .unwrap();
2012            rt.block_on(client.send_task(&task_clone))
2013        })
2014        .join()
2015        .unwrap();
2016
2017        // Record the flat-cost call
2018        self.usage.record_flat_call(agent_name, ep);
2019
2020        match response {
2021            Ok(text) => {
2022                let clean = text.trim().to_owned();
2023                if verdict_is_pass(&clean) {
2024                    StepResult {
2025                        name: format!("[agent] {display_name}"),
2026                        status: StepStatus::Passed,
2027                        message: format!("PASS: {clean}"),
2028                    }
2029                } else if verdict_is_fail(&clean) {
2030                    StepResult {
2031                        name: format!("[agent] {display_name}"),
2032                        status: StepStatus::Failed,
2033                        message: clean,
2034                    }
2035                } else {
2036                    StepResult {
2037                        name: format!("[agent] {display_name}"),
2038                        status: StepStatus::Passed,
2039                        message: format!("response: {clean}"),
2040                    }
2041                }
2042            }
2043            Err(e) => StepResult {
2044                name: format!("[agent] {display_name}"),
2045                status: StepStatus::Failed,
2046                message: format!("agent call failed: {e}"),
2047            },
2048        }
2049    }
2050
2051    /// Runs an MCP tool call step.
2052    fn run_mcp(
2053        &self,
2054        server_name: &str,
2055        tool_name: &str,
2056        args: Option<&serde_json::Value>,
2057    ) -> StepResult {
2058        let Some(ep) = self.endpoints.get(server_name) else {
2059            return StepResult {
2060                name: format!("[mcp] {server_name}:{tool_name}"),
2061                status: StepStatus::Failed,
2062                message: format!("MCP server endpoint '{server_name}' not found"),
2063            };
2064        };
2065
2066        let cmd = ep.command.as_deref().unwrap_or("");
2067        if cmd.is_empty() {
2068            return StepResult {
2069                name: format!("[mcp] {server_name}:{tool_name}"),
2070                status: StepStatus::Failed,
2071                message: format!("MCP server '{server_name}' has no command configured"),
2072            };
2073        }
2074
2075        self.reporter
2076            .debug(format!("mcp {server_name} {tool_name}"));
2077
2078        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
2079
2080        let command = cmd.to_owned();
2081        let args_vec = ep.args.clone();
2082        let tool = tool_name.to_owned();
2083
2084        let response = std::thread::spawn(move || {
2085            let mut mcp_client =
2086                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
2087            mcp_client
2088                .call_tool(&tool, &args_val)
2089                .map_err(|e| e.to_string())
2090        })
2091        .join()
2092        .unwrap();
2093
2094        // Record the flat-cost call
2095        self.usage.record_flat_call(server_name, ep);
2096
2097        match response {
2098            Ok(result) => {
2099                if result.isError {
2100                    StepResult {
2101                        name: format!("[mcp] {server_name}:{tool_name}"),
2102                        status: StepStatus::Failed,
2103                        message: result.to_string(),
2104                    }
2105                } else {
2106                    StepResult {
2107                        name: format!("[mcp] {server_name}:{tool_name}"),
2108                        status: StepStatus::Passed,
2109                        message: result.to_string(),
2110                    }
2111                }
2112            }
2113            Err(e) => StepResult {
2114                name: format!("[mcp] {server_name}:{tool_name}"),
2115                status: StepStatus::Failed,
2116                message: format!("MCP call failed: {e}"),
2117            },
2118        }
2119    }
2120
2121    // ── helpers ──────────────────────────────────────────────────────────
2122
2123    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
2124    /// the runner's default LLM config for any unset fields.
2125    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
2126        LlmConfig {
2127            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
2128                // Bedrock builds its endpoint from the resolved AWS region
2129                // when no URL is given — never inherit the default LLM URL.
2130                endpoint.url.clone()
2131            } else if endpoint.url.is_empty() {
2132                self.llm.url.clone()
2133            } else {
2134                endpoint.url.clone()
2135            },
2136            model: endpoint
2137                .model
2138                .clone()
2139                .unwrap_or_else(|| self.llm.model.clone()),
2140            api_key: endpoint
2141                .api_key
2142                .clone()
2143                .or_else(|| self.llm.api_key.clone()),
2144            headers: if endpoint.headers.is_empty() {
2145                self.llm.headers.clone()
2146            } else {
2147                endpoint.headers.clone()
2148            },
2149            timeout: self.llm.timeout,
2150            temperature: self.llm.temperature,
2151            thinking: self.llm.thinking,
2152            model_params: self.llm.model_params.clone(),
2153            cache: endpoint.cache_markers,
2154            max_attempts: endpoint.max_attempts.max(1),
2155            provider: endpoint.provider,
2156            deployment: endpoint.deployment.clone(),
2157            api_version: endpoint.api_version.clone(),
2158            auth: endpoint.auth.clone(),
2159            header_commands: endpoint.header_commands.clone(),
2160            aws: endpoint.aws.clone(),
2161        }
2162    }
2163
2164    /// Runs a single LLM call against an ordered endpoint chain (primary +
2165    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
2166    /// the first endpoint that answers wins. Returns the response together
2167    /// with the chain index of the answering endpoint (0 = primary) so the
2168    /// caller can attribute usage to the correct endpoint.
2169    ///
2170    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
2171    /// shows duration, tokens, cost and the answering endpoint per call.
2172    /// Run-level context handed to every LLM call as the FIRST block of the
2173    /// user message. Contents that are stable for the whole run ("run
2174    /// started", "target site") come first so upstream provider prefix
2175    /// caching stays effective; the current time is the last line because
2176    /// it changes on every call.
2177    fn run_context(&self) -> String {
2178        let mut parts = vec![
2179            "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
2180            "================================================================".into(),
2181            format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
2182        ];
2183        if let Some(base) = self.config.base_url.as_deref() {
2184            parts.push(format!("Target site: {base}"));
2185        }
2186        let now = SystemTime::now()
2187            .duration_since(UNIX_EPOCH)
2188            .map_or(0, |d| d.as_secs());
2189        parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
2190        parts.join("\n")
2191    }
2192
2193    #[allow(clippy::cast_possible_truncation, clippy::too_many_lines)]
2194    fn llm_call_chain(
2195        &self,
2196        chain: &[&ResolvedEndpoint],
2197        system: &str,
2198        user: &str,
2199        image: Option<&[String]>,
2200        purpose: &str,
2201    ) -> Result<(crate::costs::LlmResponse, usize), String> {
2202        if chain.is_empty() {
2203            return Err("empty LLM endpoint chain".into());
2204        }
2205        let primary = self.build_llm_for_endpoint(chain[0]);
2206        let fallbacks: Vec<LlmConfig> = chain[1..]
2207            .iter()
2208            .map(|e| self.build_llm_for_endpoint(e))
2209            .collect();
2210
2211        let (test, index) = self
2212            .current_step
2213            .borrow()
2214            .as_ref()
2215            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
2216        let primary_endpoint = chain[0].name.clone();
2217        let primary_model = primary.model.clone();
2218        self.emit_event(&TestEvent::LlmCallStarted {
2219            test: test.clone(),
2220            index,
2221            endpoint: primary_endpoint.clone(),
2222            model: primary_model.clone(),
2223            purpose: purpose.to_owned(),
2224        });
2225
2226        let started = Instant::now();
2227        let sys = system.to_owned();
2228        let context = self.run_context();
2229        let user = if context.is_empty() {
2230            user.to_owned()
2231        } else {
2232            format!("{context}\n\n{user}")
2233        };
2234        let image = image.map(<[String]>::to_vec);
2235
2236        let result = std::thread::spawn(move || {
2237            let rt = tokio::runtime::Builder::new_current_thread()
2238                .enable_all()
2239                .build()
2240                .unwrap();
2241            let call = async {
2242                match image.as_deref() {
2243                    Some(img) => {
2244                        llm_chat_vision_with_usage_chain(
2245                            &primary,
2246                            &fallbacks,
2247                            &sys,
2248                            &user,
2249                            Some(img),
2250                        )
2251                        .await
2252                    }
2253                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2254                }
2255            };
2256            rt.block_on(call)
2257        })
2258        .join()
2259        .unwrap();
2260
2261        let duration_ms = started.elapsed().as_millis() as u64;
2262        match result {
2263            Ok((lr, idx)) => {
2264                let cost = calculate_llm_cost(chain[idx], &lr.usage);
2265                let answering = chain[idx].name.clone();
2266                let model = chain[idx]
2267                    .model
2268                    .clone()
2269                    .unwrap_or_else(|| primary_model.clone());
2270                self.usage
2271                    .record_llm_call(&answering, chain[idx], &model, &lr.usage);
2272                self.emit_event(&TestEvent::LlmCallFinished {
2273                    test,
2274                    index,
2275                    endpoint: answering,
2276                    model,
2277                    purpose: purpose.to_owned(),
2278                    ok: true,
2279                    duration_ms,
2280                    input_tokens: lr.usage.prompt_tokens,
2281                    output_tokens: lr.usage.completion_tokens,
2282                    cached_input_tokens: lr.usage.cached_input_tokens,
2283                    cache_creation_input_tokens: lr.usage.cache_creation_input_tokens,
2284                    cost,
2285                    error: None,
2286                });
2287                Ok((lr, idx))
2288            }
2289            Err(e) => {
2290                self.emit_event(&TestEvent::LlmCallFinished {
2291                    test,
2292                    index,
2293                    endpoint: primary_endpoint,
2294                    model: primary_model,
2295                    purpose: purpose.to_owned(),
2296                    ok: false,
2297                    duration_ms,
2298                    input_tokens: 0,
2299                    output_tokens: 0,
2300                    cached_input_tokens: 0,
2301                    cache_creation_input_tokens: 0,
2302                    cost: 0.0,
2303                    error: Some(e.clone()),
2304                });
2305                Err(e)
2306            }
2307        }
2308    }
2309
2310    /// Resolves a CSS selector for the target element. Uses the explicit
2311    /// `selector` if provided, otherwise asks the LLM to find the element
2312    /// from the natural language `target` description and page DOM.
2313    ///
2314    /// LLM responses are sanitized and verified against the live page: a
2315    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2316    /// immediately with the raw LLM output, and a selector that matches
2317    /// nothing triggers one retry with feedback before failing.
2318    #[allow(clippy::too_many_lines)]
2319    fn resolve_selector(
2320        &self,
2321        css_override: Option<&str>,
2322        target: &str,
2323        step_endpoint: Option<&str>,
2324        test_endpoint: Option<&str>,
2325        tab: &Tab,
2326    ) -> Result<String, String> {
2327        if let Some(explicit) = css_override {
2328            return Ok(explicit.to_owned());
2329        }
2330
2331        let dom_info = extract_dom_info(tab)?;
2332        let page_content = get_page_text(tab);
2333
2334        let system = concat!(
2335            "You are a browser automation selector generator. ",
2336            "Given a web page's content and interactive elements, ",
2337            "return ONLY the best CSS selector for the described element. ",
2338            "Output nothing except the CSS selector. ",
2339            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2340            "[name=\"...\"], tag.class, tag. ",
2341            "Never output explanations, markdown, or extra text."
2342        );
2343
2344        let user = format!(
2345            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2346            page_content.url,
2347            page_content.title,
2348            truncate(&page_content.body_text, 4000),
2349            dom_info,
2350            target,
2351        );
2352
2353        let retry_user = format!(
2354            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2355            "The selector must match at least one element currently present on the page.",
2356            page_content.url,
2357            page_content.title,
2358            truncate(&page_content.body_text, 4000),
2359            dom_info,
2360            target,
2361        );
2362
2363        self.reporter.debug(format!("LLM targeting: {target}"));
2364
2365        let chain = self
2366            .endpoints
2367            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2368        let sys = system.to_owned();
2369
2370        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2371
2372        let first = call_llm(&user);
2373        let (lr, _idx) = match first {
2374            Ok(lr) => lr,
2375            Err(e) => {
2376                return Err(format!("LLM element targeting failed: {e}"));
2377            }
2378        };
2379        let clean = sanitize_selector(&lr.content);
2380        self.reporter.debug(format!("resolved selector: {clean}"));
2381
2382        if selector_is_useless(&clean) {
2383            return Err(format!(
2384                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2385                raw = lr.content.trim(),
2386            ));
2387        }
2388        if let Err(reason) = validate_selector(&clean) {
2389            return Err(format!(
2390                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2391                raw = lr.content.trim(),
2392            ));
2393        }
2394        if !selector_matches(tab, &clean).unwrap_or(false) {
2395            // One retry with feedback: flaky models occasionally invent a
2396            // selector that does not exist on the page.
2397            self.reporter.warn(format!(
2398                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2399            ));
2400            let second = call_llm(&retry_user);
2401            let (lr2, _idx2) = match second {
2402                Ok(lr2) => lr2,
2403                Err(e) => {
2404                    return Err(format!(
2405                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2406                    ));
2407                }
2408            };
2409            let clean2 = sanitize_selector(&lr2.content);
2410            self.reporter
2411                .debug(format!("resolved selector (retry): {clean2}"));
2412            if selector_is_useless(&clean2) {
2413                return Err(format!(
2414                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2415                    raw = lr2.content.trim(),
2416                    excerpt = truncate(&page_content.body_text, 300),
2417                ));
2418            }
2419            if !selector_matches(tab, &clean2).unwrap_or(false) {
2420                return Err(format!(
2421                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2422                ));
2423            }
2424            return Ok(clean2);
2425        }
2426
2427        Ok(clean)
2428    }
2429}
2430
2431/// Evaluates a JS expression that is expected to return a boolean.
2432fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2433    tab.evaluate(js, false)
2434        .map_err(|e| format!("evaluate failed: {e}"))?
2435        .value
2436        .and_then(|v| v.as_bool())
2437        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2438}
2439
2440/// Blank-page self-heal budget per wait step: how many reload cycles a
2441/// blank page gets before the wait gives up. The runner's egress and
2442/// resource bursts last minutes, so a single reload (v0.19.5) was not
2443/// enough when a burst outlived two wait budgets (immosai run #1651).
2444const MAX_BLANK_HEALS: u32 = 3;
2445/// Backoff sleep (seconds) before the 2nd/3rd blank-heal reload, giving
2446/// a transient runner outage time to clear.
2447const BLANK_HEAL_BACKOFF_SECS: u64 = 30;
2448
2449/// True when the tab rendered nothing meaningful: no body, an empty
2450/// body, or body text that is only whitespace. This is the signature
2451/// of a stalled SPA boot (index.html served, JS chunks never arrived —
2452/// runner egress stall), where any wait can only time out. A real
2453/// error page (e.g. `ERR_CONNECTION_REFUSED`) carries text and is NOT
2454/// blank, so those are left alone.
2455fn page_is_blank(tab: &Tab) -> bool {
2456    const JS: &str = "(() => { if (!document.body) return true; \
2457        const t = (document.body.innerText || '').trim(); \
2458        return document.body.childElementCount === 0 || t.length === 0; })()";
2459    eval_bool(tab, JS).unwrap_or(false)
2460}
2461
2462/// Checks whether a CSS selector matches at least one current element.
2463fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2464    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2465}
2466
2467// ── Free helper functions ──────────────────────────────────────────────
2468
2469fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2470    let name = format!("[navigate] {full_url}");
2471    match tab.navigate_to(full_url) {
2472        Ok(_) => {
2473            let _ = tab.wait_until_navigated();
2474            StepResult {
2475                name,
2476                status: StepStatus::Passed,
2477                message: format!("navigated to {full_url}"),
2478            }
2479        }
2480        Err(e) => StepResult {
2481            name,
2482            status: StepStatus::Failed,
2483            message: format!("navigation failed: {e}"),
2484        },
2485    }
2486}
2487
2488fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2489    let result = tab
2490        .evaluate(DOM_EXTRACT_JS, false)
2491        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2492
2493    let json_str = result
2494        .value
2495        .as_ref()
2496        .and_then(|v| v.as_str())
2497        .unwrap_or("[]");
2498
2499    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2500
2501    if elements.is_empty() {
2502        return Ok("(no interactive elements found)".to_owned());
2503    }
2504
2505    Ok(elements.join("\n"))
2506}
2507
2508fn get_page_text(tab: &Tab) -> PageContent {
2509    let url = tab.get_url();
2510
2511    let title = tab
2512        .evaluate("document.title", false)
2513        .ok()
2514        .and_then(|r| r.value)
2515        .and_then(|v| v.as_str().map(String::from))
2516        .unwrap_or_else(|| "unknown".to_owned());
2517
2518    let body_text = tab
2519        .evaluate(
2520            "document.body ? document.body.innerText : document.documentElement.innerText",
2521            false,
2522        )
2523        .ok()
2524        .and_then(|r| r.value)
2525        .and_then(|v| v.as_str().map(String::from))
2526        .unwrap_or_default();
2527
2528    PageContent {
2529        url,
2530        title,
2531        body_text: truncate(&body_text, 8000),
2532    }
2533}
2534
2535fn resolve_url(url: &str, base_url: &str) -> String {
2536    if url.starts_with("http://") || url.starts_with("https://") {
2537        return url.to_owned();
2538    }
2539    let base = base_url.trim_end_matches('/');
2540    if url.starts_with('/') {
2541        format!("{base}{url}")
2542    } else {
2543        format!("{base}/{url}")
2544    }
2545}
2546
2547/// Origin (`scheme://host[:port]`) of a base URL, used to scope the
2548/// per-test `Storage.clearDataForOrigin` call. Returns `None` when the
2549/// URL has no recognizable scheme/host (the clear is skipped).
2550#[must_use]
2551fn origin_of(base_url: &str) -> Option<String> {
2552    let url = if base_url.contains("://") {
2553        base_url.to_owned()
2554    } else {
2555        format!("https://{base_url}")
2556    };
2557    let (scheme, rest) = url.split_once("://")?;
2558    let authority = rest
2559        .split(['/', '?', '#'])
2560        .next()
2561        .filter(|a| !a.is_empty())?;
2562    Some(format!("{scheme}://{authority}"))
2563}
2564
2565/// True when a test performs its own login: its first navigate step
2566/// targets a login route (org `/auth/login`, tenant `/tenant/login`),
2567/// or it has no navigate step at all and rides the auto-navigate onto a
2568/// login `start_url`. Such tests need a cleared session — with a live one
2569/// the SPA bounces the login page before the form ever mounts. Tests
2570/// whose own first navigate goes to an app page keep the shared session
2571/// (most "page loads" tests rely on it) — a blanket clear broke exactly
2572/// those on immosai run #1643.
2573#[must_use]
2574fn test_targets_login(start_url: &str, steps: &[TestStep]) -> bool {
2575    for step in steps {
2576        if let TestStep::Navigate { url, .. } = step {
2577            return url.contains("login");
2578        }
2579    }
2580    start_url.contains("login")
2581}
2582
2583/// Human-readable label for a step, used when steps are skipped after an
2584/// earlier failure.
2585fn step_label(step: &TestStep) -> String {
2586    match step {
2587        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2588        TestStep::Click { target, .. } => format!("[click] {target}"),
2589        TestStep::Type { target, .. } => format!("[type] {target}"),
2590        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2591        TestStep::Assert {
2592            definition,
2593            preset,
2594            prompt,
2595            ..
2596        } => definition.as_ref().map_or_else(
2597            || {
2598                preset.as_ref().map_or_else(
2599                    || {
2600                        prompt.as_ref().map_or_else(
2601                            || "[assert]".to_owned(),
2602                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2603                        )
2604                    },
2605                    |p| format!("[assert] {p}"),
2606                )
2607            },
2608            |d| format!("[assert] {d}"),
2609        ),
2610        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2611        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2612        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2613    }
2614}
2615
2616/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2617#[must_use]
2618const fn step_kind_label(step: &TestStep) -> &'static str {
2619    match step {
2620        TestStep::Navigate { .. } => "navigate",
2621        TestStep::Click { .. } => "click",
2622        TestStep::Type { .. } => "type",
2623        TestStep::Wait { .. } => "wait",
2624        TestStep::Assert { .. } => "assert",
2625        TestStep::Screenshot { .. } => "screenshot",
2626        TestStep::Agent { .. } => "agent",
2627        TestStep::Mcp { .. } => "mcp",
2628    }
2629}
2630
2631// ── Support types ──────────────────────────────────────────────────────
2632
2633#[derive(Default)]
2634struct TestRunResult {
2635    passed: u32,
2636    failed: u32,
2637    skipped: u32,
2638    total: u32,
2639    details: Vec<StepResult>,
2640}
2641
2642struct PageContent {
2643    url: String,
2644    title: String,
2645    body_text: String,
2646}
2647
2648#[cfg(test)]
2649mod tests {
2650    use super::unix_to_rfc3339;
2651
2652    #[test]
2653    fn rfc3339_epoch_and_reference_dates() {
2654        assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2655        assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2656        assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2657        assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2658    }
2659
2660    #[test]
2661    fn rfc3339_handles_leap_years() {
2662        assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2663        assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2664    }
2665}
2666
2667#[cfg(test)]
2668mod verdict_parse_tests {
2669    use super::{verdict_is_fail, verdict_is_pass};
2670
2671    #[test]
2672    fn tolerates_markdown_punctuation_and_natural_language() {
2673        assert!(verdict_is_pass("PASS"));
2674        assert!(verdict_is_pass("pass"));
2675        assert!(verdict_is_pass("**PASS**\n\nThe page is clean."));
2676        assert!(verdict_is_pass("passes - no explicit error visible"));
2677        assert!(verdict_is_pass("  \"pass\""));
2678        assert!(!verdict_is_pass("FAIL: something broke"));
2679        assert!(!verdict_is_pass("**FAIL** broken"));
2680
2681        assert!(verdict_is_fail("**FAIL** broken"));
2682        assert!(verdict_is_fail("fails - error toast shown"));
2683        assert!(!verdict_is_fail("passes - ok"));
2684    }
2685}