Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// Formats a UNIX epoch timestamp as an RFC 3339 UTC string
26/// (`2026-09-19T16:04:00Z`) without pulling in a date dependency.
27/// Civil-from-days conversion after Howard Hinnant.
28#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31    let secs = i64::try_from(secs).unwrap_or(0);
32    let days = secs.div_euclid(86_400);
33    let rem = secs.rem_euclid(86_400);
34    let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35    let z = days + 719_468;
36    let era = z.div_euclid(146_097);
37    let doe = z.rem_euclid(146_097);
38    let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40    let mp = (5 * doy + 2) / 153;
41    let d = doy - (153 * mp + 2) / 5 + 1;
42    let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43    let y = yoe + era * 400 + i64::from(mth <= 2);
44    format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47/// One detected layout defect (`layout_no_issues` preset).
48#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50    #[serde(rename = "type")]
51    issue_type: String,
52    element: String,
53    detail: String,
54}
55
56/// In-browser DOM layout scan for `layout_no_issues`.
57///
58/// Geometry-only checks (no LLM, no pixels):
59/// 1. page horizontal overflow (`scrollWidth` > viewport width);
60/// 2. elements outside the viewport that scrolling cannot reveal
61///    (fixed elements off-screen, left/negative overflow, right-edge
62///    overflow beyond the horizontally scrollable area, and bottom
63///    overflow on a page that cannot scroll down) — below-the-fold
64///    content on a tall scrollable page is normal flow, NOT a defect;
65/// 3. text clipped by `overflow: hidden` containers whose content
66///    is measurably larger than the box;
67/// 4. interactive elements (buttons/links/inputs) whose center point
68///    is covered by a different element that would intercept the click.
69///
70/// Elements whose class matches a configured ignore prefix (default:
71/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
72/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
73/// skipped — they are intentionally 1x1 / off-screen. The prefix list
74/// is injected at `__IGNORE_CLASSES__` from
75/// [`ScenarioConfig::layout_ignore_classes`].
76///
77/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
78/// sticky headers) is excluded by the position/relation filters.
79const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81  const issues = [];
82  const push = (type, el, detail) => {
83    if (issues.length >= 30) return;
84    let element = el.tagName.toLowerCase();
85    if (el.id) element += '#' + el.id;
86    else if (typeof el.className === 'string' && el.className.trim())
87      element += '.' + el.className.trim().split(/\s+/).join('.');
88    issues.push({ type, element, detail: String(detail).slice(0, 220) });
89  };
90  const vw = document.documentElement.clientWidth || window.innerWidth;
91  const vh = document.documentElement.clientHeight || window.innerHeight;
92  if (!vw || !vh) return JSON.stringify(issues);
93  const de = document.documentElement;
94  // 1. Page-level horizontal overflow.
95  if (maxSW > vw + 2)
96    push('page-overflow-x', de,
97      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98  // Class-name prefixes to skip (injected; default = Angular CDK
99  // screen-reader helpers, which are intentionally 1x1 / off-screen).
100  const ignorePrefixes = __IGNORE_CLASSES__;
101  const isIgnored = (el) => {
102    if (typeof el.className !== 'string' || !el.className.trim()) return false;
103    const classes = el.className.trim().split(/\s+/);
104    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105  };
106  const body = document.body;
107  // The document element is not always the scroll container: the app may
108  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110  // "not scrollable" — measure the scrollable content height/width of the
111  // document element AND the body, and take the max.
112  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114  const vScrollable = maxSH > vh + 2;
115  const hScrollable = maxSW > vw + 2;
116  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117  // this element into view on the given axis — i.e. the element is inside a
118  // scrolling region, so lying beyond the viewport cut is not a defect.
119  const reachableByScroller = (el, axis) => {
120    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123    let p = el.parentElement;
124    while (p && p !== body) {
125      const pcs = getComputedStyle(p);
126      const o = pcs[ovProp];
127      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128          p[dim] > p[dimClient] + 2) return true;
129      p = p.parentElement;
130    }
131    return false;
132  };
133  const selfOverflowing = (el, cs, axis) => {
134    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138      el[dim] > el[dimClient] + 2;
139  };
140  // True when an ancestor clips this axis with overflow:hidden/clip — the
141  // element's overhang is not visible, so treat it as reachable.
142  const clippedByAncestor = (el, axis) => {
143    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144    let p = el.parentElement;
145    while (p && p !== body) {
146      const o = getComputedStyle(p)[ovProp];
147      if (o === 'hidden' || o === 'clip') return true;
148      p = p.parentElement;
149    }
150    return false;
151  };
152  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154  const hasContent = (el) =>
155    ((el.textContent || '').trim().length > 0) ||
156    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157  // 2. Elements outside the viewport that scrolling cannot reveal.
158  for (const el of all) {
159    const cs = getComputedStyle(el);
160    if (!visible(cs)) continue;
161    if (isIgnored(el)) continue;
162    const r = el.getBoundingClientRect();
163    if (r.width < 2 || r.height < 2) continue;
164    if (!hasContent(el) && el.children.length === 0) continue;
165    if (cs.position === 'fixed') {
166      // Fixed elements never move with the scroll: any edge outside the
167      // viewport is unreachable content and therefore a defect.
168      const overTop = -r.top;
169      const overLeft = -r.left;
170      const overRight = r.right - vw;
171      const overBottom = r.bottom - vh;
172      // Top/left overshoot is never reachable; bottom/right overshoot is
173      // only a defect when the fixed element cannot scroll that content
174      // into view itself (e.g. an opened Material drawer whose inner
175      // container scrolls is not a "cut-off" defect).
176      if (overTop > 2 || overLeft > 2 ||
177          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179        let where = '';
180        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186        push('element-out-of-viewport', el,
187          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188      }
189      continue;
190    }
191    if (cs.position === 'sticky') continue;
192    // Negative left overflow cannot be reached by scrolling (scrollLeft
193    // never goes below 0).
194    if (r.left < -2) {
195      push('element-out-of-viewport', el,
196        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197      continue;
198    }
199    // Negative top with the page at the top means the element sits above
200    // the document origin — also unreachable.
201    if (r.top < -2 && de.scrollTop <= 2) {
202      push('element-out-of-viewport', el,
203        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204      continue;
205    }
206    const overRight = r.right - vw;
207    // Right-edge overflow is only a defect when the page cannot scroll
208    // horizontally to reveal it (or the element sticks out past the
209    // scrollable content width itself).
210    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211      push('element-out-of-viewport', el,
212        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213      continue;
214    }
215    // Below-the-fold content on a scrollable page is normal (tall landing
216    // pages); only flag bottom overflow the user can never scroll to.
217    const overBottom = r.bottom - vh;
218    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219      push('element-out-of-viewport', el,
220        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221    }
222  }
223  // 3. Text clipped by overflow:hidden containers.
224  for (const el of all) {
225    if (isIgnored(el)) continue;
226    const cs = getComputedStyle(el);
227    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229    if (!(el.textContent || '').trim()) continue;
230    // Single-line truncation with an ellipsis is an intentional design
231    // pattern (Tailwind .truncate etc.), not a clipping defect.
232    if (cs.textOverflow === 'ellipsis') continue;
233    push('text-clipped', el,
234      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236  }
237  // 4. Interactive elements covered by a different element.
238  const interactive =
239    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240  const targets = document.querySelectorAll(interactive);
241  for (const el of targets) {
242    if (isIgnored(el)) continue;
243    const r = el.getBoundingClientRect();
244    if (r.width < 6 || r.height < 6) continue;
245    const cs = getComputedStyle(el);
246    if (!visible(cs)) continue;
247    const cx = r.left + r.width / 2;
248    const cy = r.top + r.height / 2;
249    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250    const top = document.elementFromPoint(cx, cy);
251    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252    if (isIgnored(top)) continue;
253    const tcs = getComputedStyle(top);
254    if (!visible(tcs)) continue;
255    if (tcs.pointerEvents === 'none') continue;
256    const tr = top.getBoundingClientRect();
257    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260    push('element-overlap', el,
261      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262      ' is covered by <' + tname + '>');
263  }
264  return JSON.stringify(issues);
265})()
266"#;
267
268/// How long the CDP connection stays open after the browser goes quiet.
269///
270/// `headless_chrome` ships a 30s default and tears down the entire connection
271/// when no traffic arrives for that long; a run must own its connection for
272/// its full duration instead.
273const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275/// Executes a [`Scenario`] against a real browser with optional LLM
276/// assistance for element targeting and assertions.
277pub struct ScenarioRunner {
278    config: ScenarioConfig,
279    definitions: HashMap<String, AssertDefinition>,
280    llm: LlmConfig,
281    timeout: Duration,
282    viewport_width: u32,
283    viewport_height: u32,
284    /// The viewport currently applied in the browser (CDP emulation).
285    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
286    applied_viewport: std::cell::Cell<(u32, u32)>,
287    endpoints: EndpointRegistry,
288    usage: Arc<UsageTracker>,
289    budgets: BudgetTracker,
290    /// Directory for failure artifacts (screenshots).
291    artifacts_dir: PathBuf,
292    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
293    reporter: Arc<Reporter>,
294    /// The test + step index currently executing, for LLM-call events.
295    current_step: std::cell::RefCell<Option<(String, u32)>>,
296    /// Epoch seconds of runner creation — constant for the whole run so
297    /// the LLM context preamble below is cacheable by upstream providers.
298    run_started: u64,
299    /// Whether this runner emits the `RunStarted`/`RunFinished` events.
300    /// When several scenario files are run concurrently by
301    /// [`crate::parallel::run_scenarios`], the orchestrator owns those two
302    /// events (one per batch), so each per-file runner suppresses its own.
303    emit_run_events: bool,
304}
305
306/// Aggregated results from a scenario run.
307#[derive(Debug, Default)]
308pub struct RunReport {
309    /// Number of tests that passed.
310    pub tests_passed: u32,
311    /// Number of tests that failed.
312    pub tests_failed: u32,
313    /// Number of steps that passed.
314    pub passed: u32,
315    /// Number of steps that failed.
316    pub failed: u32,
317    /// Number of steps that were skipped.
318    pub skipped: u32,
319    /// Per-step details.
320    pub details: Vec<StepResult>,
321}
322
323/// Result of a single step execution.
324#[derive(Debug, Clone)]
325pub struct StepResult {
326    /// The step name.
327    pub name: String,
328    /// Whether the step passed, failed, or was skipped.
329    pub status: StepStatus,
330    /// Human-readable result message.
331    pub message: String,
332}
333
334/// Outcome for a single step.
335pub use crate::events::StepStatus;
336
337/// Predefined assertion preset definition.
338struct AssertPreset {
339    name: &'static str,
340    system: &'static str,
341    user_template: &'static str,
342}
343
344/// Built-in assertion presets.
345#[allow(clippy::literal_string_with_formatting_args)]
346const ASSERTION_PRESETS: &[AssertPreset] = &[
347    AssertPreset {
348        name: "no_error_on_page",
349        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
350        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
351    },
352    AssertPreset {
353        name: "text_visible",
354        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
355        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
356    },
357    AssertPreset {
358        name: "element_exists",
359        system: "You are a QA tester. Check if a described UI element exists on a web page.",
360        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
361    },
362    AssertPreset {
363        name: "visual_no_issues",
364        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
365        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
366    },
367    AssertPreset {
368        name: "visual_no_overlaps",
369        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
370        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
371    },
372    AssertPreset {
373        name: "visual_fills_viewport",
374        system: "You are a visual QA engineer inspecting a website screenshot for UNDER-FILLED layouts: the rendered content stops at a fraction of the viewport and the remaining area is a large blank region the layout should have filled. Typical defects: page content ending around the halfway mark of the viewport with an empty, background-only area below it; an app shell, dashboard, or main panel collapsing to part of its expected height and leaving dead space; content that fills only the left portion of a wide viewport with unused space to the right. Do NOT fail intentionally short or centered pages: a login or signup card, a landing hero, a short article, a confirmation step, or an empty state that does not fill the viewport is normal design. Only fail when the blank region is clearly a layout defect — the page looks cut off or collapsed, or a content container (calendar, table, chart, map, panel) visibly fails to stretch to the space its own layout provides.",
375        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nDoes the page content fill the viewport as its layout intends — with no large blank region where the page appears to end partway down or partway across? Respond with exactly \"PASS\" if the page fills the viewport or is intentionally short, or \"FAIL: <describe the under-filled region and where it appears>\" if the content clearly stops at a fraction of the available page.",
376    },
377    AssertPreset {
378        name: "visual_text_visible",
379        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
380        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
381    },
382    AssertPreset {
383        name: "layout_no_issues",
384        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
385        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
386    },
387];
388
389/// Whether an LLM verdict should be read as PASS.
390///
391/// Models routinely wrap the verdict in markdown (`**PASS** …`) or prefix it
392/// with punctuation, which a bare `starts_with("pass")` misreads as a failure.
393/// Strip any leading non-alphanumerics, then match the leading word; this also
394/// accepts the natural-language forms `pass` / `passes`.
395fn verdict_is_pass(content: &str) -> bool {
396    content
397        .trim()
398        .trim_start_matches(|c: char| !c.is_alphanumeric())
399        .to_lowercase()
400        .starts_with("pass")
401}
402
403/// Whether an LLM verdict should be read as FAIL (see [`verdict_is_pass`]).
404fn verdict_is_fail(content: &str) -> bool {
405    content
406        .trim()
407        .trim_start_matches(|c: char| !c.is_alphanumeric())
408        .to_lowercase()
409        .starts_with("fail")
410}
411
412impl ScenarioRunner {
413    /// Creates a new runner with the given scenario configuration and
414    /// assertion definitions.
415    #[must_use]
416    #[allow(clippy::needless_pass_by_value)]
417    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
418        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
419    }
420
421    /// Creates a runner that reports run events through the given reporter.
422    #[must_use]
423    #[allow(clippy::needless_pass_by_value)]
424    pub fn with_reporter(
425        scenario_config: ScenarioConfig,
426        definitions: Vec<AssertDefinition>,
427        reporter: Arc<Reporter>,
428    ) -> Self {
429        Self::with_reporter_mode(scenario_config, definitions, reporter, true)
430    }
431
432    /// Creates a runner that reports run events through the given reporter
433    /// but suppresses the `RunStarted`/`RunFinished` events. Used by the
434    /// parallel orchestrator, which emits those once per batch.
435    #[must_use]
436    #[allow(clippy::needless_pass_by_value)]
437    pub fn with_reporter_parallel(
438        scenario_config: ScenarioConfig,
439        definitions: Vec<AssertDefinition>,
440        reporter: Arc<Reporter>,
441    ) -> Self {
442        Self::with_reporter_mode(scenario_config, definitions, reporter, false)
443    }
444
445    #[must_use]
446    #[allow(clippy::needless_pass_by_value)]
447    fn with_reporter_mode(
448        scenario_config: ScenarioConfig,
449        definitions: Vec<AssertDefinition>,
450        reporter: Arc<Reporter>,
451        emit_run_events: bool,
452    ) -> Self {
453        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
454            &scenario_config,
455        ));
456        let llm = LlmConfig {
457            url: scenario_config
458                .llm_url
459                .clone()
460                .unwrap_or_else(crate::llm_base_url),
461            model: scenario_config
462                .llm_model
463                .clone()
464                .unwrap_or_else(crate::llm_model),
465            api_key: scenario_config
466                .llm_api_key
467                .clone()
468                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
469            headers: if scenario_config.llm_headers.is_empty() {
470                crate::parse_headers_env()
471            } else {
472                scenario_config.llm_headers.clone()
473            },
474            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
475            temperature: scenario_config.temperature,
476            thinking: scenario_config.thinking,
477            model_params: scenario_config.model_params.clone(),
478            cache: scenario_config.cache.unwrap_or(true),
479            max_attempts: crate::default_llm_attempts(),
480            provider: crate::scenario::Provider::Openai,
481            deployment: None,
482            api_version: None,
483            auth: crate::scenario::AuthConfig::default(),
484            header_commands: std::collections::HashMap::new(),
485            aws: crate::scenario::AwsConfig::default(),
486        };
487        let endpoints = EndpointRegistry::from_config(
488            &scenario_config.endpoints,
489            Some(&llm),
490            crate::endpoints::EndpointDefaults {
491                cache: scenario_config.cache,
492                cache_pricing: scenario_config.cache_pricing,
493            },
494        );
495        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
496        let defs_map: HashMap<String, AssertDefinition> = definitions
497            .into_iter()
498            .map(|d| (d.name.clone(), d))
499            .collect();
500
501        Self {
502            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
503            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
504            viewport_height: scenario_config.viewport_height.unwrap_or(720),
505            applied_viewport: std::cell::Cell::new((0, 0)),
506            config: scenario_config.clone(),
507            definitions: defs_map,
508            llm,
509            endpoints,
510            usage: Arc::new(UsageTracker::new()),
511            budgets,
512            artifacts_dir: PathBuf::from(
513                scenario_config
514                    .artifacts_dir
515                    .unwrap_or_else(|| "artifacts".to_owned()),
516            ),
517            reporter,
518            current_step: std::cell::RefCell::new(None),
519            run_started: SystemTime::now()
520                .duration_since(UNIX_EPOCH)
521                .map_or(0, |d| d.as_secs()),
522            emit_run_events,
523        }
524    }
525
526    /// Emits an event; a sink failure degrades to a console warning so a
527    /// broken log file can never mask the run itself.
528    fn emit_event(&self, event: &TestEvent) {
529        if let Err(err) = self.reporter.emit(event) {
530            use std::io::Write as _;
531            let mut out = std::io::stderr().lock();
532            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
533        }
534    }
535
536    /// Returns a clone of the [`UsageTracker`] for reporting.
537    #[must_use]
538    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
539        Arc::clone(&self.usage)
540    }
541
542    /// Returns a reference to the [`BudgetTracker`].
543    #[must_use]
544    pub const fn budget_tracker(&self) -> &BudgetTracker {
545        &self.budgets
546    }
547
548    /// Executes all test groups in the scenario and returns a report.
549    ///
550    /// # Errors
551    ///
552    /// Returns an error if the browser fails to launch.
553    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
554    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
555        let mut report = RunReport::default();
556
557        if tests.is_empty() {
558            self.reporter.warn("No tests defined in scenario.");
559            if self.emit_run_events {
560                self.emit_event(&TestEvent::RunFinished {
561                    tests_passed: 0,
562                    tests_failed: 0,
563                    steps_passed: 0,
564                    steps_failed: 0,
565                    steps_skipped: 0,
566                    total_cost: 0.0,
567                    total_tokens: 0,
568                    total_input_tokens: 0,
569                    total_output_tokens: 0,
570                    total_cached_input_tokens: 0,
571                    total_cache_creation_input_tokens: 0,
572                    models: Vec::new(),
573                    total_calls: 0,
574                });
575            }
576            return Ok(report);
577        }
578
579        if self.emit_run_events {
580            self.emit_event(&TestEvent::RunStarted {
581                total_tests: tests.len() as u32,
582            });
583        }
584
585        let browser_headless = self.config.browser_headless.unwrap_or(true);
586
587        let launch_opts = LaunchOptions {
588            headless: browser_headless,
589            window_size: Some((self.viewport_width, self.viewport_height)),
590            sandbox: false,
591            // headless_chrome defaults this to 30s and shuts down the whole CDP
592            // connection when no messages arrive for that long. A scenario can
593            // easily exceed 30s of browser silence (slow LLM targeting/assertion
594            // calls, page waits, budget checks between steps), after which every
595            // remaining step fails with "Unable to make method calls because
596            // underlying connection is closed" — one quiet gap kills the run.
597            // Open-ended scenarios must own the connection for their full
598            // duration, so keep it alive for 6 hours.
599            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
600            ..LaunchOptions::default()
601        };
602
603        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
604        let tab = browser.new_tab().context("failed to open browser tab")?;
605        match (
606            self.config.browser_basic_auth_user.clone(),
607            self.config.browser_basic_auth_password.clone(),
608        ) {
609            (Some(username), Some(password)) if !username.is_empty() && !password.is_empty() => {
610                tab.authenticate(Some(username), Some(password))
611                    .context("failed to configure browser HTTP Basic Auth")?;
612                tab.enable_fetch(None, Some(true))
613                    .context("failed to enable browser HTTP authentication")?;
614            }
615            (None, None) => {}
616            _ => anyhow::bail!("browser Basic Auth requires both username and password"),
617        }
618        let _ = tab.set_default_timeout(self.timeout);
619
620        // Start MCP server if configured
621        #[cfg(feature = "mcp-server")]
622        if let Some(ref mcp_cfg) = self.config.mcp_server {
623            if mcp_cfg.enabled {
624                let port = mcp_cfg.port;
625                std::thread::spawn(move || {
626                    let _ = crate::mcp_server::start_mcp_server(port);
627                });
628            }
629        }
630        #[cfg(not(feature = "mcp-server"))]
631        if let Some(mcp_cfg) = &self.config.mcp_server {
632            if mcp_cfg.enabled {
633                self.reporter
634                    .warn("MCP server configured but 'mcp-server' feature not enabled");
635            }
636        }
637
638        // Start A2A agent server if configured
639        #[cfg(feature = "a2a-server")]
640        if let Some(ref a2a_cfg) = self.config.a2a_server {
641            if a2a_cfg.enabled {
642                let port = a2a_cfg.port;
643                tokio::spawn(crate::a2a_server::start_a2a_server(port));
644            }
645        }
646        #[cfg(not(feature = "a2a-server"))]
647        if let Some(a2a_cfg) = &self.config.a2a_server {
648            if a2a_cfg.enabled {
649                self.reporter
650                    .warn("A2A server configured but 'a2a-server' feature not enabled");
651            }
652        }
653
654        let retry_budget = self.config.retry_failed_tests.unwrap_or(0);
655
656        for test in tests {
657            // A shared tab on a contended runner makes some page loads
658            // stall (a JS chunk or a GraphQL call hangs mid-flight) and
659            // the test fails on a wait/assert that a fresh run passes.
660            // Re-run the WHOLE test when it failed and the retry budget
661            // allows; the fresh per-test isolation (login clear +
662            // auto-navigate) applies to the retry too. Both attempts'
663            // LLM spend stays in the budget accounting (each attempt
664            // commits its usage); only the final attempt is reported.
665            let details_len = report.details.len();
666            let passed_before = report.passed;
667            let failed_before = report.failed;
668            let skipped_before = report.skipped;
669            #[allow(unused_assignments)]
670            let mut final_result = None;
671            let mut attempt = 0;
672            loop {
673                attempt += 1;
674                self.emit_event(&TestEvent::TestStarted {
675                    test: test.name.clone(),
676                });
677
678                self.usage.reset_per_test();
679
680                let test_started = Instant::now();
681                let test_result = self.run_test(test, &tab);
682                let duration_ms = test_started.elapsed().as_millis() as u64;
683                let usage = self.usage.current_test_snapshot();
684                self.usage.commit_test(&test.name);
685
686                self.emit_event(&TestEvent::TestFinished {
687                    test: test.name.clone(),
688                    passed: test_result.passed,
689                    failed: test_result.failed,
690                    skipped: test_result.skipped,
691                    duration_ms,
692                    cost: usage.total_cost,
693                    tokens: usage.total_tokens,
694                    input_tokens: usage.total_input_tokens,
695                    output_tokens: usage.total_output_tokens,
696                    cached_input_tokens: usage.total_cached_input_tokens,
697                    cache_creation_input_tokens: usage.total_cache_creation_input_tokens,
698                    models: usage.models.clone(),
699                    calls: usage.total_calls,
700                });
701
702                final_result = Some(test_result);
703                let result = final_result.as_ref().expect("just assigned");
704                if result.failed == 0 || result.total == 0 || attempt > retry_budget {
705                    break;
706                }
707                self.reporter.warn(format!(
708                    "! retrying failed test '{}' (attempt {}/{}) — a fresh run passes on \
709                     transient page-load stalls",
710                    test.name,
711                    attempt + 1,
712                    retry_budget + 1
713                ));
714                // Roll this failed attempt out of the report so the
715                // retry's outcome replaces it.
716                report.details.truncate(details_len);
717                report.passed = passed_before;
718                report.failed = failed_before;
719                report.skipped = skipped_before;
720            }
721
722            let test_result = final_result.unwrap_or_default();
723            if test_result.failed == 0 && test_result.total > 0 {
724                report.tests_passed += 1;
725            } else if test_result.total > 0 {
726                report.tests_failed += 1;
727            }
728
729            report.passed += test_result.passed;
730            report.failed += test_result.failed;
731            report.skipped += test_result.skipped;
732            report.details.extend(test_result.details);
733        }
734
735        let global = self.usage.global_snapshot();
736        if self.emit_run_events {
737            self.emit_event(&TestEvent::RunFinished {
738                tests_passed: report.tests_passed,
739                tests_failed: report.tests_failed,
740                steps_passed: report.passed,
741                steps_failed: report.failed,
742                steps_skipped: report.skipped,
743                total_cost: global.total_cost,
744                total_tokens: global.total_tokens,
745                total_input_tokens: global.total_input_tokens,
746                total_output_tokens: global.total_output_tokens,
747                total_cached_input_tokens: global.total_cached_input_tokens,
748                total_cache_creation_input_tokens: global.total_cache_creation_input_tokens,
749                models: global.models.clone(),
750                total_calls: global.total_calls,
751            });
752        }
753
754        Ok(report)
755    }
756
757    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
758    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
759        let base_url = test
760            .base_url
761            .clone()
762            .or_else(|| self.config.base_url.clone())
763            .unwrap_or_else(crate::base_url);
764
765        // Per-test viewport override: switch the browser via CDP
766        // device-metrics emulation before this test runs.
767        let vw = test.viewport_width.unwrap_or(self.viewport_width);
768        let vh = test.viewport_height.unwrap_or(self.viewport_height);
769        if self.applied_viewport.get() != (vw, vh) {
770            self.apply_viewport(tab, vw, vh);
771            self.applied_viewport.set((vw, vh));
772        }
773
774        // Per-test isolation: every test starts from its own start_url
775        // (unless auto_navigate is disabled), so a test never inherits the
776        // previous test's page state.
777        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
778
779        let start_url = test
780            .start_url
781            .clone()
782            .or_else(|| self.config.start_url.clone())
783            .unwrap_or_else(|| "/dashboard".to_owned());
784
785        // Per-test state isolation for LOGIN tests: the tab is shared
786        // across the file's tests, and a previous test's login persists
787        // (Cognito tokens in localStorage + the hosted-UI cookies). The
788        // SPA's login pages then auto-continue authenticated visitors on
789        // boot, so a later login test never sees the form and times out
790        // waiting for `#email` (observed on immosai runs #1641/#1642: a
791        // shard's first three logins pass, the fourth onward time out).
792        // Clear cookies + origin storage ONLY for tests that perform
793        // their own login (their first navigate step targets a login
794        // route, or they have no navigate step and ride the auto-nav
795        // onto a login start_url). The many "page loads" tests that
796        // navigate app pages directly RELY on the shared session — a
797        // blanket clear bounces them off their target (run #1643: every
798        // shard failed with "the current page is the login page, not
799        // the mailboxes page"). The HTTP cache is deliberately NOT
800        // cleared — re-downloading the SPA bundle per test would only
801        // add boot time on a slow runner egress.
802        if auto_navigate && test_targets_login(&start_url, &test.steps) {
803            use headless_chrome::protocol::cdp::{Network, Storage};
804            if let Err(e) = tab.call_method(Network::ClearBrowserCookies(None)) {
805                self.reporter
806                    .warn(format!("per-test cookie clear failed: {e}"));
807            }
808            if let Some(origin) = origin_of(&base_url) {
809                if let Err(e) = tab.call_method(Storage::ClearDataForOrigin {
810                    origin,
811                    storage_Types: "all".to_string(),
812                }) {
813                    self.reporter
814                        .warn(format!("per-test storage clear failed: {e}"));
815                }
816            }
817        }
818        if auto_navigate {
819            let full_url = resolve_url(&start_url, &base_url);
820            self.reporter.debug(format!("auto-navigate: {full_url}"));
821            let _ = tab.navigate_to(&full_url);
822            let _ = tab.wait_until_navigated();
823            std::thread::sleep(Duration::from_secs(4));
824        }
825
826        let mut result = TestRunResult::default();
827
828        for (step_index, step) in test.steps.iter().enumerate() {
829            result.total += 1;
830
831            let wait_ms = match step {
832                TestStep::Navigate { wait_after_ms, .. }
833                | TestStep::Click { wait_after_ms, .. }
834                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
835                _ => None,
836            };
837
838            self.current_step
839                .replace(Some((test.name.clone(), step_index as u32)));
840            self.emit_event(&TestEvent::StepStarted {
841                test: test.name.clone(),
842                index: step_index as u32,
843                label: step_label(step),
844            });
845            let step_started = Instant::now();
846
847            let mut step_result = match step {
848                TestStep::Navigate { url, .. } => {
849                    let full_url = resolve_url(url, &base_url);
850                    run_navigate_step(&full_url, tab)
851                }
852                TestStep::Click {
853                    target,
854                    selector,
855                    endpoint,
856                    idempotent,
857                    ..
858                } => self.run_click(
859                    target,
860                    selector.as_deref(),
861                    endpoint.as_deref(),
862                    test.endpoint.as_deref(),
863                    *idempotent,
864                    tab,
865                ),
866                TestStep::Type {
867                    target,
868                    text,
869                    selector,
870                    endpoint,
871                    idempotent,
872                    ..
873                } => self.run_type(
874                    target,
875                    text,
876                    selector.as_deref(),
877                    endpoint.as_deref(),
878                    test.endpoint.as_deref(),
879                    *idempotent,
880                    tab,
881                ),
882                TestStep::Wait {
883                    target,
884                    selector,
885                    text,
886                    timeout_ms,
887                    endpoint,
888                    idempotent,
889                } => self.run_wait(
890                    target,
891                    selector.as_deref(),
892                    text.as_deref(),
893                    *timeout_ms,
894                    endpoint.as_deref(),
895                    test.endpoint.as_deref(),
896                    *idempotent,
897                    tab,
898                ),
899                TestStep::Assert {
900                    definition,
901                    preset,
902                    prompt,
903                    assert_text,
904                    endpoint,
905                    screenshot,
906                } => self.run_assert(
907                    definition.as_deref(),
908                    preset.as_deref(),
909                    prompt.as_deref(),
910                    assert_text.as_deref(),
911                    *screenshot,
912                    endpoint.as_deref(),
913                    test.endpoint.as_deref(),
914                    tab,
915                ),
916                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
917                TestStep::Agent {
918                    agent,
919                    task,
920                    definition,
921                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
922                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
923            };
924
925            // Failure diagnostics: capture the page state and a screenshot so
926            // CI logs say WHAT the page looked like when the step failed,
927            // instead of a bare "timed out: The event waited for never came".
928            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
929                let state = diagnostics::capture(tab);
930                let screenshot = diagnostics::save_screenshot(
931                    tab,
932                    &self.artifacts_dir,
933                    &test.name,
934                    &test.name,
935                    step_index,
936                    step_kind_label(step),
937                );
938                step_result.message = format!(
939                    "{base} — {excerpt}",
940                    base = step_result.message,
941                    excerpt = diagnostics::inline_excerpt(&state),
942                );
943                (Some(diagnostics::full_context(&state)), screenshot)
944            } else {
945                (None, None)
946            };
947
948            let duration_ms = step_started.elapsed().as_millis() as u64;
949            self.emit_event(&TestEvent::StepFinished {
950                test: test.name.clone(),
951                index: step_index as u32,
952                label: step_result.name.clone(),
953                status: step_result.status,
954                duration_ms,
955                message: step_result.message.clone(),
956                diagnostics: diagnostics_block,
957                screenshot: screenshot_path,
958            });
959            self.current_step.replace(None);
960
961            match step_result.status {
962                StepStatus::Passed => result.passed += 1,
963                StepStatus::Failed => result.failed += 1,
964                StepStatus::Skipped => result.skipped += 1,
965            }
966
967            // Fail fast: the first failed step ends the test and the
968            // remaining steps are reported as skipped (no LLM budget is
969            // burned asserting against a page that is already known broken).
970            if step_result.status == StepStatus::Failed
971                && !self.config.continue_on_failure
972                && step_index + 1 < test.steps.len()
973            {
974                self.emit_event(&TestEvent::Warning {
975                    message: format!(
976                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
977                        test.steps.len() - step_index - 1
978                    ),
979                });
980                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
981                    let skipped_index = step_index + 1 + offset;
982                    let label = step_label(skipped);
983                    result.total += 1;
984                    result.skipped += 1;
985                    self.emit_event(&TestEvent::StepStarted {
986                        test: test.name.clone(),
987                        index: skipped_index as u32,
988                        label: label.clone(),
989                    });
990                    self.emit_event(&TestEvent::StepFinished {
991                        test: test.name.clone(),
992                        index: skipped_index as u32,
993                        label,
994                        status: StepStatus::Skipped,
995                        duration_ms: 0,
996                        message: "skipped: previous step failed".into(),
997                        diagnostics: None,
998                        screenshot: None,
999                    });
1000                    result.details.push(StepResult {
1001                        name: step_label(skipped),
1002                        status: StepStatus::Skipped,
1003                        message: "skipped: previous step failed".into(),
1004                    });
1005                }
1006                result.details.push(step_result);
1007                return result;
1008            }
1009
1010            // Check per-test budget after each step
1011            let test_usage = self.usage.current_test_snapshot();
1012            let global_usage = self.usage.global_snapshot();
1013            let budget_status = self.budgets.check_all(
1014                &test.name,
1015                &test_usage,
1016                &global_usage,
1017                test.budget.as_ref(),
1018            );
1019            match budget_status {
1020                BudgetStatus::HardExceeded { message, .. } => {
1021                    self.emit_event(&TestEvent::Warning {
1022                        message: format!("budget exceeded: {message}"),
1023                    });
1024                    result.details.push(StepResult {
1025                        name: "[budget]".into(),
1026                        status: StepStatus::Failed,
1027                        message,
1028                    });
1029                    result.failed += 1;
1030                    return result;
1031                }
1032                BudgetStatus::SoftExceeded { message, .. } => {
1033                    self.emit_event(&TestEvent::Warning {
1034                        message: format!("budget warning: {message}"),
1035                    });
1036                }
1037                BudgetStatus::Ok => {}
1038            }
1039
1040            if let Some(ms) = wait_ms {
1041                std::thread::sleep(Duration::from_millis(ms));
1042            }
1043
1044            result.details.push(step_result);
1045        }
1046
1047        result
1048    }
1049
1050    /// Applies a viewport size to the current tab via CDP
1051    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
1052    /// overrides and the viewport matrix. The initial window size set at
1053    /// browser launch is replaced by emulation; failures are logged but
1054    /// do not fail the test (a mismatched viewport only weakens coverage).
1055    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
1056        use headless_chrome::protocol::cdp::Emulation;
1057        let _ = self;
1058        let params = Emulation::SetDeviceMetricsOverride {
1059            width,
1060            height,
1061            device_scale_factor: 1.0,
1062            mobile: false,
1063            scale: None,
1064            screen_width: Some(width),
1065            screen_height: Some(height),
1066            position_x: None,
1067            position_y: None,
1068            dont_set_visible_size: None,
1069            screen_orientation: None,
1070            viewport: None,
1071            display_feature: None,
1072            device_posture: None,
1073        };
1074        self.reporter.debug(format!("viewport: {width}x{height}"));
1075        if let Err(e) = tab.call_method(params) {
1076            self.reporter
1077                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
1078        }
1079    }
1080
1081    /// Height (px) covered by assert-step screenshots: the configured
1082    /// `screenshot_max_height` (absolute px or viewport multiple, default
1083    /// `"20x"`) resolved against the currently applied viewport, raised to
1084    /// at least the viewport height so the visible screen is always fully
1085    /// included. The capture is split into viewport-tall tiles, so this
1086    /// value bounds total coverage (and hence the number of image parts).
1087    #[must_use]
1088    fn screenshot_height_cap(&self) -> u32 {
1089        let viewport_height = self.current_viewport_height();
1090        let cap = self
1091            .config
1092            .screenshot_max_height
1093            .as_ref()
1094            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
1095        cap.max(viewport_height)
1096    }
1097
1098    /// Height of the viewport currently emulated in the browser (falling
1099    /// back to the configured default before any emulation was applied).
1100    #[must_use]
1101    const fn current_viewport_height(&self) -> u32 {
1102        let (_, height) = self.applied_viewport.get();
1103        if height > 0 {
1104            height
1105        } else {
1106            self.viewport_height
1107        }
1108    }
1109
1110    // ── step handlers ───────────────────────────────────────────────────
1111
1112    #[allow(clippy::too_many_lines)]
1113    fn run_click(
1114        &self,
1115        target: &str,
1116        selector_override: Option<&str>,
1117        step_endpoint: Option<&str>,
1118        test_endpoint: Option<&str>,
1119        idempotent: bool,
1120        tab: &Tab,
1121    ) -> StepResult {
1122        let name = format!("[click] {target}");
1123        let selector = match self.resolve_selector(
1124            selector_override,
1125            target,
1126            step_endpoint,
1127            test_endpoint,
1128            tab,
1129        ) {
1130            Ok(s) => s,
1131            Err(msg) => {
1132                if idempotent {
1133                    return StepResult {
1134                        name,
1135                        status: StepStatus::Skipped,
1136                        message: format!("skipped (idempotent): no target found — {msg}"),
1137                    };
1138                }
1139                return StepResult {
1140                    name,
1141                    status: StepStatus::Failed,
1142                    message: msg,
1143                };
1144            }
1145        };
1146
1147        // Idempotent steps probe briefly: a missing target means the
1148        // action was already done / not applicable (e.g. an
1149        // already-authenticated session), and skipping is the success
1150        // path, not a failure.
1151        let probe_secs = if idempotent { 5 } else { 10 };
1152        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1153            Ok(element) => match element.click() {
1154                Ok(_) => StepResult {
1155                    name,
1156                    status: StepStatus::Passed,
1157                    message: format!("clicked {selector}"),
1158                },
1159                Err(e) => StepResult {
1160                    name,
1161                    status: StepStatus::Failed,
1162                    message: format!("click failed on {selector}: {e}"),
1163                },
1164            },
1165            Err(e) if idempotent => StepResult {
1166                name,
1167                status: StepStatus::Skipped,
1168                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1169            },
1170            Err(e) => StepResult {
1171                name,
1172                status: StepStatus::Failed,
1173                message: format!("element {selector} not found: {e}"),
1174            },
1175        }
1176    }
1177
1178    #[allow(clippy::too_many_arguments)]
1179    fn run_type(
1180        &self,
1181        target: &str,
1182        text: &str,
1183        selector_override: Option<&str>,
1184        step_endpoint: Option<&str>,
1185        test_endpoint: Option<&str>,
1186        idempotent: bool,
1187        tab: &Tab,
1188    ) -> StepResult {
1189        let name = format!("[type] {target}");
1190        let selector = match self.resolve_selector(
1191            selector_override,
1192            target,
1193            step_endpoint,
1194            test_endpoint,
1195            tab,
1196        ) {
1197            Ok(s) => s,
1198            Err(msg) => {
1199                if idempotent {
1200                    return StepResult {
1201                        name,
1202                        status: StepStatus::Skipped,
1203                        message: format!("skipped (idempotent): no target found — {msg}"),
1204                    };
1205                }
1206                return StepResult {
1207                    name,
1208                    status: StepStatus::Failed,
1209                    message: msg,
1210                };
1211            }
1212        };
1213
1214        let probe_secs = if idempotent { 5 } else { 10 };
1215        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1216            Ok(element) => {
1217                if let Err(e) = element.click() {
1218                    return StepResult {
1219                        name,
1220                        status: StepStatus::Failed,
1221                        message: format!("click to focus {selector} failed: {e}"),
1222                    };
1223                }
1224
1225                let js = format!(
1226                    "document.querySelector('{}').value = '';",
1227                    selector.replace('\'', "\\'")
1228                );
1229                let _ = tab.evaluate(&js, false);
1230
1231                match element.type_into(text) {
1232                    Ok(_) => StepResult {
1233                        name,
1234                        status: StepStatus::Passed,
1235                        message: format!("typed {text:?} into {selector}"),
1236                    },
1237                    Err(e) => StepResult {
1238                        name,
1239                        status: StepStatus::Failed,
1240                        message: format!("type into {selector} failed: {e}"),
1241                    },
1242                }
1243            }
1244            Err(e) if idempotent => StepResult {
1245                name,
1246                status: StepStatus::Skipped,
1247                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1248            },
1249            Err(e) => StepResult {
1250                name,
1251                status: StepStatus::Failed,
1252                message: format!("element {selector} not found: {e}"),
1253            },
1254        }
1255    }
1256
1257    #[allow(clippy::too_many_arguments)]
1258    #[allow(clippy::too_many_lines)]
1259    fn run_wait(
1260        &self,
1261        target: &str,
1262        selector_override: Option<&str>,
1263        text: Option<&str>,
1264        timeout_ms: Option<u64>,
1265        step_endpoint: Option<&str>,
1266        test_endpoint: Option<&str>,
1267        idempotent: bool,
1268        tab: &Tab,
1269    ) -> StepResult {
1270        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1271        let step_name = format!("[wait] {target}");
1272
1273        // Resolve an explicit selector only (text-only waits are LLM-free).
1274        let selector = match selector_override {
1275            Some(s) => Some(s.to_owned()),
1276            None if text.is_some() => None,
1277            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1278                Ok(s) => Some(s),
1279                Err(msg) => {
1280                    if idempotent {
1281                        return StepResult {
1282                            name: step_name,
1283                            status: StepStatus::Skipped,
1284                            message: format!("skipped (idempotent): no target found — {msg}"),
1285                        };
1286                    }
1287                    return StepResult {
1288                        name: step_name,
1289                        status: StepStatus::Failed,
1290                        message: msg,
1291                    };
1292                }
1293            },
1294        };
1295
1296        if text.is_some() {
1297            let sel_js = selector
1298                .as_deref()
1299                .map(crate::selectors::selector_matches_js);
1300            let text_js = text.map(|t| {
1301                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1302                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1303            });
1304
1305            let mut reload_note = String::new();
1306            let mut reloaded = 0u32;
1307            loop {
1308                let deadline = Instant::now() + timeout;
1309                loop {
1310                    let sel_ok = sel_js
1311                        .as_ref()
1312                        .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1313                    let text_ok = text_js
1314                        .as_ref()
1315                        .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1316                    if sel_ok && text_ok {
1317                        let mut what = Vec::new();
1318                        if let Some(sel) = &selector {
1319                            what.push(format!("found {sel}"));
1320                        }
1321                        if let Some(t) = text {
1322                            what.push(format!("text {t:?} visible"));
1323                        }
1324                        return StepResult {
1325                            name: step_name,
1326                            status: StepStatus::Passed,
1327                            message: format!("{}{reload_note}", what.join(" and ")),
1328                        };
1329                    }
1330                    if Instant::now() >= deadline {
1331                        break;
1332                    }
1333                    std::thread::sleep(Duration::from_millis(250));
1334                }
1335                // Blank-page self-heal: a stalled runner egress leaves the
1336                // SPA unbooted (index.html loaded, JS chunks never arrived,
1337                // immosai runs #1646/#1647 — every failure screenshot is an
1338                // empty viewport). A wait against that page can never
1339                // succeed; reload and re-wait with the full budget. The
1340                // runner's egress/resource bursts last minutes (v0.19.5's
1341                // single reload was not enough when the burst outlived two
1342                // wait budgets — immosai run #1651), so allow up to
1343                // MAX_BLANK_HEALS heal cycles with a backoff sleep between
1344                // them.
1345                if reloaded < MAX_BLANK_HEALS && page_is_blank(tab) {
1346                    reloaded += 1;
1347                    if reloaded > 1 {
1348                        std::thread::sleep(Duration::from_secs(BLANK_HEAL_BACKOFF_SECS));
1349                    }
1350                    reload_note = format!(" (blank page — reloaded {reloaded}x and re-waited)");
1351                    let _ = tab.reload(true, None);
1352                    let _ = tab.wait_until_navigated();
1353                    std::thread::sleep(Duration::from_secs(2));
1354                    continue;
1355                }
1356                let mut what = Vec::new();
1357                if let Some(sel) = &selector {
1358                    what.push(sel.clone());
1359                }
1360                if let Some(t) = text {
1361                    what.push(format!("text {t:?}"));
1362                }
1363                let message = format!(
1364                    "wait for {} timed out after {}ms: the event waited for never came{reload_note}",
1365                    what.join(" / "),
1366                    timeout.as_millis(),
1367                );
1368                if idempotent {
1369                    return StepResult {
1370                        name: step_name,
1371                        status: StepStatus::Skipped,
1372                        message: format!("skipped (idempotent): {message}"),
1373                    };
1374                }
1375                return StepResult {
1376                    name: step_name,
1377                    status: StepStatus::Failed,
1378                    message,
1379                };
1380            }
1381        }
1382
1383        let mut reload_note = String::new();
1384        let mut reloaded = 0u32;
1385        let result = loop {
1386            match selector.as_deref() {
1387                Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1388                    Ok(_) => break Ok((sel.to_owned(), reload_note)),
1389                    Err(e) => {
1390                        // Blank-page self-heal (see the text-wait branch):
1391                        // an unbooted SPA never renders the target; reload
1392                        // and re-wait with the full budget, with a backoff
1393                        // between heal cycles for the runner's multi-minute
1394                        // egress/resource bursts (immosai run #1651).
1395                        if reloaded < MAX_BLANK_HEALS && page_is_blank(tab) {
1396                            reloaded += 1;
1397                            if reloaded > 1 {
1398                                std::thread::sleep(Duration::from_secs(BLANK_HEAL_BACKOFF_SECS));
1399                            }
1400                            format!(" (blank page — reloaded {reloaded}x and re-waited)")
1401                                .clone_into(&mut reload_note);
1402                            let _ = tab.reload(true, None);
1403                            let _ = tab.wait_until_navigated();
1404                            std::thread::sleep(Duration::from_secs(2));
1405                            continue;
1406                        }
1407                        if idempotent {
1408                            break Err((
1409                                format!(
1410                                    "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1411                                    timeout.as_millis()
1412                                ),
1413                                StepStatus::Skipped,
1414                            ));
1415                        }
1416                        break Err((
1417                            format!(
1418                                "wait for {sel} timed out after {}ms: {e}{reload_note}",
1419                                timeout.as_millis()
1420                            ),
1421                            StepStatus::Failed,
1422                        ));
1423                    }
1424                },
1425                None => {
1426                    break Err((
1427                        "wait step has neither selector nor text".to_owned(),
1428                        StepStatus::Failed,
1429                    ))
1430                }
1431            }
1432        };
1433        match result {
1434            Ok((sel, note)) => StepResult {
1435                name: step_name,
1436                status: StepStatus::Passed,
1437                message: format!("found {sel}{note}"),
1438            },
1439            Err((message, status)) => StepResult {
1440                name: step_name,
1441                status,
1442                message,
1443            },
1444        }
1445    }
1446
1447    #[allow(clippy::too_many_arguments)]
1448    fn run_assert(
1449        &self,
1450        definition: Option<&str>,
1451        preset: Option<&str>,
1452        prompt: Option<&str>,
1453        assert_text: Option<&str>,
1454        screenshot: bool,
1455        step_endpoint: Option<&str>,
1456        test_endpoint: Option<&str>,
1457        tab: &Tab,
1458    ) -> StepResult {
1459        std::thread::sleep(Duration::from_millis(500));
1460
1461        let page_content = get_page_text(tab);
1462
1463        // Vision attach: capture the full page once per assert step and
1464        // split it into viewport-tall tiles (the total coverage is bounded
1465        // by the configured height cap so vision tokens stay sane). All
1466        // tile data URLs are handed to the preset/prompt evaluation below.
1467        let image: Option<Vec<String>> = if screenshot {
1468            let endpoint = self
1469                .endpoints
1470                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1471            if !endpoint.vision {
1472                return StepResult {
1473                    name: "[assert]".into(),
1474                    status: StepStatus::Failed,
1475                    message: format!(
1476                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1477                        name = endpoint.name
1478                    ),
1479                };
1480            }
1481            match crate::vision::capture_screenshot_data_urls(
1482                tab,
1483                self.config
1484                    .screenshot_max_dimension
1485                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1486                self.screenshot_height_cap(),
1487                self.current_viewport_height(),
1488            ) {
1489                Ok(urls) => Some(urls),
1490                Err(e) => {
1491                    return StepResult {
1492                        name: "[assert]".into(),
1493                        status: StepStatus::Failed,
1494                        message: format!("screenshot capture failed: {e}"),
1495                    };
1496                }
1497            }
1498        } else {
1499            None
1500        };
1501
1502        if let Some(def_name) = definition {
1503            if let Some(def) = self.definitions.get(def_name) {
1504                return self.run_assert_def(
1505                    def,
1506                    &page_content,
1507                    image.as_deref(),
1508                    step_endpoint,
1509                    test_endpoint,
1510                    tab,
1511                );
1512            }
1513            return StepResult {
1514                name: format!("[assert] {def_name}"),
1515                status: StepStatus::Failed,
1516                message: format!("definition '{def_name}' not found"),
1517            };
1518        }
1519
1520        if let Some(preset_name) = preset {
1521            // Deterministic DOM layout scan — runs JS in the browser and
1522            // never calls the LLM (free, fast, no pixel budget).
1523            if preset_name == "layout_no_issues" {
1524                return self.run_layout_preset(tab);
1525            }
1526            return self.run_preset(
1527                preset_name,
1528                assert_text,
1529                &page_content,
1530                image.as_deref(),
1531                step_endpoint,
1532                test_endpoint,
1533            );
1534        }
1535
1536        if let Some(prompt_text) = prompt {
1537            return self.run_custom(
1538                prompt_text,
1539                &page_content,
1540                image.as_deref(),
1541                step_endpoint,
1542                test_endpoint,
1543            );
1544        }
1545
1546        StepResult {
1547            name: "[assert]".into(),
1548            status: StepStatus::Skipped,
1549            message: "no definition, preset, or prompt specified".into(),
1550        }
1551    }
1552
1553    fn run_assert_def(
1554        &self,
1555        def: &AssertDefinition,
1556        page_content: &PageContent,
1557        image: Option<&[String]>,
1558        step_endpoint: Option<&str>,
1559        test_endpoint: Option<&str>,
1560        tab: &Tab,
1561    ) -> StepResult {
1562        // Agent-based definition: delegate to an A2A agent
1563        if let Some(ref agent) = def.agent {
1564            if image.is_some() {
1565                return StepResult {
1566                    name: format!("[assert] {}", def.name),
1567                    status: StepStatus::Failed,
1568                    message: "agent-backed assertions do not support screenshots".into(),
1569                };
1570            }
1571            let task = def
1572                .task_template
1573                .as_deref()
1574                .unwrap_or("Evaluate the assertion")
1575                .replace("{url}", &page_content.url)
1576                .replace("{title}", &page_content.title)
1577                .replace("{content}", &page_content.body_text)
1578                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1579
1580            return self.run_agent_step(agent, &task, &def.name);
1581        }
1582
1583        // Custom preset: system + user_template provided in the definition
1584        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1585            return self.run_custom_preset(
1586                &def.name,
1587                system,
1588                template,
1589                def.assert_text.as_deref(),
1590                page_content,
1591                image,
1592                step_endpoint,
1593                test_endpoint,
1594            );
1595        }
1596
1597        def.preset.as_ref().map_or_else(
1598            || {
1599                def.prompt.as_ref().map_or_else(
1600                    || StepResult {
1601                        name: format!("[assert] {}", def.name),
1602                        status: StepStatus::Failed,
1603                        message: "definition has no preset, prompt, or system+user_template".into(),
1604                    },
1605                    |prompt| {
1606                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1607                    },
1608                )
1609            },
1610            |preset_name| {
1611                if preset_name == "layout_no_issues" {
1612                    return self.run_layout_preset(tab);
1613                }
1614                self.run_preset(
1615                    preset_name,
1616                    def.assert_text.as_deref(),
1617                    page_content,
1618                    image,
1619                    step_endpoint,
1620                    test_endpoint,
1621                )
1622            },
1623        )
1624    }
1625
1626    #[allow(clippy::too_many_arguments)]
1627    fn run_custom_preset(
1628        &self,
1629        name: &str,
1630        system: &str,
1631        template: &str,
1632        assert_text: Option<&str>,
1633        page_content: &PageContent,
1634        image: Option<&[String]>,
1635        step_endpoint: Option<&str>,
1636        test_endpoint: Option<&str>,
1637    ) -> StepResult {
1638        let user_prompt = template
1639            .replace("{url}", &page_content.url)
1640            .replace("{title}", &page_content.title)
1641            .replace("{content}", &page_content.body_text)
1642            .replace("{expected_text}", assert_text.unwrap_or(""))
1643            .replace("{description}", "");
1644
1645        // Custom preset definitions frequently forget the {content}
1646        // placeholder — without it the LLM has no page to evaluate and
1647        // answers "I can't determine that without seeing the page". Always
1648        // append the page context unless the template already references it.
1649        let user_prompt = if template.contains("{content}") {
1650            user_prompt
1651        } else {
1652            format!(
1653                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1654                url = page_content.url,
1655                title = page_content.title,
1656                content = page_content.body_text,
1657            )
1658        };
1659
1660        self.reporter
1661            .debug(format!("assert: {name} (custom preset)"));
1662
1663        let chain = self
1664            .endpoints
1665            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1666        let sys = system.to_owned();
1667
1668        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1669
1670        response.map_or_else(
1671            |e| StepResult {
1672                name: format!("[assert] {name}"),
1673                status: StepStatus::Failed,
1674                message: format!("LLM assertion call failed: {e}"),
1675            },
1676            |(lr, _idx)| {
1677                if verdict_is_pass(&lr.content) {
1678                    StepResult {
1679                        name: format!("[assert] {name}"),
1680                        status: StepStatus::Passed,
1681                        message: "PASS".into(),
1682                    }
1683                } else {
1684                    StepResult {
1685                        name: format!("[assert] {name}"),
1686                        status: StepStatus::Failed,
1687                        message: lr.content,
1688                    }
1689                }
1690            },
1691        )
1692    }
1693
1694    fn run_preset(
1695        &self,
1696        preset_name: &str,
1697        assert_text: Option<&str>,
1698        page_content: &PageContent,
1699        image: Option<&[String]>,
1700        step_endpoint: Option<&str>,
1701        test_endpoint: Option<&str>,
1702    ) -> StepResult {
1703        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1704            return StepResult {
1705                name: format!("[assert] {preset_name}"),
1706                status: StepStatus::Failed,
1707                message: format!("unknown assertion preset: {preset_name}"),
1708            };
1709        };
1710        if preset_name.starts_with("visual_") && image.is_none() {
1711            return StepResult {
1712                name: format!("[assert] {preset_name}"),
1713                status: StepStatus::Failed,
1714                message: format!(
1715                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1716                ),
1717            };
1718        }
1719
1720        let user_prompt = preset
1721            .user_template
1722            .replace("{url}", &page_content.url)
1723            .replace("{title}", &page_content.title)
1724            .replace("{content}", &page_content.body_text)
1725            .replace("{expected_text}", assert_text.unwrap_or(""))
1726            .replace("{description}", "");
1727
1728        // Same safety net as custom presets: never let the LLM answer with
1729        // no page context at all.
1730        let user_prompt = if preset.user_template.contains("{content}") {
1731            user_prompt
1732        } else {
1733            format!(
1734                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1735                url = page_content.url,
1736                title = page_content.title,
1737                content = page_content.body_text,
1738            )
1739        };
1740
1741        self.reporter.debug(format!("assert: {preset_name}"));
1742
1743        let chain = self
1744            .endpoints
1745            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1746        let sys = preset.system.to_owned();
1747
1748        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1749
1750        response.map_or_else(
1751            |e| StepResult {
1752                name: format!("[assert] {preset_name}"),
1753                status: StepStatus::Failed,
1754                message: format!("LLM assertion call failed: {e}"),
1755            },
1756            |(lr, _idx)| {
1757                if verdict_is_pass(&lr.content) {
1758                    StepResult {
1759                        name: format!("[assert] {preset_name}"),
1760                        status: StepStatus::Passed,
1761                        message: "PASS".into(),
1762                    }
1763                } else {
1764                    StepResult {
1765                        name: format!("[assert] {preset_name}"),
1766                        status: StepStatus::Failed,
1767                        message: lr.content,
1768                    }
1769                }
1770            },
1771        )
1772    }
1773
1774    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1775    ///
1776    /// Evaluates the layout-scan JS in the page and fails with the list of
1777    /// detected issues: horizontal page overflow, visible elements sticking
1778    /// out of the viewport, text clipped by `overflow: hidden` containers,
1779    /// and interactive elements covered by other elements. No LLM call —
1780    /// checks are geometry-based so the check is free, deterministic, and
1781    /// safe to run on every page × viewport variant.
1782    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1783        let name = "[assert] layout_no_issues".to_owned();
1784        self.reporter
1785            .debug("assert: layout_no_issues (DOM layout scan)");
1786        let js = LAYOUT_SCAN_JS.replace(
1787            "__IGNORE_CLASSES__",
1788            &serde_json::to_string(&self.config.layout_ignore_classes)
1789                .unwrap_or_else(|_| "[]".to_owned()),
1790        );
1791        let result = tab.evaluate(&js, false);
1792        let json_str = match result {
1793            Ok(r) => r
1794                .value
1795                .as_ref()
1796                .and_then(|v| v.as_str().map(String::from))
1797                .unwrap_or_else(|| "[]".to_owned()),
1798            Err(e) => {
1799                return StepResult {
1800                    name,
1801                    status: StepStatus::Failed,
1802                    message: format!("layout scan JS failed: {e}"),
1803                };
1804            }
1805        };
1806        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1807        if issues.is_empty() {
1808            return StepResult {
1809                name,
1810                status: StepStatus::Passed,
1811                message: "PASS — no layout defects detected".into(),
1812            };
1813        }
1814        let mut lines: Vec<String> = issues
1815            .iter()
1816            .take(10)
1817            .map(|i| {
1818                format!(
1819                    "- [{type_}] {element}: {detail}",
1820                    type_ = i.issue_type,
1821                    element = i.element,
1822                    detail = i.detail
1823                )
1824            })
1825            .collect();
1826        if issues.len() > 10 {
1827            lines.push(format!("- … and {} more", issues.len() - 10));
1828        }
1829        StepResult {
1830            name,
1831            status: StepStatus::Failed,
1832            message: format!(
1833                "FAIL — {} layout defect(s) detected:\n{}",
1834                issues.len(),
1835                lines.join("\n")
1836            ),
1837        }
1838    }
1839
1840    fn run_custom(
1841        &self,
1842        prompt: &str,
1843        page_content: &PageContent,
1844        image: Option<&[String]>,
1845        step_endpoint: Option<&str>,
1846        test_endpoint: Option<&str>,
1847    ) -> StepResult {
1848        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1849
1850        let mut user = format!(
1851            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1852            url = page_content.url,
1853            title = page_content.title,
1854            content = page_content.body_text,
1855        );
1856        if image.is_some() {
1857            user.push_str(
1858                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1859            );
1860        }
1861
1862        self.reporter.debug("custom assert");
1863
1864        let chain = self
1865            .endpoints
1866            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1867        let sys = system.to_owned();
1868
1869        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1870
1871        response.map_or_else(
1872            |e| StepResult {
1873                name: "[assert] custom".into(),
1874                status: StepStatus::Failed,
1875                message: format!("LLM assertion call failed: {e}"),
1876            },
1877            |(lr, _idx)| {
1878                if verdict_is_pass(&lr.content) {
1879                    StepResult {
1880                        name: "[assert] custom".into(),
1881                        status: StepStatus::Passed,
1882                        message: "PASS".into(),
1883                    }
1884                } else {
1885                    StepResult {
1886                        name: "[assert] custom".into(),
1887                        status: StepStatus::Failed,
1888                        message: lr.content,
1889                    }
1890                }
1891            },
1892        )
1893    }
1894
1895    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1896        let path = path.unwrap_or("screenshot.png");
1897
1898        match tab.capture_screenshot(
1899            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1900            None,
1901            None,
1902            true,
1903        ) {
1904            Ok(data) => {
1905                if let Err(e) = std::fs::write(path, &data) {
1906                    return StepResult {
1907                        name: format!("[screenshot] {path}"),
1908                        status: StepStatus::Failed,
1909                        message: format!("failed to write screenshot: {e}"),
1910                    };
1911                }
1912                StepResult {
1913                    name: format!("[screenshot] {path}"),
1914                    status: StepStatus::Passed,
1915                    message: format!("saved to {path}"),
1916                }
1917            }
1918            Err(e) => StepResult {
1919                name: format!("[screenshot] {path}"),
1920                status: StepStatus::Failed,
1921                message: format!("screenshot failed: {e}"),
1922            },
1923        }
1924    }
1925
1926    /// Runs an A2A agent step.
1927    #[allow(clippy::literal_string_with_formatting_args)]
1928    fn run_agent(
1929        &self,
1930        agent_name: &str,
1931        task: &str,
1932        definition: Option<&str>,
1933        _test_endpoint: Option<&str>,
1934    ) -> StepResult {
1935        // If a definition is specified, look up the task template
1936        let resolved_task = if let Some(def_name) = definition {
1937            if let Some(def) = self.definitions.get(def_name) {
1938                let tmpl = def.task_template.as_deref().unwrap_or(task);
1939                tmpl.replace("{task}", task)
1940            } else {
1941                return StepResult {
1942                    name: format!("[agent] {def_name}"),
1943                    status: StepStatus::Failed,
1944                    message: format!("definition '{def_name}' not found"),
1945                };
1946            }
1947        } else {
1948            task.to_owned()
1949        };
1950
1951        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1952    }
1953
1954    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1955        let Some(ep) = self.endpoints.get(agent_name) else {
1956            return StepResult {
1957                name: format!("[agent] {display_name}"),
1958                status: StepStatus::Failed,
1959                message: format!("agent endpoint '{agent_name}' not found"),
1960            };
1961        };
1962
1963        if ep.url.is_empty() {
1964            return StepResult {
1965                name: format!("[agent] {display_name}"),
1966                status: StepStatus::Failed,
1967                message: format!("agent endpoint '{agent_name}' has no URL"),
1968            };
1969        }
1970
1971        self.reporter.debug(format!("agent {agent_name}: {task}"));
1972
1973        let url = ep.url.clone();
1974        let client = A2aClient::new(&url, self.timeout);
1975        let task_clone = task.to_owned();
1976
1977        let response = std::thread::spawn(move || {
1978            let rt = tokio::runtime::Builder::new_current_thread()
1979                .enable_all()
1980                .build()
1981                .unwrap();
1982            rt.block_on(client.send_task(&task_clone))
1983        })
1984        .join()
1985        .unwrap();
1986
1987        // Record the flat-cost call
1988        self.usage.record_flat_call(agent_name, ep);
1989
1990        match response {
1991            Ok(text) => {
1992                let clean = text.trim().to_owned();
1993                if verdict_is_pass(&clean) {
1994                    StepResult {
1995                        name: format!("[agent] {display_name}"),
1996                        status: StepStatus::Passed,
1997                        message: format!("PASS: {clean}"),
1998                    }
1999                } else if verdict_is_fail(&clean) {
2000                    StepResult {
2001                        name: format!("[agent] {display_name}"),
2002                        status: StepStatus::Failed,
2003                        message: clean,
2004                    }
2005                } else {
2006                    StepResult {
2007                        name: format!("[agent] {display_name}"),
2008                        status: StepStatus::Passed,
2009                        message: format!("response: {clean}"),
2010                    }
2011                }
2012            }
2013            Err(e) => StepResult {
2014                name: format!("[agent] {display_name}"),
2015                status: StepStatus::Failed,
2016                message: format!("agent call failed: {e}"),
2017            },
2018        }
2019    }
2020
2021    /// Runs an MCP tool call step.
2022    fn run_mcp(
2023        &self,
2024        server_name: &str,
2025        tool_name: &str,
2026        args: Option<&serde_json::Value>,
2027    ) -> StepResult {
2028        let Some(ep) = self.endpoints.get(server_name) else {
2029            return StepResult {
2030                name: format!("[mcp] {server_name}:{tool_name}"),
2031                status: StepStatus::Failed,
2032                message: format!("MCP server endpoint '{server_name}' not found"),
2033            };
2034        };
2035
2036        let cmd = ep.command.as_deref().unwrap_or("");
2037        if cmd.is_empty() {
2038            return StepResult {
2039                name: format!("[mcp] {server_name}:{tool_name}"),
2040                status: StepStatus::Failed,
2041                message: format!("MCP server '{server_name}' has no command configured"),
2042            };
2043        }
2044
2045        self.reporter
2046            .debug(format!("mcp {server_name} {tool_name}"));
2047
2048        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
2049
2050        let command = cmd.to_owned();
2051        let args_vec = ep.args.clone();
2052        let tool = tool_name.to_owned();
2053
2054        let response = std::thread::spawn(move || {
2055            let mut mcp_client =
2056                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
2057            mcp_client
2058                .call_tool(&tool, &args_val)
2059                .map_err(|e| e.to_string())
2060        })
2061        .join()
2062        .unwrap();
2063
2064        // Record the flat-cost call
2065        self.usage.record_flat_call(server_name, ep);
2066
2067        match response {
2068            Ok(result) => {
2069                if result.isError {
2070                    StepResult {
2071                        name: format!("[mcp] {server_name}:{tool_name}"),
2072                        status: StepStatus::Failed,
2073                        message: result.to_string(),
2074                    }
2075                } else {
2076                    StepResult {
2077                        name: format!("[mcp] {server_name}:{tool_name}"),
2078                        status: StepStatus::Passed,
2079                        message: result.to_string(),
2080                    }
2081                }
2082            }
2083            Err(e) => StepResult {
2084                name: format!("[mcp] {server_name}:{tool_name}"),
2085                status: StepStatus::Failed,
2086                message: format!("MCP call failed: {e}"),
2087            },
2088        }
2089    }
2090
2091    // ── helpers ──────────────────────────────────────────────────────────
2092
2093    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
2094    /// the runner's default LLM config for any unset fields.
2095    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
2096        LlmConfig {
2097            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
2098                // Bedrock builds its endpoint from the resolved AWS region
2099                // when no URL is given — never inherit the default LLM URL.
2100                endpoint.url.clone()
2101            } else if endpoint.url.is_empty() {
2102                self.llm.url.clone()
2103            } else {
2104                endpoint.url.clone()
2105            },
2106            model: endpoint
2107                .model
2108                .clone()
2109                .unwrap_or_else(|| self.llm.model.clone()),
2110            api_key: endpoint
2111                .api_key
2112                .clone()
2113                .or_else(|| self.llm.api_key.clone()),
2114            headers: if endpoint.headers.is_empty() {
2115                self.llm.headers.clone()
2116            } else {
2117                endpoint.headers.clone()
2118            },
2119            timeout: self.llm.timeout,
2120            temperature: self.llm.temperature,
2121            thinking: self.llm.thinking,
2122            model_params: self.llm.model_params.clone(),
2123            cache: endpoint.cache_markers,
2124            max_attempts: endpoint.max_attempts.max(1),
2125            provider: endpoint.provider,
2126            deployment: endpoint.deployment.clone(),
2127            api_version: endpoint.api_version.clone(),
2128            auth: endpoint.auth.clone(),
2129            header_commands: endpoint.header_commands.clone(),
2130            aws: endpoint.aws.clone(),
2131        }
2132    }
2133
2134    /// Runs a single LLM call against an ordered endpoint chain (primary +
2135    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
2136    /// the first endpoint that answers wins. Returns the response together
2137    /// with the chain index of the answering endpoint (0 = primary) so the
2138    /// caller can attribute usage to the correct endpoint.
2139    ///
2140    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
2141    /// shows duration, tokens, cost and the answering endpoint per call.
2142    /// Run-level context handed to every LLM call as the FIRST block of the
2143    /// user message. Contents that are stable for the whole run ("run
2144    /// started", "target site") come first so upstream provider prefix
2145    /// caching stays effective; the current time is the last line because
2146    /// it changes on every call.
2147    fn run_context(&self) -> String {
2148        let mut parts = vec![
2149            "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
2150            "================================================================".into(),
2151            format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
2152        ];
2153        if let Some(base) = self.config.base_url.as_deref() {
2154            parts.push(format!("Target site: {base}"));
2155        }
2156        let now = SystemTime::now()
2157            .duration_since(UNIX_EPOCH)
2158            .map_or(0, |d| d.as_secs());
2159        parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
2160        parts.join("\n")
2161    }
2162
2163    #[allow(clippy::cast_possible_truncation, clippy::too_many_lines)]
2164    fn llm_call_chain(
2165        &self,
2166        chain: &[&ResolvedEndpoint],
2167        system: &str,
2168        user: &str,
2169        image: Option<&[String]>,
2170        purpose: &str,
2171    ) -> Result<(crate::costs::LlmResponse, usize), String> {
2172        if chain.is_empty() {
2173            return Err("empty LLM endpoint chain".into());
2174        }
2175        let primary = self.build_llm_for_endpoint(chain[0]);
2176        let fallbacks: Vec<LlmConfig> = chain[1..]
2177            .iter()
2178            .map(|e| self.build_llm_for_endpoint(e))
2179            .collect();
2180
2181        let (test, index) = self
2182            .current_step
2183            .borrow()
2184            .as_ref()
2185            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
2186        let primary_endpoint = chain[0].name.clone();
2187        let primary_model = primary.model.clone();
2188        self.emit_event(&TestEvent::LlmCallStarted {
2189            test: test.clone(),
2190            index,
2191            endpoint: primary_endpoint.clone(),
2192            model: primary_model.clone(),
2193            purpose: purpose.to_owned(),
2194        });
2195
2196        let started = Instant::now();
2197        let sys = system.to_owned();
2198        let context = self.run_context();
2199        let user = if context.is_empty() {
2200            user.to_owned()
2201        } else {
2202            format!("{context}\n\n{user}")
2203        };
2204        let image = image.map(<[String]>::to_vec);
2205
2206        let result = std::thread::spawn(move || {
2207            let rt = tokio::runtime::Builder::new_current_thread()
2208                .enable_all()
2209                .build()
2210                .unwrap();
2211            let call = async {
2212                match image.as_deref() {
2213                    Some(img) => {
2214                        llm_chat_vision_with_usage_chain(
2215                            &primary,
2216                            &fallbacks,
2217                            &sys,
2218                            &user,
2219                            Some(img),
2220                        )
2221                        .await
2222                    }
2223                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2224                }
2225            };
2226            rt.block_on(call)
2227        })
2228        .join()
2229        .unwrap();
2230
2231        let duration_ms = started.elapsed().as_millis() as u64;
2232        match result {
2233            Ok((lr, idx)) => {
2234                let cost = calculate_llm_cost(chain[idx], &lr.usage);
2235                let answering = chain[idx].name.clone();
2236                let model = chain[idx]
2237                    .model
2238                    .clone()
2239                    .unwrap_or_else(|| primary_model.clone());
2240                self.usage
2241                    .record_llm_call(&answering, chain[idx], &model, &lr.usage);
2242                self.emit_event(&TestEvent::LlmCallFinished {
2243                    test,
2244                    index,
2245                    endpoint: answering,
2246                    model,
2247                    purpose: purpose.to_owned(),
2248                    ok: true,
2249                    duration_ms,
2250                    input_tokens: lr.usage.prompt_tokens,
2251                    output_tokens: lr.usage.completion_tokens,
2252                    cached_input_tokens: lr.usage.cached_input_tokens,
2253                    cache_creation_input_tokens: lr.usage.cache_creation_input_tokens,
2254                    cost,
2255                    error: None,
2256                });
2257                Ok((lr, idx))
2258            }
2259            Err(e) => {
2260                self.emit_event(&TestEvent::LlmCallFinished {
2261                    test,
2262                    index,
2263                    endpoint: primary_endpoint,
2264                    model: primary_model,
2265                    purpose: purpose.to_owned(),
2266                    ok: false,
2267                    duration_ms,
2268                    input_tokens: 0,
2269                    output_tokens: 0,
2270                    cached_input_tokens: 0,
2271                    cache_creation_input_tokens: 0,
2272                    cost: 0.0,
2273                    error: Some(e.clone()),
2274                });
2275                Err(e)
2276            }
2277        }
2278    }
2279
2280    /// Resolves a CSS selector for the target element. Uses the explicit
2281    /// `selector` if provided, otherwise asks the LLM to find the element
2282    /// from the natural language `target` description and page DOM.
2283    ///
2284    /// LLM responses are sanitized and verified against the live page: a
2285    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2286    /// immediately with the raw LLM output, and a selector that matches
2287    /// nothing triggers one retry with feedback before failing.
2288    #[allow(clippy::too_many_lines)]
2289    fn resolve_selector(
2290        &self,
2291        css_override: Option<&str>,
2292        target: &str,
2293        step_endpoint: Option<&str>,
2294        test_endpoint: Option<&str>,
2295        tab: &Tab,
2296    ) -> Result<String, String> {
2297        if let Some(explicit) = css_override {
2298            return Ok(explicit.to_owned());
2299        }
2300
2301        let dom_info = extract_dom_info(tab)?;
2302        let page_content = get_page_text(tab);
2303
2304        let system = concat!(
2305            "You are a browser automation selector generator. ",
2306            "Given a web page's content and interactive elements, ",
2307            "return ONLY the best CSS selector for the described element. ",
2308            "Output nothing except the CSS selector. ",
2309            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2310            "[name=\"...\"], tag.class, tag. ",
2311            "Never output explanations, markdown, or extra text."
2312        );
2313
2314        let user = format!(
2315            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2316            page_content.url,
2317            page_content.title,
2318            truncate(&page_content.body_text, 4000),
2319            dom_info,
2320            target,
2321        );
2322
2323        let retry_user = format!(
2324            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2325            "The selector must match at least one element currently present on the page.",
2326            page_content.url,
2327            page_content.title,
2328            truncate(&page_content.body_text, 4000),
2329            dom_info,
2330            target,
2331        );
2332
2333        self.reporter.debug(format!("LLM targeting: {target}"));
2334
2335        let chain = self
2336            .endpoints
2337            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2338        let sys = system.to_owned();
2339
2340        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2341
2342        let first = call_llm(&user);
2343        let (lr, _idx) = match first {
2344            Ok(lr) => lr,
2345            Err(e) => {
2346                return Err(format!("LLM element targeting failed: {e}"));
2347            }
2348        };
2349        let clean = sanitize_selector(&lr.content);
2350        self.reporter.debug(format!("resolved selector: {clean}"));
2351
2352        if selector_is_useless(&clean) {
2353            return Err(format!(
2354                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2355                raw = lr.content.trim(),
2356            ));
2357        }
2358        if let Err(reason) = validate_selector(&clean) {
2359            return Err(format!(
2360                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2361                raw = lr.content.trim(),
2362            ));
2363        }
2364        if !selector_matches(tab, &clean).unwrap_or(false) {
2365            // One retry with feedback: flaky models occasionally invent a
2366            // selector that does not exist on the page.
2367            self.reporter.warn(format!(
2368                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2369            ));
2370            let second = call_llm(&retry_user);
2371            let (lr2, _idx2) = match second {
2372                Ok(lr2) => lr2,
2373                Err(e) => {
2374                    return Err(format!(
2375                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2376                    ));
2377                }
2378            };
2379            let clean2 = sanitize_selector(&lr2.content);
2380            self.reporter
2381                .debug(format!("resolved selector (retry): {clean2}"));
2382            if selector_is_useless(&clean2) {
2383                return Err(format!(
2384                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2385                    raw = lr2.content.trim(),
2386                    excerpt = truncate(&page_content.body_text, 300),
2387                ));
2388            }
2389            if !selector_matches(tab, &clean2).unwrap_or(false) {
2390                return Err(format!(
2391                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2392                ));
2393            }
2394            return Ok(clean2);
2395        }
2396
2397        Ok(clean)
2398    }
2399}
2400
2401/// Evaluates a JS expression that is expected to return a boolean.
2402fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2403    tab.evaluate(js, false)
2404        .map_err(|e| format!("evaluate failed: {e}"))?
2405        .value
2406        .and_then(|v| v.as_bool())
2407        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2408}
2409
2410/// Blank-page self-heal budget per wait step: how many reload cycles a
2411/// blank page gets before the wait gives up. The runner's egress and
2412/// resource bursts last minutes, so a single reload (v0.19.5) was not
2413/// enough when a burst outlived two wait budgets (immosai run #1651).
2414const MAX_BLANK_HEALS: u32 = 3;
2415/// Backoff sleep (seconds) before the 2nd/3rd blank-heal reload, giving
2416/// a transient runner outage time to clear.
2417const BLANK_HEAL_BACKOFF_SECS: u64 = 30;
2418
2419/// True when the tab rendered nothing meaningful: no body, an empty
2420/// body, or body text that is only whitespace. This is the signature
2421/// of a stalled SPA boot (index.html served, JS chunks never arrived —
2422/// runner egress stall), where any wait can only time out. A real
2423/// error page (e.g. `ERR_CONNECTION_REFUSED`) carries text and is NOT
2424/// blank, so those are left alone.
2425fn page_is_blank(tab: &Tab) -> bool {
2426    const JS: &str = "(() => { if (!document.body) return true; \
2427        const t = (document.body.innerText || '').trim(); \
2428        return document.body.childElementCount === 0 || t.length === 0; })()";
2429    eval_bool(tab, JS).unwrap_or(false)
2430}
2431
2432/// Checks whether a CSS selector matches at least one current element.
2433fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2434    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2435}
2436
2437// ── Free helper functions ──────────────────────────────────────────────
2438
2439fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2440    let name = format!("[navigate] {full_url}");
2441    match tab.navigate_to(full_url) {
2442        Ok(_) => {
2443            let _ = tab.wait_until_navigated();
2444            StepResult {
2445                name,
2446                status: StepStatus::Passed,
2447                message: format!("navigated to {full_url}"),
2448            }
2449        }
2450        Err(e) => StepResult {
2451            name,
2452            status: StepStatus::Failed,
2453            message: format!("navigation failed: {e}"),
2454        },
2455    }
2456}
2457
2458fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2459    let result = tab
2460        .evaluate(DOM_EXTRACT_JS, false)
2461        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2462
2463    let json_str = result
2464        .value
2465        .as_ref()
2466        .and_then(|v| v.as_str())
2467        .unwrap_or("[]");
2468
2469    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2470
2471    if elements.is_empty() {
2472        return Ok("(no interactive elements found)".to_owned());
2473    }
2474
2475    Ok(elements.join("\n"))
2476}
2477
2478fn get_page_text(tab: &Tab) -> PageContent {
2479    let url = tab.get_url();
2480
2481    let title = tab
2482        .evaluate("document.title", false)
2483        .ok()
2484        .and_then(|r| r.value)
2485        .and_then(|v| v.as_str().map(String::from))
2486        .unwrap_or_else(|| "unknown".to_owned());
2487
2488    let body_text = tab
2489        .evaluate(
2490            "document.body ? document.body.innerText : document.documentElement.innerText",
2491            false,
2492        )
2493        .ok()
2494        .and_then(|r| r.value)
2495        .and_then(|v| v.as_str().map(String::from))
2496        .unwrap_or_default();
2497
2498    PageContent {
2499        url,
2500        title,
2501        body_text: truncate(&body_text, 8000),
2502    }
2503}
2504
2505fn resolve_url(url: &str, base_url: &str) -> String {
2506    if url.starts_with("http://") || url.starts_with("https://") {
2507        return url.to_owned();
2508    }
2509    let base = base_url.trim_end_matches('/');
2510    if url.starts_with('/') {
2511        format!("{base}{url}")
2512    } else {
2513        format!("{base}/{url}")
2514    }
2515}
2516
2517/// Origin (`scheme://host[:port]`) of a base URL, used to scope the
2518/// per-test `Storage.clearDataForOrigin` call. Returns `None` when the
2519/// URL has no recognizable scheme/host (the clear is skipped).
2520#[must_use]
2521fn origin_of(base_url: &str) -> Option<String> {
2522    let url = if base_url.contains("://") {
2523        base_url.to_owned()
2524    } else {
2525        format!("https://{base_url}")
2526    };
2527    let (scheme, rest) = url.split_once("://")?;
2528    let authority = rest
2529        .split(['/', '?', '#'])
2530        .next()
2531        .filter(|a| !a.is_empty())?;
2532    Some(format!("{scheme}://{authority}"))
2533}
2534
2535/// True when a test performs its own login: its first navigate step
2536/// targets a login route (org `/auth/login`, tenant `/tenant/login`),
2537/// or it has no navigate step at all and rides the auto-navigate onto a
2538/// login `start_url`. Such tests need a cleared session — with a live one
2539/// the SPA bounces the login page before the form ever mounts. Tests
2540/// whose own first navigate goes to an app page keep the shared session
2541/// (most "page loads" tests rely on it) — a blanket clear broke exactly
2542/// those on immosai run #1643.
2543#[must_use]
2544fn test_targets_login(start_url: &str, steps: &[TestStep]) -> bool {
2545    for step in steps {
2546        if let TestStep::Navigate { url, .. } = step {
2547            return url.contains("login");
2548        }
2549    }
2550    start_url.contains("login")
2551}
2552
2553/// Human-readable label for a step, used when steps are skipped after an
2554/// earlier failure.
2555fn step_label(step: &TestStep) -> String {
2556    match step {
2557        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2558        TestStep::Click { target, .. } => format!("[click] {target}"),
2559        TestStep::Type { target, .. } => format!("[type] {target}"),
2560        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2561        TestStep::Assert {
2562            definition,
2563            preset,
2564            prompt,
2565            ..
2566        } => definition.as_ref().map_or_else(
2567            || {
2568                preset.as_ref().map_or_else(
2569                    || {
2570                        prompt.as_ref().map_or_else(
2571                            || "[assert]".to_owned(),
2572                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2573                        )
2574                    },
2575                    |p| format!("[assert] {p}"),
2576                )
2577            },
2578            |d| format!("[assert] {d}"),
2579        ),
2580        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2581        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2582        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2583    }
2584}
2585
2586/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2587#[must_use]
2588const fn step_kind_label(step: &TestStep) -> &'static str {
2589    match step {
2590        TestStep::Navigate { .. } => "navigate",
2591        TestStep::Click { .. } => "click",
2592        TestStep::Type { .. } => "type",
2593        TestStep::Wait { .. } => "wait",
2594        TestStep::Assert { .. } => "assert",
2595        TestStep::Screenshot { .. } => "screenshot",
2596        TestStep::Agent { .. } => "agent",
2597        TestStep::Mcp { .. } => "mcp",
2598    }
2599}
2600
2601// ── Support types ──────────────────────────────────────────────────────
2602
2603#[derive(Default)]
2604struct TestRunResult {
2605    passed: u32,
2606    failed: u32,
2607    skipped: u32,
2608    total: u32,
2609    details: Vec<StepResult>,
2610}
2611
2612struct PageContent {
2613    url: String,
2614    title: String,
2615    body_text: String,
2616}
2617
2618#[cfg(test)]
2619mod tests {
2620    use super::unix_to_rfc3339;
2621
2622    #[test]
2623    fn rfc3339_epoch_and_reference_dates() {
2624        assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2625        assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2626        assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2627        assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2628    }
2629
2630    #[test]
2631    fn rfc3339_handles_leap_years() {
2632        assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2633        assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2634    }
2635}
2636
2637#[cfg(test)]
2638mod verdict_parse_tests {
2639    use super::{verdict_is_fail, verdict_is_pass};
2640
2641    #[test]
2642    fn tolerates_markdown_punctuation_and_natural_language() {
2643        assert!(verdict_is_pass("PASS"));
2644        assert!(verdict_is_pass("pass"));
2645        assert!(verdict_is_pass("**PASS**\n\nThe page is clean."));
2646        assert!(verdict_is_pass("passes - no explicit error visible"));
2647        assert!(verdict_is_pass("  \"pass\""));
2648        assert!(!verdict_is_pass("FAIL: something broke"));
2649        assert!(!verdict_is_pass("**FAIL** broken"));
2650
2651        assert!(verdict_is_fail("**FAIL** broken"));
2652        assert!(verdict_is_fail("fails - error toast shown"));
2653        assert!(!verdict_is_fail("passes - ok"));
2654    }
2655}