Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// Formats a UNIX epoch timestamp as an RFC 3339 UTC string
26/// (`2026-09-19T16:04:00Z`) without pulling in a date dependency.
27/// Civil-from-days conversion after Howard Hinnant.
28#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31    let secs = i64::try_from(secs).unwrap_or(0);
32    let days = secs.div_euclid(86_400);
33    let rem = secs.rem_euclid(86_400);
34    let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35    let z = days + 719_468;
36    let era = z.div_euclid(146_097);
37    let doe = z.rem_euclid(146_097);
38    let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40    let mp = (5 * doy + 2) / 153;
41    let d = doy - (153 * mp + 2) / 5 + 1;
42    let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43    let y = yoe + era * 400 + i64::from(mth <= 2);
44    format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47/// One detected layout defect (`layout_no_issues` preset).
48#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50    #[serde(rename = "type")]
51    issue_type: String,
52    element: String,
53    detail: String,
54}
55
56/// In-browser DOM layout scan for `layout_no_issues`.
57///
58/// Geometry-only checks (no LLM, no pixels):
59/// 1. page horizontal overflow (`scrollWidth` > viewport width);
60/// 2. elements outside the viewport that scrolling cannot reveal
61///    (fixed elements off-screen, left/negative overflow, right-edge
62///    overflow beyond the horizontally scrollable area, and bottom
63///    overflow on a page that cannot scroll down) — below-the-fold
64///    content on a tall scrollable page is normal flow, NOT a defect;
65/// 3. text clipped by `overflow: hidden` containers whose content
66///    is measurably larger than the box;
67/// 4. interactive elements (buttons/links/inputs) whose center point
68///    is covered by a different element that would intercept the click.
69///
70/// Elements whose class matches a configured ignore prefix (default:
71/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
72/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
73/// skipped — they are intentionally 1x1 / off-screen. The prefix list
74/// is injected at `__IGNORE_CLASSES__` from
75/// [`ScenarioConfig::layout_ignore_classes`].
76///
77/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
78/// sticky headers) is excluded by the position/relation filters.
79const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81  const issues = [];
82  const push = (type, el, detail) => {
83    if (issues.length >= 30) return;
84    let element = el.tagName.toLowerCase();
85    if (el.id) element += '#' + el.id;
86    else if (typeof el.className === 'string' && el.className.trim())
87      element += '.' + el.className.trim().split(/\s+/).join('.');
88    issues.push({ type, element, detail: String(detail).slice(0, 220) });
89  };
90  const vw = document.documentElement.clientWidth || window.innerWidth;
91  const vh = document.documentElement.clientHeight || window.innerHeight;
92  if (!vw || !vh) return JSON.stringify(issues);
93  const de = document.documentElement;
94  // 1. Page-level horizontal overflow.
95  if (maxSW > vw + 2)
96    push('page-overflow-x', de,
97      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98  // Class-name prefixes to skip (injected; default = Angular CDK
99  // screen-reader helpers, which are intentionally 1x1 / off-screen).
100  const ignorePrefixes = __IGNORE_CLASSES__;
101  const isIgnored = (el) => {
102    if (typeof el.className !== 'string' || !el.className.trim()) return false;
103    const classes = el.className.trim().split(/\s+/);
104    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105  };
106  const body = document.body;
107  // The document element is not always the scroll container: the app may
108  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110  // "not scrollable" — measure the scrollable content height/width of the
111  // document element AND the body, and take the max.
112  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114  const vScrollable = maxSH > vh + 2;
115  const hScrollable = maxSW > vw + 2;
116  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117  // this element into view on the given axis — i.e. the element is inside a
118  // scrolling region, so lying beyond the viewport cut is not a defect.
119  const reachableByScroller = (el, axis) => {
120    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123    let p = el.parentElement;
124    while (p && p !== body) {
125      const pcs = getComputedStyle(p);
126      const o = pcs[ovProp];
127      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128          p[dim] > p[dimClient] + 2) return true;
129      p = p.parentElement;
130    }
131    return false;
132  };
133  const selfOverflowing = (el, cs, axis) => {
134    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138      el[dim] > el[dimClient] + 2;
139  };
140  // True when an ancestor clips this axis with overflow:hidden/clip — the
141  // element's overhang is not visible, so treat it as reachable.
142  const clippedByAncestor = (el, axis) => {
143    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144    let p = el.parentElement;
145    while (p && p !== body) {
146      const o = getComputedStyle(p)[ovProp];
147      if (o === 'hidden' || o === 'clip') return true;
148      p = p.parentElement;
149    }
150    return false;
151  };
152  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154  const hasContent = (el) =>
155    ((el.textContent || '').trim().length > 0) ||
156    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157  // 2. Elements outside the viewport that scrolling cannot reveal.
158  for (const el of all) {
159    const cs = getComputedStyle(el);
160    if (!visible(cs)) continue;
161    if (isIgnored(el)) continue;
162    const r = el.getBoundingClientRect();
163    if (r.width < 2 || r.height < 2) continue;
164    if (!hasContent(el) && el.children.length === 0) continue;
165    if (cs.position === 'fixed') {
166      // Fixed elements never move with the scroll: any edge outside the
167      // viewport is unreachable content and therefore a defect.
168      const overTop = -r.top;
169      const overLeft = -r.left;
170      const overRight = r.right - vw;
171      const overBottom = r.bottom - vh;
172      // Top/left overshoot is never reachable; bottom/right overshoot is
173      // only a defect when the fixed element cannot scroll that content
174      // into view itself (e.g. an opened Material drawer whose inner
175      // container scrolls is not a "cut-off" defect).
176      if (overTop > 2 || overLeft > 2 ||
177          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179        let where = '';
180        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186        push('element-out-of-viewport', el,
187          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188      }
189      continue;
190    }
191    if (cs.position === 'sticky') continue;
192    // Negative left overflow cannot be reached by scrolling (scrollLeft
193    // never goes below 0).
194    if (r.left < -2) {
195      push('element-out-of-viewport', el,
196        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197      continue;
198    }
199    // Negative top with the page at the top means the element sits above
200    // the document origin — also unreachable.
201    if (r.top < -2 && de.scrollTop <= 2) {
202      push('element-out-of-viewport', el,
203        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204      continue;
205    }
206    const overRight = r.right - vw;
207    // Right-edge overflow is only a defect when the page cannot scroll
208    // horizontally to reveal it (or the element sticks out past the
209    // scrollable content width itself).
210    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211      push('element-out-of-viewport', el,
212        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213      continue;
214    }
215    // Below-the-fold content on a scrollable page is normal (tall landing
216    // pages); only flag bottom overflow the user can never scroll to.
217    const overBottom = r.bottom - vh;
218    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219      push('element-out-of-viewport', el,
220        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221    }
222  }
223  // 3. Text clipped by overflow:hidden containers.
224  for (const el of all) {
225    if (isIgnored(el)) continue;
226    const cs = getComputedStyle(el);
227    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229    if (!(el.textContent || '').trim()) continue;
230    // Single-line truncation with an ellipsis is an intentional design
231    // pattern (Tailwind .truncate etc.), not a clipping defect.
232    if (cs.textOverflow === 'ellipsis') continue;
233    push('text-clipped', el,
234      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236  }
237  // 4. Interactive elements covered by a different element.
238  const interactive =
239    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240  const targets = document.querySelectorAll(interactive);
241  for (const el of targets) {
242    if (isIgnored(el)) continue;
243    const r = el.getBoundingClientRect();
244    if (r.width < 6 || r.height < 6) continue;
245    const cs = getComputedStyle(el);
246    if (!visible(cs)) continue;
247    const cx = r.left + r.width / 2;
248    const cy = r.top + r.height / 2;
249    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250    const top = document.elementFromPoint(cx, cy);
251    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252    if (isIgnored(top)) continue;
253    const tcs = getComputedStyle(top);
254    if (!visible(tcs)) continue;
255    if (tcs.pointerEvents === 'none') continue;
256    const tr = top.getBoundingClientRect();
257    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260    push('element-overlap', el,
261      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262      ' is covered by <' + tname + '>');
263  }
264  return JSON.stringify(issues);
265})()
266"#;
267
268/// How long the CDP connection stays open after the browser goes quiet.
269///
270/// `headless_chrome` ships a 30s default and tears down the entire connection
271/// when no traffic arrives for that long; a run must own its connection for
272/// its full duration instead.
273const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275/// Executes a [`Scenario`] against a real browser with optional LLM
276/// assistance for element targeting and assertions.
277pub struct ScenarioRunner {
278    config: ScenarioConfig,
279    definitions: HashMap<String, AssertDefinition>,
280    llm: LlmConfig,
281    timeout: Duration,
282    viewport_width: u32,
283    viewport_height: u32,
284    /// The viewport currently applied in the browser (CDP emulation).
285    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
286    applied_viewport: std::cell::Cell<(u32, u32)>,
287    endpoints: EndpointRegistry,
288    usage: Arc<UsageTracker>,
289    budgets: BudgetTracker,
290    /// Directory for failure artifacts (screenshots).
291    artifacts_dir: PathBuf,
292    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
293    reporter: Arc<Reporter>,
294    /// The test + step index currently executing, for LLM-call events.
295    current_step: std::cell::RefCell<Option<(String, u32)>>,
296    /// Epoch seconds of runner creation — constant for the whole run so
297    /// the LLM context preamble below is cacheable by upstream providers.
298    run_started: u64,
299    /// Whether this runner emits the `RunStarted`/`RunFinished` events.
300    /// When several scenario files are run concurrently by
301    /// [`crate::parallel::run_scenarios`], the orchestrator owns those two
302    /// events (one per batch), so each per-file runner suppresses its own.
303    emit_run_events: bool,
304}
305
306/// Aggregated results from a scenario run.
307#[derive(Debug, Default)]
308pub struct RunReport {
309    /// Number of tests that passed.
310    pub tests_passed: u32,
311    /// Number of tests that failed.
312    pub tests_failed: u32,
313    /// Number of steps that passed.
314    pub passed: u32,
315    /// Number of steps that failed.
316    pub failed: u32,
317    /// Number of steps that were skipped.
318    pub skipped: u32,
319    /// Per-step details.
320    pub details: Vec<StepResult>,
321}
322
323/// Result of a single step execution.
324#[derive(Debug, Clone)]
325pub struct StepResult {
326    /// The step name.
327    pub name: String,
328    /// Whether the step passed, failed, or was skipped.
329    pub status: StepStatus,
330    /// Human-readable result message.
331    pub message: String,
332}
333
334/// Outcome for a single step.
335pub use crate::events::StepStatus;
336
337/// Predefined assertion preset definition.
338struct AssertPreset {
339    name: &'static str,
340    system: &'static str,
341    user_template: &'static str,
342}
343
344/// Built-in assertion presets.
345#[allow(clippy::literal_string_with_formatting_args)]
346const ASSERTION_PRESETS: &[AssertPreset] = &[
347    AssertPreset {
348        name: "no_error_on_page",
349        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
350        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
351    },
352    AssertPreset {
353        name: "text_visible",
354        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
355        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
356    },
357    AssertPreset {
358        name: "element_exists",
359        system: "You are a QA tester. Check if a described UI element exists on a web page.",
360        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
361    },
362    AssertPreset {
363        name: "visual_no_issues",
364        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
365        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
366    },
367    AssertPreset {
368        name: "visual_no_overlaps",
369        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
370        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
371    },
372    AssertPreset {
373        name: "visual_fills_viewport",
374        system: "You are a visual QA engineer inspecting a website screenshot for UNDER-FILLED layouts: the rendered content stops at a fraction of the viewport and the remaining area is a large blank region the layout should have filled. Typical defects: page content ending around the halfway mark of the viewport with an empty, background-only area below it; an app shell, dashboard, or main panel collapsing to part of its expected height and leaving dead space; content that fills only the left portion of a wide viewport with unused space to the right. Do NOT fail intentionally short or centered pages: a login or signup card, a landing hero, a short article, a confirmation step, or an empty state that does not fill the viewport is normal design. Only fail when the blank region is clearly a layout defect — the page looks cut off or collapsed, or a content container (calendar, table, chart, map, panel) visibly fails to stretch to the space its own layout provides.",
375        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nDoes the page content fill the viewport as its layout intends — with no large blank region where the page appears to end partway down or partway across? Respond with exactly \"PASS\" if the page fills the viewport or is intentionally short, or \"FAIL: <describe the under-filled region and where it appears>\" if the content clearly stops at a fraction of the available page.",
376    },
377    AssertPreset {
378        name: "visual_text_visible",
379        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
380        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
381    },
382    AssertPreset {
383        name: "layout_no_issues",
384        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
385        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
386    },
387];
388
389/// Whether an LLM verdict should be read as PASS.
390///
391/// Models routinely wrap the verdict in markdown (`**PASS** …`) or prefix it
392/// with punctuation, which a bare `starts_with("pass")` misreads as a failure.
393/// Strip any leading non-alphanumerics, then match the leading word; this also
394/// accepts the natural-language forms `pass` / `passes`.
395fn verdict_is_pass(content: &str) -> bool {
396    content
397        .trim()
398        .trim_start_matches(|c: char| !c.is_alphanumeric())
399        .to_lowercase()
400        .starts_with("pass")
401}
402
403/// Whether an LLM verdict should be read as FAIL (see [`verdict_is_pass`]).
404fn verdict_is_fail(content: &str) -> bool {
405    content
406        .trim()
407        .trim_start_matches(|c: char| !c.is_alphanumeric())
408        .to_lowercase()
409        .starts_with("fail")
410}
411
412impl ScenarioRunner {
413    /// Creates a new runner with the given scenario configuration and
414    /// assertion definitions.
415    #[must_use]
416    #[allow(clippy::needless_pass_by_value)]
417    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
418        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
419    }
420
421    /// Creates a runner that reports run events through the given reporter.
422    #[must_use]
423    #[allow(clippy::needless_pass_by_value)]
424    pub fn with_reporter(
425        scenario_config: ScenarioConfig,
426        definitions: Vec<AssertDefinition>,
427        reporter: Arc<Reporter>,
428    ) -> Self {
429        Self::with_reporter_mode(scenario_config, definitions, reporter, true)
430    }
431
432    /// Creates a runner that reports run events through the given reporter
433    /// but suppresses the `RunStarted`/`RunFinished` events. Used by the
434    /// parallel orchestrator, which emits those once per batch.
435    #[must_use]
436    #[allow(clippy::needless_pass_by_value)]
437    pub fn with_reporter_parallel(
438        scenario_config: ScenarioConfig,
439        definitions: Vec<AssertDefinition>,
440        reporter: Arc<Reporter>,
441    ) -> Self {
442        Self::with_reporter_mode(scenario_config, definitions, reporter, false)
443    }
444
445    #[must_use]
446    #[allow(clippy::needless_pass_by_value)]
447    fn with_reporter_mode(
448        scenario_config: ScenarioConfig,
449        definitions: Vec<AssertDefinition>,
450        reporter: Arc<Reporter>,
451        emit_run_events: bool,
452    ) -> Self {
453        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
454            &scenario_config,
455        ));
456        let llm = LlmConfig {
457            url: scenario_config
458                .llm_url
459                .clone()
460                .unwrap_or_else(crate::llm_base_url),
461            model: scenario_config
462                .llm_model
463                .clone()
464                .unwrap_or_else(crate::llm_model),
465            api_key: scenario_config
466                .llm_api_key
467                .clone()
468                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
469            headers: if scenario_config.llm_headers.is_empty() {
470                crate::parse_headers_env()
471            } else {
472                scenario_config.llm_headers.clone()
473            },
474            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
475            temperature: scenario_config.temperature,
476            thinking: scenario_config.thinking,
477            model_params: scenario_config.model_params.clone(),
478            cache: scenario_config.cache.unwrap_or(true),
479            max_attempts: crate::default_llm_attempts(),
480            provider: crate::scenario::Provider::Openai,
481            deployment: None,
482            api_version: None,
483            auth: crate::scenario::AuthConfig::default(),
484            header_commands: std::collections::HashMap::new(),
485            aws: crate::scenario::AwsConfig::default(),
486        };
487        let endpoints = EndpointRegistry::from_config(
488            &scenario_config.endpoints,
489            Some(&llm),
490            crate::endpoints::EndpointDefaults {
491                cache: scenario_config.cache,
492                cache_pricing: scenario_config.cache_pricing,
493            },
494        );
495        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
496        let defs_map: HashMap<String, AssertDefinition> = definitions
497            .into_iter()
498            .map(|d| (d.name.clone(), d))
499            .collect();
500
501        Self {
502            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
503            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
504            viewport_height: scenario_config.viewport_height.unwrap_or(720),
505            applied_viewport: std::cell::Cell::new((0, 0)),
506            config: scenario_config.clone(),
507            definitions: defs_map,
508            llm,
509            endpoints,
510            usage: Arc::new(UsageTracker::new()),
511            budgets,
512            artifacts_dir: PathBuf::from(
513                scenario_config
514                    .artifacts_dir
515                    .unwrap_or_else(|| "artifacts".to_owned()),
516            ),
517            reporter,
518            current_step: std::cell::RefCell::new(None),
519            run_started: SystemTime::now()
520                .duration_since(UNIX_EPOCH)
521                .map_or(0, |d| d.as_secs()),
522            emit_run_events,
523        }
524    }
525
526    /// Emits an event; a sink failure degrades to a console warning so a
527    /// broken log file can never mask the run itself.
528    fn emit_event(&self, event: &TestEvent) {
529        if let Err(err) = self.reporter.emit(event) {
530            use std::io::Write as _;
531            let mut out = std::io::stderr().lock();
532            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
533        }
534    }
535
536    /// Returns a clone of the [`UsageTracker`] for reporting.
537    #[must_use]
538    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
539        Arc::clone(&self.usage)
540    }
541
542    /// Returns a reference to the [`BudgetTracker`].
543    #[must_use]
544    pub const fn budget_tracker(&self) -> &BudgetTracker {
545        &self.budgets
546    }
547
548    /// Executes all test groups in the scenario and returns a report.
549    ///
550    /// # Errors
551    ///
552    /// Returns an error if the browser fails to launch.
553    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
554    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
555        let mut report = RunReport::default();
556
557        if tests.is_empty() {
558            self.reporter.warn("No tests defined in scenario.");
559            if self.emit_run_events {
560                self.emit_event(&TestEvent::RunFinished {
561                    tests_passed: 0,
562                    tests_failed: 0,
563                    steps_passed: 0,
564                    steps_failed: 0,
565                    steps_skipped: 0,
566                    total_cost: 0.0,
567                    total_tokens: 0,
568                    total_input_tokens: 0,
569                    total_output_tokens: 0,
570                    total_cached_input_tokens: 0,
571                    total_cache_creation_input_tokens: 0,
572                    models: Vec::new(),
573                    total_calls: 0,
574                });
575            }
576            return Ok(report);
577        }
578
579        if self.emit_run_events {
580            self.emit_event(&TestEvent::RunStarted {
581                total_tests: tests.len() as u32,
582            });
583        }
584
585        let browser_headless = self.config.browser_headless.unwrap_or(true);
586
587        let launch_opts = LaunchOptions {
588            headless: browser_headless,
589            window_size: Some((self.viewport_width, self.viewport_height)),
590            sandbox: false,
591            // headless_chrome defaults this to 30s and shuts down the whole CDP
592            // connection when no messages arrive for that long. A scenario can
593            // easily exceed 30s of browser silence (slow LLM targeting/assertion
594            // calls, page waits, budget checks between steps), after which every
595            // remaining step fails with "Unable to make method calls because
596            // underlying connection is closed" — one quiet gap kills the run.
597            // Open-ended scenarios must own the connection for their full
598            // duration, so keep it alive for 6 hours.
599            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
600            ..LaunchOptions::default()
601        };
602
603        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
604        let tab = browser.new_tab().context("failed to open browser tab")?;
605        match (
606            self.config.browser_basic_auth_user.clone(),
607            self.config.browser_basic_auth_password.clone(),
608        ) {
609            (Some(username), Some(password)) if !username.is_empty() && !password.is_empty() => {
610                tab.authenticate(Some(username), Some(password))
611                    .context("failed to configure browser HTTP Basic Auth")?;
612                tab.enable_fetch(None, Some(true))
613                    .context("failed to enable browser HTTP authentication")?;
614            }
615            (None, None) => {}
616            _ => anyhow::bail!("browser Basic Auth requires both username and password"),
617        }
618        let _ = tab.set_default_timeout(self.timeout);
619
620        // Start MCP server if configured
621        #[cfg(feature = "mcp-server")]
622        if let Some(ref mcp_cfg) = self.config.mcp_server {
623            if mcp_cfg.enabled {
624                let port = mcp_cfg.port;
625                std::thread::spawn(move || {
626                    let _ = crate::mcp_server::start_mcp_server(port);
627                });
628            }
629        }
630        #[cfg(not(feature = "mcp-server"))]
631        if let Some(mcp_cfg) = &self.config.mcp_server {
632            if mcp_cfg.enabled {
633                self.reporter
634                    .warn("MCP server configured but 'mcp-server' feature not enabled");
635            }
636        }
637
638        // Start A2A agent server if configured
639        #[cfg(feature = "a2a-server")]
640        if let Some(ref a2a_cfg) = self.config.a2a_server {
641            if a2a_cfg.enabled {
642                let port = a2a_cfg.port;
643                tokio::spawn(crate::a2a_server::start_a2a_server(port));
644            }
645        }
646        #[cfg(not(feature = "a2a-server"))]
647        if let Some(a2a_cfg) = &self.config.a2a_server {
648            if a2a_cfg.enabled {
649                self.reporter
650                    .warn("A2A server configured but 'a2a-server' feature not enabled");
651            }
652        }
653
654        let retry_budget = self.config.retry_failed_tests.unwrap_or(0);
655
656        for test in tests {
657            // A shared tab on a contended runner makes some page loads
658            // stall (a JS chunk or a GraphQL call hangs mid-flight) and
659            // the test fails on a wait/assert that a fresh run passes.
660            // Re-run the WHOLE test when it failed and the retry budget
661            // allows; the fresh per-test isolation (login clear +
662            // auto-navigate) applies to the retry too. Both attempts'
663            // LLM spend stays in the budget accounting (each attempt
664            // commits its usage); only the final attempt is reported.
665            let details_len = report.details.len();
666            let passed_before = report.passed;
667            let failed_before = report.failed;
668            let skipped_before = report.skipped;
669            #[allow(unused_assignments)]
670            let mut final_result = None;
671            let mut attempt = 0;
672            loop {
673                attempt += 1;
674                self.emit_event(&TestEvent::TestStarted {
675                    test: test.name.clone(),
676                });
677
678                self.usage.reset_per_test();
679
680                let test_started = Instant::now();
681                let test_result = self.run_test(test, &tab);
682                let duration_ms = test_started.elapsed().as_millis() as u64;
683                let usage = self.usage.current_test_snapshot();
684                self.usage.commit_test(&test.name);
685
686                self.emit_event(&TestEvent::TestFinished {
687                    test: test.name.clone(),
688                    passed: test_result.passed,
689                    failed: test_result.failed,
690                    skipped: test_result.skipped,
691                    duration_ms,
692                    cost: usage.total_cost,
693                    tokens: usage.total_tokens,
694                    input_tokens: usage.total_input_tokens,
695                    output_tokens: usage.total_output_tokens,
696                    cached_input_tokens: usage.total_cached_input_tokens,
697                    cache_creation_input_tokens: usage.total_cache_creation_input_tokens,
698                    models: usage.models.clone(),
699                    calls: usage.total_calls,
700                });
701
702                final_result = Some(test_result);
703                let result = final_result.as_ref().expect("just assigned");
704                if result.failed == 0 || result.total == 0 || attempt > retry_budget {
705                    break;
706                }
707                self.reporter.warn(format!(
708                    "! retrying failed test '{}' (attempt {}/{}) — a fresh run passes on \
709                     transient page-load stalls",
710                    test.name,
711                    attempt + 1,
712                    retry_budget + 1
713                ));
714                // Roll this failed attempt out of the report so the
715                // retry's outcome replaces it.
716                report.details.truncate(details_len);
717                report.passed = passed_before;
718                report.failed = failed_before;
719                report.skipped = skipped_before;
720            }
721
722            let test_result = final_result.unwrap_or_default();
723            if test_result.failed == 0 && test_result.total > 0 {
724                report.tests_passed += 1;
725            } else if test_result.total > 0 {
726                report.tests_failed += 1;
727            }
728
729            report.passed += test_result.passed;
730            report.failed += test_result.failed;
731            report.skipped += test_result.skipped;
732            report.details.extend(test_result.details);
733        }
734
735        let global = self.usage.global_snapshot();
736        if self.emit_run_events {
737            self.emit_event(&TestEvent::RunFinished {
738                tests_passed: report.tests_passed,
739                tests_failed: report.tests_failed,
740                steps_passed: report.passed,
741                steps_failed: report.failed,
742                steps_skipped: report.skipped,
743                total_cost: global.total_cost,
744                total_tokens: global.total_tokens,
745                total_input_tokens: global.total_input_tokens,
746                total_output_tokens: global.total_output_tokens,
747                total_cached_input_tokens: global.total_cached_input_tokens,
748                total_cache_creation_input_tokens: global.total_cache_creation_input_tokens,
749                models: global.models.clone(),
750                total_calls: global.total_calls,
751            });
752        }
753
754        Ok(report)
755    }
756
757    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
758    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
759        let base_url = test
760            .base_url
761            .clone()
762            .or_else(|| self.config.base_url.clone())
763            .unwrap_or_else(crate::base_url);
764
765        // Per-test viewport override: switch the browser via CDP
766        // device-metrics emulation before this test runs.
767        let vw = test.viewport_width.unwrap_or(self.viewport_width);
768        let vh = test.viewport_height.unwrap_or(self.viewport_height);
769        if self.applied_viewport.get() != (vw, vh) {
770            self.apply_viewport(tab, vw, vh);
771            self.applied_viewport.set((vw, vh));
772        }
773
774        // Per-test isolation: every test starts from its own start_url
775        // (unless auto_navigate is disabled), so a test never inherits the
776        // previous test's page state.
777        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
778
779        let start_url = test
780            .start_url
781            .clone()
782            .or_else(|| self.config.start_url.clone())
783            .unwrap_or_else(|| "/dashboard".to_owned());
784
785        // Per-test state isolation for LOGIN tests: the tab is shared
786        // across the file's tests, and a previous test's login persists
787        // (Cognito tokens in localStorage + the hosted-UI cookies). The
788        // SPA's login pages then auto-continue authenticated visitors on
789        // boot, so a later login test never sees the form and times out
790        // waiting for `#email` (observed on immosai runs #1641/#1642: a
791        // shard's first three logins pass, the fourth onward time out).
792        // Clear cookies + origin storage ONLY for tests that perform
793        // their own login (their first navigate step targets a login
794        // route, or they have no navigate step and ride the auto-nav
795        // onto a login start_url). The many "page loads" tests that
796        // navigate app pages directly RELY on the shared session — a
797        // blanket clear bounces them off their target (run #1643: every
798        // shard failed with "the current page is the login page, not
799        // the mailboxes page"). The HTTP cache is deliberately NOT
800        // cleared — re-downloading the SPA bundle per test would only
801        // add boot time on a slow runner egress.
802        if auto_navigate && test_targets_login(&start_url, &test.steps) {
803            use headless_chrome::protocol::cdp::{Network, Storage};
804            if let Err(e) = tab.call_method(Network::ClearBrowserCookies(None)) {
805                self.reporter
806                    .warn(format!("per-test cookie clear failed: {e}"));
807            }
808            if let Some(origin) = origin_of(&base_url) {
809                if let Err(e) = tab.call_method(Storage::ClearDataForOrigin {
810                    origin,
811                    storage_Types: "all".to_string(),
812                }) {
813                    self.reporter
814                        .warn(format!("per-test storage clear failed: {e}"));
815                }
816            }
817        }
818        if auto_navigate {
819            let full_url = resolve_url(&start_url, &base_url);
820            self.reporter.debug(format!("auto-navigate: {full_url}"));
821            let _ = tab.navigate_to(&full_url);
822            let _ = tab.wait_until_navigated();
823            std::thread::sleep(Duration::from_secs(4));
824        }
825
826        let mut result = TestRunResult::default();
827
828        for (step_index, step) in test.steps.iter().enumerate() {
829            result.total += 1;
830
831            let wait_ms = match step {
832                TestStep::Navigate { wait_after_ms, .. }
833                | TestStep::Click { wait_after_ms, .. }
834                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
835                _ => None,
836            };
837
838            self.current_step
839                .replace(Some((test.name.clone(), step_index as u32)));
840            self.emit_event(&TestEvent::StepStarted {
841                test: test.name.clone(),
842                index: step_index as u32,
843                label: step_label(step),
844            });
845            let step_started = Instant::now();
846
847            let mut step_result = match step {
848                TestStep::Navigate { url, .. } => {
849                    let full_url = resolve_url(url, &base_url);
850                    run_navigate_step(&full_url, tab)
851                }
852                TestStep::Click {
853                    target,
854                    selector,
855                    endpoint,
856                    idempotent,
857                    ..
858                } => self.run_click(
859                    target,
860                    selector.as_deref(),
861                    endpoint.as_deref(),
862                    test.endpoint.as_deref(),
863                    *idempotent,
864                    tab,
865                ),
866                TestStep::Type {
867                    target,
868                    text,
869                    selector,
870                    endpoint,
871                    idempotent,
872                    ..
873                } => self.run_type(
874                    target,
875                    text,
876                    selector.as_deref(),
877                    endpoint.as_deref(),
878                    test.endpoint.as_deref(),
879                    *idempotent,
880                    tab,
881                ),
882                TestStep::Wait {
883                    target,
884                    selector,
885                    text,
886                    timeout_ms,
887                    endpoint,
888                    idempotent,
889                } => self.run_wait(
890                    target,
891                    selector.as_deref(),
892                    text.as_deref(),
893                    *timeout_ms,
894                    endpoint.as_deref(),
895                    test.endpoint.as_deref(),
896                    *idempotent,
897                    tab,
898                ),
899                TestStep::Assert {
900                    definition,
901                    preset,
902                    prompt,
903                    assert_text,
904                    endpoint,
905                    screenshot,
906                } => self.run_assert(
907                    definition.as_deref(),
908                    preset.as_deref(),
909                    prompt.as_deref(),
910                    assert_text.as_deref(),
911                    *screenshot,
912                    endpoint.as_deref(),
913                    test.endpoint.as_deref(),
914                    tab,
915                ),
916                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
917                TestStep::Agent {
918                    agent,
919                    task,
920                    definition,
921                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
922                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
923            };
924
925            // Failure diagnostics: capture the page state and a screenshot so
926            // CI logs say WHAT the page looked like when the step failed,
927            // instead of a bare "timed out: The event waited for never came".
928            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
929                let state = diagnostics::capture(tab);
930                let screenshot = diagnostics::save_screenshot(
931                    tab,
932                    &self.artifacts_dir,
933                    &test.name,
934                    &test.name,
935                    step_index,
936                    step_kind_label(step),
937                );
938                step_result.message = format!(
939                    "{base} — {excerpt}",
940                    base = step_result.message,
941                    excerpt = diagnostics::inline_excerpt(&state),
942                );
943                (Some(diagnostics::full_context(&state)), screenshot)
944            } else {
945                (None, None)
946            };
947
948            let duration_ms = step_started.elapsed().as_millis() as u64;
949            self.emit_event(&TestEvent::StepFinished {
950                test: test.name.clone(),
951                index: step_index as u32,
952                label: step_result.name.clone(),
953                status: step_result.status,
954                duration_ms,
955                message: step_result.message.clone(),
956                diagnostics: diagnostics_block,
957                screenshot: screenshot_path,
958            });
959            self.current_step.replace(None);
960
961            match step_result.status {
962                StepStatus::Passed => result.passed += 1,
963                StepStatus::Failed => result.failed += 1,
964                StepStatus::Skipped => result.skipped += 1,
965            }
966
967            // Fail fast: the first failed step ends the test and the
968            // remaining steps are reported as skipped (no LLM budget is
969            // burned asserting against a page that is already known broken).
970            if step_result.status == StepStatus::Failed
971                && !self.config.continue_on_failure
972                && step_index + 1 < test.steps.len()
973            {
974                self.emit_event(&TestEvent::Warning {
975                    message: format!(
976                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
977                        test.steps.len() - step_index - 1
978                    ),
979                });
980                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
981                    let skipped_index = step_index + 1 + offset;
982                    let label = step_label(skipped);
983                    result.total += 1;
984                    result.skipped += 1;
985                    self.emit_event(&TestEvent::StepStarted {
986                        test: test.name.clone(),
987                        index: skipped_index as u32,
988                        label: label.clone(),
989                    });
990                    self.emit_event(&TestEvent::StepFinished {
991                        test: test.name.clone(),
992                        index: skipped_index as u32,
993                        label,
994                        status: StepStatus::Skipped,
995                        duration_ms: 0,
996                        message: "skipped: previous step failed".into(),
997                        diagnostics: None,
998                        screenshot: None,
999                    });
1000                    result.details.push(StepResult {
1001                        name: step_label(skipped),
1002                        status: StepStatus::Skipped,
1003                        message: "skipped: previous step failed".into(),
1004                    });
1005                }
1006                result.details.push(step_result);
1007                return result;
1008            }
1009
1010            // Check per-test budget after each step
1011            let test_usage = self.usage.current_test_snapshot();
1012            let global_usage = self.usage.global_snapshot();
1013            let budget_status = self.budgets.check_all(
1014                &test.name,
1015                &test_usage,
1016                &global_usage,
1017                test.budget.as_ref(),
1018            );
1019            match budget_status {
1020                BudgetStatus::HardExceeded { message, .. } => {
1021                    self.emit_event(&TestEvent::Warning {
1022                        message: format!("budget exceeded: {message}"),
1023                    });
1024                    result.details.push(StepResult {
1025                        name: "[budget]".into(),
1026                        status: StepStatus::Failed,
1027                        message,
1028                    });
1029                    result.failed += 1;
1030                    return result;
1031                }
1032                BudgetStatus::SoftExceeded { message, .. } => {
1033                    self.emit_event(&TestEvent::Warning {
1034                        message: format!("budget warning: {message}"),
1035                    });
1036                }
1037                BudgetStatus::Ok => {}
1038            }
1039
1040            if let Some(ms) = wait_ms {
1041                std::thread::sleep(Duration::from_millis(ms));
1042            }
1043
1044            result.details.push(step_result);
1045        }
1046
1047        result
1048    }
1049
1050    /// Applies a viewport size to the current tab via CDP
1051    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
1052    /// overrides and the viewport matrix. The initial window size set at
1053    /// browser launch is replaced by emulation; failures are logged but
1054    /// do not fail the test (a mismatched viewport only weakens coverage).
1055    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
1056        use headless_chrome::protocol::cdp::Emulation;
1057        let _ = self;
1058        let params = Emulation::SetDeviceMetricsOverride {
1059            width,
1060            height,
1061            device_scale_factor: 1.0,
1062            mobile: false,
1063            scale: None,
1064            screen_width: Some(width),
1065            screen_height: Some(height),
1066            position_x: None,
1067            position_y: None,
1068            dont_set_visible_size: None,
1069            screen_orientation: None,
1070            viewport: None,
1071            display_feature: None,
1072            device_posture: None,
1073        };
1074        self.reporter.debug(format!("viewport: {width}x{height}"));
1075        if let Err(e) = tab.call_method(params) {
1076            self.reporter
1077                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
1078        }
1079    }
1080
1081    /// Height (px) covered by assert-step screenshots: the configured
1082    /// `screenshot_max_height` (absolute px or viewport multiple, default
1083    /// `"20x"`) resolved against the currently applied viewport, raised to
1084    /// at least the viewport height so the visible screen is always fully
1085    /// included. The capture is split into viewport-tall tiles, so this
1086    /// value bounds total coverage (and hence the number of image parts).
1087    #[must_use]
1088    fn screenshot_height_cap(&self) -> u32 {
1089        let viewport_height = self.current_viewport_height();
1090        let cap = self
1091            .config
1092            .screenshot_max_height
1093            .as_ref()
1094            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
1095        cap.max(viewport_height)
1096    }
1097
1098    /// Height of the viewport currently emulated in the browser (falling
1099    /// back to the configured default before any emulation was applied).
1100    #[must_use]
1101    const fn current_viewport_height(&self) -> u32 {
1102        let (_, height) = self.applied_viewport.get();
1103        if height > 0 {
1104            height
1105        } else {
1106            self.viewport_height
1107        }
1108    }
1109
1110    // ── step handlers ───────────────────────────────────────────────────
1111
1112    #[allow(clippy::too_many_lines)]
1113    fn run_click(
1114        &self,
1115        target: &str,
1116        selector_override: Option<&str>,
1117        step_endpoint: Option<&str>,
1118        test_endpoint: Option<&str>,
1119        idempotent: bool,
1120        tab: &Tab,
1121    ) -> StepResult {
1122        let name = format!("[click] {target}");
1123        let selector = match self.resolve_selector(
1124            selector_override,
1125            target,
1126            step_endpoint,
1127            test_endpoint,
1128            tab,
1129        ) {
1130            Ok(s) => s,
1131            Err(msg) => {
1132                if idempotent {
1133                    return StepResult {
1134                        name,
1135                        status: StepStatus::Skipped,
1136                        message: format!("skipped (idempotent): no target found — {msg}"),
1137                    };
1138                }
1139                return StepResult {
1140                    name,
1141                    status: StepStatus::Failed,
1142                    message: msg,
1143                };
1144            }
1145        };
1146
1147        // Idempotent steps probe briefly: a missing target means the
1148        // action was already done / not applicable (e.g. an
1149        // already-authenticated session), and skipping is the success
1150        // path, not a failure.
1151        let probe_secs = if idempotent { 5 } else { 10 };
1152        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1153            Ok(element) => match element.click() {
1154                Ok(_) => StepResult {
1155                    name,
1156                    status: StepStatus::Passed,
1157                    message: format!("clicked {selector}"),
1158                },
1159                Err(e) => StepResult {
1160                    name,
1161                    status: StepStatus::Failed,
1162                    message: format!("click failed on {selector}: {e}"),
1163                },
1164            },
1165            Err(e) if idempotent => StepResult {
1166                name,
1167                status: StepStatus::Skipped,
1168                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1169            },
1170            Err(e) => StepResult {
1171                name,
1172                status: StepStatus::Failed,
1173                message: format!("element {selector} not found: {e}"),
1174            },
1175        }
1176    }
1177
1178    #[allow(clippy::too_many_arguments)]
1179    fn run_type(
1180        &self,
1181        target: &str,
1182        text: &str,
1183        selector_override: Option<&str>,
1184        step_endpoint: Option<&str>,
1185        test_endpoint: Option<&str>,
1186        idempotent: bool,
1187        tab: &Tab,
1188    ) -> StepResult {
1189        let name = format!("[type] {target}");
1190        let selector = match self.resolve_selector(
1191            selector_override,
1192            target,
1193            step_endpoint,
1194            test_endpoint,
1195            tab,
1196        ) {
1197            Ok(s) => s,
1198            Err(msg) => {
1199                if idempotent {
1200                    return StepResult {
1201                        name,
1202                        status: StepStatus::Skipped,
1203                        message: format!("skipped (idempotent): no target found — {msg}"),
1204                    };
1205                }
1206                return StepResult {
1207                    name,
1208                    status: StepStatus::Failed,
1209                    message: msg,
1210                };
1211            }
1212        };
1213
1214        let probe_secs = if idempotent { 5 } else { 10 };
1215        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1216            Ok(element) => {
1217                if let Err(e) = element.click() {
1218                    return StepResult {
1219                        name,
1220                        status: StepStatus::Failed,
1221                        message: format!("click to focus {selector} failed: {e}"),
1222                    };
1223                }
1224
1225                let js = format!(
1226                    "document.querySelector('{}').value = '';",
1227                    selector.replace('\'', "\\'")
1228                );
1229                let _ = tab.evaluate(&js, false);
1230
1231                match element.type_into(text) {
1232                    Ok(_) => StepResult {
1233                        name,
1234                        status: StepStatus::Passed,
1235                        message: format!("typed {text:?} into {selector}"),
1236                    },
1237                    Err(e) => StepResult {
1238                        name,
1239                        status: StepStatus::Failed,
1240                        message: format!("type into {selector} failed: {e}"),
1241                    },
1242                }
1243            }
1244            Err(e) if idempotent => StepResult {
1245                name,
1246                status: StepStatus::Skipped,
1247                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1248            },
1249            Err(e) => StepResult {
1250                name,
1251                status: StepStatus::Failed,
1252                message: format!("element {selector} not found: {e}"),
1253            },
1254        }
1255    }
1256
1257    #[allow(clippy::too_many_arguments)]
1258    #[allow(clippy::too_many_lines)]
1259    fn run_wait(
1260        &self,
1261        target: &str,
1262        selector_override: Option<&str>,
1263        text: Option<&str>,
1264        timeout_ms: Option<u64>,
1265        step_endpoint: Option<&str>,
1266        test_endpoint: Option<&str>,
1267        idempotent: bool,
1268        tab: &Tab,
1269    ) -> StepResult {
1270        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1271        let step_name = format!("[wait] {target}");
1272
1273        // Resolve an explicit selector only (text-only waits are LLM-free).
1274        let selector = match selector_override {
1275            Some(s) => Some(s.to_owned()),
1276            None if text.is_some() => None,
1277            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1278                Ok(s) => Some(s),
1279                Err(msg) => {
1280                    if idempotent {
1281                        return StepResult {
1282                            name: step_name,
1283                            status: StepStatus::Skipped,
1284                            message: format!("skipped (idempotent): no target found — {msg}"),
1285                        };
1286                    }
1287                    return StepResult {
1288                        name: step_name,
1289                        status: StepStatus::Failed,
1290                        message: msg,
1291                    };
1292                }
1293            },
1294        };
1295
1296        if text.is_some() {
1297            let sel_js = selector
1298                .as_deref()
1299                .map(crate::selectors::selector_matches_js);
1300            let text_js = text.map(|t| {
1301                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1302                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1303            });
1304
1305            let deadline = Instant::now() + timeout;
1306            loop {
1307                let sel_ok = sel_js
1308                    .as_ref()
1309                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1310                let text_ok = text_js
1311                    .as_ref()
1312                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1313                if sel_ok && text_ok {
1314                    let mut what = Vec::new();
1315                    if let Some(sel) = &selector {
1316                        what.push(format!("found {sel}"));
1317                    }
1318                    if let Some(t) = text {
1319                        what.push(format!("text {t:?} visible"));
1320                    }
1321                    return StepResult {
1322                        name: step_name,
1323                        status: StepStatus::Passed,
1324                        message: what.join(" and "),
1325                    };
1326                }
1327                if Instant::now() >= deadline {
1328                    let mut what = Vec::new();
1329                    if let Some(sel) = &selector {
1330                        what.push(sel.clone());
1331                    }
1332                    if let Some(t) = text {
1333                        what.push(format!("text {t:?}"));
1334                    }
1335                    let message = format!(
1336                        "wait for {} timed out after {}ms: the event waited for never came",
1337                        what.join(" / "),
1338                        timeout.as_millis(),
1339                    );
1340                    if idempotent {
1341                        return StepResult {
1342                            name: step_name,
1343                            status: StepStatus::Skipped,
1344                            message: format!("skipped (idempotent): {message}"),
1345                        };
1346                    }
1347                    return StepResult {
1348                        name: step_name,
1349                        status: StepStatus::Failed,
1350                        message,
1351                    };
1352                }
1353                std::thread::sleep(Duration::from_millis(250));
1354            }
1355        }
1356
1357        match selector.as_deref() {
1358            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1359                Ok(_) => StepResult {
1360                    name: step_name,
1361                    status: StepStatus::Passed,
1362                    message: format!("found {sel}"),
1363                },
1364                Err(e) if idempotent => StepResult {
1365                    name: step_name,
1366                    status: StepStatus::Skipped,
1367                    message: format!(
1368                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1369                        timeout.as_millis()
1370                    ),
1371                },
1372                Err(e) => StepResult {
1373                    name: step_name,
1374                    status: StepStatus::Failed,
1375                    message: format!(
1376                        "wait for {sel} timed out after {}ms: {e}",
1377                        timeout.as_millis()
1378                    ),
1379                },
1380            },
1381            None => StepResult {
1382                name: step_name,
1383                status: StepStatus::Failed,
1384                message: "wait step has neither selector nor text".into(),
1385            },
1386        }
1387    }
1388
1389    #[allow(clippy::too_many_arguments)]
1390    fn run_assert(
1391        &self,
1392        definition: Option<&str>,
1393        preset: Option<&str>,
1394        prompt: Option<&str>,
1395        assert_text: Option<&str>,
1396        screenshot: bool,
1397        step_endpoint: Option<&str>,
1398        test_endpoint: Option<&str>,
1399        tab: &Tab,
1400    ) -> StepResult {
1401        std::thread::sleep(Duration::from_millis(500));
1402
1403        let page_content = get_page_text(tab);
1404
1405        // Vision attach: capture the full page once per assert step and
1406        // split it into viewport-tall tiles (the total coverage is bounded
1407        // by the configured height cap so vision tokens stay sane). All
1408        // tile data URLs are handed to the preset/prompt evaluation below.
1409        let image: Option<Vec<String>> = if screenshot {
1410            let endpoint = self
1411                .endpoints
1412                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1413            if !endpoint.vision {
1414                return StepResult {
1415                    name: "[assert]".into(),
1416                    status: StepStatus::Failed,
1417                    message: format!(
1418                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1419                        name = endpoint.name
1420                    ),
1421                };
1422            }
1423            match crate::vision::capture_screenshot_data_urls(
1424                tab,
1425                self.config
1426                    .screenshot_max_dimension
1427                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1428                self.screenshot_height_cap(),
1429                self.current_viewport_height(),
1430            ) {
1431                Ok(urls) => Some(urls),
1432                Err(e) => {
1433                    return StepResult {
1434                        name: "[assert]".into(),
1435                        status: StepStatus::Failed,
1436                        message: format!("screenshot capture failed: {e}"),
1437                    };
1438                }
1439            }
1440        } else {
1441            None
1442        };
1443
1444        if let Some(def_name) = definition {
1445            if let Some(def) = self.definitions.get(def_name) {
1446                return self.run_assert_def(
1447                    def,
1448                    &page_content,
1449                    image.as_deref(),
1450                    step_endpoint,
1451                    test_endpoint,
1452                    tab,
1453                );
1454            }
1455            return StepResult {
1456                name: format!("[assert] {def_name}"),
1457                status: StepStatus::Failed,
1458                message: format!("definition '{def_name}' not found"),
1459            };
1460        }
1461
1462        if let Some(preset_name) = preset {
1463            // Deterministic DOM layout scan — runs JS in the browser and
1464            // never calls the LLM (free, fast, no pixel budget).
1465            if preset_name == "layout_no_issues" {
1466                return self.run_layout_preset(tab);
1467            }
1468            return self.run_preset(
1469                preset_name,
1470                assert_text,
1471                &page_content,
1472                image.as_deref(),
1473                step_endpoint,
1474                test_endpoint,
1475            );
1476        }
1477
1478        if let Some(prompt_text) = prompt {
1479            return self.run_custom(
1480                prompt_text,
1481                &page_content,
1482                image.as_deref(),
1483                step_endpoint,
1484                test_endpoint,
1485            );
1486        }
1487
1488        StepResult {
1489            name: "[assert]".into(),
1490            status: StepStatus::Skipped,
1491            message: "no definition, preset, or prompt specified".into(),
1492        }
1493    }
1494
1495    fn run_assert_def(
1496        &self,
1497        def: &AssertDefinition,
1498        page_content: &PageContent,
1499        image: Option<&[String]>,
1500        step_endpoint: Option<&str>,
1501        test_endpoint: Option<&str>,
1502        tab: &Tab,
1503    ) -> StepResult {
1504        // Agent-based definition: delegate to an A2A agent
1505        if let Some(ref agent) = def.agent {
1506            if image.is_some() {
1507                return StepResult {
1508                    name: format!("[assert] {}", def.name),
1509                    status: StepStatus::Failed,
1510                    message: "agent-backed assertions do not support screenshots".into(),
1511                };
1512            }
1513            let task = def
1514                .task_template
1515                .as_deref()
1516                .unwrap_or("Evaluate the assertion")
1517                .replace("{url}", &page_content.url)
1518                .replace("{title}", &page_content.title)
1519                .replace("{content}", &page_content.body_text)
1520                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1521
1522            return self.run_agent_step(agent, &task, &def.name);
1523        }
1524
1525        // Custom preset: system + user_template provided in the definition
1526        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1527            return self.run_custom_preset(
1528                &def.name,
1529                system,
1530                template,
1531                def.assert_text.as_deref(),
1532                page_content,
1533                image,
1534                step_endpoint,
1535                test_endpoint,
1536            );
1537        }
1538
1539        def.preset.as_ref().map_or_else(
1540            || {
1541                def.prompt.as_ref().map_or_else(
1542                    || StepResult {
1543                        name: format!("[assert] {}", def.name),
1544                        status: StepStatus::Failed,
1545                        message: "definition has no preset, prompt, or system+user_template".into(),
1546                    },
1547                    |prompt| {
1548                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1549                    },
1550                )
1551            },
1552            |preset_name| {
1553                if preset_name == "layout_no_issues" {
1554                    return self.run_layout_preset(tab);
1555                }
1556                self.run_preset(
1557                    preset_name,
1558                    def.assert_text.as_deref(),
1559                    page_content,
1560                    image,
1561                    step_endpoint,
1562                    test_endpoint,
1563                )
1564            },
1565        )
1566    }
1567
1568    #[allow(clippy::too_many_arguments)]
1569    fn run_custom_preset(
1570        &self,
1571        name: &str,
1572        system: &str,
1573        template: &str,
1574        assert_text: Option<&str>,
1575        page_content: &PageContent,
1576        image: Option<&[String]>,
1577        step_endpoint: Option<&str>,
1578        test_endpoint: Option<&str>,
1579    ) -> StepResult {
1580        let user_prompt = template
1581            .replace("{url}", &page_content.url)
1582            .replace("{title}", &page_content.title)
1583            .replace("{content}", &page_content.body_text)
1584            .replace("{expected_text}", assert_text.unwrap_or(""))
1585            .replace("{description}", "");
1586
1587        // Custom preset definitions frequently forget the {content}
1588        // placeholder — without it the LLM has no page to evaluate and
1589        // answers "I can't determine that without seeing the page". Always
1590        // append the page context unless the template already references it.
1591        let user_prompt = if template.contains("{content}") {
1592            user_prompt
1593        } else {
1594            format!(
1595                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1596                url = page_content.url,
1597                title = page_content.title,
1598                content = page_content.body_text,
1599            )
1600        };
1601
1602        self.reporter
1603            .debug(format!("assert: {name} (custom preset)"));
1604
1605        let chain = self
1606            .endpoints
1607            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1608        let sys = system.to_owned();
1609
1610        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1611
1612        response.map_or_else(
1613            |e| StepResult {
1614                name: format!("[assert] {name}"),
1615                status: StepStatus::Failed,
1616                message: format!("LLM assertion call failed: {e}"),
1617            },
1618            |(lr, _idx)| {
1619                if verdict_is_pass(&lr.content) {
1620                    StepResult {
1621                        name: format!("[assert] {name}"),
1622                        status: StepStatus::Passed,
1623                        message: "PASS".into(),
1624                    }
1625                } else {
1626                    StepResult {
1627                        name: format!("[assert] {name}"),
1628                        status: StepStatus::Failed,
1629                        message: lr.content,
1630                    }
1631                }
1632            },
1633        )
1634    }
1635
1636    fn run_preset(
1637        &self,
1638        preset_name: &str,
1639        assert_text: Option<&str>,
1640        page_content: &PageContent,
1641        image: Option<&[String]>,
1642        step_endpoint: Option<&str>,
1643        test_endpoint: Option<&str>,
1644    ) -> StepResult {
1645        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1646            return StepResult {
1647                name: format!("[assert] {preset_name}"),
1648                status: StepStatus::Failed,
1649                message: format!("unknown assertion preset: {preset_name}"),
1650            };
1651        };
1652        if preset_name.starts_with("visual_") && image.is_none() {
1653            return StepResult {
1654                name: format!("[assert] {preset_name}"),
1655                status: StepStatus::Failed,
1656                message: format!(
1657                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1658                ),
1659            };
1660        }
1661
1662        let user_prompt = preset
1663            .user_template
1664            .replace("{url}", &page_content.url)
1665            .replace("{title}", &page_content.title)
1666            .replace("{content}", &page_content.body_text)
1667            .replace("{expected_text}", assert_text.unwrap_or(""))
1668            .replace("{description}", "");
1669
1670        // Same safety net as custom presets: never let the LLM answer with
1671        // no page context at all.
1672        let user_prompt = if preset.user_template.contains("{content}") {
1673            user_prompt
1674        } else {
1675            format!(
1676                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1677                url = page_content.url,
1678                title = page_content.title,
1679                content = page_content.body_text,
1680            )
1681        };
1682
1683        self.reporter.debug(format!("assert: {preset_name}"));
1684
1685        let chain = self
1686            .endpoints
1687            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1688        let sys = preset.system.to_owned();
1689
1690        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1691
1692        response.map_or_else(
1693            |e| StepResult {
1694                name: format!("[assert] {preset_name}"),
1695                status: StepStatus::Failed,
1696                message: format!("LLM assertion call failed: {e}"),
1697            },
1698            |(lr, _idx)| {
1699                if verdict_is_pass(&lr.content) {
1700                    StepResult {
1701                        name: format!("[assert] {preset_name}"),
1702                        status: StepStatus::Passed,
1703                        message: "PASS".into(),
1704                    }
1705                } else {
1706                    StepResult {
1707                        name: format!("[assert] {preset_name}"),
1708                        status: StepStatus::Failed,
1709                        message: lr.content,
1710                    }
1711                }
1712            },
1713        )
1714    }
1715
1716    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1717    ///
1718    /// Evaluates the layout-scan JS in the page and fails with the list of
1719    /// detected issues: horizontal page overflow, visible elements sticking
1720    /// out of the viewport, text clipped by `overflow: hidden` containers,
1721    /// and interactive elements covered by other elements. No LLM call —
1722    /// checks are geometry-based so the check is free, deterministic, and
1723    /// safe to run on every page × viewport variant.
1724    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1725        let name = "[assert] layout_no_issues".to_owned();
1726        self.reporter
1727            .debug("assert: layout_no_issues (DOM layout scan)");
1728        let js = LAYOUT_SCAN_JS.replace(
1729            "__IGNORE_CLASSES__",
1730            &serde_json::to_string(&self.config.layout_ignore_classes)
1731                .unwrap_or_else(|_| "[]".to_owned()),
1732        );
1733        let result = tab.evaluate(&js, false);
1734        let json_str = match result {
1735            Ok(r) => r
1736                .value
1737                .as_ref()
1738                .and_then(|v| v.as_str().map(String::from))
1739                .unwrap_or_else(|| "[]".to_owned()),
1740            Err(e) => {
1741                return StepResult {
1742                    name,
1743                    status: StepStatus::Failed,
1744                    message: format!("layout scan JS failed: {e}"),
1745                };
1746            }
1747        };
1748        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1749        if issues.is_empty() {
1750            return StepResult {
1751                name,
1752                status: StepStatus::Passed,
1753                message: "PASS — no layout defects detected".into(),
1754            };
1755        }
1756        let mut lines: Vec<String> = issues
1757            .iter()
1758            .take(10)
1759            .map(|i| {
1760                format!(
1761                    "- [{type_}] {element}: {detail}",
1762                    type_ = i.issue_type,
1763                    element = i.element,
1764                    detail = i.detail
1765                )
1766            })
1767            .collect();
1768        if issues.len() > 10 {
1769            lines.push(format!("- … and {} more", issues.len() - 10));
1770        }
1771        StepResult {
1772            name,
1773            status: StepStatus::Failed,
1774            message: format!(
1775                "FAIL — {} layout defect(s) detected:\n{}",
1776                issues.len(),
1777                lines.join("\n")
1778            ),
1779        }
1780    }
1781
1782    fn run_custom(
1783        &self,
1784        prompt: &str,
1785        page_content: &PageContent,
1786        image: Option<&[String]>,
1787        step_endpoint: Option<&str>,
1788        test_endpoint: Option<&str>,
1789    ) -> StepResult {
1790        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1791
1792        let mut user = format!(
1793            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1794            url = page_content.url,
1795            title = page_content.title,
1796            content = page_content.body_text,
1797        );
1798        if image.is_some() {
1799            user.push_str(
1800                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1801            );
1802        }
1803
1804        self.reporter.debug("custom assert");
1805
1806        let chain = self
1807            .endpoints
1808            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1809        let sys = system.to_owned();
1810
1811        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1812
1813        response.map_or_else(
1814            |e| StepResult {
1815                name: "[assert] custom".into(),
1816                status: StepStatus::Failed,
1817                message: format!("LLM assertion call failed: {e}"),
1818            },
1819            |(lr, _idx)| {
1820                if verdict_is_pass(&lr.content) {
1821                    StepResult {
1822                        name: "[assert] custom".into(),
1823                        status: StepStatus::Passed,
1824                        message: "PASS".into(),
1825                    }
1826                } else {
1827                    StepResult {
1828                        name: "[assert] custom".into(),
1829                        status: StepStatus::Failed,
1830                        message: lr.content,
1831                    }
1832                }
1833            },
1834        )
1835    }
1836
1837    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1838        let path = path.unwrap_or("screenshot.png");
1839
1840        match tab.capture_screenshot(
1841            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1842            None,
1843            None,
1844            true,
1845        ) {
1846            Ok(data) => {
1847                if let Err(e) = std::fs::write(path, &data) {
1848                    return StepResult {
1849                        name: format!("[screenshot] {path}"),
1850                        status: StepStatus::Failed,
1851                        message: format!("failed to write screenshot: {e}"),
1852                    };
1853                }
1854                StepResult {
1855                    name: format!("[screenshot] {path}"),
1856                    status: StepStatus::Passed,
1857                    message: format!("saved to {path}"),
1858                }
1859            }
1860            Err(e) => StepResult {
1861                name: format!("[screenshot] {path}"),
1862                status: StepStatus::Failed,
1863                message: format!("screenshot failed: {e}"),
1864            },
1865        }
1866    }
1867
1868    /// Runs an A2A agent step.
1869    #[allow(clippy::literal_string_with_formatting_args)]
1870    fn run_agent(
1871        &self,
1872        agent_name: &str,
1873        task: &str,
1874        definition: Option<&str>,
1875        _test_endpoint: Option<&str>,
1876    ) -> StepResult {
1877        // If a definition is specified, look up the task template
1878        let resolved_task = if let Some(def_name) = definition {
1879            if let Some(def) = self.definitions.get(def_name) {
1880                let tmpl = def.task_template.as_deref().unwrap_or(task);
1881                tmpl.replace("{task}", task)
1882            } else {
1883                return StepResult {
1884                    name: format!("[agent] {def_name}"),
1885                    status: StepStatus::Failed,
1886                    message: format!("definition '{def_name}' not found"),
1887                };
1888            }
1889        } else {
1890            task.to_owned()
1891        };
1892
1893        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1894    }
1895
1896    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1897        let Some(ep) = self.endpoints.get(agent_name) else {
1898            return StepResult {
1899                name: format!("[agent] {display_name}"),
1900                status: StepStatus::Failed,
1901                message: format!("agent endpoint '{agent_name}' not found"),
1902            };
1903        };
1904
1905        if ep.url.is_empty() {
1906            return StepResult {
1907                name: format!("[agent] {display_name}"),
1908                status: StepStatus::Failed,
1909                message: format!("agent endpoint '{agent_name}' has no URL"),
1910            };
1911        }
1912
1913        self.reporter.debug(format!("agent {agent_name}: {task}"));
1914
1915        let url = ep.url.clone();
1916        let client = A2aClient::new(&url, self.timeout);
1917        let task_clone = task.to_owned();
1918
1919        let response = std::thread::spawn(move || {
1920            let rt = tokio::runtime::Builder::new_current_thread()
1921                .enable_all()
1922                .build()
1923                .unwrap();
1924            rt.block_on(client.send_task(&task_clone))
1925        })
1926        .join()
1927        .unwrap();
1928
1929        // Record the flat-cost call
1930        self.usage.record_flat_call(agent_name, ep);
1931
1932        match response {
1933            Ok(text) => {
1934                let clean = text.trim().to_owned();
1935                if verdict_is_pass(&clean) {
1936                    StepResult {
1937                        name: format!("[agent] {display_name}"),
1938                        status: StepStatus::Passed,
1939                        message: format!("PASS: {clean}"),
1940                    }
1941                } else if verdict_is_fail(&clean) {
1942                    StepResult {
1943                        name: format!("[agent] {display_name}"),
1944                        status: StepStatus::Failed,
1945                        message: clean,
1946                    }
1947                } else {
1948                    StepResult {
1949                        name: format!("[agent] {display_name}"),
1950                        status: StepStatus::Passed,
1951                        message: format!("response: {clean}"),
1952                    }
1953                }
1954            }
1955            Err(e) => StepResult {
1956                name: format!("[agent] {display_name}"),
1957                status: StepStatus::Failed,
1958                message: format!("agent call failed: {e}"),
1959            },
1960        }
1961    }
1962
1963    /// Runs an MCP tool call step.
1964    fn run_mcp(
1965        &self,
1966        server_name: &str,
1967        tool_name: &str,
1968        args: Option<&serde_json::Value>,
1969    ) -> StepResult {
1970        let Some(ep) = self.endpoints.get(server_name) else {
1971            return StepResult {
1972                name: format!("[mcp] {server_name}:{tool_name}"),
1973                status: StepStatus::Failed,
1974                message: format!("MCP server endpoint '{server_name}' not found"),
1975            };
1976        };
1977
1978        let cmd = ep.command.as_deref().unwrap_or("");
1979        if cmd.is_empty() {
1980            return StepResult {
1981                name: format!("[mcp] {server_name}:{tool_name}"),
1982                status: StepStatus::Failed,
1983                message: format!("MCP server '{server_name}' has no command configured"),
1984            };
1985        }
1986
1987        self.reporter
1988            .debug(format!("mcp {server_name} {tool_name}"));
1989
1990        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1991
1992        let command = cmd.to_owned();
1993        let args_vec = ep.args.clone();
1994        let tool = tool_name.to_owned();
1995
1996        let response = std::thread::spawn(move || {
1997            let mut mcp_client =
1998                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1999            mcp_client
2000                .call_tool(&tool, &args_val)
2001                .map_err(|e| e.to_string())
2002        })
2003        .join()
2004        .unwrap();
2005
2006        // Record the flat-cost call
2007        self.usage.record_flat_call(server_name, ep);
2008
2009        match response {
2010            Ok(result) => {
2011                if result.isError {
2012                    StepResult {
2013                        name: format!("[mcp] {server_name}:{tool_name}"),
2014                        status: StepStatus::Failed,
2015                        message: result.to_string(),
2016                    }
2017                } else {
2018                    StepResult {
2019                        name: format!("[mcp] {server_name}:{tool_name}"),
2020                        status: StepStatus::Passed,
2021                        message: result.to_string(),
2022                    }
2023                }
2024            }
2025            Err(e) => StepResult {
2026                name: format!("[mcp] {server_name}:{tool_name}"),
2027                status: StepStatus::Failed,
2028                message: format!("MCP call failed: {e}"),
2029            },
2030        }
2031    }
2032
2033    // ── helpers ──────────────────────────────────────────────────────────
2034
2035    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
2036    /// the runner's default LLM config for any unset fields.
2037    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
2038        LlmConfig {
2039            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
2040                // Bedrock builds its endpoint from the resolved AWS region
2041                // when no URL is given — never inherit the default LLM URL.
2042                endpoint.url.clone()
2043            } else if endpoint.url.is_empty() {
2044                self.llm.url.clone()
2045            } else {
2046                endpoint.url.clone()
2047            },
2048            model: endpoint
2049                .model
2050                .clone()
2051                .unwrap_or_else(|| self.llm.model.clone()),
2052            api_key: endpoint
2053                .api_key
2054                .clone()
2055                .or_else(|| self.llm.api_key.clone()),
2056            headers: if endpoint.headers.is_empty() {
2057                self.llm.headers.clone()
2058            } else {
2059                endpoint.headers.clone()
2060            },
2061            timeout: self.llm.timeout,
2062            temperature: self.llm.temperature,
2063            thinking: self.llm.thinking,
2064            model_params: self.llm.model_params.clone(),
2065            cache: endpoint.cache_markers,
2066            max_attempts: endpoint.max_attempts.max(1),
2067            provider: endpoint.provider,
2068            deployment: endpoint.deployment.clone(),
2069            api_version: endpoint.api_version.clone(),
2070            auth: endpoint.auth.clone(),
2071            header_commands: endpoint.header_commands.clone(),
2072            aws: endpoint.aws.clone(),
2073        }
2074    }
2075
2076    /// Runs a single LLM call against an ordered endpoint chain (primary +
2077    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
2078    /// the first endpoint that answers wins. Returns the response together
2079    /// with the chain index of the answering endpoint (0 = primary) so the
2080    /// caller can attribute usage to the correct endpoint.
2081    ///
2082    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
2083    /// shows duration, tokens, cost and the answering endpoint per call.
2084    /// Run-level context handed to every LLM call as the FIRST block of the
2085    /// user message. Contents that are stable for the whole run ("run
2086    /// started", "target site") come first so upstream provider prefix
2087    /// caching stays effective; the current time is the last line because
2088    /// it changes on every call.
2089    fn run_context(&self) -> String {
2090        let mut parts = vec![
2091            "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
2092            "================================================================".into(),
2093            format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
2094        ];
2095        if let Some(base) = self.config.base_url.as_deref() {
2096            parts.push(format!("Target site: {base}"));
2097        }
2098        let now = SystemTime::now()
2099            .duration_since(UNIX_EPOCH)
2100            .map_or(0, |d| d.as_secs());
2101        parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
2102        parts.join("\n")
2103    }
2104
2105    #[allow(clippy::cast_possible_truncation, clippy::too_many_lines)]
2106    fn llm_call_chain(
2107        &self,
2108        chain: &[&ResolvedEndpoint],
2109        system: &str,
2110        user: &str,
2111        image: Option<&[String]>,
2112        purpose: &str,
2113    ) -> Result<(crate::costs::LlmResponse, usize), String> {
2114        if chain.is_empty() {
2115            return Err("empty LLM endpoint chain".into());
2116        }
2117        let primary = self.build_llm_for_endpoint(chain[0]);
2118        let fallbacks: Vec<LlmConfig> = chain[1..]
2119            .iter()
2120            .map(|e| self.build_llm_for_endpoint(e))
2121            .collect();
2122
2123        let (test, index) = self
2124            .current_step
2125            .borrow()
2126            .as_ref()
2127            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
2128        let primary_endpoint = chain[0].name.clone();
2129        let primary_model = primary.model.clone();
2130        self.emit_event(&TestEvent::LlmCallStarted {
2131            test: test.clone(),
2132            index,
2133            endpoint: primary_endpoint.clone(),
2134            model: primary_model.clone(),
2135            purpose: purpose.to_owned(),
2136        });
2137
2138        let started = Instant::now();
2139        let sys = system.to_owned();
2140        let context = self.run_context();
2141        let user = if context.is_empty() {
2142            user.to_owned()
2143        } else {
2144            format!("{context}\n\n{user}")
2145        };
2146        let image = image.map(<[String]>::to_vec);
2147
2148        let result = std::thread::spawn(move || {
2149            let rt = tokio::runtime::Builder::new_current_thread()
2150                .enable_all()
2151                .build()
2152                .unwrap();
2153            let call = async {
2154                match image.as_deref() {
2155                    Some(img) => {
2156                        llm_chat_vision_with_usage_chain(
2157                            &primary,
2158                            &fallbacks,
2159                            &sys,
2160                            &user,
2161                            Some(img),
2162                        )
2163                        .await
2164                    }
2165                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2166                }
2167            };
2168            rt.block_on(call)
2169        })
2170        .join()
2171        .unwrap();
2172
2173        let duration_ms = started.elapsed().as_millis() as u64;
2174        match result {
2175            Ok((lr, idx)) => {
2176                let cost = calculate_llm_cost(chain[idx], &lr.usage);
2177                let answering = chain[idx].name.clone();
2178                let model = chain[idx]
2179                    .model
2180                    .clone()
2181                    .unwrap_or_else(|| primary_model.clone());
2182                self.usage
2183                    .record_llm_call(&answering, chain[idx], &model, &lr.usage);
2184                self.emit_event(&TestEvent::LlmCallFinished {
2185                    test,
2186                    index,
2187                    endpoint: answering,
2188                    model,
2189                    purpose: purpose.to_owned(),
2190                    ok: true,
2191                    duration_ms,
2192                    input_tokens: lr.usage.prompt_tokens,
2193                    output_tokens: lr.usage.completion_tokens,
2194                    cached_input_tokens: lr.usage.cached_input_tokens,
2195                    cache_creation_input_tokens: lr.usage.cache_creation_input_tokens,
2196                    cost,
2197                    error: None,
2198                });
2199                Ok((lr, idx))
2200            }
2201            Err(e) => {
2202                self.emit_event(&TestEvent::LlmCallFinished {
2203                    test,
2204                    index,
2205                    endpoint: primary_endpoint,
2206                    model: primary_model,
2207                    purpose: purpose.to_owned(),
2208                    ok: false,
2209                    duration_ms,
2210                    input_tokens: 0,
2211                    output_tokens: 0,
2212                    cached_input_tokens: 0,
2213                    cache_creation_input_tokens: 0,
2214                    cost: 0.0,
2215                    error: Some(e.clone()),
2216                });
2217                Err(e)
2218            }
2219        }
2220    }
2221
2222    /// Resolves a CSS selector for the target element. Uses the explicit
2223    /// `selector` if provided, otherwise asks the LLM to find the element
2224    /// from the natural language `target` description and page DOM.
2225    ///
2226    /// LLM responses are sanitized and verified against the live page: a
2227    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2228    /// immediately with the raw LLM output, and a selector that matches
2229    /// nothing triggers one retry with feedback before failing.
2230    #[allow(clippy::too_many_lines)]
2231    fn resolve_selector(
2232        &self,
2233        css_override: Option<&str>,
2234        target: &str,
2235        step_endpoint: Option<&str>,
2236        test_endpoint: Option<&str>,
2237        tab: &Tab,
2238    ) -> Result<String, String> {
2239        if let Some(explicit) = css_override {
2240            return Ok(explicit.to_owned());
2241        }
2242
2243        let dom_info = extract_dom_info(tab)?;
2244        let page_content = get_page_text(tab);
2245
2246        let system = concat!(
2247            "You are a browser automation selector generator. ",
2248            "Given a web page's content and interactive elements, ",
2249            "return ONLY the best CSS selector for the described element. ",
2250            "Output nothing except the CSS selector. ",
2251            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2252            "[name=\"...\"], tag.class, tag. ",
2253            "Never output explanations, markdown, or extra text."
2254        );
2255
2256        let user = format!(
2257            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2258            page_content.url,
2259            page_content.title,
2260            truncate(&page_content.body_text, 4000),
2261            dom_info,
2262            target,
2263        );
2264
2265        let retry_user = format!(
2266            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2267            "The selector must match at least one element currently present on the page.",
2268            page_content.url,
2269            page_content.title,
2270            truncate(&page_content.body_text, 4000),
2271            dom_info,
2272            target,
2273        );
2274
2275        self.reporter.debug(format!("LLM targeting: {target}"));
2276
2277        let chain = self
2278            .endpoints
2279            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2280        let sys = system.to_owned();
2281
2282        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2283
2284        let first = call_llm(&user);
2285        let (lr, _idx) = match first {
2286            Ok(lr) => lr,
2287            Err(e) => {
2288                return Err(format!("LLM element targeting failed: {e}"));
2289            }
2290        };
2291        let clean = sanitize_selector(&lr.content);
2292        self.reporter.debug(format!("resolved selector: {clean}"));
2293
2294        if selector_is_useless(&clean) {
2295            return Err(format!(
2296                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2297                raw = lr.content.trim(),
2298            ));
2299        }
2300        if let Err(reason) = validate_selector(&clean) {
2301            return Err(format!(
2302                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2303                raw = lr.content.trim(),
2304            ));
2305        }
2306        if !selector_matches(tab, &clean).unwrap_or(false) {
2307            // One retry with feedback: flaky models occasionally invent a
2308            // selector that does not exist on the page.
2309            self.reporter.warn(format!(
2310                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2311            ));
2312            let second = call_llm(&retry_user);
2313            let (lr2, _idx2) = match second {
2314                Ok(lr2) => lr2,
2315                Err(e) => {
2316                    return Err(format!(
2317                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2318                    ));
2319                }
2320            };
2321            let clean2 = sanitize_selector(&lr2.content);
2322            self.reporter
2323                .debug(format!("resolved selector (retry): {clean2}"));
2324            if selector_is_useless(&clean2) {
2325                return Err(format!(
2326                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2327                    raw = lr2.content.trim(),
2328                    excerpt = truncate(&page_content.body_text, 300),
2329                ));
2330            }
2331            if !selector_matches(tab, &clean2).unwrap_or(false) {
2332                return Err(format!(
2333                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2334                ));
2335            }
2336            return Ok(clean2);
2337        }
2338
2339        Ok(clean)
2340    }
2341}
2342
2343/// Evaluates a JS expression that is expected to return a boolean.
2344fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2345    tab.evaluate(js, false)
2346        .map_err(|e| format!("evaluate failed: {e}"))?
2347        .value
2348        .and_then(|v| v.as_bool())
2349        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2350}
2351
2352/// Checks whether a CSS selector matches at least one current element.
2353fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2354    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2355}
2356
2357// ── Free helper functions ──────────────────────────────────────────────
2358
2359fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2360    let name = format!("[navigate] {full_url}");
2361    match tab.navigate_to(full_url) {
2362        Ok(_) => {
2363            let _ = tab.wait_until_navigated();
2364            StepResult {
2365                name,
2366                status: StepStatus::Passed,
2367                message: format!("navigated to {full_url}"),
2368            }
2369        }
2370        Err(e) => StepResult {
2371            name,
2372            status: StepStatus::Failed,
2373            message: format!("navigation failed: {e}"),
2374        },
2375    }
2376}
2377
2378fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2379    let result = tab
2380        .evaluate(DOM_EXTRACT_JS, false)
2381        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2382
2383    let json_str = result
2384        .value
2385        .as_ref()
2386        .and_then(|v| v.as_str())
2387        .unwrap_or("[]");
2388
2389    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2390
2391    if elements.is_empty() {
2392        return Ok("(no interactive elements found)".to_owned());
2393    }
2394
2395    Ok(elements.join("\n"))
2396}
2397
2398fn get_page_text(tab: &Tab) -> PageContent {
2399    let url = tab.get_url();
2400
2401    let title = tab
2402        .evaluate("document.title", false)
2403        .ok()
2404        .and_then(|r| r.value)
2405        .and_then(|v| v.as_str().map(String::from))
2406        .unwrap_or_else(|| "unknown".to_owned());
2407
2408    let body_text = tab
2409        .evaluate(
2410            "document.body ? document.body.innerText : document.documentElement.innerText",
2411            false,
2412        )
2413        .ok()
2414        .and_then(|r| r.value)
2415        .and_then(|v| v.as_str().map(String::from))
2416        .unwrap_or_default();
2417
2418    PageContent {
2419        url,
2420        title,
2421        body_text: truncate(&body_text, 8000),
2422    }
2423}
2424
2425fn resolve_url(url: &str, base_url: &str) -> String {
2426    if url.starts_with("http://") || url.starts_with("https://") {
2427        return url.to_owned();
2428    }
2429    let base = base_url.trim_end_matches('/');
2430    if url.starts_with('/') {
2431        format!("{base}{url}")
2432    } else {
2433        format!("{base}/{url}")
2434    }
2435}
2436
2437/// Origin (`scheme://host[:port]`) of a base URL, used to scope the
2438/// per-test `Storage.clearDataForOrigin` call. Returns `None` when the
2439/// URL has no recognizable scheme/host (the clear is skipped).
2440#[must_use]
2441fn origin_of(base_url: &str) -> Option<String> {
2442    let url = if base_url.contains("://") {
2443        base_url.to_owned()
2444    } else {
2445        format!("https://{base_url}")
2446    };
2447    let (scheme, rest) = url.split_once("://")?;
2448    let authority = rest
2449        .split(['/', '?', '#'])
2450        .next()
2451        .filter(|a| !a.is_empty())?;
2452    Some(format!("{scheme}://{authority}"))
2453}
2454
2455/// True when a test performs its own login: its first navigate step
2456/// targets a login route (org `/auth/login`, tenant `/tenant/login`),
2457/// or it has no navigate step at all and rides the auto-navigate onto a
2458/// login `start_url`. Such tests need a cleared session — with a live one
2459/// the SPA bounces the login page before the form ever mounts. Tests
2460/// whose own first navigate goes to an app page keep the shared session
2461/// (most "page loads" tests rely on it) — a blanket clear broke exactly
2462/// those on immosai run #1643.
2463#[must_use]
2464fn test_targets_login(start_url: &str, steps: &[TestStep]) -> bool {
2465    for step in steps {
2466        if let TestStep::Navigate { url, .. } = step {
2467            return url.contains("login");
2468        }
2469    }
2470    start_url.contains("login")
2471}
2472
2473/// Human-readable label for a step, used when steps are skipped after an
2474/// earlier failure.
2475fn step_label(step: &TestStep) -> String {
2476    match step {
2477        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2478        TestStep::Click { target, .. } => format!("[click] {target}"),
2479        TestStep::Type { target, .. } => format!("[type] {target}"),
2480        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2481        TestStep::Assert {
2482            definition,
2483            preset,
2484            prompt,
2485            ..
2486        } => definition.as_ref().map_or_else(
2487            || {
2488                preset.as_ref().map_or_else(
2489                    || {
2490                        prompt.as_ref().map_or_else(
2491                            || "[assert]".to_owned(),
2492                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2493                        )
2494                    },
2495                    |p| format!("[assert] {p}"),
2496                )
2497            },
2498            |d| format!("[assert] {d}"),
2499        ),
2500        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2501        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2502        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2503    }
2504}
2505
2506/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2507#[must_use]
2508const fn step_kind_label(step: &TestStep) -> &'static str {
2509    match step {
2510        TestStep::Navigate { .. } => "navigate",
2511        TestStep::Click { .. } => "click",
2512        TestStep::Type { .. } => "type",
2513        TestStep::Wait { .. } => "wait",
2514        TestStep::Assert { .. } => "assert",
2515        TestStep::Screenshot { .. } => "screenshot",
2516        TestStep::Agent { .. } => "agent",
2517        TestStep::Mcp { .. } => "mcp",
2518    }
2519}
2520
2521// ── Support types ──────────────────────────────────────────────────────
2522
2523#[derive(Default)]
2524struct TestRunResult {
2525    passed: u32,
2526    failed: u32,
2527    skipped: u32,
2528    total: u32,
2529    details: Vec<StepResult>,
2530}
2531
2532struct PageContent {
2533    url: String,
2534    title: String,
2535    body_text: String,
2536}
2537
2538#[cfg(test)]
2539mod tests {
2540    use super::unix_to_rfc3339;
2541
2542    #[test]
2543    fn rfc3339_epoch_and_reference_dates() {
2544        assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2545        assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2546        assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2547        assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2548    }
2549
2550    #[test]
2551    fn rfc3339_handles_leap_years() {
2552        assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2553        assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2554    }
2555}
2556
2557#[cfg(test)]
2558mod verdict_parse_tests {
2559    use super::{verdict_is_fail, verdict_is_pass};
2560
2561    #[test]
2562    fn tolerates_markdown_punctuation_and_natural_language() {
2563        assert!(verdict_is_pass("PASS"));
2564        assert!(verdict_is_pass("pass"));
2565        assert!(verdict_is_pass("**PASS**\n\nThe page is clean."));
2566        assert!(verdict_is_pass("passes - no explicit error visible"));
2567        assert!(verdict_is_pass("  \"pass\""));
2568        assert!(!verdict_is_pass("FAIL: something broke"));
2569        assert!(!verdict_is_pass("**FAIL** broken"));
2570
2571        assert!(verdict_is_fail("**FAIL** broken"));
2572        assert!(verdict_is_fail("fails - error toast shown"));
2573        assert!(!verdict_is_fail("passes - ok"));
2574    }
2575}