Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// Formats a UNIX epoch timestamp as an RFC 3339 UTC string
26/// (`2026-09-19T16:04:00Z`) without pulling in a date dependency.
27/// Civil-from-days conversion after Howard Hinnant.
28#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31    let secs = i64::try_from(secs).unwrap_or(0);
32    let days = secs.div_euclid(86_400);
33    let rem = secs.rem_euclid(86_400);
34    let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35    let z = days + 719_468;
36    let era = z.div_euclid(146_097);
37    let doe = z.rem_euclid(146_097);
38    let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40    let mp = (5 * doy + 2) / 153;
41    let d = doy - (153 * mp + 2) / 5 + 1;
42    let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43    let y = yoe + era * 400 + i64::from(mth <= 2);
44    format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47/// One detected layout defect (`layout_no_issues` preset).
48#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50    #[serde(rename = "type")]
51    issue_type: String,
52    element: String,
53    detail: String,
54}
55
56/// In-browser DOM layout scan for `layout_no_issues`.
57///
58/// Geometry-only checks (no LLM, no pixels):
59/// 1. page horizontal overflow (`scrollWidth` > viewport width);
60/// 2. elements outside the viewport that scrolling cannot reveal
61///    (fixed elements off-screen, left/negative overflow, right-edge
62///    overflow beyond the horizontally scrollable area, and bottom
63///    overflow on a page that cannot scroll down) — below-the-fold
64///    content on a tall scrollable page is normal flow, NOT a defect;
65/// 3. text clipped by `overflow: hidden` containers whose content
66///    is measurably larger than the box;
67/// 4. interactive elements (buttons/links/inputs) whose center point
68///    is covered by a different element that would intercept the click.
69///
70/// Elements whose class matches a configured ignore prefix (default:
71/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
72/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
73/// skipped — they are intentionally 1x1 / off-screen. The prefix list
74/// is injected at `__IGNORE_CLASSES__` from
75/// [`ScenarioConfig::layout_ignore_classes`].
76///
77/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
78/// sticky headers) is excluded by the position/relation filters.
79const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81  const issues = [];
82  const push = (type, el, detail) => {
83    if (issues.length >= 30) return;
84    let element = el.tagName.toLowerCase();
85    if (el.id) element += '#' + el.id;
86    else if (typeof el.className === 'string' && el.className.trim())
87      element += '.' + el.className.trim().split(/\s+/).join('.');
88    issues.push({ type, element, detail: String(detail).slice(0, 220) });
89  };
90  const vw = document.documentElement.clientWidth || window.innerWidth;
91  const vh = document.documentElement.clientHeight || window.innerHeight;
92  if (!vw || !vh) return JSON.stringify(issues);
93  const de = document.documentElement;
94  // 1. Page-level horizontal overflow.
95  if (maxSW > vw + 2)
96    push('page-overflow-x', de,
97      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98  // Class-name prefixes to skip (injected; default = Angular CDK
99  // screen-reader helpers, which are intentionally 1x1 / off-screen).
100  const ignorePrefixes = __IGNORE_CLASSES__;
101  const isIgnored = (el) => {
102    if (typeof el.className !== 'string' || !el.className.trim()) return false;
103    const classes = el.className.trim().split(/\s+/);
104    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105  };
106  const body = document.body;
107  // The document element is not always the scroll container: the app may
108  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110  // "not scrollable" — measure the scrollable content height/width of the
111  // document element AND the body, and take the max.
112  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114  const vScrollable = maxSH > vh + 2;
115  const hScrollable = maxSW > vw + 2;
116  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117  // this element into view on the given axis — i.e. the element is inside a
118  // scrolling region, so lying beyond the viewport cut is not a defect.
119  const reachableByScroller = (el, axis) => {
120    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123    let p = el.parentElement;
124    while (p && p !== body) {
125      const pcs = getComputedStyle(p);
126      const o = pcs[ovProp];
127      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128          p[dim] > p[dimClient] + 2) return true;
129      p = p.parentElement;
130    }
131    return false;
132  };
133  const selfOverflowing = (el, cs, axis) => {
134    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138      el[dim] > el[dimClient] + 2;
139  };
140  // True when an ancestor clips this axis with overflow:hidden/clip — the
141  // element's overhang is not visible, so treat it as reachable.
142  const clippedByAncestor = (el, axis) => {
143    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144    let p = el.parentElement;
145    while (p && p !== body) {
146      const o = getComputedStyle(p)[ovProp];
147      if (o === 'hidden' || o === 'clip') return true;
148      p = p.parentElement;
149    }
150    return false;
151  };
152  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154  const hasContent = (el) =>
155    ((el.textContent || '').trim().length > 0) ||
156    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157  // 2. Elements outside the viewport that scrolling cannot reveal.
158  for (const el of all) {
159    const cs = getComputedStyle(el);
160    if (!visible(cs)) continue;
161    if (isIgnored(el)) continue;
162    const r = el.getBoundingClientRect();
163    if (r.width < 2 || r.height < 2) continue;
164    if (!hasContent(el) && el.children.length === 0) continue;
165    if (cs.position === 'fixed') {
166      // Fixed elements never move with the scroll: any edge outside the
167      // viewport is unreachable content and therefore a defect.
168      const overTop = -r.top;
169      const overLeft = -r.left;
170      const overRight = r.right - vw;
171      const overBottom = r.bottom - vh;
172      // Top/left overshoot is never reachable; bottom/right overshoot is
173      // only a defect when the fixed element cannot scroll that content
174      // into view itself (e.g. an opened Material drawer whose inner
175      // container scrolls is not a "cut-off" defect).
176      if (overTop > 2 || overLeft > 2 ||
177          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179        let where = '';
180        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186        push('element-out-of-viewport', el,
187          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188      }
189      continue;
190    }
191    if (cs.position === 'sticky') continue;
192    // Negative left overflow cannot be reached by scrolling (scrollLeft
193    // never goes below 0).
194    if (r.left < -2) {
195      push('element-out-of-viewport', el,
196        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197      continue;
198    }
199    // Negative top with the page at the top means the element sits above
200    // the document origin — also unreachable.
201    if (r.top < -2 && de.scrollTop <= 2) {
202      push('element-out-of-viewport', el,
203        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204      continue;
205    }
206    const overRight = r.right - vw;
207    // Right-edge overflow is only a defect when the page cannot scroll
208    // horizontally to reveal it (or the element sticks out past the
209    // scrollable content width itself).
210    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211      push('element-out-of-viewport', el,
212        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213      continue;
214    }
215    // Below-the-fold content on a scrollable page is normal (tall landing
216    // pages); only flag bottom overflow the user can never scroll to.
217    const overBottom = r.bottom - vh;
218    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219      push('element-out-of-viewport', el,
220        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221    }
222  }
223  // 3. Text clipped by overflow:hidden containers.
224  for (const el of all) {
225    if (isIgnored(el)) continue;
226    const cs = getComputedStyle(el);
227    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229    if (!(el.textContent || '').trim()) continue;
230    // Single-line truncation with an ellipsis is an intentional design
231    // pattern (Tailwind .truncate etc.), not a clipping defect.
232    if (cs.textOverflow === 'ellipsis') continue;
233    push('text-clipped', el,
234      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236  }
237  // 4. Interactive elements covered by a different element.
238  const interactive =
239    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240  const targets = document.querySelectorAll(interactive);
241  for (const el of targets) {
242    if (isIgnored(el)) continue;
243    const r = el.getBoundingClientRect();
244    if (r.width < 6 || r.height < 6) continue;
245    const cs = getComputedStyle(el);
246    if (!visible(cs)) continue;
247    const cx = r.left + r.width / 2;
248    const cy = r.top + r.height / 2;
249    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250    const top = document.elementFromPoint(cx, cy);
251    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252    if (isIgnored(top)) continue;
253    const tcs = getComputedStyle(top);
254    if (!visible(tcs)) continue;
255    if (tcs.pointerEvents === 'none') continue;
256    const tr = top.getBoundingClientRect();
257    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260    push('element-overlap', el,
261      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262      ' is covered by <' + tname + '>');
263  }
264  return JSON.stringify(issues);
265})()
266"#;
267
268/// How long the CDP connection stays open after the browser goes quiet.
269///
270/// `headless_chrome` ships a 30s default and tears down the entire connection
271/// when no traffic arrives for that long; a run must own its connection for
272/// its full duration instead.
273const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275/// Executes a [`Scenario`] against a real browser with optional LLM
276/// assistance for element targeting and assertions.
277pub struct ScenarioRunner {
278    config: ScenarioConfig,
279    definitions: HashMap<String, AssertDefinition>,
280    llm: LlmConfig,
281    timeout: Duration,
282    viewport_width: u32,
283    viewport_height: u32,
284    /// The viewport currently applied in the browser (CDP emulation).
285    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
286    applied_viewport: std::cell::Cell<(u32, u32)>,
287    endpoints: EndpointRegistry,
288    usage: Arc<UsageTracker>,
289    budgets: BudgetTracker,
290    /// Directory for failure artifacts (screenshots).
291    artifacts_dir: PathBuf,
292    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
293    reporter: Arc<Reporter>,
294    /// The test + step index currently executing, for LLM-call events.
295    current_step: std::cell::RefCell<Option<(String, u32)>>,
296    /// Epoch seconds of runner creation — constant for the whole run so
297    /// the LLM context preamble below is cacheable by upstream providers.
298    run_started: u64,
299    /// Whether this runner emits the `RunStarted`/`RunFinished` events.
300    /// When several scenario files are run concurrently by
301    /// [`crate::parallel::run_scenarios`], the orchestrator owns those two
302    /// events (one per batch), so each per-file runner suppresses its own.
303    emit_run_events: bool,
304}
305
306/// Aggregated results from a scenario run.
307#[derive(Debug, Default)]
308pub struct RunReport {
309    /// Number of tests that passed.
310    pub tests_passed: u32,
311    /// Number of tests that failed.
312    pub tests_failed: u32,
313    /// Number of steps that passed.
314    pub passed: u32,
315    /// Number of steps that failed.
316    pub failed: u32,
317    /// Number of steps that were skipped.
318    pub skipped: u32,
319    /// Per-step details.
320    pub details: Vec<StepResult>,
321}
322
323/// Result of a single step execution.
324#[derive(Debug, Clone)]
325pub struct StepResult {
326    /// The step name.
327    pub name: String,
328    /// Whether the step passed, failed, or was skipped.
329    pub status: StepStatus,
330    /// Human-readable result message.
331    pub message: String,
332}
333
334/// Outcome for a single step.
335pub use crate::events::StepStatus;
336
337/// Predefined assertion preset definition.
338struct AssertPreset {
339    name: &'static str,
340    system: &'static str,
341    user_template: &'static str,
342}
343
344/// Built-in assertion presets.
345#[allow(clippy::literal_string_with_formatting_args)]
346const ASSERTION_PRESETS: &[AssertPreset] = &[
347    AssertPreset {
348        name: "no_error_on_page",
349        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
350        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
351    },
352    AssertPreset {
353        name: "text_visible",
354        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
355        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
356    },
357    AssertPreset {
358        name: "element_exists",
359        system: "You are a QA tester. Check if a described UI element exists on a web page.",
360        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
361    },
362    AssertPreset {
363        name: "visual_no_issues",
364        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
365        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
366    },
367    AssertPreset {
368        name: "visual_no_overlaps",
369        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
370        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
371    },
372    AssertPreset {
373        name: "visual_fills_viewport",
374        system: "You are a visual QA engineer inspecting a website screenshot for UNDER-FILLED layouts: the rendered content stops at a fraction of the viewport and the remaining area is a large blank region the layout should have filled. Typical defects: page content ending around the halfway mark of the viewport with an empty, background-only area below it; an app shell, dashboard, or main panel collapsing to part of its expected height and leaving dead space; content that fills only the left portion of a wide viewport with unused space to the right. Do NOT fail intentionally short or centered pages: a login or signup card, a landing hero, a short article, a confirmation step, or an empty state that does not fill the viewport is normal design. Only fail when the blank region is clearly a layout defect — the page looks cut off or collapsed, or a content container (calendar, table, chart, map, panel) visibly fails to stretch to the space its own layout provides.",
375        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nDoes the page content fill the viewport as its layout intends — with no large blank region where the page appears to end partway down or partway across? Respond with exactly \"PASS\" if the page fills the viewport or is intentionally short, or \"FAIL: <describe the under-filled region and where it appears>\" if the content clearly stops at a fraction of the available page.",
376    },
377    AssertPreset {
378        name: "visual_text_visible",
379        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
380        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
381    },
382    AssertPreset {
383        name: "layout_no_issues",
384        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
385        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
386    },
387];
388
389/// Whether an LLM verdict should be read as PASS.
390///
391/// Models routinely wrap the verdict in markdown (`**PASS** …`) or prefix it
392/// with punctuation, which a bare `starts_with("pass")` misreads as a failure.
393/// Strip any leading non-alphanumerics, then match the leading word; this also
394/// accepts the natural-language forms `pass` / `passes`.
395fn verdict_is_pass(content: &str) -> bool {
396    content
397        .trim()
398        .trim_start_matches(|c: char| !c.is_alphanumeric())
399        .to_lowercase()
400        .starts_with("pass")
401}
402
403/// Whether an LLM verdict should be read as FAIL (see [`verdict_is_pass`]).
404fn verdict_is_fail(content: &str) -> bool {
405    content
406        .trim()
407        .trim_start_matches(|c: char| !c.is_alphanumeric())
408        .to_lowercase()
409        .starts_with("fail")
410}
411
412impl ScenarioRunner {
413    /// Creates a new runner with the given scenario configuration and
414    /// assertion definitions.
415    #[must_use]
416    #[allow(clippy::needless_pass_by_value)]
417    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
418        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
419    }
420
421    /// Creates a runner that reports run events through the given reporter.
422    #[must_use]
423    #[allow(clippy::needless_pass_by_value)]
424    pub fn with_reporter(
425        scenario_config: ScenarioConfig,
426        definitions: Vec<AssertDefinition>,
427        reporter: Arc<Reporter>,
428    ) -> Self {
429        Self::with_reporter_mode(scenario_config, definitions, reporter, true)
430    }
431
432    /// Creates a runner that reports run events through the given reporter
433    /// but suppresses the `RunStarted`/`RunFinished` events. Used by the
434    /// parallel orchestrator, which emits those once per batch.
435    #[must_use]
436    #[allow(clippy::needless_pass_by_value)]
437    pub fn with_reporter_parallel(
438        scenario_config: ScenarioConfig,
439        definitions: Vec<AssertDefinition>,
440        reporter: Arc<Reporter>,
441    ) -> Self {
442        Self::with_reporter_mode(scenario_config, definitions, reporter, false)
443    }
444
445    #[must_use]
446    #[allow(clippy::needless_pass_by_value)]
447    fn with_reporter_mode(
448        scenario_config: ScenarioConfig,
449        definitions: Vec<AssertDefinition>,
450        reporter: Arc<Reporter>,
451        emit_run_events: bool,
452    ) -> Self {
453        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
454            &scenario_config,
455        ));
456        let llm = LlmConfig {
457            url: scenario_config
458                .llm_url
459                .clone()
460                .unwrap_or_else(crate::llm_base_url),
461            model: scenario_config
462                .llm_model
463                .clone()
464                .unwrap_or_else(crate::llm_model),
465            api_key: scenario_config
466                .llm_api_key
467                .clone()
468                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
469            headers: if scenario_config.llm_headers.is_empty() {
470                crate::parse_headers_env()
471            } else {
472                scenario_config.llm_headers.clone()
473            },
474            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
475            temperature: scenario_config.temperature,
476            thinking: scenario_config.thinking,
477            model_params: scenario_config.model_params.clone(),
478            cache: scenario_config.cache.unwrap_or(true),
479            max_attempts: crate::default_llm_attempts(),
480            provider: crate::scenario::Provider::Openai,
481            deployment: None,
482            api_version: None,
483            auth: crate::scenario::AuthConfig::default(),
484            header_commands: std::collections::HashMap::new(),
485            aws: crate::scenario::AwsConfig::default(),
486        };
487        let endpoints = EndpointRegistry::from_config(
488            &scenario_config.endpoints,
489            Some(&llm),
490            crate::endpoints::EndpointDefaults {
491                cache: scenario_config.cache,
492                cache_pricing: scenario_config.cache_pricing,
493            },
494        );
495        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
496        let defs_map: HashMap<String, AssertDefinition> = definitions
497            .into_iter()
498            .map(|d| (d.name.clone(), d))
499            .collect();
500
501        Self {
502            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
503            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
504            viewport_height: scenario_config.viewport_height.unwrap_or(720),
505            applied_viewport: std::cell::Cell::new((0, 0)),
506            config: scenario_config.clone(),
507            definitions: defs_map,
508            llm,
509            endpoints,
510            usage: Arc::new(UsageTracker::new()),
511            budgets,
512            artifacts_dir: PathBuf::from(
513                scenario_config
514                    .artifacts_dir
515                    .unwrap_or_else(|| "artifacts".to_owned()),
516            ),
517            reporter,
518            current_step: std::cell::RefCell::new(None),
519            run_started: SystemTime::now()
520                .duration_since(UNIX_EPOCH)
521                .map_or(0, |d| d.as_secs()),
522            emit_run_events,
523        }
524    }
525
526    /// Emits an event; a sink failure degrades to a console warning so a
527    /// broken log file can never mask the run itself.
528    fn emit_event(&self, event: &TestEvent) {
529        if let Err(err) = self.reporter.emit(event) {
530            use std::io::Write as _;
531            let mut out = std::io::stderr().lock();
532            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
533        }
534    }
535
536    /// Returns a clone of the [`UsageTracker`] for reporting.
537    #[must_use]
538    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
539        Arc::clone(&self.usage)
540    }
541
542    /// Returns a reference to the [`BudgetTracker`].
543    #[must_use]
544    pub const fn budget_tracker(&self) -> &BudgetTracker {
545        &self.budgets
546    }
547
548    /// Executes all test groups in the scenario and returns a report.
549    ///
550    /// # Errors
551    ///
552    /// Returns an error if the browser fails to launch.
553    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
554    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
555        let mut report = RunReport::default();
556
557        if tests.is_empty() {
558            self.reporter.warn("No tests defined in scenario.");
559            if self.emit_run_events {
560                self.emit_event(&TestEvent::RunFinished {
561                    tests_passed: 0,
562                    tests_failed: 0,
563                    steps_passed: 0,
564                    steps_failed: 0,
565                    steps_skipped: 0,
566                    total_cost: 0.0,
567                    total_tokens: 0,
568                    total_input_tokens: 0,
569                    total_output_tokens: 0,
570                    total_cached_input_tokens: 0,
571                    total_cache_creation_input_tokens: 0,
572                    models: Vec::new(),
573                    total_calls: 0,
574                });
575            }
576            return Ok(report);
577        }
578
579        if self.emit_run_events {
580            self.emit_event(&TestEvent::RunStarted {
581                total_tests: tests.len() as u32,
582            });
583        }
584
585        let browser_headless = self.config.browser_headless.unwrap_or(true);
586
587        let launch_opts = LaunchOptions {
588            headless: browser_headless,
589            window_size: Some((self.viewport_width, self.viewport_height)),
590            sandbox: false,
591            // headless_chrome defaults this to 30s and shuts down the whole CDP
592            // connection when no messages arrive for that long. A scenario can
593            // easily exceed 30s of browser silence (slow LLM targeting/assertion
594            // calls, page waits, budget checks between steps), after which every
595            // remaining step fails with "Unable to make method calls because
596            // underlying connection is closed" — one quiet gap kills the run.
597            // Open-ended scenarios must own the connection for their full
598            // duration, so keep it alive for 6 hours.
599            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
600            ..LaunchOptions::default()
601        };
602
603        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
604        let tab = browser.new_tab().context("failed to open browser tab")?;
605        match (
606            self.config.browser_basic_auth_user.clone(),
607            self.config.browser_basic_auth_password.clone(),
608        ) {
609            (Some(username), Some(password)) if !username.is_empty() && !password.is_empty() => {
610                tab.authenticate(Some(username), Some(password))
611                    .context("failed to configure browser HTTP Basic Auth")?;
612                tab.enable_fetch(None, Some(true))
613                    .context("failed to enable browser HTTP authentication")?;
614            }
615            (None, None) => {}
616            _ => anyhow::bail!("browser Basic Auth requires both username and password"),
617        }
618        let _ = tab.set_default_timeout(self.timeout);
619
620        // Start MCP server if configured
621        #[cfg(feature = "mcp-server")]
622        if let Some(ref mcp_cfg) = self.config.mcp_server {
623            if mcp_cfg.enabled {
624                let port = mcp_cfg.port;
625                std::thread::spawn(move || {
626                    let _ = crate::mcp_server::start_mcp_server(port);
627                });
628            }
629        }
630        #[cfg(not(feature = "mcp-server"))]
631        if let Some(mcp_cfg) = &self.config.mcp_server {
632            if mcp_cfg.enabled {
633                self.reporter
634                    .warn("MCP server configured but 'mcp-server' feature not enabled");
635            }
636        }
637
638        // Start A2A agent server if configured
639        #[cfg(feature = "a2a-server")]
640        if let Some(ref a2a_cfg) = self.config.a2a_server {
641            if a2a_cfg.enabled {
642                let port = a2a_cfg.port;
643                tokio::spawn(crate::a2a_server::start_a2a_server(port));
644            }
645        }
646        #[cfg(not(feature = "a2a-server"))]
647        if let Some(a2a_cfg) = &self.config.a2a_server {
648            if a2a_cfg.enabled {
649                self.reporter
650                    .warn("A2A server configured but 'a2a-server' feature not enabled");
651            }
652        }
653
654        for test in tests {
655            self.emit_event(&TestEvent::TestStarted {
656                test: test.name.clone(),
657            });
658
659            self.usage.reset_per_test();
660
661            let test_started = Instant::now();
662            let test_result = self.run_test(test, &tab);
663            let duration_ms = test_started.elapsed().as_millis() as u64;
664            let usage = self.usage.current_test_snapshot();
665            self.usage.commit_test(&test.name);
666
667            self.emit_event(&TestEvent::TestFinished {
668                test: test.name.clone(),
669                passed: test_result.passed,
670                failed: test_result.failed,
671                skipped: test_result.skipped,
672                duration_ms,
673                cost: usage.total_cost,
674                tokens: usage.total_tokens,
675                input_tokens: usage.total_input_tokens,
676                output_tokens: usage.total_output_tokens,
677                cached_input_tokens: usage.total_cached_input_tokens,
678                cache_creation_input_tokens: usage.total_cache_creation_input_tokens,
679                models: usage.models.clone(),
680                calls: usage.total_calls,
681            });
682
683            if test_result.failed == 0 && test_result.total > 0 {
684                report.tests_passed += 1;
685            } else if test_result.total > 0 {
686                report.tests_failed += 1;
687            }
688
689            report.passed += test_result.passed;
690            report.failed += test_result.failed;
691            report.skipped += test_result.skipped;
692            report.details.extend(test_result.details);
693        }
694
695        let global = self.usage.global_snapshot();
696        if self.emit_run_events {
697            self.emit_event(&TestEvent::RunFinished {
698                tests_passed: report.tests_passed,
699                tests_failed: report.tests_failed,
700                steps_passed: report.passed,
701                steps_failed: report.failed,
702                steps_skipped: report.skipped,
703                total_cost: global.total_cost,
704                total_tokens: global.total_tokens,
705                total_input_tokens: global.total_input_tokens,
706                total_output_tokens: global.total_output_tokens,
707                total_cached_input_tokens: global.total_cached_input_tokens,
708                total_cache_creation_input_tokens: global.total_cache_creation_input_tokens,
709                models: global.models.clone(),
710                total_calls: global.total_calls,
711            });
712        }
713
714        Ok(report)
715    }
716
717    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
718    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
719        let base_url = test
720            .base_url
721            .clone()
722            .or_else(|| self.config.base_url.clone())
723            .unwrap_or_else(crate::base_url);
724
725        // Per-test viewport override: switch the browser via CDP
726        // device-metrics emulation before this test runs.
727        let vw = test.viewport_width.unwrap_or(self.viewport_width);
728        let vh = test.viewport_height.unwrap_or(self.viewport_height);
729        if self.applied_viewport.get() != (vw, vh) {
730            self.apply_viewport(tab, vw, vh);
731            self.applied_viewport.set((vw, vh));
732        }
733
734        // Per-test isolation: every test starts from its own start_url
735        // (unless auto_navigate is disabled), so a test never inherits the
736        // previous test's page state.
737        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
738
739        let start_url = test
740            .start_url
741            .clone()
742            .or_else(|| self.config.start_url.clone())
743            .unwrap_or_else(|| "/dashboard".to_owned());
744
745        // Per-test state isolation for LOGIN tests: the tab is shared
746        // across the file's tests, and a previous test's login persists
747        // (Cognito tokens in localStorage + the hosted-UI cookies). The
748        // SPA's login pages then auto-continue authenticated visitors on
749        // boot, so a later login test never sees the form and times out
750        // waiting for `#email` (observed on immosai runs #1641/#1642: a
751        // shard's first three logins pass, the fourth onward time out).
752        // Clear cookies + origin storage ONLY for tests that perform
753        // their own login (their first navigate step targets a login
754        // route, or they have no navigate step and ride the auto-nav
755        // onto a login start_url). The many "page loads" tests that
756        // navigate app pages directly RELY on the shared session — a
757        // blanket clear bounces them off their target (run #1643: every
758        // shard failed with "the current page is the login page, not
759        // the mailboxes page"). The HTTP cache is deliberately NOT
760        // cleared — re-downloading the SPA bundle per test would only
761        // add boot time on a slow runner egress.
762        if auto_navigate && test_targets_login(&start_url, &test.steps) {
763            use headless_chrome::protocol::cdp::{Network, Storage};
764            if let Err(e) = tab.call_method(Network::ClearBrowserCookies(None)) {
765                self.reporter
766                    .warn(format!("per-test cookie clear failed: {e}"));
767            }
768            if let Some(origin) = origin_of(&base_url) {
769                if let Err(e) = tab.call_method(Storage::ClearDataForOrigin {
770                    origin,
771                    storage_Types: "all".to_string(),
772                }) {
773                    self.reporter
774                        .warn(format!("per-test storage clear failed: {e}"));
775                }
776            }
777        }
778        if auto_navigate {
779            let full_url = resolve_url(&start_url, &base_url);
780            self.reporter.debug(format!("auto-navigate: {full_url}"));
781            let _ = tab.navigate_to(&full_url);
782            let _ = tab.wait_until_navigated();
783            std::thread::sleep(Duration::from_secs(4));
784        }
785
786        let mut result = TestRunResult::default();
787
788        for (step_index, step) in test.steps.iter().enumerate() {
789            result.total += 1;
790
791            let wait_ms = match step {
792                TestStep::Navigate { wait_after_ms, .. }
793                | TestStep::Click { wait_after_ms, .. }
794                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
795                _ => None,
796            };
797
798            self.current_step
799                .replace(Some((test.name.clone(), step_index as u32)));
800            self.emit_event(&TestEvent::StepStarted {
801                test: test.name.clone(),
802                index: step_index as u32,
803                label: step_label(step),
804            });
805            let step_started = Instant::now();
806
807            let mut step_result = match step {
808                TestStep::Navigate { url, .. } => {
809                    let full_url = resolve_url(url, &base_url);
810                    run_navigate_step(&full_url, tab)
811                }
812                TestStep::Click {
813                    target,
814                    selector,
815                    endpoint,
816                    idempotent,
817                    ..
818                } => self.run_click(
819                    target,
820                    selector.as_deref(),
821                    endpoint.as_deref(),
822                    test.endpoint.as_deref(),
823                    *idempotent,
824                    tab,
825                ),
826                TestStep::Type {
827                    target,
828                    text,
829                    selector,
830                    endpoint,
831                    idempotent,
832                    ..
833                } => self.run_type(
834                    target,
835                    text,
836                    selector.as_deref(),
837                    endpoint.as_deref(),
838                    test.endpoint.as_deref(),
839                    *idempotent,
840                    tab,
841                ),
842                TestStep::Wait {
843                    target,
844                    selector,
845                    text,
846                    timeout_ms,
847                    endpoint,
848                    idempotent,
849                } => self.run_wait(
850                    target,
851                    selector.as_deref(),
852                    text.as_deref(),
853                    *timeout_ms,
854                    endpoint.as_deref(),
855                    test.endpoint.as_deref(),
856                    *idempotent,
857                    tab,
858                ),
859                TestStep::Assert {
860                    definition,
861                    preset,
862                    prompt,
863                    assert_text,
864                    endpoint,
865                    screenshot,
866                } => self.run_assert(
867                    definition.as_deref(),
868                    preset.as_deref(),
869                    prompt.as_deref(),
870                    assert_text.as_deref(),
871                    *screenshot,
872                    endpoint.as_deref(),
873                    test.endpoint.as_deref(),
874                    tab,
875                ),
876                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
877                TestStep::Agent {
878                    agent,
879                    task,
880                    definition,
881                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
882                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
883            };
884
885            // Failure diagnostics: capture the page state and a screenshot so
886            // CI logs say WHAT the page looked like when the step failed,
887            // instead of a bare "timed out: The event waited for never came".
888            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
889                let state = diagnostics::capture(tab);
890                let screenshot = diagnostics::save_screenshot(
891                    tab,
892                    &self.artifacts_dir,
893                    &test.name,
894                    &test.name,
895                    step_index,
896                    step_kind_label(step),
897                );
898                step_result.message = format!(
899                    "{base} — {excerpt}",
900                    base = step_result.message,
901                    excerpt = diagnostics::inline_excerpt(&state),
902                );
903                (Some(diagnostics::full_context(&state)), screenshot)
904            } else {
905                (None, None)
906            };
907
908            let duration_ms = step_started.elapsed().as_millis() as u64;
909            self.emit_event(&TestEvent::StepFinished {
910                test: test.name.clone(),
911                index: step_index as u32,
912                label: step_result.name.clone(),
913                status: step_result.status,
914                duration_ms,
915                message: step_result.message.clone(),
916                diagnostics: diagnostics_block,
917                screenshot: screenshot_path,
918            });
919            self.current_step.replace(None);
920
921            match step_result.status {
922                StepStatus::Passed => result.passed += 1,
923                StepStatus::Failed => result.failed += 1,
924                StepStatus::Skipped => result.skipped += 1,
925            }
926
927            // Fail fast: the first failed step ends the test and the
928            // remaining steps are reported as skipped (no LLM budget is
929            // burned asserting against a page that is already known broken).
930            if step_result.status == StepStatus::Failed
931                && !self.config.continue_on_failure
932                && step_index + 1 < test.steps.len()
933            {
934                self.emit_event(&TestEvent::Warning {
935                    message: format!(
936                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
937                        test.steps.len() - step_index - 1
938                    ),
939                });
940                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
941                    let skipped_index = step_index + 1 + offset;
942                    let label = step_label(skipped);
943                    result.total += 1;
944                    result.skipped += 1;
945                    self.emit_event(&TestEvent::StepStarted {
946                        test: test.name.clone(),
947                        index: skipped_index as u32,
948                        label: label.clone(),
949                    });
950                    self.emit_event(&TestEvent::StepFinished {
951                        test: test.name.clone(),
952                        index: skipped_index as u32,
953                        label,
954                        status: StepStatus::Skipped,
955                        duration_ms: 0,
956                        message: "skipped: previous step failed".into(),
957                        diagnostics: None,
958                        screenshot: None,
959                    });
960                    result.details.push(StepResult {
961                        name: step_label(skipped),
962                        status: StepStatus::Skipped,
963                        message: "skipped: previous step failed".into(),
964                    });
965                }
966                result.details.push(step_result);
967                return result;
968            }
969
970            // Check per-test budget after each step
971            let test_usage = self.usage.current_test_snapshot();
972            let global_usage = self.usage.global_snapshot();
973            let budget_status = self.budgets.check_all(
974                &test.name,
975                &test_usage,
976                &global_usage,
977                test.budget.as_ref(),
978            );
979            match budget_status {
980                BudgetStatus::HardExceeded { message, .. } => {
981                    self.emit_event(&TestEvent::Warning {
982                        message: format!("budget exceeded: {message}"),
983                    });
984                    result.details.push(StepResult {
985                        name: "[budget]".into(),
986                        status: StepStatus::Failed,
987                        message,
988                    });
989                    result.failed += 1;
990                    return result;
991                }
992                BudgetStatus::SoftExceeded { message, .. } => {
993                    self.emit_event(&TestEvent::Warning {
994                        message: format!("budget warning: {message}"),
995                    });
996                }
997                BudgetStatus::Ok => {}
998            }
999
1000            if let Some(ms) = wait_ms {
1001                std::thread::sleep(Duration::from_millis(ms));
1002            }
1003
1004            result.details.push(step_result);
1005        }
1006
1007        result
1008    }
1009
1010    /// Applies a viewport size to the current tab via CDP
1011    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
1012    /// overrides and the viewport matrix. The initial window size set at
1013    /// browser launch is replaced by emulation; failures are logged but
1014    /// do not fail the test (a mismatched viewport only weakens coverage).
1015    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
1016        use headless_chrome::protocol::cdp::Emulation;
1017        let _ = self;
1018        let params = Emulation::SetDeviceMetricsOverride {
1019            width,
1020            height,
1021            device_scale_factor: 1.0,
1022            mobile: false,
1023            scale: None,
1024            screen_width: Some(width),
1025            screen_height: Some(height),
1026            position_x: None,
1027            position_y: None,
1028            dont_set_visible_size: None,
1029            screen_orientation: None,
1030            viewport: None,
1031            display_feature: None,
1032            device_posture: None,
1033        };
1034        self.reporter.debug(format!("viewport: {width}x{height}"));
1035        if let Err(e) = tab.call_method(params) {
1036            self.reporter
1037                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
1038        }
1039    }
1040
1041    /// Height (px) covered by assert-step screenshots: the configured
1042    /// `screenshot_max_height` (absolute px or viewport multiple, default
1043    /// `"20x"`) resolved against the currently applied viewport, raised to
1044    /// at least the viewport height so the visible screen is always fully
1045    /// included. The capture is split into viewport-tall tiles, so this
1046    /// value bounds total coverage (and hence the number of image parts).
1047    #[must_use]
1048    fn screenshot_height_cap(&self) -> u32 {
1049        let viewport_height = self.current_viewport_height();
1050        let cap = self
1051            .config
1052            .screenshot_max_height
1053            .as_ref()
1054            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
1055        cap.max(viewport_height)
1056    }
1057
1058    /// Height of the viewport currently emulated in the browser (falling
1059    /// back to the configured default before any emulation was applied).
1060    #[must_use]
1061    const fn current_viewport_height(&self) -> u32 {
1062        let (_, height) = self.applied_viewport.get();
1063        if height > 0 {
1064            height
1065        } else {
1066            self.viewport_height
1067        }
1068    }
1069
1070    // ── step handlers ───────────────────────────────────────────────────
1071
1072    #[allow(clippy::too_many_lines)]
1073    fn run_click(
1074        &self,
1075        target: &str,
1076        selector_override: Option<&str>,
1077        step_endpoint: Option<&str>,
1078        test_endpoint: Option<&str>,
1079        idempotent: bool,
1080        tab: &Tab,
1081    ) -> StepResult {
1082        let name = format!("[click] {target}");
1083        let selector = match self.resolve_selector(
1084            selector_override,
1085            target,
1086            step_endpoint,
1087            test_endpoint,
1088            tab,
1089        ) {
1090            Ok(s) => s,
1091            Err(msg) => {
1092                if idempotent {
1093                    return StepResult {
1094                        name,
1095                        status: StepStatus::Skipped,
1096                        message: format!("skipped (idempotent): no target found — {msg}"),
1097                    };
1098                }
1099                return StepResult {
1100                    name,
1101                    status: StepStatus::Failed,
1102                    message: msg,
1103                };
1104            }
1105        };
1106
1107        // Idempotent steps probe briefly: a missing target means the
1108        // action was already done / not applicable (e.g. an
1109        // already-authenticated session), and skipping is the success
1110        // path, not a failure.
1111        let probe_secs = if idempotent { 5 } else { 10 };
1112        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1113            Ok(element) => match element.click() {
1114                Ok(_) => StepResult {
1115                    name,
1116                    status: StepStatus::Passed,
1117                    message: format!("clicked {selector}"),
1118                },
1119                Err(e) => StepResult {
1120                    name,
1121                    status: StepStatus::Failed,
1122                    message: format!("click failed on {selector}: {e}"),
1123                },
1124            },
1125            Err(e) if idempotent => StepResult {
1126                name,
1127                status: StepStatus::Skipped,
1128                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1129            },
1130            Err(e) => StepResult {
1131                name,
1132                status: StepStatus::Failed,
1133                message: format!("element {selector} not found: {e}"),
1134            },
1135        }
1136    }
1137
1138    #[allow(clippy::too_many_arguments)]
1139    fn run_type(
1140        &self,
1141        target: &str,
1142        text: &str,
1143        selector_override: Option<&str>,
1144        step_endpoint: Option<&str>,
1145        test_endpoint: Option<&str>,
1146        idempotent: bool,
1147        tab: &Tab,
1148    ) -> StepResult {
1149        let name = format!("[type] {target}");
1150        let selector = match self.resolve_selector(
1151            selector_override,
1152            target,
1153            step_endpoint,
1154            test_endpoint,
1155            tab,
1156        ) {
1157            Ok(s) => s,
1158            Err(msg) => {
1159                if idempotent {
1160                    return StepResult {
1161                        name,
1162                        status: StepStatus::Skipped,
1163                        message: format!("skipped (idempotent): no target found — {msg}"),
1164                    };
1165                }
1166                return StepResult {
1167                    name,
1168                    status: StepStatus::Failed,
1169                    message: msg,
1170                };
1171            }
1172        };
1173
1174        let probe_secs = if idempotent { 5 } else { 10 };
1175        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1176            Ok(element) => {
1177                if let Err(e) = element.click() {
1178                    return StepResult {
1179                        name,
1180                        status: StepStatus::Failed,
1181                        message: format!("click to focus {selector} failed: {e}"),
1182                    };
1183                }
1184
1185                let js = format!(
1186                    "document.querySelector('{}').value = '';",
1187                    selector.replace('\'', "\\'")
1188                );
1189                let _ = tab.evaluate(&js, false);
1190
1191                match element.type_into(text) {
1192                    Ok(_) => StepResult {
1193                        name,
1194                        status: StepStatus::Passed,
1195                        message: format!("typed {text:?} into {selector}"),
1196                    },
1197                    Err(e) => StepResult {
1198                        name,
1199                        status: StepStatus::Failed,
1200                        message: format!("type into {selector} failed: {e}"),
1201                    },
1202                }
1203            }
1204            Err(e) if idempotent => StepResult {
1205                name,
1206                status: StepStatus::Skipped,
1207                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1208            },
1209            Err(e) => StepResult {
1210                name,
1211                status: StepStatus::Failed,
1212                message: format!("element {selector} not found: {e}"),
1213            },
1214        }
1215    }
1216
1217    #[allow(clippy::too_many_arguments)]
1218    #[allow(clippy::too_many_lines)]
1219    fn run_wait(
1220        &self,
1221        target: &str,
1222        selector_override: Option<&str>,
1223        text: Option<&str>,
1224        timeout_ms: Option<u64>,
1225        step_endpoint: Option<&str>,
1226        test_endpoint: Option<&str>,
1227        idempotent: bool,
1228        tab: &Tab,
1229    ) -> StepResult {
1230        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1231        let step_name = format!("[wait] {target}");
1232
1233        // Resolve an explicit selector only (text-only waits are LLM-free).
1234        let selector = match selector_override {
1235            Some(s) => Some(s.to_owned()),
1236            None if text.is_some() => None,
1237            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1238                Ok(s) => Some(s),
1239                Err(msg) => {
1240                    if idempotent {
1241                        return StepResult {
1242                            name: step_name,
1243                            status: StepStatus::Skipped,
1244                            message: format!("skipped (idempotent): no target found — {msg}"),
1245                        };
1246                    }
1247                    return StepResult {
1248                        name: step_name,
1249                        status: StepStatus::Failed,
1250                        message: msg,
1251                    };
1252                }
1253            },
1254        };
1255
1256        if text.is_some() {
1257            let sel_js = selector
1258                .as_deref()
1259                .map(crate::selectors::selector_matches_js);
1260            let text_js = text.map(|t| {
1261                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1262                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1263            });
1264
1265            let deadline = Instant::now() + timeout;
1266            loop {
1267                let sel_ok = sel_js
1268                    .as_ref()
1269                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1270                let text_ok = text_js
1271                    .as_ref()
1272                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1273                if sel_ok && text_ok {
1274                    let mut what = Vec::new();
1275                    if let Some(sel) = &selector {
1276                        what.push(format!("found {sel}"));
1277                    }
1278                    if let Some(t) = text {
1279                        what.push(format!("text {t:?} visible"));
1280                    }
1281                    return StepResult {
1282                        name: step_name,
1283                        status: StepStatus::Passed,
1284                        message: what.join(" and "),
1285                    };
1286                }
1287                if Instant::now() >= deadline {
1288                    let mut what = Vec::new();
1289                    if let Some(sel) = &selector {
1290                        what.push(sel.clone());
1291                    }
1292                    if let Some(t) = text {
1293                        what.push(format!("text {t:?}"));
1294                    }
1295                    let message = format!(
1296                        "wait for {} timed out after {}ms: the event waited for never came",
1297                        what.join(" / "),
1298                        timeout.as_millis(),
1299                    );
1300                    if idempotent {
1301                        return StepResult {
1302                            name: step_name,
1303                            status: StepStatus::Skipped,
1304                            message: format!("skipped (idempotent): {message}"),
1305                        };
1306                    }
1307                    return StepResult {
1308                        name: step_name,
1309                        status: StepStatus::Failed,
1310                        message,
1311                    };
1312                }
1313                std::thread::sleep(Duration::from_millis(250));
1314            }
1315        }
1316
1317        match selector.as_deref() {
1318            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1319                Ok(_) => StepResult {
1320                    name: step_name,
1321                    status: StepStatus::Passed,
1322                    message: format!("found {sel}"),
1323                },
1324                Err(e) if idempotent => StepResult {
1325                    name: step_name,
1326                    status: StepStatus::Skipped,
1327                    message: format!(
1328                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1329                        timeout.as_millis()
1330                    ),
1331                },
1332                Err(e) => StepResult {
1333                    name: step_name,
1334                    status: StepStatus::Failed,
1335                    message: format!(
1336                        "wait for {sel} timed out after {}ms: {e}",
1337                        timeout.as_millis()
1338                    ),
1339                },
1340            },
1341            None => StepResult {
1342                name: step_name,
1343                status: StepStatus::Failed,
1344                message: "wait step has neither selector nor text".into(),
1345            },
1346        }
1347    }
1348
1349    #[allow(clippy::too_many_arguments)]
1350    fn run_assert(
1351        &self,
1352        definition: Option<&str>,
1353        preset: Option<&str>,
1354        prompt: Option<&str>,
1355        assert_text: Option<&str>,
1356        screenshot: bool,
1357        step_endpoint: Option<&str>,
1358        test_endpoint: Option<&str>,
1359        tab: &Tab,
1360    ) -> StepResult {
1361        std::thread::sleep(Duration::from_millis(500));
1362
1363        let page_content = get_page_text(tab);
1364
1365        // Vision attach: capture the full page once per assert step and
1366        // split it into viewport-tall tiles (the total coverage is bounded
1367        // by the configured height cap so vision tokens stay sane). All
1368        // tile data URLs are handed to the preset/prompt evaluation below.
1369        let image: Option<Vec<String>> = if screenshot {
1370            let endpoint = self
1371                .endpoints
1372                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1373            if !endpoint.vision {
1374                return StepResult {
1375                    name: "[assert]".into(),
1376                    status: StepStatus::Failed,
1377                    message: format!(
1378                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1379                        name = endpoint.name
1380                    ),
1381                };
1382            }
1383            match crate::vision::capture_screenshot_data_urls(
1384                tab,
1385                self.config
1386                    .screenshot_max_dimension
1387                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1388                self.screenshot_height_cap(),
1389                self.current_viewport_height(),
1390            ) {
1391                Ok(urls) => Some(urls),
1392                Err(e) => {
1393                    return StepResult {
1394                        name: "[assert]".into(),
1395                        status: StepStatus::Failed,
1396                        message: format!("screenshot capture failed: {e}"),
1397                    };
1398                }
1399            }
1400        } else {
1401            None
1402        };
1403
1404        if let Some(def_name) = definition {
1405            if let Some(def) = self.definitions.get(def_name) {
1406                return self.run_assert_def(
1407                    def,
1408                    &page_content,
1409                    image.as_deref(),
1410                    step_endpoint,
1411                    test_endpoint,
1412                    tab,
1413                );
1414            }
1415            return StepResult {
1416                name: format!("[assert] {def_name}"),
1417                status: StepStatus::Failed,
1418                message: format!("definition '{def_name}' not found"),
1419            };
1420        }
1421
1422        if let Some(preset_name) = preset {
1423            // Deterministic DOM layout scan — runs JS in the browser and
1424            // never calls the LLM (free, fast, no pixel budget).
1425            if preset_name == "layout_no_issues" {
1426                return self.run_layout_preset(tab);
1427            }
1428            return self.run_preset(
1429                preset_name,
1430                assert_text,
1431                &page_content,
1432                image.as_deref(),
1433                step_endpoint,
1434                test_endpoint,
1435            );
1436        }
1437
1438        if let Some(prompt_text) = prompt {
1439            return self.run_custom(
1440                prompt_text,
1441                &page_content,
1442                image.as_deref(),
1443                step_endpoint,
1444                test_endpoint,
1445            );
1446        }
1447
1448        StepResult {
1449            name: "[assert]".into(),
1450            status: StepStatus::Skipped,
1451            message: "no definition, preset, or prompt specified".into(),
1452        }
1453    }
1454
1455    fn run_assert_def(
1456        &self,
1457        def: &AssertDefinition,
1458        page_content: &PageContent,
1459        image: Option<&[String]>,
1460        step_endpoint: Option<&str>,
1461        test_endpoint: Option<&str>,
1462        tab: &Tab,
1463    ) -> StepResult {
1464        // Agent-based definition: delegate to an A2A agent
1465        if let Some(ref agent) = def.agent {
1466            if image.is_some() {
1467                return StepResult {
1468                    name: format!("[assert] {}", def.name),
1469                    status: StepStatus::Failed,
1470                    message: "agent-backed assertions do not support screenshots".into(),
1471                };
1472            }
1473            let task = def
1474                .task_template
1475                .as_deref()
1476                .unwrap_or("Evaluate the assertion")
1477                .replace("{url}", &page_content.url)
1478                .replace("{title}", &page_content.title)
1479                .replace("{content}", &page_content.body_text)
1480                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1481
1482            return self.run_agent_step(agent, &task, &def.name);
1483        }
1484
1485        // Custom preset: system + user_template provided in the definition
1486        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1487            return self.run_custom_preset(
1488                &def.name,
1489                system,
1490                template,
1491                def.assert_text.as_deref(),
1492                page_content,
1493                image,
1494                step_endpoint,
1495                test_endpoint,
1496            );
1497        }
1498
1499        def.preset.as_ref().map_or_else(
1500            || {
1501                def.prompt.as_ref().map_or_else(
1502                    || StepResult {
1503                        name: format!("[assert] {}", def.name),
1504                        status: StepStatus::Failed,
1505                        message: "definition has no preset, prompt, or system+user_template".into(),
1506                    },
1507                    |prompt| {
1508                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1509                    },
1510                )
1511            },
1512            |preset_name| {
1513                if preset_name == "layout_no_issues" {
1514                    return self.run_layout_preset(tab);
1515                }
1516                self.run_preset(
1517                    preset_name,
1518                    def.assert_text.as_deref(),
1519                    page_content,
1520                    image,
1521                    step_endpoint,
1522                    test_endpoint,
1523                )
1524            },
1525        )
1526    }
1527
1528    #[allow(clippy::too_many_arguments)]
1529    fn run_custom_preset(
1530        &self,
1531        name: &str,
1532        system: &str,
1533        template: &str,
1534        assert_text: Option<&str>,
1535        page_content: &PageContent,
1536        image: Option<&[String]>,
1537        step_endpoint: Option<&str>,
1538        test_endpoint: Option<&str>,
1539    ) -> StepResult {
1540        let user_prompt = template
1541            .replace("{url}", &page_content.url)
1542            .replace("{title}", &page_content.title)
1543            .replace("{content}", &page_content.body_text)
1544            .replace("{expected_text}", assert_text.unwrap_or(""))
1545            .replace("{description}", "");
1546
1547        // Custom preset definitions frequently forget the {content}
1548        // placeholder — without it the LLM has no page to evaluate and
1549        // answers "I can't determine that without seeing the page". Always
1550        // append the page context unless the template already references it.
1551        let user_prompt = if template.contains("{content}") {
1552            user_prompt
1553        } else {
1554            format!(
1555                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1556                url = page_content.url,
1557                title = page_content.title,
1558                content = page_content.body_text,
1559            )
1560        };
1561
1562        self.reporter
1563            .debug(format!("assert: {name} (custom preset)"));
1564
1565        let chain = self
1566            .endpoints
1567            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1568        let sys = system.to_owned();
1569
1570        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1571
1572        response.map_or_else(
1573            |e| StepResult {
1574                name: format!("[assert] {name}"),
1575                status: StepStatus::Failed,
1576                message: format!("LLM assertion call failed: {e}"),
1577            },
1578            |(lr, _idx)| {
1579                if verdict_is_pass(&lr.content) {
1580                    StepResult {
1581                        name: format!("[assert] {name}"),
1582                        status: StepStatus::Passed,
1583                        message: "PASS".into(),
1584                    }
1585                } else {
1586                    StepResult {
1587                        name: format!("[assert] {name}"),
1588                        status: StepStatus::Failed,
1589                        message: lr.content,
1590                    }
1591                }
1592            },
1593        )
1594    }
1595
1596    fn run_preset(
1597        &self,
1598        preset_name: &str,
1599        assert_text: Option<&str>,
1600        page_content: &PageContent,
1601        image: Option<&[String]>,
1602        step_endpoint: Option<&str>,
1603        test_endpoint: Option<&str>,
1604    ) -> StepResult {
1605        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1606            return StepResult {
1607                name: format!("[assert] {preset_name}"),
1608                status: StepStatus::Failed,
1609                message: format!("unknown assertion preset: {preset_name}"),
1610            };
1611        };
1612        if preset_name.starts_with("visual_") && image.is_none() {
1613            return StepResult {
1614                name: format!("[assert] {preset_name}"),
1615                status: StepStatus::Failed,
1616                message: format!(
1617                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1618                ),
1619            };
1620        }
1621
1622        let user_prompt = preset
1623            .user_template
1624            .replace("{url}", &page_content.url)
1625            .replace("{title}", &page_content.title)
1626            .replace("{content}", &page_content.body_text)
1627            .replace("{expected_text}", assert_text.unwrap_or(""))
1628            .replace("{description}", "");
1629
1630        // Same safety net as custom presets: never let the LLM answer with
1631        // no page context at all.
1632        let user_prompt = if preset.user_template.contains("{content}") {
1633            user_prompt
1634        } else {
1635            format!(
1636                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1637                url = page_content.url,
1638                title = page_content.title,
1639                content = page_content.body_text,
1640            )
1641        };
1642
1643        self.reporter.debug(format!("assert: {preset_name}"));
1644
1645        let chain = self
1646            .endpoints
1647            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1648        let sys = preset.system.to_owned();
1649
1650        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1651
1652        response.map_or_else(
1653            |e| StepResult {
1654                name: format!("[assert] {preset_name}"),
1655                status: StepStatus::Failed,
1656                message: format!("LLM assertion call failed: {e}"),
1657            },
1658            |(lr, _idx)| {
1659                if verdict_is_pass(&lr.content) {
1660                    StepResult {
1661                        name: format!("[assert] {preset_name}"),
1662                        status: StepStatus::Passed,
1663                        message: "PASS".into(),
1664                    }
1665                } else {
1666                    StepResult {
1667                        name: format!("[assert] {preset_name}"),
1668                        status: StepStatus::Failed,
1669                        message: lr.content,
1670                    }
1671                }
1672            },
1673        )
1674    }
1675
1676    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1677    ///
1678    /// Evaluates the layout-scan JS in the page and fails with the list of
1679    /// detected issues: horizontal page overflow, visible elements sticking
1680    /// out of the viewport, text clipped by `overflow: hidden` containers,
1681    /// and interactive elements covered by other elements. No LLM call —
1682    /// checks are geometry-based so the check is free, deterministic, and
1683    /// safe to run on every page × viewport variant.
1684    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1685        let name = "[assert] layout_no_issues".to_owned();
1686        self.reporter
1687            .debug("assert: layout_no_issues (DOM layout scan)");
1688        let js = LAYOUT_SCAN_JS.replace(
1689            "__IGNORE_CLASSES__",
1690            &serde_json::to_string(&self.config.layout_ignore_classes)
1691                .unwrap_or_else(|_| "[]".to_owned()),
1692        );
1693        let result = tab.evaluate(&js, false);
1694        let json_str = match result {
1695            Ok(r) => r
1696                .value
1697                .as_ref()
1698                .and_then(|v| v.as_str().map(String::from))
1699                .unwrap_or_else(|| "[]".to_owned()),
1700            Err(e) => {
1701                return StepResult {
1702                    name,
1703                    status: StepStatus::Failed,
1704                    message: format!("layout scan JS failed: {e}"),
1705                };
1706            }
1707        };
1708        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1709        if issues.is_empty() {
1710            return StepResult {
1711                name,
1712                status: StepStatus::Passed,
1713                message: "PASS — no layout defects detected".into(),
1714            };
1715        }
1716        let mut lines: Vec<String> = issues
1717            .iter()
1718            .take(10)
1719            .map(|i| {
1720                format!(
1721                    "- [{type_}] {element}: {detail}",
1722                    type_ = i.issue_type,
1723                    element = i.element,
1724                    detail = i.detail
1725                )
1726            })
1727            .collect();
1728        if issues.len() > 10 {
1729            lines.push(format!("- … and {} more", issues.len() - 10));
1730        }
1731        StepResult {
1732            name,
1733            status: StepStatus::Failed,
1734            message: format!(
1735                "FAIL — {} layout defect(s) detected:\n{}",
1736                issues.len(),
1737                lines.join("\n")
1738            ),
1739        }
1740    }
1741
1742    fn run_custom(
1743        &self,
1744        prompt: &str,
1745        page_content: &PageContent,
1746        image: Option<&[String]>,
1747        step_endpoint: Option<&str>,
1748        test_endpoint: Option<&str>,
1749    ) -> StepResult {
1750        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1751
1752        let mut user = format!(
1753            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1754            url = page_content.url,
1755            title = page_content.title,
1756            content = page_content.body_text,
1757        );
1758        if image.is_some() {
1759            user.push_str(
1760                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1761            );
1762        }
1763
1764        self.reporter.debug("custom assert");
1765
1766        let chain = self
1767            .endpoints
1768            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1769        let sys = system.to_owned();
1770
1771        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1772
1773        response.map_or_else(
1774            |e| StepResult {
1775                name: "[assert] custom".into(),
1776                status: StepStatus::Failed,
1777                message: format!("LLM assertion call failed: {e}"),
1778            },
1779            |(lr, _idx)| {
1780                if verdict_is_pass(&lr.content) {
1781                    StepResult {
1782                        name: "[assert] custom".into(),
1783                        status: StepStatus::Passed,
1784                        message: "PASS".into(),
1785                    }
1786                } else {
1787                    StepResult {
1788                        name: "[assert] custom".into(),
1789                        status: StepStatus::Failed,
1790                        message: lr.content,
1791                    }
1792                }
1793            },
1794        )
1795    }
1796
1797    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1798        let path = path.unwrap_or("screenshot.png");
1799
1800        match tab.capture_screenshot(
1801            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1802            None,
1803            None,
1804            true,
1805        ) {
1806            Ok(data) => {
1807                if let Err(e) = std::fs::write(path, &data) {
1808                    return StepResult {
1809                        name: format!("[screenshot] {path}"),
1810                        status: StepStatus::Failed,
1811                        message: format!("failed to write screenshot: {e}"),
1812                    };
1813                }
1814                StepResult {
1815                    name: format!("[screenshot] {path}"),
1816                    status: StepStatus::Passed,
1817                    message: format!("saved to {path}"),
1818                }
1819            }
1820            Err(e) => StepResult {
1821                name: format!("[screenshot] {path}"),
1822                status: StepStatus::Failed,
1823                message: format!("screenshot failed: {e}"),
1824            },
1825        }
1826    }
1827
1828    /// Runs an A2A agent step.
1829    #[allow(clippy::literal_string_with_formatting_args)]
1830    fn run_agent(
1831        &self,
1832        agent_name: &str,
1833        task: &str,
1834        definition: Option<&str>,
1835        _test_endpoint: Option<&str>,
1836    ) -> StepResult {
1837        // If a definition is specified, look up the task template
1838        let resolved_task = if let Some(def_name) = definition {
1839            if let Some(def) = self.definitions.get(def_name) {
1840                let tmpl = def.task_template.as_deref().unwrap_or(task);
1841                tmpl.replace("{task}", task)
1842            } else {
1843                return StepResult {
1844                    name: format!("[agent] {def_name}"),
1845                    status: StepStatus::Failed,
1846                    message: format!("definition '{def_name}' not found"),
1847                };
1848            }
1849        } else {
1850            task.to_owned()
1851        };
1852
1853        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1854    }
1855
1856    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1857        let Some(ep) = self.endpoints.get(agent_name) else {
1858            return StepResult {
1859                name: format!("[agent] {display_name}"),
1860                status: StepStatus::Failed,
1861                message: format!("agent endpoint '{agent_name}' not found"),
1862            };
1863        };
1864
1865        if ep.url.is_empty() {
1866            return StepResult {
1867                name: format!("[agent] {display_name}"),
1868                status: StepStatus::Failed,
1869                message: format!("agent endpoint '{agent_name}' has no URL"),
1870            };
1871        }
1872
1873        self.reporter.debug(format!("agent {agent_name}: {task}"));
1874
1875        let url = ep.url.clone();
1876        let client = A2aClient::new(&url, self.timeout);
1877        let task_clone = task.to_owned();
1878
1879        let response = std::thread::spawn(move || {
1880            let rt = tokio::runtime::Builder::new_current_thread()
1881                .enable_all()
1882                .build()
1883                .unwrap();
1884            rt.block_on(client.send_task(&task_clone))
1885        })
1886        .join()
1887        .unwrap();
1888
1889        // Record the flat-cost call
1890        self.usage.record_flat_call(agent_name, ep);
1891
1892        match response {
1893            Ok(text) => {
1894                let clean = text.trim().to_owned();
1895                if verdict_is_pass(&clean) {
1896                    StepResult {
1897                        name: format!("[agent] {display_name}"),
1898                        status: StepStatus::Passed,
1899                        message: format!("PASS: {clean}"),
1900                    }
1901                } else if verdict_is_fail(&clean) {
1902                    StepResult {
1903                        name: format!("[agent] {display_name}"),
1904                        status: StepStatus::Failed,
1905                        message: clean,
1906                    }
1907                } else {
1908                    StepResult {
1909                        name: format!("[agent] {display_name}"),
1910                        status: StepStatus::Passed,
1911                        message: format!("response: {clean}"),
1912                    }
1913                }
1914            }
1915            Err(e) => StepResult {
1916                name: format!("[agent] {display_name}"),
1917                status: StepStatus::Failed,
1918                message: format!("agent call failed: {e}"),
1919            },
1920        }
1921    }
1922
1923    /// Runs an MCP tool call step.
1924    fn run_mcp(
1925        &self,
1926        server_name: &str,
1927        tool_name: &str,
1928        args: Option<&serde_json::Value>,
1929    ) -> StepResult {
1930        let Some(ep) = self.endpoints.get(server_name) else {
1931            return StepResult {
1932                name: format!("[mcp] {server_name}:{tool_name}"),
1933                status: StepStatus::Failed,
1934                message: format!("MCP server endpoint '{server_name}' not found"),
1935            };
1936        };
1937
1938        let cmd = ep.command.as_deref().unwrap_or("");
1939        if cmd.is_empty() {
1940            return StepResult {
1941                name: format!("[mcp] {server_name}:{tool_name}"),
1942                status: StepStatus::Failed,
1943                message: format!("MCP server '{server_name}' has no command configured"),
1944            };
1945        }
1946
1947        self.reporter
1948            .debug(format!("mcp {server_name} {tool_name}"));
1949
1950        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1951
1952        let command = cmd.to_owned();
1953        let args_vec = ep.args.clone();
1954        let tool = tool_name.to_owned();
1955
1956        let response = std::thread::spawn(move || {
1957            let mut mcp_client =
1958                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1959            mcp_client
1960                .call_tool(&tool, &args_val)
1961                .map_err(|e| e.to_string())
1962        })
1963        .join()
1964        .unwrap();
1965
1966        // Record the flat-cost call
1967        self.usage.record_flat_call(server_name, ep);
1968
1969        match response {
1970            Ok(result) => {
1971                if result.isError {
1972                    StepResult {
1973                        name: format!("[mcp] {server_name}:{tool_name}"),
1974                        status: StepStatus::Failed,
1975                        message: result.to_string(),
1976                    }
1977                } else {
1978                    StepResult {
1979                        name: format!("[mcp] {server_name}:{tool_name}"),
1980                        status: StepStatus::Passed,
1981                        message: result.to_string(),
1982                    }
1983                }
1984            }
1985            Err(e) => StepResult {
1986                name: format!("[mcp] {server_name}:{tool_name}"),
1987                status: StepStatus::Failed,
1988                message: format!("MCP call failed: {e}"),
1989            },
1990        }
1991    }
1992
1993    // ── helpers ──────────────────────────────────────────────────────────
1994
1995    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1996    /// the runner's default LLM config for any unset fields.
1997    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1998        LlmConfig {
1999            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
2000                // Bedrock builds its endpoint from the resolved AWS region
2001                // when no URL is given — never inherit the default LLM URL.
2002                endpoint.url.clone()
2003            } else if endpoint.url.is_empty() {
2004                self.llm.url.clone()
2005            } else {
2006                endpoint.url.clone()
2007            },
2008            model: endpoint
2009                .model
2010                .clone()
2011                .unwrap_or_else(|| self.llm.model.clone()),
2012            api_key: endpoint
2013                .api_key
2014                .clone()
2015                .or_else(|| self.llm.api_key.clone()),
2016            headers: if endpoint.headers.is_empty() {
2017                self.llm.headers.clone()
2018            } else {
2019                endpoint.headers.clone()
2020            },
2021            timeout: self.llm.timeout,
2022            temperature: self.llm.temperature,
2023            thinking: self.llm.thinking,
2024            model_params: self.llm.model_params.clone(),
2025            cache: endpoint.cache_markers,
2026            max_attempts: endpoint.max_attempts.max(1),
2027            provider: endpoint.provider,
2028            deployment: endpoint.deployment.clone(),
2029            api_version: endpoint.api_version.clone(),
2030            auth: endpoint.auth.clone(),
2031            header_commands: endpoint.header_commands.clone(),
2032            aws: endpoint.aws.clone(),
2033        }
2034    }
2035
2036    /// Runs a single LLM call against an ordered endpoint chain (primary +
2037    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
2038    /// the first endpoint that answers wins. Returns the response together
2039    /// with the chain index of the answering endpoint (0 = primary) so the
2040    /// caller can attribute usage to the correct endpoint.
2041    ///
2042    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
2043    /// shows duration, tokens, cost and the answering endpoint per call.
2044    /// Run-level context handed to every LLM call as the FIRST block of the
2045    /// user message. Contents that are stable for the whole run ("run
2046    /// started", "target site") come first so upstream provider prefix
2047    /// caching stays effective; the current time is the last line because
2048    /// it changes on every call.
2049    fn run_context(&self) -> String {
2050        let mut parts = vec![
2051            "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
2052            "================================================================".into(),
2053            format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
2054        ];
2055        if let Some(base) = self.config.base_url.as_deref() {
2056            parts.push(format!("Target site: {base}"));
2057        }
2058        let now = SystemTime::now()
2059            .duration_since(UNIX_EPOCH)
2060            .map_or(0, |d| d.as_secs());
2061        parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
2062        parts.join("\n")
2063    }
2064
2065    #[allow(clippy::cast_possible_truncation, clippy::too_many_lines)]
2066    fn llm_call_chain(
2067        &self,
2068        chain: &[&ResolvedEndpoint],
2069        system: &str,
2070        user: &str,
2071        image: Option<&[String]>,
2072        purpose: &str,
2073    ) -> Result<(crate::costs::LlmResponse, usize), String> {
2074        if chain.is_empty() {
2075            return Err("empty LLM endpoint chain".into());
2076        }
2077        let primary = self.build_llm_for_endpoint(chain[0]);
2078        let fallbacks: Vec<LlmConfig> = chain[1..]
2079            .iter()
2080            .map(|e| self.build_llm_for_endpoint(e))
2081            .collect();
2082
2083        let (test, index) = self
2084            .current_step
2085            .borrow()
2086            .as_ref()
2087            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
2088        let primary_endpoint = chain[0].name.clone();
2089        let primary_model = primary.model.clone();
2090        self.emit_event(&TestEvent::LlmCallStarted {
2091            test: test.clone(),
2092            index,
2093            endpoint: primary_endpoint.clone(),
2094            model: primary_model.clone(),
2095            purpose: purpose.to_owned(),
2096        });
2097
2098        let started = Instant::now();
2099        let sys = system.to_owned();
2100        let context = self.run_context();
2101        let user = if context.is_empty() {
2102            user.to_owned()
2103        } else {
2104            format!("{context}\n\n{user}")
2105        };
2106        let image = image.map(<[String]>::to_vec);
2107
2108        let result = std::thread::spawn(move || {
2109            let rt = tokio::runtime::Builder::new_current_thread()
2110                .enable_all()
2111                .build()
2112                .unwrap();
2113            let call = async {
2114                match image.as_deref() {
2115                    Some(img) => {
2116                        llm_chat_vision_with_usage_chain(
2117                            &primary,
2118                            &fallbacks,
2119                            &sys,
2120                            &user,
2121                            Some(img),
2122                        )
2123                        .await
2124                    }
2125                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2126                }
2127            };
2128            rt.block_on(call)
2129        })
2130        .join()
2131        .unwrap();
2132
2133        let duration_ms = started.elapsed().as_millis() as u64;
2134        match result {
2135            Ok((lr, idx)) => {
2136                let cost = calculate_llm_cost(chain[idx], &lr.usage);
2137                let answering = chain[idx].name.clone();
2138                let model = chain[idx]
2139                    .model
2140                    .clone()
2141                    .unwrap_or_else(|| primary_model.clone());
2142                self.usage
2143                    .record_llm_call(&answering, chain[idx], &model, &lr.usage);
2144                self.emit_event(&TestEvent::LlmCallFinished {
2145                    test,
2146                    index,
2147                    endpoint: answering,
2148                    model,
2149                    purpose: purpose.to_owned(),
2150                    ok: true,
2151                    duration_ms,
2152                    input_tokens: lr.usage.prompt_tokens,
2153                    output_tokens: lr.usage.completion_tokens,
2154                    cached_input_tokens: lr.usage.cached_input_tokens,
2155                    cache_creation_input_tokens: lr.usage.cache_creation_input_tokens,
2156                    cost,
2157                    error: None,
2158                });
2159                Ok((lr, idx))
2160            }
2161            Err(e) => {
2162                self.emit_event(&TestEvent::LlmCallFinished {
2163                    test,
2164                    index,
2165                    endpoint: primary_endpoint,
2166                    model: primary_model,
2167                    purpose: purpose.to_owned(),
2168                    ok: false,
2169                    duration_ms,
2170                    input_tokens: 0,
2171                    output_tokens: 0,
2172                    cached_input_tokens: 0,
2173                    cache_creation_input_tokens: 0,
2174                    cost: 0.0,
2175                    error: Some(e.clone()),
2176                });
2177                Err(e)
2178            }
2179        }
2180    }
2181
2182    /// Resolves a CSS selector for the target element. Uses the explicit
2183    /// `selector` if provided, otherwise asks the LLM to find the element
2184    /// from the natural language `target` description and page DOM.
2185    ///
2186    /// LLM responses are sanitized and verified against the live page: a
2187    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2188    /// immediately with the raw LLM output, and a selector that matches
2189    /// nothing triggers one retry with feedback before failing.
2190    #[allow(clippy::too_many_lines)]
2191    fn resolve_selector(
2192        &self,
2193        css_override: Option<&str>,
2194        target: &str,
2195        step_endpoint: Option<&str>,
2196        test_endpoint: Option<&str>,
2197        tab: &Tab,
2198    ) -> Result<String, String> {
2199        if let Some(explicit) = css_override {
2200            return Ok(explicit.to_owned());
2201        }
2202
2203        let dom_info = extract_dom_info(tab)?;
2204        let page_content = get_page_text(tab);
2205
2206        let system = concat!(
2207            "You are a browser automation selector generator. ",
2208            "Given a web page's content and interactive elements, ",
2209            "return ONLY the best CSS selector for the described element. ",
2210            "Output nothing except the CSS selector. ",
2211            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2212            "[name=\"...\"], tag.class, tag. ",
2213            "Never output explanations, markdown, or extra text."
2214        );
2215
2216        let user = format!(
2217            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2218            page_content.url,
2219            page_content.title,
2220            truncate(&page_content.body_text, 4000),
2221            dom_info,
2222            target,
2223        );
2224
2225        let retry_user = format!(
2226            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2227            "The selector must match at least one element currently present on the page.",
2228            page_content.url,
2229            page_content.title,
2230            truncate(&page_content.body_text, 4000),
2231            dom_info,
2232            target,
2233        );
2234
2235        self.reporter.debug(format!("LLM targeting: {target}"));
2236
2237        let chain = self
2238            .endpoints
2239            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2240        let sys = system.to_owned();
2241
2242        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2243
2244        let first = call_llm(&user);
2245        let (lr, _idx) = match first {
2246            Ok(lr) => lr,
2247            Err(e) => {
2248                return Err(format!("LLM element targeting failed: {e}"));
2249            }
2250        };
2251        let clean = sanitize_selector(&lr.content);
2252        self.reporter.debug(format!("resolved selector: {clean}"));
2253
2254        if selector_is_useless(&clean) {
2255            return Err(format!(
2256                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2257                raw = lr.content.trim(),
2258            ));
2259        }
2260        if let Err(reason) = validate_selector(&clean) {
2261            return Err(format!(
2262                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2263                raw = lr.content.trim(),
2264            ));
2265        }
2266        if !selector_matches(tab, &clean).unwrap_or(false) {
2267            // One retry with feedback: flaky models occasionally invent a
2268            // selector that does not exist on the page.
2269            self.reporter.warn(format!(
2270                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2271            ));
2272            let second = call_llm(&retry_user);
2273            let (lr2, _idx2) = match second {
2274                Ok(lr2) => lr2,
2275                Err(e) => {
2276                    return Err(format!(
2277                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2278                    ));
2279                }
2280            };
2281            let clean2 = sanitize_selector(&lr2.content);
2282            self.reporter
2283                .debug(format!("resolved selector (retry): {clean2}"));
2284            if selector_is_useless(&clean2) {
2285                return Err(format!(
2286                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2287                    raw = lr2.content.trim(),
2288                    excerpt = truncate(&page_content.body_text, 300),
2289                ));
2290            }
2291            if !selector_matches(tab, &clean2).unwrap_or(false) {
2292                return Err(format!(
2293                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2294                ));
2295            }
2296            return Ok(clean2);
2297        }
2298
2299        Ok(clean)
2300    }
2301}
2302
2303/// Evaluates a JS expression that is expected to return a boolean.
2304fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2305    tab.evaluate(js, false)
2306        .map_err(|e| format!("evaluate failed: {e}"))?
2307        .value
2308        .and_then(|v| v.as_bool())
2309        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2310}
2311
2312/// Checks whether a CSS selector matches at least one current element.
2313fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2314    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2315}
2316
2317// ── Free helper functions ──────────────────────────────────────────────
2318
2319fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2320    let name = format!("[navigate] {full_url}");
2321    match tab.navigate_to(full_url) {
2322        Ok(_) => {
2323            let _ = tab.wait_until_navigated();
2324            StepResult {
2325                name,
2326                status: StepStatus::Passed,
2327                message: format!("navigated to {full_url}"),
2328            }
2329        }
2330        Err(e) => StepResult {
2331            name,
2332            status: StepStatus::Failed,
2333            message: format!("navigation failed: {e}"),
2334        },
2335    }
2336}
2337
2338fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2339    let result = tab
2340        .evaluate(DOM_EXTRACT_JS, false)
2341        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2342
2343    let json_str = result
2344        .value
2345        .as_ref()
2346        .and_then(|v| v.as_str())
2347        .unwrap_or("[]");
2348
2349    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2350
2351    if elements.is_empty() {
2352        return Ok("(no interactive elements found)".to_owned());
2353    }
2354
2355    Ok(elements.join("\n"))
2356}
2357
2358fn get_page_text(tab: &Tab) -> PageContent {
2359    let url = tab.get_url();
2360
2361    let title = tab
2362        .evaluate("document.title", false)
2363        .ok()
2364        .and_then(|r| r.value)
2365        .and_then(|v| v.as_str().map(String::from))
2366        .unwrap_or_else(|| "unknown".to_owned());
2367
2368    let body_text = tab
2369        .evaluate(
2370            "document.body ? document.body.innerText : document.documentElement.innerText",
2371            false,
2372        )
2373        .ok()
2374        .and_then(|r| r.value)
2375        .and_then(|v| v.as_str().map(String::from))
2376        .unwrap_or_default();
2377
2378    PageContent {
2379        url,
2380        title,
2381        body_text: truncate(&body_text, 8000),
2382    }
2383}
2384
2385fn resolve_url(url: &str, base_url: &str) -> String {
2386    if url.starts_with("http://") || url.starts_with("https://") {
2387        return url.to_owned();
2388    }
2389    let base = base_url.trim_end_matches('/');
2390    if url.starts_with('/') {
2391        format!("{base}{url}")
2392    } else {
2393        format!("{base}/{url}")
2394    }
2395}
2396
2397/// Origin (`scheme://host[:port]`) of a base URL, used to scope the
2398/// per-test `Storage.clearDataForOrigin` call. Returns `None` when the
2399/// URL has no recognizable scheme/host (the clear is skipped).
2400#[must_use]
2401fn origin_of(base_url: &str) -> Option<String> {
2402    let url = if base_url.contains("://") {
2403        base_url.to_owned()
2404    } else {
2405        format!("https://{base_url}")
2406    };
2407    let (scheme, rest) = url.split_once("://")?;
2408    let authority = rest
2409        .split(['/', '?', '#'])
2410        .next()
2411        .filter(|a| !a.is_empty())?;
2412    Some(format!("{scheme}://{authority}"))
2413}
2414
2415/// True when a test performs its own login: its first navigate step
2416/// targets a login route (org `/auth/login`, tenant `/tenant/login`),
2417/// or it has no navigate step at all and rides the auto-navigate onto a
2418/// login `start_url`. Such tests need a cleared session — with a live one
2419/// the SPA bounces the login page before the form ever mounts. Tests
2420/// whose own first navigate goes to an app page keep the shared session
2421/// (most "page loads" tests rely on it) — a blanket clear broke exactly
2422/// those on immosai run #1643.
2423#[must_use]
2424fn test_targets_login(start_url: &str, steps: &[TestStep]) -> bool {
2425    let first_navigate = steps.iter().find_map(|s| match s {
2426        TestStep::Navigate { url, .. } => Some(url.clone()),
2427        _ => None,
2428    });
2429    match first_navigate {
2430        Some(url) => url.contains("login"),
2431        // No own navigation: the test rides the auto-navigated start_url.
2432        None => start_url.contains("login"),
2433    }
2434}
2435
2436/// Human-readable label for a step, used when steps are skipped after an
2437/// earlier failure.
2438fn step_label(step: &TestStep) -> String {
2439    match step {
2440        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2441        TestStep::Click { target, .. } => format!("[click] {target}"),
2442        TestStep::Type { target, .. } => format!("[type] {target}"),
2443        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2444        TestStep::Assert {
2445            definition,
2446            preset,
2447            prompt,
2448            ..
2449        } => definition.as_ref().map_or_else(
2450            || {
2451                preset.as_ref().map_or_else(
2452                    || {
2453                        prompt.as_ref().map_or_else(
2454                            || "[assert]".to_owned(),
2455                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2456                        )
2457                    },
2458                    |p| format!("[assert] {p}"),
2459                )
2460            },
2461            |d| format!("[assert] {d}"),
2462        ),
2463        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2464        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2465        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2466    }
2467}
2468
2469/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2470#[must_use]
2471const fn step_kind_label(step: &TestStep) -> &'static str {
2472    match step {
2473        TestStep::Navigate { .. } => "navigate",
2474        TestStep::Click { .. } => "click",
2475        TestStep::Type { .. } => "type",
2476        TestStep::Wait { .. } => "wait",
2477        TestStep::Assert { .. } => "assert",
2478        TestStep::Screenshot { .. } => "screenshot",
2479        TestStep::Agent { .. } => "agent",
2480        TestStep::Mcp { .. } => "mcp",
2481    }
2482}
2483
2484// ── Support types ──────────────────────────────────────────────────────
2485
2486#[derive(Default)]
2487struct TestRunResult {
2488    passed: u32,
2489    failed: u32,
2490    skipped: u32,
2491    total: u32,
2492    details: Vec<StepResult>,
2493}
2494
2495struct PageContent {
2496    url: String,
2497    title: String,
2498    body_text: String,
2499}
2500
2501#[cfg(test)]
2502mod tests {
2503    use super::unix_to_rfc3339;
2504
2505    #[test]
2506    fn rfc3339_epoch_and_reference_dates() {
2507        assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2508        assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2509        assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2510        assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2511    }
2512
2513    #[test]
2514    fn rfc3339_handles_leap_years() {
2515        assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2516        assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2517    }
2518}
2519
2520#[cfg(test)]
2521mod verdict_parse_tests {
2522    use super::{verdict_is_fail, verdict_is_pass};
2523
2524    #[test]
2525    fn tolerates_markdown_punctuation_and_natural_language() {
2526        assert!(verdict_is_pass("PASS"));
2527        assert!(verdict_is_pass("pass"));
2528        assert!(verdict_is_pass("**PASS**\n\nThe page is clean."));
2529        assert!(verdict_is_pass("passes - no explicit error visible"));
2530        assert!(verdict_is_pass("  \"pass\""));
2531        assert!(!verdict_is_pass("FAIL: something broke"));
2532        assert!(!verdict_is_pass("**FAIL** broken"));
2533
2534        assert!(verdict_is_fail("**FAIL** broken"));
2535        assert!(verdict_is_fail("fails - error toast shown"));
2536        assert!(!verdict_is_fail("passes - ok"));
2537    }
2538}