Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// Formats a UNIX epoch timestamp as an RFC 3339 UTC string
26/// (`2026-09-19T16:04:00Z`) without pulling in a date dependency.
27/// Civil-from-days conversion after Howard Hinnant.
28#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31    let secs = i64::try_from(secs).unwrap_or(0);
32    let days = secs.div_euclid(86_400);
33    let rem = secs.rem_euclid(86_400);
34    let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35    let z = days + 719_468;
36    let era = z.div_euclid(146_097);
37    let doe = z.rem_euclid(146_097);
38    let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40    let mp = (5 * doy + 2) / 153;
41    let d = doy - (153 * mp + 2) / 5 + 1;
42    let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43    let y = yoe + era * 400 + i64::from(mth <= 2);
44    format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47/// One detected layout defect (`layout_no_issues` preset).
48#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50    #[serde(rename = "type")]
51    issue_type: String,
52    element: String,
53    detail: String,
54}
55
56/// In-browser DOM layout scan for `layout_no_issues`.
57///
58/// Geometry-only checks (no LLM, no pixels):
59/// 1. page horizontal overflow (`scrollWidth` > viewport width);
60/// 2. elements outside the viewport that scrolling cannot reveal
61///    (fixed elements off-screen, left/negative overflow, right-edge
62///    overflow beyond the horizontally scrollable area, and bottom
63///    overflow on a page that cannot scroll down) — below-the-fold
64///    content on a tall scrollable page is normal flow, NOT a defect;
65/// 3. text clipped by `overflow: hidden` containers whose content
66///    is measurably larger than the box;
67/// 4. interactive elements (buttons/links/inputs) whose center point
68///    is covered by a different element that would intercept the click.
69///
70/// Elements whose class matches a configured ignore prefix (default:
71/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
72/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
73/// skipped — they are intentionally 1x1 / off-screen. The prefix list
74/// is injected at `__IGNORE_CLASSES__` from
75/// [`ScenarioConfig::layout_ignore_classes`].
76///
77/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
78/// sticky headers) is excluded by the position/relation filters.
79const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81  const issues = [];
82  const push = (type, el, detail) => {
83    if (issues.length >= 30) return;
84    let element = el.tagName.toLowerCase();
85    if (el.id) element += '#' + el.id;
86    else if (typeof el.className === 'string' && el.className.trim())
87      element += '.' + el.className.trim().split(/\s+/).join('.');
88    issues.push({ type, element, detail: String(detail).slice(0, 220) });
89  };
90  const vw = document.documentElement.clientWidth || window.innerWidth;
91  const vh = document.documentElement.clientHeight || window.innerHeight;
92  if (!vw || !vh) return JSON.stringify(issues);
93  const de = document.documentElement;
94  // 1. Page-level horizontal overflow.
95  if (maxSW > vw + 2)
96    push('page-overflow-x', de,
97      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98  // Class-name prefixes to skip (injected; default = Angular CDK
99  // screen-reader helpers, which are intentionally 1x1 / off-screen).
100  const ignorePrefixes = __IGNORE_CLASSES__;
101  const isIgnored = (el) => {
102    if (typeof el.className !== 'string' || !el.className.trim()) return false;
103    const classes = el.className.trim().split(/\s+/);
104    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105  };
106  const body = document.body;
107  // The document element is not always the scroll container: the app may
108  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110  // "not scrollable" — measure the scrollable content height/width of the
111  // document element AND the body, and take the max.
112  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114  const vScrollable = maxSH > vh + 2;
115  const hScrollable = maxSW > vw + 2;
116  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117  // this element into view on the given axis — i.e. the element is inside a
118  // scrolling region, so lying beyond the viewport cut is not a defect.
119  const reachableByScroller = (el, axis) => {
120    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123    let p = el.parentElement;
124    while (p && p !== body) {
125      const pcs = getComputedStyle(p);
126      const o = pcs[ovProp];
127      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128          p[dim] > p[dimClient] + 2) return true;
129      p = p.parentElement;
130    }
131    return false;
132  };
133  const selfOverflowing = (el, cs, axis) => {
134    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138      el[dim] > el[dimClient] + 2;
139  };
140  // True when an ancestor clips this axis with overflow:hidden/clip — the
141  // element's overhang is not visible, so treat it as reachable.
142  const clippedByAncestor = (el, axis) => {
143    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144    let p = el.parentElement;
145    while (p && p !== body) {
146      const o = getComputedStyle(p)[ovProp];
147      if (o === 'hidden' || o === 'clip') return true;
148      p = p.parentElement;
149    }
150    return false;
151  };
152  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154  const hasContent = (el) =>
155    ((el.textContent || '').trim().length > 0) ||
156    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157  // 2. Elements outside the viewport that scrolling cannot reveal.
158  for (const el of all) {
159    const cs = getComputedStyle(el);
160    if (!visible(cs)) continue;
161    if (isIgnored(el)) continue;
162    const r = el.getBoundingClientRect();
163    if (r.width < 2 || r.height < 2) continue;
164    if (!hasContent(el) && el.children.length === 0) continue;
165    if (cs.position === 'fixed') {
166      // Fixed elements never move with the scroll: any edge outside the
167      // viewport is unreachable content and therefore a defect.
168      const overTop = -r.top;
169      const overLeft = -r.left;
170      const overRight = r.right - vw;
171      const overBottom = r.bottom - vh;
172      // Top/left overshoot is never reachable; bottom/right overshoot is
173      // only a defect when the fixed element cannot scroll that content
174      // into view itself (e.g. an opened Material drawer whose inner
175      // container scrolls is not a "cut-off" defect).
176      if (overTop > 2 || overLeft > 2 ||
177          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179        let where = '';
180        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186        push('element-out-of-viewport', el,
187          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188      }
189      continue;
190    }
191    if (cs.position === 'sticky') continue;
192    // Negative left overflow cannot be reached by scrolling (scrollLeft
193    // never goes below 0).
194    if (r.left < -2) {
195      push('element-out-of-viewport', el,
196        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197      continue;
198    }
199    // Negative top with the page at the top means the element sits above
200    // the document origin — also unreachable.
201    if (r.top < -2 && de.scrollTop <= 2) {
202      push('element-out-of-viewport', el,
203        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204      continue;
205    }
206    const overRight = r.right - vw;
207    // Right-edge overflow is only a defect when the page cannot scroll
208    // horizontally to reveal it (or the element sticks out past the
209    // scrollable content width itself).
210    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211      push('element-out-of-viewport', el,
212        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213      continue;
214    }
215    // Below-the-fold content on a scrollable page is normal (tall landing
216    // pages); only flag bottom overflow the user can never scroll to.
217    const overBottom = r.bottom - vh;
218    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219      push('element-out-of-viewport', el,
220        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221    }
222  }
223  // 3. Text clipped by overflow:hidden containers.
224  for (const el of all) {
225    if (isIgnored(el)) continue;
226    const cs = getComputedStyle(el);
227    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229    if (!(el.textContent || '').trim()) continue;
230    // Single-line truncation with an ellipsis is an intentional design
231    // pattern (Tailwind .truncate etc.), not a clipping defect.
232    if (cs.textOverflow === 'ellipsis') continue;
233    push('text-clipped', el,
234      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236  }
237  // 4. Interactive elements covered by a different element.
238  const interactive =
239    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240  const targets = document.querySelectorAll(interactive);
241  for (const el of targets) {
242    if (isIgnored(el)) continue;
243    const r = el.getBoundingClientRect();
244    if (r.width < 6 || r.height < 6) continue;
245    const cs = getComputedStyle(el);
246    if (!visible(cs)) continue;
247    const cx = r.left + r.width / 2;
248    const cy = r.top + r.height / 2;
249    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250    const top = document.elementFromPoint(cx, cy);
251    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252    if (isIgnored(top)) continue;
253    const tcs = getComputedStyle(top);
254    if (!visible(tcs)) continue;
255    if (tcs.pointerEvents === 'none') continue;
256    const tr = top.getBoundingClientRect();
257    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260    push('element-overlap', el,
261      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262      ' is covered by <' + tname + '>');
263  }
264  return JSON.stringify(issues);
265})()
266"#;
267
268/// How long the CDP connection stays open after the browser goes quiet.
269///
270/// `headless_chrome` ships a 30s default and tears down the entire connection
271/// when no traffic arrives for that long; a run must own its connection for
272/// its full duration instead.
273const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275/// Executes a [`Scenario`] against a real browser with optional LLM
276/// assistance for element targeting and assertions.
277pub struct ScenarioRunner {
278    config: ScenarioConfig,
279    definitions: HashMap<String, AssertDefinition>,
280    llm: LlmConfig,
281    timeout: Duration,
282    viewport_width: u32,
283    viewport_height: u32,
284    /// The viewport currently applied in the browser (CDP emulation).
285    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
286    applied_viewport: std::cell::Cell<(u32, u32)>,
287    endpoints: EndpointRegistry,
288    usage: Arc<UsageTracker>,
289    budgets: BudgetTracker,
290    /// Directory for failure artifacts (screenshots).
291    artifacts_dir: PathBuf,
292    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
293    reporter: Arc<Reporter>,
294    /// The test + step index currently executing, for LLM-call events.
295    current_step: std::cell::RefCell<Option<(String, u32)>>,
296    /// Epoch seconds of runner creation — constant for the whole run so
297    /// the LLM context preamble below is cacheable by upstream providers.
298    run_started: u64,
299    /// Whether this runner emits the `RunStarted`/`RunFinished` events.
300    /// When several scenario files are run concurrently by
301    /// [`crate::parallel::run_scenarios`], the orchestrator owns those two
302    /// events (one per batch), so each per-file runner suppresses its own.
303    emit_run_events: bool,
304}
305
306/// Aggregated results from a scenario run.
307#[derive(Debug, Default)]
308pub struct RunReport {
309    /// Number of tests that passed.
310    pub tests_passed: u32,
311    /// Number of tests that failed.
312    pub tests_failed: u32,
313    /// Number of steps that passed.
314    pub passed: u32,
315    /// Number of steps that failed.
316    pub failed: u32,
317    /// Number of steps that were skipped.
318    pub skipped: u32,
319    /// Per-step details.
320    pub details: Vec<StepResult>,
321}
322
323/// Result of a single step execution.
324#[derive(Debug, Clone)]
325pub struct StepResult {
326    /// The step name.
327    pub name: String,
328    /// Whether the step passed, failed, or was skipped.
329    pub status: StepStatus,
330    /// Human-readable result message.
331    pub message: String,
332}
333
334/// Outcome for a single step.
335pub use crate::events::StepStatus;
336
337/// Predefined assertion preset definition.
338struct AssertPreset {
339    name: &'static str,
340    system: &'static str,
341    user_template: &'static str,
342}
343
344/// Built-in assertion presets.
345#[allow(clippy::literal_string_with_formatting_args)]
346const ASSERTION_PRESETS: &[AssertPreset] = &[
347    AssertPreset {
348        name: "no_error_on_page",
349        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
350        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
351    },
352    AssertPreset {
353        name: "text_visible",
354        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
355        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
356    },
357    AssertPreset {
358        name: "element_exists",
359        system: "You are a QA tester. Check if a described UI element exists on a web page.",
360        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
361    },
362    AssertPreset {
363        name: "visual_no_issues",
364        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
365        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
366    },
367    AssertPreset {
368        name: "visual_no_overlaps",
369        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
370        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
371    },
372    AssertPreset {
373        name: "visual_text_visible",
374        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
375        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
376    },
377    AssertPreset {
378        name: "layout_no_issues",
379        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
380        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
381    },
382];
383
384impl ScenarioRunner {
385    /// Creates a new runner with the given scenario configuration and
386    /// assertion definitions.
387    #[must_use]
388    #[allow(clippy::needless_pass_by_value)]
389    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
390        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
391    }
392
393    /// Creates a runner that reports run events through the given reporter.
394    #[must_use]
395    #[allow(clippy::needless_pass_by_value)]
396    pub fn with_reporter(
397        scenario_config: ScenarioConfig,
398        definitions: Vec<AssertDefinition>,
399        reporter: Arc<Reporter>,
400    ) -> Self {
401        Self::with_reporter_mode(scenario_config, definitions, reporter, true)
402    }
403
404    /// Creates a runner that reports run events through the given reporter
405    /// but suppresses the `RunStarted`/`RunFinished` events. Used by the
406    /// parallel orchestrator, which emits those once per batch.
407    #[must_use]
408    #[allow(clippy::needless_pass_by_value)]
409    pub fn with_reporter_parallel(
410        scenario_config: ScenarioConfig,
411        definitions: Vec<AssertDefinition>,
412        reporter: Arc<Reporter>,
413    ) -> Self {
414        Self::with_reporter_mode(scenario_config, definitions, reporter, false)
415    }
416
417    #[must_use]
418    #[allow(clippy::needless_pass_by_value)]
419    fn with_reporter_mode(
420        scenario_config: ScenarioConfig,
421        definitions: Vec<AssertDefinition>,
422        reporter: Arc<Reporter>,
423        emit_run_events: bool,
424    ) -> Self {
425        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
426            &scenario_config,
427        ));
428        let llm = LlmConfig {
429            url: scenario_config
430                .llm_url
431                .clone()
432                .unwrap_or_else(crate::llm_base_url),
433            model: scenario_config
434                .llm_model
435                .clone()
436                .unwrap_or_else(crate::llm_model),
437            api_key: scenario_config
438                .llm_api_key
439                .clone()
440                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
441            headers: if scenario_config.llm_headers.is_empty() {
442                crate::parse_headers_env()
443            } else {
444                scenario_config.llm_headers.clone()
445            },
446            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
447            temperature: scenario_config.temperature,
448            thinking: scenario_config.thinking,
449            model_params: scenario_config.model_params.clone(),
450            max_attempts: crate::default_llm_attempts(),
451            provider: crate::scenario::Provider::Openai,
452            deployment: None,
453            api_version: None,
454            auth: crate::scenario::AuthConfig::default(),
455            header_commands: std::collections::HashMap::new(),
456            aws: crate::scenario::AwsConfig::default(),
457        };
458        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
459        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
460        let defs_map: HashMap<String, AssertDefinition> = definitions
461            .into_iter()
462            .map(|d| (d.name.clone(), d))
463            .collect();
464
465        Self {
466            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
467            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
468            viewport_height: scenario_config.viewport_height.unwrap_or(720),
469            applied_viewport: std::cell::Cell::new((0, 0)),
470            config: scenario_config.clone(),
471            definitions: defs_map,
472            llm,
473            endpoints,
474            usage: Arc::new(UsageTracker::new()),
475            budgets,
476            artifacts_dir: PathBuf::from(
477                scenario_config
478                    .artifacts_dir
479                    .unwrap_or_else(|| "artifacts".to_owned()),
480            ),
481            reporter,
482            current_step: std::cell::RefCell::new(None),
483            run_started: SystemTime::now()
484                .duration_since(UNIX_EPOCH)
485                .map_or(0, |d| d.as_secs()),
486            emit_run_events,
487        }
488    }
489
490    /// Emits an event; a sink failure degrades to a console warning so a
491    /// broken log file can never mask the run itself.
492    fn emit_event(&self, event: &TestEvent) {
493        if let Err(err) = self.reporter.emit(event) {
494            use std::io::Write as _;
495            let mut out = std::io::stderr().lock();
496            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
497        }
498    }
499
500    /// Returns a clone of the [`UsageTracker`] for reporting.
501    #[must_use]
502    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
503        Arc::clone(&self.usage)
504    }
505
506    /// Returns a reference to the [`BudgetTracker`].
507    #[must_use]
508    pub const fn budget_tracker(&self) -> &BudgetTracker {
509        &self.budgets
510    }
511
512    /// Executes all test groups in the scenario and returns a report.
513    ///
514    /// # Errors
515    ///
516    /// Returns an error if the browser fails to launch.
517    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
518    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
519        let mut report = RunReport::default();
520
521        if tests.is_empty() {
522            self.reporter.warn("No tests defined in scenario.");
523            if self.emit_run_events {
524                self.emit_event(&TestEvent::RunFinished {
525                    tests_passed: 0,
526                    tests_failed: 0,
527                    steps_passed: 0,
528                    steps_failed: 0,
529                    steps_skipped: 0,
530                    total_cost: 0.0,
531                    total_tokens: 0,
532                    total_input_tokens: 0,
533                    total_output_tokens: 0,
534                    total_cached_input_tokens: 0,
535                    models: Vec::new(),
536                    total_calls: 0,
537                });
538            }
539            return Ok(report);
540        }
541
542        if self.emit_run_events {
543            self.emit_event(&TestEvent::RunStarted {
544                total_tests: tests.len() as u32,
545            });
546        }
547
548        let browser_headless = self.config.browser_headless.unwrap_or(true);
549
550        let launch_opts = LaunchOptions {
551            headless: browser_headless,
552            window_size: Some((self.viewport_width, self.viewport_height)),
553            sandbox: false,
554            // headless_chrome defaults this to 30s and shuts down the whole CDP
555            // connection when no messages arrive for that long. A scenario can
556            // easily exceed 30s of browser silence (slow LLM targeting/assertion
557            // calls, page waits, budget checks between steps), after which every
558            // remaining step fails with "Unable to make method calls because
559            // underlying connection is closed" — one quiet gap kills the run.
560            // Open-ended scenarios must own the connection for their full
561            // duration, so keep it alive for 6 hours.
562            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
563            ..LaunchOptions::default()
564        };
565
566        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
567        let tab = browser.new_tab().context("failed to open browser tab")?;
568        let _ = tab.set_default_timeout(self.timeout);
569
570        // Start MCP server if configured
571        #[cfg(feature = "mcp-server")]
572        if let Some(ref mcp_cfg) = self.config.mcp_server {
573            if mcp_cfg.enabled {
574                let port = mcp_cfg.port;
575                std::thread::spawn(move || {
576                    let _ = crate::mcp_server::start_mcp_server(port);
577                });
578            }
579        }
580        #[cfg(not(feature = "mcp-server"))]
581        if let Some(mcp_cfg) = &self.config.mcp_server {
582            if mcp_cfg.enabled {
583                self.reporter
584                    .warn("MCP server configured but 'mcp-server' feature not enabled");
585            }
586        }
587
588        // Start A2A agent server if configured
589        #[cfg(feature = "a2a-server")]
590        if let Some(ref a2a_cfg) = self.config.a2a_server {
591            if a2a_cfg.enabled {
592                let port = a2a_cfg.port;
593                tokio::spawn(crate::a2a_server::start_a2a_server(port));
594            }
595        }
596        #[cfg(not(feature = "a2a-server"))]
597        if let Some(a2a_cfg) = &self.config.a2a_server {
598            if a2a_cfg.enabled {
599                self.reporter
600                    .warn("A2A server configured but 'a2a-server' feature not enabled");
601            }
602        }
603
604        for test in tests {
605            self.emit_event(&TestEvent::TestStarted {
606                test: test.name.clone(),
607            });
608
609            self.usage.reset_per_test();
610
611            let test_started = Instant::now();
612            let test_result = self.run_test(test, &tab);
613            let duration_ms = test_started.elapsed().as_millis() as u64;
614            let usage = self.usage.current_test_snapshot();
615            self.usage.commit_test(&test.name);
616
617            self.emit_event(&TestEvent::TestFinished {
618                test: test.name.clone(),
619                passed: test_result.passed,
620                failed: test_result.failed,
621                skipped: test_result.skipped,
622                duration_ms,
623                cost: usage.total_cost,
624                tokens: usage.total_tokens,
625                input_tokens: usage.total_input_tokens,
626                output_tokens: usage.total_output_tokens,
627                cached_input_tokens: usage.total_cached_input_tokens,
628                models: usage.models.clone(),
629                calls: usage.total_calls,
630            });
631
632            if test_result.failed == 0 && test_result.total > 0 {
633                report.tests_passed += 1;
634            } else if test_result.total > 0 {
635                report.tests_failed += 1;
636            }
637
638            report.passed += test_result.passed;
639            report.failed += test_result.failed;
640            report.skipped += test_result.skipped;
641            report.details.extend(test_result.details);
642        }
643
644        let global = self.usage.global_snapshot();
645        if self.emit_run_events {
646            self.emit_event(&TestEvent::RunFinished {
647                tests_passed: report.tests_passed,
648                tests_failed: report.tests_failed,
649                steps_passed: report.passed,
650                steps_failed: report.failed,
651                steps_skipped: report.skipped,
652                total_cost: global.total_cost,
653                total_tokens: global.total_tokens,
654                total_input_tokens: global.total_input_tokens,
655                total_output_tokens: global.total_output_tokens,
656                total_cached_input_tokens: global.total_cached_input_tokens,
657                models: global.models.clone(),
658                total_calls: global.total_calls,
659            });
660        }
661
662        Ok(report)
663    }
664
665    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
666    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
667        let base_url = test
668            .base_url
669            .clone()
670            .or_else(|| self.config.base_url.clone())
671            .unwrap_or_else(crate::base_url);
672
673        // Per-test viewport override: switch the browser via CDP
674        // device-metrics emulation before this test runs.
675        let vw = test.viewport_width.unwrap_or(self.viewport_width);
676        let vh = test.viewport_height.unwrap_or(self.viewport_height);
677        if self.applied_viewport.get() != (vw, vh) {
678            self.apply_viewport(tab, vw, vh);
679            self.applied_viewport.set((vw, vh));
680        }
681
682        // Per-test isolation: every test starts from its own start_url
683        // (unless auto_navigate is disabled), so a test never inherits the
684        // previous test's page state.
685        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
686
687        let start_url = test
688            .start_url
689            .clone()
690            .or_else(|| self.config.start_url.clone())
691            .unwrap_or_else(|| "/dashboard".to_owned());
692
693        if auto_navigate {
694            let full_url = resolve_url(&start_url, &base_url);
695            self.reporter.debug(format!("auto-navigate: {full_url}"));
696            let _ = tab.navigate_to(&full_url);
697            let _ = tab.wait_until_navigated();
698            std::thread::sleep(Duration::from_secs(4));
699        }
700
701        let mut result = TestRunResult::default();
702
703        for (step_index, step) in test.steps.iter().enumerate() {
704            result.total += 1;
705
706            let wait_ms = match step {
707                TestStep::Navigate { wait_after_ms, .. }
708                | TestStep::Click { wait_after_ms, .. }
709                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
710                _ => None,
711            };
712
713            self.current_step
714                .replace(Some((test.name.clone(), step_index as u32)));
715            self.emit_event(&TestEvent::StepStarted {
716                test: test.name.clone(),
717                index: step_index as u32,
718                label: step_label(step),
719            });
720            let step_started = Instant::now();
721
722            let mut step_result = match step {
723                TestStep::Navigate { url, .. } => {
724                    let full_url = resolve_url(url, &base_url);
725                    run_navigate_step(&full_url, tab)
726                }
727                TestStep::Click {
728                    target,
729                    selector,
730                    endpoint,
731                    idempotent,
732                    ..
733                } => self.run_click(
734                    target,
735                    selector.as_deref(),
736                    endpoint.as_deref(),
737                    test.endpoint.as_deref(),
738                    *idempotent,
739                    tab,
740                ),
741                TestStep::Type {
742                    target,
743                    text,
744                    selector,
745                    endpoint,
746                    idempotent,
747                    ..
748                } => self.run_type(
749                    target,
750                    text,
751                    selector.as_deref(),
752                    endpoint.as_deref(),
753                    test.endpoint.as_deref(),
754                    *idempotent,
755                    tab,
756                ),
757                TestStep::Wait {
758                    target,
759                    selector,
760                    text,
761                    timeout_ms,
762                    endpoint,
763                    idempotent,
764                } => self.run_wait(
765                    target,
766                    selector.as_deref(),
767                    text.as_deref(),
768                    *timeout_ms,
769                    endpoint.as_deref(),
770                    test.endpoint.as_deref(),
771                    *idempotent,
772                    tab,
773                ),
774                TestStep::Assert {
775                    definition,
776                    preset,
777                    prompt,
778                    assert_text,
779                    endpoint,
780                    screenshot,
781                } => self.run_assert(
782                    definition.as_deref(),
783                    preset.as_deref(),
784                    prompt.as_deref(),
785                    assert_text.as_deref(),
786                    *screenshot,
787                    endpoint.as_deref(),
788                    test.endpoint.as_deref(),
789                    tab,
790                ),
791                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
792                TestStep::Agent {
793                    agent,
794                    task,
795                    definition,
796                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
797                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
798            };
799
800            // Failure diagnostics: capture the page state and a screenshot so
801            // CI logs say WHAT the page looked like when the step failed,
802            // instead of a bare "timed out: The event waited for never came".
803            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
804                let state = diagnostics::capture(tab);
805                let screenshot = diagnostics::save_screenshot(
806                    tab,
807                    &self.artifacts_dir,
808                    &test.name,
809                    &test.name,
810                    step_index,
811                    step_kind_label(step),
812                );
813                step_result.message = format!(
814                    "{base} — {excerpt}",
815                    base = step_result.message,
816                    excerpt = diagnostics::inline_excerpt(&state),
817                );
818                (Some(diagnostics::full_context(&state)), screenshot)
819            } else {
820                (None, None)
821            };
822
823            let duration_ms = step_started.elapsed().as_millis() as u64;
824            self.emit_event(&TestEvent::StepFinished {
825                test: test.name.clone(),
826                index: step_index as u32,
827                label: step_result.name.clone(),
828                status: step_result.status,
829                duration_ms,
830                message: step_result.message.clone(),
831                diagnostics: diagnostics_block,
832                screenshot: screenshot_path,
833            });
834            self.current_step.replace(None);
835
836            match step_result.status {
837                StepStatus::Passed => result.passed += 1,
838                StepStatus::Failed => result.failed += 1,
839                StepStatus::Skipped => result.skipped += 1,
840            }
841
842            // Fail fast: the first failed step ends the test and the
843            // remaining steps are reported as skipped (no LLM budget is
844            // burned asserting against a page that is already known broken).
845            if step_result.status == StepStatus::Failed
846                && !self.config.continue_on_failure
847                && step_index + 1 < test.steps.len()
848            {
849                self.emit_event(&TestEvent::Warning {
850                    message: format!(
851                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
852                        test.steps.len() - step_index - 1
853                    ),
854                });
855                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
856                    let skipped_index = step_index + 1 + offset;
857                    let label = step_label(skipped);
858                    result.total += 1;
859                    result.skipped += 1;
860                    self.emit_event(&TestEvent::StepStarted {
861                        test: test.name.clone(),
862                        index: skipped_index as u32,
863                        label: label.clone(),
864                    });
865                    self.emit_event(&TestEvent::StepFinished {
866                        test: test.name.clone(),
867                        index: skipped_index as u32,
868                        label,
869                        status: StepStatus::Skipped,
870                        duration_ms: 0,
871                        message: "skipped: previous step failed".into(),
872                        diagnostics: None,
873                        screenshot: None,
874                    });
875                    result.details.push(StepResult {
876                        name: step_label(skipped),
877                        status: StepStatus::Skipped,
878                        message: "skipped: previous step failed".into(),
879                    });
880                }
881                result.details.push(step_result);
882                return result;
883            }
884
885            // Check per-test budget after each step
886            let test_usage = self.usage.current_test_snapshot();
887            let global_usage = self.usage.global_snapshot();
888            let budget_status = self.budgets.check_all(
889                &test.name,
890                &test_usage,
891                &global_usage,
892                test.budget.as_ref(),
893            );
894            match budget_status {
895                BudgetStatus::HardExceeded { message, .. } => {
896                    self.emit_event(&TestEvent::Warning {
897                        message: format!("budget exceeded: {message}"),
898                    });
899                    result.details.push(StepResult {
900                        name: "[budget]".into(),
901                        status: StepStatus::Failed,
902                        message,
903                    });
904                    result.failed += 1;
905                    return result;
906                }
907                BudgetStatus::SoftExceeded { message, .. } => {
908                    self.emit_event(&TestEvent::Warning {
909                        message: format!("budget warning: {message}"),
910                    });
911                }
912                BudgetStatus::Ok => {}
913            }
914
915            if let Some(ms) = wait_ms {
916                std::thread::sleep(Duration::from_millis(ms));
917            }
918
919            result.details.push(step_result);
920        }
921
922        result
923    }
924
925    /// Applies a viewport size to the current tab via CDP
926    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
927    /// overrides and the viewport matrix. The initial window size set at
928    /// browser launch is replaced by emulation; failures are logged but
929    /// do not fail the test (a mismatched viewport only weakens coverage).
930    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
931        use headless_chrome::protocol::cdp::Emulation;
932        let _ = self;
933        let params = Emulation::SetDeviceMetricsOverride {
934            width,
935            height,
936            device_scale_factor: 1.0,
937            mobile: false,
938            scale: None,
939            screen_width: Some(width),
940            screen_height: Some(height),
941            position_x: None,
942            position_y: None,
943            dont_set_visible_size: None,
944            screen_orientation: None,
945            viewport: None,
946            display_feature: None,
947            device_posture: None,
948        };
949        self.reporter.debug(format!("viewport: {width}x{height}"));
950        if let Err(e) = tab.call_method(params) {
951            self.reporter
952                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
953        }
954    }
955
956    /// Height (px) covered by assert-step screenshots: the configured
957    /// `screenshot_max_height` (absolute px or viewport multiple, default
958    /// `"20x"`) resolved against the currently applied viewport, raised to
959    /// at least the viewport height so the visible screen is always fully
960    /// included. The capture is split into viewport-tall tiles, so this
961    /// value bounds total coverage (and hence the number of image parts).
962    #[must_use]
963    fn screenshot_height_cap(&self) -> u32 {
964        let viewport_height = self.current_viewport_height();
965        let cap = self
966            .config
967            .screenshot_max_height
968            .as_ref()
969            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
970        cap.max(viewport_height)
971    }
972
973    /// Height of the viewport currently emulated in the browser (falling
974    /// back to the configured default before any emulation was applied).
975    #[must_use]
976    const fn current_viewport_height(&self) -> u32 {
977        let (_, height) = self.applied_viewport.get();
978        if height > 0 {
979            height
980        } else {
981            self.viewport_height
982        }
983    }
984
985    // ── step handlers ───────────────────────────────────────────────────
986
987    #[allow(clippy::too_many_lines)]
988    fn run_click(
989        &self,
990        target: &str,
991        selector_override: Option<&str>,
992        step_endpoint: Option<&str>,
993        test_endpoint: Option<&str>,
994        idempotent: bool,
995        tab: &Tab,
996    ) -> StepResult {
997        let name = format!("[click] {target}");
998        let selector = match self.resolve_selector(
999            selector_override,
1000            target,
1001            step_endpoint,
1002            test_endpoint,
1003            tab,
1004        ) {
1005            Ok(s) => s,
1006            Err(msg) => {
1007                if idempotent {
1008                    return StepResult {
1009                        name,
1010                        status: StepStatus::Skipped,
1011                        message: format!("skipped (idempotent): no target found — {msg}"),
1012                    };
1013                }
1014                return StepResult {
1015                    name,
1016                    status: StepStatus::Failed,
1017                    message: msg,
1018                };
1019            }
1020        };
1021
1022        // Idempotent steps probe briefly: a missing target means the
1023        // action was already done / not applicable (e.g. an
1024        // already-authenticated session), and skipping is the success
1025        // path, not a failure.
1026        let probe_secs = if idempotent { 5 } else { 10 };
1027        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1028            Ok(element) => match element.click() {
1029                Ok(_) => StepResult {
1030                    name,
1031                    status: StepStatus::Passed,
1032                    message: format!("clicked {selector}"),
1033                },
1034                Err(e) => StepResult {
1035                    name,
1036                    status: StepStatus::Failed,
1037                    message: format!("click failed on {selector}: {e}"),
1038                },
1039            },
1040            Err(e) if idempotent => StepResult {
1041                name,
1042                status: StepStatus::Skipped,
1043                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1044            },
1045            Err(e) => StepResult {
1046                name,
1047                status: StepStatus::Failed,
1048                message: format!("element {selector} not found: {e}"),
1049            },
1050        }
1051    }
1052
1053    #[allow(clippy::too_many_arguments)]
1054    fn run_type(
1055        &self,
1056        target: &str,
1057        text: &str,
1058        selector_override: Option<&str>,
1059        step_endpoint: Option<&str>,
1060        test_endpoint: Option<&str>,
1061        idempotent: bool,
1062        tab: &Tab,
1063    ) -> StepResult {
1064        let name = format!("[type] {target}");
1065        let selector = match self.resolve_selector(
1066            selector_override,
1067            target,
1068            step_endpoint,
1069            test_endpoint,
1070            tab,
1071        ) {
1072            Ok(s) => s,
1073            Err(msg) => {
1074                if idempotent {
1075                    return StepResult {
1076                        name,
1077                        status: StepStatus::Skipped,
1078                        message: format!("skipped (idempotent): no target found — {msg}"),
1079                    };
1080                }
1081                return StepResult {
1082                    name,
1083                    status: StepStatus::Failed,
1084                    message: msg,
1085                };
1086            }
1087        };
1088
1089        let probe_secs = if idempotent { 5 } else { 10 };
1090        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1091            Ok(element) => {
1092                if let Err(e) = element.click() {
1093                    return StepResult {
1094                        name,
1095                        status: StepStatus::Failed,
1096                        message: format!("click to focus {selector} failed: {e}"),
1097                    };
1098                }
1099
1100                let js = format!(
1101                    "document.querySelector('{}').value = '';",
1102                    selector.replace('\'', "\\'")
1103                );
1104                let _ = tab.evaluate(&js, false);
1105
1106                match element.type_into(text) {
1107                    Ok(_) => StepResult {
1108                        name,
1109                        status: StepStatus::Passed,
1110                        message: format!("typed {text:?} into {selector}"),
1111                    },
1112                    Err(e) => StepResult {
1113                        name,
1114                        status: StepStatus::Failed,
1115                        message: format!("type into {selector} failed: {e}"),
1116                    },
1117                }
1118            }
1119            Err(e) if idempotent => StepResult {
1120                name,
1121                status: StepStatus::Skipped,
1122                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1123            },
1124            Err(e) => StepResult {
1125                name,
1126                status: StepStatus::Failed,
1127                message: format!("element {selector} not found: {e}"),
1128            },
1129        }
1130    }
1131
1132    #[allow(clippy::too_many_arguments)]
1133    #[allow(clippy::too_many_lines)]
1134    fn run_wait(
1135        &self,
1136        target: &str,
1137        selector_override: Option<&str>,
1138        text: Option<&str>,
1139        timeout_ms: Option<u64>,
1140        step_endpoint: Option<&str>,
1141        test_endpoint: Option<&str>,
1142        idempotent: bool,
1143        tab: &Tab,
1144    ) -> StepResult {
1145        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1146        let step_name = format!("[wait] {target}");
1147
1148        // Resolve an explicit selector only (text-only waits are LLM-free).
1149        let selector = match selector_override {
1150            Some(s) => Some(s.to_owned()),
1151            None if text.is_some() => None,
1152            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1153                Ok(s) => Some(s),
1154                Err(msg) => {
1155                    if idempotent {
1156                        return StepResult {
1157                            name: step_name,
1158                            status: StepStatus::Skipped,
1159                            message: format!("skipped (idempotent): no target found — {msg}"),
1160                        };
1161                    }
1162                    return StepResult {
1163                        name: step_name,
1164                        status: StepStatus::Failed,
1165                        message: msg,
1166                    };
1167                }
1168            },
1169        };
1170
1171        if text.is_some() {
1172            let sel_js = selector
1173                .as_deref()
1174                .map(crate::selectors::selector_matches_js);
1175            let text_js = text.map(|t| {
1176                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1177                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1178            });
1179
1180            let deadline = Instant::now() + timeout;
1181            loop {
1182                let sel_ok = sel_js
1183                    .as_ref()
1184                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1185                let text_ok = text_js
1186                    .as_ref()
1187                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1188                if sel_ok && text_ok {
1189                    let mut what = Vec::new();
1190                    if let Some(sel) = &selector {
1191                        what.push(format!("found {sel}"));
1192                    }
1193                    if let Some(t) = text {
1194                        what.push(format!("text {t:?} visible"));
1195                    }
1196                    return StepResult {
1197                        name: step_name,
1198                        status: StepStatus::Passed,
1199                        message: what.join(" and "),
1200                    };
1201                }
1202                if Instant::now() >= deadline {
1203                    let mut what = Vec::new();
1204                    if let Some(sel) = &selector {
1205                        what.push(sel.clone());
1206                    }
1207                    if let Some(t) = text {
1208                        what.push(format!("text {t:?}"));
1209                    }
1210                    let message = format!(
1211                        "wait for {} timed out after {}ms: the event waited for never came",
1212                        what.join(" / "),
1213                        timeout.as_millis(),
1214                    );
1215                    if idempotent {
1216                        return StepResult {
1217                            name: step_name,
1218                            status: StepStatus::Skipped,
1219                            message: format!("skipped (idempotent): {message}"),
1220                        };
1221                    }
1222                    return StepResult {
1223                        name: step_name,
1224                        status: StepStatus::Failed,
1225                        message,
1226                    };
1227                }
1228                std::thread::sleep(Duration::from_millis(250));
1229            }
1230        }
1231
1232        match selector.as_deref() {
1233            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1234                Ok(_) => StepResult {
1235                    name: step_name,
1236                    status: StepStatus::Passed,
1237                    message: format!("found {sel}"),
1238                },
1239                Err(e) if idempotent => StepResult {
1240                    name: step_name,
1241                    status: StepStatus::Skipped,
1242                    message: format!(
1243                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1244                        timeout.as_millis()
1245                    ),
1246                },
1247                Err(e) => StepResult {
1248                    name: step_name,
1249                    status: StepStatus::Failed,
1250                    message: format!(
1251                        "wait for {sel} timed out after {}ms: {e}",
1252                        timeout.as_millis()
1253                    ),
1254                },
1255            },
1256            None => StepResult {
1257                name: step_name,
1258                status: StepStatus::Failed,
1259                message: "wait step has neither selector nor text".into(),
1260            },
1261        }
1262    }
1263
1264    #[allow(clippy::too_many_arguments)]
1265    fn run_assert(
1266        &self,
1267        definition: Option<&str>,
1268        preset: Option<&str>,
1269        prompt: Option<&str>,
1270        assert_text: Option<&str>,
1271        screenshot: bool,
1272        step_endpoint: Option<&str>,
1273        test_endpoint: Option<&str>,
1274        tab: &Tab,
1275    ) -> StepResult {
1276        std::thread::sleep(Duration::from_millis(500));
1277
1278        let page_content = get_page_text(tab);
1279
1280        // Vision attach: capture the full page once per assert step and
1281        // split it into viewport-tall tiles (the total coverage is bounded
1282        // by the configured height cap so vision tokens stay sane). All
1283        // tile data URLs are handed to the preset/prompt evaluation below.
1284        let image: Option<Vec<String>> = if screenshot {
1285            let endpoint = self
1286                .endpoints
1287                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1288            if !endpoint.vision {
1289                return StepResult {
1290                    name: "[assert]".into(),
1291                    status: StepStatus::Failed,
1292                    message: format!(
1293                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1294                        name = endpoint.name
1295                    ),
1296                };
1297            }
1298            match crate::vision::capture_screenshot_data_urls(
1299                tab,
1300                self.config
1301                    .screenshot_max_dimension
1302                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1303                self.screenshot_height_cap(),
1304                self.current_viewport_height(),
1305            ) {
1306                Ok(urls) => Some(urls),
1307                Err(e) => {
1308                    return StepResult {
1309                        name: "[assert]".into(),
1310                        status: StepStatus::Failed,
1311                        message: format!("screenshot capture failed: {e}"),
1312                    };
1313                }
1314            }
1315        } else {
1316            None
1317        };
1318
1319        if let Some(def_name) = definition {
1320            if let Some(def) = self.definitions.get(def_name) {
1321                return self.run_assert_def(
1322                    def,
1323                    &page_content,
1324                    image.as_deref(),
1325                    step_endpoint,
1326                    test_endpoint,
1327                    tab,
1328                );
1329            }
1330            return StepResult {
1331                name: format!("[assert] {def_name}"),
1332                status: StepStatus::Failed,
1333                message: format!("definition '{def_name}' not found"),
1334            };
1335        }
1336
1337        if let Some(preset_name) = preset {
1338            // Deterministic DOM layout scan — runs JS in the browser and
1339            // never calls the LLM (free, fast, no pixel budget).
1340            if preset_name == "layout_no_issues" {
1341                return self.run_layout_preset(tab);
1342            }
1343            return self.run_preset(
1344                preset_name,
1345                assert_text,
1346                &page_content,
1347                image.as_deref(),
1348                step_endpoint,
1349                test_endpoint,
1350            );
1351        }
1352
1353        if let Some(prompt_text) = prompt {
1354            return self.run_custom(
1355                prompt_text,
1356                &page_content,
1357                image.as_deref(),
1358                step_endpoint,
1359                test_endpoint,
1360            );
1361        }
1362
1363        StepResult {
1364            name: "[assert]".into(),
1365            status: StepStatus::Skipped,
1366            message: "no definition, preset, or prompt specified".into(),
1367        }
1368    }
1369
1370    fn run_assert_def(
1371        &self,
1372        def: &AssertDefinition,
1373        page_content: &PageContent,
1374        image: Option<&[String]>,
1375        step_endpoint: Option<&str>,
1376        test_endpoint: Option<&str>,
1377        tab: &Tab,
1378    ) -> StepResult {
1379        // Agent-based definition: delegate to an A2A agent
1380        if let Some(ref agent) = def.agent {
1381            if image.is_some() {
1382                return StepResult {
1383                    name: format!("[assert] {}", def.name),
1384                    status: StepStatus::Failed,
1385                    message: "agent-backed assertions do not support screenshots".into(),
1386                };
1387            }
1388            let task = def
1389                .task_template
1390                .as_deref()
1391                .unwrap_or("Evaluate the assertion")
1392                .replace("{url}", &page_content.url)
1393                .replace("{title}", &page_content.title)
1394                .replace("{content}", &page_content.body_text)
1395                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1396
1397            return self.run_agent_step(agent, &task, &def.name);
1398        }
1399
1400        // Custom preset: system + user_template provided in the definition
1401        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1402            return self.run_custom_preset(
1403                &def.name,
1404                system,
1405                template,
1406                def.assert_text.as_deref(),
1407                page_content,
1408                image,
1409                step_endpoint,
1410                test_endpoint,
1411            );
1412        }
1413
1414        def.preset.as_ref().map_or_else(
1415            || {
1416                def.prompt.as_ref().map_or_else(
1417                    || StepResult {
1418                        name: format!("[assert] {}", def.name),
1419                        status: StepStatus::Failed,
1420                        message: "definition has no preset, prompt, or system+user_template".into(),
1421                    },
1422                    |prompt| {
1423                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1424                    },
1425                )
1426            },
1427            |preset_name| {
1428                if preset_name == "layout_no_issues" {
1429                    return self.run_layout_preset(tab);
1430                }
1431                self.run_preset(
1432                    preset_name,
1433                    def.assert_text.as_deref(),
1434                    page_content,
1435                    image,
1436                    step_endpoint,
1437                    test_endpoint,
1438                )
1439            },
1440        )
1441    }
1442
1443    #[allow(clippy::too_many_arguments)]
1444    fn run_custom_preset(
1445        &self,
1446        name: &str,
1447        system: &str,
1448        template: &str,
1449        assert_text: Option<&str>,
1450        page_content: &PageContent,
1451        image: Option<&[String]>,
1452        step_endpoint: Option<&str>,
1453        test_endpoint: Option<&str>,
1454    ) -> StepResult {
1455        let user_prompt = template
1456            .replace("{url}", &page_content.url)
1457            .replace("{title}", &page_content.title)
1458            .replace("{content}", &page_content.body_text)
1459            .replace("{expected_text}", assert_text.unwrap_or(""))
1460            .replace("{description}", "");
1461
1462        // Custom preset definitions frequently forget the {content}
1463        // placeholder — without it the LLM has no page to evaluate and
1464        // answers "I can't determine that without seeing the page". Always
1465        // append the page context unless the template already references it.
1466        let user_prompt = if template.contains("{content}") {
1467            user_prompt
1468        } else {
1469            format!(
1470                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1471                url = page_content.url,
1472                title = page_content.title,
1473                content = page_content.body_text,
1474            )
1475        };
1476
1477        self.reporter
1478            .debug(format!("assert: {name} (custom preset)"));
1479
1480        let chain = self
1481            .endpoints
1482            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1483        let sys = system.to_owned();
1484
1485        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1486
1487        response.map_or_else(
1488            |e| StepResult {
1489                name: format!("[assert] {name}"),
1490                status: StepStatus::Failed,
1491                message: format!("LLM assertion call failed: {e}"),
1492            },
1493            |(lr, _idx)| {
1494                let content_lower = lr.content.to_lowercase().trim().to_owned();
1495                if content_lower.starts_with("pass") {
1496                    StepResult {
1497                        name: format!("[assert] {name}"),
1498                        status: StepStatus::Passed,
1499                        message: "PASS".into(),
1500                    }
1501                } else {
1502                    StepResult {
1503                        name: format!("[assert] {name}"),
1504                        status: StepStatus::Failed,
1505                        message: lr.content,
1506                    }
1507                }
1508            },
1509        )
1510    }
1511
1512    fn run_preset(
1513        &self,
1514        preset_name: &str,
1515        assert_text: Option<&str>,
1516        page_content: &PageContent,
1517        image: Option<&[String]>,
1518        step_endpoint: Option<&str>,
1519        test_endpoint: Option<&str>,
1520    ) -> StepResult {
1521        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1522            return StepResult {
1523                name: format!("[assert] {preset_name}"),
1524                status: StepStatus::Failed,
1525                message: format!("unknown assertion preset: {preset_name}"),
1526            };
1527        };
1528        if preset_name.starts_with("visual_") && image.is_none() {
1529            return StepResult {
1530                name: format!("[assert] {preset_name}"),
1531                status: StepStatus::Failed,
1532                message: format!(
1533                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1534                ),
1535            };
1536        }
1537
1538        let user_prompt = preset
1539            .user_template
1540            .replace("{url}", &page_content.url)
1541            .replace("{title}", &page_content.title)
1542            .replace("{content}", &page_content.body_text)
1543            .replace("{expected_text}", assert_text.unwrap_or(""))
1544            .replace("{description}", "");
1545
1546        // Same safety net as custom presets: never let the LLM answer with
1547        // no page context at all.
1548        let user_prompt = if preset.user_template.contains("{content}") {
1549            user_prompt
1550        } else {
1551            format!(
1552                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1553                url = page_content.url,
1554                title = page_content.title,
1555                content = page_content.body_text,
1556            )
1557        };
1558
1559        self.reporter.debug(format!("assert: {preset_name}"));
1560
1561        let chain = self
1562            .endpoints
1563            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1564        let sys = preset.system.to_owned();
1565
1566        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1567
1568        response.map_or_else(
1569            |e| StepResult {
1570                name: format!("[assert] {preset_name}"),
1571                status: StepStatus::Failed,
1572                message: format!("LLM assertion call failed: {e}"),
1573            },
1574            |(lr, _idx)| {
1575                let content_lower = lr.content.to_lowercase().trim().to_owned();
1576                if content_lower.starts_with("pass") {
1577                    StepResult {
1578                        name: format!("[assert] {preset_name}"),
1579                        status: StepStatus::Passed,
1580                        message: "PASS".into(),
1581                    }
1582                } else {
1583                    StepResult {
1584                        name: format!("[assert] {preset_name}"),
1585                        status: StepStatus::Failed,
1586                        message: lr.content,
1587                    }
1588                }
1589            },
1590        )
1591    }
1592
1593    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1594    ///
1595    /// Evaluates the layout-scan JS in the page and fails with the list of
1596    /// detected issues: horizontal page overflow, visible elements sticking
1597    /// out of the viewport, text clipped by `overflow: hidden` containers,
1598    /// and interactive elements covered by other elements. No LLM call —
1599    /// checks are geometry-based so the check is free, deterministic, and
1600    /// safe to run on every page × viewport variant.
1601    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1602        let name = "[assert] layout_no_issues".to_owned();
1603        self.reporter
1604            .debug("assert: layout_no_issues (DOM layout scan)");
1605        let js = LAYOUT_SCAN_JS.replace(
1606            "__IGNORE_CLASSES__",
1607            &serde_json::to_string(&self.config.layout_ignore_classes)
1608                .unwrap_or_else(|_| "[]".to_owned()),
1609        );
1610        let result = tab.evaluate(&js, false);
1611        let json_str = match result {
1612            Ok(r) => r
1613                .value
1614                .as_ref()
1615                .and_then(|v| v.as_str().map(String::from))
1616                .unwrap_or_else(|| "[]".to_owned()),
1617            Err(e) => {
1618                return StepResult {
1619                    name,
1620                    status: StepStatus::Failed,
1621                    message: format!("layout scan JS failed: {e}"),
1622                };
1623            }
1624        };
1625        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1626        if issues.is_empty() {
1627            return StepResult {
1628                name,
1629                status: StepStatus::Passed,
1630                message: "PASS — no layout defects detected".into(),
1631            };
1632        }
1633        let mut lines: Vec<String> = issues
1634            .iter()
1635            .take(10)
1636            .map(|i| {
1637                format!(
1638                    "- [{type_}] {element}: {detail}",
1639                    type_ = i.issue_type,
1640                    element = i.element,
1641                    detail = i.detail
1642                )
1643            })
1644            .collect();
1645        if issues.len() > 10 {
1646            lines.push(format!("- … and {} more", issues.len() - 10));
1647        }
1648        StepResult {
1649            name,
1650            status: StepStatus::Failed,
1651            message: format!(
1652                "FAIL — {} layout defect(s) detected:\n{}",
1653                issues.len(),
1654                lines.join("\n")
1655            ),
1656        }
1657    }
1658
1659    fn run_custom(
1660        &self,
1661        prompt: &str,
1662        page_content: &PageContent,
1663        image: Option<&[String]>,
1664        step_endpoint: Option<&str>,
1665        test_endpoint: Option<&str>,
1666    ) -> StepResult {
1667        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1668
1669        let mut user = format!(
1670            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1671            url = page_content.url,
1672            title = page_content.title,
1673            content = page_content.body_text,
1674        );
1675        if image.is_some() {
1676            user.push_str(
1677                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1678            );
1679        }
1680
1681        self.reporter.debug("custom assert");
1682
1683        let chain = self
1684            .endpoints
1685            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1686        let sys = system.to_owned();
1687
1688        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1689
1690        response.map_or_else(
1691            |e| StepResult {
1692                name: "[assert] custom".into(),
1693                status: StepStatus::Failed,
1694                message: format!("LLM assertion call failed: {e}"),
1695            },
1696            |(lr, _idx)| {
1697                let content_lower = lr.content.to_lowercase().trim().to_owned();
1698                if content_lower.starts_with("pass") {
1699                    StepResult {
1700                        name: "[assert] custom".into(),
1701                        status: StepStatus::Passed,
1702                        message: "PASS".into(),
1703                    }
1704                } else {
1705                    StepResult {
1706                        name: "[assert] custom".into(),
1707                        status: StepStatus::Failed,
1708                        message: lr.content,
1709                    }
1710                }
1711            },
1712        )
1713    }
1714
1715    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1716        let path = path.unwrap_or("screenshot.png");
1717
1718        match tab.capture_screenshot(
1719            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1720            None,
1721            None,
1722            true,
1723        ) {
1724            Ok(data) => {
1725                if let Err(e) = std::fs::write(path, &data) {
1726                    return StepResult {
1727                        name: format!("[screenshot] {path}"),
1728                        status: StepStatus::Failed,
1729                        message: format!("failed to write screenshot: {e}"),
1730                    };
1731                }
1732                StepResult {
1733                    name: format!("[screenshot] {path}"),
1734                    status: StepStatus::Passed,
1735                    message: format!("saved to {path}"),
1736                }
1737            }
1738            Err(e) => StepResult {
1739                name: format!("[screenshot] {path}"),
1740                status: StepStatus::Failed,
1741                message: format!("screenshot failed: {e}"),
1742            },
1743        }
1744    }
1745
1746    /// Runs an A2A agent step.
1747    #[allow(clippy::literal_string_with_formatting_args)]
1748    fn run_agent(
1749        &self,
1750        agent_name: &str,
1751        task: &str,
1752        definition: Option<&str>,
1753        _test_endpoint: Option<&str>,
1754    ) -> StepResult {
1755        // If a definition is specified, look up the task template
1756        let resolved_task = if let Some(def_name) = definition {
1757            if let Some(def) = self.definitions.get(def_name) {
1758                let tmpl = def.task_template.as_deref().unwrap_or(task);
1759                tmpl.replace("{task}", task)
1760            } else {
1761                return StepResult {
1762                    name: format!("[agent] {def_name}"),
1763                    status: StepStatus::Failed,
1764                    message: format!("definition '{def_name}' not found"),
1765                };
1766            }
1767        } else {
1768            task.to_owned()
1769        };
1770
1771        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1772    }
1773
1774    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1775        let Some(ep) = self.endpoints.get(agent_name) else {
1776            return StepResult {
1777                name: format!("[agent] {display_name}"),
1778                status: StepStatus::Failed,
1779                message: format!("agent endpoint '{agent_name}' not found"),
1780            };
1781        };
1782
1783        if ep.url.is_empty() {
1784            return StepResult {
1785                name: format!("[agent] {display_name}"),
1786                status: StepStatus::Failed,
1787                message: format!("agent endpoint '{agent_name}' has no URL"),
1788            };
1789        }
1790
1791        self.reporter.debug(format!("agent {agent_name}: {task}"));
1792
1793        let url = ep.url.clone();
1794        let client = A2aClient::new(&url, self.timeout);
1795        let task_clone = task.to_owned();
1796
1797        let response = std::thread::spawn(move || {
1798            let rt = tokio::runtime::Builder::new_current_thread()
1799                .enable_all()
1800                .build()
1801                .unwrap();
1802            rt.block_on(client.send_task(&task_clone))
1803        })
1804        .join()
1805        .unwrap();
1806
1807        // Record the flat-cost call
1808        self.usage.record_flat_call(agent_name, ep);
1809
1810        match response {
1811            Ok(text) => {
1812                let clean = text.trim().to_owned();
1813                let lower = clean.to_lowercase();
1814                if lower.starts_with("pass") {
1815                    StepResult {
1816                        name: format!("[agent] {display_name}"),
1817                        status: StepStatus::Passed,
1818                        message: format!("PASS: {clean}"),
1819                    }
1820                } else if lower.starts_with("fail") {
1821                    StepResult {
1822                        name: format!("[agent] {display_name}"),
1823                        status: StepStatus::Failed,
1824                        message: clean,
1825                    }
1826                } else {
1827                    StepResult {
1828                        name: format!("[agent] {display_name}"),
1829                        status: StepStatus::Passed,
1830                        message: format!("response: {clean}"),
1831                    }
1832                }
1833            }
1834            Err(e) => StepResult {
1835                name: format!("[agent] {display_name}"),
1836                status: StepStatus::Failed,
1837                message: format!("agent call failed: {e}"),
1838            },
1839        }
1840    }
1841
1842    /// Runs an MCP tool call step.
1843    fn run_mcp(
1844        &self,
1845        server_name: &str,
1846        tool_name: &str,
1847        args: Option<&serde_json::Value>,
1848    ) -> StepResult {
1849        let Some(ep) = self.endpoints.get(server_name) else {
1850            return StepResult {
1851                name: format!("[mcp] {server_name}:{tool_name}"),
1852                status: StepStatus::Failed,
1853                message: format!("MCP server endpoint '{server_name}' not found"),
1854            };
1855        };
1856
1857        let cmd = ep.command.as_deref().unwrap_or("");
1858        if cmd.is_empty() {
1859            return StepResult {
1860                name: format!("[mcp] {server_name}:{tool_name}"),
1861                status: StepStatus::Failed,
1862                message: format!("MCP server '{server_name}' has no command configured"),
1863            };
1864        }
1865
1866        self.reporter
1867            .debug(format!("mcp {server_name} {tool_name}"));
1868
1869        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1870
1871        let command = cmd.to_owned();
1872        let args_vec = ep.args.clone();
1873        let tool = tool_name.to_owned();
1874
1875        let response = std::thread::spawn(move || {
1876            let mut mcp_client =
1877                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1878            mcp_client
1879                .call_tool(&tool, &args_val)
1880                .map_err(|e| e.to_string())
1881        })
1882        .join()
1883        .unwrap();
1884
1885        // Record the flat-cost call
1886        self.usage.record_flat_call(server_name, ep);
1887
1888        match response {
1889            Ok(result) => {
1890                if result.isError {
1891                    StepResult {
1892                        name: format!("[mcp] {server_name}:{tool_name}"),
1893                        status: StepStatus::Failed,
1894                        message: result.to_string(),
1895                    }
1896                } else {
1897                    StepResult {
1898                        name: format!("[mcp] {server_name}:{tool_name}"),
1899                        status: StepStatus::Passed,
1900                        message: result.to_string(),
1901                    }
1902                }
1903            }
1904            Err(e) => StepResult {
1905                name: format!("[mcp] {server_name}:{tool_name}"),
1906                status: StepStatus::Failed,
1907                message: format!("MCP call failed: {e}"),
1908            },
1909        }
1910    }
1911
1912    // ── helpers ──────────────────────────────────────────────────────────
1913
1914    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1915    /// the runner's default LLM config for any unset fields.
1916    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1917        LlmConfig {
1918            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
1919                // Bedrock builds its endpoint from the resolved AWS region
1920                // when no URL is given — never inherit the default LLM URL.
1921                endpoint.url.clone()
1922            } else if endpoint.url.is_empty() {
1923                self.llm.url.clone()
1924            } else {
1925                endpoint.url.clone()
1926            },
1927            model: endpoint
1928                .model
1929                .clone()
1930                .unwrap_or_else(|| self.llm.model.clone()),
1931            api_key: endpoint
1932                .api_key
1933                .clone()
1934                .or_else(|| self.llm.api_key.clone()),
1935            headers: if endpoint.headers.is_empty() {
1936                self.llm.headers.clone()
1937            } else {
1938                endpoint.headers.clone()
1939            },
1940            timeout: self.llm.timeout,
1941            temperature: self.llm.temperature,
1942            thinking: self.llm.thinking,
1943            model_params: self.llm.model_params.clone(),
1944            max_attempts: endpoint.max_attempts.max(1),
1945            provider: endpoint.provider,
1946            deployment: endpoint.deployment.clone(),
1947            api_version: endpoint.api_version.clone(),
1948            auth: endpoint.auth.clone(),
1949            header_commands: endpoint.header_commands.clone(),
1950            aws: endpoint.aws.clone(),
1951        }
1952    }
1953
1954    /// Runs a single LLM call against an ordered endpoint chain (primary +
1955    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
1956    /// the first endpoint that answers wins. Returns the response together
1957    /// with the chain index of the answering endpoint (0 = primary) so the
1958    /// caller can attribute usage to the correct endpoint.
1959    ///
1960    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
1961    /// shows duration, tokens, cost and the answering endpoint per call.
1962    /// Run-level context handed to every LLM call as the FIRST block of the
1963    /// user message. Contents that are stable for the whole run ("run
1964    /// started", "target site") come first so upstream provider prefix
1965    /// caching stays effective; the current time is the last line because
1966    /// it changes on every call.
1967    fn run_context(&self) -> String {
1968        let mut parts = vec![
1969            "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
1970            "================================================================".into(),
1971            format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
1972        ];
1973        if let Some(base) = self.config.base_url.as_deref() {
1974            parts.push(format!("Target site: {base}"));
1975        }
1976        let now = SystemTime::now()
1977            .duration_since(UNIX_EPOCH)
1978            .map_or(0, |d| d.as_secs());
1979        parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
1980        parts.join("\n")
1981    }
1982
1983    #[allow(clippy::cast_possible_truncation, clippy::too_many_lines)]
1984    fn llm_call_chain(
1985        &self,
1986        chain: &[&ResolvedEndpoint],
1987        system: &str,
1988        user: &str,
1989        image: Option<&[String]>,
1990        purpose: &str,
1991    ) -> Result<(crate::costs::LlmResponse, usize), String> {
1992        if chain.is_empty() {
1993            return Err("empty LLM endpoint chain".into());
1994        }
1995        let primary = self.build_llm_for_endpoint(chain[0]);
1996        let fallbacks: Vec<LlmConfig> = chain[1..]
1997            .iter()
1998            .map(|e| self.build_llm_for_endpoint(e))
1999            .collect();
2000
2001        let (test, index) = self
2002            .current_step
2003            .borrow()
2004            .as_ref()
2005            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
2006        let primary_endpoint = chain[0].name.clone();
2007        let primary_model = primary.model.clone();
2008        self.emit_event(&TestEvent::LlmCallStarted {
2009            test: test.clone(),
2010            index,
2011            endpoint: primary_endpoint.clone(),
2012            model: primary_model.clone(),
2013            purpose: purpose.to_owned(),
2014        });
2015
2016        let started = Instant::now();
2017        let sys = system.to_owned();
2018        let context = self.run_context();
2019        let user = if context.is_empty() {
2020            user.to_owned()
2021        } else {
2022            format!("{context}\n\n{user}")
2023        };
2024        let image = image.map(<[String]>::to_vec);
2025
2026        let result = std::thread::spawn(move || {
2027            let rt = tokio::runtime::Builder::new_current_thread()
2028                .enable_all()
2029                .build()
2030                .unwrap();
2031            let call = async {
2032                match image.as_deref() {
2033                    Some(img) => {
2034                        llm_chat_vision_with_usage_chain(
2035                            &primary,
2036                            &fallbacks,
2037                            &sys,
2038                            &user,
2039                            Some(img),
2040                        )
2041                        .await
2042                    }
2043                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2044                }
2045            };
2046            rt.block_on(call)
2047        })
2048        .join()
2049        .unwrap();
2050
2051        let duration_ms = started.elapsed().as_millis() as u64;
2052        match result {
2053            Ok((lr, idx)) => {
2054                let cost = calculate_llm_cost(
2055                    chain[idx],
2056                    lr.usage.prompt_tokens,
2057                    lr.usage.completion_tokens,
2058                );
2059                let answering = chain[idx].name.clone();
2060                let model = chain[idx]
2061                    .model
2062                    .clone()
2063                    .unwrap_or_else(|| primary_model.clone());
2064                self.usage
2065                    .record_llm_call(&answering, chain[idx], &model, &lr.usage);
2066                self.emit_event(&TestEvent::LlmCallFinished {
2067                    test,
2068                    index,
2069                    endpoint: answering,
2070                    model,
2071                    purpose: purpose.to_owned(),
2072                    ok: true,
2073                    duration_ms,
2074                    input_tokens: lr.usage.prompt_tokens,
2075                    output_tokens: lr.usage.completion_tokens,
2076                    cached_input_tokens: lr.usage.cached_input_tokens,
2077                    cost,
2078                    error: None,
2079                });
2080                Ok((lr, idx))
2081            }
2082            Err(e) => {
2083                self.emit_event(&TestEvent::LlmCallFinished {
2084                    test,
2085                    index,
2086                    endpoint: primary_endpoint,
2087                    model: primary_model,
2088                    purpose: purpose.to_owned(),
2089                    ok: false,
2090                    duration_ms,
2091                    input_tokens: 0,
2092                    output_tokens: 0,
2093                    cached_input_tokens: 0,
2094                    cost: 0.0,
2095                    error: Some(e.clone()),
2096                });
2097                Err(e)
2098            }
2099        }
2100    }
2101
2102    /// Resolves a CSS selector for the target element. Uses the explicit
2103    /// `selector` if provided, otherwise asks the LLM to find the element
2104    /// from the natural language `target` description and page DOM.
2105    ///
2106    /// LLM responses are sanitized and verified against the live page: a
2107    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2108    /// immediately with the raw LLM output, and a selector that matches
2109    /// nothing triggers one retry with feedback before failing.
2110    #[allow(clippy::too_many_lines)]
2111    fn resolve_selector(
2112        &self,
2113        css_override: Option<&str>,
2114        target: &str,
2115        step_endpoint: Option<&str>,
2116        test_endpoint: Option<&str>,
2117        tab: &Tab,
2118    ) -> Result<String, String> {
2119        if let Some(explicit) = css_override {
2120            return Ok(explicit.to_owned());
2121        }
2122
2123        let dom_info = extract_dom_info(tab)?;
2124        let page_content = get_page_text(tab);
2125
2126        let system = concat!(
2127            "You are a browser automation selector generator. ",
2128            "Given a web page's content and interactive elements, ",
2129            "return ONLY the best CSS selector for the described element. ",
2130            "Output nothing except the CSS selector. ",
2131            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2132            "[name=\"...\"], tag.class, tag. ",
2133            "Never output explanations, markdown, or extra text."
2134        );
2135
2136        let user = format!(
2137            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2138            page_content.url,
2139            page_content.title,
2140            truncate(&page_content.body_text, 4000),
2141            dom_info,
2142            target,
2143        );
2144
2145        let retry_user = format!(
2146            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2147            "The selector must match at least one element currently present on the page.",
2148            page_content.url,
2149            page_content.title,
2150            truncate(&page_content.body_text, 4000),
2151            dom_info,
2152            target,
2153        );
2154
2155        self.reporter.debug(format!("LLM targeting: {target}"));
2156
2157        let chain = self
2158            .endpoints
2159            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2160        let sys = system.to_owned();
2161
2162        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2163
2164        let first = call_llm(&user);
2165        let (lr, _idx) = match first {
2166            Ok(lr) => lr,
2167            Err(e) => {
2168                return Err(format!("LLM element targeting failed: {e}"));
2169            }
2170        };
2171        let clean = sanitize_selector(&lr.content);
2172        self.reporter.debug(format!("resolved selector: {clean}"));
2173
2174        if selector_is_useless(&clean) {
2175            return Err(format!(
2176                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2177                raw = lr.content.trim(),
2178            ));
2179        }
2180        if let Err(reason) = validate_selector(&clean) {
2181            return Err(format!(
2182                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2183                raw = lr.content.trim(),
2184            ));
2185        }
2186        if !selector_matches(tab, &clean).unwrap_or(false) {
2187            // One retry with feedback: flaky models occasionally invent a
2188            // selector that does not exist on the page.
2189            self.reporter.warn(format!(
2190                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2191            ));
2192            let second = call_llm(&retry_user);
2193            let (lr2, _idx2) = match second {
2194                Ok(lr2) => lr2,
2195                Err(e) => {
2196                    return Err(format!(
2197                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2198                    ));
2199                }
2200            };
2201            let clean2 = sanitize_selector(&lr2.content);
2202            self.reporter
2203                .debug(format!("resolved selector (retry): {clean2}"));
2204            if selector_is_useless(&clean2) {
2205                return Err(format!(
2206                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2207                    raw = lr2.content.trim(),
2208                    excerpt = truncate(&page_content.body_text, 300),
2209                ));
2210            }
2211            if !selector_matches(tab, &clean2).unwrap_or(false) {
2212                return Err(format!(
2213                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2214                ));
2215            }
2216            return Ok(clean2);
2217        }
2218
2219        Ok(clean)
2220    }
2221}
2222
2223/// Evaluates a JS expression that is expected to return a boolean.
2224fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2225    tab.evaluate(js, false)
2226        .map_err(|e| format!("evaluate failed: {e}"))?
2227        .value
2228        .and_then(|v| v.as_bool())
2229        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2230}
2231
2232/// Checks whether a CSS selector matches at least one current element.
2233fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2234    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2235}
2236
2237// ── Free helper functions ──────────────────────────────────────────────
2238
2239fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2240    let name = format!("[navigate] {full_url}");
2241    match tab.navigate_to(full_url) {
2242        Ok(_) => {
2243            let _ = tab.wait_until_navigated();
2244            StepResult {
2245                name,
2246                status: StepStatus::Passed,
2247                message: format!("navigated to {full_url}"),
2248            }
2249        }
2250        Err(e) => StepResult {
2251            name,
2252            status: StepStatus::Failed,
2253            message: format!("navigation failed: {e}"),
2254        },
2255    }
2256}
2257
2258fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2259    let result = tab
2260        .evaluate(DOM_EXTRACT_JS, false)
2261        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2262
2263    let json_str = result
2264        .value
2265        .as_ref()
2266        .and_then(|v| v.as_str())
2267        .unwrap_or("[]");
2268
2269    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2270
2271    if elements.is_empty() {
2272        return Ok("(no interactive elements found)".to_owned());
2273    }
2274
2275    Ok(elements.join("\n"))
2276}
2277
2278fn get_page_text(tab: &Tab) -> PageContent {
2279    let url = tab.get_url();
2280
2281    let title = tab
2282        .evaluate("document.title", false)
2283        .ok()
2284        .and_then(|r| r.value)
2285        .and_then(|v| v.as_str().map(String::from))
2286        .unwrap_or_else(|| "unknown".to_owned());
2287
2288    let body_text = tab
2289        .evaluate(
2290            "document.body ? document.body.innerText : document.documentElement.innerText",
2291            false,
2292        )
2293        .ok()
2294        .and_then(|r| r.value)
2295        .and_then(|v| v.as_str().map(String::from))
2296        .unwrap_or_default();
2297
2298    PageContent {
2299        url,
2300        title,
2301        body_text: truncate(&body_text, 8000),
2302    }
2303}
2304
2305fn resolve_url(url: &str, base_url: &str) -> String {
2306    if url.starts_with("http://") || url.starts_with("https://") {
2307        return url.to_owned();
2308    }
2309    let base = base_url.trim_end_matches('/');
2310    if url.starts_with('/') {
2311        format!("{base}{url}")
2312    } else {
2313        format!("{base}/{url}")
2314    }
2315}
2316
2317/// Human-readable label for a step, used when steps are skipped after an
2318/// earlier failure.
2319fn step_label(step: &TestStep) -> String {
2320    match step {
2321        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2322        TestStep::Click { target, .. } => format!("[click] {target}"),
2323        TestStep::Type { target, .. } => format!("[type] {target}"),
2324        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2325        TestStep::Assert {
2326            definition,
2327            preset,
2328            prompt,
2329            ..
2330        } => definition.as_ref().map_or_else(
2331            || {
2332                preset.as_ref().map_or_else(
2333                    || {
2334                        prompt.as_ref().map_or_else(
2335                            || "[assert]".to_owned(),
2336                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2337                        )
2338                    },
2339                    |p| format!("[assert] {p}"),
2340                )
2341            },
2342            |d| format!("[assert] {d}"),
2343        ),
2344        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2345        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2346        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2347    }
2348}
2349
2350/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2351#[must_use]
2352const fn step_kind_label(step: &TestStep) -> &'static str {
2353    match step {
2354        TestStep::Navigate { .. } => "navigate",
2355        TestStep::Click { .. } => "click",
2356        TestStep::Type { .. } => "type",
2357        TestStep::Wait { .. } => "wait",
2358        TestStep::Assert { .. } => "assert",
2359        TestStep::Screenshot { .. } => "screenshot",
2360        TestStep::Agent { .. } => "agent",
2361        TestStep::Mcp { .. } => "mcp",
2362    }
2363}
2364
2365// ── Support types ──────────────────────────────────────────────────────
2366
2367#[derive(Default)]
2368struct TestRunResult {
2369    passed: u32,
2370    failed: u32,
2371    skipped: u32,
2372    total: u32,
2373    details: Vec<StepResult>,
2374}
2375
2376struct PageContent {
2377    url: String,
2378    title: String,
2379    body_text: String,
2380}
2381
2382#[cfg(test)]
2383mod tests {
2384    use super::unix_to_rfc3339;
2385
2386    #[test]
2387    fn rfc3339_epoch_and_reference_dates() {
2388        assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2389        assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2390        assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2391        assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2392    }
2393
2394    #[test]
2395    fn rfc3339_handles_leap_years() {
2396        assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2397        assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2398    }
2399}