Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// Formats a UNIX epoch timestamp as an RFC 3339 UTC string
26/// (`2026-09-19T16:04:00Z`) without pulling in a date dependency.
27/// Civil-from-days conversion after Howard Hinnant.
28#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31    let secs = i64::try_from(secs).unwrap_or(0);
32    let days = secs.div_euclid(86_400);
33    let rem = secs.rem_euclid(86_400);
34    let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35    let z = days + 719_468;
36    let era = z.div_euclid(146_097);
37    let doe = z.rem_euclid(146_097);
38    let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40    let mp = (5 * doy + 2) / 153;
41    let d = doy - (153 * mp + 2) / 5 + 1;
42    let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43    let y = yoe + era * 400 + i64::from(mth <= 2);
44    format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47/// One detected layout defect (`layout_no_issues` preset).
48#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50    #[serde(rename = "type")]
51    issue_type: String,
52    element: String,
53    detail: String,
54}
55
56/// In-browser DOM layout scan for `layout_no_issues`.
57///
58/// Geometry-only checks (no LLM, no pixels):
59/// 1. page horizontal overflow (`scrollWidth` > viewport width);
60/// 2. elements outside the viewport that scrolling cannot reveal
61///    (fixed elements off-screen, left/negative overflow, right-edge
62///    overflow beyond the horizontally scrollable area, and bottom
63///    overflow on a page that cannot scroll down) — below-the-fold
64///    content on a tall scrollable page is normal flow, NOT a defect;
65/// 3. text clipped by `overflow: hidden` containers whose content
66///    is measurably larger than the box;
67/// 4. interactive elements (buttons/links/inputs) whose center point
68///    is covered by a different element that would intercept the click.
69///
70/// Elements whose class matches a configured ignore prefix (default:
71/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
72/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
73/// skipped — they are intentionally 1x1 / off-screen. The prefix list
74/// is injected at `__IGNORE_CLASSES__` from
75/// [`ScenarioConfig::layout_ignore_classes`].
76///
77/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
78/// sticky headers) is excluded by the position/relation filters.
79const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81  const issues = [];
82  const push = (type, el, detail) => {
83    if (issues.length >= 30) return;
84    let element = el.tagName.toLowerCase();
85    if (el.id) element += '#' + el.id;
86    else if (typeof el.className === 'string' && el.className.trim())
87      element += '.' + el.className.trim().split(/\s+/).join('.');
88    issues.push({ type, element, detail: String(detail).slice(0, 220) });
89  };
90  const vw = document.documentElement.clientWidth || window.innerWidth;
91  const vh = document.documentElement.clientHeight || window.innerHeight;
92  if (!vw || !vh) return JSON.stringify(issues);
93  const de = document.documentElement;
94  // 1. Page-level horizontal overflow.
95  if (maxSW > vw + 2)
96    push('page-overflow-x', de,
97      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98  // Class-name prefixes to skip (injected; default = Angular CDK
99  // screen-reader helpers, which are intentionally 1x1 / off-screen).
100  const ignorePrefixes = __IGNORE_CLASSES__;
101  const isIgnored = (el) => {
102    if (typeof el.className !== 'string' || !el.className.trim()) return false;
103    const classes = el.className.trim().split(/\s+/);
104    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105  };
106  const body = document.body;
107  // The document element is not always the scroll container: the app may
108  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110  // "not scrollable" — measure the scrollable content height/width of the
111  // document element AND the body, and take the max.
112  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114  const vScrollable = maxSH > vh + 2;
115  const hScrollable = maxSW > vw + 2;
116  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117  // this element into view on the given axis — i.e. the element is inside a
118  // scrolling region, so lying beyond the viewport cut is not a defect.
119  const reachableByScroller = (el, axis) => {
120    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123    let p = el.parentElement;
124    while (p && p !== body) {
125      const pcs = getComputedStyle(p);
126      const o = pcs[ovProp];
127      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128          p[dim] > p[dimClient] + 2) return true;
129      p = p.parentElement;
130    }
131    return false;
132  };
133  const selfOverflowing = (el, cs, axis) => {
134    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138      el[dim] > el[dimClient] + 2;
139  };
140  // True when an ancestor clips this axis with overflow:hidden/clip — the
141  // element's overhang is not visible, so treat it as reachable.
142  const clippedByAncestor = (el, axis) => {
143    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144    let p = el.parentElement;
145    while (p && p !== body) {
146      const o = getComputedStyle(p)[ovProp];
147      if (o === 'hidden' || o === 'clip') return true;
148      p = p.parentElement;
149    }
150    return false;
151  };
152  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154  const hasContent = (el) =>
155    ((el.textContent || '').trim().length > 0) ||
156    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157  // 2. Elements outside the viewport that scrolling cannot reveal.
158  for (const el of all) {
159    const cs = getComputedStyle(el);
160    if (!visible(cs)) continue;
161    if (isIgnored(el)) continue;
162    const r = el.getBoundingClientRect();
163    if (r.width < 2 || r.height < 2) continue;
164    if (!hasContent(el) && el.children.length === 0) continue;
165    if (cs.position === 'fixed') {
166      // Fixed elements never move with the scroll: any edge outside the
167      // viewport is unreachable content and therefore a defect.
168      const overTop = -r.top;
169      const overLeft = -r.left;
170      const overRight = r.right - vw;
171      const overBottom = r.bottom - vh;
172      // Top/left overshoot is never reachable; bottom/right overshoot is
173      // only a defect when the fixed element cannot scroll that content
174      // into view itself (e.g. an opened Material drawer whose inner
175      // container scrolls is not a "cut-off" defect).
176      if (overTop > 2 || overLeft > 2 ||
177          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179        let where = '';
180        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186        push('element-out-of-viewport', el,
187          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188      }
189      continue;
190    }
191    if (cs.position === 'sticky') continue;
192    // Negative left overflow cannot be reached by scrolling (scrollLeft
193    // never goes below 0).
194    if (r.left < -2) {
195      push('element-out-of-viewport', el,
196        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197      continue;
198    }
199    // Negative top with the page at the top means the element sits above
200    // the document origin — also unreachable.
201    if (r.top < -2 && de.scrollTop <= 2) {
202      push('element-out-of-viewport', el,
203        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204      continue;
205    }
206    const overRight = r.right - vw;
207    // Right-edge overflow is only a defect when the page cannot scroll
208    // horizontally to reveal it (or the element sticks out past the
209    // scrollable content width itself).
210    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211      push('element-out-of-viewport', el,
212        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213      continue;
214    }
215    // Below-the-fold content on a scrollable page is normal (tall landing
216    // pages); only flag bottom overflow the user can never scroll to.
217    const overBottom = r.bottom - vh;
218    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219      push('element-out-of-viewport', el,
220        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221    }
222  }
223  // 3. Text clipped by overflow:hidden containers.
224  for (const el of all) {
225    if (isIgnored(el)) continue;
226    const cs = getComputedStyle(el);
227    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229    if (!(el.textContent || '').trim()) continue;
230    // Single-line truncation with an ellipsis is an intentional design
231    // pattern (Tailwind .truncate etc.), not a clipping defect.
232    if (cs.textOverflow === 'ellipsis') continue;
233    push('text-clipped', el,
234      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236  }
237  // 4. Interactive elements covered by a different element.
238  const interactive =
239    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240  const targets = document.querySelectorAll(interactive);
241  for (const el of targets) {
242    if (isIgnored(el)) continue;
243    const r = el.getBoundingClientRect();
244    if (r.width < 6 || r.height < 6) continue;
245    const cs = getComputedStyle(el);
246    if (!visible(cs)) continue;
247    const cx = r.left + r.width / 2;
248    const cy = r.top + r.height / 2;
249    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250    const top = document.elementFromPoint(cx, cy);
251    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252    if (isIgnored(top)) continue;
253    const tcs = getComputedStyle(top);
254    if (!visible(tcs)) continue;
255    if (tcs.pointerEvents === 'none') continue;
256    const tr = top.getBoundingClientRect();
257    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260    push('element-overlap', el,
261      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262      ' is covered by <' + tname + '>');
263  }
264  return JSON.stringify(issues);
265})()
266"#;
267
268/// How long the CDP connection stays open after the browser goes quiet.
269///
270/// `headless_chrome` ships a 30s default and tears down the entire connection
271/// when no traffic arrives for that long; a run must own its connection for
272/// its full duration instead.
273const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275/// Executes a [`Scenario`] against a real browser with optional LLM
276/// assistance for element targeting and assertions.
277pub struct ScenarioRunner {
278    config: ScenarioConfig,
279    definitions: HashMap<String, AssertDefinition>,
280    llm: LlmConfig,
281    timeout: Duration,
282    viewport_width: u32,
283    viewport_height: u32,
284    /// The viewport currently applied in the browser (CDP emulation).
285    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
286    applied_viewport: std::cell::Cell<(u32, u32)>,
287    endpoints: EndpointRegistry,
288    usage: Arc<UsageTracker>,
289    budgets: BudgetTracker,
290    /// Directory for failure artifacts (screenshots).
291    artifacts_dir: PathBuf,
292    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
293    reporter: Arc<Reporter>,
294    /// The test + step index currently executing, for LLM-call events.
295    current_step: std::cell::RefCell<Option<(String, u32)>>,
296    /// Epoch seconds of runner creation — constant for the whole run so
297    /// the LLM context preamble below is cacheable by upstream providers.
298    run_started: u64,
299    /// Whether this runner emits the `RunStarted`/`RunFinished` events.
300    /// When several scenario files are run concurrently by
301    /// [`crate::parallel::run_scenarios`], the orchestrator owns those two
302    /// events (one per batch), so each per-file runner suppresses its own.
303    emit_run_events: bool,
304}
305
306/// Aggregated results from a scenario run.
307#[derive(Debug, Default)]
308pub struct RunReport {
309    /// Number of tests that passed.
310    pub tests_passed: u32,
311    /// Number of tests that failed.
312    pub tests_failed: u32,
313    /// Number of steps that passed.
314    pub passed: u32,
315    /// Number of steps that failed.
316    pub failed: u32,
317    /// Number of steps that were skipped.
318    pub skipped: u32,
319    /// Per-step details.
320    pub details: Vec<StepResult>,
321}
322
323/// Result of a single step execution.
324#[derive(Debug, Clone)]
325pub struct StepResult {
326    /// The step name.
327    pub name: String,
328    /// Whether the step passed, failed, or was skipped.
329    pub status: StepStatus,
330    /// Human-readable result message.
331    pub message: String,
332}
333
334/// Outcome for a single step.
335pub use crate::events::StepStatus;
336
337/// Predefined assertion preset definition.
338struct AssertPreset {
339    name: &'static str,
340    system: &'static str,
341    user_template: &'static str,
342}
343
344/// Built-in assertion presets.
345#[allow(clippy::literal_string_with_formatting_args)]
346const ASSERTION_PRESETS: &[AssertPreset] = &[
347    AssertPreset {
348        name: "no_error_on_page",
349        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
350        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
351    },
352    AssertPreset {
353        name: "text_visible",
354        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
355        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
356    },
357    AssertPreset {
358        name: "element_exists",
359        system: "You are a QA tester. Check if a described UI element exists on a web page.",
360        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
361    },
362    AssertPreset {
363        name: "visual_no_issues",
364        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
365        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
366    },
367    AssertPreset {
368        name: "visual_no_overlaps",
369        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
370        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
371    },
372    AssertPreset {
373        name: "visual_text_visible",
374        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
375        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
376    },
377    AssertPreset {
378        name: "layout_no_issues",
379        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
380        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
381    },
382];
383
384impl ScenarioRunner {
385    /// Creates a new runner with the given scenario configuration and
386    /// assertion definitions.
387    #[must_use]
388    #[allow(clippy::needless_pass_by_value)]
389    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
390        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
391    }
392
393    /// Creates a runner that reports run events through the given reporter.
394    #[must_use]
395    #[allow(clippy::needless_pass_by_value)]
396    pub fn with_reporter(
397        scenario_config: ScenarioConfig,
398        definitions: Vec<AssertDefinition>,
399        reporter: Arc<Reporter>,
400    ) -> Self {
401        Self::with_reporter_mode(scenario_config, definitions, reporter, true)
402    }
403
404    /// Creates a runner that reports run events through the given reporter
405    /// but suppresses the `RunStarted`/`RunFinished` events. Used by the
406    /// parallel orchestrator, which emits those once per batch.
407    #[must_use]
408    #[allow(clippy::needless_pass_by_value)]
409    pub fn with_reporter_parallel(
410        scenario_config: ScenarioConfig,
411        definitions: Vec<AssertDefinition>,
412        reporter: Arc<Reporter>,
413    ) -> Self {
414        Self::with_reporter_mode(scenario_config, definitions, reporter, false)
415    }
416
417    #[must_use]
418    #[allow(clippy::needless_pass_by_value)]
419    fn with_reporter_mode(
420        scenario_config: ScenarioConfig,
421        definitions: Vec<AssertDefinition>,
422        reporter: Arc<Reporter>,
423        emit_run_events: bool,
424    ) -> Self {
425        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
426            &scenario_config,
427        ));
428        let llm = LlmConfig {
429            url: scenario_config
430                .llm_url
431                .clone()
432                .unwrap_or_else(crate::llm_base_url),
433            model: scenario_config
434                .llm_model
435                .clone()
436                .unwrap_or_else(crate::llm_model),
437            api_key: scenario_config
438                .llm_api_key
439                .clone()
440                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
441            headers: if scenario_config.llm_headers.is_empty() {
442                crate::parse_headers_env()
443            } else {
444                scenario_config.llm_headers.clone()
445            },
446            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
447            temperature: scenario_config.temperature,
448            thinking: scenario_config.thinking,
449            model_params: scenario_config.model_params.clone(),
450            max_attempts: crate::default_llm_attempts(),
451            provider: crate::scenario::Provider::Openai,
452            deployment: None,
453            api_version: None,
454            auth: crate::scenario::AuthConfig::default(),
455            header_commands: std::collections::HashMap::new(),
456            aws: crate::scenario::AwsConfig::default(),
457        };
458        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
459        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
460        let defs_map: HashMap<String, AssertDefinition> = definitions
461            .into_iter()
462            .map(|d| (d.name.clone(), d))
463            .collect();
464
465        Self {
466            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
467            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
468            viewport_height: scenario_config.viewport_height.unwrap_or(720),
469            applied_viewport: std::cell::Cell::new((0, 0)),
470            config: scenario_config.clone(),
471            definitions: defs_map,
472            llm,
473            endpoints,
474            usage: Arc::new(UsageTracker::new()),
475            budgets,
476            artifacts_dir: PathBuf::from(
477                scenario_config
478                    .artifacts_dir
479                    .unwrap_or_else(|| "artifacts".to_owned()),
480            ),
481            reporter,
482            current_step: std::cell::RefCell::new(None),
483            run_started: SystemTime::now()
484                .duration_since(UNIX_EPOCH)
485                .map_or(0, |d| d.as_secs()),
486            emit_run_events,
487        }
488    }
489
490    /// Emits an event; a sink failure degrades to a console warning so a
491    /// broken log file can never mask the run itself.
492    fn emit_event(&self, event: &TestEvent) {
493        if let Err(err) = self.reporter.emit(event) {
494            use std::io::Write as _;
495            let mut out = std::io::stderr().lock();
496            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
497        }
498    }
499
500    /// Returns a clone of the [`UsageTracker`] for reporting.
501    #[must_use]
502    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
503        Arc::clone(&self.usage)
504    }
505
506    /// Returns a reference to the [`BudgetTracker`].
507    #[must_use]
508    pub const fn budget_tracker(&self) -> &BudgetTracker {
509        &self.budgets
510    }
511
512    /// Executes all test groups in the scenario and returns a report.
513    ///
514    /// # Errors
515    ///
516    /// Returns an error if the browser fails to launch.
517    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
518    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
519        let mut report = RunReport::default();
520
521        if tests.is_empty() {
522            self.reporter.warn("No tests defined in scenario.");
523            if self.emit_run_events {
524                self.emit_event(&TestEvent::RunFinished {
525                    tests_passed: 0,
526                    tests_failed: 0,
527                    steps_passed: 0,
528                    steps_failed: 0,
529                    steps_skipped: 0,
530                    total_cost: 0.0,
531                    total_tokens: 0,
532                    total_input_tokens: 0,
533                    total_output_tokens: 0,
534                    total_cached_input_tokens: 0,
535                    total_cache_creation_input_tokens: 0,
536                    models: Vec::new(),
537                    total_calls: 0,
538                });
539            }
540            return Ok(report);
541        }
542
543        if self.emit_run_events {
544            self.emit_event(&TestEvent::RunStarted {
545                total_tests: tests.len() as u32,
546            });
547        }
548
549        let browser_headless = self.config.browser_headless.unwrap_or(true);
550
551        let launch_opts = LaunchOptions {
552            headless: browser_headless,
553            window_size: Some((self.viewport_width, self.viewport_height)),
554            sandbox: false,
555            // headless_chrome defaults this to 30s and shuts down the whole CDP
556            // connection when no messages arrive for that long. A scenario can
557            // easily exceed 30s of browser silence (slow LLM targeting/assertion
558            // calls, page waits, budget checks between steps), after which every
559            // remaining step fails with "Unable to make method calls because
560            // underlying connection is closed" — one quiet gap kills the run.
561            // Open-ended scenarios must own the connection for their full
562            // duration, so keep it alive for 6 hours.
563            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
564            ..LaunchOptions::default()
565        };
566
567        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
568        let tab = browser.new_tab().context("failed to open browser tab")?;
569        let _ = tab.set_default_timeout(self.timeout);
570
571        // Start MCP server if configured
572        #[cfg(feature = "mcp-server")]
573        if let Some(ref mcp_cfg) = self.config.mcp_server {
574            if mcp_cfg.enabled {
575                let port = mcp_cfg.port;
576                std::thread::spawn(move || {
577                    let _ = crate::mcp_server::start_mcp_server(port);
578                });
579            }
580        }
581        #[cfg(not(feature = "mcp-server"))]
582        if let Some(mcp_cfg) = &self.config.mcp_server {
583            if mcp_cfg.enabled {
584                self.reporter
585                    .warn("MCP server configured but 'mcp-server' feature not enabled");
586            }
587        }
588
589        // Start A2A agent server if configured
590        #[cfg(feature = "a2a-server")]
591        if let Some(ref a2a_cfg) = self.config.a2a_server {
592            if a2a_cfg.enabled {
593                let port = a2a_cfg.port;
594                tokio::spawn(crate::a2a_server::start_a2a_server(port));
595            }
596        }
597        #[cfg(not(feature = "a2a-server"))]
598        if let Some(a2a_cfg) = &self.config.a2a_server {
599            if a2a_cfg.enabled {
600                self.reporter
601                    .warn("A2A server configured but 'a2a-server' feature not enabled");
602            }
603        }
604
605        for test in tests {
606            self.emit_event(&TestEvent::TestStarted {
607                test: test.name.clone(),
608            });
609
610            self.usage.reset_per_test();
611
612            let test_started = Instant::now();
613            let test_result = self.run_test(test, &tab);
614            let duration_ms = test_started.elapsed().as_millis() as u64;
615            let usage = self.usage.current_test_snapshot();
616            self.usage.commit_test(&test.name);
617
618            self.emit_event(&TestEvent::TestFinished {
619                test: test.name.clone(),
620                passed: test_result.passed,
621                failed: test_result.failed,
622                skipped: test_result.skipped,
623                duration_ms,
624                cost: usage.total_cost,
625                tokens: usage.total_tokens,
626                input_tokens: usage.total_input_tokens,
627                output_tokens: usage.total_output_tokens,
628                cached_input_tokens: usage.total_cached_input_tokens,
629                cache_creation_input_tokens: usage.total_cache_creation_input_tokens,
630                models: usage.models.clone(),
631                calls: usage.total_calls,
632            });
633
634            if test_result.failed == 0 && test_result.total > 0 {
635                report.tests_passed += 1;
636            } else if test_result.total > 0 {
637                report.tests_failed += 1;
638            }
639
640            report.passed += test_result.passed;
641            report.failed += test_result.failed;
642            report.skipped += test_result.skipped;
643            report.details.extend(test_result.details);
644        }
645
646        let global = self.usage.global_snapshot();
647        if self.emit_run_events {
648            self.emit_event(&TestEvent::RunFinished {
649                tests_passed: report.tests_passed,
650                tests_failed: report.tests_failed,
651                steps_passed: report.passed,
652                steps_failed: report.failed,
653                steps_skipped: report.skipped,
654                total_cost: global.total_cost,
655                total_tokens: global.total_tokens,
656                total_input_tokens: global.total_input_tokens,
657                total_output_tokens: global.total_output_tokens,
658                total_cached_input_tokens: global.total_cached_input_tokens,
659                total_cache_creation_input_tokens: global.total_cache_creation_input_tokens,
660                models: global.models.clone(),
661                total_calls: global.total_calls,
662            });
663        }
664
665        Ok(report)
666    }
667
668    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
669    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
670        let base_url = test
671            .base_url
672            .clone()
673            .or_else(|| self.config.base_url.clone())
674            .unwrap_or_else(crate::base_url);
675
676        // Per-test viewport override: switch the browser via CDP
677        // device-metrics emulation before this test runs.
678        let vw = test.viewport_width.unwrap_or(self.viewport_width);
679        let vh = test.viewport_height.unwrap_or(self.viewport_height);
680        if self.applied_viewport.get() != (vw, vh) {
681            self.apply_viewport(tab, vw, vh);
682            self.applied_viewport.set((vw, vh));
683        }
684
685        // Per-test isolation: every test starts from its own start_url
686        // (unless auto_navigate is disabled), so a test never inherits the
687        // previous test's page state.
688        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
689
690        let start_url = test
691            .start_url
692            .clone()
693            .or_else(|| self.config.start_url.clone())
694            .unwrap_or_else(|| "/dashboard".to_owned());
695
696        if auto_navigate {
697            let full_url = resolve_url(&start_url, &base_url);
698            self.reporter.debug(format!("auto-navigate: {full_url}"));
699            let _ = tab.navigate_to(&full_url);
700            let _ = tab.wait_until_navigated();
701            std::thread::sleep(Duration::from_secs(4));
702        }
703
704        let mut result = TestRunResult::default();
705
706        for (step_index, step) in test.steps.iter().enumerate() {
707            result.total += 1;
708
709            let wait_ms = match step {
710                TestStep::Navigate { wait_after_ms, .. }
711                | TestStep::Click { wait_after_ms, .. }
712                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
713                _ => None,
714            };
715
716            self.current_step
717                .replace(Some((test.name.clone(), step_index as u32)));
718            self.emit_event(&TestEvent::StepStarted {
719                test: test.name.clone(),
720                index: step_index as u32,
721                label: step_label(step),
722            });
723            let step_started = Instant::now();
724
725            let mut step_result = match step {
726                TestStep::Navigate { url, .. } => {
727                    let full_url = resolve_url(url, &base_url);
728                    run_navigate_step(&full_url, tab)
729                }
730                TestStep::Click {
731                    target,
732                    selector,
733                    endpoint,
734                    idempotent,
735                    ..
736                } => self.run_click(
737                    target,
738                    selector.as_deref(),
739                    endpoint.as_deref(),
740                    test.endpoint.as_deref(),
741                    *idempotent,
742                    tab,
743                ),
744                TestStep::Type {
745                    target,
746                    text,
747                    selector,
748                    endpoint,
749                    idempotent,
750                    ..
751                } => self.run_type(
752                    target,
753                    text,
754                    selector.as_deref(),
755                    endpoint.as_deref(),
756                    test.endpoint.as_deref(),
757                    *idempotent,
758                    tab,
759                ),
760                TestStep::Wait {
761                    target,
762                    selector,
763                    text,
764                    timeout_ms,
765                    endpoint,
766                    idempotent,
767                } => self.run_wait(
768                    target,
769                    selector.as_deref(),
770                    text.as_deref(),
771                    *timeout_ms,
772                    endpoint.as_deref(),
773                    test.endpoint.as_deref(),
774                    *idempotent,
775                    tab,
776                ),
777                TestStep::Assert {
778                    definition,
779                    preset,
780                    prompt,
781                    assert_text,
782                    endpoint,
783                    screenshot,
784                } => self.run_assert(
785                    definition.as_deref(),
786                    preset.as_deref(),
787                    prompt.as_deref(),
788                    assert_text.as_deref(),
789                    *screenshot,
790                    endpoint.as_deref(),
791                    test.endpoint.as_deref(),
792                    tab,
793                ),
794                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
795                TestStep::Agent {
796                    agent,
797                    task,
798                    definition,
799                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
800                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
801            };
802
803            // Failure diagnostics: capture the page state and a screenshot so
804            // CI logs say WHAT the page looked like when the step failed,
805            // instead of a bare "timed out: The event waited for never came".
806            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
807                let state = diagnostics::capture(tab);
808                let screenshot = diagnostics::save_screenshot(
809                    tab,
810                    &self.artifacts_dir,
811                    &test.name,
812                    &test.name,
813                    step_index,
814                    step_kind_label(step),
815                );
816                step_result.message = format!(
817                    "{base} — {excerpt}",
818                    base = step_result.message,
819                    excerpt = diagnostics::inline_excerpt(&state),
820                );
821                (Some(diagnostics::full_context(&state)), screenshot)
822            } else {
823                (None, None)
824            };
825
826            let duration_ms = step_started.elapsed().as_millis() as u64;
827            self.emit_event(&TestEvent::StepFinished {
828                test: test.name.clone(),
829                index: step_index as u32,
830                label: step_result.name.clone(),
831                status: step_result.status,
832                duration_ms,
833                message: step_result.message.clone(),
834                diagnostics: diagnostics_block,
835                screenshot: screenshot_path,
836            });
837            self.current_step.replace(None);
838
839            match step_result.status {
840                StepStatus::Passed => result.passed += 1,
841                StepStatus::Failed => result.failed += 1,
842                StepStatus::Skipped => result.skipped += 1,
843            }
844
845            // Fail fast: the first failed step ends the test and the
846            // remaining steps are reported as skipped (no LLM budget is
847            // burned asserting against a page that is already known broken).
848            if step_result.status == StepStatus::Failed
849                && !self.config.continue_on_failure
850                && step_index + 1 < test.steps.len()
851            {
852                self.emit_event(&TestEvent::Warning {
853                    message: format!(
854                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
855                        test.steps.len() - step_index - 1
856                    ),
857                });
858                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
859                    let skipped_index = step_index + 1 + offset;
860                    let label = step_label(skipped);
861                    result.total += 1;
862                    result.skipped += 1;
863                    self.emit_event(&TestEvent::StepStarted {
864                        test: test.name.clone(),
865                        index: skipped_index as u32,
866                        label: label.clone(),
867                    });
868                    self.emit_event(&TestEvent::StepFinished {
869                        test: test.name.clone(),
870                        index: skipped_index as u32,
871                        label,
872                        status: StepStatus::Skipped,
873                        duration_ms: 0,
874                        message: "skipped: previous step failed".into(),
875                        diagnostics: None,
876                        screenshot: None,
877                    });
878                    result.details.push(StepResult {
879                        name: step_label(skipped),
880                        status: StepStatus::Skipped,
881                        message: "skipped: previous step failed".into(),
882                    });
883                }
884                result.details.push(step_result);
885                return result;
886            }
887
888            // Check per-test budget after each step
889            let test_usage = self.usage.current_test_snapshot();
890            let global_usage = self.usage.global_snapshot();
891            let budget_status = self.budgets.check_all(
892                &test.name,
893                &test_usage,
894                &global_usage,
895                test.budget.as_ref(),
896            );
897            match budget_status {
898                BudgetStatus::HardExceeded { message, .. } => {
899                    self.emit_event(&TestEvent::Warning {
900                        message: format!("budget exceeded: {message}"),
901                    });
902                    result.details.push(StepResult {
903                        name: "[budget]".into(),
904                        status: StepStatus::Failed,
905                        message,
906                    });
907                    result.failed += 1;
908                    return result;
909                }
910                BudgetStatus::SoftExceeded { message, .. } => {
911                    self.emit_event(&TestEvent::Warning {
912                        message: format!("budget warning: {message}"),
913                    });
914                }
915                BudgetStatus::Ok => {}
916            }
917
918            if let Some(ms) = wait_ms {
919                std::thread::sleep(Duration::from_millis(ms));
920            }
921
922            result.details.push(step_result);
923        }
924
925        result
926    }
927
928    /// Applies a viewport size to the current tab via CDP
929    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
930    /// overrides and the viewport matrix. The initial window size set at
931    /// browser launch is replaced by emulation; failures are logged but
932    /// do not fail the test (a mismatched viewport only weakens coverage).
933    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
934        use headless_chrome::protocol::cdp::Emulation;
935        let _ = self;
936        let params = Emulation::SetDeviceMetricsOverride {
937            width,
938            height,
939            device_scale_factor: 1.0,
940            mobile: false,
941            scale: None,
942            screen_width: Some(width),
943            screen_height: Some(height),
944            position_x: None,
945            position_y: None,
946            dont_set_visible_size: None,
947            screen_orientation: None,
948            viewport: None,
949            display_feature: None,
950            device_posture: None,
951        };
952        self.reporter.debug(format!("viewport: {width}x{height}"));
953        if let Err(e) = tab.call_method(params) {
954            self.reporter
955                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
956        }
957    }
958
959    /// Height (px) covered by assert-step screenshots: the configured
960    /// `screenshot_max_height` (absolute px or viewport multiple, default
961    /// `"20x"`) resolved against the currently applied viewport, raised to
962    /// at least the viewport height so the visible screen is always fully
963    /// included. The capture is split into viewport-tall tiles, so this
964    /// value bounds total coverage (and hence the number of image parts).
965    #[must_use]
966    fn screenshot_height_cap(&self) -> u32 {
967        let viewport_height = self.current_viewport_height();
968        let cap = self
969            .config
970            .screenshot_max_height
971            .as_ref()
972            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
973        cap.max(viewport_height)
974    }
975
976    /// Height of the viewport currently emulated in the browser (falling
977    /// back to the configured default before any emulation was applied).
978    #[must_use]
979    const fn current_viewport_height(&self) -> u32 {
980        let (_, height) = self.applied_viewport.get();
981        if height > 0 {
982            height
983        } else {
984            self.viewport_height
985        }
986    }
987
988    // ── step handlers ───────────────────────────────────────────────────
989
990    #[allow(clippy::too_many_lines)]
991    fn run_click(
992        &self,
993        target: &str,
994        selector_override: Option<&str>,
995        step_endpoint: Option<&str>,
996        test_endpoint: Option<&str>,
997        idempotent: bool,
998        tab: &Tab,
999    ) -> StepResult {
1000        let name = format!("[click] {target}");
1001        let selector = match self.resolve_selector(
1002            selector_override,
1003            target,
1004            step_endpoint,
1005            test_endpoint,
1006            tab,
1007        ) {
1008            Ok(s) => s,
1009            Err(msg) => {
1010                if idempotent {
1011                    return StepResult {
1012                        name,
1013                        status: StepStatus::Skipped,
1014                        message: format!("skipped (idempotent): no target found — {msg}"),
1015                    };
1016                }
1017                return StepResult {
1018                    name,
1019                    status: StepStatus::Failed,
1020                    message: msg,
1021                };
1022            }
1023        };
1024
1025        // Idempotent steps probe briefly: a missing target means the
1026        // action was already done / not applicable (e.g. an
1027        // already-authenticated session), and skipping is the success
1028        // path, not a failure.
1029        let probe_secs = if idempotent { 5 } else { 10 };
1030        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1031            Ok(element) => match element.click() {
1032                Ok(_) => StepResult {
1033                    name,
1034                    status: StepStatus::Passed,
1035                    message: format!("clicked {selector}"),
1036                },
1037                Err(e) => StepResult {
1038                    name,
1039                    status: StepStatus::Failed,
1040                    message: format!("click failed on {selector}: {e}"),
1041                },
1042            },
1043            Err(e) if idempotent => StepResult {
1044                name,
1045                status: StepStatus::Skipped,
1046                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1047            },
1048            Err(e) => StepResult {
1049                name,
1050                status: StepStatus::Failed,
1051                message: format!("element {selector} not found: {e}"),
1052            },
1053        }
1054    }
1055
1056    #[allow(clippy::too_many_arguments)]
1057    fn run_type(
1058        &self,
1059        target: &str,
1060        text: &str,
1061        selector_override: Option<&str>,
1062        step_endpoint: Option<&str>,
1063        test_endpoint: Option<&str>,
1064        idempotent: bool,
1065        tab: &Tab,
1066    ) -> StepResult {
1067        let name = format!("[type] {target}");
1068        let selector = match self.resolve_selector(
1069            selector_override,
1070            target,
1071            step_endpoint,
1072            test_endpoint,
1073            tab,
1074        ) {
1075            Ok(s) => s,
1076            Err(msg) => {
1077                if idempotent {
1078                    return StepResult {
1079                        name,
1080                        status: StepStatus::Skipped,
1081                        message: format!("skipped (idempotent): no target found — {msg}"),
1082                    };
1083                }
1084                return StepResult {
1085                    name,
1086                    status: StepStatus::Failed,
1087                    message: msg,
1088                };
1089            }
1090        };
1091
1092        let probe_secs = if idempotent { 5 } else { 10 };
1093        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1094            Ok(element) => {
1095                if let Err(e) = element.click() {
1096                    return StepResult {
1097                        name,
1098                        status: StepStatus::Failed,
1099                        message: format!("click to focus {selector} failed: {e}"),
1100                    };
1101                }
1102
1103                let js = format!(
1104                    "document.querySelector('{}').value = '';",
1105                    selector.replace('\'', "\\'")
1106                );
1107                let _ = tab.evaluate(&js, false);
1108
1109                match element.type_into(text) {
1110                    Ok(_) => StepResult {
1111                        name,
1112                        status: StepStatus::Passed,
1113                        message: format!("typed {text:?} into {selector}"),
1114                    },
1115                    Err(e) => StepResult {
1116                        name,
1117                        status: StepStatus::Failed,
1118                        message: format!("type into {selector} failed: {e}"),
1119                    },
1120                }
1121            }
1122            Err(e) if idempotent => StepResult {
1123                name,
1124                status: StepStatus::Skipped,
1125                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1126            },
1127            Err(e) => StepResult {
1128                name,
1129                status: StepStatus::Failed,
1130                message: format!("element {selector} not found: {e}"),
1131            },
1132        }
1133    }
1134
1135    #[allow(clippy::too_many_arguments)]
1136    #[allow(clippy::too_many_lines)]
1137    fn run_wait(
1138        &self,
1139        target: &str,
1140        selector_override: Option<&str>,
1141        text: Option<&str>,
1142        timeout_ms: Option<u64>,
1143        step_endpoint: Option<&str>,
1144        test_endpoint: Option<&str>,
1145        idempotent: bool,
1146        tab: &Tab,
1147    ) -> StepResult {
1148        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1149        let step_name = format!("[wait] {target}");
1150
1151        // Resolve an explicit selector only (text-only waits are LLM-free).
1152        let selector = match selector_override {
1153            Some(s) => Some(s.to_owned()),
1154            None if text.is_some() => None,
1155            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1156                Ok(s) => Some(s),
1157                Err(msg) => {
1158                    if idempotent {
1159                        return StepResult {
1160                            name: step_name,
1161                            status: StepStatus::Skipped,
1162                            message: format!("skipped (idempotent): no target found — {msg}"),
1163                        };
1164                    }
1165                    return StepResult {
1166                        name: step_name,
1167                        status: StepStatus::Failed,
1168                        message: msg,
1169                    };
1170                }
1171            },
1172        };
1173
1174        if text.is_some() {
1175            let sel_js = selector
1176                .as_deref()
1177                .map(crate::selectors::selector_matches_js);
1178            let text_js = text.map(|t| {
1179                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1180                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1181            });
1182
1183            let deadline = Instant::now() + timeout;
1184            loop {
1185                let sel_ok = sel_js
1186                    .as_ref()
1187                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1188                let text_ok = text_js
1189                    .as_ref()
1190                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1191                if sel_ok && text_ok {
1192                    let mut what = Vec::new();
1193                    if let Some(sel) = &selector {
1194                        what.push(format!("found {sel}"));
1195                    }
1196                    if let Some(t) = text {
1197                        what.push(format!("text {t:?} visible"));
1198                    }
1199                    return StepResult {
1200                        name: step_name,
1201                        status: StepStatus::Passed,
1202                        message: what.join(" and "),
1203                    };
1204                }
1205                if Instant::now() >= deadline {
1206                    let mut what = Vec::new();
1207                    if let Some(sel) = &selector {
1208                        what.push(sel.clone());
1209                    }
1210                    if let Some(t) = text {
1211                        what.push(format!("text {t:?}"));
1212                    }
1213                    let message = format!(
1214                        "wait for {} timed out after {}ms: the event waited for never came",
1215                        what.join(" / "),
1216                        timeout.as_millis(),
1217                    );
1218                    if idempotent {
1219                        return StepResult {
1220                            name: step_name,
1221                            status: StepStatus::Skipped,
1222                            message: format!("skipped (idempotent): {message}"),
1223                        };
1224                    }
1225                    return StepResult {
1226                        name: step_name,
1227                        status: StepStatus::Failed,
1228                        message,
1229                    };
1230                }
1231                std::thread::sleep(Duration::from_millis(250));
1232            }
1233        }
1234
1235        match selector.as_deref() {
1236            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1237                Ok(_) => StepResult {
1238                    name: step_name,
1239                    status: StepStatus::Passed,
1240                    message: format!("found {sel}"),
1241                },
1242                Err(e) if idempotent => StepResult {
1243                    name: step_name,
1244                    status: StepStatus::Skipped,
1245                    message: format!(
1246                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1247                        timeout.as_millis()
1248                    ),
1249                },
1250                Err(e) => StepResult {
1251                    name: step_name,
1252                    status: StepStatus::Failed,
1253                    message: format!(
1254                        "wait for {sel} timed out after {}ms: {e}",
1255                        timeout.as_millis()
1256                    ),
1257                },
1258            },
1259            None => StepResult {
1260                name: step_name,
1261                status: StepStatus::Failed,
1262                message: "wait step has neither selector nor text".into(),
1263            },
1264        }
1265    }
1266
1267    #[allow(clippy::too_many_arguments)]
1268    fn run_assert(
1269        &self,
1270        definition: Option<&str>,
1271        preset: Option<&str>,
1272        prompt: Option<&str>,
1273        assert_text: Option<&str>,
1274        screenshot: bool,
1275        step_endpoint: Option<&str>,
1276        test_endpoint: Option<&str>,
1277        tab: &Tab,
1278    ) -> StepResult {
1279        std::thread::sleep(Duration::from_millis(500));
1280
1281        let page_content = get_page_text(tab);
1282
1283        // Vision attach: capture the full page once per assert step and
1284        // split it into viewport-tall tiles (the total coverage is bounded
1285        // by the configured height cap so vision tokens stay sane). All
1286        // tile data URLs are handed to the preset/prompt evaluation below.
1287        let image: Option<Vec<String>> = if screenshot {
1288            let endpoint = self
1289                .endpoints
1290                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1291            if !endpoint.vision {
1292                return StepResult {
1293                    name: "[assert]".into(),
1294                    status: StepStatus::Failed,
1295                    message: format!(
1296                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1297                        name = endpoint.name
1298                    ),
1299                };
1300            }
1301            match crate::vision::capture_screenshot_data_urls(
1302                tab,
1303                self.config
1304                    .screenshot_max_dimension
1305                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1306                self.screenshot_height_cap(),
1307                self.current_viewport_height(),
1308            ) {
1309                Ok(urls) => Some(urls),
1310                Err(e) => {
1311                    return StepResult {
1312                        name: "[assert]".into(),
1313                        status: StepStatus::Failed,
1314                        message: format!("screenshot capture failed: {e}"),
1315                    };
1316                }
1317            }
1318        } else {
1319            None
1320        };
1321
1322        if let Some(def_name) = definition {
1323            if let Some(def) = self.definitions.get(def_name) {
1324                return self.run_assert_def(
1325                    def,
1326                    &page_content,
1327                    image.as_deref(),
1328                    step_endpoint,
1329                    test_endpoint,
1330                    tab,
1331                );
1332            }
1333            return StepResult {
1334                name: format!("[assert] {def_name}"),
1335                status: StepStatus::Failed,
1336                message: format!("definition '{def_name}' not found"),
1337            };
1338        }
1339
1340        if let Some(preset_name) = preset {
1341            // Deterministic DOM layout scan — runs JS in the browser and
1342            // never calls the LLM (free, fast, no pixel budget).
1343            if preset_name == "layout_no_issues" {
1344                return self.run_layout_preset(tab);
1345            }
1346            return self.run_preset(
1347                preset_name,
1348                assert_text,
1349                &page_content,
1350                image.as_deref(),
1351                step_endpoint,
1352                test_endpoint,
1353            );
1354        }
1355
1356        if let Some(prompt_text) = prompt {
1357            return self.run_custom(
1358                prompt_text,
1359                &page_content,
1360                image.as_deref(),
1361                step_endpoint,
1362                test_endpoint,
1363            );
1364        }
1365
1366        StepResult {
1367            name: "[assert]".into(),
1368            status: StepStatus::Skipped,
1369            message: "no definition, preset, or prompt specified".into(),
1370        }
1371    }
1372
1373    fn run_assert_def(
1374        &self,
1375        def: &AssertDefinition,
1376        page_content: &PageContent,
1377        image: Option<&[String]>,
1378        step_endpoint: Option<&str>,
1379        test_endpoint: Option<&str>,
1380        tab: &Tab,
1381    ) -> StepResult {
1382        // Agent-based definition: delegate to an A2A agent
1383        if let Some(ref agent) = def.agent {
1384            if image.is_some() {
1385                return StepResult {
1386                    name: format!("[assert] {}", def.name),
1387                    status: StepStatus::Failed,
1388                    message: "agent-backed assertions do not support screenshots".into(),
1389                };
1390            }
1391            let task = def
1392                .task_template
1393                .as_deref()
1394                .unwrap_or("Evaluate the assertion")
1395                .replace("{url}", &page_content.url)
1396                .replace("{title}", &page_content.title)
1397                .replace("{content}", &page_content.body_text)
1398                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1399
1400            return self.run_agent_step(agent, &task, &def.name);
1401        }
1402
1403        // Custom preset: system + user_template provided in the definition
1404        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1405            return self.run_custom_preset(
1406                &def.name,
1407                system,
1408                template,
1409                def.assert_text.as_deref(),
1410                page_content,
1411                image,
1412                step_endpoint,
1413                test_endpoint,
1414            );
1415        }
1416
1417        def.preset.as_ref().map_or_else(
1418            || {
1419                def.prompt.as_ref().map_or_else(
1420                    || StepResult {
1421                        name: format!("[assert] {}", def.name),
1422                        status: StepStatus::Failed,
1423                        message: "definition has no preset, prompt, or system+user_template".into(),
1424                    },
1425                    |prompt| {
1426                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1427                    },
1428                )
1429            },
1430            |preset_name| {
1431                if preset_name == "layout_no_issues" {
1432                    return self.run_layout_preset(tab);
1433                }
1434                self.run_preset(
1435                    preset_name,
1436                    def.assert_text.as_deref(),
1437                    page_content,
1438                    image,
1439                    step_endpoint,
1440                    test_endpoint,
1441                )
1442            },
1443        )
1444    }
1445
1446    #[allow(clippy::too_many_arguments)]
1447    fn run_custom_preset(
1448        &self,
1449        name: &str,
1450        system: &str,
1451        template: &str,
1452        assert_text: Option<&str>,
1453        page_content: &PageContent,
1454        image: Option<&[String]>,
1455        step_endpoint: Option<&str>,
1456        test_endpoint: Option<&str>,
1457    ) -> StepResult {
1458        let user_prompt = template
1459            .replace("{url}", &page_content.url)
1460            .replace("{title}", &page_content.title)
1461            .replace("{content}", &page_content.body_text)
1462            .replace("{expected_text}", assert_text.unwrap_or(""))
1463            .replace("{description}", "");
1464
1465        // Custom preset definitions frequently forget the {content}
1466        // placeholder — without it the LLM has no page to evaluate and
1467        // answers "I can't determine that without seeing the page". Always
1468        // append the page context unless the template already references it.
1469        let user_prompt = if template.contains("{content}") {
1470            user_prompt
1471        } else {
1472            format!(
1473                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1474                url = page_content.url,
1475                title = page_content.title,
1476                content = page_content.body_text,
1477            )
1478        };
1479
1480        self.reporter
1481            .debug(format!("assert: {name} (custom preset)"));
1482
1483        let chain = self
1484            .endpoints
1485            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1486        let sys = system.to_owned();
1487
1488        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1489
1490        response.map_or_else(
1491            |e| StepResult {
1492                name: format!("[assert] {name}"),
1493                status: StepStatus::Failed,
1494                message: format!("LLM assertion call failed: {e}"),
1495            },
1496            |(lr, _idx)| {
1497                let content_lower = lr.content.to_lowercase().trim().to_owned();
1498                if content_lower.starts_with("pass") {
1499                    StepResult {
1500                        name: format!("[assert] {name}"),
1501                        status: StepStatus::Passed,
1502                        message: "PASS".into(),
1503                    }
1504                } else {
1505                    StepResult {
1506                        name: format!("[assert] {name}"),
1507                        status: StepStatus::Failed,
1508                        message: lr.content,
1509                    }
1510                }
1511            },
1512        )
1513    }
1514
1515    fn run_preset(
1516        &self,
1517        preset_name: &str,
1518        assert_text: Option<&str>,
1519        page_content: &PageContent,
1520        image: Option<&[String]>,
1521        step_endpoint: Option<&str>,
1522        test_endpoint: Option<&str>,
1523    ) -> StepResult {
1524        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1525            return StepResult {
1526                name: format!("[assert] {preset_name}"),
1527                status: StepStatus::Failed,
1528                message: format!("unknown assertion preset: {preset_name}"),
1529            };
1530        };
1531        if preset_name.starts_with("visual_") && image.is_none() {
1532            return StepResult {
1533                name: format!("[assert] {preset_name}"),
1534                status: StepStatus::Failed,
1535                message: format!(
1536                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1537                ),
1538            };
1539        }
1540
1541        let user_prompt = preset
1542            .user_template
1543            .replace("{url}", &page_content.url)
1544            .replace("{title}", &page_content.title)
1545            .replace("{content}", &page_content.body_text)
1546            .replace("{expected_text}", assert_text.unwrap_or(""))
1547            .replace("{description}", "");
1548
1549        // Same safety net as custom presets: never let the LLM answer with
1550        // no page context at all.
1551        let user_prompt = if preset.user_template.contains("{content}") {
1552            user_prompt
1553        } else {
1554            format!(
1555                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1556                url = page_content.url,
1557                title = page_content.title,
1558                content = page_content.body_text,
1559            )
1560        };
1561
1562        self.reporter.debug(format!("assert: {preset_name}"));
1563
1564        let chain = self
1565            .endpoints
1566            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1567        let sys = preset.system.to_owned();
1568
1569        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1570
1571        response.map_or_else(
1572            |e| StepResult {
1573                name: format!("[assert] {preset_name}"),
1574                status: StepStatus::Failed,
1575                message: format!("LLM assertion call failed: {e}"),
1576            },
1577            |(lr, _idx)| {
1578                let content_lower = lr.content.to_lowercase().trim().to_owned();
1579                if content_lower.starts_with("pass") {
1580                    StepResult {
1581                        name: format!("[assert] {preset_name}"),
1582                        status: StepStatus::Passed,
1583                        message: "PASS".into(),
1584                    }
1585                } else {
1586                    StepResult {
1587                        name: format!("[assert] {preset_name}"),
1588                        status: StepStatus::Failed,
1589                        message: lr.content,
1590                    }
1591                }
1592            },
1593        )
1594    }
1595
1596    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1597    ///
1598    /// Evaluates the layout-scan JS in the page and fails with the list of
1599    /// detected issues: horizontal page overflow, visible elements sticking
1600    /// out of the viewport, text clipped by `overflow: hidden` containers,
1601    /// and interactive elements covered by other elements. No LLM call —
1602    /// checks are geometry-based so the check is free, deterministic, and
1603    /// safe to run on every page × viewport variant.
1604    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1605        let name = "[assert] layout_no_issues".to_owned();
1606        self.reporter
1607            .debug("assert: layout_no_issues (DOM layout scan)");
1608        let js = LAYOUT_SCAN_JS.replace(
1609            "__IGNORE_CLASSES__",
1610            &serde_json::to_string(&self.config.layout_ignore_classes)
1611                .unwrap_or_else(|_| "[]".to_owned()),
1612        );
1613        let result = tab.evaluate(&js, false);
1614        let json_str = match result {
1615            Ok(r) => r
1616                .value
1617                .as_ref()
1618                .and_then(|v| v.as_str().map(String::from))
1619                .unwrap_or_else(|| "[]".to_owned()),
1620            Err(e) => {
1621                return StepResult {
1622                    name,
1623                    status: StepStatus::Failed,
1624                    message: format!("layout scan JS failed: {e}"),
1625                };
1626            }
1627        };
1628        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1629        if issues.is_empty() {
1630            return StepResult {
1631                name,
1632                status: StepStatus::Passed,
1633                message: "PASS — no layout defects detected".into(),
1634            };
1635        }
1636        let mut lines: Vec<String> = issues
1637            .iter()
1638            .take(10)
1639            .map(|i| {
1640                format!(
1641                    "- [{type_}] {element}: {detail}",
1642                    type_ = i.issue_type,
1643                    element = i.element,
1644                    detail = i.detail
1645                )
1646            })
1647            .collect();
1648        if issues.len() > 10 {
1649            lines.push(format!("- … and {} more", issues.len() - 10));
1650        }
1651        StepResult {
1652            name,
1653            status: StepStatus::Failed,
1654            message: format!(
1655                "FAIL — {} layout defect(s) detected:\n{}",
1656                issues.len(),
1657                lines.join("\n")
1658            ),
1659        }
1660    }
1661
1662    fn run_custom(
1663        &self,
1664        prompt: &str,
1665        page_content: &PageContent,
1666        image: Option<&[String]>,
1667        step_endpoint: Option<&str>,
1668        test_endpoint: Option<&str>,
1669    ) -> StepResult {
1670        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1671
1672        let mut user = format!(
1673            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1674            url = page_content.url,
1675            title = page_content.title,
1676            content = page_content.body_text,
1677        );
1678        if image.is_some() {
1679            user.push_str(
1680                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1681            );
1682        }
1683
1684        self.reporter.debug("custom assert");
1685
1686        let chain = self
1687            .endpoints
1688            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1689        let sys = system.to_owned();
1690
1691        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1692
1693        response.map_or_else(
1694            |e| StepResult {
1695                name: "[assert] custom".into(),
1696                status: StepStatus::Failed,
1697                message: format!("LLM assertion call failed: {e}"),
1698            },
1699            |(lr, _idx)| {
1700                let content_lower = lr.content.to_lowercase().trim().to_owned();
1701                if content_lower.starts_with("pass") {
1702                    StepResult {
1703                        name: "[assert] custom".into(),
1704                        status: StepStatus::Passed,
1705                        message: "PASS".into(),
1706                    }
1707                } else {
1708                    StepResult {
1709                        name: "[assert] custom".into(),
1710                        status: StepStatus::Failed,
1711                        message: lr.content,
1712                    }
1713                }
1714            },
1715        )
1716    }
1717
1718    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1719        let path = path.unwrap_or("screenshot.png");
1720
1721        match tab.capture_screenshot(
1722            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1723            None,
1724            None,
1725            true,
1726        ) {
1727            Ok(data) => {
1728                if let Err(e) = std::fs::write(path, &data) {
1729                    return StepResult {
1730                        name: format!("[screenshot] {path}"),
1731                        status: StepStatus::Failed,
1732                        message: format!("failed to write screenshot: {e}"),
1733                    };
1734                }
1735                StepResult {
1736                    name: format!("[screenshot] {path}"),
1737                    status: StepStatus::Passed,
1738                    message: format!("saved to {path}"),
1739                }
1740            }
1741            Err(e) => StepResult {
1742                name: format!("[screenshot] {path}"),
1743                status: StepStatus::Failed,
1744                message: format!("screenshot failed: {e}"),
1745            },
1746        }
1747    }
1748
1749    /// Runs an A2A agent step.
1750    #[allow(clippy::literal_string_with_formatting_args)]
1751    fn run_agent(
1752        &self,
1753        agent_name: &str,
1754        task: &str,
1755        definition: Option<&str>,
1756        _test_endpoint: Option<&str>,
1757    ) -> StepResult {
1758        // If a definition is specified, look up the task template
1759        let resolved_task = if let Some(def_name) = definition {
1760            if let Some(def) = self.definitions.get(def_name) {
1761                let tmpl = def.task_template.as_deref().unwrap_or(task);
1762                tmpl.replace("{task}", task)
1763            } else {
1764                return StepResult {
1765                    name: format!("[agent] {def_name}"),
1766                    status: StepStatus::Failed,
1767                    message: format!("definition '{def_name}' not found"),
1768                };
1769            }
1770        } else {
1771            task.to_owned()
1772        };
1773
1774        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1775    }
1776
1777    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1778        let Some(ep) = self.endpoints.get(agent_name) else {
1779            return StepResult {
1780                name: format!("[agent] {display_name}"),
1781                status: StepStatus::Failed,
1782                message: format!("agent endpoint '{agent_name}' not found"),
1783            };
1784        };
1785
1786        if ep.url.is_empty() {
1787            return StepResult {
1788                name: format!("[agent] {display_name}"),
1789                status: StepStatus::Failed,
1790                message: format!("agent endpoint '{agent_name}' has no URL"),
1791            };
1792        }
1793
1794        self.reporter.debug(format!("agent {agent_name}: {task}"));
1795
1796        let url = ep.url.clone();
1797        let client = A2aClient::new(&url, self.timeout);
1798        let task_clone = task.to_owned();
1799
1800        let response = std::thread::spawn(move || {
1801            let rt = tokio::runtime::Builder::new_current_thread()
1802                .enable_all()
1803                .build()
1804                .unwrap();
1805            rt.block_on(client.send_task(&task_clone))
1806        })
1807        .join()
1808        .unwrap();
1809
1810        // Record the flat-cost call
1811        self.usage.record_flat_call(agent_name, ep);
1812
1813        match response {
1814            Ok(text) => {
1815                let clean = text.trim().to_owned();
1816                let lower = clean.to_lowercase();
1817                if lower.starts_with("pass") {
1818                    StepResult {
1819                        name: format!("[agent] {display_name}"),
1820                        status: StepStatus::Passed,
1821                        message: format!("PASS: {clean}"),
1822                    }
1823                } else if lower.starts_with("fail") {
1824                    StepResult {
1825                        name: format!("[agent] {display_name}"),
1826                        status: StepStatus::Failed,
1827                        message: clean,
1828                    }
1829                } else {
1830                    StepResult {
1831                        name: format!("[agent] {display_name}"),
1832                        status: StepStatus::Passed,
1833                        message: format!("response: {clean}"),
1834                    }
1835                }
1836            }
1837            Err(e) => StepResult {
1838                name: format!("[agent] {display_name}"),
1839                status: StepStatus::Failed,
1840                message: format!("agent call failed: {e}"),
1841            },
1842        }
1843    }
1844
1845    /// Runs an MCP tool call step.
1846    fn run_mcp(
1847        &self,
1848        server_name: &str,
1849        tool_name: &str,
1850        args: Option<&serde_json::Value>,
1851    ) -> StepResult {
1852        let Some(ep) = self.endpoints.get(server_name) else {
1853            return StepResult {
1854                name: format!("[mcp] {server_name}:{tool_name}"),
1855                status: StepStatus::Failed,
1856                message: format!("MCP server endpoint '{server_name}' not found"),
1857            };
1858        };
1859
1860        let cmd = ep.command.as_deref().unwrap_or("");
1861        if cmd.is_empty() {
1862            return StepResult {
1863                name: format!("[mcp] {server_name}:{tool_name}"),
1864                status: StepStatus::Failed,
1865                message: format!("MCP server '{server_name}' has no command configured"),
1866            };
1867        }
1868
1869        self.reporter
1870            .debug(format!("mcp {server_name} {tool_name}"));
1871
1872        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1873
1874        let command = cmd.to_owned();
1875        let args_vec = ep.args.clone();
1876        let tool = tool_name.to_owned();
1877
1878        let response = std::thread::spawn(move || {
1879            let mut mcp_client =
1880                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1881            mcp_client
1882                .call_tool(&tool, &args_val)
1883                .map_err(|e| e.to_string())
1884        })
1885        .join()
1886        .unwrap();
1887
1888        // Record the flat-cost call
1889        self.usage.record_flat_call(server_name, ep);
1890
1891        match response {
1892            Ok(result) => {
1893                if result.isError {
1894                    StepResult {
1895                        name: format!("[mcp] {server_name}:{tool_name}"),
1896                        status: StepStatus::Failed,
1897                        message: result.to_string(),
1898                    }
1899                } else {
1900                    StepResult {
1901                        name: format!("[mcp] {server_name}:{tool_name}"),
1902                        status: StepStatus::Passed,
1903                        message: result.to_string(),
1904                    }
1905                }
1906            }
1907            Err(e) => StepResult {
1908                name: format!("[mcp] {server_name}:{tool_name}"),
1909                status: StepStatus::Failed,
1910                message: format!("MCP call failed: {e}"),
1911            },
1912        }
1913    }
1914
1915    // ── helpers ──────────────────────────────────────────────────────────
1916
1917    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1918    /// the runner's default LLM config for any unset fields.
1919    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1920        LlmConfig {
1921            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
1922                // Bedrock builds its endpoint from the resolved AWS region
1923                // when no URL is given — never inherit the default LLM URL.
1924                endpoint.url.clone()
1925            } else if endpoint.url.is_empty() {
1926                self.llm.url.clone()
1927            } else {
1928                endpoint.url.clone()
1929            },
1930            model: endpoint
1931                .model
1932                .clone()
1933                .unwrap_or_else(|| self.llm.model.clone()),
1934            api_key: endpoint
1935                .api_key
1936                .clone()
1937                .or_else(|| self.llm.api_key.clone()),
1938            headers: if endpoint.headers.is_empty() {
1939                self.llm.headers.clone()
1940            } else {
1941                endpoint.headers.clone()
1942            },
1943            timeout: self.llm.timeout,
1944            temperature: self.llm.temperature,
1945            thinking: self.llm.thinking,
1946            model_params: self.llm.model_params.clone(),
1947            max_attempts: endpoint.max_attempts.max(1),
1948            provider: endpoint.provider,
1949            deployment: endpoint.deployment.clone(),
1950            api_version: endpoint.api_version.clone(),
1951            auth: endpoint.auth.clone(),
1952            header_commands: endpoint.header_commands.clone(),
1953            aws: endpoint.aws.clone(),
1954        }
1955    }
1956
1957    /// Runs a single LLM call against an ordered endpoint chain (primary +
1958    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
1959    /// the first endpoint that answers wins. Returns the response together
1960    /// with the chain index of the answering endpoint (0 = primary) so the
1961    /// caller can attribute usage to the correct endpoint.
1962    ///
1963    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
1964    /// shows duration, tokens, cost and the answering endpoint per call.
1965    /// Run-level context handed to every LLM call as the FIRST block of the
1966    /// user message. Contents that are stable for the whole run ("run
1967    /// started", "target site") come first so upstream provider prefix
1968    /// caching stays effective; the current time is the last line because
1969    /// it changes on every call.
1970    fn run_context(&self) -> String {
1971        let mut parts = vec![
1972            "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
1973            "================================================================".into(),
1974            format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
1975        ];
1976        if let Some(base) = self.config.base_url.as_deref() {
1977            parts.push(format!("Target site: {base}"));
1978        }
1979        let now = SystemTime::now()
1980            .duration_since(UNIX_EPOCH)
1981            .map_or(0, |d| d.as_secs());
1982        parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
1983        parts.join("\n")
1984    }
1985
1986    #[allow(clippy::cast_possible_truncation, clippy::too_many_lines)]
1987    fn llm_call_chain(
1988        &self,
1989        chain: &[&ResolvedEndpoint],
1990        system: &str,
1991        user: &str,
1992        image: Option<&[String]>,
1993        purpose: &str,
1994    ) -> Result<(crate::costs::LlmResponse, usize), String> {
1995        if chain.is_empty() {
1996            return Err("empty LLM endpoint chain".into());
1997        }
1998        let primary = self.build_llm_for_endpoint(chain[0]);
1999        let fallbacks: Vec<LlmConfig> = chain[1..]
2000            .iter()
2001            .map(|e| self.build_llm_for_endpoint(e))
2002            .collect();
2003
2004        let (test, index) = self
2005            .current_step
2006            .borrow()
2007            .as_ref()
2008            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
2009        let primary_endpoint = chain[0].name.clone();
2010        let primary_model = primary.model.clone();
2011        self.emit_event(&TestEvent::LlmCallStarted {
2012            test: test.clone(),
2013            index,
2014            endpoint: primary_endpoint.clone(),
2015            model: primary_model.clone(),
2016            purpose: purpose.to_owned(),
2017        });
2018
2019        let started = Instant::now();
2020        let sys = system.to_owned();
2021        let context = self.run_context();
2022        let user = if context.is_empty() {
2023            user.to_owned()
2024        } else {
2025            format!("{context}\n\n{user}")
2026        };
2027        let image = image.map(<[String]>::to_vec);
2028
2029        let result = std::thread::spawn(move || {
2030            let rt = tokio::runtime::Builder::new_current_thread()
2031                .enable_all()
2032                .build()
2033                .unwrap();
2034            let call = async {
2035                match image.as_deref() {
2036                    Some(img) => {
2037                        llm_chat_vision_with_usage_chain(
2038                            &primary,
2039                            &fallbacks,
2040                            &sys,
2041                            &user,
2042                            Some(img),
2043                        )
2044                        .await
2045                    }
2046                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2047                }
2048            };
2049            rt.block_on(call)
2050        })
2051        .join()
2052        .unwrap();
2053
2054        let duration_ms = started.elapsed().as_millis() as u64;
2055        match result {
2056            Ok((lr, idx)) => {
2057                let cost = calculate_llm_cost(
2058                    chain[idx],
2059                    lr.usage.prompt_tokens,
2060                    lr.usage.completion_tokens,
2061                );
2062                let answering = chain[idx].name.clone();
2063                let model = chain[idx]
2064                    .model
2065                    .clone()
2066                    .unwrap_or_else(|| primary_model.clone());
2067                self.usage
2068                    .record_llm_call(&answering, chain[idx], &model, &lr.usage);
2069                self.emit_event(&TestEvent::LlmCallFinished {
2070                    test,
2071                    index,
2072                    endpoint: answering,
2073                    model,
2074                    purpose: purpose.to_owned(),
2075                    ok: true,
2076                    duration_ms,
2077                    input_tokens: lr.usage.prompt_tokens,
2078                    output_tokens: lr.usage.completion_tokens,
2079                    cached_input_tokens: lr.usage.cached_input_tokens,
2080                    cache_creation_input_tokens: lr.usage.cache_creation_input_tokens,
2081                    cost,
2082                    error: None,
2083                });
2084                Ok((lr, idx))
2085            }
2086            Err(e) => {
2087                self.emit_event(&TestEvent::LlmCallFinished {
2088                    test,
2089                    index,
2090                    endpoint: primary_endpoint,
2091                    model: primary_model,
2092                    purpose: purpose.to_owned(),
2093                    ok: false,
2094                    duration_ms,
2095                    input_tokens: 0,
2096                    output_tokens: 0,
2097                    cached_input_tokens: 0,
2098                    cache_creation_input_tokens: 0,
2099                    cost: 0.0,
2100                    error: Some(e.clone()),
2101                });
2102                Err(e)
2103            }
2104        }
2105    }
2106
2107    /// Resolves a CSS selector for the target element. Uses the explicit
2108    /// `selector` if provided, otherwise asks the LLM to find the element
2109    /// from the natural language `target` description and page DOM.
2110    ///
2111    /// LLM responses are sanitized and verified against the live page: a
2112    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2113    /// immediately with the raw LLM output, and a selector that matches
2114    /// nothing triggers one retry with feedback before failing.
2115    #[allow(clippy::too_many_lines)]
2116    fn resolve_selector(
2117        &self,
2118        css_override: Option<&str>,
2119        target: &str,
2120        step_endpoint: Option<&str>,
2121        test_endpoint: Option<&str>,
2122        tab: &Tab,
2123    ) -> Result<String, String> {
2124        if let Some(explicit) = css_override {
2125            return Ok(explicit.to_owned());
2126        }
2127
2128        let dom_info = extract_dom_info(tab)?;
2129        let page_content = get_page_text(tab);
2130
2131        let system = concat!(
2132            "You are a browser automation selector generator. ",
2133            "Given a web page's content and interactive elements, ",
2134            "return ONLY the best CSS selector for the described element. ",
2135            "Output nothing except the CSS selector. ",
2136            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2137            "[name=\"...\"], tag.class, tag. ",
2138            "Never output explanations, markdown, or extra text."
2139        );
2140
2141        let user = format!(
2142            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2143            page_content.url,
2144            page_content.title,
2145            truncate(&page_content.body_text, 4000),
2146            dom_info,
2147            target,
2148        );
2149
2150        let retry_user = format!(
2151            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2152            "The selector must match at least one element currently present on the page.",
2153            page_content.url,
2154            page_content.title,
2155            truncate(&page_content.body_text, 4000),
2156            dom_info,
2157            target,
2158        );
2159
2160        self.reporter.debug(format!("LLM targeting: {target}"));
2161
2162        let chain = self
2163            .endpoints
2164            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2165        let sys = system.to_owned();
2166
2167        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2168
2169        let first = call_llm(&user);
2170        let (lr, _idx) = match first {
2171            Ok(lr) => lr,
2172            Err(e) => {
2173                return Err(format!("LLM element targeting failed: {e}"));
2174            }
2175        };
2176        let clean = sanitize_selector(&lr.content);
2177        self.reporter.debug(format!("resolved selector: {clean}"));
2178
2179        if selector_is_useless(&clean) {
2180            return Err(format!(
2181                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2182                raw = lr.content.trim(),
2183            ));
2184        }
2185        if let Err(reason) = validate_selector(&clean) {
2186            return Err(format!(
2187                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2188                raw = lr.content.trim(),
2189            ));
2190        }
2191        if !selector_matches(tab, &clean).unwrap_or(false) {
2192            // One retry with feedback: flaky models occasionally invent a
2193            // selector that does not exist on the page.
2194            self.reporter.warn(format!(
2195                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2196            ));
2197            let second = call_llm(&retry_user);
2198            let (lr2, _idx2) = match second {
2199                Ok(lr2) => lr2,
2200                Err(e) => {
2201                    return Err(format!(
2202                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2203                    ));
2204                }
2205            };
2206            let clean2 = sanitize_selector(&lr2.content);
2207            self.reporter
2208                .debug(format!("resolved selector (retry): {clean2}"));
2209            if selector_is_useless(&clean2) {
2210                return Err(format!(
2211                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2212                    raw = lr2.content.trim(),
2213                    excerpt = truncate(&page_content.body_text, 300),
2214                ));
2215            }
2216            if !selector_matches(tab, &clean2).unwrap_or(false) {
2217                return Err(format!(
2218                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2219                ));
2220            }
2221            return Ok(clean2);
2222        }
2223
2224        Ok(clean)
2225    }
2226}
2227
2228/// Evaluates a JS expression that is expected to return a boolean.
2229fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2230    tab.evaluate(js, false)
2231        .map_err(|e| format!("evaluate failed: {e}"))?
2232        .value
2233        .and_then(|v| v.as_bool())
2234        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2235}
2236
2237/// Checks whether a CSS selector matches at least one current element.
2238fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2239    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2240}
2241
2242// ── Free helper functions ──────────────────────────────────────────────
2243
2244fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2245    let name = format!("[navigate] {full_url}");
2246    match tab.navigate_to(full_url) {
2247        Ok(_) => {
2248            let _ = tab.wait_until_navigated();
2249            StepResult {
2250                name,
2251                status: StepStatus::Passed,
2252                message: format!("navigated to {full_url}"),
2253            }
2254        }
2255        Err(e) => StepResult {
2256            name,
2257            status: StepStatus::Failed,
2258            message: format!("navigation failed: {e}"),
2259        },
2260    }
2261}
2262
2263fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2264    let result = tab
2265        .evaluate(DOM_EXTRACT_JS, false)
2266        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2267
2268    let json_str = result
2269        .value
2270        .as_ref()
2271        .and_then(|v| v.as_str())
2272        .unwrap_or("[]");
2273
2274    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2275
2276    if elements.is_empty() {
2277        return Ok("(no interactive elements found)".to_owned());
2278    }
2279
2280    Ok(elements.join("\n"))
2281}
2282
2283fn get_page_text(tab: &Tab) -> PageContent {
2284    let url = tab.get_url();
2285
2286    let title = tab
2287        .evaluate("document.title", false)
2288        .ok()
2289        .and_then(|r| r.value)
2290        .and_then(|v| v.as_str().map(String::from))
2291        .unwrap_or_else(|| "unknown".to_owned());
2292
2293    let body_text = tab
2294        .evaluate(
2295            "document.body ? document.body.innerText : document.documentElement.innerText",
2296            false,
2297        )
2298        .ok()
2299        .and_then(|r| r.value)
2300        .and_then(|v| v.as_str().map(String::from))
2301        .unwrap_or_default();
2302
2303    PageContent {
2304        url,
2305        title,
2306        body_text: truncate(&body_text, 8000),
2307    }
2308}
2309
2310fn resolve_url(url: &str, base_url: &str) -> String {
2311    if url.starts_with("http://") || url.starts_with("https://") {
2312        return url.to_owned();
2313    }
2314    let base = base_url.trim_end_matches('/');
2315    if url.starts_with('/') {
2316        format!("{base}{url}")
2317    } else {
2318        format!("{base}/{url}")
2319    }
2320}
2321
2322/// Human-readable label for a step, used when steps are skipped after an
2323/// earlier failure.
2324fn step_label(step: &TestStep) -> String {
2325    match step {
2326        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2327        TestStep::Click { target, .. } => format!("[click] {target}"),
2328        TestStep::Type { target, .. } => format!("[type] {target}"),
2329        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2330        TestStep::Assert {
2331            definition,
2332            preset,
2333            prompt,
2334            ..
2335        } => definition.as_ref().map_or_else(
2336            || {
2337                preset.as_ref().map_or_else(
2338                    || {
2339                        prompt.as_ref().map_or_else(
2340                            || "[assert]".to_owned(),
2341                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2342                        )
2343                    },
2344                    |p| format!("[assert] {p}"),
2345                )
2346            },
2347            |d| format!("[assert] {d}"),
2348        ),
2349        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2350        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2351        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2352    }
2353}
2354
2355/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2356#[must_use]
2357const fn step_kind_label(step: &TestStep) -> &'static str {
2358    match step {
2359        TestStep::Navigate { .. } => "navigate",
2360        TestStep::Click { .. } => "click",
2361        TestStep::Type { .. } => "type",
2362        TestStep::Wait { .. } => "wait",
2363        TestStep::Assert { .. } => "assert",
2364        TestStep::Screenshot { .. } => "screenshot",
2365        TestStep::Agent { .. } => "agent",
2366        TestStep::Mcp { .. } => "mcp",
2367    }
2368}
2369
2370// ── Support types ──────────────────────────────────────────────────────
2371
2372#[derive(Default)]
2373struct TestRunResult {
2374    passed: u32,
2375    failed: u32,
2376    skipped: u32,
2377    total: u32,
2378    details: Vec<StepResult>,
2379}
2380
2381struct PageContent {
2382    url: String,
2383    title: String,
2384    body_text: String,
2385}
2386
2387#[cfg(test)]
2388mod tests {
2389    use super::unix_to_rfc3339;
2390
2391    #[test]
2392    fn rfc3339_epoch_and_reference_dates() {
2393        assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2394        assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2395        assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2396        assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2397    }
2398
2399    #[test]
2400    fn rfc3339_handles_leap_years() {
2401        assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2402        assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2403    }
2404}