Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// Formats a UNIX epoch timestamp as an RFC 3339 UTC string
26/// (`2026-09-19T16:04:00Z`) without pulling in a date dependency.
27/// Civil-from-days conversion after Howard Hinnant.
28#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31    let secs = i64::try_from(secs).unwrap_or(0);
32    let days = secs.div_euclid(86_400);
33    let rem = secs.rem_euclid(86_400);
34    let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35    let z = days + 719_468;
36    let era = z.div_euclid(146_097);
37    let doe = z.rem_euclid(146_097);
38    let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40    let mp = (5 * doy + 2) / 153;
41    let d = doy - (153 * mp + 2) / 5 + 1;
42    let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43    let y = yoe + era * 400 + i64::from(mth <= 2);
44    format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47/// One detected layout defect (`layout_no_issues` preset).
48#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50    #[serde(rename = "type")]
51    issue_type: String,
52    element: String,
53    detail: String,
54}
55
56/// In-browser DOM layout scan for `layout_no_issues`.
57///
58/// Geometry-only checks (no LLM, no pixels):
59/// 1. page horizontal overflow (`scrollWidth` > viewport width);
60/// 2. elements outside the viewport that scrolling cannot reveal
61///    (fixed elements off-screen, left/negative overflow, right-edge
62///    overflow beyond the horizontally scrollable area, and bottom
63///    overflow on a page that cannot scroll down) — below-the-fold
64///    content on a tall scrollable page is normal flow, NOT a defect;
65/// 3. text clipped by `overflow: hidden` containers whose content
66///    is measurably larger than the box;
67/// 4. interactive elements (buttons/links/inputs) whose center point
68///    is covered by a different element that would intercept the click.
69///
70/// Elements whose class matches a configured ignore prefix (default:
71/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
72/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
73/// skipped — they are intentionally 1x1 / off-screen. The prefix list
74/// is injected at `__IGNORE_CLASSES__` from
75/// [`ScenarioConfig::layout_ignore_classes`].
76///
77/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
78/// sticky headers) is excluded by the position/relation filters.
79const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81  const issues = [];
82  const push = (type, el, detail) => {
83    if (issues.length >= 30) return;
84    let element = el.tagName.toLowerCase();
85    if (el.id) element += '#' + el.id;
86    else if (typeof el.className === 'string' && el.className.trim())
87      element += '.' + el.className.trim().split(/\s+/).join('.');
88    issues.push({ type, element, detail: String(detail).slice(0, 220) });
89  };
90  const vw = document.documentElement.clientWidth || window.innerWidth;
91  const vh = document.documentElement.clientHeight || window.innerHeight;
92  if (!vw || !vh) return JSON.stringify(issues);
93  const de = document.documentElement;
94  // 1. Page-level horizontal overflow.
95  if (maxSW > vw + 2)
96    push('page-overflow-x', de,
97      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98  // Class-name prefixes to skip (injected; default = Angular CDK
99  // screen-reader helpers, which are intentionally 1x1 / off-screen).
100  const ignorePrefixes = __IGNORE_CLASSES__;
101  const isIgnored = (el) => {
102    if (typeof el.className !== 'string' || !el.className.trim()) return false;
103    const classes = el.className.trim().split(/\s+/);
104    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105  };
106  const body = document.body;
107  // The document element is not always the scroll container: the app may
108  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110  // "not scrollable" — measure the scrollable content height/width of the
111  // document element AND the body, and take the max.
112  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114  const vScrollable = maxSH > vh + 2;
115  const hScrollable = maxSW > vw + 2;
116  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117  // this element into view on the given axis — i.e. the element is inside a
118  // scrolling region, so lying beyond the viewport cut is not a defect.
119  const reachableByScroller = (el, axis) => {
120    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123    let p = el.parentElement;
124    while (p && p !== body) {
125      const pcs = getComputedStyle(p);
126      const o = pcs[ovProp];
127      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128          p[dim] > p[dimClient] + 2) return true;
129      p = p.parentElement;
130    }
131    return false;
132  };
133  const selfOverflowing = (el, cs, axis) => {
134    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138      el[dim] > el[dimClient] + 2;
139  };
140  // True when an ancestor clips this axis with overflow:hidden/clip — the
141  // element's overhang is not visible, so treat it as reachable.
142  const clippedByAncestor = (el, axis) => {
143    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144    let p = el.parentElement;
145    while (p && p !== body) {
146      const o = getComputedStyle(p)[ovProp];
147      if (o === 'hidden' || o === 'clip') return true;
148      p = p.parentElement;
149    }
150    return false;
151  };
152  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154  const hasContent = (el) =>
155    ((el.textContent || '').trim().length > 0) ||
156    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157  // 2. Elements outside the viewport that scrolling cannot reveal.
158  for (const el of all) {
159    const cs = getComputedStyle(el);
160    if (!visible(cs)) continue;
161    if (isIgnored(el)) continue;
162    const r = el.getBoundingClientRect();
163    if (r.width < 2 || r.height < 2) continue;
164    if (!hasContent(el) && el.children.length === 0) continue;
165    if (cs.position === 'fixed') {
166      // Fixed elements never move with the scroll: any edge outside the
167      // viewport is unreachable content and therefore a defect.
168      const overTop = -r.top;
169      const overLeft = -r.left;
170      const overRight = r.right - vw;
171      const overBottom = r.bottom - vh;
172      // Top/left overshoot is never reachable; bottom/right overshoot is
173      // only a defect when the fixed element cannot scroll that content
174      // into view itself (e.g. an opened Material drawer whose inner
175      // container scrolls is not a "cut-off" defect).
176      if (overTop > 2 || overLeft > 2 ||
177          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179        let where = '';
180        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186        push('element-out-of-viewport', el,
187          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188      }
189      continue;
190    }
191    if (cs.position === 'sticky') continue;
192    // Negative left overflow cannot be reached by scrolling (scrollLeft
193    // never goes below 0).
194    if (r.left < -2) {
195      push('element-out-of-viewport', el,
196        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197      continue;
198    }
199    // Negative top with the page at the top means the element sits above
200    // the document origin — also unreachable.
201    if (r.top < -2 && de.scrollTop <= 2) {
202      push('element-out-of-viewport', el,
203        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204      continue;
205    }
206    const overRight = r.right - vw;
207    // Right-edge overflow is only a defect when the page cannot scroll
208    // horizontally to reveal it (or the element sticks out past the
209    // scrollable content width itself).
210    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211      push('element-out-of-viewport', el,
212        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213      continue;
214    }
215    // Below-the-fold content on a scrollable page is normal (tall landing
216    // pages); only flag bottom overflow the user can never scroll to.
217    const overBottom = r.bottom - vh;
218    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219      push('element-out-of-viewport', el,
220        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221    }
222  }
223  // 3. Text clipped by overflow:hidden containers.
224  for (const el of all) {
225    if (isIgnored(el)) continue;
226    const cs = getComputedStyle(el);
227    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229    if (!(el.textContent || '').trim()) continue;
230    // Single-line truncation with an ellipsis is an intentional design
231    // pattern (Tailwind .truncate etc.), not a clipping defect.
232    if (cs.textOverflow === 'ellipsis') continue;
233    push('text-clipped', el,
234      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236  }
237  // 4. Interactive elements covered by a different element.
238  const interactive =
239    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240  const targets = document.querySelectorAll(interactive);
241  for (const el of targets) {
242    if (isIgnored(el)) continue;
243    const r = el.getBoundingClientRect();
244    if (r.width < 6 || r.height < 6) continue;
245    const cs = getComputedStyle(el);
246    if (!visible(cs)) continue;
247    const cx = r.left + r.width / 2;
248    const cy = r.top + r.height / 2;
249    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250    const top = document.elementFromPoint(cx, cy);
251    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252    if (isIgnored(top)) continue;
253    const tcs = getComputedStyle(top);
254    if (!visible(tcs)) continue;
255    if (tcs.pointerEvents === 'none') continue;
256    const tr = top.getBoundingClientRect();
257    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260    push('element-overlap', el,
261      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262      ' is covered by <' + tname + '>');
263  }
264  return JSON.stringify(issues);
265})()
266"#;
267
268/// How long the CDP connection stays open after the browser goes quiet.
269///
270/// `headless_chrome` ships a 30s default and tears down the entire connection
271/// when no traffic arrives for that long; a run must own its connection for
272/// its full duration instead.
273const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275/// Executes a [`Scenario`] against a real browser with optional LLM
276/// assistance for element targeting and assertions.
277pub struct ScenarioRunner {
278    config: ScenarioConfig,
279    definitions: HashMap<String, AssertDefinition>,
280    llm: LlmConfig,
281    timeout: Duration,
282    viewport_width: u32,
283    viewport_height: u32,
284    /// The viewport currently applied in the browser (CDP emulation).
285    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
286    applied_viewport: std::cell::Cell<(u32, u32)>,
287    endpoints: EndpointRegistry,
288    usage: Arc<UsageTracker>,
289    budgets: BudgetTracker,
290    /// Directory for failure artifacts (screenshots).
291    artifacts_dir: PathBuf,
292    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
293    reporter: Arc<Reporter>,
294    /// The test + step index currently executing, for LLM-call events.
295    current_step: std::cell::RefCell<Option<(String, u32)>>,
296    /// Epoch seconds of runner creation — constant for the whole run so
297    /// the LLM context preamble below is cacheable by upstream providers.
298    run_started: u64,
299    /// Whether this runner emits the `RunStarted`/`RunFinished` events.
300    /// When several scenario files are run concurrently by
301    /// [`crate::parallel::run_scenarios`], the orchestrator owns those two
302    /// events (one per batch), so each per-file runner suppresses its own.
303    emit_run_events: bool,
304}
305
306/// Aggregated results from a scenario run.
307#[derive(Debug, Default)]
308pub struct RunReport {
309    /// Number of tests that passed.
310    pub tests_passed: u32,
311    /// Number of tests that failed.
312    pub tests_failed: u32,
313    /// Number of steps that passed.
314    pub passed: u32,
315    /// Number of steps that failed.
316    pub failed: u32,
317    /// Number of steps that were skipped.
318    pub skipped: u32,
319    /// Per-step details.
320    pub details: Vec<StepResult>,
321}
322
323/// Result of a single step execution.
324#[derive(Debug, Clone)]
325pub struct StepResult {
326    /// The step name.
327    pub name: String,
328    /// Whether the step passed, failed, or was skipped.
329    pub status: StepStatus,
330    /// Human-readable result message.
331    pub message: String,
332}
333
334/// Outcome for a single step.
335pub use crate::events::StepStatus;
336
337/// Predefined assertion preset definition.
338struct AssertPreset {
339    name: &'static str,
340    system: &'static str,
341    user_template: &'static str,
342}
343
344/// Built-in assertion presets.
345#[allow(clippy::literal_string_with_formatting_args)]
346const ASSERTION_PRESETS: &[AssertPreset] = &[
347    AssertPreset {
348        name: "no_error_on_page",
349        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
350        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
351    },
352    AssertPreset {
353        name: "text_visible",
354        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
355        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
356    },
357    AssertPreset {
358        name: "element_exists",
359        system: "You are a QA tester. Check if a described UI element exists on a web page.",
360        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
361    },
362    AssertPreset {
363        name: "visual_no_issues",
364        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
365        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
366    },
367    AssertPreset {
368        name: "visual_no_overlaps",
369        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
370        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
371    },
372    AssertPreset {
373        name: "visual_text_visible",
374        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
375        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
376    },
377    AssertPreset {
378        name: "layout_no_issues",
379        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
380        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
381    },
382];
383
384impl ScenarioRunner {
385    /// Creates a new runner with the given scenario configuration and
386    /// assertion definitions.
387    #[must_use]
388    #[allow(clippy::needless_pass_by_value)]
389    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
390        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
391    }
392
393    /// Creates a runner that reports run events through the given reporter.
394    #[must_use]
395    #[allow(clippy::needless_pass_by_value)]
396    pub fn with_reporter(
397        scenario_config: ScenarioConfig,
398        definitions: Vec<AssertDefinition>,
399        reporter: Arc<Reporter>,
400    ) -> Self {
401        Self::with_reporter_mode(scenario_config, definitions, reporter, true)
402    }
403
404    /// Creates a runner that reports run events through the given reporter
405    /// but suppresses the `RunStarted`/`RunFinished` events. Used by the
406    /// parallel orchestrator, which emits those once per batch.
407    #[must_use]
408    #[allow(clippy::needless_pass_by_value)]
409    pub fn with_reporter_parallel(
410        scenario_config: ScenarioConfig,
411        definitions: Vec<AssertDefinition>,
412        reporter: Arc<Reporter>,
413    ) -> Self {
414        Self::with_reporter_mode(scenario_config, definitions, reporter, false)
415    }
416
417    #[must_use]
418    #[allow(clippy::needless_pass_by_value)]
419    fn with_reporter_mode(
420        scenario_config: ScenarioConfig,
421        definitions: Vec<AssertDefinition>,
422        reporter: Arc<Reporter>,
423        emit_run_events: bool,
424    ) -> Self {
425        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
426            &scenario_config,
427        ));
428        let llm = LlmConfig {
429            url: scenario_config
430                .llm_url
431                .clone()
432                .unwrap_or_else(crate::llm_base_url),
433            model: scenario_config
434                .llm_model
435                .clone()
436                .unwrap_or_else(crate::llm_model),
437            api_key: scenario_config
438                .llm_api_key
439                .clone()
440                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
441            headers: if scenario_config.llm_headers.is_empty() {
442                crate::parse_headers_env()
443            } else {
444                scenario_config.llm_headers.clone()
445            },
446            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
447            temperature: scenario_config.temperature,
448            thinking: scenario_config.thinking,
449            model_params: scenario_config.model_params.clone(),
450            cache: scenario_config.cache.unwrap_or(true),
451            max_attempts: crate::default_llm_attempts(),
452            provider: crate::scenario::Provider::Openai,
453            deployment: None,
454            api_version: None,
455            auth: crate::scenario::AuthConfig::default(),
456            header_commands: std::collections::HashMap::new(),
457            aws: crate::scenario::AwsConfig::default(),
458        };
459        let endpoints = EndpointRegistry::from_config(
460            &scenario_config.endpoints,
461            Some(&llm),
462            crate::endpoints::EndpointDefaults {
463                cache: scenario_config.cache,
464                cache_pricing: scenario_config.cache_pricing,
465            },
466        );
467        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
468        let defs_map: HashMap<String, AssertDefinition> = definitions
469            .into_iter()
470            .map(|d| (d.name.clone(), d))
471            .collect();
472
473        Self {
474            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
475            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
476            viewport_height: scenario_config.viewport_height.unwrap_or(720),
477            applied_viewport: std::cell::Cell::new((0, 0)),
478            config: scenario_config.clone(),
479            definitions: defs_map,
480            llm,
481            endpoints,
482            usage: Arc::new(UsageTracker::new()),
483            budgets,
484            artifacts_dir: PathBuf::from(
485                scenario_config
486                    .artifacts_dir
487                    .unwrap_or_else(|| "artifacts".to_owned()),
488            ),
489            reporter,
490            current_step: std::cell::RefCell::new(None),
491            run_started: SystemTime::now()
492                .duration_since(UNIX_EPOCH)
493                .map_or(0, |d| d.as_secs()),
494            emit_run_events,
495        }
496    }
497
498    /// Emits an event; a sink failure degrades to a console warning so a
499    /// broken log file can never mask the run itself.
500    fn emit_event(&self, event: &TestEvent) {
501        if let Err(err) = self.reporter.emit(event) {
502            use std::io::Write as _;
503            let mut out = std::io::stderr().lock();
504            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
505        }
506    }
507
508    /// Returns a clone of the [`UsageTracker`] for reporting.
509    #[must_use]
510    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
511        Arc::clone(&self.usage)
512    }
513
514    /// Returns a reference to the [`BudgetTracker`].
515    #[must_use]
516    pub const fn budget_tracker(&self) -> &BudgetTracker {
517        &self.budgets
518    }
519
520    /// Executes all test groups in the scenario and returns a report.
521    ///
522    /// # Errors
523    ///
524    /// Returns an error if the browser fails to launch.
525    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
526    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
527        let mut report = RunReport::default();
528
529        if tests.is_empty() {
530            self.reporter.warn("No tests defined in scenario.");
531            if self.emit_run_events {
532                self.emit_event(&TestEvent::RunFinished {
533                    tests_passed: 0,
534                    tests_failed: 0,
535                    steps_passed: 0,
536                    steps_failed: 0,
537                    steps_skipped: 0,
538                    total_cost: 0.0,
539                    total_tokens: 0,
540                    total_input_tokens: 0,
541                    total_output_tokens: 0,
542                    total_cached_input_tokens: 0,
543                    total_cache_creation_input_tokens: 0,
544                    models: Vec::new(),
545                    total_calls: 0,
546                });
547            }
548            return Ok(report);
549        }
550
551        if self.emit_run_events {
552            self.emit_event(&TestEvent::RunStarted {
553                total_tests: tests.len() as u32,
554            });
555        }
556
557        let browser_headless = self.config.browser_headless.unwrap_or(true);
558
559        let launch_opts = LaunchOptions {
560            headless: browser_headless,
561            window_size: Some((self.viewport_width, self.viewport_height)),
562            sandbox: false,
563            // headless_chrome defaults this to 30s and shuts down the whole CDP
564            // connection when no messages arrive for that long. A scenario can
565            // easily exceed 30s of browser silence (slow LLM targeting/assertion
566            // calls, page waits, budget checks between steps), after which every
567            // remaining step fails with "Unable to make method calls because
568            // underlying connection is closed" — one quiet gap kills the run.
569            // Open-ended scenarios must own the connection for their full
570            // duration, so keep it alive for 6 hours.
571            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
572            ..LaunchOptions::default()
573        };
574
575        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
576        let tab = browser.new_tab().context("failed to open browser tab")?;
577        let _ = tab.set_default_timeout(self.timeout);
578
579        // Start MCP server if configured
580        #[cfg(feature = "mcp-server")]
581        if let Some(ref mcp_cfg) = self.config.mcp_server {
582            if mcp_cfg.enabled {
583                let port = mcp_cfg.port;
584                std::thread::spawn(move || {
585                    let _ = crate::mcp_server::start_mcp_server(port);
586                });
587            }
588        }
589        #[cfg(not(feature = "mcp-server"))]
590        if let Some(mcp_cfg) = &self.config.mcp_server {
591            if mcp_cfg.enabled {
592                self.reporter
593                    .warn("MCP server configured but 'mcp-server' feature not enabled");
594            }
595        }
596
597        // Start A2A agent server if configured
598        #[cfg(feature = "a2a-server")]
599        if let Some(ref a2a_cfg) = self.config.a2a_server {
600            if a2a_cfg.enabled {
601                let port = a2a_cfg.port;
602                tokio::spawn(crate::a2a_server::start_a2a_server(port));
603            }
604        }
605        #[cfg(not(feature = "a2a-server"))]
606        if let Some(a2a_cfg) = &self.config.a2a_server {
607            if a2a_cfg.enabled {
608                self.reporter
609                    .warn("A2A server configured but 'a2a-server' feature not enabled");
610            }
611        }
612
613        for test in tests {
614            self.emit_event(&TestEvent::TestStarted {
615                test: test.name.clone(),
616            });
617
618            self.usage.reset_per_test();
619
620            let test_started = Instant::now();
621            let test_result = self.run_test(test, &tab);
622            let duration_ms = test_started.elapsed().as_millis() as u64;
623            let usage = self.usage.current_test_snapshot();
624            self.usage.commit_test(&test.name);
625
626            self.emit_event(&TestEvent::TestFinished {
627                test: test.name.clone(),
628                passed: test_result.passed,
629                failed: test_result.failed,
630                skipped: test_result.skipped,
631                duration_ms,
632                cost: usage.total_cost,
633                tokens: usage.total_tokens,
634                input_tokens: usage.total_input_tokens,
635                output_tokens: usage.total_output_tokens,
636                cached_input_tokens: usage.total_cached_input_tokens,
637                cache_creation_input_tokens: usage.total_cache_creation_input_tokens,
638                models: usage.models.clone(),
639                calls: usage.total_calls,
640            });
641
642            if test_result.failed == 0 && test_result.total > 0 {
643                report.tests_passed += 1;
644            } else if test_result.total > 0 {
645                report.tests_failed += 1;
646            }
647
648            report.passed += test_result.passed;
649            report.failed += test_result.failed;
650            report.skipped += test_result.skipped;
651            report.details.extend(test_result.details);
652        }
653
654        let global = self.usage.global_snapshot();
655        if self.emit_run_events {
656            self.emit_event(&TestEvent::RunFinished {
657                tests_passed: report.tests_passed,
658                tests_failed: report.tests_failed,
659                steps_passed: report.passed,
660                steps_failed: report.failed,
661                steps_skipped: report.skipped,
662                total_cost: global.total_cost,
663                total_tokens: global.total_tokens,
664                total_input_tokens: global.total_input_tokens,
665                total_output_tokens: global.total_output_tokens,
666                total_cached_input_tokens: global.total_cached_input_tokens,
667                total_cache_creation_input_tokens: global.total_cache_creation_input_tokens,
668                models: global.models.clone(),
669                total_calls: global.total_calls,
670            });
671        }
672
673        Ok(report)
674    }
675
676    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
677    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
678        let base_url = test
679            .base_url
680            .clone()
681            .or_else(|| self.config.base_url.clone())
682            .unwrap_or_else(crate::base_url);
683
684        // Per-test viewport override: switch the browser via CDP
685        // device-metrics emulation before this test runs.
686        let vw = test.viewport_width.unwrap_or(self.viewport_width);
687        let vh = test.viewport_height.unwrap_or(self.viewport_height);
688        if self.applied_viewport.get() != (vw, vh) {
689            self.apply_viewport(tab, vw, vh);
690            self.applied_viewport.set((vw, vh));
691        }
692
693        // Per-test isolation: every test starts from its own start_url
694        // (unless auto_navigate is disabled), so a test never inherits the
695        // previous test's page state.
696        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
697
698        let start_url = test
699            .start_url
700            .clone()
701            .or_else(|| self.config.start_url.clone())
702            .unwrap_or_else(|| "/dashboard".to_owned());
703
704        if auto_navigate {
705            let full_url = resolve_url(&start_url, &base_url);
706            self.reporter.debug(format!("auto-navigate: {full_url}"));
707            let _ = tab.navigate_to(&full_url);
708            let _ = tab.wait_until_navigated();
709            std::thread::sleep(Duration::from_secs(4));
710        }
711
712        let mut result = TestRunResult::default();
713
714        for (step_index, step) in test.steps.iter().enumerate() {
715            result.total += 1;
716
717            let wait_ms = match step {
718                TestStep::Navigate { wait_after_ms, .. }
719                | TestStep::Click { wait_after_ms, .. }
720                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
721                _ => None,
722            };
723
724            self.current_step
725                .replace(Some((test.name.clone(), step_index as u32)));
726            self.emit_event(&TestEvent::StepStarted {
727                test: test.name.clone(),
728                index: step_index as u32,
729                label: step_label(step),
730            });
731            let step_started = Instant::now();
732
733            let mut step_result = match step {
734                TestStep::Navigate { url, .. } => {
735                    let full_url = resolve_url(url, &base_url);
736                    run_navigate_step(&full_url, tab)
737                }
738                TestStep::Click {
739                    target,
740                    selector,
741                    endpoint,
742                    idempotent,
743                    ..
744                } => self.run_click(
745                    target,
746                    selector.as_deref(),
747                    endpoint.as_deref(),
748                    test.endpoint.as_deref(),
749                    *idempotent,
750                    tab,
751                ),
752                TestStep::Type {
753                    target,
754                    text,
755                    selector,
756                    endpoint,
757                    idempotent,
758                    ..
759                } => self.run_type(
760                    target,
761                    text,
762                    selector.as_deref(),
763                    endpoint.as_deref(),
764                    test.endpoint.as_deref(),
765                    *idempotent,
766                    tab,
767                ),
768                TestStep::Wait {
769                    target,
770                    selector,
771                    text,
772                    timeout_ms,
773                    endpoint,
774                    idempotent,
775                } => self.run_wait(
776                    target,
777                    selector.as_deref(),
778                    text.as_deref(),
779                    *timeout_ms,
780                    endpoint.as_deref(),
781                    test.endpoint.as_deref(),
782                    *idempotent,
783                    tab,
784                ),
785                TestStep::Assert {
786                    definition,
787                    preset,
788                    prompt,
789                    assert_text,
790                    endpoint,
791                    screenshot,
792                } => self.run_assert(
793                    definition.as_deref(),
794                    preset.as_deref(),
795                    prompt.as_deref(),
796                    assert_text.as_deref(),
797                    *screenshot,
798                    endpoint.as_deref(),
799                    test.endpoint.as_deref(),
800                    tab,
801                ),
802                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
803                TestStep::Agent {
804                    agent,
805                    task,
806                    definition,
807                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
808                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
809            };
810
811            // Failure diagnostics: capture the page state and a screenshot so
812            // CI logs say WHAT the page looked like when the step failed,
813            // instead of a bare "timed out: The event waited for never came".
814            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
815                let state = diagnostics::capture(tab);
816                let screenshot = diagnostics::save_screenshot(
817                    tab,
818                    &self.artifacts_dir,
819                    &test.name,
820                    &test.name,
821                    step_index,
822                    step_kind_label(step),
823                );
824                step_result.message = format!(
825                    "{base} — {excerpt}",
826                    base = step_result.message,
827                    excerpt = diagnostics::inline_excerpt(&state),
828                );
829                (Some(diagnostics::full_context(&state)), screenshot)
830            } else {
831                (None, None)
832            };
833
834            let duration_ms = step_started.elapsed().as_millis() as u64;
835            self.emit_event(&TestEvent::StepFinished {
836                test: test.name.clone(),
837                index: step_index as u32,
838                label: step_result.name.clone(),
839                status: step_result.status,
840                duration_ms,
841                message: step_result.message.clone(),
842                diagnostics: diagnostics_block,
843                screenshot: screenshot_path,
844            });
845            self.current_step.replace(None);
846
847            match step_result.status {
848                StepStatus::Passed => result.passed += 1,
849                StepStatus::Failed => result.failed += 1,
850                StepStatus::Skipped => result.skipped += 1,
851            }
852
853            // Fail fast: the first failed step ends the test and the
854            // remaining steps are reported as skipped (no LLM budget is
855            // burned asserting against a page that is already known broken).
856            if step_result.status == StepStatus::Failed
857                && !self.config.continue_on_failure
858                && step_index + 1 < test.steps.len()
859            {
860                self.emit_event(&TestEvent::Warning {
861                    message: format!(
862                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
863                        test.steps.len() - step_index - 1
864                    ),
865                });
866                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
867                    let skipped_index = step_index + 1 + offset;
868                    let label = step_label(skipped);
869                    result.total += 1;
870                    result.skipped += 1;
871                    self.emit_event(&TestEvent::StepStarted {
872                        test: test.name.clone(),
873                        index: skipped_index as u32,
874                        label: label.clone(),
875                    });
876                    self.emit_event(&TestEvent::StepFinished {
877                        test: test.name.clone(),
878                        index: skipped_index as u32,
879                        label,
880                        status: StepStatus::Skipped,
881                        duration_ms: 0,
882                        message: "skipped: previous step failed".into(),
883                        diagnostics: None,
884                        screenshot: None,
885                    });
886                    result.details.push(StepResult {
887                        name: step_label(skipped),
888                        status: StepStatus::Skipped,
889                        message: "skipped: previous step failed".into(),
890                    });
891                }
892                result.details.push(step_result);
893                return result;
894            }
895
896            // Check per-test budget after each step
897            let test_usage = self.usage.current_test_snapshot();
898            let global_usage = self.usage.global_snapshot();
899            let budget_status = self.budgets.check_all(
900                &test.name,
901                &test_usage,
902                &global_usage,
903                test.budget.as_ref(),
904            );
905            match budget_status {
906                BudgetStatus::HardExceeded { message, .. } => {
907                    self.emit_event(&TestEvent::Warning {
908                        message: format!("budget exceeded: {message}"),
909                    });
910                    result.details.push(StepResult {
911                        name: "[budget]".into(),
912                        status: StepStatus::Failed,
913                        message,
914                    });
915                    result.failed += 1;
916                    return result;
917                }
918                BudgetStatus::SoftExceeded { message, .. } => {
919                    self.emit_event(&TestEvent::Warning {
920                        message: format!("budget warning: {message}"),
921                    });
922                }
923                BudgetStatus::Ok => {}
924            }
925
926            if let Some(ms) = wait_ms {
927                std::thread::sleep(Duration::from_millis(ms));
928            }
929
930            result.details.push(step_result);
931        }
932
933        result
934    }
935
936    /// Applies a viewport size to the current tab via CDP
937    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
938    /// overrides and the viewport matrix. The initial window size set at
939    /// browser launch is replaced by emulation; failures are logged but
940    /// do not fail the test (a mismatched viewport only weakens coverage).
941    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
942        use headless_chrome::protocol::cdp::Emulation;
943        let _ = self;
944        let params = Emulation::SetDeviceMetricsOverride {
945            width,
946            height,
947            device_scale_factor: 1.0,
948            mobile: false,
949            scale: None,
950            screen_width: Some(width),
951            screen_height: Some(height),
952            position_x: None,
953            position_y: None,
954            dont_set_visible_size: None,
955            screen_orientation: None,
956            viewport: None,
957            display_feature: None,
958            device_posture: None,
959        };
960        self.reporter.debug(format!("viewport: {width}x{height}"));
961        if let Err(e) = tab.call_method(params) {
962            self.reporter
963                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
964        }
965    }
966
967    /// Height (px) covered by assert-step screenshots: the configured
968    /// `screenshot_max_height` (absolute px or viewport multiple, default
969    /// `"20x"`) resolved against the currently applied viewport, raised to
970    /// at least the viewport height so the visible screen is always fully
971    /// included. The capture is split into viewport-tall tiles, so this
972    /// value bounds total coverage (and hence the number of image parts).
973    #[must_use]
974    fn screenshot_height_cap(&self) -> u32 {
975        let viewport_height = self.current_viewport_height();
976        let cap = self
977            .config
978            .screenshot_max_height
979            .as_ref()
980            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
981        cap.max(viewport_height)
982    }
983
984    /// Height of the viewport currently emulated in the browser (falling
985    /// back to the configured default before any emulation was applied).
986    #[must_use]
987    const fn current_viewport_height(&self) -> u32 {
988        let (_, height) = self.applied_viewport.get();
989        if height > 0 {
990            height
991        } else {
992            self.viewport_height
993        }
994    }
995
996    // ── step handlers ───────────────────────────────────────────────────
997
998    #[allow(clippy::too_many_lines)]
999    fn run_click(
1000        &self,
1001        target: &str,
1002        selector_override: Option<&str>,
1003        step_endpoint: Option<&str>,
1004        test_endpoint: Option<&str>,
1005        idempotent: bool,
1006        tab: &Tab,
1007    ) -> StepResult {
1008        let name = format!("[click] {target}");
1009        let selector = match self.resolve_selector(
1010            selector_override,
1011            target,
1012            step_endpoint,
1013            test_endpoint,
1014            tab,
1015        ) {
1016            Ok(s) => s,
1017            Err(msg) => {
1018                if idempotent {
1019                    return StepResult {
1020                        name,
1021                        status: StepStatus::Skipped,
1022                        message: format!("skipped (idempotent): no target found — {msg}"),
1023                    };
1024                }
1025                return StepResult {
1026                    name,
1027                    status: StepStatus::Failed,
1028                    message: msg,
1029                };
1030            }
1031        };
1032
1033        // Idempotent steps probe briefly: a missing target means the
1034        // action was already done / not applicable (e.g. an
1035        // already-authenticated session), and skipping is the success
1036        // path, not a failure.
1037        let probe_secs = if idempotent { 5 } else { 10 };
1038        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1039            Ok(element) => match element.click() {
1040                Ok(_) => StepResult {
1041                    name,
1042                    status: StepStatus::Passed,
1043                    message: format!("clicked {selector}"),
1044                },
1045                Err(e) => StepResult {
1046                    name,
1047                    status: StepStatus::Failed,
1048                    message: format!("click failed on {selector}: {e}"),
1049                },
1050            },
1051            Err(e) if idempotent => StepResult {
1052                name,
1053                status: StepStatus::Skipped,
1054                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1055            },
1056            Err(e) => StepResult {
1057                name,
1058                status: StepStatus::Failed,
1059                message: format!("element {selector} not found: {e}"),
1060            },
1061        }
1062    }
1063
1064    #[allow(clippy::too_many_arguments)]
1065    fn run_type(
1066        &self,
1067        target: &str,
1068        text: &str,
1069        selector_override: Option<&str>,
1070        step_endpoint: Option<&str>,
1071        test_endpoint: Option<&str>,
1072        idempotent: bool,
1073        tab: &Tab,
1074    ) -> StepResult {
1075        let name = format!("[type] {target}");
1076        let selector = match self.resolve_selector(
1077            selector_override,
1078            target,
1079            step_endpoint,
1080            test_endpoint,
1081            tab,
1082        ) {
1083            Ok(s) => s,
1084            Err(msg) => {
1085                if idempotent {
1086                    return StepResult {
1087                        name,
1088                        status: StepStatus::Skipped,
1089                        message: format!("skipped (idempotent): no target found — {msg}"),
1090                    };
1091                }
1092                return StepResult {
1093                    name,
1094                    status: StepStatus::Failed,
1095                    message: msg,
1096                };
1097            }
1098        };
1099
1100        let probe_secs = if idempotent { 5 } else { 10 };
1101        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1102            Ok(element) => {
1103                if let Err(e) = element.click() {
1104                    return StepResult {
1105                        name,
1106                        status: StepStatus::Failed,
1107                        message: format!("click to focus {selector} failed: {e}"),
1108                    };
1109                }
1110
1111                let js = format!(
1112                    "document.querySelector('{}').value = '';",
1113                    selector.replace('\'', "\\'")
1114                );
1115                let _ = tab.evaluate(&js, false);
1116
1117                match element.type_into(text) {
1118                    Ok(_) => StepResult {
1119                        name,
1120                        status: StepStatus::Passed,
1121                        message: format!("typed {text:?} into {selector}"),
1122                    },
1123                    Err(e) => StepResult {
1124                        name,
1125                        status: StepStatus::Failed,
1126                        message: format!("type into {selector} failed: {e}"),
1127                    },
1128                }
1129            }
1130            Err(e) if idempotent => StepResult {
1131                name,
1132                status: StepStatus::Skipped,
1133                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1134            },
1135            Err(e) => StepResult {
1136                name,
1137                status: StepStatus::Failed,
1138                message: format!("element {selector} not found: {e}"),
1139            },
1140        }
1141    }
1142
1143    #[allow(clippy::too_many_arguments)]
1144    #[allow(clippy::too_many_lines)]
1145    fn run_wait(
1146        &self,
1147        target: &str,
1148        selector_override: Option<&str>,
1149        text: Option<&str>,
1150        timeout_ms: Option<u64>,
1151        step_endpoint: Option<&str>,
1152        test_endpoint: Option<&str>,
1153        idempotent: bool,
1154        tab: &Tab,
1155    ) -> StepResult {
1156        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1157        let step_name = format!("[wait] {target}");
1158
1159        // Resolve an explicit selector only (text-only waits are LLM-free).
1160        let selector = match selector_override {
1161            Some(s) => Some(s.to_owned()),
1162            None if text.is_some() => None,
1163            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1164                Ok(s) => Some(s),
1165                Err(msg) => {
1166                    if idempotent {
1167                        return StepResult {
1168                            name: step_name,
1169                            status: StepStatus::Skipped,
1170                            message: format!("skipped (idempotent): no target found — {msg}"),
1171                        };
1172                    }
1173                    return StepResult {
1174                        name: step_name,
1175                        status: StepStatus::Failed,
1176                        message: msg,
1177                    };
1178                }
1179            },
1180        };
1181
1182        if text.is_some() {
1183            let sel_js = selector
1184                .as_deref()
1185                .map(crate::selectors::selector_matches_js);
1186            let text_js = text.map(|t| {
1187                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1188                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1189            });
1190
1191            let deadline = Instant::now() + timeout;
1192            loop {
1193                let sel_ok = sel_js
1194                    .as_ref()
1195                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1196                let text_ok = text_js
1197                    .as_ref()
1198                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1199                if sel_ok && text_ok {
1200                    let mut what = Vec::new();
1201                    if let Some(sel) = &selector {
1202                        what.push(format!("found {sel}"));
1203                    }
1204                    if let Some(t) = text {
1205                        what.push(format!("text {t:?} visible"));
1206                    }
1207                    return StepResult {
1208                        name: step_name,
1209                        status: StepStatus::Passed,
1210                        message: what.join(" and "),
1211                    };
1212                }
1213                if Instant::now() >= deadline {
1214                    let mut what = Vec::new();
1215                    if let Some(sel) = &selector {
1216                        what.push(sel.clone());
1217                    }
1218                    if let Some(t) = text {
1219                        what.push(format!("text {t:?}"));
1220                    }
1221                    let message = format!(
1222                        "wait for {} timed out after {}ms: the event waited for never came",
1223                        what.join(" / "),
1224                        timeout.as_millis(),
1225                    );
1226                    if idempotent {
1227                        return StepResult {
1228                            name: step_name,
1229                            status: StepStatus::Skipped,
1230                            message: format!("skipped (idempotent): {message}"),
1231                        };
1232                    }
1233                    return StepResult {
1234                        name: step_name,
1235                        status: StepStatus::Failed,
1236                        message,
1237                    };
1238                }
1239                std::thread::sleep(Duration::from_millis(250));
1240            }
1241        }
1242
1243        match selector.as_deref() {
1244            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1245                Ok(_) => StepResult {
1246                    name: step_name,
1247                    status: StepStatus::Passed,
1248                    message: format!("found {sel}"),
1249                },
1250                Err(e) if idempotent => StepResult {
1251                    name: step_name,
1252                    status: StepStatus::Skipped,
1253                    message: format!(
1254                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1255                        timeout.as_millis()
1256                    ),
1257                },
1258                Err(e) => StepResult {
1259                    name: step_name,
1260                    status: StepStatus::Failed,
1261                    message: format!(
1262                        "wait for {sel} timed out after {}ms: {e}",
1263                        timeout.as_millis()
1264                    ),
1265                },
1266            },
1267            None => StepResult {
1268                name: step_name,
1269                status: StepStatus::Failed,
1270                message: "wait step has neither selector nor text".into(),
1271            },
1272        }
1273    }
1274
1275    #[allow(clippy::too_many_arguments)]
1276    fn run_assert(
1277        &self,
1278        definition: Option<&str>,
1279        preset: Option<&str>,
1280        prompt: Option<&str>,
1281        assert_text: Option<&str>,
1282        screenshot: bool,
1283        step_endpoint: Option<&str>,
1284        test_endpoint: Option<&str>,
1285        tab: &Tab,
1286    ) -> StepResult {
1287        std::thread::sleep(Duration::from_millis(500));
1288
1289        let page_content = get_page_text(tab);
1290
1291        // Vision attach: capture the full page once per assert step and
1292        // split it into viewport-tall tiles (the total coverage is bounded
1293        // by the configured height cap so vision tokens stay sane). All
1294        // tile data URLs are handed to the preset/prompt evaluation below.
1295        let image: Option<Vec<String>> = if screenshot {
1296            let endpoint = self
1297                .endpoints
1298                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1299            if !endpoint.vision {
1300                return StepResult {
1301                    name: "[assert]".into(),
1302                    status: StepStatus::Failed,
1303                    message: format!(
1304                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1305                        name = endpoint.name
1306                    ),
1307                };
1308            }
1309            match crate::vision::capture_screenshot_data_urls(
1310                tab,
1311                self.config
1312                    .screenshot_max_dimension
1313                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1314                self.screenshot_height_cap(),
1315                self.current_viewport_height(),
1316            ) {
1317                Ok(urls) => Some(urls),
1318                Err(e) => {
1319                    return StepResult {
1320                        name: "[assert]".into(),
1321                        status: StepStatus::Failed,
1322                        message: format!("screenshot capture failed: {e}"),
1323                    };
1324                }
1325            }
1326        } else {
1327            None
1328        };
1329
1330        if let Some(def_name) = definition {
1331            if let Some(def) = self.definitions.get(def_name) {
1332                return self.run_assert_def(
1333                    def,
1334                    &page_content,
1335                    image.as_deref(),
1336                    step_endpoint,
1337                    test_endpoint,
1338                    tab,
1339                );
1340            }
1341            return StepResult {
1342                name: format!("[assert] {def_name}"),
1343                status: StepStatus::Failed,
1344                message: format!("definition '{def_name}' not found"),
1345            };
1346        }
1347
1348        if let Some(preset_name) = preset {
1349            // Deterministic DOM layout scan — runs JS in the browser and
1350            // never calls the LLM (free, fast, no pixel budget).
1351            if preset_name == "layout_no_issues" {
1352                return self.run_layout_preset(tab);
1353            }
1354            return self.run_preset(
1355                preset_name,
1356                assert_text,
1357                &page_content,
1358                image.as_deref(),
1359                step_endpoint,
1360                test_endpoint,
1361            );
1362        }
1363
1364        if let Some(prompt_text) = prompt {
1365            return self.run_custom(
1366                prompt_text,
1367                &page_content,
1368                image.as_deref(),
1369                step_endpoint,
1370                test_endpoint,
1371            );
1372        }
1373
1374        StepResult {
1375            name: "[assert]".into(),
1376            status: StepStatus::Skipped,
1377            message: "no definition, preset, or prompt specified".into(),
1378        }
1379    }
1380
1381    fn run_assert_def(
1382        &self,
1383        def: &AssertDefinition,
1384        page_content: &PageContent,
1385        image: Option<&[String]>,
1386        step_endpoint: Option<&str>,
1387        test_endpoint: Option<&str>,
1388        tab: &Tab,
1389    ) -> StepResult {
1390        // Agent-based definition: delegate to an A2A agent
1391        if let Some(ref agent) = def.agent {
1392            if image.is_some() {
1393                return StepResult {
1394                    name: format!("[assert] {}", def.name),
1395                    status: StepStatus::Failed,
1396                    message: "agent-backed assertions do not support screenshots".into(),
1397                };
1398            }
1399            let task = def
1400                .task_template
1401                .as_deref()
1402                .unwrap_or("Evaluate the assertion")
1403                .replace("{url}", &page_content.url)
1404                .replace("{title}", &page_content.title)
1405                .replace("{content}", &page_content.body_text)
1406                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1407
1408            return self.run_agent_step(agent, &task, &def.name);
1409        }
1410
1411        // Custom preset: system + user_template provided in the definition
1412        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1413            return self.run_custom_preset(
1414                &def.name,
1415                system,
1416                template,
1417                def.assert_text.as_deref(),
1418                page_content,
1419                image,
1420                step_endpoint,
1421                test_endpoint,
1422            );
1423        }
1424
1425        def.preset.as_ref().map_or_else(
1426            || {
1427                def.prompt.as_ref().map_or_else(
1428                    || StepResult {
1429                        name: format!("[assert] {}", def.name),
1430                        status: StepStatus::Failed,
1431                        message: "definition has no preset, prompt, or system+user_template".into(),
1432                    },
1433                    |prompt| {
1434                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1435                    },
1436                )
1437            },
1438            |preset_name| {
1439                if preset_name == "layout_no_issues" {
1440                    return self.run_layout_preset(tab);
1441                }
1442                self.run_preset(
1443                    preset_name,
1444                    def.assert_text.as_deref(),
1445                    page_content,
1446                    image,
1447                    step_endpoint,
1448                    test_endpoint,
1449                )
1450            },
1451        )
1452    }
1453
1454    #[allow(clippy::too_many_arguments)]
1455    fn run_custom_preset(
1456        &self,
1457        name: &str,
1458        system: &str,
1459        template: &str,
1460        assert_text: Option<&str>,
1461        page_content: &PageContent,
1462        image: Option<&[String]>,
1463        step_endpoint: Option<&str>,
1464        test_endpoint: Option<&str>,
1465    ) -> StepResult {
1466        let user_prompt = template
1467            .replace("{url}", &page_content.url)
1468            .replace("{title}", &page_content.title)
1469            .replace("{content}", &page_content.body_text)
1470            .replace("{expected_text}", assert_text.unwrap_or(""))
1471            .replace("{description}", "");
1472
1473        // Custom preset definitions frequently forget the {content}
1474        // placeholder — without it the LLM has no page to evaluate and
1475        // answers "I can't determine that without seeing the page". Always
1476        // append the page context unless the template already references it.
1477        let user_prompt = if template.contains("{content}") {
1478            user_prompt
1479        } else {
1480            format!(
1481                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1482                url = page_content.url,
1483                title = page_content.title,
1484                content = page_content.body_text,
1485            )
1486        };
1487
1488        self.reporter
1489            .debug(format!("assert: {name} (custom preset)"));
1490
1491        let chain = self
1492            .endpoints
1493            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1494        let sys = system.to_owned();
1495
1496        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1497
1498        response.map_or_else(
1499            |e| StepResult {
1500                name: format!("[assert] {name}"),
1501                status: StepStatus::Failed,
1502                message: format!("LLM assertion call failed: {e}"),
1503            },
1504            |(lr, _idx)| {
1505                let content_lower = lr.content.to_lowercase().trim().to_owned();
1506                if content_lower.starts_with("pass") {
1507                    StepResult {
1508                        name: format!("[assert] {name}"),
1509                        status: StepStatus::Passed,
1510                        message: "PASS".into(),
1511                    }
1512                } else {
1513                    StepResult {
1514                        name: format!("[assert] {name}"),
1515                        status: StepStatus::Failed,
1516                        message: lr.content,
1517                    }
1518                }
1519            },
1520        )
1521    }
1522
1523    fn run_preset(
1524        &self,
1525        preset_name: &str,
1526        assert_text: Option<&str>,
1527        page_content: &PageContent,
1528        image: Option<&[String]>,
1529        step_endpoint: Option<&str>,
1530        test_endpoint: Option<&str>,
1531    ) -> StepResult {
1532        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1533            return StepResult {
1534                name: format!("[assert] {preset_name}"),
1535                status: StepStatus::Failed,
1536                message: format!("unknown assertion preset: {preset_name}"),
1537            };
1538        };
1539        if preset_name.starts_with("visual_") && image.is_none() {
1540            return StepResult {
1541                name: format!("[assert] {preset_name}"),
1542                status: StepStatus::Failed,
1543                message: format!(
1544                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1545                ),
1546            };
1547        }
1548
1549        let user_prompt = preset
1550            .user_template
1551            .replace("{url}", &page_content.url)
1552            .replace("{title}", &page_content.title)
1553            .replace("{content}", &page_content.body_text)
1554            .replace("{expected_text}", assert_text.unwrap_or(""))
1555            .replace("{description}", "");
1556
1557        // Same safety net as custom presets: never let the LLM answer with
1558        // no page context at all.
1559        let user_prompt = if preset.user_template.contains("{content}") {
1560            user_prompt
1561        } else {
1562            format!(
1563                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1564                url = page_content.url,
1565                title = page_content.title,
1566                content = page_content.body_text,
1567            )
1568        };
1569
1570        self.reporter.debug(format!("assert: {preset_name}"));
1571
1572        let chain = self
1573            .endpoints
1574            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1575        let sys = preset.system.to_owned();
1576
1577        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1578
1579        response.map_or_else(
1580            |e| StepResult {
1581                name: format!("[assert] {preset_name}"),
1582                status: StepStatus::Failed,
1583                message: format!("LLM assertion call failed: {e}"),
1584            },
1585            |(lr, _idx)| {
1586                let content_lower = lr.content.to_lowercase().trim().to_owned();
1587                if content_lower.starts_with("pass") {
1588                    StepResult {
1589                        name: format!("[assert] {preset_name}"),
1590                        status: StepStatus::Passed,
1591                        message: "PASS".into(),
1592                    }
1593                } else {
1594                    StepResult {
1595                        name: format!("[assert] {preset_name}"),
1596                        status: StepStatus::Failed,
1597                        message: lr.content,
1598                    }
1599                }
1600            },
1601        )
1602    }
1603
1604    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1605    ///
1606    /// Evaluates the layout-scan JS in the page and fails with the list of
1607    /// detected issues: horizontal page overflow, visible elements sticking
1608    /// out of the viewport, text clipped by `overflow: hidden` containers,
1609    /// and interactive elements covered by other elements. No LLM call —
1610    /// checks are geometry-based so the check is free, deterministic, and
1611    /// safe to run on every page × viewport variant.
1612    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1613        let name = "[assert] layout_no_issues".to_owned();
1614        self.reporter
1615            .debug("assert: layout_no_issues (DOM layout scan)");
1616        let js = LAYOUT_SCAN_JS.replace(
1617            "__IGNORE_CLASSES__",
1618            &serde_json::to_string(&self.config.layout_ignore_classes)
1619                .unwrap_or_else(|_| "[]".to_owned()),
1620        );
1621        let result = tab.evaluate(&js, false);
1622        let json_str = match result {
1623            Ok(r) => r
1624                .value
1625                .as_ref()
1626                .and_then(|v| v.as_str().map(String::from))
1627                .unwrap_or_else(|| "[]".to_owned()),
1628            Err(e) => {
1629                return StepResult {
1630                    name,
1631                    status: StepStatus::Failed,
1632                    message: format!("layout scan JS failed: {e}"),
1633                };
1634            }
1635        };
1636        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1637        if issues.is_empty() {
1638            return StepResult {
1639                name,
1640                status: StepStatus::Passed,
1641                message: "PASS — no layout defects detected".into(),
1642            };
1643        }
1644        let mut lines: Vec<String> = issues
1645            .iter()
1646            .take(10)
1647            .map(|i| {
1648                format!(
1649                    "- [{type_}] {element}: {detail}",
1650                    type_ = i.issue_type,
1651                    element = i.element,
1652                    detail = i.detail
1653                )
1654            })
1655            .collect();
1656        if issues.len() > 10 {
1657            lines.push(format!("- … and {} more", issues.len() - 10));
1658        }
1659        StepResult {
1660            name,
1661            status: StepStatus::Failed,
1662            message: format!(
1663                "FAIL — {} layout defect(s) detected:\n{}",
1664                issues.len(),
1665                lines.join("\n")
1666            ),
1667        }
1668    }
1669
1670    fn run_custom(
1671        &self,
1672        prompt: &str,
1673        page_content: &PageContent,
1674        image: Option<&[String]>,
1675        step_endpoint: Option<&str>,
1676        test_endpoint: Option<&str>,
1677    ) -> StepResult {
1678        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1679
1680        let mut user = format!(
1681            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1682            url = page_content.url,
1683            title = page_content.title,
1684            content = page_content.body_text,
1685        );
1686        if image.is_some() {
1687            user.push_str(
1688                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1689            );
1690        }
1691
1692        self.reporter.debug("custom assert");
1693
1694        let chain = self
1695            .endpoints
1696            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1697        let sys = system.to_owned();
1698
1699        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1700
1701        response.map_or_else(
1702            |e| StepResult {
1703                name: "[assert] custom".into(),
1704                status: StepStatus::Failed,
1705                message: format!("LLM assertion call failed: {e}"),
1706            },
1707            |(lr, _idx)| {
1708                let content_lower = lr.content.to_lowercase().trim().to_owned();
1709                if content_lower.starts_with("pass") {
1710                    StepResult {
1711                        name: "[assert] custom".into(),
1712                        status: StepStatus::Passed,
1713                        message: "PASS".into(),
1714                    }
1715                } else {
1716                    StepResult {
1717                        name: "[assert] custom".into(),
1718                        status: StepStatus::Failed,
1719                        message: lr.content,
1720                    }
1721                }
1722            },
1723        )
1724    }
1725
1726    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1727        let path = path.unwrap_or("screenshot.png");
1728
1729        match tab.capture_screenshot(
1730            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1731            None,
1732            None,
1733            true,
1734        ) {
1735            Ok(data) => {
1736                if let Err(e) = std::fs::write(path, &data) {
1737                    return StepResult {
1738                        name: format!("[screenshot] {path}"),
1739                        status: StepStatus::Failed,
1740                        message: format!("failed to write screenshot: {e}"),
1741                    };
1742                }
1743                StepResult {
1744                    name: format!("[screenshot] {path}"),
1745                    status: StepStatus::Passed,
1746                    message: format!("saved to {path}"),
1747                }
1748            }
1749            Err(e) => StepResult {
1750                name: format!("[screenshot] {path}"),
1751                status: StepStatus::Failed,
1752                message: format!("screenshot failed: {e}"),
1753            },
1754        }
1755    }
1756
1757    /// Runs an A2A agent step.
1758    #[allow(clippy::literal_string_with_formatting_args)]
1759    fn run_agent(
1760        &self,
1761        agent_name: &str,
1762        task: &str,
1763        definition: Option<&str>,
1764        _test_endpoint: Option<&str>,
1765    ) -> StepResult {
1766        // If a definition is specified, look up the task template
1767        let resolved_task = if let Some(def_name) = definition {
1768            if let Some(def) = self.definitions.get(def_name) {
1769                let tmpl = def.task_template.as_deref().unwrap_or(task);
1770                tmpl.replace("{task}", task)
1771            } else {
1772                return StepResult {
1773                    name: format!("[agent] {def_name}"),
1774                    status: StepStatus::Failed,
1775                    message: format!("definition '{def_name}' not found"),
1776                };
1777            }
1778        } else {
1779            task.to_owned()
1780        };
1781
1782        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1783    }
1784
1785    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1786        let Some(ep) = self.endpoints.get(agent_name) else {
1787            return StepResult {
1788                name: format!("[agent] {display_name}"),
1789                status: StepStatus::Failed,
1790                message: format!("agent endpoint '{agent_name}' not found"),
1791            };
1792        };
1793
1794        if ep.url.is_empty() {
1795            return StepResult {
1796                name: format!("[agent] {display_name}"),
1797                status: StepStatus::Failed,
1798                message: format!("agent endpoint '{agent_name}' has no URL"),
1799            };
1800        }
1801
1802        self.reporter.debug(format!("agent {agent_name}: {task}"));
1803
1804        let url = ep.url.clone();
1805        let client = A2aClient::new(&url, self.timeout);
1806        let task_clone = task.to_owned();
1807
1808        let response = std::thread::spawn(move || {
1809            let rt = tokio::runtime::Builder::new_current_thread()
1810                .enable_all()
1811                .build()
1812                .unwrap();
1813            rt.block_on(client.send_task(&task_clone))
1814        })
1815        .join()
1816        .unwrap();
1817
1818        // Record the flat-cost call
1819        self.usage.record_flat_call(agent_name, ep);
1820
1821        match response {
1822            Ok(text) => {
1823                let clean = text.trim().to_owned();
1824                let lower = clean.to_lowercase();
1825                if lower.starts_with("pass") {
1826                    StepResult {
1827                        name: format!("[agent] {display_name}"),
1828                        status: StepStatus::Passed,
1829                        message: format!("PASS: {clean}"),
1830                    }
1831                } else if lower.starts_with("fail") {
1832                    StepResult {
1833                        name: format!("[agent] {display_name}"),
1834                        status: StepStatus::Failed,
1835                        message: clean,
1836                    }
1837                } else {
1838                    StepResult {
1839                        name: format!("[agent] {display_name}"),
1840                        status: StepStatus::Passed,
1841                        message: format!("response: {clean}"),
1842                    }
1843                }
1844            }
1845            Err(e) => StepResult {
1846                name: format!("[agent] {display_name}"),
1847                status: StepStatus::Failed,
1848                message: format!("agent call failed: {e}"),
1849            },
1850        }
1851    }
1852
1853    /// Runs an MCP tool call step.
1854    fn run_mcp(
1855        &self,
1856        server_name: &str,
1857        tool_name: &str,
1858        args: Option<&serde_json::Value>,
1859    ) -> StepResult {
1860        let Some(ep) = self.endpoints.get(server_name) else {
1861            return StepResult {
1862                name: format!("[mcp] {server_name}:{tool_name}"),
1863                status: StepStatus::Failed,
1864                message: format!("MCP server endpoint '{server_name}' not found"),
1865            };
1866        };
1867
1868        let cmd = ep.command.as_deref().unwrap_or("");
1869        if cmd.is_empty() {
1870            return StepResult {
1871                name: format!("[mcp] {server_name}:{tool_name}"),
1872                status: StepStatus::Failed,
1873                message: format!("MCP server '{server_name}' has no command configured"),
1874            };
1875        }
1876
1877        self.reporter
1878            .debug(format!("mcp {server_name} {tool_name}"));
1879
1880        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1881
1882        let command = cmd.to_owned();
1883        let args_vec = ep.args.clone();
1884        let tool = tool_name.to_owned();
1885
1886        let response = std::thread::spawn(move || {
1887            let mut mcp_client =
1888                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1889            mcp_client
1890                .call_tool(&tool, &args_val)
1891                .map_err(|e| e.to_string())
1892        })
1893        .join()
1894        .unwrap();
1895
1896        // Record the flat-cost call
1897        self.usage.record_flat_call(server_name, ep);
1898
1899        match response {
1900            Ok(result) => {
1901                if result.isError {
1902                    StepResult {
1903                        name: format!("[mcp] {server_name}:{tool_name}"),
1904                        status: StepStatus::Failed,
1905                        message: result.to_string(),
1906                    }
1907                } else {
1908                    StepResult {
1909                        name: format!("[mcp] {server_name}:{tool_name}"),
1910                        status: StepStatus::Passed,
1911                        message: result.to_string(),
1912                    }
1913                }
1914            }
1915            Err(e) => StepResult {
1916                name: format!("[mcp] {server_name}:{tool_name}"),
1917                status: StepStatus::Failed,
1918                message: format!("MCP call failed: {e}"),
1919            },
1920        }
1921    }
1922
1923    // ── helpers ──────────────────────────────────────────────────────────
1924
1925    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1926    /// the runner's default LLM config for any unset fields.
1927    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1928        LlmConfig {
1929            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
1930                // Bedrock builds its endpoint from the resolved AWS region
1931                // when no URL is given — never inherit the default LLM URL.
1932                endpoint.url.clone()
1933            } else if endpoint.url.is_empty() {
1934                self.llm.url.clone()
1935            } else {
1936                endpoint.url.clone()
1937            },
1938            model: endpoint
1939                .model
1940                .clone()
1941                .unwrap_or_else(|| self.llm.model.clone()),
1942            api_key: endpoint
1943                .api_key
1944                .clone()
1945                .or_else(|| self.llm.api_key.clone()),
1946            headers: if endpoint.headers.is_empty() {
1947                self.llm.headers.clone()
1948            } else {
1949                endpoint.headers.clone()
1950            },
1951            timeout: self.llm.timeout,
1952            temperature: self.llm.temperature,
1953            thinking: self.llm.thinking,
1954            model_params: self.llm.model_params.clone(),
1955            cache: endpoint.cache_markers,
1956            max_attempts: endpoint.max_attempts.max(1),
1957            provider: endpoint.provider,
1958            deployment: endpoint.deployment.clone(),
1959            api_version: endpoint.api_version.clone(),
1960            auth: endpoint.auth.clone(),
1961            header_commands: endpoint.header_commands.clone(),
1962            aws: endpoint.aws.clone(),
1963        }
1964    }
1965
1966    /// Runs a single LLM call against an ordered endpoint chain (primary +
1967    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
1968    /// the first endpoint that answers wins. Returns the response together
1969    /// with the chain index of the answering endpoint (0 = primary) so the
1970    /// caller can attribute usage to the correct endpoint.
1971    ///
1972    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
1973    /// shows duration, tokens, cost and the answering endpoint per call.
1974    /// Run-level context handed to every LLM call as the FIRST block of the
1975    /// user message. Contents that are stable for the whole run ("run
1976    /// started", "target site") come first so upstream provider prefix
1977    /// caching stays effective; the current time is the last line because
1978    /// it changes on every call.
1979    fn run_context(&self) -> String {
1980        let mut parts = vec![
1981            "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
1982            "================================================================".into(),
1983            format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
1984        ];
1985        if let Some(base) = self.config.base_url.as_deref() {
1986            parts.push(format!("Target site: {base}"));
1987        }
1988        let now = SystemTime::now()
1989            .duration_since(UNIX_EPOCH)
1990            .map_or(0, |d| d.as_secs());
1991        parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
1992        parts.join("\n")
1993    }
1994
1995    #[allow(clippy::cast_possible_truncation, clippy::too_many_lines)]
1996    fn llm_call_chain(
1997        &self,
1998        chain: &[&ResolvedEndpoint],
1999        system: &str,
2000        user: &str,
2001        image: Option<&[String]>,
2002        purpose: &str,
2003    ) -> Result<(crate::costs::LlmResponse, usize), String> {
2004        if chain.is_empty() {
2005            return Err("empty LLM endpoint chain".into());
2006        }
2007        let primary = self.build_llm_for_endpoint(chain[0]);
2008        let fallbacks: Vec<LlmConfig> = chain[1..]
2009            .iter()
2010            .map(|e| self.build_llm_for_endpoint(e))
2011            .collect();
2012
2013        let (test, index) = self
2014            .current_step
2015            .borrow()
2016            .as_ref()
2017            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
2018        let primary_endpoint = chain[0].name.clone();
2019        let primary_model = primary.model.clone();
2020        self.emit_event(&TestEvent::LlmCallStarted {
2021            test: test.clone(),
2022            index,
2023            endpoint: primary_endpoint.clone(),
2024            model: primary_model.clone(),
2025            purpose: purpose.to_owned(),
2026        });
2027
2028        let started = Instant::now();
2029        let sys = system.to_owned();
2030        let context = self.run_context();
2031        let user = if context.is_empty() {
2032            user.to_owned()
2033        } else {
2034            format!("{context}\n\n{user}")
2035        };
2036        let image = image.map(<[String]>::to_vec);
2037
2038        let result = std::thread::spawn(move || {
2039            let rt = tokio::runtime::Builder::new_current_thread()
2040                .enable_all()
2041                .build()
2042                .unwrap();
2043            let call = async {
2044                match image.as_deref() {
2045                    Some(img) => {
2046                        llm_chat_vision_with_usage_chain(
2047                            &primary,
2048                            &fallbacks,
2049                            &sys,
2050                            &user,
2051                            Some(img),
2052                        )
2053                        .await
2054                    }
2055                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2056                }
2057            };
2058            rt.block_on(call)
2059        })
2060        .join()
2061        .unwrap();
2062
2063        let duration_ms = started.elapsed().as_millis() as u64;
2064        match result {
2065            Ok((lr, idx)) => {
2066                let cost = calculate_llm_cost(chain[idx], &lr.usage);
2067                let answering = chain[idx].name.clone();
2068                let model = chain[idx]
2069                    .model
2070                    .clone()
2071                    .unwrap_or_else(|| primary_model.clone());
2072                self.usage
2073                    .record_llm_call(&answering, chain[idx], &model, &lr.usage);
2074                self.emit_event(&TestEvent::LlmCallFinished {
2075                    test,
2076                    index,
2077                    endpoint: answering,
2078                    model,
2079                    purpose: purpose.to_owned(),
2080                    ok: true,
2081                    duration_ms,
2082                    input_tokens: lr.usage.prompt_tokens,
2083                    output_tokens: lr.usage.completion_tokens,
2084                    cached_input_tokens: lr.usage.cached_input_tokens,
2085                    cache_creation_input_tokens: lr.usage.cache_creation_input_tokens,
2086                    cost,
2087                    error: None,
2088                });
2089                Ok((lr, idx))
2090            }
2091            Err(e) => {
2092                self.emit_event(&TestEvent::LlmCallFinished {
2093                    test,
2094                    index,
2095                    endpoint: primary_endpoint,
2096                    model: primary_model,
2097                    purpose: purpose.to_owned(),
2098                    ok: false,
2099                    duration_ms,
2100                    input_tokens: 0,
2101                    output_tokens: 0,
2102                    cached_input_tokens: 0,
2103                    cache_creation_input_tokens: 0,
2104                    cost: 0.0,
2105                    error: Some(e.clone()),
2106                });
2107                Err(e)
2108            }
2109        }
2110    }
2111
2112    /// Resolves a CSS selector for the target element. Uses the explicit
2113    /// `selector` if provided, otherwise asks the LLM to find the element
2114    /// from the natural language `target` description and page DOM.
2115    ///
2116    /// LLM responses are sanitized and verified against the live page: a
2117    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2118    /// immediately with the raw LLM output, and a selector that matches
2119    /// nothing triggers one retry with feedback before failing.
2120    #[allow(clippy::too_many_lines)]
2121    fn resolve_selector(
2122        &self,
2123        css_override: Option<&str>,
2124        target: &str,
2125        step_endpoint: Option<&str>,
2126        test_endpoint: Option<&str>,
2127        tab: &Tab,
2128    ) -> Result<String, String> {
2129        if let Some(explicit) = css_override {
2130            return Ok(explicit.to_owned());
2131        }
2132
2133        let dom_info = extract_dom_info(tab)?;
2134        let page_content = get_page_text(tab);
2135
2136        let system = concat!(
2137            "You are a browser automation selector generator. ",
2138            "Given a web page's content and interactive elements, ",
2139            "return ONLY the best CSS selector for the described element. ",
2140            "Output nothing except the CSS selector. ",
2141            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2142            "[name=\"...\"], tag.class, tag. ",
2143            "Never output explanations, markdown, or extra text."
2144        );
2145
2146        let user = format!(
2147            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2148            page_content.url,
2149            page_content.title,
2150            truncate(&page_content.body_text, 4000),
2151            dom_info,
2152            target,
2153        );
2154
2155        let retry_user = format!(
2156            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2157            "The selector must match at least one element currently present on the page.",
2158            page_content.url,
2159            page_content.title,
2160            truncate(&page_content.body_text, 4000),
2161            dom_info,
2162            target,
2163        );
2164
2165        self.reporter.debug(format!("LLM targeting: {target}"));
2166
2167        let chain = self
2168            .endpoints
2169            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2170        let sys = system.to_owned();
2171
2172        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2173
2174        let first = call_llm(&user);
2175        let (lr, _idx) = match first {
2176            Ok(lr) => lr,
2177            Err(e) => {
2178                return Err(format!("LLM element targeting failed: {e}"));
2179            }
2180        };
2181        let clean = sanitize_selector(&lr.content);
2182        self.reporter.debug(format!("resolved selector: {clean}"));
2183
2184        if selector_is_useless(&clean) {
2185            return Err(format!(
2186                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2187                raw = lr.content.trim(),
2188            ));
2189        }
2190        if let Err(reason) = validate_selector(&clean) {
2191            return Err(format!(
2192                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2193                raw = lr.content.trim(),
2194            ));
2195        }
2196        if !selector_matches(tab, &clean).unwrap_or(false) {
2197            // One retry with feedback: flaky models occasionally invent a
2198            // selector that does not exist on the page.
2199            self.reporter.warn(format!(
2200                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2201            ));
2202            let second = call_llm(&retry_user);
2203            let (lr2, _idx2) = match second {
2204                Ok(lr2) => lr2,
2205                Err(e) => {
2206                    return Err(format!(
2207                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2208                    ));
2209                }
2210            };
2211            let clean2 = sanitize_selector(&lr2.content);
2212            self.reporter
2213                .debug(format!("resolved selector (retry): {clean2}"));
2214            if selector_is_useless(&clean2) {
2215                return Err(format!(
2216                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2217                    raw = lr2.content.trim(),
2218                    excerpt = truncate(&page_content.body_text, 300),
2219                ));
2220            }
2221            if !selector_matches(tab, &clean2).unwrap_or(false) {
2222                return Err(format!(
2223                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2224                ));
2225            }
2226            return Ok(clean2);
2227        }
2228
2229        Ok(clean)
2230    }
2231}
2232
2233/// Evaluates a JS expression that is expected to return a boolean.
2234fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2235    tab.evaluate(js, false)
2236        .map_err(|e| format!("evaluate failed: {e}"))?
2237        .value
2238        .and_then(|v| v.as_bool())
2239        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2240}
2241
2242/// Checks whether a CSS selector matches at least one current element.
2243fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2244    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2245}
2246
2247// ── Free helper functions ──────────────────────────────────────────────
2248
2249fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2250    let name = format!("[navigate] {full_url}");
2251    match tab.navigate_to(full_url) {
2252        Ok(_) => {
2253            let _ = tab.wait_until_navigated();
2254            StepResult {
2255                name,
2256                status: StepStatus::Passed,
2257                message: format!("navigated to {full_url}"),
2258            }
2259        }
2260        Err(e) => StepResult {
2261            name,
2262            status: StepStatus::Failed,
2263            message: format!("navigation failed: {e}"),
2264        },
2265    }
2266}
2267
2268fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2269    let result = tab
2270        .evaluate(DOM_EXTRACT_JS, false)
2271        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2272
2273    let json_str = result
2274        .value
2275        .as_ref()
2276        .and_then(|v| v.as_str())
2277        .unwrap_or("[]");
2278
2279    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2280
2281    if elements.is_empty() {
2282        return Ok("(no interactive elements found)".to_owned());
2283    }
2284
2285    Ok(elements.join("\n"))
2286}
2287
2288fn get_page_text(tab: &Tab) -> PageContent {
2289    let url = tab.get_url();
2290
2291    let title = tab
2292        .evaluate("document.title", false)
2293        .ok()
2294        .and_then(|r| r.value)
2295        .and_then(|v| v.as_str().map(String::from))
2296        .unwrap_or_else(|| "unknown".to_owned());
2297
2298    let body_text = tab
2299        .evaluate(
2300            "document.body ? document.body.innerText : document.documentElement.innerText",
2301            false,
2302        )
2303        .ok()
2304        .and_then(|r| r.value)
2305        .and_then(|v| v.as_str().map(String::from))
2306        .unwrap_or_default();
2307
2308    PageContent {
2309        url,
2310        title,
2311        body_text: truncate(&body_text, 8000),
2312    }
2313}
2314
2315fn resolve_url(url: &str, base_url: &str) -> String {
2316    if url.starts_with("http://") || url.starts_with("https://") {
2317        return url.to_owned();
2318    }
2319    let base = base_url.trim_end_matches('/');
2320    if url.starts_with('/') {
2321        format!("{base}{url}")
2322    } else {
2323        format!("{base}/{url}")
2324    }
2325}
2326
2327/// Human-readable label for a step, used when steps are skipped after an
2328/// earlier failure.
2329fn step_label(step: &TestStep) -> String {
2330    match step {
2331        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2332        TestStep::Click { target, .. } => format!("[click] {target}"),
2333        TestStep::Type { target, .. } => format!("[type] {target}"),
2334        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2335        TestStep::Assert {
2336            definition,
2337            preset,
2338            prompt,
2339            ..
2340        } => definition.as_ref().map_or_else(
2341            || {
2342                preset.as_ref().map_or_else(
2343                    || {
2344                        prompt.as_ref().map_or_else(
2345                            || "[assert]".to_owned(),
2346                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2347                        )
2348                    },
2349                    |p| format!("[assert] {p}"),
2350                )
2351            },
2352            |d| format!("[assert] {d}"),
2353        ),
2354        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2355        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2356        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2357    }
2358}
2359
2360/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2361#[must_use]
2362const fn step_kind_label(step: &TestStep) -> &'static str {
2363    match step {
2364        TestStep::Navigate { .. } => "navigate",
2365        TestStep::Click { .. } => "click",
2366        TestStep::Type { .. } => "type",
2367        TestStep::Wait { .. } => "wait",
2368        TestStep::Assert { .. } => "assert",
2369        TestStep::Screenshot { .. } => "screenshot",
2370        TestStep::Agent { .. } => "agent",
2371        TestStep::Mcp { .. } => "mcp",
2372    }
2373}
2374
2375// ── Support types ──────────────────────────────────────────────────────
2376
2377#[derive(Default)]
2378struct TestRunResult {
2379    passed: u32,
2380    failed: u32,
2381    skipped: u32,
2382    total: u32,
2383    details: Vec<StepResult>,
2384}
2385
2386struct PageContent {
2387    url: String,
2388    title: String,
2389    body_text: String,
2390}
2391
2392#[cfg(test)]
2393mod tests {
2394    use super::unix_to_rfc3339;
2395
2396    #[test]
2397    fn rfc3339_epoch_and_reference_dates() {
2398        assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2399        assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2400        assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2401        assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2402    }
2403
2404    #[test]
2405    fn rfc3339_handles_leap_years() {
2406        assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2407        assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2408    }
2409}