Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// Formats a UNIX epoch timestamp as an RFC 3339 UTC string
26/// (`2026-09-19T16:04:00Z`) without pulling in a date dependency.
27/// Civil-from-days conversion after Howard Hinnant.
28#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31    let secs = i64::try_from(secs).unwrap_or(0);
32    let days = secs.div_euclid(86_400);
33    let rem = secs.rem_euclid(86_400);
34    let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35    let z = days + 719_468;
36    let era = z.div_euclid(146_097);
37    let doe = z.rem_euclid(146_097);
38    let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40    let mp = (5 * doy + 2) / 153;
41    let d = doy - (153 * mp + 2) / 5 + 1;
42    let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43    let y = yoe + era * 400 + i64::from(mth <= 2);
44    format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47/// One detected layout defect (`layout_no_issues` preset).
48#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50    #[serde(rename = "type")]
51    issue_type: String,
52    element: String,
53    detail: String,
54}
55
56/// In-browser DOM layout scan for `layout_no_issues`.
57///
58/// Geometry-only checks (no LLM, no pixels):
59/// 1. page horizontal overflow (`scrollWidth` > viewport width);
60/// 2. elements outside the viewport that scrolling cannot reveal
61///    (fixed elements off-screen, left/negative overflow, right-edge
62///    overflow beyond the horizontally scrollable area, and bottom
63///    overflow on a page that cannot scroll down) — below-the-fold
64///    content on a tall scrollable page is normal flow, NOT a defect;
65/// 3. text clipped by `overflow: hidden` containers whose content
66///    is measurably larger than the box;
67/// 4. interactive elements (buttons/links/inputs) whose center point
68///    is covered by a different element that would intercept the click.
69///
70/// Elements whose class matches a configured ignore prefix (default:
71/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
72/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
73/// skipped — they are intentionally 1x1 / off-screen. The prefix list
74/// is injected at `__IGNORE_CLASSES__` from
75/// [`ScenarioConfig::layout_ignore_classes`].
76///
77/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
78/// sticky headers) is excluded by the position/relation filters.
79const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81  const issues = [];
82  const push = (type, el, detail) => {
83    if (issues.length >= 30) return;
84    let element = el.tagName.toLowerCase();
85    if (el.id) element += '#' + el.id;
86    else if (typeof el.className === 'string' && el.className.trim())
87      element += '.' + el.className.trim().split(/\s+/).join('.');
88    issues.push({ type, element, detail: String(detail).slice(0, 220) });
89  };
90  const vw = document.documentElement.clientWidth || window.innerWidth;
91  const vh = document.documentElement.clientHeight || window.innerHeight;
92  if (!vw || !vh) return JSON.stringify(issues);
93  const de = document.documentElement;
94  // 1. Page-level horizontal overflow.
95  if (maxSW > vw + 2)
96    push('page-overflow-x', de,
97      'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98  // Class-name prefixes to skip (injected; default = Angular CDK
99  // screen-reader helpers, which are intentionally 1x1 / off-screen).
100  const ignorePrefixes = __IGNORE_CLASSES__;
101  const isIgnored = (el) => {
102    if (typeof el.className !== 'string' || !el.className.trim()) return false;
103    const classes = el.className.trim().split(/\s+/);
104    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105  };
106  const body = document.body;
107  // The document element is not always the scroll container: the app may
108  // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109  // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110  // "not scrollable" — measure the scrollable content height/width of the
111  // document element AND the body, and take the max.
112  const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113  const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114  const vScrollable = maxSH > vh + 2;
115  const hScrollable = maxSW > vw + 2;
116  // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117  // this element into view on the given axis — i.e. the element is inside a
118  // scrolling region, so lying beyond the viewport cut is not a defect.
119  const reachableByScroller = (el, axis) => {
120    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123    let p = el.parentElement;
124    while (p && p !== body) {
125      const pcs = getComputedStyle(p);
126      const o = pcs[ovProp];
127      if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128          p[dim] > p[dimClient] + 2) return true;
129      p = p.parentElement;
130    }
131    return false;
132  };
133  const selfOverflowing = (el, cs, axis) => {
134    const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135    const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136    const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137    return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138      el[dim] > el[dimClient] + 2;
139  };
140  // True when an ancestor clips this axis with overflow:hidden/clip — the
141  // element's overhang is not visible, so treat it as reachable.
142  const clippedByAncestor = (el, axis) => {
143    const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144    let p = el.parentElement;
145    while (p && p !== body) {
146      const o = getComputedStyle(p)[ovProp];
147      if (o === 'hidden' || o === 'clip') return true;
148      p = p.parentElement;
149    }
150    return false;
151  };
152  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154  const hasContent = (el) =>
155    ((el.textContent || '').trim().length > 0) ||
156    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157  // 2. Elements outside the viewport that scrolling cannot reveal.
158  for (const el of all) {
159    const cs = getComputedStyle(el);
160    if (!visible(cs)) continue;
161    if (isIgnored(el)) continue;
162    const r = el.getBoundingClientRect();
163    if (r.width < 2 || r.height < 2) continue;
164    if (!hasContent(el) && el.children.length === 0) continue;
165    if (cs.position === 'fixed') {
166      // Fixed elements never move with the scroll: any edge outside the
167      // viewport is unreachable content and therefore a defect.
168      const overTop = -r.top;
169      const overLeft = -r.left;
170      const overRight = r.right - vw;
171      const overBottom = r.bottom - vh;
172      // Top/left overshoot is never reachable; bottom/right overshoot is
173      // only a defect when the fixed element cannot scroll that content
174      // into view itself (e.g. an opened Material drawer whose inner
175      // container scrolls is not a "cut-off" defect).
176      if (overTop > 2 || overLeft > 2 ||
177          (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178          (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179        let where = '';
180        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186        push('element-out-of-viewport', el,
187          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188      }
189      continue;
190    }
191    if (cs.position === 'sticky') continue;
192    // Negative left overflow cannot be reached by scrolling (scrollLeft
193    // never goes below 0).
194    if (r.left < -2) {
195      push('element-out-of-viewport', el,
196        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197      continue;
198    }
199    // Negative top with the page at the top means the element sits above
200    // the document origin — also unreachable.
201    if (r.top < -2 && de.scrollTop <= 2) {
202      push('element-out-of-viewport', el,
203        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204      continue;
205    }
206    const overRight = r.right - vw;
207    // Right-edge overflow is only a defect when the page cannot scroll
208    // horizontally to reveal it (or the element sticks out past the
209    // scrollable content width itself).
210    if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211      push('element-out-of-viewport', el,
212        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213      continue;
214    }
215    // Below-the-fold content on a scrollable page is normal (tall landing
216    // pages); only flag bottom overflow the user can never scroll to.
217    const overBottom = r.bottom - vh;
218    if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219      push('element-out-of-viewport', el,
220        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221    }
222  }
223  // 3. Text clipped by overflow:hidden containers.
224  for (const el of all) {
225    if (isIgnored(el)) continue;
226    const cs = getComputedStyle(el);
227    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229    if (!(el.textContent || '').trim()) continue;
230    // Single-line truncation with an ellipsis is an intentional design
231    // pattern (Tailwind .truncate etc.), not a clipping defect.
232    if (cs.textOverflow === 'ellipsis') continue;
233    push('text-clipped', el,
234      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236  }
237  // 4. Interactive elements covered by a different element.
238  const interactive =
239    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240  const targets = document.querySelectorAll(interactive);
241  for (const el of targets) {
242    if (isIgnored(el)) continue;
243    const r = el.getBoundingClientRect();
244    if (r.width < 6 || r.height < 6) continue;
245    const cs = getComputedStyle(el);
246    if (!visible(cs)) continue;
247    const cx = r.left + r.width / 2;
248    const cy = r.top + r.height / 2;
249    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250    const top = document.elementFromPoint(cx, cy);
251    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252    if (isIgnored(top)) continue;
253    const tcs = getComputedStyle(top);
254    if (!visible(tcs)) continue;
255    if (tcs.pointerEvents === 'none') continue;
256    const tr = top.getBoundingClientRect();
257    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260    push('element-overlap', el,
261      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262      ' is covered by <' + tname + '>');
263  }
264  return JSON.stringify(issues);
265})()
266"#;
267
268/// How long the CDP connection stays open after the browser goes quiet.
269///
270/// `headless_chrome` ships a 30s default and tears down the entire connection
271/// when no traffic arrives for that long; a run must own its connection for
272/// its full duration instead.
273const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275/// Executes a [`Scenario`] against a real browser with optional LLM
276/// assistance for element targeting and assertions.
277pub struct ScenarioRunner {
278    config: ScenarioConfig,
279    definitions: HashMap<String, AssertDefinition>,
280    llm: LlmConfig,
281    timeout: Duration,
282    viewport_width: u32,
283    viewport_height: u32,
284    /// The viewport currently applied in the browser (CDP emulation).
285    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
286    applied_viewport: std::cell::Cell<(u32, u32)>,
287    endpoints: EndpointRegistry,
288    usage: Arc<UsageTracker>,
289    budgets: BudgetTracker,
290    /// Directory for failure artifacts (screenshots).
291    artifacts_dir: PathBuf,
292    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
293    reporter: Arc<Reporter>,
294    /// The test + step index currently executing, for LLM-call events.
295    current_step: std::cell::RefCell<Option<(String, u32)>>,
296    /// Epoch seconds of runner creation — constant for the whole run so
297    /// the LLM context preamble below is cacheable by upstream providers.
298    run_started: u64,
299    /// Whether this runner emits the `RunStarted`/`RunFinished` events.
300    /// When several scenario files are run concurrently by
301    /// [`crate::parallel::run_scenarios`], the orchestrator owns those two
302    /// events (one per batch), so each per-file runner suppresses its own.
303    emit_run_events: bool,
304}
305
306/// Aggregated results from a scenario run.
307#[derive(Debug, Default)]
308pub struct RunReport {
309    /// Number of tests that passed.
310    pub tests_passed: u32,
311    /// Number of tests that failed.
312    pub tests_failed: u32,
313    /// Number of steps that passed.
314    pub passed: u32,
315    /// Number of steps that failed.
316    pub failed: u32,
317    /// Number of steps that were skipped.
318    pub skipped: u32,
319    /// Per-step details.
320    pub details: Vec<StepResult>,
321}
322
323/// Result of a single step execution.
324#[derive(Debug, Clone)]
325pub struct StepResult {
326    /// The step name.
327    pub name: String,
328    /// Whether the step passed, failed, or was skipped.
329    pub status: StepStatus,
330    /// Human-readable result message.
331    pub message: String,
332}
333
334/// Outcome for a single step.
335pub use crate::events::StepStatus;
336
337/// Predefined assertion preset definition.
338struct AssertPreset {
339    name: &'static str,
340    system: &'static str,
341    user_template: &'static str,
342}
343
344/// Built-in assertion presets.
345#[allow(clippy::literal_string_with_formatting_args)]
346const ASSERTION_PRESETS: &[AssertPreset] = &[
347    AssertPreset {
348        name: "no_error_on_page",
349        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
350        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
351    },
352    AssertPreset {
353        name: "text_visible",
354        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
355        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
356    },
357    AssertPreset {
358        name: "element_exists",
359        system: "You are a QA tester. Check if a described UI element exists on a web page.",
360        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
361    },
362    AssertPreset {
363        name: "visual_no_issues",
364        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
365        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
366    },
367    AssertPreset {
368        name: "visual_no_overlaps",
369        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
370        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
371    },
372    AssertPreset {
373        name: "visual_text_visible",
374        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
375        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
376    },
377    AssertPreset {
378        name: "layout_no_issues",
379        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
380        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
381    },
382];
383
384impl ScenarioRunner {
385    /// Creates a new runner with the given scenario configuration and
386    /// assertion definitions.
387    #[must_use]
388    #[allow(clippy::needless_pass_by_value)]
389    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
390        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
391    }
392
393    /// Creates a runner that reports run events through the given reporter.
394    #[must_use]
395    #[allow(clippy::needless_pass_by_value)]
396    pub fn with_reporter(
397        scenario_config: ScenarioConfig,
398        definitions: Vec<AssertDefinition>,
399        reporter: Arc<Reporter>,
400    ) -> Self {
401        Self::with_reporter_mode(scenario_config, definitions, reporter, true)
402    }
403
404    /// Creates a runner that reports run events through the given reporter
405    /// but suppresses the `RunStarted`/`RunFinished` events. Used by the
406    /// parallel orchestrator, which emits those once per batch.
407    #[must_use]
408    #[allow(clippy::needless_pass_by_value)]
409    pub fn with_reporter_parallel(
410        scenario_config: ScenarioConfig,
411        definitions: Vec<AssertDefinition>,
412        reporter: Arc<Reporter>,
413    ) -> Self {
414        Self::with_reporter_mode(scenario_config, definitions, reporter, false)
415    }
416
417    #[must_use]
418    #[allow(clippy::needless_pass_by_value)]
419    fn with_reporter_mode(
420        scenario_config: ScenarioConfig,
421        definitions: Vec<AssertDefinition>,
422        reporter: Arc<Reporter>,
423        emit_run_events: bool,
424    ) -> Self {
425        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
426            &scenario_config,
427        ));
428        let llm = LlmConfig {
429            url: scenario_config
430                .llm_url
431                .clone()
432                .unwrap_or_else(crate::llm_base_url),
433            model: scenario_config
434                .llm_model
435                .clone()
436                .unwrap_or_else(crate::llm_model),
437            api_key: scenario_config
438                .llm_api_key
439                .clone()
440                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
441            headers: if scenario_config.llm_headers.is_empty() {
442                crate::parse_headers_env()
443            } else {
444                scenario_config.llm_headers.clone()
445            },
446            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
447            temperature: scenario_config.temperature,
448            thinking: scenario_config.thinking,
449            model_params: scenario_config.model_params.clone(),
450            max_attempts: crate::default_llm_attempts(),
451            provider: crate::scenario::Provider::Openai,
452            deployment: None,
453            api_version: None,
454            auth: crate::scenario::AuthConfig::default(),
455            header_commands: std::collections::HashMap::new(),
456            aws: crate::scenario::AwsConfig::default(),
457        };
458        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
459        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
460        let defs_map: HashMap<String, AssertDefinition> = definitions
461            .into_iter()
462            .map(|d| (d.name.clone(), d))
463            .collect();
464
465        Self {
466            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
467            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
468            viewport_height: scenario_config.viewport_height.unwrap_or(720),
469            applied_viewport: std::cell::Cell::new((0, 0)),
470            config: scenario_config.clone(),
471            definitions: defs_map,
472            llm,
473            endpoints,
474            usage: Arc::new(UsageTracker::new()),
475            budgets,
476            artifacts_dir: PathBuf::from(
477                scenario_config
478                    .artifacts_dir
479                    .unwrap_or_else(|| "artifacts".to_owned()),
480            ),
481            reporter,
482            current_step: std::cell::RefCell::new(None),
483            run_started: SystemTime::now()
484                .duration_since(UNIX_EPOCH)
485                .map_or(0, |d| d.as_secs()),
486            emit_run_events,
487        }
488    }
489
490    /// Emits an event; a sink failure degrades to a console warning so a
491    /// broken log file can never mask the run itself.
492    fn emit_event(&self, event: &TestEvent) {
493        if let Err(err) = self.reporter.emit(event) {
494            use std::io::Write as _;
495            let mut out = std::io::stderr().lock();
496            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
497        }
498    }
499
500    /// Returns a clone of the [`UsageTracker`] for reporting.
501    #[must_use]
502    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
503        Arc::clone(&self.usage)
504    }
505
506    /// Returns a reference to the [`BudgetTracker`].
507    #[must_use]
508    pub const fn budget_tracker(&self) -> &BudgetTracker {
509        &self.budgets
510    }
511
512    /// Executes all test groups in the scenario and returns a report.
513    ///
514    /// # Errors
515    ///
516    /// Returns an error if the browser fails to launch.
517    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
518    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
519        let mut report = RunReport::default();
520
521        if tests.is_empty() {
522            self.reporter.warn("No tests defined in scenario.");
523            if self.emit_run_events {
524                self.emit_event(&TestEvent::RunFinished {
525                    tests_passed: 0,
526                    tests_failed: 0,
527                    steps_passed: 0,
528                    steps_failed: 0,
529                    steps_skipped: 0,
530                    total_cost: 0.0,
531                    total_tokens: 0,
532                    total_calls: 0,
533                });
534            }
535            return Ok(report);
536        }
537
538        if self.emit_run_events {
539            self.emit_event(&TestEvent::RunStarted {
540                total_tests: tests.len() as u32,
541            });
542        }
543
544        let browser_headless = self.config.browser_headless.unwrap_or(true);
545
546        let launch_opts = LaunchOptions {
547            headless: browser_headless,
548            window_size: Some((self.viewport_width, self.viewport_height)),
549            sandbox: false,
550            // headless_chrome defaults this to 30s and shuts down the whole CDP
551            // connection when no messages arrive for that long. A scenario can
552            // easily exceed 30s of browser silence (slow LLM targeting/assertion
553            // calls, page waits, budget checks between steps), after which every
554            // remaining step fails with "Unable to make method calls because
555            // underlying connection is closed" — one quiet gap kills the run.
556            // Open-ended scenarios must own the connection for their full
557            // duration, so keep it alive for 6 hours.
558            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
559            ..LaunchOptions::default()
560        };
561
562        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
563        let tab = browser.new_tab().context("failed to open browser tab")?;
564        let _ = tab.set_default_timeout(self.timeout);
565
566        // Start MCP server if configured
567        #[cfg(feature = "mcp-server")]
568        if let Some(ref mcp_cfg) = self.config.mcp_server {
569            if mcp_cfg.enabled {
570                let port = mcp_cfg.port;
571                std::thread::spawn(move || {
572                    let _ = crate::mcp_server::start_mcp_server(port);
573                });
574            }
575        }
576        #[cfg(not(feature = "mcp-server"))]
577        if let Some(mcp_cfg) = &self.config.mcp_server {
578            if mcp_cfg.enabled {
579                self.reporter
580                    .warn("MCP server configured but 'mcp-server' feature not enabled");
581            }
582        }
583
584        // Start A2A agent server if configured
585        #[cfg(feature = "a2a-server")]
586        if let Some(ref a2a_cfg) = self.config.a2a_server {
587            if a2a_cfg.enabled {
588                let port = a2a_cfg.port;
589                tokio::spawn(crate::a2a_server::start_a2a_server(port));
590            }
591        }
592        #[cfg(not(feature = "a2a-server"))]
593        if let Some(a2a_cfg) = &self.config.a2a_server {
594            if a2a_cfg.enabled {
595                self.reporter
596                    .warn("A2A server configured but 'a2a-server' feature not enabled");
597            }
598        }
599
600        for test in tests {
601            self.emit_event(&TestEvent::TestStarted {
602                test: test.name.clone(),
603            });
604
605            self.usage.reset_per_test();
606
607            let test_started = Instant::now();
608            let test_result = self.run_test(test, &tab);
609            let duration_ms = test_started.elapsed().as_millis() as u64;
610            let usage = self.usage.current_test_snapshot();
611            self.usage.commit_test(&test.name);
612
613            self.emit_event(&TestEvent::TestFinished {
614                test: test.name.clone(),
615                passed: test_result.passed,
616                failed: test_result.failed,
617                skipped: test_result.skipped,
618                duration_ms,
619                cost: usage.total_cost,
620                tokens: usage.total_tokens,
621                calls: usage.total_calls,
622            });
623
624            if test_result.failed == 0 && test_result.total > 0 {
625                report.tests_passed += 1;
626            } else if test_result.total > 0 {
627                report.tests_failed += 1;
628            }
629
630            report.passed += test_result.passed;
631            report.failed += test_result.failed;
632            report.skipped += test_result.skipped;
633            report.details.extend(test_result.details);
634        }
635
636        let global = self.usage.global_snapshot();
637        if self.emit_run_events {
638            self.emit_event(&TestEvent::RunFinished {
639                tests_passed: report.tests_passed,
640                tests_failed: report.tests_failed,
641                steps_passed: report.passed,
642                steps_failed: report.failed,
643                steps_skipped: report.skipped,
644                total_cost: global.total_cost,
645                total_tokens: global.total_tokens,
646                total_calls: global.total_calls,
647            });
648        }
649
650        Ok(report)
651    }
652
653    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
654    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
655        let base_url = test
656            .base_url
657            .clone()
658            .or_else(|| self.config.base_url.clone())
659            .unwrap_or_else(crate::base_url);
660
661        // Per-test viewport override: switch the browser via CDP
662        // device-metrics emulation before this test runs.
663        let vw = test.viewport_width.unwrap_or(self.viewport_width);
664        let vh = test.viewport_height.unwrap_or(self.viewport_height);
665        if self.applied_viewport.get() != (vw, vh) {
666            self.apply_viewport(tab, vw, vh);
667            self.applied_viewport.set((vw, vh));
668        }
669
670        // Per-test isolation: every test starts from its own start_url
671        // (unless auto_navigate is disabled), so a test never inherits the
672        // previous test's page state.
673        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
674
675        let start_url = test
676            .start_url
677            .clone()
678            .or_else(|| self.config.start_url.clone())
679            .unwrap_or_else(|| "/dashboard".to_owned());
680
681        if auto_navigate {
682            let full_url = resolve_url(&start_url, &base_url);
683            self.reporter.debug(format!("auto-navigate: {full_url}"));
684            let _ = tab.navigate_to(&full_url);
685            let _ = tab.wait_until_navigated();
686            std::thread::sleep(Duration::from_secs(4));
687        }
688
689        let mut result = TestRunResult::default();
690
691        for (step_index, step) in test.steps.iter().enumerate() {
692            result.total += 1;
693
694            let wait_ms = match step {
695                TestStep::Navigate { wait_after_ms, .. }
696                | TestStep::Click { wait_after_ms, .. }
697                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
698                _ => None,
699            };
700
701            self.current_step
702                .replace(Some((test.name.clone(), step_index as u32)));
703            self.emit_event(&TestEvent::StepStarted {
704                test: test.name.clone(),
705                index: step_index as u32,
706                label: step_label(step),
707            });
708            let step_started = Instant::now();
709
710            let mut step_result = match step {
711                TestStep::Navigate { url, .. } => {
712                    let full_url = resolve_url(url, &base_url);
713                    run_navigate_step(&full_url, tab)
714                }
715                TestStep::Click {
716                    target,
717                    selector,
718                    endpoint,
719                    idempotent,
720                    ..
721                } => self.run_click(
722                    target,
723                    selector.as_deref(),
724                    endpoint.as_deref(),
725                    test.endpoint.as_deref(),
726                    *idempotent,
727                    tab,
728                ),
729                TestStep::Type {
730                    target,
731                    text,
732                    selector,
733                    endpoint,
734                    idempotent,
735                    ..
736                } => self.run_type(
737                    target,
738                    text,
739                    selector.as_deref(),
740                    endpoint.as_deref(),
741                    test.endpoint.as_deref(),
742                    *idempotent,
743                    tab,
744                ),
745                TestStep::Wait {
746                    target,
747                    selector,
748                    text,
749                    timeout_ms,
750                    endpoint,
751                    idempotent,
752                } => self.run_wait(
753                    target,
754                    selector.as_deref(),
755                    text.as_deref(),
756                    *timeout_ms,
757                    endpoint.as_deref(),
758                    test.endpoint.as_deref(),
759                    *idempotent,
760                    tab,
761                ),
762                TestStep::Assert {
763                    definition,
764                    preset,
765                    prompt,
766                    assert_text,
767                    endpoint,
768                    screenshot,
769                } => self.run_assert(
770                    definition.as_deref(),
771                    preset.as_deref(),
772                    prompt.as_deref(),
773                    assert_text.as_deref(),
774                    *screenshot,
775                    endpoint.as_deref(),
776                    test.endpoint.as_deref(),
777                    tab,
778                ),
779                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
780                TestStep::Agent {
781                    agent,
782                    task,
783                    definition,
784                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
785                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
786            };
787
788            // Failure diagnostics: capture the page state and a screenshot so
789            // CI logs say WHAT the page looked like when the step failed,
790            // instead of a bare "timed out: The event waited for never came".
791            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
792                let state = diagnostics::capture(tab);
793                let screenshot = diagnostics::save_screenshot(
794                    tab,
795                    &self.artifacts_dir,
796                    &test.name,
797                    &test.name,
798                    step_index,
799                    step_kind_label(step),
800                );
801                step_result.message = format!(
802                    "{base} — {excerpt}",
803                    base = step_result.message,
804                    excerpt = diagnostics::inline_excerpt(&state),
805                );
806                (Some(diagnostics::full_context(&state)), screenshot)
807            } else {
808                (None, None)
809            };
810
811            let duration_ms = step_started.elapsed().as_millis() as u64;
812            self.emit_event(&TestEvent::StepFinished {
813                test: test.name.clone(),
814                index: step_index as u32,
815                label: step_result.name.clone(),
816                status: step_result.status,
817                duration_ms,
818                message: step_result.message.clone(),
819                diagnostics: diagnostics_block,
820                screenshot: screenshot_path,
821            });
822            self.current_step.replace(None);
823
824            match step_result.status {
825                StepStatus::Passed => result.passed += 1,
826                StepStatus::Failed => result.failed += 1,
827                StepStatus::Skipped => result.skipped += 1,
828            }
829
830            // Fail fast: the first failed step ends the test and the
831            // remaining steps are reported as skipped (no LLM budget is
832            // burned asserting against a page that is already known broken).
833            if step_result.status == StepStatus::Failed
834                && !self.config.continue_on_failure
835                && step_index + 1 < test.steps.len()
836            {
837                self.emit_event(&TestEvent::Warning {
838                    message: format!(
839                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
840                        test.steps.len() - step_index - 1
841                    ),
842                });
843                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
844                    let skipped_index = step_index + 1 + offset;
845                    let label = step_label(skipped);
846                    result.total += 1;
847                    result.skipped += 1;
848                    self.emit_event(&TestEvent::StepStarted {
849                        test: test.name.clone(),
850                        index: skipped_index as u32,
851                        label: label.clone(),
852                    });
853                    self.emit_event(&TestEvent::StepFinished {
854                        test: test.name.clone(),
855                        index: skipped_index as u32,
856                        label,
857                        status: StepStatus::Skipped,
858                        duration_ms: 0,
859                        message: "skipped: previous step failed".into(),
860                        diagnostics: None,
861                        screenshot: None,
862                    });
863                    result.details.push(StepResult {
864                        name: step_label(skipped),
865                        status: StepStatus::Skipped,
866                        message: "skipped: previous step failed".into(),
867                    });
868                }
869                result.details.push(step_result);
870                return result;
871            }
872
873            // Check per-test budget after each step
874            let test_usage = self.usage.current_test_snapshot();
875            let global_usage = self.usage.global_snapshot();
876            let budget_status = self.budgets.check_all(
877                &test.name,
878                &test_usage,
879                &global_usage,
880                test.budget.as_ref(),
881            );
882            match budget_status {
883                BudgetStatus::HardExceeded { message, .. } => {
884                    self.emit_event(&TestEvent::Warning {
885                        message: format!("budget exceeded: {message}"),
886                    });
887                    result.details.push(StepResult {
888                        name: "[budget]".into(),
889                        status: StepStatus::Failed,
890                        message,
891                    });
892                    result.failed += 1;
893                    return result;
894                }
895                BudgetStatus::SoftExceeded { message, .. } => {
896                    self.emit_event(&TestEvent::Warning {
897                        message: format!("budget warning: {message}"),
898                    });
899                }
900                BudgetStatus::Ok => {}
901            }
902
903            if let Some(ms) = wait_ms {
904                std::thread::sleep(Duration::from_millis(ms));
905            }
906
907            result.details.push(step_result);
908        }
909
910        result
911    }
912
913    /// Applies a viewport size to the current tab via CDP
914    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
915    /// overrides and the viewport matrix. The initial window size set at
916    /// browser launch is replaced by emulation; failures are logged but
917    /// do not fail the test (a mismatched viewport only weakens coverage).
918    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
919        use headless_chrome::protocol::cdp::Emulation;
920        let _ = self;
921        let params = Emulation::SetDeviceMetricsOverride {
922            width,
923            height,
924            device_scale_factor: 1.0,
925            mobile: false,
926            scale: None,
927            screen_width: Some(width),
928            screen_height: Some(height),
929            position_x: None,
930            position_y: None,
931            dont_set_visible_size: None,
932            screen_orientation: None,
933            viewport: None,
934            display_feature: None,
935            device_posture: None,
936        };
937        self.reporter.debug(format!("viewport: {width}x{height}"));
938        if let Err(e) = tab.call_method(params) {
939            self.reporter
940                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
941        }
942    }
943
944    /// Height (px) covered by assert-step screenshots: the configured
945    /// `screenshot_max_height` (absolute px or viewport multiple, default
946    /// `"20x"`) resolved against the currently applied viewport, raised to
947    /// at least the viewport height so the visible screen is always fully
948    /// included. The capture is split into viewport-tall tiles, so this
949    /// value bounds total coverage (and hence the number of image parts).
950    #[must_use]
951    fn screenshot_height_cap(&self) -> u32 {
952        let viewport_height = self.current_viewport_height();
953        let cap = self
954            .config
955            .screenshot_max_height
956            .as_ref()
957            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
958        cap.max(viewport_height)
959    }
960
961    /// Height of the viewport currently emulated in the browser (falling
962    /// back to the configured default before any emulation was applied).
963    #[must_use]
964    const fn current_viewport_height(&self) -> u32 {
965        let (_, height) = self.applied_viewport.get();
966        if height > 0 {
967            height
968        } else {
969            self.viewport_height
970        }
971    }
972
973    // ── step handlers ───────────────────────────────────────────────────
974
975    #[allow(clippy::too_many_lines)]
976    fn run_click(
977        &self,
978        target: &str,
979        selector_override: Option<&str>,
980        step_endpoint: Option<&str>,
981        test_endpoint: Option<&str>,
982        idempotent: bool,
983        tab: &Tab,
984    ) -> StepResult {
985        let name = format!("[click] {target}");
986        let selector = match self.resolve_selector(
987            selector_override,
988            target,
989            step_endpoint,
990            test_endpoint,
991            tab,
992        ) {
993            Ok(s) => s,
994            Err(msg) => {
995                if idempotent {
996                    return StepResult {
997                        name,
998                        status: StepStatus::Skipped,
999                        message: format!("skipped (idempotent): no target found — {msg}"),
1000                    };
1001                }
1002                return StepResult {
1003                    name,
1004                    status: StepStatus::Failed,
1005                    message: msg,
1006                };
1007            }
1008        };
1009
1010        // Idempotent steps probe briefly: a missing target means the
1011        // action was already done / not applicable (e.g. an
1012        // already-authenticated session), and skipping is the success
1013        // path, not a failure.
1014        let probe_secs = if idempotent { 5 } else { 10 };
1015        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1016            Ok(element) => match element.click() {
1017                Ok(_) => StepResult {
1018                    name,
1019                    status: StepStatus::Passed,
1020                    message: format!("clicked {selector}"),
1021                },
1022                Err(e) => StepResult {
1023                    name,
1024                    status: StepStatus::Failed,
1025                    message: format!("click failed on {selector}: {e}"),
1026                },
1027            },
1028            Err(e) if idempotent => StepResult {
1029                name,
1030                status: StepStatus::Skipped,
1031                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1032            },
1033            Err(e) => StepResult {
1034                name,
1035                status: StepStatus::Failed,
1036                message: format!("element {selector} not found: {e}"),
1037            },
1038        }
1039    }
1040
1041    #[allow(clippy::too_many_arguments)]
1042    fn run_type(
1043        &self,
1044        target: &str,
1045        text: &str,
1046        selector_override: Option<&str>,
1047        step_endpoint: Option<&str>,
1048        test_endpoint: Option<&str>,
1049        idempotent: bool,
1050        tab: &Tab,
1051    ) -> StepResult {
1052        let name = format!("[type] {target}");
1053        let selector = match self.resolve_selector(
1054            selector_override,
1055            target,
1056            step_endpoint,
1057            test_endpoint,
1058            tab,
1059        ) {
1060            Ok(s) => s,
1061            Err(msg) => {
1062                if idempotent {
1063                    return StepResult {
1064                        name,
1065                        status: StepStatus::Skipped,
1066                        message: format!("skipped (idempotent): no target found — {msg}"),
1067                    };
1068                }
1069                return StepResult {
1070                    name,
1071                    status: StepStatus::Failed,
1072                    message: msg,
1073                };
1074            }
1075        };
1076
1077        let probe_secs = if idempotent { 5 } else { 10 };
1078        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1079            Ok(element) => {
1080                if let Err(e) = element.click() {
1081                    return StepResult {
1082                        name,
1083                        status: StepStatus::Failed,
1084                        message: format!("click to focus {selector} failed: {e}"),
1085                    };
1086                }
1087
1088                let js = format!(
1089                    "document.querySelector('{}').value = '';",
1090                    selector.replace('\'', "\\'")
1091                );
1092                let _ = tab.evaluate(&js, false);
1093
1094                match element.type_into(text) {
1095                    Ok(_) => StepResult {
1096                        name,
1097                        status: StepStatus::Passed,
1098                        message: format!("typed {text:?} into {selector}"),
1099                    },
1100                    Err(e) => StepResult {
1101                        name,
1102                        status: StepStatus::Failed,
1103                        message: format!("type into {selector} failed: {e}"),
1104                    },
1105                }
1106            }
1107            Err(e) if idempotent => StepResult {
1108                name,
1109                status: StepStatus::Skipped,
1110                message: format!("skipped (idempotent): element {selector} not present — {e}"),
1111            },
1112            Err(e) => StepResult {
1113                name,
1114                status: StepStatus::Failed,
1115                message: format!("element {selector} not found: {e}"),
1116            },
1117        }
1118    }
1119
1120    #[allow(clippy::too_many_arguments)]
1121    #[allow(clippy::too_many_lines)]
1122    fn run_wait(
1123        &self,
1124        target: &str,
1125        selector_override: Option<&str>,
1126        text: Option<&str>,
1127        timeout_ms: Option<u64>,
1128        step_endpoint: Option<&str>,
1129        test_endpoint: Option<&str>,
1130        idempotent: bool,
1131        tab: &Tab,
1132    ) -> StepResult {
1133        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1134        let step_name = format!("[wait] {target}");
1135
1136        // Resolve an explicit selector only (text-only waits are LLM-free).
1137        let selector = match selector_override {
1138            Some(s) => Some(s.to_owned()),
1139            None if text.is_some() => None,
1140            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1141                Ok(s) => Some(s),
1142                Err(msg) => {
1143                    if idempotent {
1144                        return StepResult {
1145                            name: step_name,
1146                            status: StepStatus::Skipped,
1147                            message: format!("skipped (idempotent): no target found — {msg}"),
1148                        };
1149                    }
1150                    return StepResult {
1151                        name: step_name,
1152                        status: StepStatus::Failed,
1153                        message: msg,
1154                    };
1155                }
1156            },
1157        };
1158
1159        if text.is_some() {
1160            let sel_js = selector
1161                .as_deref()
1162                .map(crate::selectors::selector_matches_js);
1163            let text_js = text.map(|t| {
1164                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1165                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1166            });
1167
1168            let deadline = Instant::now() + timeout;
1169            loop {
1170                let sel_ok = sel_js
1171                    .as_ref()
1172                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1173                let text_ok = text_js
1174                    .as_ref()
1175                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1176                if sel_ok && text_ok {
1177                    let mut what = Vec::new();
1178                    if let Some(sel) = &selector {
1179                        what.push(format!("found {sel}"));
1180                    }
1181                    if let Some(t) = text {
1182                        what.push(format!("text {t:?} visible"));
1183                    }
1184                    return StepResult {
1185                        name: step_name,
1186                        status: StepStatus::Passed,
1187                        message: what.join(" and "),
1188                    };
1189                }
1190                if Instant::now() >= deadline {
1191                    let mut what = Vec::new();
1192                    if let Some(sel) = &selector {
1193                        what.push(sel.clone());
1194                    }
1195                    if let Some(t) = text {
1196                        what.push(format!("text {t:?}"));
1197                    }
1198                    let message = format!(
1199                        "wait for {} timed out after {}ms: the event waited for never came",
1200                        what.join(" / "),
1201                        timeout.as_millis(),
1202                    );
1203                    if idempotent {
1204                        return StepResult {
1205                            name: step_name,
1206                            status: StepStatus::Skipped,
1207                            message: format!("skipped (idempotent): {message}"),
1208                        };
1209                    }
1210                    return StepResult {
1211                        name: step_name,
1212                        status: StepStatus::Failed,
1213                        message,
1214                    };
1215                }
1216                std::thread::sleep(Duration::from_millis(250));
1217            }
1218        }
1219
1220        match selector.as_deref() {
1221            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1222                Ok(_) => StepResult {
1223                    name: step_name,
1224                    status: StepStatus::Passed,
1225                    message: format!("found {sel}"),
1226                },
1227                Err(e) if idempotent => StepResult {
1228                    name: step_name,
1229                    status: StepStatus::Skipped,
1230                    message: format!(
1231                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1232                        timeout.as_millis()
1233                    ),
1234                },
1235                Err(e) => StepResult {
1236                    name: step_name,
1237                    status: StepStatus::Failed,
1238                    message: format!(
1239                        "wait for {sel} timed out after {}ms: {e}",
1240                        timeout.as_millis()
1241                    ),
1242                },
1243            },
1244            None => StepResult {
1245                name: step_name,
1246                status: StepStatus::Failed,
1247                message: "wait step has neither selector nor text".into(),
1248            },
1249        }
1250    }
1251
1252    #[allow(clippy::too_many_arguments)]
1253    fn run_assert(
1254        &self,
1255        definition: Option<&str>,
1256        preset: Option<&str>,
1257        prompt: Option<&str>,
1258        assert_text: Option<&str>,
1259        screenshot: bool,
1260        step_endpoint: Option<&str>,
1261        test_endpoint: Option<&str>,
1262        tab: &Tab,
1263    ) -> StepResult {
1264        std::thread::sleep(Duration::from_millis(500));
1265
1266        let page_content = get_page_text(tab);
1267
1268        // Vision attach: capture the full page once per assert step and
1269        // split it into viewport-tall tiles (the total coverage is bounded
1270        // by the configured height cap so vision tokens stay sane). All
1271        // tile data URLs are handed to the preset/prompt evaluation below.
1272        let image: Option<Vec<String>> = if screenshot {
1273            let endpoint = self
1274                .endpoints
1275                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1276            if !endpoint.vision {
1277                return StepResult {
1278                    name: "[assert]".into(),
1279                    status: StepStatus::Failed,
1280                    message: format!(
1281                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1282                        name = endpoint.name
1283                    ),
1284                };
1285            }
1286            match crate::vision::capture_screenshot_data_urls(
1287                tab,
1288                self.config
1289                    .screenshot_max_dimension
1290                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1291                self.screenshot_height_cap(),
1292                self.current_viewport_height(),
1293            ) {
1294                Ok(urls) => Some(urls),
1295                Err(e) => {
1296                    return StepResult {
1297                        name: "[assert]".into(),
1298                        status: StepStatus::Failed,
1299                        message: format!("screenshot capture failed: {e}"),
1300                    };
1301                }
1302            }
1303        } else {
1304            None
1305        };
1306
1307        if let Some(def_name) = definition {
1308            if let Some(def) = self.definitions.get(def_name) {
1309                return self.run_assert_def(
1310                    def,
1311                    &page_content,
1312                    image.as_deref(),
1313                    step_endpoint,
1314                    test_endpoint,
1315                    tab,
1316                );
1317            }
1318            return StepResult {
1319                name: format!("[assert] {def_name}"),
1320                status: StepStatus::Failed,
1321                message: format!("definition '{def_name}' not found"),
1322            };
1323        }
1324
1325        if let Some(preset_name) = preset {
1326            // Deterministic DOM layout scan — runs JS in the browser and
1327            // never calls the LLM (free, fast, no pixel budget).
1328            if preset_name == "layout_no_issues" {
1329                return self.run_layout_preset(tab);
1330            }
1331            return self.run_preset(
1332                preset_name,
1333                assert_text,
1334                &page_content,
1335                image.as_deref(),
1336                step_endpoint,
1337                test_endpoint,
1338            );
1339        }
1340
1341        if let Some(prompt_text) = prompt {
1342            return self.run_custom(
1343                prompt_text,
1344                &page_content,
1345                image.as_deref(),
1346                step_endpoint,
1347                test_endpoint,
1348            );
1349        }
1350
1351        StepResult {
1352            name: "[assert]".into(),
1353            status: StepStatus::Skipped,
1354            message: "no definition, preset, or prompt specified".into(),
1355        }
1356    }
1357
1358    fn run_assert_def(
1359        &self,
1360        def: &AssertDefinition,
1361        page_content: &PageContent,
1362        image: Option<&[String]>,
1363        step_endpoint: Option<&str>,
1364        test_endpoint: Option<&str>,
1365        tab: &Tab,
1366    ) -> StepResult {
1367        // Agent-based definition: delegate to an A2A agent
1368        if let Some(ref agent) = def.agent {
1369            if image.is_some() {
1370                return StepResult {
1371                    name: format!("[assert] {}", def.name),
1372                    status: StepStatus::Failed,
1373                    message: "agent-backed assertions do not support screenshots".into(),
1374                };
1375            }
1376            let task = def
1377                .task_template
1378                .as_deref()
1379                .unwrap_or("Evaluate the assertion")
1380                .replace("{url}", &page_content.url)
1381                .replace("{title}", &page_content.title)
1382                .replace("{content}", &page_content.body_text)
1383                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1384
1385            return self.run_agent_step(agent, &task, &def.name);
1386        }
1387
1388        // Custom preset: system + user_template provided in the definition
1389        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1390            return self.run_custom_preset(
1391                &def.name,
1392                system,
1393                template,
1394                def.assert_text.as_deref(),
1395                page_content,
1396                image,
1397                step_endpoint,
1398                test_endpoint,
1399            );
1400        }
1401
1402        def.preset.as_ref().map_or_else(
1403            || {
1404                def.prompt.as_ref().map_or_else(
1405                    || StepResult {
1406                        name: format!("[assert] {}", def.name),
1407                        status: StepStatus::Failed,
1408                        message: "definition has no preset, prompt, or system+user_template".into(),
1409                    },
1410                    |prompt| {
1411                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1412                    },
1413                )
1414            },
1415            |preset_name| {
1416                if preset_name == "layout_no_issues" {
1417                    return self.run_layout_preset(tab);
1418                }
1419                self.run_preset(
1420                    preset_name,
1421                    def.assert_text.as_deref(),
1422                    page_content,
1423                    image,
1424                    step_endpoint,
1425                    test_endpoint,
1426                )
1427            },
1428        )
1429    }
1430
1431    #[allow(clippy::too_many_arguments)]
1432    fn run_custom_preset(
1433        &self,
1434        name: &str,
1435        system: &str,
1436        template: &str,
1437        assert_text: Option<&str>,
1438        page_content: &PageContent,
1439        image: Option<&[String]>,
1440        step_endpoint: Option<&str>,
1441        test_endpoint: Option<&str>,
1442    ) -> StepResult {
1443        let user_prompt = template
1444            .replace("{url}", &page_content.url)
1445            .replace("{title}", &page_content.title)
1446            .replace("{content}", &page_content.body_text)
1447            .replace("{expected_text}", assert_text.unwrap_or(""))
1448            .replace("{description}", "");
1449
1450        // Custom preset definitions frequently forget the {content}
1451        // placeholder — without it the LLM has no page to evaluate and
1452        // answers "I can't determine that without seeing the page". Always
1453        // append the page context unless the template already references it.
1454        let user_prompt = if template.contains("{content}") {
1455            user_prompt
1456        } else {
1457            format!(
1458                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1459                url = page_content.url,
1460                title = page_content.title,
1461                content = page_content.body_text,
1462            )
1463        };
1464
1465        self.reporter
1466            .debug(format!("assert: {name} (custom preset)"));
1467
1468        let chain = self
1469            .endpoints
1470            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1471        let sys = system.to_owned();
1472
1473        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1474
1475        response.map_or_else(
1476            |e| StepResult {
1477                name: format!("[assert] {name}"),
1478                status: StepStatus::Failed,
1479                message: format!("LLM assertion call failed: {e}"),
1480            },
1481            |(lr, idx)| {
1482                self.usage.record_llm_call(
1483                    &chain[idx].name,
1484                    chain[idx],
1485                    lr.usage.prompt_tokens,
1486                    lr.usage.completion_tokens,
1487                );
1488                let content_lower = lr.content.to_lowercase().trim().to_owned();
1489                if content_lower.starts_with("pass") {
1490                    StepResult {
1491                        name: format!("[assert] {name}"),
1492                        status: StepStatus::Passed,
1493                        message: "PASS".into(),
1494                    }
1495                } else {
1496                    StepResult {
1497                        name: format!("[assert] {name}"),
1498                        status: StepStatus::Failed,
1499                        message: lr.content,
1500                    }
1501                }
1502            },
1503        )
1504    }
1505
1506    fn run_preset(
1507        &self,
1508        preset_name: &str,
1509        assert_text: Option<&str>,
1510        page_content: &PageContent,
1511        image: Option<&[String]>,
1512        step_endpoint: Option<&str>,
1513        test_endpoint: Option<&str>,
1514    ) -> StepResult {
1515        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1516            return StepResult {
1517                name: format!("[assert] {preset_name}"),
1518                status: StepStatus::Failed,
1519                message: format!("unknown assertion preset: {preset_name}"),
1520            };
1521        };
1522        if preset_name.starts_with("visual_") && image.is_none() {
1523            return StepResult {
1524                name: format!("[assert] {preset_name}"),
1525                status: StepStatus::Failed,
1526                message: format!(
1527                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1528                ),
1529            };
1530        }
1531
1532        let user_prompt = preset
1533            .user_template
1534            .replace("{url}", &page_content.url)
1535            .replace("{title}", &page_content.title)
1536            .replace("{content}", &page_content.body_text)
1537            .replace("{expected_text}", assert_text.unwrap_or(""))
1538            .replace("{description}", "");
1539
1540        // Same safety net as custom presets: never let the LLM answer with
1541        // no page context at all.
1542        let user_prompt = if preset.user_template.contains("{content}") {
1543            user_prompt
1544        } else {
1545            format!(
1546                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1547                url = page_content.url,
1548                title = page_content.title,
1549                content = page_content.body_text,
1550            )
1551        };
1552
1553        self.reporter.debug(format!("assert: {preset_name}"));
1554
1555        let chain = self
1556            .endpoints
1557            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1558        let sys = preset.system.to_owned();
1559
1560        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1561
1562        response.map_or_else(
1563            |e| StepResult {
1564                name: format!("[assert] {preset_name}"),
1565                status: StepStatus::Failed,
1566                message: format!("LLM assertion call failed: {e}"),
1567            },
1568            |(lr, idx)| {
1569                self.usage.record_llm_call(
1570                    &chain[idx].name,
1571                    chain[idx],
1572                    lr.usage.prompt_tokens,
1573                    lr.usage.completion_tokens,
1574                );
1575                let content_lower = lr.content.to_lowercase().trim().to_owned();
1576                if content_lower.starts_with("pass") {
1577                    StepResult {
1578                        name: format!("[assert] {preset_name}"),
1579                        status: StepStatus::Passed,
1580                        message: "PASS".into(),
1581                    }
1582                } else {
1583                    StepResult {
1584                        name: format!("[assert] {preset_name}"),
1585                        status: StepStatus::Failed,
1586                        message: lr.content,
1587                    }
1588                }
1589            },
1590        )
1591    }
1592
1593    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1594    ///
1595    /// Evaluates the layout-scan JS in the page and fails with the list of
1596    /// detected issues: horizontal page overflow, visible elements sticking
1597    /// out of the viewport, text clipped by `overflow: hidden` containers,
1598    /// and interactive elements covered by other elements. No LLM call —
1599    /// checks are geometry-based so the check is free, deterministic, and
1600    /// safe to run on every page × viewport variant.
1601    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1602        let name = "[assert] layout_no_issues".to_owned();
1603        self.reporter
1604            .debug("assert: layout_no_issues (DOM layout scan)");
1605        let js = LAYOUT_SCAN_JS.replace(
1606            "__IGNORE_CLASSES__",
1607            &serde_json::to_string(&self.config.layout_ignore_classes)
1608                .unwrap_or_else(|_| "[]".to_owned()),
1609        );
1610        let result = tab.evaluate(&js, false);
1611        let json_str = match result {
1612            Ok(r) => r
1613                .value
1614                .as_ref()
1615                .and_then(|v| v.as_str().map(String::from))
1616                .unwrap_or_else(|| "[]".to_owned()),
1617            Err(e) => {
1618                return StepResult {
1619                    name,
1620                    status: StepStatus::Failed,
1621                    message: format!("layout scan JS failed: {e}"),
1622                };
1623            }
1624        };
1625        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1626        if issues.is_empty() {
1627            return StepResult {
1628                name,
1629                status: StepStatus::Passed,
1630                message: "PASS — no layout defects detected".into(),
1631            };
1632        }
1633        let mut lines: Vec<String> = issues
1634            .iter()
1635            .take(10)
1636            .map(|i| {
1637                format!(
1638                    "- [{type_}] {element}: {detail}",
1639                    type_ = i.issue_type,
1640                    element = i.element,
1641                    detail = i.detail
1642                )
1643            })
1644            .collect();
1645        if issues.len() > 10 {
1646            lines.push(format!("- … and {} more", issues.len() - 10));
1647        }
1648        StepResult {
1649            name,
1650            status: StepStatus::Failed,
1651            message: format!(
1652                "FAIL — {} layout defect(s) detected:\n{}",
1653                issues.len(),
1654                lines.join("\n")
1655            ),
1656        }
1657    }
1658
1659    fn run_custom(
1660        &self,
1661        prompt: &str,
1662        page_content: &PageContent,
1663        image: Option<&[String]>,
1664        step_endpoint: Option<&str>,
1665        test_endpoint: Option<&str>,
1666    ) -> StepResult {
1667        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1668
1669        let mut user = format!(
1670            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1671            url = page_content.url,
1672            title = page_content.title,
1673            content = page_content.body_text,
1674        );
1675        if image.is_some() {
1676            user.push_str(
1677                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1678            );
1679        }
1680
1681        self.reporter.debug("custom assert");
1682
1683        let chain = self
1684            .endpoints
1685            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1686        let sys = system.to_owned();
1687
1688        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1689
1690        response.map_or_else(
1691            |e| StepResult {
1692                name: "[assert] custom".into(),
1693                status: StepStatus::Failed,
1694                message: format!("LLM assertion call failed: {e}"),
1695            },
1696            |(lr, idx)| {
1697                self.usage.record_llm_call(
1698                    &chain[idx].name,
1699                    chain[idx],
1700                    lr.usage.prompt_tokens,
1701                    lr.usage.completion_tokens,
1702                );
1703                let content_lower = lr.content.to_lowercase().trim().to_owned();
1704                if content_lower.starts_with("pass") {
1705                    StepResult {
1706                        name: "[assert] custom".into(),
1707                        status: StepStatus::Passed,
1708                        message: "PASS".into(),
1709                    }
1710                } else {
1711                    StepResult {
1712                        name: "[assert] custom".into(),
1713                        status: StepStatus::Failed,
1714                        message: lr.content,
1715                    }
1716                }
1717            },
1718        )
1719    }
1720
1721    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1722        let path = path.unwrap_or("screenshot.png");
1723
1724        match tab.capture_screenshot(
1725            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1726            None,
1727            None,
1728            true,
1729        ) {
1730            Ok(data) => {
1731                if let Err(e) = std::fs::write(path, &data) {
1732                    return StepResult {
1733                        name: format!("[screenshot] {path}"),
1734                        status: StepStatus::Failed,
1735                        message: format!("failed to write screenshot: {e}"),
1736                    };
1737                }
1738                StepResult {
1739                    name: format!("[screenshot] {path}"),
1740                    status: StepStatus::Passed,
1741                    message: format!("saved to {path}"),
1742                }
1743            }
1744            Err(e) => StepResult {
1745                name: format!("[screenshot] {path}"),
1746                status: StepStatus::Failed,
1747                message: format!("screenshot failed: {e}"),
1748            },
1749        }
1750    }
1751
1752    /// Runs an A2A agent step.
1753    #[allow(clippy::literal_string_with_formatting_args)]
1754    fn run_agent(
1755        &self,
1756        agent_name: &str,
1757        task: &str,
1758        definition: Option<&str>,
1759        _test_endpoint: Option<&str>,
1760    ) -> StepResult {
1761        // If a definition is specified, look up the task template
1762        let resolved_task = if let Some(def_name) = definition {
1763            if let Some(def) = self.definitions.get(def_name) {
1764                let tmpl = def.task_template.as_deref().unwrap_or(task);
1765                tmpl.replace("{task}", task)
1766            } else {
1767                return StepResult {
1768                    name: format!("[agent] {def_name}"),
1769                    status: StepStatus::Failed,
1770                    message: format!("definition '{def_name}' not found"),
1771                };
1772            }
1773        } else {
1774            task.to_owned()
1775        };
1776
1777        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1778    }
1779
1780    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1781        let Some(ep) = self.endpoints.get(agent_name) else {
1782            return StepResult {
1783                name: format!("[agent] {display_name}"),
1784                status: StepStatus::Failed,
1785                message: format!("agent endpoint '{agent_name}' not found"),
1786            };
1787        };
1788
1789        if ep.url.is_empty() {
1790            return StepResult {
1791                name: format!("[agent] {display_name}"),
1792                status: StepStatus::Failed,
1793                message: format!("agent endpoint '{agent_name}' has no URL"),
1794            };
1795        }
1796
1797        self.reporter.debug(format!("agent {agent_name}: {task}"));
1798
1799        let url = ep.url.clone();
1800        let client = A2aClient::new(&url, self.timeout);
1801        let task_clone = task.to_owned();
1802
1803        let response = std::thread::spawn(move || {
1804            let rt = tokio::runtime::Builder::new_current_thread()
1805                .enable_all()
1806                .build()
1807                .unwrap();
1808            rt.block_on(client.send_task(&task_clone))
1809        })
1810        .join()
1811        .unwrap();
1812
1813        // Record the flat-cost call
1814        self.usage.record_flat_call(agent_name, ep);
1815
1816        match response {
1817            Ok(text) => {
1818                let clean = text.trim().to_owned();
1819                let lower = clean.to_lowercase();
1820                if lower.starts_with("pass") {
1821                    StepResult {
1822                        name: format!("[agent] {display_name}"),
1823                        status: StepStatus::Passed,
1824                        message: format!("PASS: {clean}"),
1825                    }
1826                } else if lower.starts_with("fail") {
1827                    StepResult {
1828                        name: format!("[agent] {display_name}"),
1829                        status: StepStatus::Failed,
1830                        message: clean,
1831                    }
1832                } else {
1833                    StepResult {
1834                        name: format!("[agent] {display_name}"),
1835                        status: StepStatus::Passed,
1836                        message: format!("response: {clean}"),
1837                    }
1838                }
1839            }
1840            Err(e) => StepResult {
1841                name: format!("[agent] {display_name}"),
1842                status: StepStatus::Failed,
1843                message: format!("agent call failed: {e}"),
1844            },
1845        }
1846    }
1847
1848    /// Runs an MCP tool call step.
1849    fn run_mcp(
1850        &self,
1851        server_name: &str,
1852        tool_name: &str,
1853        args: Option<&serde_json::Value>,
1854    ) -> StepResult {
1855        let Some(ep) = self.endpoints.get(server_name) else {
1856            return StepResult {
1857                name: format!("[mcp] {server_name}:{tool_name}"),
1858                status: StepStatus::Failed,
1859                message: format!("MCP server endpoint '{server_name}' not found"),
1860            };
1861        };
1862
1863        let cmd = ep.command.as_deref().unwrap_or("");
1864        if cmd.is_empty() {
1865            return StepResult {
1866                name: format!("[mcp] {server_name}:{tool_name}"),
1867                status: StepStatus::Failed,
1868                message: format!("MCP server '{server_name}' has no command configured"),
1869            };
1870        }
1871
1872        self.reporter
1873            .debug(format!("mcp {server_name} {tool_name}"));
1874
1875        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1876
1877        let command = cmd.to_owned();
1878        let args_vec = ep.args.clone();
1879        let tool = tool_name.to_owned();
1880
1881        let response = std::thread::spawn(move || {
1882            let mut mcp_client =
1883                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1884            mcp_client
1885                .call_tool(&tool, &args_val)
1886                .map_err(|e| e.to_string())
1887        })
1888        .join()
1889        .unwrap();
1890
1891        // Record the flat-cost call
1892        self.usage.record_flat_call(server_name, ep);
1893
1894        match response {
1895            Ok(result) => {
1896                if result.isError {
1897                    StepResult {
1898                        name: format!("[mcp] {server_name}:{tool_name}"),
1899                        status: StepStatus::Failed,
1900                        message: result.to_string(),
1901                    }
1902                } else {
1903                    StepResult {
1904                        name: format!("[mcp] {server_name}:{tool_name}"),
1905                        status: StepStatus::Passed,
1906                        message: result.to_string(),
1907                    }
1908                }
1909            }
1910            Err(e) => StepResult {
1911                name: format!("[mcp] {server_name}:{tool_name}"),
1912                status: StepStatus::Failed,
1913                message: format!("MCP call failed: {e}"),
1914            },
1915        }
1916    }
1917
1918    // ── helpers ──────────────────────────────────────────────────────────
1919
1920    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1921    /// the runner's default LLM config for any unset fields.
1922    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1923        LlmConfig {
1924            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
1925                // Bedrock builds its endpoint from the resolved AWS region
1926                // when no URL is given — never inherit the default LLM URL.
1927                endpoint.url.clone()
1928            } else if endpoint.url.is_empty() {
1929                self.llm.url.clone()
1930            } else {
1931                endpoint.url.clone()
1932            },
1933            model: endpoint
1934                .model
1935                .clone()
1936                .unwrap_or_else(|| self.llm.model.clone()),
1937            api_key: endpoint
1938                .api_key
1939                .clone()
1940                .or_else(|| self.llm.api_key.clone()),
1941            headers: if endpoint.headers.is_empty() {
1942                self.llm.headers.clone()
1943            } else {
1944                endpoint.headers.clone()
1945            },
1946            timeout: self.llm.timeout,
1947            temperature: self.llm.temperature,
1948            thinking: self.llm.thinking,
1949            model_params: self.llm.model_params.clone(),
1950            max_attempts: endpoint.max_attempts.max(1),
1951            provider: endpoint.provider,
1952            deployment: endpoint.deployment.clone(),
1953            api_version: endpoint.api_version.clone(),
1954            auth: endpoint.auth.clone(),
1955            header_commands: endpoint.header_commands.clone(),
1956            aws: endpoint.aws.clone(),
1957        }
1958    }
1959
1960    /// Runs a single LLM call against an ordered endpoint chain (primary +
1961    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
1962    /// the first endpoint that answers wins. Returns the response together
1963    /// with the chain index of the answering endpoint (0 = primary) so the
1964    /// caller can attribute usage to the correct endpoint.
1965    ///
1966    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
1967    /// shows duration, tokens, cost and the answering endpoint per call.
1968    /// Run-level context handed to every LLM call as the FIRST block of the
1969    /// user message. Contents that are stable for the whole run ("run
1970    /// started", "target site") come first so upstream provider prefix
1971    /// caching stays effective; the current time is the last line because
1972    /// it changes on every call.
1973    fn run_context(&self) -> String {
1974        let mut parts = vec![
1975            "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
1976            "================================================================".into(),
1977            format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
1978        ];
1979        if let Some(base) = self.config.base_url.as_deref() {
1980            parts.push(format!("Target site: {base}"));
1981        }
1982        let now = SystemTime::now()
1983            .duration_since(UNIX_EPOCH)
1984            .map_or(0, |d| d.as_secs());
1985        parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
1986        parts.join("\n")
1987    }
1988
1989    #[allow(clippy::cast_possible_truncation)]
1990    fn llm_call_chain(
1991        &self,
1992        chain: &[&ResolvedEndpoint],
1993        system: &str,
1994        user: &str,
1995        image: Option<&[String]>,
1996        purpose: &str,
1997    ) -> Result<(crate::costs::LlmResponse, usize), String> {
1998        if chain.is_empty() {
1999            return Err("empty LLM endpoint chain".into());
2000        }
2001        let primary = self.build_llm_for_endpoint(chain[0]);
2002        let fallbacks: Vec<LlmConfig> = chain[1..]
2003            .iter()
2004            .map(|e| self.build_llm_for_endpoint(e))
2005            .collect();
2006
2007        let (test, index) = self
2008            .current_step
2009            .borrow()
2010            .as_ref()
2011            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
2012        let primary_endpoint = chain[0].name.clone();
2013        let primary_model = primary.model.clone();
2014        self.emit_event(&TestEvent::LlmCallStarted {
2015            test: test.clone(),
2016            index,
2017            endpoint: primary_endpoint.clone(),
2018            model: primary_model.clone(),
2019            purpose: purpose.to_owned(),
2020        });
2021
2022        let started = Instant::now();
2023        let sys = system.to_owned();
2024        let context = self.run_context();
2025        let user = if context.is_empty() {
2026            user.to_owned()
2027        } else {
2028            format!("{context}\n\n{user}")
2029        };
2030        let image = image.map(<[String]>::to_vec);
2031
2032        let result = std::thread::spawn(move || {
2033            let rt = tokio::runtime::Builder::new_current_thread()
2034                .enable_all()
2035                .build()
2036                .unwrap();
2037            let call = async {
2038                match image.as_deref() {
2039                    Some(img) => {
2040                        llm_chat_vision_with_usage_chain(
2041                            &primary,
2042                            &fallbacks,
2043                            &sys,
2044                            &user,
2045                            Some(img),
2046                        )
2047                        .await
2048                    }
2049                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2050                }
2051            };
2052            rt.block_on(call)
2053        })
2054        .join()
2055        .unwrap();
2056
2057        let duration_ms = started.elapsed().as_millis() as u64;
2058        match result {
2059            Ok((lr, idx)) => {
2060                let cost = calculate_llm_cost(
2061                    chain[idx],
2062                    lr.usage.prompt_tokens,
2063                    lr.usage.completion_tokens,
2064                );
2065                let answering = chain[idx].name.clone();
2066                let model = chain[idx]
2067                    .model
2068                    .clone()
2069                    .unwrap_or_else(|| primary_model.clone());
2070                self.emit_event(&TestEvent::LlmCallFinished {
2071                    test,
2072                    index,
2073                    endpoint: answering,
2074                    model,
2075                    purpose: purpose.to_owned(),
2076                    ok: true,
2077                    duration_ms,
2078                    input_tokens: lr.usage.prompt_tokens,
2079                    output_tokens: lr.usage.completion_tokens,
2080                    cost,
2081                    error: None,
2082                });
2083                Ok((lr, idx))
2084            }
2085            Err(e) => {
2086                self.emit_event(&TestEvent::LlmCallFinished {
2087                    test,
2088                    index,
2089                    endpoint: primary_endpoint,
2090                    model: primary_model,
2091                    purpose: purpose.to_owned(),
2092                    ok: false,
2093                    duration_ms,
2094                    input_tokens: 0,
2095                    output_tokens: 0,
2096                    cost: 0.0,
2097                    error: Some(e.clone()),
2098                });
2099                Err(e)
2100            }
2101        }
2102    }
2103
2104    /// Resolves a CSS selector for the target element. Uses the explicit
2105    /// `selector` if provided, otherwise asks the LLM to find the element
2106    /// from the natural language `target` description and page DOM.
2107    ///
2108    /// LLM responses are sanitized and verified against the live page: a
2109    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
2110    /// immediately with the raw LLM output, and a selector that matches
2111    /// nothing triggers one retry with feedback before failing.
2112    #[allow(clippy::too_many_lines)]
2113    fn resolve_selector(
2114        &self,
2115        css_override: Option<&str>,
2116        target: &str,
2117        step_endpoint: Option<&str>,
2118        test_endpoint: Option<&str>,
2119        tab: &Tab,
2120    ) -> Result<String, String> {
2121        if let Some(explicit) = css_override {
2122            return Ok(explicit.to_owned());
2123        }
2124
2125        let dom_info = extract_dom_info(tab)?;
2126        let page_content = get_page_text(tab);
2127
2128        let system = concat!(
2129            "You are a browser automation selector generator. ",
2130            "Given a web page's content and interactive elements, ",
2131            "return ONLY the best CSS selector for the described element. ",
2132            "Output nothing except the CSS selector. ",
2133            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2134            "[name=\"...\"], tag.class, tag. ",
2135            "Never output explanations, markdown, or extra text."
2136        );
2137
2138        let user = format!(
2139            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2140            page_content.url,
2141            page_content.title,
2142            truncate(&page_content.body_text, 4000),
2143            dom_info,
2144            target,
2145        );
2146
2147        let retry_user = format!(
2148            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2149            "The selector must match at least one element currently present on the page.",
2150            page_content.url,
2151            page_content.title,
2152            truncate(&page_content.body_text, 4000),
2153            dom_info,
2154            target,
2155        );
2156
2157        self.reporter.debug(format!("LLM targeting: {target}"));
2158
2159        let chain = self
2160            .endpoints
2161            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2162        let sys = system.to_owned();
2163
2164        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2165
2166        let first = call_llm(&user);
2167        let (lr, idx) = match first {
2168            Ok(lr) => lr,
2169            Err(e) => {
2170                return Err(format!("LLM element targeting failed: {e}"));
2171            }
2172        };
2173        self.usage.record_llm_call(
2174            &chain[idx].name,
2175            chain[idx],
2176            lr.usage.prompt_tokens,
2177            lr.usage.completion_tokens,
2178        );
2179        let clean = sanitize_selector(&lr.content);
2180        self.reporter.debug(format!("resolved selector: {clean}"));
2181
2182        if selector_is_useless(&clean) {
2183            return Err(format!(
2184                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2185                raw = lr.content.trim(),
2186            ));
2187        }
2188        if let Err(reason) = validate_selector(&clean) {
2189            return Err(format!(
2190                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2191                raw = lr.content.trim(),
2192            ));
2193        }
2194        if !selector_matches(tab, &clean).unwrap_or(false) {
2195            // One retry with feedback: flaky models occasionally invent a
2196            // selector that does not exist on the page.
2197            self.reporter.warn(format!(
2198                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2199            ));
2200            let second = call_llm(&retry_user);
2201            let (lr2, idx2) = match second {
2202                Ok(lr2) => lr2,
2203                Err(e) => {
2204                    return Err(format!(
2205                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2206                    ));
2207                }
2208            };
2209            self.usage.record_llm_call(
2210                &chain[idx2].name,
2211                chain[idx2],
2212                lr2.usage.prompt_tokens,
2213                lr2.usage.completion_tokens,
2214            );
2215            let clean2 = sanitize_selector(&lr2.content);
2216            self.reporter
2217                .debug(format!("resolved selector (retry): {clean2}"));
2218            if selector_is_useless(&clean2) {
2219                return Err(format!(
2220                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2221                    raw = lr2.content.trim(),
2222                    excerpt = truncate(&page_content.body_text, 300),
2223                ));
2224            }
2225            if !selector_matches(tab, &clean2).unwrap_or(false) {
2226                return Err(format!(
2227                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2228                ));
2229            }
2230            return Ok(clean2);
2231        }
2232
2233        Ok(clean)
2234    }
2235}
2236
2237/// Evaluates a JS expression that is expected to return a boolean.
2238fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2239    tab.evaluate(js, false)
2240        .map_err(|e| format!("evaluate failed: {e}"))?
2241        .value
2242        .and_then(|v| v.as_bool())
2243        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2244}
2245
2246/// Checks whether a CSS selector matches at least one current element.
2247fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2248    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2249}
2250
2251// ── Free helper functions ──────────────────────────────────────────────
2252
2253fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2254    let name = format!("[navigate] {full_url}");
2255    match tab.navigate_to(full_url) {
2256        Ok(_) => {
2257            let _ = tab.wait_until_navigated();
2258            StepResult {
2259                name,
2260                status: StepStatus::Passed,
2261                message: format!("navigated to {full_url}"),
2262            }
2263        }
2264        Err(e) => StepResult {
2265            name,
2266            status: StepStatus::Failed,
2267            message: format!("navigation failed: {e}"),
2268        },
2269    }
2270}
2271
2272fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2273    let result = tab
2274        .evaluate(DOM_EXTRACT_JS, false)
2275        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2276
2277    let json_str = result
2278        .value
2279        .as_ref()
2280        .and_then(|v| v.as_str())
2281        .unwrap_or("[]");
2282
2283    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2284
2285    if elements.is_empty() {
2286        return Ok("(no interactive elements found)".to_owned());
2287    }
2288
2289    Ok(elements.join("\n"))
2290}
2291
2292fn get_page_text(tab: &Tab) -> PageContent {
2293    let url = tab.get_url();
2294
2295    let title = tab
2296        .evaluate("document.title", false)
2297        .ok()
2298        .and_then(|r| r.value)
2299        .and_then(|v| v.as_str().map(String::from))
2300        .unwrap_or_else(|| "unknown".to_owned());
2301
2302    let body_text = tab
2303        .evaluate(
2304            "document.body ? document.body.innerText : document.documentElement.innerText",
2305            false,
2306        )
2307        .ok()
2308        .and_then(|r| r.value)
2309        .and_then(|v| v.as_str().map(String::from))
2310        .unwrap_or_default();
2311
2312    PageContent {
2313        url,
2314        title,
2315        body_text: truncate(&body_text, 8000),
2316    }
2317}
2318
2319fn resolve_url(url: &str, base_url: &str) -> String {
2320    if url.starts_with("http://") || url.starts_with("https://") {
2321        return url.to_owned();
2322    }
2323    let base = base_url.trim_end_matches('/');
2324    if url.starts_with('/') {
2325        format!("{base}{url}")
2326    } else {
2327        format!("{base}/{url}")
2328    }
2329}
2330
2331/// Human-readable label for a step, used when steps are skipped after an
2332/// earlier failure.
2333fn step_label(step: &TestStep) -> String {
2334    match step {
2335        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2336        TestStep::Click { target, .. } => format!("[click] {target}"),
2337        TestStep::Type { target, .. } => format!("[type] {target}"),
2338        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2339        TestStep::Assert {
2340            definition,
2341            preset,
2342            prompt,
2343            ..
2344        } => definition.as_ref().map_or_else(
2345            || {
2346                preset.as_ref().map_or_else(
2347                    || {
2348                        prompt.as_ref().map_or_else(
2349                            || "[assert]".to_owned(),
2350                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2351                        )
2352                    },
2353                    |p| format!("[assert] {p}"),
2354                )
2355            },
2356            |d| format!("[assert] {d}"),
2357        ),
2358        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2359        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2360        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2361    }
2362}
2363
2364/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2365#[must_use]
2366const fn step_kind_label(step: &TestStep) -> &'static str {
2367    match step {
2368        TestStep::Navigate { .. } => "navigate",
2369        TestStep::Click { .. } => "click",
2370        TestStep::Type { .. } => "type",
2371        TestStep::Wait { .. } => "wait",
2372        TestStep::Assert { .. } => "assert",
2373        TestStep::Screenshot { .. } => "screenshot",
2374        TestStep::Agent { .. } => "agent",
2375        TestStep::Mcp { .. } => "mcp",
2376    }
2377}
2378
2379// ── Support types ──────────────────────────────────────────────────────
2380
2381#[derive(Default)]
2382struct TestRunResult {
2383    passed: u32,
2384    failed: u32,
2385    skipped: u32,
2386    total: u32,
2387    details: Vec<StepResult>,
2388}
2389
2390struct PageContent {
2391    url: String,
2392    title: String,
2393    body_text: String,
2394}
2395
2396#[cfg(test)]
2397mod tests {
2398    use super::unix_to_rfc3339;
2399
2400    #[test]
2401    fn rfc3339_epoch_and_reference_dates() {
2402        assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2403        assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2404        assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2405        assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2406    }
2407
2408    #[test]
2409    fn rfc3339_handles_leap_years() {
2410        assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2411        assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2412    }
2413}