Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25/// One detected layout defect (`layout_no_issues` preset).
26#[derive(Debug, serde::Deserialize)]
27struct LayoutIssue {
28    #[serde(rename = "type")]
29    issue_type: String,
30    element: String,
31    detail: String,
32}
33
34/// In-browser DOM layout scan for `layout_no_issues`.
35///
36/// Geometry-only checks (no LLM, no pixels):
37/// 1. page horizontal overflow (`scrollWidth` > viewport width);
38/// 2. elements outside the viewport that scrolling cannot reveal
39///    (fixed elements off-screen, left/negative overflow, right-edge
40///    overflow beyond the horizontally scrollable area, and bottom
41///    overflow on a page that cannot scroll down) — below-the-fold
42///    content on a tall scrollable page is normal flow, NOT a defect;
43/// 3. text clipped by `overflow: hidden` containers whose content
44///    is measurably larger than the box;
45/// 4. interactive elements (buttons/links/inputs) whose center point
46///    is covered by a different element that would intercept the click.
47///
48/// Elements whose class matches a configured ignore prefix (default:
49/// the Angular CDK screen-reader helpers `cdk-visually-hidden`,
50/// `cdk-describedby-message-container`, `cdk-overlay-container`) are
51/// skipped — they are intentionally 1x1 / off-screen. The prefix list
52/// is injected at `__IGNORE_CLASSES__` from
53/// [`ScenarioConfig::layout_ignore_classes`].
54///
55/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
56/// sticky headers) is excluded by the position/relation filters.
57const LAYOUT_SCAN_JS: &str = r#"
58(() => {
59  const issues = [];
60  const push = (type, el, detail) => {
61    if (issues.length >= 30) return;
62    let element = el.tagName.toLowerCase();
63    if (el.id) element += '#' + el.id;
64    else if (typeof el.className === 'string' && el.className.trim())
65      element += '.' + el.className.trim().split(/\s+/).join('.');
66    issues.push({ type, element, detail: String(detail).slice(0, 220) });
67  };
68  const vw = document.documentElement.clientWidth || window.innerWidth;
69  const vh = document.documentElement.clientHeight || window.innerHeight;
70  if (!vw || !vh) return JSON.stringify(issues);
71  const de = document.documentElement;
72  // 1. Page-level horizontal overflow.
73  if (de.scrollWidth > vw + 2)
74    push('page-overflow-x', de,
75      'page scrollWidth ' + de.scrollWidth + ' exceeds viewport width ' + vw);
76  // Class-name prefixes to skip (injected; default = Angular CDK
77  // screen-reader helpers, which are intentionally 1x1 / off-screen).
78  const ignorePrefixes = __IGNORE_CLASSES__;
79  const isIgnored = (el) => {
80    if (typeof el.className !== 'string' || !el.className.trim()) return false;
81    const classes = el.className.trim().split(/\s+/);
82    return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
83  };
84  const vScrollable = de.scrollHeight > vh + 2;
85  const hScrollable = de.scrollWidth > vw + 2;
86  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
87  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
88  const hasContent = (el) =>
89    ((el.textContent || '').trim().length > 0) ||
90    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
91  // 2. Elements outside the viewport that scrolling cannot reveal.
92  for (const el of all) {
93    const cs = getComputedStyle(el);
94    if (!visible(cs)) continue;
95    if (isIgnored(el)) continue;
96    const r = el.getBoundingClientRect();
97    if (r.width < 2 || r.height < 2) continue;
98    if (!hasContent(el) && el.children.length === 0) continue;
99    if (cs.position === 'fixed') {
100      // Fixed elements never move with the scroll: any edge outside the
101      // viewport is unreachable content and therefore a defect.
102      const overTop = -r.top;
103      const overLeft = -r.left;
104      const overRight = r.right - vw;
105      const overBottom = r.bottom - vh;
106      if (overTop > 2 || overLeft > 2 || overRight > 2 || overBottom > 2) {
107        let where = '';
108        if (overTop > 2 && overLeft > 2) where = 'top+left edges';
109        else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
110        else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
111        else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
112        else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
113        else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
114        push('element-out-of-viewport', el,
115          'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
116      }
117      continue;
118    }
119    if (cs.position === 'sticky') continue;
120    // Negative left overflow cannot be reached by scrolling (scrollLeft
121    // never goes below 0).
122    if (r.left < -2) {
123      push('element-out-of-viewport', el,
124        'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
125      continue;
126    }
127    // Negative top with the page at the top means the element sits above
128    // the document origin — also unreachable.
129    if (r.top < -2 && de.scrollTop <= 2) {
130      push('element-out-of-viewport', el,
131        'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
132      continue;
133    }
134    const overRight = r.right - vw;
135    // Right-edge overflow is only a defect when the page cannot scroll
136    // horizontally to reveal it (or the element sticks out past the
137    // scrollable content width itself).
138    if (overRight > 2 && (!hScrollable || r.right > de.scrollWidth + 2)) {
139      push('element-out-of-viewport', el,
140        'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
141      continue;
142    }
143    // Below-the-fold content on a scrollable page is normal (tall landing
144    // pages); only flag bottom overflow the user can never scroll to.
145    const overBottom = r.bottom - vh;
146    if (overBottom > 2 && (!vScrollable || r.bottom > de.scrollHeight + 2)) {
147      push('element-out-of-viewport', el,
148        'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
149    }
150  }
151  // 3. Text clipped by overflow:hidden containers.
152  for (const el of all) {
153    if (isIgnored(el)) continue;
154    const cs = getComputedStyle(el);
155    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
156    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
157    if (!(el.textContent || '').trim()) continue;
158    push('text-clipped', el,
159      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
160      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
161  }
162  // 4. Interactive elements covered by a different element.
163  const interactive =
164    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
165  const targets = document.querySelectorAll(interactive);
166  for (const el of targets) {
167    if (isIgnored(el)) continue;
168    const r = el.getBoundingClientRect();
169    if (r.width < 6 || r.height < 6) continue;
170    const cs = getComputedStyle(el);
171    if (!visible(cs)) continue;
172    const cx = r.left + r.width / 2;
173    const cy = r.top + r.height / 2;
174    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
175    const top = document.elementFromPoint(cx, cy);
176    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
177    if (isIgnored(top)) continue;
178    const tcs = getComputedStyle(top);
179    if (!visible(tcs)) continue;
180    if (tcs.pointerEvents === 'none') continue;
181    const tr = top.getBoundingClientRect();
182    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
183    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
184      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
185    push('element-overlap', el,
186      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
187      ' is covered by <' + tname + '>');
188  }
189  return JSON.stringify(issues);
190})()
191"#;
192
193/// How long the CDP connection stays open after the browser goes quiet.
194///
195/// `headless_chrome` ships a 30s default and tears down the entire connection
196/// when no traffic arrives for that long; a run must own its connection for
197/// its full duration instead.
198const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
199
200/// Executes a [`Scenario`] against a real browser with optional LLM
201/// assistance for element targeting and assertions.
202pub struct ScenarioRunner {
203    config: ScenarioConfig,
204    definitions: HashMap<String, AssertDefinition>,
205    llm: LlmConfig,
206    timeout: Duration,
207    viewport_width: u32,
208    viewport_height: u32,
209    /// The viewport currently applied in the browser (CDP emulation).
210    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
211    applied_viewport: std::cell::Cell<(u32, u32)>,
212    endpoints: EndpointRegistry,
213    usage: Arc<UsageTracker>,
214    budgets: BudgetTracker,
215    /// Directory for failure artifacts (screenshots).
216    artifacts_dir: PathBuf,
217    /// Event sink for test reporting (console, `JSONL`, `JUnit`, trace).
218    reporter: Arc<Reporter>,
219    /// The test + step index currently executing, for LLM-call events.
220    current_step: std::cell::RefCell<Option<(String, u32)>>,
221}
222
223/// Aggregated results from a scenario run.
224#[derive(Debug, Default)]
225pub struct RunReport {
226    /// Number of tests that passed.
227    pub tests_passed: u32,
228    /// Number of tests that failed.
229    pub tests_failed: u32,
230    /// Number of steps that passed.
231    pub passed: u32,
232    /// Number of steps that failed.
233    pub failed: u32,
234    /// Number of steps that were skipped.
235    pub skipped: u32,
236    /// Per-step details.
237    pub details: Vec<StepResult>,
238}
239
240/// Result of a single step execution.
241#[derive(Debug)]
242pub struct StepResult {
243    /// The step name.
244    pub name: String,
245    /// Whether the step passed, failed, or was skipped.
246    pub status: StepStatus,
247    /// Human-readable result message.
248    pub message: String,
249}
250
251/// Outcome for a single step.
252pub use crate::events::StepStatus;
253
254/// Predefined assertion preset definition.
255struct AssertPreset {
256    name: &'static str,
257    system: &'static str,
258    user_template: &'static str,
259}
260
261/// Built-in assertion presets.
262#[allow(clippy::literal_string_with_formatting_args)]
263const ASSERTION_PRESETS: &[AssertPreset] = &[
264    AssertPreset {
265        name: "no_error_on_page",
266        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
267        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
268    },
269    AssertPreset {
270        name: "text_visible",
271        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
272        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
273    },
274    AssertPreset {
275        name: "element_exists",
276        system: "You are a QA tester. Check if a described UI element exists on a web page.",
277        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
278    },
279    AssertPreset {
280        name: "visual_no_issues",
281        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
282        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
283    },
284    AssertPreset {
285        name: "visual_no_overlaps",
286        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
287        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
288    },
289    AssertPreset {
290        name: "visual_text_visible",
291        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
292        user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
293    },
294    AssertPreset {
295        name: "layout_no_issues",
296        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
297        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
298    },
299];
300
301impl ScenarioRunner {
302    /// Creates a new runner with the given scenario configuration and
303    /// assertion definitions.
304    #[must_use]
305    #[allow(clippy::needless_pass_by_value)]
306    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
307        Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
308    }
309
310    /// Creates a runner that reports run events through the given reporter.
311    #[must_use]
312    #[allow(clippy::needless_pass_by_value)]
313    pub fn with_reporter(
314        scenario_config: ScenarioConfig,
315        definitions: Vec<AssertDefinition>,
316        reporter: Arc<Reporter>,
317    ) -> Self {
318        reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
319            &scenario_config,
320        ));
321        let llm = LlmConfig {
322            url: scenario_config
323                .llm_url
324                .clone()
325                .unwrap_or_else(crate::llm_base_url),
326            model: scenario_config
327                .llm_model
328                .clone()
329                .unwrap_or_else(crate::llm_model),
330            api_key: scenario_config
331                .llm_api_key
332                .clone()
333                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
334            headers: if scenario_config.llm_headers.is_empty() {
335                crate::parse_headers_env()
336            } else {
337                scenario_config.llm_headers.clone()
338            },
339            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
340            temperature: scenario_config.temperature,
341            thinking: scenario_config.thinking,
342            model_params: scenario_config.model_params.clone(),
343            max_attempts: crate::default_llm_attempts(),
344            provider: crate::scenario::Provider::Openai,
345            deployment: None,
346            api_version: None,
347            auth: crate::scenario::AuthConfig::default(),
348            header_commands: std::collections::HashMap::new(),
349            aws: crate::scenario::AwsConfig::default(),
350        };
351        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
352        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
353        let defs_map: HashMap<String, AssertDefinition> = definitions
354            .into_iter()
355            .map(|d| (d.name.clone(), d))
356            .collect();
357
358        Self {
359            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
360            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
361            viewport_height: scenario_config.viewport_height.unwrap_or(720),
362            applied_viewport: std::cell::Cell::new((0, 0)),
363            config: scenario_config.clone(),
364            definitions: defs_map,
365            llm,
366            endpoints,
367            usage: Arc::new(UsageTracker::new()),
368            budgets,
369            artifacts_dir: PathBuf::from(
370                scenario_config
371                    .artifacts_dir
372                    .unwrap_or_else(|| "artifacts".to_owned()),
373            ),
374            reporter,
375            current_step: std::cell::RefCell::new(None),
376        }
377    }
378
379    /// Emits an event; a sink failure degrades to a console warning so a
380    /// broken log file can never mask the run itself.
381    fn emit_event(&self, event: &TestEvent) {
382        if let Err(err) = self.reporter.emit(event) {
383            use std::io::Write as _;
384            let mut out = std::io::stderr().lock();
385            let _ = out.write_fmt(format_args!("  ! failed to record event: {err}\n"));
386        }
387    }
388
389    /// Returns a clone of the [`UsageTracker`] for reporting.
390    #[must_use]
391    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
392        Arc::clone(&self.usage)
393    }
394
395    /// Returns a reference to the [`BudgetTracker`].
396    #[must_use]
397    pub const fn budget_tracker(&self) -> &BudgetTracker {
398        &self.budgets
399    }
400
401    /// Executes all test groups in the scenario and returns a report.
402    ///
403    /// # Errors
404    ///
405    /// Returns an error if the browser fails to launch.
406    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
407    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
408        let mut report = RunReport::default();
409
410        if tests.is_empty() {
411            self.reporter.warn("No tests defined in scenario.");
412            self.emit_event(&TestEvent::RunFinished {
413                tests_passed: 0,
414                tests_failed: 0,
415                steps_passed: 0,
416                steps_failed: 0,
417                steps_skipped: 0,
418                total_cost: 0.0,
419                total_tokens: 0,
420                total_calls: 0,
421            });
422            return Ok(report);
423        }
424
425        self.emit_event(&TestEvent::RunStarted {
426            total_tests: tests.len() as u32,
427        });
428
429        let browser_headless = self.config.browser_headless.unwrap_or(true);
430
431        let launch_opts = LaunchOptions {
432            headless: browser_headless,
433            window_size: Some((self.viewport_width, self.viewport_height)),
434            sandbox: false,
435            // headless_chrome defaults this to 30s and shuts down the whole CDP
436            // connection when no messages arrive for that long. A scenario can
437            // easily exceed 30s of browser silence (slow LLM targeting/assertion
438            // calls, page waits, budget checks between steps), after which every
439            // remaining step fails with "Unable to make method calls because
440            // underlying connection is closed" — one quiet gap kills the run.
441            // Open-ended scenarios must own the connection for their full
442            // duration, so keep it alive for 6 hours.
443            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
444            ..LaunchOptions::default()
445        };
446
447        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
448        let tab = browser.new_tab().context("failed to open browser tab")?;
449        let _ = tab.set_default_timeout(self.timeout);
450
451        // Start MCP server if configured
452        #[cfg(feature = "mcp-server")]
453        if let Some(ref mcp_cfg) = self.config.mcp_server {
454            if mcp_cfg.enabled {
455                let port = mcp_cfg.port;
456                std::thread::spawn(move || {
457                    let _ = crate::mcp_server::start_mcp_server(port);
458                });
459            }
460        }
461        #[cfg(not(feature = "mcp-server"))]
462        if let Some(mcp_cfg) = &self.config.mcp_server {
463            if mcp_cfg.enabled {
464                self.reporter
465                    .warn("MCP server configured but 'mcp-server' feature not enabled");
466            }
467        }
468
469        // Start A2A agent server if configured
470        #[cfg(feature = "a2a-server")]
471        if let Some(ref a2a_cfg) = self.config.a2a_server {
472            if a2a_cfg.enabled {
473                let port = a2a_cfg.port;
474                tokio::spawn(crate::a2a_server::start_a2a_server(port));
475            }
476        }
477        #[cfg(not(feature = "a2a-server"))]
478        if let Some(a2a_cfg) = &self.config.a2a_server {
479            if a2a_cfg.enabled {
480                self.reporter
481                    .warn("A2A server configured but 'a2a-server' feature not enabled");
482            }
483        }
484
485        for test in tests {
486            self.emit_event(&TestEvent::TestStarted {
487                test: test.name.clone(),
488            });
489
490            self.usage.reset_per_test();
491
492            let test_started = Instant::now();
493            let test_result = self.run_test(test, &tab);
494            let duration_ms = test_started.elapsed().as_millis() as u64;
495            let usage = self.usage.current_test_snapshot();
496            self.usage.commit_test(&test.name);
497
498            self.emit_event(&TestEvent::TestFinished {
499                test: test.name.clone(),
500                passed: test_result.passed,
501                failed: test_result.failed,
502                skipped: test_result.skipped,
503                duration_ms,
504                cost: usage.total_cost,
505                tokens: usage.total_tokens,
506                calls: usage.total_calls,
507            });
508
509            if test_result.failed == 0 && test_result.total > 0 {
510                report.tests_passed += 1;
511            } else if test_result.total > 0 {
512                report.tests_failed += 1;
513            }
514
515            report.passed += test_result.passed;
516            report.failed += test_result.failed;
517            report.skipped += test_result.skipped;
518            report.details.extend(test_result.details);
519        }
520
521        let global = self.usage.global_snapshot();
522        self.emit_event(&TestEvent::RunFinished {
523            tests_passed: report.tests_passed,
524            tests_failed: report.tests_failed,
525            steps_passed: report.passed,
526            steps_failed: report.failed,
527            steps_skipped: report.skipped,
528            total_cost: global.total_cost,
529            total_tokens: global.total_tokens,
530            total_calls: global.total_calls,
531        });
532
533        Ok(report)
534    }
535
536    #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
537    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
538        let base_url = test
539            .base_url
540            .clone()
541            .or_else(|| self.config.base_url.clone())
542            .unwrap_or_else(crate::base_url);
543
544        // Per-test viewport override: switch the browser via CDP
545        // device-metrics emulation before this test runs.
546        let vw = test.viewport_width.unwrap_or(self.viewport_width);
547        let vh = test.viewport_height.unwrap_or(self.viewport_height);
548        if self.applied_viewport.get() != (vw, vh) {
549            self.apply_viewport(tab, vw, vh);
550            self.applied_viewport.set((vw, vh));
551        }
552
553        // Per-test isolation: every test starts from its own start_url
554        // (unless auto_navigate is disabled), so a test never inherits the
555        // previous test's page state.
556        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
557
558        let start_url = test
559            .start_url
560            .clone()
561            .or_else(|| self.config.start_url.clone())
562            .unwrap_or_else(|| "/dashboard".to_owned());
563
564        if auto_navigate {
565            let full_url = resolve_url(&start_url, &base_url);
566            self.reporter.debug(format!("auto-navigate: {full_url}"));
567            let _ = tab.navigate_to(&full_url);
568            let _ = tab.wait_until_navigated();
569            std::thread::sleep(Duration::from_secs(4));
570        }
571
572        let mut result = TestRunResult::default();
573
574        for (step_index, step) in test.steps.iter().enumerate() {
575            result.total += 1;
576
577            let wait_ms = match step {
578                TestStep::Navigate { wait_after_ms, .. }
579                | TestStep::Click { wait_after_ms, .. }
580                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
581                _ => None,
582            };
583
584            self.current_step
585                .replace(Some((test.name.clone(), step_index as u32)));
586            self.emit_event(&TestEvent::StepStarted {
587                test: test.name.clone(),
588                index: step_index as u32,
589                label: step_label(step),
590            });
591            let step_started = Instant::now();
592
593            let mut step_result = match step {
594                TestStep::Navigate { url, .. } => {
595                    let full_url = resolve_url(url, &base_url);
596                    run_navigate_step(&full_url, tab)
597                }
598                TestStep::Click {
599                    target,
600                    selector,
601                    endpoint,
602                    idempotent,
603                    ..
604                } => self.run_click(
605                    target,
606                    selector.as_deref(),
607                    endpoint.as_deref(),
608                    test.endpoint.as_deref(),
609                    *idempotent,
610                    tab,
611                ),
612                TestStep::Type {
613                    target,
614                    text,
615                    selector,
616                    endpoint,
617                    idempotent,
618                    ..
619                } => self.run_type(
620                    target,
621                    text,
622                    selector.as_deref(),
623                    endpoint.as_deref(),
624                    test.endpoint.as_deref(),
625                    *idempotent,
626                    tab,
627                ),
628                TestStep::Wait {
629                    target,
630                    selector,
631                    text,
632                    timeout_ms,
633                    endpoint,
634                    idempotent,
635                } => self.run_wait(
636                    target,
637                    selector.as_deref(),
638                    text.as_deref(),
639                    *timeout_ms,
640                    endpoint.as_deref(),
641                    test.endpoint.as_deref(),
642                    *idempotent,
643                    tab,
644                ),
645                TestStep::Assert {
646                    definition,
647                    preset,
648                    prompt,
649                    assert_text,
650                    endpoint,
651                    screenshot,
652                } => self.run_assert(
653                    definition.as_deref(),
654                    preset.as_deref(),
655                    prompt.as_deref(),
656                    assert_text.as_deref(),
657                    *screenshot,
658                    endpoint.as_deref(),
659                    test.endpoint.as_deref(),
660                    tab,
661                ),
662                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
663                TestStep::Agent {
664                    agent,
665                    task,
666                    definition,
667                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
668                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
669            };
670
671            // Failure diagnostics: capture the page state and a screenshot so
672            // CI logs say WHAT the page looked like when the step failed,
673            // instead of a bare "timed out: The event waited for never came".
674            let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
675                let state = diagnostics::capture(tab);
676                let screenshot = diagnostics::save_screenshot(
677                    tab,
678                    &self.artifacts_dir,
679                    &test.name,
680                    &test.name,
681                    step_index,
682                    step_kind_label(step),
683                );
684                step_result.message = format!(
685                    "{base} — {excerpt}",
686                    base = step_result.message,
687                    excerpt = diagnostics::inline_excerpt(&state),
688                );
689                (Some(diagnostics::full_context(&state)), screenshot)
690            } else {
691                (None, None)
692            };
693
694            let duration_ms = step_started.elapsed().as_millis() as u64;
695            self.emit_event(&TestEvent::StepFinished {
696                test: test.name.clone(),
697                index: step_index as u32,
698                label: step_result.name.clone(),
699                status: step_result.status,
700                duration_ms,
701                message: step_result.message.clone(),
702                diagnostics: diagnostics_block,
703                screenshot: screenshot_path,
704            });
705            self.current_step.replace(None);
706
707            match step_result.status {
708                StepStatus::Passed => result.passed += 1,
709                StepStatus::Failed => result.failed += 1,
710                StepStatus::Skipped => result.skipped += 1,
711            }
712
713            // Fail fast: the first failed step ends the test and the
714            // remaining steps are reported as skipped (no LLM budget is
715            // burned asserting against a page that is already known broken).
716            if step_result.status == StepStatus::Failed
717                && !self.config.continue_on_failure
718                && step_index + 1 < test.steps.len()
719            {
720                self.emit_event(&TestEvent::Warning {
721                    message: format!(
722                        "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
723                        test.steps.len() - step_index - 1
724                    ),
725                });
726                for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
727                    let skipped_index = step_index + 1 + offset;
728                    let label = step_label(skipped);
729                    result.total += 1;
730                    result.skipped += 1;
731                    self.emit_event(&TestEvent::StepStarted {
732                        test: test.name.clone(),
733                        index: skipped_index as u32,
734                        label: label.clone(),
735                    });
736                    self.emit_event(&TestEvent::StepFinished {
737                        test: test.name.clone(),
738                        index: skipped_index as u32,
739                        label,
740                        status: StepStatus::Skipped,
741                        duration_ms: 0,
742                        message: "skipped: previous step failed".into(),
743                        diagnostics: None,
744                        screenshot: None,
745                    });
746                    result.details.push(StepResult {
747                        name: step_label(skipped),
748                        status: StepStatus::Skipped,
749                        message: "skipped: previous step failed".into(),
750                    });
751                }
752                result.details.push(step_result);
753                return result;
754            }
755
756            // Check per-test budget after each step
757            let test_usage = self.usage.current_test_snapshot();
758            let global_usage = self.usage.global_snapshot();
759            let budget_status = self.budgets.check_all(
760                &test.name,
761                &test_usage,
762                &global_usage,
763                test.budget.as_ref(),
764            );
765            match budget_status {
766                BudgetStatus::HardExceeded { message, .. } => {
767                    self.emit_event(&TestEvent::Warning {
768                        message: format!("budget exceeded: {message}"),
769                    });
770                    result.details.push(StepResult {
771                        name: "[budget]".into(),
772                        status: StepStatus::Failed,
773                        message,
774                    });
775                    result.failed += 1;
776                    return result;
777                }
778                BudgetStatus::SoftExceeded { message, .. } => {
779                    self.emit_event(&TestEvent::Warning {
780                        message: format!("budget warning: {message}"),
781                    });
782                }
783                BudgetStatus::Ok => {}
784            }
785
786            if let Some(ms) = wait_ms {
787                std::thread::sleep(Duration::from_millis(ms));
788            }
789
790            result.details.push(step_result);
791        }
792
793        result
794    }
795
796    /// Applies a viewport size to the current tab via CDP
797    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
798    /// overrides and the viewport matrix. The initial window size set at
799    /// browser launch is replaced by emulation; failures are logged but
800    /// do not fail the test (a mismatched viewport only weakens coverage).
801    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
802        use headless_chrome::protocol::cdp::Emulation;
803        let _ = self;
804        let params = Emulation::SetDeviceMetricsOverride {
805            width,
806            height,
807            device_scale_factor: 1.0,
808            mobile: false,
809            scale: None,
810            screen_width: Some(width),
811            screen_height: Some(height),
812            position_x: None,
813            position_y: None,
814            dont_set_visible_size: None,
815            screen_orientation: None,
816            viewport: None,
817            display_feature: None,
818            device_posture: None,
819        };
820        self.reporter.debug(format!("viewport: {width}x{height}"));
821        if let Err(e) = tab.call_method(params) {
822            self.reporter
823                .warn(format!("viewport switch to {width}x{height} failed: {e}"));
824        }
825    }
826
827    /// Height (px) covered by assert-step screenshots: the configured
828    /// `screenshot_max_height` (absolute px or viewport multiple, default
829    /// `"20x"`) resolved against the currently applied viewport, raised to
830    /// at least the viewport height so the visible screen is always fully
831    /// included. The capture is split into viewport-tall tiles, so this
832    /// value bounds total coverage (and hence the number of image parts).
833    #[must_use]
834    fn screenshot_height_cap(&self) -> u32 {
835        let viewport_height = self.current_viewport_height();
836        let cap = self
837            .config
838            .screenshot_max_height
839            .as_ref()
840            .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
841        cap.max(viewport_height)
842    }
843
844    /// Height of the viewport currently emulated in the browser (falling
845    /// back to the configured default before any emulation was applied).
846    #[must_use]
847    const fn current_viewport_height(&self) -> u32 {
848        let (_, height) = self.applied_viewport.get();
849        if height > 0 {
850            height
851        } else {
852            self.viewport_height
853        }
854    }
855
856    // ── step handlers ───────────────────────────────────────────────────
857
858    #[allow(clippy::too_many_lines)]
859    fn run_click(
860        &self,
861        target: &str,
862        selector_override: Option<&str>,
863        step_endpoint: Option<&str>,
864        test_endpoint: Option<&str>,
865        idempotent: bool,
866        tab: &Tab,
867    ) -> StepResult {
868        let name = format!("[click] {target}");
869        let selector = match self.resolve_selector(
870            selector_override,
871            target,
872            step_endpoint,
873            test_endpoint,
874            tab,
875        ) {
876            Ok(s) => s,
877            Err(msg) => {
878                if idempotent {
879                    return StepResult {
880                        name,
881                        status: StepStatus::Skipped,
882                        message: format!("skipped (idempotent): no target found — {msg}"),
883                    };
884                }
885                return StepResult {
886                    name,
887                    status: StepStatus::Failed,
888                    message: msg,
889                };
890            }
891        };
892
893        // Idempotent steps probe briefly: a missing target means the
894        // action was already done / not applicable (e.g. an
895        // already-authenticated session), and skipping is the success
896        // path, not a failure.
897        let probe_secs = if idempotent { 5 } else { 10 };
898        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
899            Ok(element) => match element.click() {
900                Ok(_) => StepResult {
901                    name,
902                    status: StepStatus::Passed,
903                    message: format!("clicked {selector}"),
904                },
905                Err(e) => StepResult {
906                    name,
907                    status: StepStatus::Failed,
908                    message: format!("click failed on {selector}: {e}"),
909                },
910            },
911            Err(e) if idempotent => StepResult {
912                name,
913                status: StepStatus::Skipped,
914                message: format!("skipped (idempotent): element {selector} not present — {e}"),
915            },
916            Err(e) => StepResult {
917                name,
918                status: StepStatus::Failed,
919                message: format!("element {selector} not found: {e}"),
920            },
921        }
922    }
923
924    #[allow(clippy::too_many_arguments)]
925    fn run_type(
926        &self,
927        target: &str,
928        text: &str,
929        selector_override: Option<&str>,
930        step_endpoint: Option<&str>,
931        test_endpoint: Option<&str>,
932        idempotent: bool,
933        tab: &Tab,
934    ) -> StepResult {
935        let name = format!("[type] {target}");
936        let selector = match self.resolve_selector(
937            selector_override,
938            target,
939            step_endpoint,
940            test_endpoint,
941            tab,
942        ) {
943            Ok(s) => s,
944            Err(msg) => {
945                if idempotent {
946                    return StepResult {
947                        name,
948                        status: StepStatus::Skipped,
949                        message: format!("skipped (idempotent): no target found — {msg}"),
950                    };
951                }
952                return StepResult {
953                    name,
954                    status: StepStatus::Failed,
955                    message: msg,
956                };
957            }
958        };
959
960        let probe_secs = if idempotent { 5 } else { 10 };
961        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
962            Ok(element) => {
963                if let Err(e) = element.click() {
964                    return StepResult {
965                        name,
966                        status: StepStatus::Failed,
967                        message: format!("click to focus {selector} failed: {e}"),
968                    };
969                }
970
971                let js = format!(
972                    "document.querySelector('{}').value = '';",
973                    selector.replace('\'', "\\'")
974                );
975                let _ = tab.evaluate(&js, false);
976
977                match element.type_into(text) {
978                    Ok(_) => StepResult {
979                        name,
980                        status: StepStatus::Passed,
981                        message: format!("typed {text:?} into {selector}"),
982                    },
983                    Err(e) => StepResult {
984                        name,
985                        status: StepStatus::Failed,
986                        message: format!("type into {selector} failed: {e}"),
987                    },
988                }
989            }
990            Err(e) if idempotent => StepResult {
991                name,
992                status: StepStatus::Skipped,
993                message: format!("skipped (idempotent): element {selector} not present — {e}"),
994            },
995            Err(e) => StepResult {
996                name,
997                status: StepStatus::Failed,
998                message: format!("element {selector} not found: {e}"),
999            },
1000        }
1001    }
1002
1003    #[allow(clippy::too_many_arguments)]
1004    #[allow(clippy::too_many_lines)]
1005    fn run_wait(
1006        &self,
1007        target: &str,
1008        selector_override: Option<&str>,
1009        text: Option<&str>,
1010        timeout_ms: Option<u64>,
1011        step_endpoint: Option<&str>,
1012        test_endpoint: Option<&str>,
1013        idempotent: bool,
1014        tab: &Tab,
1015    ) -> StepResult {
1016        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1017        let step_name = format!("[wait] {target}");
1018
1019        // Resolve an explicit selector only (text-only waits are LLM-free).
1020        let selector = match selector_override {
1021            Some(s) => Some(s.to_owned()),
1022            None if text.is_some() => None,
1023            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1024                Ok(s) => Some(s),
1025                Err(msg) => {
1026                    if idempotent {
1027                        return StepResult {
1028                            name: step_name,
1029                            status: StepStatus::Skipped,
1030                            message: format!("skipped (idempotent): no target found — {msg}"),
1031                        };
1032                    }
1033                    return StepResult {
1034                        name: step_name,
1035                        status: StepStatus::Failed,
1036                        message: msg,
1037                    };
1038                }
1039            },
1040        };
1041
1042        if text.is_some() {
1043            let sel_js = selector
1044                .as_deref()
1045                .map(crate::selectors::selector_matches_js);
1046            let text_js = text.map(|t| {
1047                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1048                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1049            });
1050
1051            let deadline = Instant::now() + timeout;
1052            loop {
1053                let sel_ok = sel_js
1054                    .as_ref()
1055                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1056                let text_ok = text_js
1057                    .as_ref()
1058                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1059                if sel_ok && text_ok {
1060                    let mut what = Vec::new();
1061                    if let Some(sel) = &selector {
1062                        what.push(format!("found {sel}"));
1063                    }
1064                    if let Some(t) = text {
1065                        what.push(format!("text {t:?} visible"));
1066                    }
1067                    return StepResult {
1068                        name: step_name,
1069                        status: StepStatus::Passed,
1070                        message: what.join(" and "),
1071                    };
1072                }
1073                if Instant::now() >= deadline {
1074                    let mut what = Vec::new();
1075                    if let Some(sel) = &selector {
1076                        what.push(sel.clone());
1077                    }
1078                    if let Some(t) = text {
1079                        what.push(format!("text {t:?}"));
1080                    }
1081                    let message = format!(
1082                        "wait for {} timed out after {}ms: the event waited for never came",
1083                        what.join(" / "),
1084                        timeout.as_millis(),
1085                    );
1086                    if idempotent {
1087                        return StepResult {
1088                            name: step_name,
1089                            status: StepStatus::Skipped,
1090                            message: format!("skipped (idempotent): {message}"),
1091                        };
1092                    }
1093                    return StepResult {
1094                        name: step_name,
1095                        status: StepStatus::Failed,
1096                        message,
1097                    };
1098                }
1099                std::thread::sleep(Duration::from_millis(250));
1100            }
1101        }
1102
1103        match selector.as_deref() {
1104            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1105                Ok(_) => StepResult {
1106                    name: step_name,
1107                    status: StepStatus::Passed,
1108                    message: format!("found {sel}"),
1109                },
1110                Err(e) if idempotent => StepResult {
1111                    name: step_name,
1112                    status: StepStatus::Skipped,
1113                    message: format!(
1114                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1115                        timeout.as_millis()
1116                    ),
1117                },
1118                Err(e) => StepResult {
1119                    name: step_name,
1120                    status: StepStatus::Failed,
1121                    message: format!(
1122                        "wait for {sel} timed out after {}ms: {e}",
1123                        timeout.as_millis()
1124                    ),
1125                },
1126            },
1127            None => StepResult {
1128                name: step_name,
1129                status: StepStatus::Failed,
1130                message: "wait step has neither selector nor text".into(),
1131            },
1132        }
1133    }
1134
1135    #[allow(clippy::too_many_arguments)]
1136    fn run_assert(
1137        &self,
1138        definition: Option<&str>,
1139        preset: Option<&str>,
1140        prompt: Option<&str>,
1141        assert_text: Option<&str>,
1142        screenshot: bool,
1143        step_endpoint: Option<&str>,
1144        test_endpoint: Option<&str>,
1145        tab: &Tab,
1146    ) -> StepResult {
1147        std::thread::sleep(Duration::from_millis(500));
1148
1149        let page_content = get_page_text(tab);
1150
1151        // Vision attach: capture the full page once per assert step and
1152        // split it into viewport-tall tiles (the total coverage is bounded
1153        // by the configured height cap so vision tokens stay sane). All
1154        // tile data URLs are handed to the preset/prompt evaluation below.
1155        let image: Option<Vec<String>> = if screenshot {
1156            let endpoint = self
1157                .endpoints
1158                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1159            if !endpoint.vision {
1160                return StepResult {
1161                    name: "[assert]".into(),
1162                    status: StepStatus::Failed,
1163                    message: format!(
1164                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1165                        name = endpoint.name
1166                    ),
1167                };
1168            }
1169            match crate::vision::capture_screenshot_data_urls(
1170                tab,
1171                self.config
1172                    .screenshot_max_dimension
1173                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1174                self.screenshot_height_cap(),
1175                self.current_viewport_height(),
1176            ) {
1177                Ok(urls) => Some(urls),
1178                Err(e) => {
1179                    return StepResult {
1180                        name: "[assert]".into(),
1181                        status: StepStatus::Failed,
1182                        message: format!("screenshot capture failed: {e}"),
1183                    };
1184                }
1185            }
1186        } else {
1187            None
1188        };
1189
1190        if let Some(def_name) = definition {
1191            if let Some(def) = self.definitions.get(def_name) {
1192                return self.run_assert_def(
1193                    def,
1194                    &page_content,
1195                    image.as_deref(),
1196                    step_endpoint,
1197                    test_endpoint,
1198                    tab,
1199                );
1200            }
1201            return StepResult {
1202                name: format!("[assert] {def_name}"),
1203                status: StepStatus::Failed,
1204                message: format!("definition '{def_name}' not found"),
1205            };
1206        }
1207
1208        if let Some(preset_name) = preset {
1209            // Deterministic DOM layout scan — runs JS in the browser and
1210            // never calls the LLM (free, fast, no pixel budget).
1211            if preset_name == "layout_no_issues" {
1212                return self.run_layout_preset(tab);
1213            }
1214            return self.run_preset(
1215                preset_name,
1216                assert_text,
1217                &page_content,
1218                image.as_deref(),
1219                step_endpoint,
1220                test_endpoint,
1221            );
1222        }
1223
1224        if let Some(prompt_text) = prompt {
1225            return self.run_custom(
1226                prompt_text,
1227                &page_content,
1228                image.as_deref(),
1229                step_endpoint,
1230                test_endpoint,
1231            );
1232        }
1233
1234        StepResult {
1235            name: "[assert]".into(),
1236            status: StepStatus::Skipped,
1237            message: "no definition, preset, or prompt specified".into(),
1238        }
1239    }
1240
1241    fn run_assert_def(
1242        &self,
1243        def: &AssertDefinition,
1244        page_content: &PageContent,
1245        image: Option<&[String]>,
1246        step_endpoint: Option<&str>,
1247        test_endpoint: Option<&str>,
1248        tab: &Tab,
1249    ) -> StepResult {
1250        // Agent-based definition: delegate to an A2A agent
1251        if let Some(ref agent) = def.agent {
1252            if image.is_some() {
1253                return StepResult {
1254                    name: format!("[assert] {}", def.name),
1255                    status: StepStatus::Failed,
1256                    message: "agent-backed assertions do not support screenshots".into(),
1257                };
1258            }
1259            let task = def
1260                .task_template
1261                .as_deref()
1262                .unwrap_or("Evaluate the assertion")
1263                .replace("{url}", &page_content.url)
1264                .replace("{title}", &page_content.title)
1265                .replace("{content}", &page_content.body_text)
1266                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1267
1268            return self.run_agent_step(agent, &task, &def.name);
1269        }
1270
1271        // Custom preset: system + user_template provided in the definition
1272        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1273            return self.run_custom_preset(
1274                &def.name,
1275                system,
1276                template,
1277                def.assert_text.as_deref(),
1278                page_content,
1279                image,
1280                step_endpoint,
1281                test_endpoint,
1282            );
1283        }
1284
1285        def.preset.as_ref().map_or_else(
1286            || {
1287                def.prompt.as_ref().map_or_else(
1288                    || StepResult {
1289                        name: format!("[assert] {}", def.name),
1290                        status: StepStatus::Failed,
1291                        message: "definition has no preset, prompt, or system+user_template".into(),
1292                    },
1293                    |prompt| {
1294                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1295                    },
1296                )
1297            },
1298            |preset_name| {
1299                if preset_name == "layout_no_issues" {
1300                    return self.run_layout_preset(tab);
1301                }
1302                self.run_preset(
1303                    preset_name,
1304                    def.assert_text.as_deref(),
1305                    page_content,
1306                    image,
1307                    step_endpoint,
1308                    test_endpoint,
1309                )
1310            },
1311        )
1312    }
1313
1314    #[allow(clippy::too_many_arguments)]
1315    fn run_custom_preset(
1316        &self,
1317        name: &str,
1318        system: &str,
1319        template: &str,
1320        assert_text: Option<&str>,
1321        page_content: &PageContent,
1322        image: Option<&[String]>,
1323        step_endpoint: Option<&str>,
1324        test_endpoint: Option<&str>,
1325    ) -> StepResult {
1326        let user_prompt = template
1327            .replace("{url}", &page_content.url)
1328            .replace("{title}", &page_content.title)
1329            .replace("{content}", &page_content.body_text)
1330            .replace("{expected_text}", assert_text.unwrap_or(""))
1331            .replace("{description}", "");
1332
1333        // Custom preset definitions frequently forget the {content}
1334        // placeholder — without it the LLM has no page to evaluate and
1335        // answers "I can't determine that without seeing the page". Always
1336        // append the page context unless the template already references it.
1337        let user_prompt = if template.contains("{content}") {
1338            user_prompt
1339        } else {
1340            format!(
1341                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1342                url = page_content.url,
1343                title = page_content.title,
1344                content = page_content.body_text,
1345            )
1346        };
1347
1348        self.reporter
1349            .debug(format!("assert: {name} (custom preset)"));
1350
1351        let chain = self
1352            .endpoints
1353            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1354        let sys = system.to_owned();
1355
1356        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1357
1358        response.map_or_else(
1359            |e| StepResult {
1360                name: format!("[assert] {name}"),
1361                status: StepStatus::Failed,
1362                message: format!("LLM assertion call failed: {e}"),
1363            },
1364            |(lr, idx)| {
1365                self.usage.record_llm_call(
1366                    &chain[idx].name,
1367                    chain[idx],
1368                    lr.usage.prompt_tokens,
1369                    lr.usage.completion_tokens,
1370                );
1371                let content_lower = lr.content.to_lowercase().trim().to_owned();
1372                if content_lower.starts_with("pass") {
1373                    StepResult {
1374                        name: format!("[assert] {name}"),
1375                        status: StepStatus::Passed,
1376                        message: "PASS".into(),
1377                    }
1378                } else {
1379                    StepResult {
1380                        name: format!("[assert] {name}"),
1381                        status: StepStatus::Failed,
1382                        message: lr.content,
1383                    }
1384                }
1385            },
1386        )
1387    }
1388
1389    fn run_preset(
1390        &self,
1391        preset_name: &str,
1392        assert_text: Option<&str>,
1393        page_content: &PageContent,
1394        image: Option<&[String]>,
1395        step_endpoint: Option<&str>,
1396        test_endpoint: Option<&str>,
1397    ) -> StepResult {
1398        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1399            return StepResult {
1400                name: format!("[assert] {preset_name}"),
1401                status: StepStatus::Failed,
1402                message: format!("unknown assertion preset: {preset_name}"),
1403            };
1404        };
1405        if preset_name.starts_with("visual_") && image.is_none() {
1406            return StepResult {
1407                name: format!("[assert] {preset_name}"),
1408                status: StepStatus::Failed,
1409                message: format!(
1410                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1411                ),
1412            };
1413        }
1414
1415        let user_prompt = preset
1416            .user_template
1417            .replace("{url}", &page_content.url)
1418            .replace("{title}", &page_content.title)
1419            .replace("{content}", &page_content.body_text)
1420            .replace("{expected_text}", assert_text.unwrap_or(""))
1421            .replace("{description}", "");
1422
1423        // Same safety net as custom presets: never let the LLM answer with
1424        // no page context at all.
1425        let user_prompt = if preset.user_template.contains("{content}") {
1426            user_prompt
1427        } else {
1428            format!(
1429                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1430                url = page_content.url,
1431                title = page_content.title,
1432                content = page_content.body_text,
1433            )
1434        };
1435
1436        self.reporter.debug(format!("assert: {preset_name}"));
1437
1438        let chain = self
1439            .endpoints
1440            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1441        let sys = preset.system.to_owned();
1442
1443        let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1444
1445        response.map_or_else(
1446            |e| StepResult {
1447                name: format!("[assert] {preset_name}"),
1448                status: StepStatus::Failed,
1449                message: format!("LLM assertion call failed: {e}"),
1450            },
1451            |(lr, idx)| {
1452                self.usage.record_llm_call(
1453                    &chain[idx].name,
1454                    chain[idx],
1455                    lr.usage.prompt_tokens,
1456                    lr.usage.completion_tokens,
1457                );
1458                let content_lower = lr.content.to_lowercase().trim().to_owned();
1459                if content_lower.starts_with("pass") {
1460                    StepResult {
1461                        name: format!("[assert] {preset_name}"),
1462                        status: StepStatus::Passed,
1463                        message: "PASS".into(),
1464                    }
1465                } else {
1466                    StepResult {
1467                        name: format!("[assert] {preset_name}"),
1468                        status: StepStatus::Failed,
1469                        message: lr.content,
1470                    }
1471                }
1472            },
1473        )
1474    }
1475
1476    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1477    ///
1478    /// Evaluates the layout-scan JS in the page and fails with the list of
1479    /// detected issues: horizontal page overflow, visible elements sticking
1480    /// out of the viewport, text clipped by `overflow: hidden` containers,
1481    /// and interactive elements covered by other elements. No LLM call —
1482    /// checks are geometry-based so the check is free, deterministic, and
1483    /// safe to run on every page × viewport variant.
1484    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1485        let name = "[assert] layout_no_issues".to_owned();
1486        self.reporter
1487            .debug("assert: layout_no_issues (DOM layout scan)");
1488        let js = LAYOUT_SCAN_JS.replace(
1489            "__IGNORE_CLASSES__",
1490            &serde_json::to_string(&self.config.layout_ignore_classes)
1491                .unwrap_or_else(|_| "[]".to_owned()),
1492        );
1493        let result = tab.evaluate(&js, false);
1494        let json_str = match result {
1495            Ok(r) => r
1496                .value
1497                .as_ref()
1498                .and_then(|v| v.as_str().map(String::from))
1499                .unwrap_or_else(|| "[]".to_owned()),
1500            Err(e) => {
1501                return StepResult {
1502                    name,
1503                    status: StepStatus::Failed,
1504                    message: format!("layout scan JS failed: {e}"),
1505                };
1506            }
1507        };
1508        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1509        if issues.is_empty() {
1510            return StepResult {
1511                name,
1512                status: StepStatus::Passed,
1513                message: "PASS — no layout defects detected".into(),
1514            };
1515        }
1516        let mut lines: Vec<String> = issues
1517            .iter()
1518            .take(10)
1519            .map(|i| {
1520                format!(
1521                    "- [{type_}] {element}: {detail}",
1522                    type_ = i.issue_type,
1523                    element = i.element,
1524                    detail = i.detail
1525                )
1526            })
1527            .collect();
1528        if issues.len() > 10 {
1529            lines.push(format!("- … and {} more", issues.len() - 10));
1530        }
1531        StepResult {
1532            name,
1533            status: StepStatus::Failed,
1534            message: format!(
1535                "FAIL — {} layout defect(s) detected:\n{}",
1536                issues.len(),
1537                lines.join("\n")
1538            ),
1539        }
1540    }
1541
1542    fn run_custom(
1543        &self,
1544        prompt: &str,
1545        page_content: &PageContent,
1546        image: Option<&[String]>,
1547        step_endpoint: Option<&str>,
1548        test_endpoint: Option<&str>,
1549    ) -> StepResult {
1550        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1551
1552        let mut user = format!(
1553            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1554            url = page_content.url,
1555            title = page_content.title,
1556            content = page_content.body_text,
1557        );
1558        if image.is_some() {
1559            user.push_str(
1560                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1561            );
1562        }
1563
1564        self.reporter.debug("custom assert");
1565
1566        let chain = self
1567            .endpoints
1568            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1569        let sys = system.to_owned();
1570
1571        let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1572
1573        response.map_or_else(
1574            |e| StepResult {
1575                name: "[assert] custom".into(),
1576                status: StepStatus::Failed,
1577                message: format!("LLM assertion call failed: {e}"),
1578            },
1579            |(lr, idx)| {
1580                self.usage.record_llm_call(
1581                    &chain[idx].name,
1582                    chain[idx],
1583                    lr.usage.prompt_tokens,
1584                    lr.usage.completion_tokens,
1585                );
1586                let content_lower = lr.content.to_lowercase().trim().to_owned();
1587                if content_lower.starts_with("pass") {
1588                    StepResult {
1589                        name: "[assert] custom".into(),
1590                        status: StepStatus::Passed,
1591                        message: "PASS".into(),
1592                    }
1593                } else {
1594                    StepResult {
1595                        name: "[assert] custom".into(),
1596                        status: StepStatus::Failed,
1597                        message: lr.content,
1598                    }
1599                }
1600            },
1601        )
1602    }
1603
1604    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1605        let path = path.unwrap_or("screenshot.png");
1606
1607        match tab.capture_screenshot(
1608            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1609            None,
1610            None,
1611            true,
1612        ) {
1613            Ok(data) => {
1614                if let Err(e) = std::fs::write(path, &data) {
1615                    return StepResult {
1616                        name: format!("[screenshot] {path}"),
1617                        status: StepStatus::Failed,
1618                        message: format!("failed to write screenshot: {e}"),
1619                    };
1620                }
1621                StepResult {
1622                    name: format!("[screenshot] {path}"),
1623                    status: StepStatus::Passed,
1624                    message: format!("saved to {path}"),
1625                }
1626            }
1627            Err(e) => StepResult {
1628                name: format!("[screenshot] {path}"),
1629                status: StepStatus::Failed,
1630                message: format!("screenshot failed: {e}"),
1631            },
1632        }
1633    }
1634
1635    /// Runs an A2A agent step.
1636    #[allow(clippy::literal_string_with_formatting_args)]
1637    fn run_agent(
1638        &self,
1639        agent_name: &str,
1640        task: &str,
1641        definition: Option<&str>,
1642        _test_endpoint: Option<&str>,
1643    ) -> StepResult {
1644        // If a definition is specified, look up the task template
1645        let resolved_task = if let Some(def_name) = definition {
1646            if let Some(def) = self.definitions.get(def_name) {
1647                let tmpl = def.task_template.as_deref().unwrap_or(task);
1648                tmpl.replace("{task}", task)
1649            } else {
1650                return StepResult {
1651                    name: format!("[agent] {def_name}"),
1652                    status: StepStatus::Failed,
1653                    message: format!("definition '{def_name}' not found"),
1654                };
1655            }
1656        } else {
1657            task.to_owned()
1658        };
1659
1660        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1661    }
1662
1663    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1664        let Some(ep) = self.endpoints.get(agent_name) else {
1665            return StepResult {
1666                name: format!("[agent] {display_name}"),
1667                status: StepStatus::Failed,
1668                message: format!("agent endpoint '{agent_name}' not found"),
1669            };
1670        };
1671
1672        if ep.url.is_empty() {
1673            return StepResult {
1674                name: format!("[agent] {display_name}"),
1675                status: StepStatus::Failed,
1676                message: format!("agent endpoint '{agent_name}' has no URL"),
1677            };
1678        }
1679
1680        self.reporter.debug(format!("agent {agent_name}: {task}"));
1681
1682        let url = ep.url.clone();
1683        let client = A2aClient::new(&url, self.timeout);
1684        let task_clone = task.to_owned();
1685
1686        let response = std::thread::spawn(move || {
1687            let rt = tokio::runtime::Builder::new_current_thread()
1688                .enable_all()
1689                .build()
1690                .unwrap();
1691            rt.block_on(client.send_task(&task_clone))
1692        })
1693        .join()
1694        .unwrap();
1695
1696        // Record the flat-cost call
1697        self.usage.record_flat_call(agent_name, ep);
1698
1699        match response {
1700            Ok(text) => {
1701                let clean = text.trim().to_owned();
1702                let lower = clean.to_lowercase();
1703                if lower.starts_with("pass") {
1704                    StepResult {
1705                        name: format!("[agent] {display_name}"),
1706                        status: StepStatus::Passed,
1707                        message: format!("PASS: {clean}"),
1708                    }
1709                } else if lower.starts_with("fail") {
1710                    StepResult {
1711                        name: format!("[agent] {display_name}"),
1712                        status: StepStatus::Failed,
1713                        message: clean,
1714                    }
1715                } else {
1716                    StepResult {
1717                        name: format!("[agent] {display_name}"),
1718                        status: StepStatus::Passed,
1719                        message: format!("response: {clean}"),
1720                    }
1721                }
1722            }
1723            Err(e) => StepResult {
1724                name: format!("[agent] {display_name}"),
1725                status: StepStatus::Failed,
1726                message: format!("agent call failed: {e}"),
1727            },
1728        }
1729    }
1730
1731    /// Runs an MCP tool call step.
1732    fn run_mcp(
1733        &self,
1734        server_name: &str,
1735        tool_name: &str,
1736        args: Option<&serde_json::Value>,
1737    ) -> StepResult {
1738        let Some(ep) = self.endpoints.get(server_name) else {
1739            return StepResult {
1740                name: format!("[mcp] {server_name}:{tool_name}"),
1741                status: StepStatus::Failed,
1742                message: format!("MCP server endpoint '{server_name}' not found"),
1743            };
1744        };
1745
1746        let cmd = ep.command.as_deref().unwrap_or("");
1747        if cmd.is_empty() {
1748            return StepResult {
1749                name: format!("[mcp] {server_name}:{tool_name}"),
1750                status: StepStatus::Failed,
1751                message: format!("MCP server '{server_name}' has no command configured"),
1752            };
1753        }
1754
1755        self.reporter
1756            .debug(format!("mcp {server_name} {tool_name}"));
1757
1758        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1759
1760        let command = cmd.to_owned();
1761        let args_vec = ep.args.clone();
1762        let tool = tool_name.to_owned();
1763
1764        let response = std::thread::spawn(move || {
1765            let mut mcp_client =
1766                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1767            mcp_client
1768                .call_tool(&tool, &args_val)
1769                .map_err(|e| e.to_string())
1770        })
1771        .join()
1772        .unwrap();
1773
1774        // Record the flat-cost call
1775        self.usage.record_flat_call(server_name, ep);
1776
1777        match response {
1778            Ok(result) => {
1779                if result.isError {
1780                    StepResult {
1781                        name: format!("[mcp] {server_name}:{tool_name}"),
1782                        status: StepStatus::Failed,
1783                        message: result.to_string(),
1784                    }
1785                } else {
1786                    StepResult {
1787                        name: format!("[mcp] {server_name}:{tool_name}"),
1788                        status: StepStatus::Passed,
1789                        message: result.to_string(),
1790                    }
1791                }
1792            }
1793            Err(e) => StepResult {
1794                name: format!("[mcp] {server_name}:{tool_name}"),
1795                status: StepStatus::Failed,
1796                message: format!("MCP call failed: {e}"),
1797            },
1798        }
1799    }
1800
1801    // ── helpers ──────────────────────────────────────────────────────────
1802
1803    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1804    /// the runner's default LLM config for any unset fields.
1805    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1806        LlmConfig {
1807            url: if endpoint.provider == crate::scenario::Provider::Bedrock {
1808                // Bedrock builds its endpoint from the resolved AWS region
1809                // when no URL is given — never inherit the default LLM URL.
1810                endpoint.url.clone()
1811            } else if endpoint.url.is_empty() {
1812                self.llm.url.clone()
1813            } else {
1814                endpoint.url.clone()
1815            },
1816            model: endpoint
1817                .model
1818                .clone()
1819                .unwrap_or_else(|| self.llm.model.clone()),
1820            api_key: endpoint
1821                .api_key
1822                .clone()
1823                .or_else(|| self.llm.api_key.clone()),
1824            headers: if endpoint.headers.is_empty() {
1825                self.llm.headers.clone()
1826            } else {
1827                endpoint.headers.clone()
1828            },
1829            timeout: self.llm.timeout,
1830            temperature: self.llm.temperature,
1831            thinking: self.llm.thinking,
1832            model_params: self.llm.model_params.clone(),
1833            max_attempts: endpoint.max_attempts.max(1),
1834            provider: endpoint.provider,
1835            deployment: endpoint.deployment.clone(),
1836            api_version: endpoint.api_version.clone(),
1837            auth: endpoint.auth.clone(),
1838            header_commands: endpoint.header_commands.clone(),
1839            aws: endpoint.aws.clone(),
1840        }
1841    }
1842
1843    /// Runs a single LLM call against an ordered endpoint chain (primary +
1844    /// fallbacks). Every endpoint gets its own `max_attempts` retry budget;
1845    /// the first endpoint that answers wins. Returns the response together
1846    /// with the chain index of the answering endpoint (0 = primary) so the
1847    /// caller can attribute usage to the correct endpoint.
1848    ///
1849    /// Emits an `LlmCallStarted`/`LlmCallFinished` event pair so the report
1850    /// shows duration, tokens, cost and the answering endpoint per call.
1851    #[allow(clippy::cast_possible_truncation)]
1852    fn llm_call_chain(
1853        &self,
1854        chain: &[&ResolvedEndpoint],
1855        system: &str,
1856        user: &str,
1857        image: Option<&[String]>,
1858        purpose: &str,
1859    ) -> Result<(crate::costs::LlmResponse, usize), String> {
1860        if chain.is_empty() {
1861            return Err("empty LLM endpoint chain".into());
1862        }
1863        let primary = self.build_llm_for_endpoint(chain[0]);
1864        let fallbacks: Vec<LlmConfig> = chain[1..]
1865            .iter()
1866            .map(|e| self.build_llm_for_endpoint(e))
1867            .collect();
1868
1869        let (test, index) = self
1870            .current_step
1871            .borrow()
1872            .as_ref()
1873            .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
1874        let primary_endpoint = chain[0].name.clone();
1875        let primary_model = primary.model.clone();
1876        self.emit_event(&TestEvent::LlmCallStarted {
1877            test: test.clone(),
1878            index,
1879            endpoint: primary_endpoint.clone(),
1880            model: primary_model.clone(),
1881            purpose: purpose.to_owned(),
1882        });
1883
1884        let started = Instant::now();
1885        let sys = system.to_owned();
1886        let user = user.to_owned();
1887        let image = image.map(<[String]>::to_vec);
1888
1889        let result = std::thread::spawn(move || {
1890            let rt = tokio::runtime::Builder::new_current_thread()
1891                .enable_all()
1892                .build()
1893                .unwrap();
1894            let call = async {
1895                match image.as_deref() {
1896                    Some(img) => {
1897                        llm_chat_vision_with_usage_chain(
1898                            &primary,
1899                            &fallbacks,
1900                            &sys,
1901                            &user,
1902                            Some(img),
1903                        )
1904                        .await
1905                    }
1906                    None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
1907                }
1908            };
1909            rt.block_on(call)
1910        })
1911        .join()
1912        .unwrap();
1913
1914        let duration_ms = started.elapsed().as_millis() as u64;
1915        match result {
1916            Ok((lr, idx)) => {
1917                let cost = calculate_llm_cost(
1918                    chain[idx],
1919                    lr.usage.prompt_tokens,
1920                    lr.usage.completion_tokens,
1921                );
1922                let answering = chain[idx].name.clone();
1923                let model = chain[idx]
1924                    .model
1925                    .clone()
1926                    .unwrap_or_else(|| primary_model.clone());
1927                self.emit_event(&TestEvent::LlmCallFinished {
1928                    test,
1929                    index,
1930                    endpoint: answering,
1931                    model,
1932                    purpose: purpose.to_owned(),
1933                    ok: true,
1934                    duration_ms,
1935                    input_tokens: lr.usage.prompt_tokens,
1936                    output_tokens: lr.usage.completion_tokens,
1937                    cost,
1938                    error: None,
1939                });
1940                Ok((lr, idx))
1941            }
1942            Err(e) => {
1943                self.emit_event(&TestEvent::LlmCallFinished {
1944                    test,
1945                    index,
1946                    endpoint: primary_endpoint,
1947                    model: primary_model,
1948                    purpose: purpose.to_owned(),
1949                    ok: false,
1950                    duration_ms,
1951                    input_tokens: 0,
1952                    output_tokens: 0,
1953                    cost: 0.0,
1954                    error: Some(e.clone()),
1955                });
1956                Err(e)
1957            }
1958        }
1959    }
1960
1961    /// Resolves a CSS selector for the target element. Uses the explicit
1962    /// `selector` if provided, otherwise asks the LLM to find the element
1963    /// from the natural language `target` description and page DOM.
1964    ///
1965    /// LLM responses are sanitized and verified against the live page: a
1966    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
1967    /// immediately with the raw LLM output, and a selector that matches
1968    /// nothing triggers one retry with feedback before failing.
1969    #[allow(clippy::too_many_lines)]
1970    fn resolve_selector(
1971        &self,
1972        css_override: Option<&str>,
1973        target: &str,
1974        step_endpoint: Option<&str>,
1975        test_endpoint: Option<&str>,
1976        tab: &Tab,
1977    ) -> Result<String, String> {
1978        if let Some(explicit) = css_override {
1979            return Ok(explicit.to_owned());
1980        }
1981
1982        let dom_info = extract_dom_info(tab)?;
1983        let page_content = get_page_text(tab);
1984
1985        let system = concat!(
1986            "You are a browser automation selector generator. ",
1987            "Given a web page's content and interactive elements, ",
1988            "return ONLY the best CSS selector for the described element. ",
1989            "Output nothing except the CSS selector. ",
1990            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
1991            "[name=\"...\"], tag.class, tag. ",
1992            "Never output explanations, markdown, or extra text."
1993        );
1994
1995        let user = format!(
1996            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
1997            page_content.url,
1998            page_content.title,
1999            truncate(&page_content.body_text, 4000),
2000            dom_info,
2001            target,
2002        );
2003
2004        let retry_user = format!(
2005            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2006            "The selector must match at least one element currently present on the page.",
2007            page_content.url,
2008            page_content.title,
2009            truncate(&page_content.body_text, 4000),
2010            dom_info,
2011            target,
2012        );
2013
2014        self.reporter.debug(format!("LLM targeting: {target}"));
2015
2016        let chain = self
2017            .endpoints
2018            .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2019        let sys = system.to_owned();
2020
2021        let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2022
2023        let first = call_llm(&user);
2024        let (lr, idx) = match first {
2025            Ok(lr) => lr,
2026            Err(e) => {
2027                return Err(format!("LLM element targeting failed: {e}"));
2028            }
2029        };
2030        self.usage.record_llm_call(
2031            &chain[idx].name,
2032            chain[idx],
2033            lr.usage.prompt_tokens,
2034            lr.usage.completion_tokens,
2035        );
2036        let clean = sanitize_selector(&lr.content);
2037        self.reporter.debug(format!("resolved selector: {clean}"));
2038
2039        if selector_is_useless(&clean) {
2040            return Err(format!(
2041                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2042                raw = lr.content.trim(),
2043            ));
2044        }
2045        if let Err(reason) = validate_selector(&clean) {
2046            return Err(format!(
2047                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2048                raw = lr.content.trim(),
2049            ));
2050        }
2051        if !selector_matches(tab, &clean).unwrap_or(false) {
2052            // One retry with feedback: flaky models occasionally invent a
2053            // selector that does not exist on the page.
2054            self.reporter.warn(format!(
2055                "selector {clean} matches nothing — retrying LLM targeting with feedback"
2056            ));
2057            let second = call_llm(&retry_user);
2058            let (lr2, idx2) = match second {
2059                Ok(lr2) => lr2,
2060                Err(e) => {
2061                    return Err(format!(
2062                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2063                    ));
2064                }
2065            };
2066            self.usage.record_llm_call(
2067                &chain[idx2].name,
2068                chain[idx2],
2069                lr2.usage.prompt_tokens,
2070                lr2.usage.completion_tokens,
2071            );
2072            let clean2 = sanitize_selector(&lr2.content);
2073            self.reporter
2074                .debug(format!("resolved selector (retry): {clean2}"));
2075            if selector_is_useless(&clean2) {
2076                return Err(format!(
2077                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2078                    raw = lr2.content.trim(),
2079                    excerpt = truncate(&page_content.body_text, 300),
2080                ));
2081            }
2082            if !selector_matches(tab, &clean2).unwrap_or(false) {
2083                return Err(format!(
2084                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2085                ));
2086            }
2087            return Ok(clean2);
2088        }
2089
2090        Ok(clean)
2091    }
2092}
2093
2094/// Evaluates a JS expression that is expected to return a boolean.
2095fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2096    tab.evaluate(js, false)
2097        .map_err(|e| format!("evaluate failed: {e}"))?
2098        .value
2099        .and_then(|v| v.as_bool())
2100        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2101}
2102
2103/// Checks whether a CSS selector matches at least one current element.
2104fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2105    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2106}
2107
2108// ── Free helper functions ──────────────────────────────────────────────
2109
2110fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2111    let name = format!("[navigate] {full_url}");
2112    match tab.navigate_to(full_url) {
2113        Ok(_) => {
2114            let _ = tab.wait_until_navigated();
2115            StepResult {
2116                name,
2117                status: StepStatus::Passed,
2118                message: format!("navigated to {full_url}"),
2119            }
2120        }
2121        Err(e) => StepResult {
2122            name,
2123            status: StepStatus::Failed,
2124            message: format!("navigation failed: {e}"),
2125        },
2126    }
2127}
2128
2129fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2130    let result = tab
2131        .evaluate(DOM_EXTRACT_JS, false)
2132        .map_err(|e| format!("DOM extraction failed: {e}"))?;
2133
2134    let json_str = result
2135        .value
2136        .as_ref()
2137        .and_then(|v| v.as_str())
2138        .unwrap_or("[]");
2139
2140    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2141
2142    if elements.is_empty() {
2143        return Ok("(no interactive elements found)".to_owned());
2144    }
2145
2146    Ok(elements.join("\n"))
2147}
2148
2149fn get_page_text(tab: &Tab) -> PageContent {
2150    let url = tab.get_url();
2151
2152    let title = tab
2153        .evaluate("document.title", false)
2154        .ok()
2155        .and_then(|r| r.value)
2156        .and_then(|v| v.as_str().map(String::from))
2157        .unwrap_or_else(|| "unknown".to_owned());
2158
2159    let body_text = tab
2160        .evaluate(
2161            "document.body ? document.body.innerText : document.documentElement.innerText",
2162            false,
2163        )
2164        .ok()
2165        .and_then(|r| r.value)
2166        .and_then(|v| v.as_str().map(String::from))
2167        .unwrap_or_default();
2168
2169    PageContent {
2170        url,
2171        title,
2172        body_text: truncate(&body_text, 8000),
2173    }
2174}
2175
2176fn resolve_url(url: &str, base_url: &str) -> String {
2177    if url.starts_with("http://") || url.starts_with("https://") {
2178        return url.to_owned();
2179    }
2180    let base = base_url.trim_end_matches('/');
2181    if url.starts_with('/') {
2182        format!("{base}{url}")
2183    } else {
2184        format!("{base}/{url}")
2185    }
2186}
2187
2188/// Human-readable label for a step, used when steps are skipped after an
2189/// earlier failure.
2190fn step_label(step: &TestStep) -> String {
2191    match step {
2192        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2193        TestStep::Click { target, .. } => format!("[click] {target}"),
2194        TestStep::Type { target, .. } => format!("[type] {target}"),
2195        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2196        TestStep::Assert {
2197            definition,
2198            preset,
2199            prompt,
2200            ..
2201        } => definition.as_ref().map_or_else(
2202            || {
2203                preset.as_ref().map_or_else(
2204                    || {
2205                        prompt.as_ref().map_or_else(
2206                            || "[assert]".to_owned(),
2207                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2208                        )
2209                    },
2210                    |p| format!("[assert] {p}"),
2211                )
2212            },
2213            |d| format!("[assert] {d}"),
2214        ),
2215        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2216        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2217        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2218    }
2219}
2220
2221/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2222#[must_use]
2223const fn step_kind_label(step: &TestStep) -> &'static str {
2224    match step {
2225        TestStep::Navigate { .. } => "navigate",
2226        TestStep::Click { .. } => "click",
2227        TestStep::Type { .. } => "type",
2228        TestStep::Wait { .. } => "wait",
2229        TestStep::Assert { .. } => "assert",
2230        TestStep::Screenshot { .. } => "screenshot",
2231        TestStep::Agent { .. } => "agent",
2232        TestStep::Mcp { .. } => "mcp",
2233    }
2234}
2235
2236// ── Support types ──────────────────────────────────────────────────────
2237
2238#[derive(Default)]
2239struct TestRunResult {
2240    passed: u32,
2241    failed: u32,
2242    skipped: u32,
2243    total: u32,
2244    details: Vec<StepResult>,
2245}
2246
2247struct PageContent {
2248    url: String,
2249    title: String,
2250    body_text: String,
2251}