Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::UsageTracker;
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, TaskType};
14use crate::llm_chat_vision_with_usage;
15use crate::llm_chat_with_usage;
16use crate::mcp_client::McpClient;
17use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
18use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
19use crate::truncate;
20use crate::LlmConfig;
21use crate::DOM_EXTRACT_JS;
22
23/// One detected layout defect (`layout_no_issues` preset).
24#[derive(Debug, serde::Deserialize)]
25struct LayoutIssue {
26    #[serde(rename = "type")]
27    issue_type: String,
28    element: String,
29    detail: String,
30}
31
32/// In-browser DOM layout scan for `layout_no_issues`.
33///
34/// Geometry-only checks (no LLM, no pixels):
35/// 1. page horizontal overflow (`scrollWidth` > viewport width);
36/// 2. visible, non-fixed elements that stick out of the viewport
37///    (right/bottom edge) while still partially on screen;
38/// 3. text clipped by `overflow: hidden` containers whose content
39///    is measurably larger than the box;
40/// 4. interactive elements (buttons/links/inputs) whose center point
41///    is covered by a different element that would intercept the click.
42///
43/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
44/// fixed headers) is excluded by the position/relation filters.
45const LAYOUT_SCAN_JS: &str = r#"
46(() => {
47  const issues = [];
48  const push = (type, el, detail) => {
49    if (issues.length >= 30) return;
50    let element = el.tagName.toLowerCase();
51    if (el.id) element += '#' + el.id;
52    else if (typeof el.className === 'string' && el.className.trim())
53      element += '.' + el.className.trim().split(/\s+/).join('.');
54    issues.push({ type, element, detail: String(detail).slice(0, 220) });
55  };
56  const vw = document.documentElement.clientWidth || window.innerWidth;
57  const vh = document.documentElement.clientHeight || window.innerHeight;
58  if (!vw || !vh) return JSON.stringify(issues);
59  const de = document.documentElement;
60  // 1. Page-level horizontal overflow.
61  if (de.scrollWidth > vw + 2)
62    push('page-overflow-x', de,
63      'page scrollWidth ' + de.scrollWidth + ' exceeds viewport width ' + vw);
64  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
65  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
66  const hasContent = (el) =>
67    ((el.textContent || '').trim().length > 0) ||
68    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
69  // 2. Elements sticking out of the viewport (partially visible only).
70  for (const el of all) {
71    const cs = getComputedStyle(el);
72    if (!visible(cs)) continue;
73    const r = el.getBoundingClientRect();
74    if (r.width < 2 || r.height < 2) continue;
75    if (cs.position === 'fixed' || cs.position === 'sticky') continue;
76    if (!hasContent(el) && el.children.length === 0) continue;
77    if (r.top >= vh || r.left >= vw) continue; // fully offscreen = normal scroll content
78    const overRight = r.right - vw;
79    const overBottom = r.bottom - vh;
80    if (overRight > 2 || overBottom > 2) {
81      let where = '';
82      if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
83      else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
84      else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
85      push('element-out-of-viewport', el, 'extends ' + Math.max(overRight, overBottom).toFixed(0) + 'px past the ' + where);
86    }
87  }
88  // 3. Text clipped by overflow:hidden containers.
89  for (const el of all) {
90    const cs = getComputedStyle(el);
91    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
92    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
93    if (!(el.textContent || '').trim()) continue;
94    push('text-clipped', el,
95      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
96      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
97  }
98  // 4. Interactive elements covered by a different element.
99  const interactive =
100    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
101  const targets = document.querySelectorAll(interactive);
102  for (const el of targets) {
103    const r = el.getBoundingClientRect();
104    if (r.width < 6 || r.height < 6) continue;
105    const cs = getComputedStyle(el);
106    if (!visible(cs)) continue;
107    const cx = r.left + r.width / 2;
108    const cy = r.top + r.height / 2;
109    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
110    const top = document.elementFromPoint(cx, cy);
111    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
112    const tcs = getComputedStyle(top);
113    if (!visible(tcs)) continue;
114    if (tcs.pointerEvents === 'none') continue;
115    const tr = top.getBoundingClientRect();
116    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
117    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
118      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
119    push('element-overlap', el,
120      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
121      ' is covered by <' + tname + '>');
122  }
123  return JSON.stringify(issues);
124})()
125"#;
126
127/// How long the CDP connection stays open after the browser goes quiet.
128///
129/// `headless_chrome` ships a 30s default and tears down the entire connection
130/// when no traffic arrives for that long; a run must own its connection for
131/// its full duration instead.
132const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
133
134/// Executes a [`Scenario`] against a real browser with optional LLM
135/// assistance for element targeting and assertions.
136pub struct ScenarioRunner {
137    config: ScenarioConfig,
138    definitions: HashMap<String, AssertDefinition>,
139    llm: LlmConfig,
140    timeout: Duration,
141    viewport_width: u32,
142    viewport_height: u32,
143    /// The viewport currently applied in the browser (CDP emulation).
144    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
145    applied_viewport: std::cell::Cell<(u32, u32)>,
146    endpoints: EndpointRegistry,
147    usage: Arc<UsageTracker>,
148    budgets: BudgetTracker,
149    /// Directory for failure artifacts (screenshots).
150    artifacts_dir: PathBuf,
151}
152
153/// Aggregated results from a scenario run.
154#[derive(Debug, Default)]
155pub struct RunReport {
156    /// Number of tests that passed.
157    pub tests_passed: u32,
158    /// Number of tests that failed.
159    pub tests_failed: u32,
160    /// Number of steps that passed.
161    pub passed: u32,
162    /// Number of steps that failed.
163    pub failed: u32,
164    /// Number of steps that were skipped.
165    pub skipped: u32,
166    /// Per-step details.
167    pub details: Vec<StepResult>,
168}
169
170/// Result of a single step execution.
171#[derive(Debug)]
172pub struct StepResult {
173    /// The step name.
174    pub name: String,
175    /// Whether the step passed, failed, or was skipped.
176    pub status: StepStatus,
177    /// Human-readable result message.
178    pub message: String,
179}
180
181/// Outcome for a single step.
182#[derive(Debug, PartialEq, Eq)]
183pub enum StepStatus {
184    /// Step executed successfully and all assertions passed.
185    Passed,
186    /// Step execution or assertion failed.
187    Failed,
188    /// Step was skipped.
189    Skipped,
190}
191
192/// Predefined assertion preset definition.
193struct AssertPreset {
194    name: &'static str,
195    system: &'static str,
196    user_template: &'static str,
197}
198
199/// Built-in assertion presets.
200#[allow(clippy::literal_string_with_formatting_args)]
201const ASSERTION_PRESETS: &[AssertPreset] = &[
202    AssertPreset {
203        name: "no_error_on_page",
204        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
205        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
206    },
207    AssertPreset {
208        name: "text_visible",
209        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
210        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
211    },
212    AssertPreset {
213        name: "element_exists",
214        system: "You are a QA tester. Check if a described UI element exists on a web page.",
215        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
216    },
217    AssertPreset {
218        name: "visual_no_issues",
219        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
220        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
221    },
222    AssertPreset {
223        name: "visual_no_overlaps",
224        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
225        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
226    },
227    AssertPreset {
228        name: "visual_text_visible",
229        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
230        user_template: "Inspect the attached screenshot and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
231    },
232    AssertPreset {
233        name: "layout_no_issues",
234        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
235        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
236    },
237];
238
239impl ScenarioRunner {
240    /// Creates a new runner with the given scenario configuration and
241    /// assertion definitions.
242    #[must_use]
243    #[allow(clippy::needless_pass_by_value)]
244    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
245        let llm = LlmConfig {
246            url: scenario_config
247                .llm_url
248                .clone()
249                .unwrap_or_else(crate::llm_base_url),
250            model: scenario_config
251                .llm_model
252                .clone()
253                .unwrap_or_else(crate::llm_model),
254            api_key: scenario_config
255                .llm_api_key
256                .clone()
257                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
258            headers: if scenario_config.llm_headers.is_empty() {
259                crate::parse_headers_env()
260            } else {
261                scenario_config.llm_headers.clone()
262            },
263            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
264            temperature: scenario_config.temperature,
265            thinking: scenario_config.thinking,
266            model_params: scenario_config.model_params.clone(),
267        };
268        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
269        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
270        let defs_map: HashMap<String, AssertDefinition> = definitions
271            .into_iter()
272            .map(|d| (d.name.clone(), d))
273            .collect();
274
275        Self {
276            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
277            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
278            viewport_height: scenario_config.viewport_height.unwrap_or(720),
279            applied_viewport: std::cell::Cell::new((0, 0)),
280            config: scenario_config.clone(),
281            definitions: defs_map,
282            llm,
283            endpoints,
284            usage: Arc::new(UsageTracker::new()),
285            budgets,
286            artifacts_dir: PathBuf::from(
287                scenario_config
288                    .artifacts_dir
289                    .unwrap_or_else(|| "artifacts".to_owned()),
290            ),
291        }
292    }
293
294    /// Returns a clone of the [`UsageTracker`] for reporting.
295    #[must_use]
296    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
297        Arc::clone(&self.usage)
298    }
299
300    /// Returns a reference to the [`BudgetTracker`].
301    #[must_use]
302    pub const fn budget_tracker(&self) -> &BudgetTracker {
303        &self.budgets
304    }
305
306    /// Executes all test groups in the scenario and returns a report.
307    ///
308    /// # Errors
309    ///
310    /// Returns an error if the browser fails to launch.
311    #[allow(clippy::too_many_lines)]
312    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
313        let mut report = RunReport::default();
314
315        if tests.is_empty() {
316            eprintln!("No tests defined in scenario.");
317            return Ok(report);
318        }
319
320        let browser_headless = self.config.browser_headless.unwrap_or(true);
321
322        let launch_opts = LaunchOptions {
323            headless: browser_headless,
324            window_size: Some((self.viewport_width, self.viewport_height)),
325            sandbox: false,
326            // headless_chrome defaults this to 30s and shuts down the whole CDP
327            // connection when no messages arrive for that long. A scenario can
328            // easily exceed 30s of browser silence (slow LLM targeting/assertion
329            // calls, page waits, budget checks between steps), after which every
330            // remaining step fails with "Unable to make method calls because
331            // underlying connection is closed" — one quiet gap kills the run.
332            // Open-ended scenarios must own the connection for their full
333            // duration, so keep it alive for 6 hours.
334            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
335            ..LaunchOptions::default()
336        };
337
338        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
339        let tab = browser.new_tab().context("failed to open browser tab")?;
340        let _ = tab.set_default_timeout(self.timeout);
341
342        // Start MCP server if configured
343        #[cfg(feature = "mcp-server")]
344        if let Some(ref mcp_cfg) = self.config.mcp_server {
345            if mcp_cfg.enabled {
346                let port = mcp_cfg.port;
347                std::thread::spawn(move || {
348                    let _ = crate::mcp_server::start_mcp_server(port);
349                });
350            }
351        }
352        #[cfg(not(feature = "mcp-server"))]
353        if let Some(mcp_cfg) = &self.config.mcp_server {
354            if mcp_cfg.enabled {
355                eprintln!("  ⚠️  MCP server configured but 'mcp-server' feature not enabled");
356            }
357        }
358
359        // Start A2A agent server if configured
360        #[cfg(feature = "a2a-server")]
361        if let Some(ref a2a_cfg) = self.config.a2a_server {
362            if a2a_cfg.enabled {
363                let port = a2a_cfg.port;
364                tokio::spawn(crate::a2a_server::start_a2a_server(port));
365            }
366        }
367        #[cfg(not(feature = "a2a-server"))]
368        if let Some(a2a_cfg) = &self.config.a2a_server {
369            if a2a_cfg.enabled {
370                eprintln!("  ⚠️  A2A server configured but 'a2a-server' feature not enabled");
371            }
372        }
373
374        for test in tests {
375            eprintln!("\n╔══════════════════════════════");
376            eprintln!("║  Test: {}", test.name);
377            eprintln!("╚══════════════════════════════");
378
379            self.usage.reset_per_test();
380
381            let test_result = self.run_test(test, &tab);
382            self.usage.commit_test(&test.name);
383
384            if test_result.failed == 0 && test_result.total > 0 {
385                report.tests_passed += 1;
386                eprintln!("  Test ✅ Passed");
387            } else if test_result.total > 0 {
388                report.tests_failed += 1;
389                eprintln!("  Test ❌ Failed");
390            }
391
392            report.passed += test_result.passed;
393            report.failed += test_result.failed;
394            report.skipped += test_result.skipped;
395            report.details.extend(test_result.details);
396        }
397
398        Ok(report)
399    }
400
401    #[allow(clippy::too_many_lines)]
402    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
403        let base_url = test
404            .base_url
405            .clone()
406            .or_else(|| self.config.base_url.clone())
407            .unwrap_or_else(crate::base_url);
408
409        // Per-test viewport override: switch the browser via CDP
410        // device-metrics emulation before this test runs.
411        let vw = test.viewport_width.unwrap_or(self.viewport_width);
412        let vh = test.viewport_height.unwrap_or(self.viewport_height);
413        if self.applied_viewport.get() != (vw, vh) {
414            self.apply_viewport(tab, vw, vh);
415            self.applied_viewport.set((vw, vh));
416        }
417
418        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
419
420        let start_url = test
421            .start_url
422            .clone()
423            .or_else(|| self.config.start_url.clone())
424            .unwrap_or_else(|| "/dashboard".to_owned());
425
426        if auto_navigate {
427            let full_url = resolve_url(&start_url, &base_url);
428            eprintln!("  → auto-navigate: {full_url}");
429            let _ = tab.navigate_to(&full_url);
430            let _ = tab.wait_until_navigated();
431            std::thread::sleep(Duration::from_secs(4));
432        }
433
434        let mut result = TestRunResult::default();
435
436        for (step_index, step) in test.steps.iter().enumerate() {
437            result.total += 1;
438
439            let wait_ms = match step {
440                TestStep::Navigate { wait_after_ms, .. }
441                | TestStep::Click { wait_after_ms, .. }
442                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
443                _ => None,
444            };
445
446            let mut step_result = match step {
447                TestStep::Navigate { url, .. } => {
448                    let full_url = resolve_url(url, &base_url);
449                    run_navigate_step(&full_url, tab)
450                }
451                TestStep::Click {
452                    target,
453                    selector,
454                    endpoint,
455                    ..
456                } => self.run_click(
457                    target,
458                    selector.as_deref(),
459                    endpoint.as_deref(),
460                    test.endpoint.as_deref(),
461                    tab,
462                ),
463                TestStep::Type {
464                    target,
465                    text,
466                    selector,
467                    endpoint,
468                    ..
469                } => self.run_type(
470                    target,
471                    text,
472                    selector.as_deref(),
473                    endpoint.as_deref(),
474                    test.endpoint.as_deref(),
475                    tab,
476                ),
477                TestStep::Wait {
478                    target,
479                    selector,
480                    text,
481                    timeout_ms,
482                    endpoint,
483                } => self.run_wait(
484                    target,
485                    selector.as_deref(),
486                    text.as_deref(),
487                    *timeout_ms,
488                    endpoint.as_deref(),
489                    test.endpoint.as_deref(),
490                    tab,
491                ),
492                TestStep::Assert {
493                    definition,
494                    preset,
495                    prompt,
496                    assert_text,
497                    endpoint,
498                    screenshot,
499                } => self.run_assert(
500                    definition.as_deref(),
501                    preset.as_deref(),
502                    prompt.as_deref(),
503                    assert_text.as_deref(),
504                    *screenshot,
505                    endpoint.as_deref(),
506                    test.endpoint.as_deref(),
507                    tab,
508                ),
509                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
510                TestStep::Agent {
511                    agent,
512                    task,
513                    definition,
514                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
515                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
516            };
517
518            // Failure diagnostics: capture the page state and a screenshot so
519            // CI logs say WHAT the page looked like when the step failed,
520            // instead of a bare "timed out: The event waited for never came".
521            if step_result.status == StepStatus::Failed {
522                let state = diagnostics::capture(tab);
523                let screenshot = diagnostics::save_screenshot(
524                    tab,
525                    &self.artifacts_dir,
526                    &test.name,
527                    &test.name,
528                    step_index,
529                    step_kind_label(step),
530                );
531                step_result.message = format!(
532                    "{base} — {excerpt}",
533                    base = step_result.message,
534                    excerpt = diagnostics::inline_excerpt(&state),
535                );
536                eprintln!(
537                    "{}",
538                    diagnostics::full_context(&state, screenshot.as_deref())
539                );
540            }
541
542            eprintln!(
543                "    {} {} — {}",
544                if step_result.status == StepStatus::Passed {
545                    "✅"
546                } else if step_result.status == StepStatus::Failed {
547                    "❌"
548                } else {
549                    "⏭️"
550                },
551                step_result.name,
552                step_result.message,
553            );
554
555            match step_result.status {
556                StepStatus::Passed => result.passed += 1,
557                StepStatus::Failed => result.failed += 1,
558                StepStatus::Skipped => result.skipped += 1,
559            }
560
561            // Fail fast: the first failed step ends the test and the
562            // remaining steps are reported as skipped (no LLM budget is
563            // burned asserting against a page that is already known broken).
564            if step_result.status == StepStatus::Failed
565                && !self.config.continue_on_failure
566                && step_index + 1 < test.steps.len()
567            {
568                eprintln!(
569                    "      ⏭️  failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
570                    test.steps.len() - step_index - 1
571                );
572                for skipped in &test.steps[step_index + 1..] {
573                    result.total += 1;
574                    result.skipped += 1;
575                    eprintln!(
576                        "    ⏭️  {} — skipped: previous step failed",
577                        step_label(skipped)
578                    );
579                    result.details.push(StepResult {
580                        name: step_label(skipped),
581                        status: StepStatus::Skipped,
582                        message: "skipped: previous step failed".into(),
583                    });
584                }
585                result.details.push(step_result);
586                return result;
587            }
588
589            // Check per-test budget after each step
590            let test_usage = self.usage.current_test_snapshot();
591            let global_usage = self.usage.global_snapshot();
592            let budget_status = self.budgets.check_all(
593                &test.name,
594                &test_usage,
595                &global_usage,
596                test.budget.as_ref(),
597            );
598            match budget_status {
599                BudgetStatus::HardExceeded { message, .. } => {
600                    crate::reporting::print_budget_error(&message);
601                    result.details.push(StepResult {
602                        name: "[budget]".into(),
603                        status: StepStatus::Failed,
604                        message,
605                    });
606                    result.failed += 1;
607                    return result;
608                }
609                BudgetStatus::SoftExceeded { message, .. } => {
610                    crate::reporting::print_budget_warning(&message);
611                }
612                BudgetStatus::Ok => {}
613            }
614
615            if let Some(ms) = wait_ms {
616                std::thread::sleep(Duration::from_millis(ms));
617            }
618
619            result.details.push(step_result);
620        }
621
622        result
623    }
624
625    /// Applies a viewport size to the current tab via CDP
626    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
627    /// overrides and the viewport matrix. The initial window size set at
628    /// browser launch is replaced by emulation; failures are logged but
629    /// do not fail the test (a mismatched viewport only weakens coverage).
630    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
631        use headless_chrome::protocol::cdp::Emulation;
632        let _ = self;
633        let params = Emulation::SetDeviceMetricsOverride {
634            width,
635            height,
636            device_scale_factor: 1.0,
637            mobile: false,
638            scale: None,
639            screen_width: Some(width),
640            screen_height: Some(height),
641            position_x: None,
642            position_y: None,
643            dont_set_visible_size: None,
644            screen_orientation: None,
645            viewport: None,
646            display_feature: None,
647            device_posture: None,
648        };
649        eprintln!("      ↻ viewport: {width}x{height}");
650        if let Err(e) = tab.call_method(params) {
651            eprintln!("      ⚠️  viewport switch to {width}x{height} failed: {e}");
652        }
653    }
654
655    // ── step handlers ───────────────────────────────────────────────────
656
657    fn run_click(
658        &self,
659        target: &str,
660        selector_override: Option<&str>,
661        step_endpoint: Option<&str>,
662        test_endpoint: Option<&str>,
663        tab: &Tab,
664    ) -> StepResult {
665        let selector = match self.resolve_selector(
666            selector_override,
667            target,
668            step_endpoint,
669            test_endpoint,
670            tab,
671        ) {
672            Ok(s) => s,
673            Err(msg) => {
674                return StepResult {
675                    name: format!("[click] {target}"),
676                    status: StepStatus::Failed,
677                    message: msg,
678                };
679            }
680        };
681
682        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(10)) {
683            Ok(element) => match element.click() {
684                Ok(_) => StepResult {
685                    name: format!("[click] {target}"),
686                    status: StepStatus::Passed,
687                    message: format!("clicked {selector}"),
688                },
689                Err(e) => StepResult {
690                    name: format!("[click] {target}"),
691                    status: StepStatus::Failed,
692                    message: format!("click failed on {selector}: {e}"),
693                },
694            },
695            Err(e) => StepResult {
696                name: format!("[click] {target}"),
697                status: StepStatus::Failed,
698                message: format!("element {selector} not found: {e}"),
699            },
700        }
701    }
702
703    #[allow(clippy::too_many_arguments)]
704    fn run_type(
705        &self,
706        target: &str,
707        text: &str,
708        selector_override: Option<&str>,
709        step_endpoint: Option<&str>,
710        test_endpoint: Option<&str>,
711        tab: &Tab,
712    ) -> StepResult {
713        let selector = match self.resolve_selector(
714            selector_override,
715            target,
716            step_endpoint,
717            test_endpoint,
718            tab,
719        ) {
720            Ok(s) => s,
721            Err(msg) => {
722                return StepResult {
723                    name: format!("[type] {target}"),
724                    status: StepStatus::Failed,
725                    message: msg,
726                };
727            }
728        };
729
730        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(10)) {
731            Ok(element) => {
732                if let Err(e) = element.click() {
733                    return StepResult {
734                        name: format!("[type] {target}"),
735                        status: StepStatus::Failed,
736                        message: format!("click to focus {selector} failed: {e}"),
737                    };
738                }
739
740                let js = format!(
741                    "document.querySelector('{}').value = '';",
742                    selector.replace('\'', "\\'")
743                );
744                let _ = tab.evaluate(&js, false);
745
746                match element.type_into(text) {
747                    Ok(_) => StepResult {
748                        name: format!("[type] {target}"),
749                        status: StepStatus::Passed,
750                        message: format!("typed {text:?} into {selector}"),
751                    },
752                    Err(e) => StepResult {
753                        name: format!("[type] {target}"),
754                        status: StepStatus::Failed,
755                        message: format!("type into {selector} failed: {e}"),
756                    },
757                }
758            }
759            Err(e) => StepResult {
760                name: format!("[type] {target}"),
761                status: StepStatus::Failed,
762                message: format!("element {selector} not found: {e}"),
763            },
764        }
765    }
766
767    #[allow(clippy::too_many_arguments)]
768    fn run_wait(
769        &self,
770        target: &str,
771        selector_override: Option<&str>,
772        text: Option<&str>,
773        timeout_ms: Option<u64>,
774        step_endpoint: Option<&str>,
775        test_endpoint: Option<&str>,
776        tab: &Tab,
777    ) -> StepResult {
778        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
779        let step_name = format!("[wait] {target}");
780
781        // Resolve an explicit selector only (text-only waits are LLM-free).
782        let selector = match selector_override {
783            Some(s) => Some(s.to_owned()),
784            None if text.is_some() => None,
785            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
786                Ok(s) => Some(s),
787                Err(msg) => {
788                    return StepResult {
789                        name: step_name,
790                        status: StepStatus::Failed,
791                        message: msg,
792                    };
793                }
794            },
795        };
796
797        if text.is_some() {
798            let sel_js = selector
799                .as_deref()
800                .map(crate::selectors::selector_matches_js);
801            let text_js = text.map(|t| {
802                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
803                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
804            });
805
806            let deadline = Instant::now() + timeout;
807            loop {
808                let sel_ok = sel_js
809                    .as_ref()
810                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
811                let text_ok = text_js
812                    .as_ref()
813                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
814                if sel_ok && text_ok {
815                    let mut what = Vec::new();
816                    if let Some(sel) = &selector {
817                        what.push(format!("found {sel}"));
818                    }
819                    if let Some(t) = text {
820                        what.push(format!("text {t:?} visible"));
821                    }
822                    return StepResult {
823                        name: step_name,
824                        status: StepStatus::Passed,
825                        message: what.join(" and "),
826                    };
827                }
828                if Instant::now() >= deadline {
829                    let mut what = Vec::new();
830                    if let Some(sel) = &selector {
831                        what.push(sel.clone());
832                    }
833                    if let Some(t) = text {
834                        what.push(format!("text {t:?}"));
835                    }
836                    return StepResult {
837                        name: step_name,
838                        status: StepStatus::Failed,
839                        message: format!(
840                            "wait for {} timed out after {}ms: the event waited for never came",
841                            what.join(" / "),
842                            timeout.as_millis(),
843                        ),
844                    };
845                }
846                std::thread::sleep(Duration::from_millis(250));
847            }
848        }
849
850        match selector.as_deref() {
851            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
852                Ok(_) => StepResult {
853                    name: step_name,
854                    status: StepStatus::Passed,
855                    message: format!("found {sel}"),
856                },
857                Err(e) => StepResult {
858                    name: step_name,
859                    status: StepStatus::Failed,
860                    message: format!(
861                        "wait for {sel} timed out after {}ms: {e}",
862                        timeout.as_millis()
863                    ),
864                },
865            },
866            None => StepResult {
867                name: step_name,
868                status: StepStatus::Failed,
869                message: "wait step has neither selector nor text".into(),
870            },
871        }
872    }
873
874    #[allow(clippy::too_many_arguments)]
875    fn run_assert(
876        &self,
877        definition: Option<&str>,
878        preset: Option<&str>,
879        prompt: Option<&str>,
880        assert_text: Option<&str>,
881        screenshot: bool,
882        step_endpoint: Option<&str>,
883        test_endpoint: Option<&str>,
884        tab: &Tab,
885    ) -> StepResult {
886        std::thread::sleep(Duration::from_millis(500));
887
888        let page_content = get_page_text(tab);
889
890        // Vision attach: capture the viewport once per assert step and hand
891        // the JPEG data URL to the preset/prompt evaluation below.
892        let image = if screenshot {
893            let endpoint = self
894                .endpoints
895                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
896            if !endpoint.vision {
897                return StepResult {
898                    name: "[assert]".into(),
899                    status: StepStatus::Failed,
900                    message: format!(
901                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
902                        name = endpoint.name
903                    ),
904                };
905            }
906            match crate::vision::capture_screenshot_data_url(
907                tab,
908                self.config
909                    .screenshot_max_dimension
910                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
911            ) {
912                Ok(data_url) => Some(data_url),
913                Err(e) => {
914                    return StepResult {
915                        name: "[assert]".into(),
916                        status: StepStatus::Failed,
917                        message: format!("screenshot capture failed: {e}"),
918                    };
919                }
920            }
921        } else {
922            None
923        };
924
925        if let Some(def_name) = definition {
926            if let Some(def) = self.definitions.get(def_name) {
927                return self.run_assert_def(
928                    def,
929                    &page_content,
930                    image.as_deref(),
931                    step_endpoint,
932                    test_endpoint,
933                    tab,
934                );
935            }
936            return StepResult {
937                name: format!("[assert] {def_name}"),
938                status: StepStatus::Failed,
939                message: format!("definition '{def_name}' not found"),
940            };
941        }
942
943        if let Some(preset_name) = preset {
944            // Deterministic DOM layout scan — runs JS in the browser and
945            // never calls the LLM (free, fast, no pixel budget).
946            if preset_name == "layout_no_issues" {
947                return self.run_layout_preset(tab);
948            }
949            return self.run_preset(
950                preset_name,
951                assert_text,
952                &page_content,
953                image.as_deref(),
954                step_endpoint,
955                test_endpoint,
956            );
957        }
958
959        if let Some(prompt_text) = prompt {
960            return self.run_custom(
961                prompt_text,
962                &page_content,
963                image.as_deref(),
964                step_endpoint,
965                test_endpoint,
966            );
967        }
968
969        StepResult {
970            name: "[assert]".into(),
971            status: StepStatus::Skipped,
972            message: "no definition, preset, or prompt specified".into(),
973        }
974    }
975
976    fn run_assert_def(
977        &self,
978        def: &AssertDefinition,
979        page_content: &PageContent,
980        image: Option<&str>,
981        step_endpoint: Option<&str>,
982        test_endpoint: Option<&str>,
983        tab: &Tab,
984    ) -> StepResult {
985        // Agent-based definition: delegate to an A2A agent
986        if let Some(ref agent) = def.agent {
987            if image.is_some() {
988                return StepResult {
989                    name: format!("[assert] {}", def.name),
990                    status: StepStatus::Failed,
991                    message: "agent-backed assertions do not support screenshots".into(),
992                };
993            }
994            let task = def
995                .task_template
996                .as_deref()
997                .unwrap_or("Evaluate the assertion")
998                .replace("{url}", &page_content.url)
999                .replace("{title}", &page_content.title)
1000                .replace("{content}", &page_content.body_text)
1001                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1002
1003            return self.run_agent_step(agent, &task, &def.name);
1004        }
1005
1006        // Custom preset: system + user_template provided in the definition
1007        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1008            return self.run_custom_preset(
1009                &def.name,
1010                system,
1011                template,
1012                def.assert_text.as_deref(),
1013                page_content,
1014                image,
1015                step_endpoint,
1016                test_endpoint,
1017            );
1018        }
1019
1020        def.preset.as_ref().map_or_else(
1021            || {
1022                def.prompt.as_ref().map_or_else(
1023                    || StepResult {
1024                        name: format!("[assert] {}", def.name),
1025                        status: StepStatus::Failed,
1026                        message: "definition has no preset, prompt, or system+user_template".into(),
1027                    },
1028                    |prompt| {
1029                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1030                    },
1031                )
1032            },
1033            |preset_name| {
1034                if preset_name == "layout_no_issues" {
1035                    return self.run_layout_preset(tab);
1036                }
1037                self.run_preset(
1038                    preset_name,
1039                    def.assert_text.as_deref(),
1040                    page_content,
1041                    image,
1042                    step_endpoint,
1043                    test_endpoint,
1044                )
1045            },
1046        )
1047    }
1048
1049    #[allow(clippy::too_many_arguments)]
1050    fn run_custom_preset(
1051        &self,
1052        name: &str,
1053        system: &str,
1054        template: &str,
1055        assert_text: Option<&str>,
1056        page_content: &PageContent,
1057        image: Option<&str>,
1058        step_endpoint: Option<&str>,
1059        test_endpoint: Option<&str>,
1060    ) -> StepResult {
1061        let user_prompt = template
1062            .replace("{url}", &page_content.url)
1063            .replace("{title}", &page_content.title)
1064            .replace("{content}", &page_content.body_text)
1065            .replace("{expected_text}", assert_text.unwrap_or(""))
1066            .replace("{description}", "");
1067
1068        // Custom preset definitions frequently forget the {content}
1069        // placeholder — without it the LLM has no page to evaluate and
1070        // answers "I can't determine that without seeing the page". Always
1071        // append the page context unless the template already references it.
1072        let user_prompt = if template.contains("{content}") {
1073            user_prompt
1074        } else {
1075            format!(
1076                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1077                url = page_content.url,
1078                title = page_content.title,
1079                content = page_content.body_text,
1080            )
1081        };
1082
1083        eprintln!("      assert: {name} (custom preset)");
1084
1085        let endpoint = self
1086            .endpoints
1087            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1088        let llm = self.build_llm_for_endpoint(endpoint);
1089        let usage = Arc::clone(&self.usage);
1090        let endpoint_name = endpoint.name.clone();
1091        let sys = system.to_owned();
1092        let image = image.map(str::to_owned);
1093
1094        let response = std::thread::spawn(move || {
1095            let rt = tokio::runtime::Builder::new_current_thread()
1096                .enable_all()
1097                .build()
1098                .unwrap();
1099            let call = async {
1100                match image.as_deref() {
1101                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user_prompt, img).await,
1102                    None => llm_chat_with_usage(&llm, &sys, &user_prompt).await,
1103                }
1104            };
1105            rt.block_on(call)
1106        })
1107        .join()
1108        .unwrap();
1109
1110        response.map_or_else(
1111            |e| StepResult {
1112                name: format!("[assert] {name}"),
1113                status: StepStatus::Failed,
1114                message: format!("LLM assertion call failed: {e}"),
1115            },
1116            |lr| {
1117                usage.record_llm_call(
1118                    &endpoint_name,
1119                    endpoint,
1120                    lr.usage.prompt_tokens,
1121                    lr.usage.completion_tokens,
1122                );
1123                let content_lower = lr.content.to_lowercase().trim().to_owned();
1124                if content_lower.starts_with("pass") {
1125                    StepResult {
1126                        name: format!("[assert] {name}"),
1127                        status: StepStatus::Passed,
1128                        message: "PASS".into(),
1129                    }
1130                } else {
1131                    StepResult {
1132                        name: format!("[assert] {name}"),
1133                        status: StepStatus::Failed,
1134                        message: lr.content,
1135                    }
1136                }
1137            },
1138        )
1139    }
1140
1141    fn run_preset(
1142        &self,
1143        preset_name: &str,
1144        assert_text: Option<&str>,
1145        page_content: &PageContent,
1146        image: Option<&str>,
1147        step_endpoint: Option<&str>,
1148        test_endpoint: Option<&str>,
1149    ) -> StepResult {
1150        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1151            return StepResult {
1152                name: format!("[assert] {preset_name}"),
1153                status: StepStatus::Failed,
1154                message: format!("unknown assertion preset: {preset_name}"),
1155            };
1156        };
1157        if preset_name.starts_with("visual_") && image.is_none() {
1158            return StepResult {
1159                name: format!("[assert] {preset_name}"),
1160                status: StepStatus::Failed,
1161                message: format!(
1162                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1163                ),
1164            };
1165        }
1166
1167        let user_prompt = preset
1168            .user_template
1169            .replace("{url}", &page_content.url)
1170            .replace("{title}", &page_content.title)
1171            .replace("{content}", &page_content.body_text)
1172            .replace("{expected_text}", assert_text.unwrap_or(""))
1173            .replace("{description}", "");
1174
1175        // Same safety net as custom presets: never let the LLM answer with
1176        // no page context at all.
1177        let user_prompt = if preset.user_template.contains("{content}") {
1178            user_prompt
1179        } else {
1180            format!(
1181                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1182                url = page_content.url,
1183                title = page_content.title,
1184                content = page_content.body_text,
1185            )
1186        };
1187
1188        eprintln!("      assert: {preset_name}");
1189
1190        let endpoint = self
1191            .endpoints
1192            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1193        let llm = self.build_llm_for_endpoint(endpoint);
1194        let usage = Arc::clone(&self.usage);
1195        let endpoint_name = endpoint.name.clone();
1196        let sys = preset.system.to_owned();
1197        let image = image.map(str::to_owned);
1198
1199        let response = std::thread::spawn(move || {
1200            let rt = tokio::runtime::Builder::new_current_thread()
1201                .enable_all()
1202                .build()
1203                .unwrap();
1204            let call = async {
1205                match image.as_deref() {
1206                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user_prompt, img).await,
1207                    None => llm_chat_with_usage(&llm, &sys, &user_prompt).await,
1208                }
1209            };
1210            rt.block_on(call)
1211        })
1212        .join()
1213        .unwrap();
1214
1215        response.map_or_else(
1216            |e| StepResult {
1217                name: format!("[assert] {preset_name}"),
1218                status: StepStatus::Failed,
1219                message: format!("LLM assertion call failed: {e}"),
1220            },
1221            |lr| {
1222                usage.record_llm_call(
1223                    &endpoint_name,
1224                    endpoint,
1225                    lr.usage.prompt_tokens,
1226                    lr.usage.completion_tokens,
1227                );
1228                let content_lower = lr.content.to_lowercase().trim().to_owned();
1229                if content_lower.starts_with("pass") {
1230                    StepResult {
1231                        name: format!("[assert] {preset_name}"),
1232                        status: StepStatus::Passed,
1233                        message: "PASS".into(),
1234                    }
1235                } else {
1236                    StepResult {
1237                        name: format!("[assert] {preset_name}"),
1238                        status: StepStatus::Failed,
1239                        message: lr.content,
1240                    }
1241                }
1242            },
1243        )
1244    }
1245
1246    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1247    ///
1248    /// Evaluates the layout-scan JS in the page and fails with the list of
1249    /// detected issues: horizontal page overflow, visible elements sticking
1250    /// out of the viewport, text clipped by `overflow: hidden` containers,
1251    /// and interactive elements covered by other elements. No LLM call —
1252    /// checks are geometry-based so the check is free, deterministic, and
1253    /// safe to run on every page × viewport variant.
1254    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1255        let _ = self;
1256        let name = "[assert] layout_no_issues".to_owned();
1257        eprintln!("      assert: layout_no_issues (DOM layout scan)");
1258        let result = tab.evaluate(LAYOUT_SCAN_JS, false);
1259        let json_str = match result {
1260            Ok(r) => r
1261                .value
1262                .as_ref()
1263                .and_then(|v| v.as_str().map(String::from))
1264                .unwrap_or_else(|| "[]".to_owned()),
1265            Err(e) => {
1266                return StepResult {
1267                    name,
1268                    status: StepStatus::Failed,
1269                    message: format!("layout scan JS failed: {e}"),
1270                };
1271            }
1272        };
1273        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1274        if issues.is_empty() {
1275            return StepResult {
1276                name,
1277                status: StepStatus::Passed,
1278                message: "PASS — no layout defects detected".into(),
1279            };
1280        }
1281        let mut lines: Vec<String> = issues
1282            .iter()
1283            .take(10)
1284            .map(|i| {
1285                format!(
1286                    "- [{type_}] {element}: {detail}",
1287                    type_ = i.issue_type,
1288                    element = i.element,
1289                    detail = i.detail
1290                )
1291            })
1292            .collect();
1293        if issues.len() > 10 {
1294            lines.push(format!("- … and {} more", issues.len() - 10));
1295        }
1296        StepResult {
1297            name,
1298            status: StepStatus::Failed,
1299            message: format!(
1300                "FAIL — {} layout defect(s) detected:\n{}",
1301                issues.len(),
1302                lines.join("\n")
1303            ),
1304        }
1305    }
1306
1307    fn run_custom(
1308        &self,
1309        prompt: &str,
1310        page_content: &PageContent,
1311        image: Option<&str>,
1312        step_endpoint: Option<&str>,
1313        test_endpoint: Option<&str>,
1314    ) -> StepResult {
1315        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1316
1317        let mut user = format!(
1318            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1319            url = page_content.url,
1320            title = page_content.title,
1321            content = page_content.body_text,
1322        );
1323        if image.is_some() {
1324            user.push_str(
1325                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1326            );
1327        }
1328
1329        eprintln!("      custom assert");
1330
1331        let endpoint = self
1332            .endpoints
1333            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1334        let llm = self.build_llm_for_endpoint(endpoint);
1335        let usage = Arc::clone(&self.usage);
1336        let endpoint_name = endpoint.name.clone();
1337        let sys = system.to_owned();
1338        let image = image.map(str::to_owned);
1339
1340        let response = std::thread::spawn(move || {
1341            let rt = tokio::runtime::Builder::new_current_thread()
1342                .enable_all()
1343                .build()
1344                .unwrap();
1345            let call = async {
1346                match image.as_deref() {
1347                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user, img).await,
1348                    None => llm_chat_with_usage(&llm, &sys, &user).await,
1349                }
1350            };
1351            rt.block_on(call)
1352        })
1353        .join()
1354        .unwrap();
1355
1356        response.map_or_else(
1357            |e| StepResult {
1358                name: "[assert] custom".into(),
1359                status: StepStatus::Failed,
1360                message: format!("LLM assertion call failed: {e}"),
1361            },
1362            |lr| {
1363                usage.record_llm_call(
1364                    &endpoint_name,
1365                    endpoint,
1366                    lr.usage.prompt_tokens,
1367                    lr.usage.completion_tokens,
1368                );
1369                let content_lower = lr.content.to_lowercase().trim().to_owned();
1370                if content_lower.starts_with("pass") {
1371                    StepResult {
1372                        name: "[assert] custom".into(),
1373                        status: StepStatus::Passed,
1374                        message: "PASS".into(),
1375                    }
1376                } else {
1377                    StepResult {
1378                        name: "[assert] custom".into(),
1379                        status: StepStatus::Failed,
1380                        message: lr.content,
1381                    }
1382                }
1383            },
1384        )
1385    }
1386
1387    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1388        let path = path.unwrap_or("screenshot.png");
1389
1390        match tab.capture_screenshot(
1391            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1392            None,
1393            None,
1394            true,
1395        ) {
1396            Ok(data) => {
1397                if let Err(e) = std::fs::write(path, &data) {
1398                    return StepResult {
1399                        name: format!("[screenshot] {path}"),
1400                        status: StepStatus::Failed,
1401                        message: format!("failed to write screenshot: {e}"),
1402                    };
1403                }
1404                StepResult {
1405                    name: format!("[screenshot] {path}"),
1406                    status: StepStatus::Passed,
1407                    message: format!("saved to {path}"),
1408                }
1409            }
1410            Err(e) => StepResult {
1411                name: format!("[screenshot] {path}"),
1412                status: StepStatus::Failed,
1413                message: format!("screenshot failed: {e}"),
1414            },
1415        }
1416    }
1417
1418    /// Runs an A2A agent step.
1419    #[allow(clippy::literal_string_with_formatting_args)]
1420    fn run_agent(
1421        &self,
1422        agent_name: &str,
1423        task: &str,
1424        definition: Option<&str>,
1425        _test_endpoint: Option<&str>,
1426    ) -> StepResult {
1427        // If a definition is specified, look up the task template
1428        let resolved_task = if let Some(def_name) = definition {
1429            if let Some(def) = self.definitions.get(def_name) {
1430                let tmpl = def.task_template.as_deref().unwrap_or(task);
1431                tmpl.replace("{task}", task)
1432            } else {
1433                return StepResult {
1434                    name: format!("[agent] {def_name}"),
1435                    status: StepStatus::Failed,
1436                    message: format!("definition '{def_name}' not found"),
1437                };
1438            }
1439        } else {
1440            task.to_owned()
1441        };
1442
1443        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1444    }
1445
1446    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1447        let Some(ep) = self.endpoints.get(agent_name) else {
1448            return StepResult {
1449                name: format!("[agent] {display_name}"),
1450                status: StepStatus::Failed,
1451                message: format!("agent endpoint '{agent_name}' not found"),
1452            };
1453        };
1454
1455        if ep.url.is_empty() {
1456            return StepResult {
1457                name: format!("[agent] {display_name}"),
1458                status: StepStatus::Failed,
1459                message: format!("agent endpoint '{agent_name}' has no URL"),
1460            };
1461        }
1462
1463        eprintln!("      → agent {agent_name}: {task}");
1464
1465        let url = ep.url.clone();
1466        let client = A2aClient::new(&url, self.timeout);
1467        let task_clone = task.to_owned();
1468
1469        let response = std::thread::spawn(move || {
1470            let rt = tokio::runtime::Builder::new_current_thread()
1471                .enable_all()
1472                .build()
1473                .unwrap();
1474            rt.block_on(client.send_task(&task_clone))
1475        })
1476        .join()
1477        .unwrap();
1478
1479        // Record the flat-cost call
1480        self.usage.record_flat_call(agent_name, ep);
1481
1482        match response {
1483            Ok(text) => {
1484                let clean = text.trim().to_owned();
1485                let lower = clean.to_lowercase();
1486                if lower.starts_with("pass") {
1487                    StepResult {
1488                        name: format!("[agent] {display_name}"),
1489                        status: StepStatus::Passed,
1490                        message: format!("PASS: {clean}"),
1491                    }
1492                } else if lower.starts_with("fail") {
1493                    StepResult {
1494                        name: format!("[agent] {display_name}"),
1495                        status: StepStatus::Failed,
1496                        message: clean,
1497                    }
1498                } else {
1499                    StepResult {
1500                        name: format!("[agent] {display_name}"),
1501                        status: StepStatus::Passed,
1502                        message: format!("response: {clean}"),
1503                    }
1504                }
1505            }
1506            Err(e) => StepResult {
1507                name: format!("[agent] {display_name}"),
1508                status: StepStatus::Failed,
1509                message: format!("agent call failed: {e}"),
1510            },
1511        }
1512    }
1513
1514    /// Runs an MCP tool call step.
1515    fn run_mcp(
1516        &self,
1517        server_name: &str,
1518        tool_name: &str,
1519        args: Option<&serde_json::Value>,
1520    ) -> StepResult {
1521        let Some(ep) = self.endpoints.get(server_name) else {
1522            return StepResult {
1523                name: format!("[mcp] {server_name}:{tool_name}"),
1524                status: StepStatus::Failed,
1525                message: format!("MCP server endpoint '{server_name}' not found"),
1526            };
1527        };
1528
1529        let cmd = ep.command.as_deref().unwrap_or("");
1530        if cmd.is_empty() {
1531            return StepResult {
1532                name: format!("[mcp] {server_name}:{tool_name}"),
1533                status: StepStatus::Failed,
1534                message: format!("MCP server '{server_name}' has no command configured"),
1535            };
1536        }
1537
1538        eprintln!("      → mcp {server_name} {tool_name}");
1539
1540        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1541
1542        let command = cmd.to_owned();
1543        let args_vec = ep.args.clone();
1544        let tool = tool_name.to_owned();
1545
1546        let response = std::thread::spawn(move || {
1547            let mut mcp_client =
1548                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1549            mcp_client
1550                .call_tool(&tool, &args_val)
1551                .map_err(|e| e.to_string())
1552        })
1553        .join()
1554        .unwrap();
1555
1556        // Record the flat-cost call
1557        self.usage.record_flat_call(server_name, ep);
1558
1559        match response {
1560            Ok(result) => {
1561                if result.isError {
1562                    StepResult {
1563                        name: format!("[mcp] {server_name}:{tool_name}"),
1564                        status: StepStatus::Failed,
1565                        message: result.to_string(),
1566                    }
1567                } else {
1568                    StepResult {
1569                        name: format!("[mcp] {server_name}:{tool_name}"),
1570                        status: StepStatus::Passed,
1571                        message: result.to_string(),
1572                    }
1573                }
1574            }
1575            Err(e) => StepResult {
1576                name: format!("[mcp] {server_name}:{tool_name}"),
1577                status: StepStatus::Failed,
1578                message: format!("MCP call failed: {e}"),
1579            },
1580        }
1581    }
1582
1583    // ── helpers ──────────────────────────────────────────────────────────
1584
1585    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1586    /// the runner's default LLM config for any unset fields.
1587    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1588        LlmConfig {
1589            url: if endpoint.url.is_empty() {
1590                self.llm.url.clone()
1591            } else {
1592                endpoint.url.clone()
1593            },
1594            model: endpoint
1595                .model
1596                .clone()
1597                .unwrap_or_else(|| self.llm.model.clone()),
1598            api_key: endpoint
1599                .api_key
1600                .clone()
1601                .or_else(|| self.llm.api_key.clone()),
1602            headers: if endpoint.headers.is_empty() {
1603                self.llm.headers.clone()
1604            } else {
1605                endpoint.headers.clone()
1606            },
1607            timeout: self.llm.timeout,
1608            temperature: self.llm.temperature,
1609            thinking: self.llm.thinking,
1610            model_params: self.llm.model_params.clone(),
1611        }
1612    }
1613
1614    /// Resolves a CSS selector for the target element. Uses the explicit
1615    /// `selector` if provided, otherwise asks the LLM to find the element
1616    /// from the natural language `target` description and page DOM.
1617    ///
1618    /// LLM responses are sanitized and verified against the live page: a
1619    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
1620    /// immediately with the raw LLM output, and a selector that matches
1621    /// nothing triggers one retry with feedback before failing.
1622    #[allow(clippy::too_many_lines)]
1623    fn resolve_selector(
1624        &self,
1625        css_override: Option<&str>,
1626        target: &str,
1627        step_endpoint: Option<&str>,
1628        test_endpoint: Option<&str>,
1629        tab: &Tab,
1630    ) -> Result<String, String> {
1631        if let Some(explicit) = css_override {
1632            return Ok(explicit.to_owned());
1633        }
1634
1635        let dom_info = extract_dom_info(tab)?;
1636        let page_content = get_page_text(tab);
1637
1638        let system = concat!(
1639            "You are a browser automation selector generator. ",
1640            "Given a web page's content and interactive elements, ",
1641            "return ONLY the best CSS selector for the described element. ",
1642            "Output nothing except the CSS selector. ",
1643            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
1644            "[name=\"...\"], tag.class, tag. ",
1645            "Never output explanations, markdown, or extra text."
1646        );
1647
1648        let user = format!(
1649            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
1650            page_content.url,
1651            page_content.title,
1652            truncate(&page_content.body_text, 4000),
1653            dom_info,
1654            target,
1655        );
1656
1657        let retry_user = format!(
1658            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
1659            "The selector must match at least one element currently present on the page.",
1660            page_content.url,
1661            page_content.title,
1662            truncate(&page_content.body_text, 4000),
1663            dom_info,
1664            target,
1665        );
1666
1667        eprintln!("      LLM targeting: {target}");
1668
1669        let endpoint = self
1670            .endpoints
1671            .resolve(step_endpoint.or(test_endpoint), TaskType::Targeting);
1672        let llm = self.build_llm_for_endpoint(endpoint);
1673        let usage = Arc::clone(&self.usage);
1674        let endpoint_name = endpoint.name.clone();
1675        let endpoint_clone = endpoint.clone();
1676        let sys = system.to_owned();
1677
1678        let call_llm = |prompt: &str| {
1679            let llm = llm.clone();
1680            let sys = sys.clone();
1681            let prompt = prompt.to_owned();
1682            std::thread::spawn(move || {
1683                let rt = tokio::runtime::Builder::new_current_thread()
1684                    .enable_all()
1685                    .build()
1686                    .unwrap();
1687                rt.block_on(llm_chat_with_usage(&llm, &sys, &prompt))
1688            })
1689            .join()
1690            .unwrap()
1691        };
1692
1693        let first = call_llm(&user);
1694        let lr = match first {
1695            Ok(lr) => lr,
1696            Err(e) => {
1697                return Err(format!("LLM element targeting failed: {e}"));
1698            }
1699        };
1700        usage.record_llm_call(
1701            &endpoint_name,
1702            &endpoint_clone,
1703            lr.usage.prompt_tokens,
1704            lr.usage.completion_tokens,
1705        );
1706        let clean = sanitize_selector(&lr.content);
1707        eprintln!("      resolved selector: {clean}");
1708
1709        if selector_is_useless(&clean) {
1710            return Err(format!(
1711                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
1712                raw = lr.content.trim(),
1713            ));
1714        }
1715        if let Err(reason) = validate_selector(&clean) {
1716            return Err(format!(
1717                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
1718                raw = lr.content.trim(),
1719            ));
1720        }
1721        if !selector_matches(tab, &clean).unwrap_or(false) {
1722            // One retry with feedback: flaky models occasionally invent a
1723            // selector that does not exist on the page.
1724            eprintln!(
1725                "      selector {clean} matches nothing — retrying LLM targeting with feedback"
1726            );
1727            let second = call_llm(&retry_user);
1728            let lr2 = match second {
1729                Ok(lr2) => lr2,
1730                Err(e) => {
1731                    return Err(format!(
1732                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
1733                    ));
1734                }
1735            };
1736            usage.record_llm_call(
1737                &endpoint_name,
1738                &endpoint_clone,
1739                lr2.usage.prompt_tokens,
1740                lr2.usage.completion_tokens,
1741            );
1742            let clean2 = sanitize_selector(&lr2.content);
1743            eprintln!("      resolved selector (retry): {clean2}");
1744            if selector_is_useless(&clean2) {
1745                return Err(format!(
1746                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
1747                    raw = lr2.content.trim(),
1748                    excerpt = truncate(&page_content.body_text, 300),
1749                ));
1750            }
1751            if !selector_matches(tab, &clean2).unwrap_or(false) {
1752                return Err(format!(
1753                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
1754                ));
1755            }
1756            return Ok(clean2);
1757        }
1758
1759        Ok(clean)
1760    }
1761}
1762
1763/// Evaluates a JS expression that is expected to return a boolean.
1764fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
1765    tab.evaluate(js, false)
1766        .map_err(|e| format!("evaluate failed: {e}"))?
1767        .value
1768        .and_then(|v| v.as_bool())
1769        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
1770}
1771
1772/// Checks whether a CSS selector matches at least one current element.
1773fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
1774    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
1775}
1776
1777// ── Free helper functions ──────────────────────────────────────────────
1778
1779fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
1780    let name = format!("[navigate] {full_url}");
1781    match tab.navigate_to(full_url) {
1782        Ok(_) => {
1783            let _ = tab.wait_until_navigated();
1784            StepResult {
1785                name,
1786                status: StepStatus::Passed,
1787                message: format!("navigated to {full_url}"),
1788            }
1789        }
1790        Err(e) => StepResult {
1791            name,
1792            status: StepStatus::Failed,
1793            message: format!("navigation failed: {e}"),
1794        },
1795    }
1796}
1797
1798fn extract_dom_info(tab: &Tab) -> Result<String, String> {
1799    let result = tab
1800        .evaluate(DOM_EXTRACT_JS, false)
1801        .map_err(|e| format!("DOM extraction failed: {e}"))?;
1802
1803    let json_str = result
1804        .value
1805        .as_ref()
1806        .and_then(|v| v.as_str())
1807        .unwrap_or("[]");
1808
1809    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
1810
1811    if elements.is_empty() {
1812        return Ok("(no interactive elements found)".to_owned());
1813    }
1814
1815    Ok(elements.join("\n"))
1816}
1817
1818fn get_page_text(tab: &Tab) -> PageContent {
1819    let url = tab.get_url();
1820
1821    let title = tab
1822        .evaluate("document.title", false)
1823        .ok()
1824        .and_then(|r| r.value)
1825        .and_then(|v| v.as_str().map(String::from))
1826        .unwrap_or_else(|| "unknown".to_owned());
1827
1828    let body_text = tab
1829        .evaluate(
1830            "document.body ? document.body.innerText : document.documentElement.innerText",
1831            false,
1832        )
1833        .ok()
1834        .and_then(|r| r.value)
1835        .and_then(|v| v.as_str().map(String::from))
1836        .unwrap_or_default();
1837
1838    PageContent {
1839        url,
1840        title,
1841        body_text: truncate(&body_text, 8000),
1842    }
1843}
1844
1845fn resolve_url(url: &str, base_url: &str) -> String {
1846    if url.starts_with("http://") || url.starts_with("https://") {
1847        return url.to_owned();
1848    }
1849    let base = base_url.trim_end_matches('/');
1850    if url.starts_with('/') {
1851        format!("{base}{url}")
1852    } else {
1853        format!("{base}/{url}")
1854    }
1855}
1856
1857/// Human-readable label for a step, used when steps are skipped after an
1858/// earlier failure.
1859fn step_label(step: &TestStep) -> String {
1860    match step {
1861        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
1862        TestStep::Click { target, .. } => format!("[click] {target}"),
1863        TestStep::Type { target, .. } => format!("[type] {target}"),
1864        TestStep::Wait { target, .. } => format!("[wait] {target}"),
1865        TestStep::Assert {
1866            definition,
1867            preset,
1868            prompt,
1869            ..
1870        } => definition.as_ref().map_or_else(
1871            || {
1872                preset.as_ref().map_or_else(
1873                    || {
1874                        prompt.as_ref().map_or_else(
1875                            || "[assert]".to_owned(),
1876                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
1877                        )
1878                    },
1879                    |p| format!("[assert] {p}"),
1880                )
1881            },
1882            |d| format!("[assert] {d}"),
1883        ),
1884        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
1885        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
1886        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
1887    }
1888}
1889
1890/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
1891#[must_use]
1892const fn step_kind_label(step: &TestStep) -> &'static str {
1893    match step {
1894        TestStep::Navigate { .. } => "navigate",
1895        TestStep::Click { .. } => "click",
1896        TestStep::Type { .. } => "type",
1897        TestStep::Wait { .. } => "wait",
1898        TestStep::Assert { .. } => "assert",
1899        TestStep::Screenshot { .. } => "screenshot",
1900        TestStep::Agent { .. } => "agent",
1901        TestStep::Mcp { .. } => "mcp",
1902    }
1903}
1904
1905// ── Support types ──────────────────────────────────────────────────────
1906
1907#[derive(Default)]
1908struct TestRunResult {
1909    passed: u32,
1910    failed: u32,
1911    skipped: u32,
1912    total: u32,
1913    details: Vec<StepResult>,
1914}
1915
1916struct PageContent {
1917    url: String,
1918    title: String,
1919    body_text: String,
1920}