Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::UsageTracker;
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, TaskType};
14use crate::llm_chat_vision_with_usage;
15use crate::llm_chat_with_usage;
16use crate::mcp_client::McpClient;
17use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
18use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
19use crate::truncate;
20use crate::LlmConfig;
21use crate::DOM_EXTRACT_JS;
22
23/// One detected layout defect (`layout_no_issues` preset).
24#[derive(Debug, serde::Deserialize)]
25struct LayoutIssue {
26    #[serde(rename = "type")]
27    issue_type: String,
28    element: String,
29    detail: String,
30}
31
32/// In-browser DOM layout scan for `layout_no_issues`.
33///
34/// Geometry-only checks (no LLM, no pixels):
35/// 1. page horizontal overflow (`scrollWidth` > viewport width);
36/// 2. visible, non-fixed elements that stick out of the viewport
37///    (right/bottom edge) while still partially on screen;
38/// 3. text clipped by `overflow: hidden` containers whose content
39///    is measurably larger than the box;
40/// 4. interactive elements (buttons/links/inputs) whose center point
41///    is covered by a different element that would intercept the click.
42///
43/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
44/// fixed headers) is excluded by the position/relation filters.
45const LAYOUT_SCAN_JS: &str = r#"
46(() => {
47  const issues = [];
48  const push = (type, el, detail) => {
49    if (issues.length >= 30) return;
50    let element = el.tagName.toLowerCase();
51    if (el.id) element += '#' + el.id;
52    else if (typeof el.className === 'string' && el.className.trim())
53      element += '.' + el.className.trim().split(/\s+/).join('.');
54    issues.push({ type, element, detail: String(detail).slice(0, 220) });
55  };
56  const vw = document.documentElement.clientWidth || window.innerWidth;
57  const vh = document.documentElement.clientHeight || window.innerHeight;
58  if (!vw || !vh) return JSON.stringify(issues);
59  const de = document.documentElement;
60  // 1. Page-level horizontal overflow.
61  if (de.scrollWidth > vw + 2)
62    push('page-overflow-x', de,
63      'page scrollWidth ' + de.scrollWidth + ' exceeds viewport width ' + vw);
64  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
65  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
66  const hasContent = (el) =>
67    ((el.textContent || '').trim().length > 0) ||
68    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
69  // 2. Elements sticking out of the viewport (partially visible only).
70  for (const el of all) {
71    const cs = getComputedStyle(el);
72    if (!visible(cs)) continue;
73    const r = el.getBoundingClientRect();
74    if (r.width < 2 || r.height < 2) continue;
75    if (cs.position === 'fixed' || cs.position === 'sticky') continue;
76    if (!hasContent(el) && el.children.length === 0) continue;
77    if (r.top >= vh || r.left >= vw) continue; // fully offscreen = normal scroll content
78    const overRight = r.right - vw;
79    const overBottom = r.bottom - vh;
80    if (overRight > 2 || overBottom > 2) {
81      let where = '';
82      if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
83      else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
84      else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
85      push('element-out-of-viewport', el, 'extends ' + Math.max(overRight, overBottom).toFixed(0) + 'px past the ' + where);
86    }
87  }
88  // 3. Text clipped by overflow:hidden containers.
89  for (const el of all) {
90    const cs = getComputedStyle(el);
91    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
92    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
93    if (!(el.textContent || '').trim()) continue;
94    push('text-clipped', el,
95      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
96      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
97  }
98  // 4. Interactive elements covered by a different element.
99  const interactive =
100    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
101  const targets = document.querySelectorAll(interactive);
102  for (const el of targets) {
103    const r = el.getBoundingClientRect();
104    if (r.width < 6 || r.height < 6) continue;
105    const cs = getComputedStyle(el);
106    if (!visible(cs)) continue;
107    const cx = r.left + r.width / 2;
108    const cy = r.top + r.height / 2;
109    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
110    const top = document.elementFromPoint(cx, cy);
111    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
112    const tcs = getComputedStyle(top);
113    if (!visible(tcs)) continue;
114    if (tcs.pointerEvents === 'none') continue;
115    const tr = top.getBoundingClientRect();
116    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
117    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
118      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
119    push('element-overlap', el,
120      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
121      ' is covered by <' + tname + '>');
122  }
123  return JSON.stringify(issues);
124})()
125"#;
126
127/// How long the CDP connection stays open after the browser goes quiet.
128///
129/// `headless_chrome` ships a 30s default and tears down the entire connection
130/// when no traffic arrives for that long; a run must own its connection for
131/// its full duration instead.
132const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
133
134/// Executes a [`Scenario`] against a real browser with optional LLM
135/// assistance for element targeting and assertions.
136pub struct ScenarioRunner {
137    config: ScenarioConfig,
138    definitions: HashMap<String, AssertDefinition>,
139    llm: LlmConfig,
140    timeout: Duration,
141    viewport_width: u32,
142    viewport_height: u32,
143    /// The viewport currently applied in the browser (CDP emulation).
144    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
145    applied_viewport: std::cell::Cell<(u32, u32)>,
146    endpoints: EndpointRegistry,
147    usage: Arc<UsageTracker>,
148    budgets: BudgetTracker,
149    /// Directory for failure artifacts (screenshots).
150    artifacts_dir: PathBuf,
151}
152
153/// Aggregated results from a scenario run.
154#[derive(Debug, Default)]
155pub struct RunReport {
156    /// Number of tests that passed.
157    pub tests_passed: u32,
158    /// Number of tests that failed.
159    pub tests_failed: u32,
160    /// Number of steps that passed.
161    pub passed: u32,
162    /// Number of steps that failed.
163    pub failed: u32,
164    /// Number of steps that were skipped.
165    pub skipped: u32,
166    /// Per-step details.
167    pub details: Vec<StepResult>,
168}
169
170/// Result of a single step execution.
171#[derive(Debug)]
172pub struct StepResult {
173    /// The step name.
174    pub name: String,
175    /// Whether the step passed, failed, or was skipped.
176    pub status: StepStatus,
177    /// Human-readable result message.
178    pub message: String,
179}
180
181/// Outcome for a single step.
182#[derive(Debug, PartialEq, Eq)]
183pub enum StepStatus {
184    /// Step executed successfully and all assertions passed.
185    Passed,
186    /// Step execution or assertion failed.
187    Failed,
188    /// Step was skipped.
189    Skipped,
190}
191
192/// Predefined assertion preset definition.
193struct AssertPreset {
194    name: &'static str,
195    system: &'static str,
196    user_template: &'static str,
197}
198
199/// Built-in assertion presets.
200#[allow(clippy::literal_string_with_formatting_args)]
201const ASSERTION_PRESETS: &[AssertPreset] = &[
202    AssertPreset {
203        name: "no_error_on_page",
204        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
205        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
206    },
207    AssertPreset {
208        name: "text_visible",
209        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
210        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
211    },
212    AssertPreset {
213        name: "element_exists",
214        system: "You are a QA tester. Check if a described UI element exists on a web page.",
215        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
216    },
217    AssertPreset {
218        name: "visual_no_issues",
219        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
220        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
221    },
222    AssertPreset {
223        name: "visual_no_overlaps",
224        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
225        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
226    },
227    AssertPreset {
228        name: "visual_text_visible",
229        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
230        user_template: "Inspect the attached screenshot and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
231    },
232    AssertPreset {
233        name: "layout_no_issues",
234        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
235        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
236    },
237];
238
239impl ScenarioRunner {
240    /// Creates a new runner with the given scenario configuration and
241    /// assertion definitions.
242    #[must_use]
243    #[allow(clippy::needless_pass_by_value)]
244    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
245        let llm = LlmConfig {
246            url: scenario_config
247                .llm_url
248                .clone()
249                .unwrap_or_else(crate::llm_base_url),
250            model: scenario_config
251                .llm_model
252                .clone()
253                .unwrap_or_else(crate::llm_model),
254            api_key: scenario_config
255                .llm_api_key
256                .clone()
257                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
258            headers: if scenario_config.llm_headers.is_empty() {
259                crate::parse_headers_env()
260            } else {
261                scenario_config.llm_headers.clone()
262            },
263            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
264            temperature: scenario_config.temperature,
265            thinking: scenario_config.thinking,
266            model_params: scenario_config.model_params.clone(),
267        };
268        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
269        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
270        let defs_map: HashMap<String, AssertDefinition> = definitions
271            .into_iter()
272            .map(|d| (d.name.clone(), d))
273            .collect();
274
275        Self {
276            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
277            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
278            viewport_height: scenario_config.viewport_height.unwrap_or(720),
279            applied_viewport: std::cell::Cell::new((0, 0)),
280            config: scenario_config.clone(),
281            definitions: defs_map,
282            llm,
283            endpoints,
284            usage: Arc::new(UsageTracker::new()),
285            budgets,
286            artifacts_dir: PathBuf::from(
287                scenario_config
288                    .artifacts_dir
289                    .unwrap_or_else(|| "artifacts".to_owned()),
290            ),
291        }
292    }
293
294    /// Returns a clone of the [`UsageTracker`] for reporting.
295    #[must_use]
296    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
297        Arc::clone(&self.usage)
298    }
299
300    /// Returns a reference to the [`BudgetTracker`].
301    #[must_use]
302    pub const fn budget_tracker(&self) -> &BudgetTracker {
303        &self.budgets
304    }
305
306    /// Executes all test groups in the scenario and returns a report.
307    ///
308    /// # Errors
309    ///
310    /// Returns an error if the browser fails to launch.
311    #[allow(clippy::too_many_lines)]
312    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
313        let mut report = RunReport::default();
314
315        if tests.is_empty() {
316            eprintln!("No tests defined in scenario.");
317            return Ok(report);
318        }
319
320        let browser_headless = self.config.browser_headless.unwrap_or(true);
321
322        let launch_opts = LaunchOptions {
323            headless: browser_headless,
324            window_size: Some((self.viewport_width, self.viewport_height)),
325            sandbox: false,
326            // headless_chrome defaults this to 30s and shuts down the whole CDP
327            // connection when no messages arrive for that long. A scenario can
328            // easily exceed 30s of browser silence (slow LLM targeting/assertion
329            // calls, page waits, budget checks between steps), after which every
330            // remaining step fails with "Unable to make method calls because
331            // underlying connection is closed" — one quiet gap kills the run.
332            // Open-ended scenarios must own the connection for their full
333            // duration, so keep it alive for 6 hours.
334            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
335            ..LaunchOptions::default()
336        };
337
338        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
339        let tab = browser.new_tab().context("failed to open browser tab")?;
340        let _ = tab.set_default_timeout(self.timeout);
341
342        // Start MCP server if configured
343        #[cfg(feature = "mcp-server")]
344        if let Some(ref mcp_cfg) = self.config.mcp_server {
345            if mcp_cfg.enabled {
346                let port = mcp_cfg.port;
347                std::thread::spawn(move || {
348                    let _ = crate::mcp_server::start_mcp_server(port);
349                });
350            }
351        }
352        #[cfg(not(feature = "mcp-server"))]
353        if let Some(mcp_cfg) = &self.config.mcp_server {
354            if mcp_cfg.enabled {
355                eprintln!("  ⚠️  MCP server configured but 'mcp-server' feature not enabled");
356            }
357        }
358
359        // Start A2A agent server if configured
360        #[cfg(feature = "a2a-server")]
361        if let Some(ref a2a_cfg) = self.config.a2a_server {
362            if a2a_cfg.enabled {
363                let port = a2a_cfg.port;
364                tokio::spawn(crate::a2a_server::start_a2a_server(port));
365            }
366        }
367        #[cfg(not(feature = "a2a-server"))]
368        if let Some(a2a_cfg) = &self.config.a2a_server {
369            if a2a_cfg.enabled {
370                eprintln!("  ⚠️  A2A server configured but 'a2a-server' feature not enabled");
371            }
372        }
373
374        for test in tests {
375            eprintln!("\n╔══════════════════════════════");
376            eprintln!("║  Test: {}", test.name);
377            eprintln!("╚══════════════════════════════");
378
379            self.usage.reset_per_test();
380
381            let test_result = self.run_test(test, &tab);
382            self.usage.commit_test(&test.name);
383
384            if test_result.failed == 0 && test_result.total > 0 {
385                report.tests_passed += 1;
386                eprintln!("  Test ✅ Passed");
387            } else if test_result.total > 0 {
388                report.tests_failed += 1;
389                eprintln!("  Test ❌ Failed");
390            }
391
392            report.passed += test_result.passed;
393            report.failed += test_result.failed;
394            report.skipped += test_result.skipped;
395            report.details.extend(test_result.details);
396        }
397
398        Ok(report)
399    }
400
401    #[allow(clippy::too_many_lines)]
402    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
403        let base_url = test
404            .base_url
405            .clone()
406            .or_else(|| self.config.base_url.clone())
407            .unwrap_or_else(crate::base_url);
408
409        // Per-test viewport override: switch the browser via CDP
410        // device-metrics emulation before this test runs.
411        let vw = test.viewport_width.unwrap_or(self.viewport_width);
412        let vh = test.viewport_height.unwrap_or(self.viewport_height);
413        if self.applied_viewport.get() != (vw, vh) {
414            self.apply_viewport(tab, vw, vh);
415            self.applied_viewport.set((vw, vh));
416        }
417
418        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
419
420        let start_url = test
421            .start_url
422            .clone()
423            .or_else(|| self.config.start_url.clone())
424            .unwrap_or_else(|| "/dashboard".to_owned());
425
426        if auto_navigate {
427            let full_url = resolve_url(&start_url, &base_url);
428            eprintln!("  → auto-navigate: {full_url}");
429            let _ = tab.navigate_to(&full_url);
430            let _ = tab.wait_until_navigated();
431            std::thread::sleep(Duration::from_secs(4));
432        }
433
434        let mut result = TestRunResult::default();
435
436        for (step_index, step) in test.steps.iter().enumerate() {
437            result.total += 1;
438
439            let wait_ms = match step {
440                TestStep::Navigate { wait_after_ms, .. }
441                | TestStep::Click { wait_after_ms, .. }
442                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
443                _ => None,
444            };
445
446            let mut step_result = match step {
447                TestStep::Navigate { url, .. } => {
448                    let full_url = resolve_url(url, &base_url);
449                    run_navigate_step(&full_url, tab)
450                }
451                TestStep::Login {
452                    url,
453                    email,
454                    password,
455                    wait_after_ms,
456                } => run_login_step(
457                    &resolve_url(url, &base_url),
458                    email,
459                    password,
460                    *wait_after_ms,
461                    tab,
462                ),
463                TestStep::Click {
464                    target,
465                    selector,
466                    endpoint,
467                    ..
468                } => self.run_click(
469                    target,
470                    selector.as_deref(),
471                    endpoint.as_deref(),
472                    test.endpoint.as_deref(),
473                    tab,
474                ),
475                TestStep::Type {
476                    target,
477                    text,
478                    selector,
479                    endpoint,
480                    ..
481                } => self.run_type(
482                    target,
483                    text,
484                    selector.as_deref(),
485                    endpoint.as_deref(),
486                    test.endpoint.as_deref(),
487                    tab,
488                ),
489                TestStep::Wait {
490                    target,
491                    selector,
492                    text,
493                    timeout_ms,
494                    endpoint,
495                } => self.run_wait(
496                    target,
497                    selector.as_deref(),
498                    text.as_deref(),
499                    *timeout_ms,
500                    endpoint.as_deref(),
501                    test.endpoint.as_deref(),
502                    tab,
503                ),
504                TestStep::Assert {
505                    definition,
506                    preset,
507                    prompt,
508                    assert_text,
509                    endpoint,
510                    screenshot,
511                } => self.run_assert(
512                    definition.as_deref(),
513                    preset.as_deref(),
514                    prompt.as_deref(),
515                    assert_text.as_deref(),
516                    *screenshot,
517                    endpoint.as_deref(),
518                    test.endpoint.as_deref(),
519                    tab,
520                ),
521                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
522                TestStep::Agent {
523                    agent,
524                    task,
525                    definition,
526                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
527                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
528            };
529
530            // Failure diagnostics: capture the page state and a screenshot so
531            // CI logs say WHAT the page looked like when the step failed,
532            // instead of a bare "timed out: The event waited for never came".
533            if step_result.status == StepStatus::Failed {
534                let state = diagnostics::capture(tab);
535                let screenshot = diagnostics::save_screenshot(
536                    tab,
537                    &self.artifacts_dir,
538                    &test.name,
539                    &test.name,
540                    step_index,
541                    step_kind_label(step),
542                );
543                step_result.message = format!(
544                    "{base} — {excerpt}",
545                    base = step_result.message,
546                    excerpt = diagnostics::inline_excerpt(&state),
547                );
548                eprintln!(
549                    "{}",
550                    diagnostics::full_context(&state, screenshot.as_deref())
551                );
552            }
553
554            eprintln!(
555                "    {} {} — {}",
556                if step_result.status == StepStatus::Passed {
557                    "✅"
558                } else if step_result.status == StepStatus::Failed {
559                    "❌"
560                } else {
561                    "⏭️"
562                },
563                step_result.name,
564                step_result.message,
565            );
566
567            match step_result.status {
568                StepStatus::Passed => result.passed += 1,
569                StepStatus::Failed => result.failed += 1,
570                StepStatus::Skipped => result.skipped += 1,
571            }
572
573            // Fail fast: the first failed step ends the test and the
574            // remaining steps are reported as skipped (no LLM budget is
575            // burned asserting against a page that is already known broken).
576            if step_result.status == StepStatus::Failed
577                && !self.config.continue_on_failure
578                && step_index + 1 < test.steps.len()
579            {
580                eprintln!(
581                    "      ⏭️  failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
582                    test.steps.len() - step_index - 1
583                );
584                for skipped in &test.steps[step_index + 1..] {
585                    result.total += 1;
586                    result.skipped += 1;
587                    eprintln!(
588                        "    ⏭️  {} — skipped: previous step failed",
589                        step_label(skipped)
590                    );
591                    result.details.push(StepResult {
592                        name: step_label(skipped),
593                        status: StepStatus::Skipped,
594                        message: "skipped: previous step failed".into(),
595                    });
596                }
597                result.details.push(step_result);
598                return result;
599            }
600
601            // Check per-test budget after each step
602            let test_usage = self.usage.current_test_snapshot();
603            let global_usage = self.usage.global_snapshot();
604            let budget_status = self.budgets.check_all(
605                &test.name,
606                &test_usage,
607                &global_usage,
608                test.budget.as_ref(),
609            );
610            match budget_status {
611                BudgetStatus::HardExceeded { message, .. } => {
612                    crate::reporting::print_budget_error(&message);
613                    result.details.push(StepResult {
614                        name: "[budget]".into(),
615                        status: StepStatus::Failed,
616                        message,
617                    });
618                    result.failed += 1;
619                    return result;
620                }
621                BudgetStatus::SoftExceeded { message, .. } => {
622                    crate::reporting::print_budget_warning(&message);
623                }
624                BudgetStatus::Ok => {}
625            }
626
627            if let Some(ms) = wait_ms {
628                std::thread::sleep(Duration::from_millis(ms));
629            }
630
631            result.details.push(step_result);
632        }
633
634        result
635    }
636
637    /// Applies a viewport size to the current tab via CDP
638    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
639    /// overrides and the viewport matrix. The initial window size set at
640    /// browser launch is replaced by emulation; failures are logged but
641    /// do not fail the test (a mismatched viewport only weakens coverage).
642    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
643        use headless_chrome::protocol::cdp::Emulation;
644        let _ = self;
645        let params = Emulation::SetDeviceMetricsOverride {
646            width,
647            height,
648            device_scale_factor: 1.0,
649            mobile: false,
650            scale: None,
651            screen_width: Some(width),
652            screen_height: Some(height),
653            position_x: None,
654            position_y: None,
655            dont_set_visible_size: None,
656            screen_orientation: None,
657            viewport: None,
658            display_feature: None,
659            device_posture: None,
660        };
661        eprintln!("      ↻ viewport: {width}x{height}");
662        if let Err(e) = tab.call_method(params) {
663            eprintln!("      ⚠️  viewport switch to {width}x{height} failed: {e}");
664        }
665    }
666
667    // ── step handlers ───────────────────────────────────────────────────
668
669    fn run_click(
670        &self,
671        target: &str,
672        selector_override: Option<&str>,
673        step_endpoint: Option<&str>,
674        test_endpoint: Option<&str>,
675        tab: &Tab,
676    ) -> StepResult {
677        let selector = match self.resolve_selector(
678            selector_override,
679            target,
680            step_endpoint,
681            test_endpoint,
682            tab,
683        ) {
684            Ok(s) => s,
685            Err(msg) => {
686                return StepResult {
687                    name: format!("[click] {target}"),
688                    status: StepStatus::Failed,
689                    message: msg,
690                };
691            }
692        };
693
694        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(10)) {
695            Ok(element) => match element.click() {
696                Ok(_) => StepResult {
697                    name: format!("[click] {target}"),
698                    status: StepStatus::Passed,
699                    message: format!("clicked {selector}"),
700                },
701                Err(e) => StepResult {
702                    name: format!("[click] {target}"),
703                    status: StepStatus::Failed,
704                    message: format!("click failed on {selector}: {e}"),
705                },
706            },
707            Err(e) => StepResult {
708                name: format!("[click] {target}"),
709                status: StepStatus::Failed,
710                message: format!("element {selector} not found: {e}"),
711            },
712        }
713    }
714
715    #[allow(clippy::too_many_arguments)]
716    fn run_type(
717        &self,
718        target: &str,
719        text: &str,
720        selector_override: Option<&str>,
721        step_endpoint: Option<&str>,
722        test_endpoint: Option<&str>,
723        tab: &Tab,
724    ) -> StepResult {
725        let selector = match self.resolve_selector(
726            selector_override,
727            target,
728            step_endpoint,
729            test_endpoint,
730            tab,
731        ) {
732            Ok(s) => s,
733            Err(msg) => {
734                return StepResult {
735                    name: format!("[type] {target}"),
736                    status: StepStatus::Failed,
737                    message: msg,
738                };
739            }
740        };
741
742        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(10)) {
743            Ok(element) => {
744                if let Err(e) = element.click() {
745                    return StepResult {
746                        name: format!("[type] {target}"),
747                        status: StepStatus::Failed,
748                        message: format!("click to focus {selector} failed: {e}"),
749                    };
750                }
751
752                let js = format!(
753                    "document.querySelector('{}').value = '';",
754                    selector.replace('\'', "\\'")
755                );
756                let _ = tab.evaluate(&js, false);
757
758                match element.type_into(text) {
759                    Ok(_) => StepResult {
760                        name: format!("[type] {target}"),
761                        status: StepStatus::Passed,
762                        message: format!("typed {text:?} into {selector}"),
763                    },
764                    Err(e) => StepResult {
765                        name: format!("[type] {target}"),
766                        status: StepStatus::Failed,
767                        message: format!("type into {selector} failed: {e}"),
768                    },
769                }
770            }
771            Err(e) => StepResult {
772                name: format!("[type] {target}"),
773                status: StepStatus::Failed,
774                message: format!("element {selector} not found: {e}"),
775            },
776        }
777    }
778
779    #[allow(clippy::too_many_arguments)]
780    fn run_wait(
781        &self,
782        target: &str,
783        selector_override: Option<&str>,
784        text: Option<&str>,
785        timeout_ms: Option<u64>,
786        step_endpoint: Option<&str>,
787        test_endpoint: Option<&str>,
788        tab: &Tab,
789    ) -> StepResult {
790        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
791        let step_name = format!("[wait] {target}");
792
793        // Resolve an explicit selector only (text-only waits are LLM-free).
794        let selector = match selector_override {
795            Some(s) => Some(s.to_owned()),
796            None if text.is_some() => None,
797            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
798                Ok(s) => Some(s),
799                Err(msg) => {
800                    return StepResult {
801                        name: step_name,
802                        status: StepStatus::Failed,
803                        message: msg,
804                    };
805                }
806            },
807        };
808
809        if text.is_some() {
810            let sel_js = selector
811                .as_deref()
812                .map(crate::selectors::selector_matches_js);
813            let text_js = text.map(|t| {
814                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
815                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
816            });
817
818            let deadline = Instant::now() + timeout;
819            loop {
820                let sel_ok = sel_js
821                    .as_ref()
822                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
823                let text_ok = text_js
824                    .as_ref()
825                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
826                if sel_ok && text_ok {
827                    let mut what = Vec::new();
828                    if let Some(sel) = &selector {
829                        what.push(format!("found {sel}"));
830                    }
831                    if let Some(t) = text {
832                        what.push(format!("text {t:?} visible"));
833                    }
834                    return StepResult {
835                        name: step_name,
836                        status: StepStatus::Passed,
837                        message: what.join(" and "),
838                    };
839                }
840                if Instant::now() >= deadline {
841                    let mut what = Vec::new();
842                    if let Some(sel) = &selector {
843                        what.push(sel.clone());
844                    }
845                    if let Some(t) = text {
846                        what.push(format!("text {t:?}"));
847                    }
848                    return StepResult {
849                        name: step_name,
850                        status: StepStatus::Failed,
851                        message: format!(
852                            "wait for {} timed out after {}ms: the event waited for never came",
853                            what.join(" / "),
854                            timeout.as_millis(),
855                        ),
856                    };
857                }
858                std::thread::sleep(Duration::from_millis(250));
859            }
860        }
861
862        match selector.as_deref() {
863            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
864                Ok(_) => StepResult {
865                    name: step_name,
866                    status: StepStatus::Passed,
867                    message: format!("found {sel}"),
868                },
869                Err(e) => StepResult {
870                    name: step_name,
871                    status: StepStatus::Failed,
872                    message: format!(
873                        "wait for {sel} timed out after {}ms: {e}",
874                        timeout.as_millis()
875                    ),
876                },
877            },
878            None => StepResult {
879                name: step_name,
880                status: StepStatus::Failed,
881                message: "wait step has neither selector nor text".into(),
882            },
883        }
884    }
885
886    #[allow(clippy::too_many_arguments)]
887    fn run_assert(
888        &self,
889        definition: Option<&str>,
890        preset: Option<&str>,
891        prompt: Option<&str>,
892        assert_text: Option<&str>,
893        screenshot: bool,
894        step_endpoint: Option<&str>,
895        test_endpoint: Option<&str>,
896        tab: &Tab,
897    ) -> StepResult {
898        std::thread::sleep(Duration::from_millis(500));
899
900        let page_content = get_page_text(tab);
901
902        // Vision attach: capture the viewport once per assert step and hand
903        // the JPEG data URL to the preset/prompt evaluation below.
904        let image = if screenshot {
905            let endpoint = self
906                .endpoints
907                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
908            if !endpoint.vision {
909                return StepResult {
910                    name: "[assert]".into(),
911                    status: StepStatus::Failed,
912                    message: format!(
913                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
914                        name = endpoint.name
915                    ),
916                };
917            }
918            match crate::vision::capture_screenshot_data_url(
919                tab,
920                self.config
921                    .screenshot_max_dimension
922                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
923            ) {
924                Ok(data_url) => Some(data_url),
925                Err(e) => {
926                    return StepResult {
927                        name: "[assert]".into(),
928                        status: StepStatus::Failed,
929                        message: format!("screenshot capture failed: {e}"),
930                    };
931                }
932            }
933        } else {
934            None
935        };
936
937        if let Some(def_name) = definition {
938            if let Some(def) = self.definitions.get(def_name) {
939                return self.run_assert_def(
940                    def,
941                    &page_content,
942                    image.as_deref(),
943                    step_endpoint,
944                    test_endpoint,
945                    tab,
946                );
947            }
948            return StepResult {
949                name: format!("[assert] {def_name}"),
950                status: StepStatus::Failed,
951                message: format!("definition '{def_name}' not found"),
952            };
953        }
954
955        if let Some(preset_name) = preset {
956            // Deterministic DOM layout scan — runs JS in the browser and
957            // never calls the LLM (free, fast, no pixel budget).
958            if preset_name == "layout_no_issues" {
959                return self.run_layout_preset(tab);
960            }
961            return self.run_preset(
962                preset_name,
963                assert_text,
964                &page_content,
965                image.as_deref(),
966                step_endpoint,
967                test_endpoint,
968            );
969        }
970
971        if let Some(prompt_text) = prompt {
972            return self.run_custom(
973                prompt_text,
974                &page_content,
975                image.as_deref(),
976                step_endpoint,
977                test_endpoint,
978            );
979        }
980
981        StepResult {
982            name: "[assert]".into(),
983            status: StepStatus::Skipped,
984            message: "no definition, preset, or prompt specified".into(),
985        }
986    }
987
988    fn run_assert_def(
989        &self,
990        def: &AssertDefinition,
991        page_content: &PageContent,
992        image: Option<&str>,
993        step_endpoint: Option<&str>,
994        test_endpoint: Option<&str>,
995        tab: &Tab,
996    ) -> StepResult {
997        // Agent-based definition: delegate to an A2A agent
998        if let Some(ref agent) = def.agent {
999            if image.is_some() {
1000                return StepResult {
1001                    name: format!("[assert] {}", def.name),
1002                    status: StepStatus::Failed,
1003                    message: "agent-backed assertions do not support screenshots".into(),
1004                };
1005            }
1006            let task = def
1007                .task_template
1008                .as_deref()
1009                .unwrap_or("Evaluate the assertion")
1010                .replace("{url}", &page_content.url)
1011                .replace("{title}", &page_content.title)
1012                .replace("{content}", &page_content.body_text)
1013                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1014
1015            return self.run_agent_step(agent, &task, &def.name);
1016        }
1017
1018        // Custom preset: system + user_template provided in the definition
1019        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1020            return self.run_custom_preset(
1021                &def.name,
1022                system,
1023                template,
1024                def.assert_text.as_deref(),
1025                page_content,
1026                image,
1027                step_endpoint,
1028                test_endpoint,
1029            );
1030        }
1031
1032        def.preset.as_ref().map_or_else(
1033            || {
1034                def.prompt.as_ref().map_or_else(
1035                    || StepResult {
1036                        name: format!("[assert] {}", def.name),
1037                        status: StepStatus::Failed,
1038                        message: "definition has no preset, prompt, or system+user_template".into(),
1039                    },
1040                    |prompt| {
1041                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1042                    },
1043                )
1044            },
1045            |preset_name| {
1046                if preset_name == "layout_no_issues" {
1047                    return self.run_layout_preset(tab);
1048                }
1049                self.run_preset(
1050                    preset_name,
1051                    def.assert_text.as_deref(),
1052                    page_content,
1053                    image,
1054                    step_endpoint,
1055                    test_endpoint,
1056                )
1057            },
1058        )
1059    }
1060
1061    #[allow(clippy::too_many_arguments)]
1062    fn run_custom_preset(
1063        &self,
1064        name: &str,
1065        system: &str,
1066        template: &str,
1067        assert_text: Option<&str>,
1068        page_content: &PageContent,
1069        image: Option<&str>,
1070        step_endpoint: Option<&str>,
1071        test_endpoint: Option<&str>,
1072    ) -> StepResult {
1073        let user_prompt = template
1074            .replace("{url}", &page_content.url)
1075            .replace("{title}", &page_content.title)
1076            .replace("{content}", &page_content.body_text)
1077            .replace("{expected_text}", assert_text.unwrap_or(""))
1078            .replace("{description}", "");
1079
1080        // Custom preset definitions frequently forget the {content}
1081        // placeholder — without it the LLM has no page to evaluate and
1082        // answers "I can't determine that without seeing the page". Always
1083        // append the page context unless the template already references it.
1084        let user_prompt = if template.contains("{content}") {
1085            user_prompt
1086        } else {
1087            format!(
1088                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1089                url = page_content.url,
1090                title = page_content.title,
1091                content = page_content.body_text,
1092            )
1093        };
1094
1095        eprintln!("      assert: {name} (custom preset)");
1096
1097        let endpoint = self
1098            .endpoints
1099            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1100        let llm = self.build_llm_for_endpoint(endpoint);
1101        let usage = Arc::clone(&self.usage);
1102        let endpoint_name = endpoint.name.clone();
1103        let sys = system.to_owned();
1104        let image = image.map(str::to_owned);
1105
1106        let response = std::thread::spawn(move || {
1107            let rt = tokio::runtime::Builder::new_current_thread()
1108                .enable_all()
1109                .build()
1110                .unwrap();
1111            let call = async {
1112                match image.as_deref() {
1113                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user_prompt, img).await,
1114                    None => llm_chat_with_usage(&llm, &sys, &user_prompt).await,
1115                }
1116            };
1117            rt.block_on(call)
1118        })
1119        .join()
1120        .unwrap();
1121
1122        response.map_or_else(
1123            |e| StepResult {
1124                name: format!("[assert] {name}"),
1125                status: StepStatus::Failed,
1126                message: format!("LLM assertion call failed: {e}"),
1127            },
1128            |lr| {
1129                usage.record_llm_call(
1130                    &endpoint_name,
1131                    endpoint,
1132                    lr.usage.prompt_tokens,
1133                    lr.usage.completion_tokens,
1134                );
1135                let content_lower = lr.content.to_lowercase().trim().to_owned();
1136                if content_lower.starts_with("pass") {
1137                    StepResult {
1138                        name: format!("[assert] {name}"),
1139                        status: StepStatus::Passed,
1140                        message: "PASS".into(),
1141                    }
1142                } else {
1143                    StepResult {
1144                        name: format!("[assert] {name}"),
1145                        status: StepStatus::Failed,
1146                        message: lr.content,
1147                    }
1148                }
1149            },
1150        )
1151    }
1152
1153    fn run_preset(
1154        &self,
1155        preset_name: &str,
1156        assert_text: Option<&str>,
1157        page_content: &PageContent,
1158        image: Option<&str>,
1159        step_endpoint: Option<&str>,
1160        test_endpoint: Option<&str>,
1161    ) -> StepResult {
1162        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1163            return StepResult {
1164                name: format!("[assert] {preset_name}"),
1165                status: StepStatus::Failed,
1166                message: format!("unknown assertion preset: {preset_name}"),
1167            };
1168        };
1169        if preset_name.starts_with("visual_") && image.is_none() {
1170            return StepResult {
1171                name: format!("[assert] {preset_name}"),
1172                status: StepStatus::Failed,
1173                message: format!(
1174                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1175                ),
1176            };
1177        }
1178
1179        let user_prompt = preset
1180            .user_template
1181            .replace("{url}", &page_content.url)
1182            .replace("{title}", &page_content.title)
1183            .replace("{content}", &page_content.body_text)
1184            .replace("{expected_text}", assert_text.unwrap_or(""))
1185            .replace("{description}", "");
1186
1187        // Same safety net as custom presets: never let the LLM answer with
1188        // no page context at all.
1189        let user_prompt = if preset.user_template.contains("{content}") {
1190            user_prompt
1191        } else {
1192            format!(
1193                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1194                url = page_content.url,
1195                title = page_content.title,
1196                content = page_content.body_text,
1197            )
1198        };
1199
1200        eprintln!("      assert: {preset_name}");
1201
1202        let endpoint = self
1203            .endpoints
1204            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1205        let llm = self.build_llm_for_endpoint(endpoint);
1206        let usage = Arc::clone(&self.usage);
1207        let endpoint_name = endpoint.name.clone();
1208        let sys = preset.system.to_owned();
1209        let image = image.map(str::to_owned);
1210
1211        let response = std::thread::spawn(move || {
1212            let rt = tokio::runtime::Builder::new_current_thread()
1213                .enable_all()
1214                .build()
1215                .unwrap();
1216            let call = async {
1217                match image.as_deref() {
1218                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user_prompt, img).await,
1219                    None => llm_chat_with_usage(&llm, &sys, &user_prompt).await,
1220                }
1221            };
1222            rt.block_on(call)
1223        })
1224        .join()
1225        .unwrap();
1226
1227        response.map_or_else(
1228            |e| StepResult {
1229                name: format!("[assert] {preset_name}"),
1230                status: StepStatus::Failed,
1231                message: format!("LLM assertion call failed: {e}"),
1232            },
1233            |lr| {
1234                usage.record_llm_call(
1235                    &endpoint_name,
1236                    endpoint,
1237                    lr.usage.prompt_tokens,
1238                    lr.usage.completion_tokens,
1239                );
1240                let content_lower = lr.content.to_lowercase().trim().to_owned();
1241                if content_lower.starts_with("pass") {
1242                    StepResult {
1243                        name: format!("[assert] {preset_name}"),
1244                        status: StepStatus::Passed,
1245                        message: "PASS".into(),
1246                    }
1247                } else {
1248                    StepResult {
1249                        name: format!("[assert] {preset_name}"),
1250                        status: StepStatus::Failed,
1251                        message: lr.content,
1252                    }
1253                }
1254            },
1255        )
1256    }
1257
1258    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1259    ///
1260    /// Evaluates the layout-scan JS in the page and fails with the list of
1261    /// detected issues: horizontal page overflow, visible elements sticking
1262    /// out of the viewport, text clipped by `overflow: hidden` containers,
1263    /// and interactive elements covered by other elements. No LLM call —
1264    /// checks are geometry-based so the check is free, deterministic, and
1265    /// safe to run on every page × viewport variant.
1266    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1267        let _ = self;
1268        let name = "[assert] layout_no_issues".to_owned();
1269        eprintln!("      assert: layout_no_issues (DOM layout scan)");
1270        let result = tab.evaluate(LAYOUT_SCAN_JS, false);
1271        let json_str = match result {
1272            Ok(r) => r
1273                .value
1274                .as_ref()
1275                .and_then(|v| v.as_str().map(String::from))
1276                .unwrap_or_else(|| "[]".to_owned()),
1277            Err(e) => {
1278                return StepResult {
1279                    name,
1280                    status: StepStatus::Failed,
1281                    message: format!("layout scan JS failed: {e}"),
1282                };
1283            }
1284        };
1285        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1286        if issues.is_empty() {
1287            return StepResult {
1288                name,
1289                status: StepStatus::Passed,
1290                message: "PASS — no layout defects detected".into(),
1291            };
1292        }
1293        let mut lines: Vec<String> = issues
1294            .iter()
1295            .take(10)
1296            .map(|i| {
1297                format!(
1298                    "- [{type_}] {element}: {detail}",
1299                    type_ = i.issue_type,
1300                    element = i.element,
1301                    detail = i.detail
1302                )
1303            })
1304            .collect();
1305        if issues.len() > 10 {
1306            lines.push(format!("- … and {} more", issues.len() - 10));
1307        }
1308        StepResult {
1309            name,
1310            status: StepStatus::Failed,
1311            message: format!(
1312                "FAIL — {} layout defect(s) detected:\n{}",
1313                issues.len(),
1314                lines.join("\n")
1315            ),
1316        }
1317    }
1318
1319    fn run_custom(
1320        &self,
1321        prompt: &str,
1322        page_content: &PageContent,
1323        image: Option<&str>,
1324        step_endpoint: Option<&str>,
1325        test_endpoint: Option<&str>,
1326    ) -> StepResult {
1327        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1328
1329        let mut user = format!(
1330            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1331            url = page_content.url,
1332            title = page_content.title,
1333            content = page_content.body_text,
1334        );
1335        if image.is_some() {
1336            user.push_str(
1337                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1338            );
1339        }
1340
1341        eprintln!("      custom assert");
1342
1343        let endpoint = self
1344            .endpoints
1345            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1346        let llm = self.build_llm_for_endpoint(endpoint);
1347        let usage = Arc::clone(&self.usage);
1348        let endpoint_name = endpoint.name.clone();
1349        let sys = system.to_owned();
1350        let image = image.map(str::to_owned);
1351
1352        let response = std::thread::spawn(move || {
1353            let rt = tokio::runtime::Builder::new_current_thread()
1354                .enable_all()
1355                .build()
1356                .unwrap();
1357            let call = async {
1358                match image.as_deref() {
1359                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user, img).await,
1360                    None => llm_chat_with_usage(&llm, &sys, &user).await,
1361                }
1362            };
1363            rt.block_on(call)
1364        })
1365        .join()
1366        .unwrap();
1367
1368        response.map_or_else(
1369            |e| StepResult {
1370                name: "[assert] custom".into(),
1371                status: StepStatus::Failed,
1372                message: format!("LLM assertion call failed: {e}"),
1373            },
1374            |lr| {
1375                usage.record_llm_call(
1376                    &endpoint_name,
1377                    endpoint,
1378                    lr.usage.prompt_tokens,
1379                    lr.usage.completion_tokens,
1380                );
1381                let content_lower = lr.content.to_lowercase().trim().to_owned();
1382                if content_lower.starts_with("pass") {
1383                    StepResult {
1384                        name: "[assert] custom".into(),
1385                        status: StepStatus::Passed,
1386                        message: "PASS".into(),
1387                    }
1388                } else {
1389                    StepResult {
1390                        name: "[assert] custom".into(),
1391                        status: StepStatus::Failed,
1392                        message: lr.content,
1393                    }
1394                }
1395            },
1396        )
1397    }
1398
1399    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1400        let path = path.unwrap_or("screenshot.png");
1401
1402        match tab.capture_screenshot(
1403            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1404            None,
1405            None,
1406            true,
1407        ) {
1408            Ok(data) => {
1409                if let Err(e) = std::fs::write(path, &data) {
1410                    return StepResult {
1411                        name: format!("[screenshot] {path}"),
1412                        status: StepStatus::Failed,
1413                        message: format!("failed to write screenshot: {e}"),
1414                    };
1415                }
1416                StepResult {
1417                    name: format!("[screenshot] {path}"),
1418                    status: StepStatus::Passed,
1419                    message: format!("saved to {path}"),
1420                }
1421            }
1422            Err(e) => StepResult {
1423                name: format!("[screenshot] {path}"),
1424                status: StepStatus::Failed,
1425                message: format!("screenshot failed: {e}"),
1426            },
1427        }
1428    }
1429
1430    /// Runs an A2A agent step.
1431    #[allow(clippy::literal_string_with_formatting_args)]
1432    fn run_agent(
1433        &self,
1434        agent_name: &str,
1435        task: &str,
1436        definition: Option<&str>,
1437        _test_endpoint: Option<&str>,
1438    ) -> StepResult {
1439        // If a definition is specified, look up the task template
1440        let resolved_task = if let Some(def_name) = definition {
1441            if let Some(def) = self.definitions.get(def_name) {
1442                let tmpl = def.task_template.as_deref().unwrap_or(task);
1443                tmpl.replace("{task}", task)
1444            } else {
1445                return StepResult {
1446                    name: format!("[agent] {def_name}"),
1447                    status: StepStatus::Failed,
1448                    message: format!("definition '{def_name}' not found"),
1449                };
1450            }
1451        } else {
1452            task.to_owned()
1453        };
1454
1455        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1456    }
1457
1458    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1459        let Some(ep) = self.endpoints.get(agent_name) else {
1460            return StepResult {
1461                name: format!("[agent] {display_name}"),
1462                status: StepStatus::Failed,
1463                message: format!("agent endpoint '{agent_name}' not found"),
1464            };
1465        };
1466
1467        if ep.url.is_empty() {
1468            return StepResult {
1469                name: format!("[agent] {display_name}"),
1470                status: StepStatus::Failed,
1471                message: format!("agent endpoint '{agent_name}' has no URL"),
1472            };
1473        }
1474
1475        eprintln!("      → agent {agent_name}: {task}");
1476
1477        let url = ep.url.clone();
1478        let client = A2aClient::new(&url, self.timeout);
1479        let task_clone = task.to_owned();
1480
1481        let response = std::thread::spawn(move || {
1482            let rt = tokio::runtime::Builder::new_current_thread()
1483                .enable_all()
1484                .build()
1485                .unwrap();
1486            rt.block_on(client.send_task(&task_clone))
1487        })
1488        .join()
1489        .unwrap();
1490
1491        // Record the flat-cost call
1492        self.usage.record_flat_call(agent_name, ep);
1493
1494        match response {
1495            Ok(text) => {
1496                let clean = text.trim().to_owned();
1497                let lower = clean.to_lowercase();
1498                if lower.starts_with("pass") {
1499                    StepResult {
1500                        name: format!("[agent] {display_name}"),
1501                        status: StepStatus::Passed,
1502                        message: format!("PASS: {clean}"),
1503                    }
1504                } else if lower.starts_with("fail") {
1505                    StepResult {
1506                        name: format!("[agent] {display_name}"),
1507                        status: StepStatus::Failed,
1508                        message: clean,
1509                    }
1510                } else {
1511                    StepResult {
1512                        name: format!("[agent] {display_name}"),
1513                        status: StepStatus::Passed,
1514                        message: format!("response: {clean}"),
1515                    }
1516                }
1517            }
1518            Err(e) => StepResult {
1519                name: format!("[agent] {display_name}"),
1520                status: StepStatus::Failed,
1521                message: format!("agent call failed: {e}"),
1522            },
1523        }
1524    }
1525
1526    /// Runs an MCP tool call step.
1527    fn run_mcp(
1528        &self,
1529        server_name: &str,
1530        tool_name: &str,
1531        args: Option<&serde_json::Value>,
1532    ) -> StepResult {
1533        let Some(ep) = self.endpoints.get(server_name) else {
1534            return StepResult {
1535                name: format!("[mcp] {server_name}:{tool_name}"),
1536                status: StepStatus::Failed,
1537                message: format!("MCP server endpoint '{server_name}' not found"),
1538            };
1539        };
1540
1541        let cmd = ep.command.as_deref().unwrap_or("");
1542        if cmd.is_empty() {
1543            return StepResult {
1544                name: format!("[mcp] {server_name}:{tool_name}"),
1545                status: StepStatus::Failed,
1546                message: format!("MCP server '{server_name}' has no command configured"),
1547            };
1548        }
1549
1550        eprintln!("      → mcp {server_name} {tool_name}");
1551
1552        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1553
1554        let command = cmd.to_owned();
1555        let args_vec = ep.args.clone();
1556        let tool = tool_name.to_owned();
1557
1558        let response = std::thread::spawn(move || {
1559            let mut mcp_client =
1560                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1561            mcp_client
1562                .call_tool(&tool, &args_val)
1563                .map_err(|e| e.to_string())
1564        })
1565        .join()
1566        .unwrap();
1567
1568        // Record the flat-cost call
1569        self.usage.record_flat_call(server_name, ep);
1570
1571        match response {
1572            Ok(result) => {
1573                if result.isError {
1574                    StepResult {
1575                        name: format!("[mcp] {server_name}:{tool_name}"),
1576                        status: StepStatus::Failed,
1577                        message: result.to_string(),
1578                    }
1579                } else {
1580                    StepResult {
1581                        name: format!("[mcp] {server_name}:{tool_name}"),
1582                        status: StepStatus::Passed,
1583                        message: result.to_string(),
1584                    }
1585                }
1586            }
1587            Err(e) => StepResult {
1588                name: format!("[mcp] {server_name}:{tool_name}"),
1589                status: StepStatus::Failed,
1590                message: format!("MCP call failed: {e}"),
1591            },
1592        }
1593    }
1594
1595    // ── helpers ──────────────────────────────────────────────────────────
1596
1597    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1598    /// the runner's default LLM config for any unset fields.
1599    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1600        LlmConfig {
1601            url: if endpoint.url.is_empty() {
1602                self.llm.url.clone()
1603            } else {
1604                endpoint.url.clone()
1605            },
1606            model: endpoint
1607                .model
1608                .clone()
1609                .unwrap_or_else(|| self.llm.model.clone()),
1610            api_key: endpoint
1611                .api_key
1612                .clone()
1613                .or_else(|| self.llm.api_key.clone()),
1614            headers: if endpoint.headers.is_empty() {
1615                self.llm.headers.clone()
1616            } else {
1617                endpoint.headers.clone()
1618            },
1619            timeout: self.llm.timeout,
1620            temperature: self.llm.temperature,
1621            thinking: self.llm.thinking,
1622            model_params: self.llm.model_params.clone(),
1623        }
1624    }
1625
1626    /// Resolves a CSS selector for the target element. Uses the explicit
1627    /// `selector` if provided, otherwise asks the LLM to find the element
1628    /// from the natural language `target` description and page DOM.
1629    ///
1630    /// LLM responses are sanitized and verified against the live page: a
1631    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
1632    /// immediately with the raw LLM output, and a selector that matches
1633    /// nothing triggers one retry with feedback before failing.
1634    #[allow(clippy::too_many_lines)]
1635    fn resolve_selector(
1636        &self,
1637        css_override: Option<&str>,
1638        target: &str,
1639        step_endpoint: Option<&str>,
1640        test_endpoint: Option<&str>,
1641        tab: &Tab,
1642    ) -> Result<String, String> {
1643        if let Some(explicit) = css_override {
1644            return Ok(explicit.to_owned());
1645        }
1646
1647        let dom_info = extract_dom_info(tab)?;
1648        let page_content = get_page_text(tab);
1649
1650        let system = concat!(
1651            "You are a browser automation selector generator. ",
1652            "Given a web page's content and interactive elements, ",
1653            "return ONLY the best CSS selector for the described element. ",
1654            "Output nothing except the CSS selector. ",
1655            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
1656            "[name=\"...\"], tag.class, tag. ",
1657            "Never output explanations, markdown, or extra text."
1658        );
1659
1660        let user = format!(
1661            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
1662            page_content.url,
1663            page_content.title,
1664            truncate(&page_content.body_text, 4000),
1665            dom_info,
1666            target,
1667        );
1668
1669        let retry_user = format!(
1670            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
1671            "The selector must match at least one element currently present on the page.",
1672            page_content.url,
1673            page_content.title,
1674            truncate(&page_content.body_text, 4000),
1675            dom_info,
1676            target,
1677        );
1678
1679        eprintln!("      LLM targeting: {target}");
1680
1681        let endpoint = self
1682            .endpoints
1683            .resolve(step_endpoint.or(test_endpoint), TaskType::Targeting);
1684        let llm = self.build_llm_for_endpoint(endpoint);
1685        let usage = Arc::clone(&self.usage);
1686        let endpoint_name = endpoint.name.clone();
1687        let endpoint_clone = endpoint.clone();
1688        let sys = system.to_owned();
1689
1690        let call_llm = |prompt: &str| {
1691            let llm = llm.clone();
1692            let sys = sys.clone();
1693            let prompt = prompt.to_owned();
1694            std::thread::spawn(move || {
1695                let rt = tokio::runtime::Builder::new_current_thread()
1696                    .enable_all()
1697                    .build()
1698                    .unwrap();
1699                rt.block_on(llm_chat_with_usage(&llm, &sys, &prompt))
1700            })
1701            .join()
1702            .unwrap()
1703        };
1704
1705        let first = call_llm(&user);
1706        let lr = match first {
1707            Ok(lr) => lr,
1708            Err(e) => {
1709                return Err(format!("LLM element targeting failed: {e}"));
1710            }
1711        };
1712        usage.record_llm_call(
1713            &endpoint_name,
1714            &endpoint_clone,
1715            lr.usage.prompt_tokens,
1716            lr.usage.completion_tokens,
1717        );
1718        let clean = sanitize_selector(&lr.content);
1719        eprintln!("      resolved selector: {clean}");
1720
1721        if selector_is_useless(&clean) {
1722            return Err(format!(
1723                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
1724                raw = lr.content.trim(),
1725            ));
1726        }
1727        if let Err(reason) = validate_selector(&clean) {
1728            return Err(format!(
1729                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
1730                raw = lr.content.trim(),
1731            ));
1732        }
1733        if !selector_matches(tab, &clean).unwrap_or(false) {
1734            // One retry with feedback: flaky models occasionally invent a
1735            // selector that does not exist on the page.
1736            eprintln!(
1737                "      selector {clean} matches nothing — retrying LLM targeting with feedback"
1738            );
1739            let second = call_llm(&retry_user);
1740            let lr2 = match second {
1741                Ok(lr2) => lr2,
1742                Err(e) => {
1743                    return Err(format!(
1744                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
1745                    ));
1746                }
1747            };
1748            usage.record_llm_call(
1749                &endpoint_name,
1750                &endpoint_clone,
1751                lr2.usage.prompt_tokens,
1752                lr2.usage.completion_tokens,
1753            );
1754            let clean2 = sanitize_selector(&lr2.content);
1755            eprintln!("      resolved selector (retry): {clean2}");
1756            if selector_is_useless(&clean2) {
1757                return Err(format!(
1758                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
1759                    raw = lr2.content.trim(),
1760                    excerpt = truncate(&page_content.body_text, 300),
1761                ));
1762            }
1763            if !selector_matches(tab, &clean2).unwrap_or(false) {
1764                return Err(format!(
1765                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
1766                ));
1767            }
1768            return Ok(clean2);
1769        }
1770
1771        Ok(clean)
1772    }
1773}
1774
1775/// Evaluates a JS expression that is expected to return a boolean.
1776fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
1777    tab.evaluate(js, false)
1778        .map_err(|e| format!("evaluate failed: {e}"))?
1779        .value
1780        .and_then(|v| v.as_bool())
1781        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
1782}
1783
1784/// Checks whether a CSS selector matches at least one current element.
1785fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
1786    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
1787}
1788
1789// ── Free helper functions ──────────────────────────────────────────────
1790
1791fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
1792    let name = format!("[navigate] {full_url}");
1793    match tab.navigate_to(full_url) {
1794        Ok(_) => {
1795            let _ = tab.wait_until_navigated();
1796            StepResult {
1797                name,
1798                status: StepStatus::Passed,
1799                message: format!("navigated to {full_url}"),
1800            }
1801        }
1802        Err(e) => StepResult {
1803            name,
1804            status: StepStatus::Failed,
1805            message: format!("navigation failed: {e}"),
1806        },
1807    }
1808}
1809
1810/// Idempotent login step: navigate to the login URL and, if a login
1811/// form is rendered, fill it, wait for the bot-protection token,
1812/// submit, and wait for the authenticated shell. When the app is
1813/// already authenticated the login form never appears and the step
1814/// passes silently — this keeps viewport-matrix variants and repeated
1815/// logins in one browser session green.
1816fn run_login_step(
1817    full_url: &str,
1818    email: &str,
1819    password: &str,
1820    wait_after_ms: Option<u64>,
1821    tab: &Tab,
1822) -> StepResult {
1823    let name = format!("[login] {full_url}");
1824    match tab.navigate_to(full_url) {
1825        Ok(_) => {
1826            let _ = tab.wait_until_navigated();
1827            std::thread::sleep(Duration::from_millis(3000));
1828        }
1829        Err(e) => {
1830            return StepResult {
1831                name,
1832                status: StepStatus::Failed,
1833                message: format!("navigation failed: {e}"),
1834            };
1835        }
1836    }
1837
1838    let form_present = eval_bool(tab, "document.querySelector('#email') !== null").unwrap_or(false);
1839    if !form_present {
1840        return StepResult {
1841            name,
1842            status: StepStatus::Passed,
1843            message: "already authenticated (no login form rendered)".into(),
1844        };
1845    }
1846
1847    for (selector, text) in [("#email", email), ("#password", password)] {
1848        match tab.wait_for_element_with_custom_timeout(selector, Duration::from_secs(5)) {
1849            Ok(element) => {
1850                let _ = element.click();
1851                let js = format!("document.querySelector('{selector}').value = '';");
1852                let _ = tab.evaluate(&js, false);
1853                if let Err(e) = element.type_into(text) {
1854                    return StepResult {
1855                        name,
1856                        status: StepStatus::Failed,
1857                        message: format!("typing into {selector} failed: {e}"),
1858                    };
1859                }
1860            }
1861            Err(e) => {
1862                return StepResult {
1863                    name,
1864                    status: StepStatus::Failed,
1865                    message: format!("login input {selector} not found: {e}"),
1866                };
1867            }
1868        }
1869        std::thread::sleep(Duration::from_millis(300));
1870    }
1871
1872    // Bot-protection / Turnstile token. The dev estate uses the
1873    // Turnstile test-mode widget; wait for the hidden field value.
1874    let token_ok = (0..20).any(|_| {
1875        std::thread::sleep(Duration::from_millis(1500));
1876        eval_bool(
1877            tab,
1878            "document.querySelector('input[name=\"cf-turnstile-response\"][value]:not([value=\"\"])') !== null",
1879        )
1880        .unwrap_or(false)
1881    });
1882    if !token_ok {
1883        return StepResult {
1884            name,
1885            status: StepStatus::Failed,
1886            message: "bot-protection token never appeared (is the Turnstile widget in test mode?)"
1887                .into(),
1888        };
1889    }
1890
1891    let sign_in_ok = match tab.wait_for_element_with_custom_timeout(
1892        "button.btn--landing.btn--primary",
1893        Duration::from_secs(5),
1894    ) {
1895        Ok(element) => element.click().is_ok(),
1896        Err(_) => false,
1897    };
1898    if !sign_in_ok {
1899        return StepResult {
1900            name,
1901            status: StepStatus::Failed,
1902            message: "sign-in button not found or not clickable".into(),
1903        };
1904    }
1905
1906    // Wait for the authenticated shell.
1907    let authed = (0..20).any(|_| {
1908        std::thread::sleep(Duration::from_millis(1500));
1909        eval_bool(tab, "document.querySelector('app-account-shell') !== null").unwrap_or(false)
1910    });
1911    if !authed {
1912        return StepResult {
1913            name,
1914            status: StepStatus::Failed,
1915            message:
1916                "login submitted but the authenticated shell never appeared (bad credentials?)"
1917                    .into(),
1918        };
1919    }
1920
1921    if let Some(ms) = wait_after_ms {
1922        std::thread::sleep(Duration::from_millis(ms));
1923    }
1924
1925    StepResult {
1926        name,
1927        status: StepStatus::Passed,
1928        message: "authenticated".into(),
1929    }
1930}
1931
1932fn extract_dom_info(tab: &Tab) -> Result<String, String> {
1933    let result = tab
1934        .evaluate(DOM_EXTRACT_JS, false)
1935        .map_err(|e| format!("DOM extraction failed: {e}"))?;
1936
1937    let json_str = result
1938        .value
1939        .as_ref()
1940        .and_then(|v| v.as_str())
1941        .unwrap_or("[]");
1942
1943    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
1944
1945    if elements.is_empty() {
1946        return Ok("(no interactive elements found)".to_owned());
1947    }
1948
1949    Ok(elements.join("\n"))
1950}
1951
1952fn get_page_text(tab: &Tab) -> PageContent {
1953    let url = tab.get_url();
1954
1955    let title = tab
1956        .evaluate("document.title", false)
1957        .ok()
1958        .and_then(|r| r.value)
1959        .and_then(|v| v.as_str().map(String::from))
1960        .unwrap_or_else(|| "unknown".to_owned());
1961
1962    let body_text = tab
1963        .evaluate(
1964            "document.body ? document.body.innerText : document.documentElement.innerText",
1965            false,
1966        )
1967        .ok()
1968        .and_then(|r| r.value)
1969        .and_then(|v| v.as_str().map(String::from))
1970        .unwrap_or_default();
1971
1972    PageContent {
1973        url,
1974        title,
1975        body_text: truncate(&body_text, 8000),
1976    }
1977}
1978
1979fn resolve_url(url: &str, base_url: &str) -> String {
1980    if url.starts_with("http://") || url.starts_with("https://") {
1981        return url.to_owned();
1982    }
1983    let base = base_url.trim_end_matches('/');
1984    if url.starts_with('/') {
1985        format!("{base}{url}")
1986    } else {
1987        format!("{base}/{url}")
1988    }
1989}
1990
1991/// Human-readable label for a step, used when steps are skipped after an
1992/// earlier failure.
1993fn step_label(step: &TestStep) -> String {
1994    match step {
1995        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
1996        TestStep::Login { url, .. } => format!("[login] {url}"),
1997        TestStep::Click { target, .. } => format!("[click] {target}"),
1998        TestStep::Type { target, .. } => format!("[type] {target}"),
1999        TestStep::Wait { target, .. } => format!("[wait] {target}"),
2000        TestStep::Assert {
2001            definition,
2002            preset,
2003            prompt,
2004            ..
2005        } => definition.as_ref().map_or_else(
2006            || {
2007                preset.as_ref().map_or_else(
2008                    || {
2009                        prompt.as_ref().map_or_else(
2010                            || "[assert]".to_owned(),
2011                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2012                        )
2013                    },
2014                    |p| format!("[assert] {p}"),
2015                )
2016            },
2017            |d| format!("[assert] {d}"),
2018        ),
2019        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2020        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2021        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2022    }
2023}
2024
2025/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
2026#[must_use]
2027const fn step_kind_label(step: &TestStep) -> &'static str {
2028    match step {
2029        TestStep::Navigate { .. } => "navigate",
2030        TestStep::Login { .. } => "login",
2031        TestStep::Click { .. } => "click",
2032        TestStep::Type { .. } => "type",
2033        TestStep::Wait { .. } => "wait",
2034        TestStep::Assert { .. } => "assert",
2035        TestStep::Screenshot { .. } => "screenshot",
2036        TestStep::Agent { .. } => "agent",
2037        TestStep::Mcp { .. } => "mcp",
2038    }
2039}
2040
2041// ── Support types ──────────────────────────────────────────────────────
2042
2043#[derive(Default)]
2044struct TestRunResult {
2045    passed: u32,
2046    failed: u32,
2047    skipped: u32,
2048    total: u32,
2049    details: Vec<StepResult>,
2050}
2051
2052struct PageContent {
2053    url: String,
2054    title: String,
2055    body_text: String,
2056}