Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::UsageTracker;
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, TaskType};
14use crate::llm_chat_vision_with_usage;
15use crate::llm_chat_with_usage;
16use crate::mcp_client::McpClient;
17use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
18use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
19use crate::truncate;
20use crate::LlmConfig;
21use crate::DOM_EXTRACT_JS;
22
23/// One detected layout defect (`layout_no_issues` preset).
24#[derive(Debug, serde::Deserialize)]
25struct LayoutIssue {
26    #[serde(rename = "type")]
27    issue_type: String,
28    element: String,
29    detail: String,
30}
31
32/// In-browser DOM layout scan for `layout_no_issues`.
33///
34/// Geometry-only checks (no LLM, no pixels):
35/// 1. page horizontal overflow (`scrollWidth` > viewport width);
36/// 2. visible, non-fixed elements that stick out of the viewport
37///    (right/bottom edge) while still partially on screen;
38/// 3. text clipped by `overflow: hidden` containers whose content
39///    is measurably larger than the box;
40/// 4. interactive elements (buttons/links/inputs) whose center point
41///    is covered by a different element that would intercept the click.
42///
43/// Intentional stacking (off-canvas drawers, dropdown menus, badges,
44/// fixed headers) is excluded by the position/relation filters.
45const LAYOUT_SCAN_JS: &str = r#"
46(() => {
47  const issues = [];
48  const push = (type, el, detail) => {
49    if (issues.length >= 30) return;
50    let element = el.tagName.toLowerCase();
51    if (el.id) element += '#' + el.id;
52    else if (typeof el.className === 'string' && el.className.trim())
53      element += '.' + el.className.trim().split(/\s+/).join('.');
54    issues.push({ type, element, detail: String(detail).slice(0, 220) });
55  };
56  const vw = document.documentElement.clientWidth || window.innerWidth;
57  const vh = document.documentElement.clientHeight || window.innerHeight;
58  if (!vw || !vh) return JSON.stringify(issues);
59  const de = document.documentElement;
60  // 1. Page-level horizontal overflow.
61  if (de.scrollWidth > vw + 2)
62    push('page-overflow-x', de,
63      'page scrollWidth ' + de.scrollWidth + ' exceeds viewport width ' + vw);
64  const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
65  const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
66  const hasContent = (el) =>
67    ((el.textContent || '').trim().length > 0) ||
68    !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
69  // 2. Elements sticking out of the viewport (partially visible only).
70  for (const el of all) {
71    const cs = getComputedStyle(el);
72    if (!visible(cs)) continue;
73    const r = el.getBoundingClientRect();
74    if (r.width < 2 || r.height < 2) continue;
75    if (cs.position === 'fixed' || cs.position === 'sticky') continue;
76    if (!hasContent(el) && el.children.length === 0) continue;
77    if (r.top >= vh || r.left >= vw) continue; // fully offscreen = normal scroll content
78    const overRight = r.right - vw;
79    const overBottom = r.bottom - vh;
80    if (overRight > 2 || overBottom > 2) {
81      let where = '';
82      if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
83      else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
84      else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
85      push('element-out-of-viewport', el, 'extends ' + Math.max(overRight, overBottom).toFixed(0) + 'px past the ' + where);
86    }
87  }
88  // 3. Text clipped by overflow:hidden containers.
89  for (const el of all) {
90    const cs = getComputedStyle(el);
91    if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
92    if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
93    if (!(el.textContent || '').trim()) continue;
94    push('text-clipped', el,
95      'content ' + el.scrollWidth + 'x' + el.scrollHeight +
96      ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
97  }
98  // 4. Interactive elements covered by a different element.
99  const interactive =
100    'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
101  const targets = document.querySelectorAll(interactive);
102  for (const el of targets) {
103    const r = el.getBoundingClientRect();
104    if (r.width < 6 || r.height < 6) continue;
105    const cs = getComputedStyle(el);
106    if (!visible(cs)) continue;
107    const cx = r.left + r.width / 2;
108    const cy = r.top + r.height / 2;
109    if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
110    const top = document.elementFromPoint(cx, cy);
111    if (!top || top === el || el.contains(top) || top.contains(el)) continue;
112    const tcs = getComputedStyle(top);
113    if (!visible(tcs)) continue;
114    if (tcs.pointerEvents === 'none') continue;
115    const tr = top.getBoundingClientRect();
116    if (tr.width * tr.height < r.width * r.height * 0.25) continue;
117    const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
118      (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
119    push('element-overlap', el,
120      'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
121      ' is covered by <' + tname + '>');
122  }
123  return JSON.stringify(issues);
124})()
125"#;
126
127/// How long the CDP connection stays open after the browser goes quiet.
128///
129/// `headless_chrome` ships a 30s default and tears down the entire connection
130/// when no traffic arrives for that long; a run must own its connection for
131/// its full duration instead.
132const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
133
134/// Executes a [`Scenario`] against a real browser with optional LLM
135/// assistance for element targeting and assertions.
136pub struct ScenarioRunner {
137    config: ScenarioConfig,
138    definitions: HashMap<String, AssertDefinition>,
139    llm: LlmConfig,
140    timeout: Duration,
141    viewport_width: u32,
142    viewport_height: u32,
143    /// The viewport currently applied in the browser (CDP emulation).
144    /// Per-test overrides switch it mid-run; `(0, 0)` = not yet applied.
145    applied_viewport: std::cell::Cell<(u32, u32)>,
146    endpoints: EndpointRegistry,
147    usage: Arc<UsageTracker>,
148    budgets: BudgetTracker,
149    /// Directory for failure artifacts (screenshots).
150    artifacts_dir: PathBuf,
151}
152
153/// Aggregated results from a scenario run.
154#[derive(Debug, Default)]
155pub struct RunReport {
156    /// Number of tests that passed.
157    pub tests_passed: u32,
158    /// Number of tests that failed.
159    pub tests_failed: u32,
160    /// Number of steps that passed.
161    pub passed: u32,
162    /// Number of steps that failed.
163    pub failed: u32,
164    /// Number of steps that were skipped.
165    pub skipped: u32,
166    /// Per-step details.
167    pub details: Vec<StepResult>,
168}
169
170/// Result of a single step execution.
171#[derive(Debug)]
172pub struct StepResult {
173    /// The step name.
174    pub name: String,
175    /// Whether the step passed, failed, or was skipped.
176    pub status: StepStatus,
177    /// Human-readable result message.
178    pub message: String,
179}
180
181/// Outcome for a single step.
182#[derive(Debug, PartialEq, Eq)]
183pub enum StepStatus {
184    /// Step executed successfully and all assertions passed.
185    Passed,
186    /// Step execution or assertion failed.
187    Failed,
188    /// Step was skipped.
189    Skipped,
190}
191
192/// Predefined assertion preset definition.
193struct AssertPreset {
194    name: &'static str,
195    system: &'static str,
196    user_template: &'static str,
197}
198
199/// Built-in assertion presets.
200#[allow(clippy::literal_string_with_formatting_args)]
201const ASSERTION_PRESETS: &[AssertPreset] = &[
202    AssertPreset {
203        name: "no_error_on_page",
204        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
205        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
206    },
207    AssertPreset {
208        name: "text_visible",
209        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
210        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
211    },
212    AssertPreset {
213        name: "element_exists",
214        system: "You are a QA tester. Check if a described UI element exists on a web page.",
215        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
216    },
217    AssertPreset {
218        name: "visual_no_issues",
219        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
220        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
221    },
222    AssertPreset {
223        name: "visual_no_overlaps",
224        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
225        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
226    },
227    AssertPreset {
228        name: "visual_text_visible",
229        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
230        user_template: "Inspect the attached screenshot and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
231    },
232    AssertPreset {
233        name: "layout_no_issues",
234        system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
235        user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
236    },
237];
238
239impl ScenarioRunner {
240    /// Creates a new runner with the given scenario configuration and
241    /// assertion definitions.
242    #[must_use]
243    #[allow(clippy::needless_pass_by_value)]
244    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
245        let llm = LlmConfig {
246            url: scenario_config
247                .llm_url
248                .clone()
249                .unwrap_or_else(crate::llm_base_url),
250            model: scenario_config
251                .llm_model
252                .clone()
253                .unwrap_or_else(crate::llm_model),
254            api_key: scenario_config
255                .llm_api_key
256                .clone()
257                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
258            headers: if scenario_config.llm_headers.is_empty() {
259                crate::parse_headers_env()
260            } else {
261                scenario_config.llm_headers.clone()
262            },
263            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
264            temperature: scenario_config.temperature,
265            thinking: scenario_config.thinking,
266            model_params: scenario_config.model_params.clone(),
267        };
268        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
269        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
270        let defs_map: HashMap<String, AssertDefinition> = definitions
271            .into_iter()
272            .map(|d| (d.name.clone(), d))
273            .collect();
274
275        Self {
276            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
277            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
278            viewport_height: scenario_config.viewport_height.unwrap_or(720),
279            applied_viewport: std::cell::Cell::new((0, 0)),
280            config: scenario_config.clone(),
281            definitions: defs_map,
282            llm,
283            endpoints,
284            usage: Arc::new(UsageTracker::new()),
285            budgets,
286            artifacts_dir: PathBuf::from(
287                scenario_config
288                    .artifacts_dir
289                    .unwrap_or_else(|| "artifacts".to_owned()),
290            ),
291        }
292    }
293
294    /// Returns a clone of the [`UsageTracker`] for reporting.
295    #[must_use]
296    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
297        Arc::clone(&self.usage)
298    }
299
300    /// Returns a reference to the [`BudgetTracker`].
301    #[must_use]
302    pub const fn budget_tracker(&self) -> &BudgetTracker {
303        &self.budgets
304    }
305
306    /// Executes all test groups in the scenario and returns a report.
307    ///
308    /// # Errors
309    ///
310    /// Returns an error if the browser fails to launch.
311    #[allow(clippy::too_many_lines)]
312    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
313        let mut report = RunReport::default();
314
315        if tests.is_empty() {
316            eprintln!("No tests defined in scenario.");
317            return Ok(report);
318        }
319
320        let browser_headless = self.config.browser_headless.unwrap_or(true);
321
322        let launch_opts = LaunchOptions {
323            headless: browser_headless,
324            window_size: Some((self.viewport_width, self.viewport_height)),
325            sandbox: false,
326            // headless_chrome defaults this to 30s and shuts down the whole CDP
327            // connection when no messages arrive for that long. A scenario can
328            // easily exceed 30s of browser silence (slow LLM targeting/assertion
329            // calls, page waits, budget checks between steps), after which every
330            // remaining step fails with "Unable to make method calls because
331            // underlying connection is closed" — one quiet gap kills the run.
332            // Open-ended scenarios must own the connection for their full
333            // duration, so keep it alive for 6 hours.
334            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
335            ..LaunchOptions::default()
336        };
337
338        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
339        let tab = browser.new_tab().context("failed to open browser tab")?;
340        let _ = tab.set_default_timeout(self.timeout);
341
342        // Start MCP server if configured
343        #[cfg(feature = "mcp-server")]
344        if let Some(ref mcp_cfg) = self.config.mcp_server {
345            if mcp_cfg.enabled {
346                let port = mcp_cfg.port;
347                std::thread::spawn(move || {
348                    let _ = crate::mcp_server::start_mcp_server(port);
349                });
350            }
351        }
352        #[cfg(not(feature = "mcp-server"))]
353        if let Some(mcp_cfg) = &self.config.mcp_server {
354            if mcp_cfg.enabled {
355                eprintln!("  ⚠️  MCP server configured but 'mcp-server' feature not enabled");
356            }
357        }
358
359        // Start A2A agent server if configured
360        #[cfg(feature = "a2a-server")]
361        if let Some(ref a2a_cfg) = self.config.a2a_server {
362            if a2a_cfg.enabled {
363                let port = a2a_cfg.port;
364                tokio::spawn(crate::a2a_server::start_a2a_server(port));
365            }
366        }
367        #[cfg(not(feature = "a2a-server"))]
368        if let Some(a2a_cfg) = &self.config.a2a_server {
369            if a2a_cfg.enabled {
370                eprintln!("  ⚠️  A2A server configured but 'a2a-server' feature not enabled");
371            }
372        }
373
374        for test in tests {
375            eprintln!("\n╔══════════════════════════════");
376            eprintln!("║  Test: {}", test.name);
377            eprintln!("╚══════════════════════════════");
378
379            self.usage.reset_per_test();
380
381            let test_result = self.run_test(test, &tab);
382            self.usage.commit_test(&test.name);
383
384            if test_result.failed == 0 && test_result.total > 0 {
385                report.tests_passed += 1;
386                eprintln!("  Test ✅ Passed");
387            } else if test_result.total > 0 {
388                report.tests_failed += 1;
389                eprintln!("  Test ❌ Failed");
390            }
391
392            report.passed += test_result.passed;
393            report.failed += test_result.failed;
394            report.skipped += test_result.skipped;
395            report.details.extend(test_result.details);
396        }
397
398        Ok(report)
399    }
400
401    #[allow(clippy::too_many_lines)]
402    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
403        let base_url = test
404            .base_url
405            .clone()
406            .or_else(|| self.config.base_url.clone())
407            .unwrap_or_else(crate::base_url);
408
409        // Per-test viewport override: switch the browser via CDP
410        // device-metrics emulation before this test runs.
411        let vw = test.viewport_width.unwrap_or(self.viewport_width);
412        let vh = test.viewport_height.unwrap_or(self.viewport_height);
413        if self.applied_viewport.get() != (vw, vh) {
414            self.apply_viewport(tab, vw, vh);
415            self.applied_viewport.set((vw, vh));
416        }
417
418        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
419
420        let start_url = test
421            .start_url
422            .clone()
423            .or_else(|| self.config.start_url.clone())
424            .unwrap_or_else(|| "/dashboard".to_owned());
425
426        if auto_navigate {
427            let full_url = resolve_url(&start_url, &base_url);
428            eprintln!("  → auto-navigate: {full_url}");
429            let _ = tab.navigate_to(&full_url);
430            let _ = tab.wait_until_navigated();
431            std::thread::sleep(Duration::from_secs(4));
432        }
433
434        let mut result = TestRunResult::default();
435
436        for (step_index, step) in test.steps.iter().enumerate() {
437            result.total += 1;
438
439            let wait_ms = match step {
440                TestStep::Navigate { wait_after_ms, .. }
441                | TestStep::Click { wait_after_ms, .. }
442                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
443                _ => None,
444            };
445
446            let mut step_result = match step {
447                TestStep::Navigate { url, .. } => {
448                    let full_url = resolve_url(url, &base_url);
449                    run_navigate_step(&full_url, tab)
450                }
451                TestStep::Click {
452                    target,
453                    selector,
454                    endpoint,
455                    idempotent,
456                    ..
457                } => self.run_click(
458                    target,
459                    selector.as_deref(),
460                    endpoint.as_deref(),
461                    test.endpoint.as_deref(),
462                    *idempotent,
463                    tab,
464                ),
465                TestStep::Type {
466                    target,
467                    text,
468                    selector,
469                    endpoint,
470                    idempotent,
471                    ..
472                } => self.run_type(
473                    target,
474                    text,
475                    selector.as_deref(),
476                    endpoint.as_deref(),
477                    test.endpoint.as_deref(),
478                    *idempotent,
479                    tab,
480                ),
481                TestStep::Wait {
482                    target,
483                    selector,
484                    text,
485                    timeout_ms,
486                    endpoint,
487                    idempotent,
488                } => self.run_wait(
489                    target,
490                    selector.as_deref(),
491                    text.as_deref(),
492                    *timeout_ms,
493                    endpoint.as_deref(),
494                    test.endpoint.as_deref(),
495                    *idempotent,
496                    tab,
497                ),
498                TestStep::Assert {
499                    definition,
500                    preset,
501                    prompt,
502                    assert_text,
503                    endpoint,
504                    screenshot,
505                } => self.run_assert(
506                    definition.as_deref(),
507                    preset.as_deref(),
508                    prompt.as_deref(),
509                    assert_text.as_deref(),
510                    *screenshot,
511                    endpoint.as_deref(),
512                    test.endpoint.as_deref(),
513                    tab,
514                ),
515                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
516                TestStep::Agent {
517                    agent,
518                    task,
519                    definition,
520                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
521                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
522            };
523
524            // Failure diagnostics: capture the page state and a screenshot so
525            // CI logs say WHAT the page looked like when the step failed,
526            // instead of a bare "timed out: The event waited for never came".
527            if step_result.status == StepStatus::Failed {
528                let state = diagnostics::capture(tab);
529                let screenshot = diagnostics::save_screenshot(
530                    tab,
531                    &self.artifacts_dir,
532                    &test.name,
533                    &test.name,
534                    step_index,
535                    step_kind_label(step),
536                );
537                step_result.message = format!(
538                    "{base} — {excerpt}",
539                    base = step_result.message,
540                    excerpt = diagnostics::inline_excerpt(&state),
541                );
542                eprintln!(
543                    "{}",
544                    diagnostics::full_context(&state, screenshot.as_deref())
545                );
546            }
547
548            eprintln!(
549                "    {} {} — {}",
550                if step_result.status == StepStatus::Passed {
551                    "✅"
552                } else if step_result.status == StepStatus::Failed {
553                    "❌"
554                } else {
555                    "⏭️"
556                },
557                step_result.name,
558                step_result.message,
559            );
560
561            match step_result.status {
562                StepStatus::Passed => result.passed += 1,
563                StepStatus::Failed => result.failed += 1,
564                StepStatus::Skipped => result.skipped += 1,
565            }
566
567            // Fail fast: the first failed step ends the test and the
568            // remaining steps are reported as skipped (no LLM budget is
569            // burned asserting against a page that is already known broken).
570            if step_result.status == StepStatus::Failed
571                && !self.config.continue_on_failure
572                && step_index + 1 < test.steps.len()
573            {
574                eprintln!(
575                    "      ⏭️  failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
576                    test.steps.len() - step_index - 1
577                );
578                for skipped in &test.steps[step_index + 1..] {
579                    result.total += 1;
580                    result.skipped += 1;
581                    eprintln!(
582                        "    ⏭️  {} — skipped: previous step failed",
583                        step_label(skipped)
584                    );
585                    result.details.push(StepResult {
586                        name: step_label(skipped),
587                        status: StepStatus::Skipped,
588                        message: "skipped: previous step failed".into(),
589                    });
590                }
591                result.details.push(step_result);
592                return result;
593            }
594
595            // Check per-test budget after each step
596            let test_usage = self.usage.current_test_snapshot();
597            let global_usage = self.usage.global_snapshot();
598            let budget_status = self.budgets.check_all(
599                &test.name,
600                &test_usage,
601                &global_usage,
602                test.budget.as_ref(),
603            );
604            match budget_status {
605                BudgetStatus::HardExceeded { message, .. } => {
606                    crate::reporting::print_budget_error(&message);
607                    result.details.push(StepResult {
608                        name: "[budget]".into(),
609                        status: StepStatus::Failed,
610                        message,
611                    });
612                    result.failed += 1;
613                    return result;
614                }
615                BudgetStatus::SoftExceeded { message, .. } => {
616                    crate::reporting::print_budget_warning(&message);
617                }
618                BudgetStatus::Ok => {}
619            }
620
621            if let Some(ms) = wait_ms {
622                std::thread::sleep(Duration::from_millis(ms));
623            }
624
625            result.details.push(step_result);
626        }
627
628        result
629    }
630
631    /// Applies a viewport size to the current tab via CDP
632    /// `Emulation.setDeviceMetricsOverride`. Used by per-test viewport
633    /// overrides and the viewport matrix. The initial window size set at
634    /// browser launch is replaced by emulation; failures are logged but
635    /// do not fail the test (a mismatched viewport only weakens coverage).
636    fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
637        use headless_chrome::protocol::cdp::Emulation;
638        let _ = self;
639        let params = Emulation::SetDeviceMetricsOverride {
640            width,
641            height,
642            device_scale_factor: 1.0,
643            mobile: false,
644            scale: None,
645            screen_width: Some(width),
646            screen_height: Some(height),
647            position_x: None,
648            position_y: None,
649            dont_set_visible_size: None,
650            screen_orientation: None,
651            viewport: None,
652            display_feature: None,
653            device_posture: None,
654        };
655        eprintln!("      ↻ viewport: {width}x{height}");
656        if let Err(e) = tab.call_method(params) {
657            eprintln!("      ⚠️  viewport switch to {width}x{height} failed: {e}");
658        }
659    }
660
661    // ── step handlers ───────────────────────────────────────────────────
662
663    #[allow(clippy::too_many_lines)]
664    fn run_click(
665        &self,
666        target: &str,
667        selector_override: Option<&str>,
668        step_endpoint: Option<&str>,
669        test_endpoint: Option<&str>,
670        idempotent: bool,
671        tab: &Tab,
672    ) -> StepResult {
673        let name = format!("[click] {target}");
674        let selector = match self.resolve_selector(
675            selector_override,
676            target,
677            step_endpoint,
678            test_endpoint,
679            tab,
680        ) {
681            Ok(s) => s,
682            Err(msg) => {
683                if idempotent {
684                    return StepResult {
685                        name,
686                        status: StepStatus::Skipped,
687                        message: format!("skipped (idempotent): no target found — {msg}"),
688                    };
689                }
690                return StepResult {
691                    name,
692                    status: StepStatus::Failed,
693                    message: msg,
694                };
695            }
696        };
697
698        // Idempotent steps probe briefly: a missing target means the
699        // action was already done / not applicable (e.g. an
700        // already-authenticated session), and skipping is the success
701        // path, not a failure.
702        let probe_secs = if idempotent { 5 } else { 10 };
703        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
704            Ok(element) => match element.click() {
705                Ok(_) => StepResult {
706                    name,
707                    status: StepStatus::Passed,
708                    message: format!("clicked {selector}"),
709                },
710                Err(e) => StepResult {
711                    name,
712                    status: StepStatus::Failed,
713                    message: format!("click failed on {selector}: {e}"),
714                },
715            },
716            Err(e) if idempotent => StepResult {
717                name,
718                status: StepStatus::Skipped,
719                message: format!("skipped (idempotent): element {selector} not present — {e}"),
720            },
721            Err(e) => StepResult {
722                name,
723                status: StepStatus::Failed,
724                message: format!("element {selector} not found: {e}"),
725            },
726        }
727    }
728
729    #[allow(clippy::too_many_arguments)]
730    fn run_type(
731        &self,
732        target: &str,
733        text: &str,
734        selector_override: Option<&str>,
735        step_endpoint: Option<&str>,
736        test_endpoint: Option<&str>,
737        idempotent: bool,
738        tab: &Tab,
739    ) -> StepResult {
740        let name = format!("[type] {target}");
741        let selector = match self.resolve_selector(
742            selector_override,
743            target,
744            step_endpoint,
745            test_endpoint,
746            tab,
747        ) {
748            Ok(s) => s,
749            Err(msg) => {
750                if idempotent {
751                    return StepResult {
752                        name,
753                        status: StepStatus::Skipped,
754                        message: format!("skipped (idempotent): no target found — {msg}"),
755                    };
756                }
757                return StepResult {
758                    name,
759                    status: StepStatus::Failed,
760                    message: msg,
761                };
762            }
763        };
764
765        let probe_secs = if idempotent { 5 } else { 10 };
766        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
767            Ok(element) => {
768                if let Err(e) = element.click() {
769                    return StepResult {
770                        name,
771                        status: StepStatus::Failed,
772                        message: format!("click to focus {selector} failed: {e}"),
773                    };
774                }
775
776                let js = format!(
777                    "document.querySelector('{}').value = '';",
778                    selector.replace('\'', "\\'")
779                );
780                let _ = tab.evaluate(&js, false);
781
782                match element.type_into(text) {
783                    Ok(_) => StepResult {
784                        name,
785                        status: StepStatus::Passed,
786                        message: format!("typed {text:?} into {selector}"),
787                    },
788                    Err(e) => StepResult {
789                        name,
790                        status: StepStatus::Failed,
791                        message: format!("type into {selector} failed: {e}"),
792                    },
793                }
794            }
795            Err(e) if idempotent => StepResult {
796                name,
797                status: StepStatus::Skipped,
798                message: format!("skipped (idempotent): element {selector} not present — {e}"),
799            },
800            Err(e) => StepResult {
801                name,
802                status: StepStatus::Failed,
803                message: format!("element {selector} not found: {e}"),
804            },
805        }
806    }
807
808    #[allow(clippy::too_many_arguments)]
809    #[allow(clippy::too_many_lines)]
810    fn run_wait(
811        &self,
812        target: &str,
813        selector_override: Option<&str>,
814        text: Option<&str>,
815        timeout_ms: Option<u64>,
816        step_endpoint: Option<&str>,
817        test_endpoint: Option<&str>,
818        idempotent: bool,
819        tab: &Tab,
820    ) -> StepResult {
821        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
822        let step_name = format!("[wait] {target}");
823
824        // Resolve an explicit selector only (text-only waits are LLM-free).
825        let selector = match selector_override {
826            Some(s) => Some(s.to_owned()),
827            None if text.is_some() => None,
828            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
829                Ok(s) => Some(s),
830                Err(msg) => {
831                    if idempotent {
832                        return StepResult {
833                            name: step_name,
834                            status: StepStatus::Skipped,
835                            message: format!("skipped (idempotent): no target found — {msg}"),
836                        };
837                    }
838                    return StepResult {
839                        name: step_name,
840                        status: StepStatus::Failed,
841                        message: msg,
842                    };
843                }
844            },
845        };
846
847        if text.is_some() {
848            let sel_js = selector
849                .as_deref()
850                .map(crate::selectors::selector_matches_js);
851            let text_js = text.map(|t| {
852                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
853                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
854            });
855
856            let deadline = Instant::now() + timeout;
857            loop {
858                let sel_ok = sel_js
859                    .as_ref()
860                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
861                let text_ok = text_js
862                    .as_ref()
863                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
864                if sel_ok && text_ok {
865                    let mut what = Vec::new();
866                    if let Some(sel) = &selector {
867                        what.push(format!("found {sel}"));
868                    }
869                    if let Some(t) = text {
870                        what.push(format!("text {t:?} visible"));
871                    }
872                    return StepResult {
873                        name: step_name,
874                        status: StepStatus::Passed,
875                        message: what.join(" and "),
876                    };
877                }
878                if Instant::now() >= deadline {
879                    let mut what = Vec::new();
880                    if let Some(sel) = &selector {
881                        what.push(sel.clone());
882                    }
883                    if let Some(t) = text {
884                        what.push(format!("text {t:?}"));
885                    }
886                    let message = format!(
887                        "wait for {} timed out after {}ms: the event waited for never came",
888                        what.join(" / "),
889                        timeout.as_millis(),
890                    );
891                    if idempotent {
892                        return StepResult {
893                            name: step_name,
894                            status: StepStatus::Skipped,
895                            message: format!("skipped (idempotent): {message}"),
896                        };
897                    }
898                    return StepResult {
899                        name: step_name,
900                        status: StepStatus::Failed,
901                        message,
902                    };
903                }
904                std::thread::sleep(Duration::from_millis(250));
905            }
906        }
907
908        match selector.as_deref() {
909            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
910                Ok(_) => StepResult {
911                    name: step_name,
912                    status: StepStatus::Passed,
913                    message: format!("found {sel}"),
914                },
915                Err(e) if idempotent => StepResult {
916                    name: step_name,
917                    status: StepStatus::Skipped,
918                    message: format!(
919                        "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
920                        timeout.as_millis()
921                    ),
922                },
923                Err(e) => StepResult {
924                    name: step_name,
925                    status: StepStatus::Failed,
926                    message: format!(
927                        "wait for {sel} timed out after {}ms: {e}",
928                        timeout.as_millis()
929                    ),
930                },
931            },
932            None => StepResult {
933                name: step_name,
934                status: StepStatus::Failed,
935                message: "wait step has neither selector nor text".into(),
936            },
937        }
938    }
939
940    #[allow(clippy::too_many_arguments)]
941    fn run_assert(
942        &self,
943        definition: Option<&str>,
944        preset: Option<&str>,
945        prompt: Option<&str>,
946        assert_text: Option<&str>,
947        screenshot: bool,
948        step_endpoint: Option<&str>,
949        test_endpoint: Option<&str>,
950        tab: &Tab,
951    ) -> StepResult {
952        std::thread::sleep(Duration::from_millis(500));
953
954        let page_content = get_page_text(tab);
955
956        // Vision attach: capture the viewport once per assert step and hand
957        // the JPEG data URL to the preset/prompt evaluation below.
958        let image = if screenshot {
959            let endpoint = self
960                .endpoints
961                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
962            if !endpoint.vision {
963                return StepResult {
964                    name: "[assert]".into(),
965                    status: StepStatus::Failed,
966                    message: format!(
967                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
968                        name = endpoint.name
969                    ),
970                };
971            }
972            match crate::vision::capture_screenshot_data_url(
973                tab,
974                self.config
975                    .screenshot_max_dimension
976                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
977            ) {
978                Ok(data_url) => Some(data_url),
979                Err(e) => {
980                    return StepResult {
981                        name: "[assert]".into(),
982                        status: StepStatus::Failed,
983                        message: format!("screenshot capture failed: {e}"),
984                    };
985                }
986            }
987        } else {
988            None
989        };
990
991        if let Some(def_name) = definition {
992            if let Some(def) = self.definitions.get(def_name) {
993                return self.run_assert_def(
994                    def,
995                    &page_content,
996                    image.as_deref(),
997                    step_endpoint,
998                    test_endpoint,
999                    tab,
1000                );
1001            }
1002            return StepResult {
1003                name: format!("[assert] {def_name}"),
1004                status: StepStatus::Failed,
1005                message: format!("definition '{def_name}' not found"),
1006            };
1007        }
1008
1009        if let Some(preset_name) = preset {
1010            // Deterministic DOM layout scan — runs JS in the browser and
1011            // never calls the LLM (free, fast, no pixel budget).
1012            if preset_name == "layout_no_issues" {
1013                return self.run_layout_preset(tab);
1014            }
1015            return self.run_preset(
1016                preset_name,
1017                assert_text,
1018                &page_content,
1019                image.as_deref(),
1020                step_endpoint,
1021                test_endpoint,
1022            );
1023        }
1024
1025        if let Some(prompt_text) = prompt {
1026            return self.run_custom(
1027                prompt_text,
1028                &page_content,
1029                image.as_deref(),
1030                step_endpoint,
1031                test_endpoint,
1032            );
1033        }
1034
1035        StepResult {
1036            name: "[assert]".into(),
1037            status: StepStatus::Skipped,
1038            message: "no definition, preset, or prompt specified".into(),
1039        }
1040    }
1041
1042    fn run_assert_def(
1043        &self,
1044        def: &AssertDefinition,
1045        page_content: &PageContent,
1046        image: Option<&str>,
1047        step_endpoint: Option<&str>,
1048        test_endpoint: Option<&str>,
1049        tab: &Tab,
1050    ) -> StepResult {
1051        // Agent-based definition: delegate to an A2A agent
1052        if let Some(ref agent) = def.agent {
1053            if image.is_some() {
1054                return StepResult {
1055                    name: format!("[assert] {}", def.name),
1056                    status: StepStatus::Failed,
1057                    message: "agent-backed assertions do not support screenshots".into(),
1058                };
1059            }
1060            let task = def
1061                .task_template
1062                .as_deref()
1063                .unwrap_or("Evaluate the assertion")
1064                .replace("{url}", &page_content.url)
1065                .replace("{title}", &page_content.title)
1066                .replace("{content}", &page_content.body_text)
1067                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1068
1069            return self.run_agent_step(agent, &task, &def.name);
1070        }
1071
1072        // Custom preset: system + user_template provided in the definition
1073        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1074            return self.run_custom_preset(
1075                &def.name,
1076                system,
1077                template,
1078                def.assert_text.as_deref(),
1079                page_content,
1080                image,
1081                step_endpoint,
1082                test_endpoint,
1083            );
1084        }
1085
1086        def.preset.as_ref().map_or_else(
1087            || {
1088                def.prompt.as_ref().map_or_else(
1089                    || StepResult {
1090                        name: format!("[assert] {}", def.name),
1091                        status: StepStatus::Failed,
1092                        message: "definition has no preset, prompt, or system+user_template".into(),
1093                    },
1094                    |prompt| {
1095                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1096                    },
1097                )
1098            },
1099            |preset_name| {
1100                if preset_name == "layout_no_issues" {
1101                    return self.run_layout_preset(tab);
1102                }
1103                self.run_preset(
1104                    preset_name,
1105                    def.assert_text.as_deref(),
1106                    page_content,
1107                    image,
1108                    step_endpoint,
1109                    test_endpoint,
1110                )
1111            },
1112        )
1113    }
1114
1115    #[allow(clippy::too_many_arguments)]
1116    fn run_custom_preset(
1117        &self,
1118        name: &str,
1119        system: &str,
1120        template: &str,
1121        assert_text: Option<&str>,
1122        page_content: &PageContent,
1123        image: Option<&str>,
1124        step_endpoint: Option<&str>,
1125        test_endpoint: Option<&str>,
1126    ) -> StepResult {
1127        let user_prompt = template
1128            .replace("{url}", &page_content.url)
1129            .replace("{title}", &page_content.title)
1130            .replace("{content}", &page_content.body_text)
1131            .replace("{expected_text}", assert_text.unwrap_or(""))
1132            .replace("{description}", "");
1133
1134        // Custom preset definitions frequently forget the {content}
1135        // placeholder — without it the LLM has no page to evaluate and
1136        // answers "I can't determine that without seeing the page". Always
1137        // append the page context unless the template already references it.
1138        let user_prompt = if template.contains("{content}") {
1139            user_prompt
1140        } else {
1141            format!(
1142                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1143                url = page_content.url,
1144                title = page_content.title,
1145                content = page_content.body_text,
1146            )
1147        };
1148
1149        eprintln!("      assert: {name} (custom preset)");
1150
1151        let endpoint = self
1152            .endpoints
1153            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1154        let llm = self.build_llm_for_endpoint(endpoint);
1155        let usage = Arc::clone(&self.usage);
1156        let endpoint_name = endpoint.name.clone();
1157        let sys = system.to_owned();
1158        let image = image.map(str::to_owned);
1159
1160        let response = std::thread::spawn(move || {
1161            let rt = tokio::runtime::Builder::new_current_thread()
1162                .enable_all()
1163                .build()
1164                .unwrap();
1165            let call = async {
1166                match image.as_deref() {
1167                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user_prompt, img).await,
1168                    None => llm_chat_with_usage(&llm, &sys, &user_prompt).await,
1169                }
1170            };
1171            rt.block_on(call)
1172        })
1173        .join()
1174        .unwrap();
1175
1176        response.map_or_else(
1177            |e| StepResult {
1178                name: format!("[assert] {name}"),
1179                status: StepStatus::Failed,
1180                message: format!("LLM assertion call failed: {e}"),
1181            },
1182            |lr| {
1183                usage.record_llm_call(
1184                    &endpoint_name,
1185                    endpoint,
1186                    lr.usage.prompt_tokens,
1187                    lr.usage.completion_tokens,
1188                );
1189                let content_lower = lr.content.to_lowercase().trim().to_owned();
1190                if content_lower.starts_with("pass") {
1191                    StepResult {
1192                        name: format!("[assert] {name}"),
1193                        status: StepStatus::Passed,
1194                        message: "PASS".into(),
1195                    }
1196                } else {
1197                    StepResult {
1198                        name: format!("[assert] {name}"),
1199                        status: StepStatus::Failed,
1200                        message: lr.content,
1201                    }
1202                }
1203            },
1204        )
1205    }
1206
1207    fn run_preset(
1208        &self,
1209        preset_name: &str,
1210        assert_text: Option<&str>,
1211        page_content: &PageContent,
1212        image: Option<&str>,
1213        step_endpoint: Option<&str>,
1214        test_endpoint: Option<&str>,
1215    ) -> StepResult {
1216        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1217            return StepResult {
1218                name: format!("[assert] {preset_name}"),
1219                status: StepStatus::Failed,
1220                message: format!("unknown assertion preset: {preset_name}"),
1221            };
1222        };
1223        if preset_name.starts_with("visual_") && image.is_none() {
1224            return StepResult {
1225                name: format!("[assert] {preset_name}"),
1226                status: StepStatus::Failed,
1227                message: format!(
1228                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1229                ),
1230            };
1231        }
1232
1233        let user_prompt = preset
1234            .user_template
1235            .replace("{url}", &page_content.url)
1236            .replace("{title}", &page_content.title)
1237            .replace("{content}", &page_content.body_text)
1238            .replace("{expected_text}", assert_text.unwrap_or(""))
1239            .replace("{description}", "");
1240
1241        // Same safety net as custom presets: never let the LLM answer with
1242        // no page context at all.
1243        let user_prompt = if preset.user_template.contains("{content}") {
1244            user_prompt
1245        } else {
1246            format!(
1247                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1248                url = page_content.url,
1249                title = page_content.title,
1250                content = page_content.body_text,
1251            )
1252        };
1253
1254        eprintln!("      assert: {preset_name}");
1255
1256        let endpoint = self
1257            .endpoints
1258            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1259        let llm = self.build_llm_for_endpoint(endpoint);
1260        let usage = Arc::clone(&self.usage);
1261        let endpoint_name = endpoint.name.clone();
1262        let sys = preset.system.to_owned();
1263        let image = image.map(str::to_owned);
1264
1265        let response = std::thread::spawn(move || {
1266            let rt = tokio::runtime::Builder::new_current_thread()
1267                .enable_all()
1268                .build()
1269                .unwrap();
1270            let call = async {
1271                match image.as_deref() {
1272                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user_prompt, img).await,
1273                    None => llm_chat_with_usage(&llm, &sys, &user_prompt).await,
1274                }
1275            };
1276            rt.block_on(call)
1277        })
1278        .join()
1279        .unwrap();
1280
1281        response.map_or_else(
1282            |e| StepResult {
1283                name: format!("[assert] {preset_name}"),
1284                status: StepStatus::Failed,
1285                message: format!("LLM assertion call failed: {e}"),
1286            },
1287            |lr| {
1288                usage.record_llm_call(
1289                    &endpoint_name,
1290                    endpoint,
1291                    lr.usage.prompt_tokens,
1292                    lr.usage.completion_tokens,
1293                );
1294                let content_lower = lr.content.to_lowercase().trim().to_owned();
1295                if content_lower.starts_with("pass") {
1296                    StepResult {
1297                        name: format!("[assert] {preset_name}"),
1298                        status: StepStatus::Passed,
1299                        message: "PASS".into(),
1300                    }
1301                } else {
1302                    StepResult {
1303                        name: format!("[assert] {preset_name}"),
1304                        status: StepStatus::Failed,
1305                        message: lr.content,
1306                    }
1307                }
1308            },
1309        )
1310    }
1311
1312    /// Runs the deterministic DOM layout scan (`layout_no_issues`).
1313    ///
1314    /// Evaluates the layout-scan JS in the page and fails with the list of
1315    /// detected issues: horizontal page overflow, visible elements sticking
1316    /// out of the viewport, text clipped by `overflow: hidden` containers,
1317    /// and interactive elements covered by other elements. No LLM call —
1318    /// checks are geometry-based so the check is free, deterministic, and
1319    /// safe to run on every page × viewport variant.
1320    fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1321        let _ = self;
1322        let name = "[assert] layout_no_issues".to_owned();
1323        eprintln!("      assert: layout_no_issues (DOM layout scan)");
1324        let result = tab.evaluate(LAYOUT_SCAN_JS, false);
1325        let json_str = match result {
1326            Ok(r) => r
1327                .value
1328                .as_ref()
1329                .and_then(|v| v.as_str().map(String::from))
1330                .unwrap_or_else(|| "[]".to_owned()),
1331            Err(e) => {
1332                return StepResult {
1333                    name,
1334                    status: StepStatus::Failed,
1335                    message: format!("layout scan JS failed: {e}"),
1336                };
1337            }
1338        };
1339        let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1340        if issues.is_empty() {
1341            return StepResult {
1342                name,
1343                status: StepStatus::Passed,
1344                message: "PASS — no layout defects detected".into(),
1345            };
1346        }
1347        let mut lines: Vec<String> = issues
1348            .iter()
1349            .take(10)
1350            .map(|i| {
1351                format!(
1352                    "- [{type_}] {element}: {detail}",
1353                    type_ = i.issue_type,
1354                    element = i.element,
1355                    detail = i.detail
1356                )
1357            })
1358            .collect();
1359        if issues.len() > 10 {
1360            lines.push(format!("- … and {} more", issues.len() - 10));
1361        }
1362        StepResult {
1363            name,
1364            status: StepStatus::Failed,
1365            message: format!(
1366                "FAIL — {} layout defect(s) detected:\n{}",
1367                issues.len(),
1368                lines.join("\n")
1369            ),
1370        }
1371    }
1372
1373    fn run_custom(
1374        &self,
1375        prompt: &str,
1376        page_content: &PageContent,
1377        image: Option<&str>,
1378        step_endpoint: Option<&str>,
1379        test_endpoint: Option<&str>,
1380    ) -> StepResult {
1381        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1382
1383        let mut user = format!(
1384            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1385            url = page_content.url,
1386            title = page_content.title,
1387            content = page_content.body_text,
1388        );
1389        if image.is_some() {
1390            user.push_str(
1391                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1392            );
1393        }
1394
1395        eprintln!("      custom assert");
1396
1397        let endpoint = self
1398            .endpoints
1399            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1400        let llm = self.build_llm_for_endpoint(endpoint);
1401        let usage = Arc::clone(&self.usage);
1402        let endpoint_name = endpoint.name.clone();
1403        let sys = system.to_owned();
1404        let image = image.map(str::to_owned);
1405
1406        let response = std::thread::spawn(move || {
1407            let rt = tokio::runtime::Builder::new_current_thread()
1408                .enable_all()
1409                .build()
1410                .unwrap();
1411            let call = async {
1412                match image.as_deref() {
1413                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user, img).await,
1414                    None => llm_chat_with_usage(&llm, &sys, &user).await,
1415                }
1416            };
1417            rt.block_on(call)
1418        })
1419        .join()
1420        .unwrap();
1421
1422        response.map_or_else(
1423            |e| StepResult {
1424                name: "[assert] custom".into(),
1425                status: StepStatus::Failed,
1426                message: format!("LLM assertion call failed: {e}"),
1427            },
1428            |lr| {
1429                usage.record_llm_call(
1430                    &endpoint_name,
1431                    endpoint,
1432                    lr.usage.prompt_tokens,
1433                    lr.usage.completion_tokens,
1434                );
1435                let content_lower = lr.content.to_lowercase().trim().to_owned();
1436                if content_lower.starts_with("pass") {
1437                    StepResult {
1438                        name: "[assert] custom".into(),
1439                        status: StepStatus::Passed,
1440                        message: "PASS".into(),
1441                    }
1442                } else {
1443                    StepResult {
1444                        name: "[assert] custom".into(),
1445                        status: StepStatus::Failed,
1446                        message: lr.content,
1447                    }
1448                }
1449            },
1450        )
1451    }
1452
1453    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1454        let path = path.unwrap_or("screenshot.png");
1455
1456        match tab.capture_screenshot(
1457            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1458            None,
1459            None,
1460            true,
1461        ) {
1462            Ok(data) => {
1463                if let Err(e) = std::fs::write(path, &data) {
1464                    return StepResult {
1465                        name: format!("[screenshot] {path}"),
1466                        status: StepStatus::Failed,
1467                        message: format!("failed to write screenshot: {e}"),
1468                    };
1469                }
1470                StepResult {
1471                    name: format!("[screenshot] {path}"),
1472                    status: StepStatus::Passed,
1473                    message: format!("saved to {path}"),
1474                }
1475            }
1476            Err(e) => StepResult {
1477                name: format!("[screenshot] {path}"),
1478                status: StepStatus::Failed,
1479                message: format!("screenshot failed: {e}"),
1480            },
1481        }
1482    }
1483
1484    /// Runs an A2A agent step.
1485    #[allow(clippy::literal_string_with_formatting_args)]
1486    fn run_agent(
1487        &self,
1488        agent_name: &str,
1489        task: &str,
1490        definition: Option<&str>,
1491        _test_endpoint: Option<&str>,
1492    ) -> StepResult {
1493        // If a definition is specified, look up the task template
1494        let resolved_task = if let Some(def_name) = definition {
1495            if let Some(def) = self.definitions.get(def_name) {
1496                let tmpl = def.task_template.as_deref().unwrap_or(task);
1497                tmpl.replace("{task}", task)
1498            } else {
1499                return StepResult {
1500                    name: format!("[agent] {def_name}"),
1501                    status: StepStatus::Failed,
1502                    message: format!("definition '{def_name}' not found"),
1503                };
1504            }
1505        } else {
1506            task.to_owned()
1507        };
1508
1509        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1510    }
1511
1512    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1513        let Some(ep) = self.endpoints.get(agent_name) else {
1514            return StepResult {
1515                name: format!("[agent] {display_name}"),
1516                status: StepStatus::Failed,
1517                message: format!("agent endpoint '{agent_name}' not found"),
1518            };
1519        };
1520
1521        if ep.url.is_empty() {
1522            return StepResult {
1523                name: format!("[agent] {display_name}"),
1524                status: StepStatus::Failed,
1525                message: format!("agent endpoint '{agent_name}' has no URL"),
1526            };
1527        }
1528
1529        eprintln!("      → agent {agent_name}: {task}");
1530
1531        let url = ep.url.clone();
1532        let client = A2aClient::new(&url, self.timeout);
1533        let task_clone = task.to_owned();
1534
1535        let response = std::thread::spawn(move || {
1536            let rt = tokio::runtime::Builder::new_current_thread()
1537                .enable_all()
1538                .build()
1539                .unwrap();
1540            rt.block_on(client.send_task(&task_clone))
1541        })
1542        .join()
1543        .unwrap();
1544
1545        // Record the flat-cost call
1546        self.usage.record_flat_call(agent_name, ep);
1547
1548        match response {
1549            Ok(text) => {
1550                let clean = text.trim().to_owned();
1551                let lower = clean.to_lowercase();
1552                if lower.starts_with("pass") {
1553                    StepResult {
1554                        name: format!("[agent] {display_name}"),
1555                        status: StepStatus::Passed,
1556                        message: format!("PASS: {clean}"),
1557                    }
1558                } else if lower.starts_with("fail") {
1559                    StepResult {
1560                        name: format!("[agent] {display_name}"),
1561                        status: StepStatus::Failed,
1562                        message: clean,
1563                    }
1564                } else {
1565                    StepResult {
1566                        name: format!("[agent] {display_name}"),
1567                        status: StepStatus::Passed,
1568                        message: format!("response: {clean}"),
1569                    }
1570                }
1571            }
1572            Err(e) => StepResult {
1573                name: format!("[agent] {display_name}"),
1574                status: StepStatus::Failed,
1575                message: format!("agent call failed: {e}"),
1576            },
1577        }
1578    }
1579
1580    /// Runs an MCP tool call step.
1581    fn run_mcp(
1582        &self,
1583        server_name: &str,
1584        tool_name: &str,
1585        args: Option<&serde_json::Value>,
1586    ) -> StepResult {
1587        let Some(ep) = self.endpoints.get(server_name) else {
1588            return StepResult {
1589                name: format!("[mcp] {server_name}:{tool_name}"),
1590                status: StepStatus::Failed,
1591                message: format!("MCP server endpoint '{server_name}' not found"),
1592            };
1593        };
1594
1595        let cmd = ep.command.as_deref().unwrap_or("");
1596        if cmd.is_empty() {
1597            return StepResult {
1598                name: format!("[mcp] {server_name}:{tool_name}"),
1599                status: StepStatus::Failed,
1600                message: format!("MCP server '{server_name}' has no command configured"),
1601            };
1602        }
1603
1604        eprintln!("      → mcp {server_name} {tool_name}");
1605
1606        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1607
1608        let command = cmd.to_owned();
1609        let args_vec = ep.args.clone();
1610        let tool = tool_name.to_owned();
1611
1612        let response = std::thread::spawn(move || {
1613            let mut mcp_client =
1614                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1615            mcp_client
1616                .call_tool(&tool, &args_val)
1617                .map_err(|e| e.to_string())
1618        })
1619        .join()
1620        .unwrap();
1621
1622        // Record the flat-cost call
1623        self.usage.record_flat_call(server_name, ep);
1624
1625        match response {
1626            Ok(result) => {
1627                if result.isError {
1628                    StepResult {
1629                        name: format!("[mcp] {server_name}:{tool_name}"),
1630                        status: StepStatus::Failed,
1631                        message: result.to_string(),
1632                    }
1633                } else {
1634                    StepResult {
1635                        name: format!("[mcp] {server_name}:{tool_name}"),
1636                        status: StepStatus::Passed,
1637                        message: result.to_string(),
1638                    }
1639                }
1640            }
1641            Err(e) => StepResult {
1642                name: format!("[mcp] {server_name}:{tool_name}"),
1643                status: StepStatus::Failed,
1644                message: format!("MCP call failed: {e}"),
1645            },
1646        }
1647    }
1648
1649    // ── helpers ──────────────────────────────────────────────────────────
1650
1651    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1652    /// the runner's default LLM config for any unset fields.
1653    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1654        LlmConfig {
1655            url: if endpoint.url.is_empty() {
1656                self.llm.url.clone()
1657            } else {
1658                endpoint.url.clone()
1659            },
1660            model: endpoint
1661                .model
1662                .clone()
1663                .unwrap_or_else(|| self.llm.model.clone()),
1664            api_key: endpoint
1665                .api_key
1666                .clone()
1667                .or_else(|| self.llm.api_key.clone()),
1668            headers: if endpoint.headers.is_empty() {
1669                self.llm.headers.clone()
1670            } else {
1671                endpoint.headers.clone()
1672            },
1673            timeout: self.llm.timeout,
1674            temperature: self.llm.temperature,
1675            thinking: self.llm.thinking,
1676            model_params: self.llm.model_params.clone(),
1677        }
1678    }
1679
1680    /// Resolves a CSS selector for the target element. Uses the explicit
1681    /// `selector` if provided, otherwise asks the LLM to find the element
1682    /// from the natural language `target` description and page DOM.
1683    ///
1684    /// LLM responses are sanitized and verified against the live page: a
1685    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
1686    /// immediately with the raw LLM output, and a selector that matches
1687    /// nothing triggers one retry with feedback before failing.
1688    #[allow(clippy::too_many_lines)]
1689    fn resolve_selector(
1690        &self,
1691        css_override: Option<&str>,
1692        target: &str,
1693        step_endpoint: Option<&str>,
1694        test_endpoint: Option<&str>,
1695        tab: &Tab,
1696    ) -> Result<String, String> {
1697        if let Some(explicit) = css_override {
1698            return Ok(explicit.to_owned());
1699        }
1700
1701        let dom_info = extract_dom_info(tab)?;
1702        let page_content = get_page_text(tab);
1703
1704        let system = concat!(
1705            "You are a browser automation selector generator. ",
1706            "Given a web page's content and interactive elements, ",
1707            "return ONLY the best CSS selector for the described element. ",
1708            "Output nothing except the CSS selector. ",
1709            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
1710            "[name=\"...\"], tag.class, tag. ",
1711            "Never output explanations, markdown, or extra text."
1712        );
1713
1714        let user = format!(
1715            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
1716            page_content.url,
1717            page_content.title,
1718            truncate(&page_content.body_text, 4000),
1719            dom_info,
1720            target,
1721        );
1722
1723        let retry_user = format!(
1724            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
1725            "The selector must match at least one element currently present on the page.",
1726            page_content.url,
1727            page_content.title,
1728            truncate(&page_content.body_text, 4000),
1729            dom_info,
1730            target,
1731        );
1732
1733        eprintln!("      LLM targeting: {target}");
1734
1735        let endpoint = self
1736            .endpoints
1737            .resolve(step_endpoint.or(test_endpoint), TaskType::Targeting);
1738        let llm = self.build_llm_for_endpoint(endpoint);
1739        let usage = Arc::clone(&self.usage);
1740        let endpoint_name = endpoint.name.clone();
1741        let endpoint_clone = endpoint.clone();
1742        let sys = system.to_owned();
1743
1744        let call_llm = |prompt: &str| {
1745            let llm = llm.clone();
1746            let sys = sys.clone();
1747            let prompt = prompt.to_owned();
1748            std::thread::spawn(move || {
1749                let rt = tokio::runtime::Builder::new_current_thread()
1750                    .enable_all()
1751                    .build()
1752                    .unwrap();
1753                rt.block_on(llm_chat_with_usage(&llm, &sys, &prompt))
1754            })
1755            .join()
1756            .unwrap()
1757        };
1758
1759        let first = call_llm(&user);
1760        let lr = match first {
1761            Ok(lr) => lr,
1762            Err(e) => {
1763                return Err(format!("LLM element targeting failed: {e}"));
1764            }
1765        };
1766        usage.record_llm_call(
1767            &endpoint_name,
1768            &endpoint_clone,
1769            lr.usage.prompt_tokens,
1770            lr.usage.completion_tokens,
1771        );
1772        let clean = sanitize_selector(&lr.content);
1773        eprintln!("      resolved selector: {clean}");
1774
1775        if selector_is_useless(&clean) {
1776            return Err(format!(
1777                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
1778                raw = lr.content.trim(),
1779            ));
1780        }
1781        if let Err(reason) = validate_selector(&clean) {
1782            return Err(format!(
1783                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
1784                raw = lr.content.trim(),
1785            ));
1786        }
1787        if !selector_matches(tab, &clean).unwrap_or(false) {
1788            // One retry with feedback: flaky models occasionally invent a
1789            // selector that does not exist on the page.
1790            eprintln!(
1791                "      selector {clean} matches nothing — retrying LLM targeting with feedback"
1792            );
1793            let second = call_llm(&retry_user);
1794            let lr2 = match second {
1795                Ok(lr2) => lr2,
1796                Err(e) => {
1797                    return Err(format!(
1798                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
1799                    ));
1800                }
1801            };
1802            usage.record_llm_call(
1803                &endpoint_name,
1804                &endpoint_clone,
1805                lr2.usage.prompt_tokens,
1806                lr2.usage.completion_tokens,
1807            );
1808            let clean2 = sanitize_selector(&lr2.content);
1809            eprintln!("      resolved selector (retry): {clean2}");
1810            if selector_is_useless(&clean2) {
1811                return Err(format!(
1812                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
1813                    raw = lr2.content.trim(),
1814                    excerpt = truncate(&page_content.body_text, 300),
1815                ));
1816            }
1817            if !selector_matches(tab, &clean2).unwrap_or(false) {
1818                return Err(format!(
1819                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
1820                ));
1821            }
1822            return Ok(clean2);
1823        }
1824
1825        Ok(clean)
1826    }
1827}
1828
1829/// Evaluates a JS expression that is expected to return a boolean.
1830fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
1831    tab.evaluate(js, false)
1832        .map_err(|e| format!("evaluate failed: {e}"))?
1833        .value
1834        .and_then(|v| v.as_bool())
1835        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
1836}
1837
1838/// Checks whether a CSS selector matches at least one current element.
1839fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
1840    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
1841}
1842
1843// ── Free helper functions ──────────────────────────────────────────────
1844
1845fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
1846    let name = format!("[navigate] {full_url}");
1847    match tab.navigate_to(full_url) {
1848        Ok(_) => {
1849            let _ = tab.wait_until_navigated();
1850            StepResult {
1851                name,
1852                status: StepStatus::Passed,
1853                message: format!("navigated to {full_url}"),
1854            }
1855        }
1856        Err(e) => StepResult {
1857            name,
1858            status: StepStatus::Failed,
1859            message: format!("navigation failed: {e}"),
1860        },
1861    }
1862}
1863
1864fn extract_dom_info(tab: &Tab) -> Result<String, String> {
1865    let result = tab
1866        .evaluate(DOM_EXTRACT_JS, false)
1867        .map_err(|e| format!("DOM extraction failed: {e}"))?;
1868
1869    let json_str = result
1870        .value
1871        .as_ref()
1872        .and_then(|v| v.as_str())
1873        .unwrap_or("[]");
1874
1875    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
1876
1877    if elements.is_empty() {
1878        return Ok("(no interactive elements found)".to_owned());
1879    }
1880
1881    Ok(elements.join("\n"))
1882}
1883
1884fn get_page_text(tab: &Tab) -> PageContent {
1885    let url = tab.get_url();
1886
1887    let title = tab
1888        .evaluate("document.title", false)
1889        .ok()
1890        .and_then(|r| r.value)
1891        .and_then(|v| v.as_str().map(String::from))
1892        .unwrap_or_else(|| "unknown".to_owned());
1893
1894    let body_text = tab
1895        .evaluate(
1896            "document.body ? document.body.innerText : document.documentElement.innerText",
1897            false,
1898        )
1899        .ok()
1900        .and_then(|r| r.value)
1901        .and_then(|v| v.as_str().map(String::from))
1902        .unwrap_or_default();
1903
1904    PageContent {
1905        url,
1906        title,
1907        body_text: truncate(&body_text, 8000),
1908    }
1909}
1910
1911fn resolve_url(url: &str, base_url: &str) -> String {
1912    if url.starts_with("http://") || url.starts_with("https://") {
1913        return url.to_owned();
1914    }
1915    let base = base_url.trim_end_matches('/');
1916    if url.starts_with('/') {
1917        format!("{base}{url}")
1918    } else {
1919        format!("{base}/{url}")
1920    }
1921}
1922
1923/// Human-readable label for a step, used when steps are skipped after an
1924/// earlier failure.
1925fn step_label(step: &TestStep) -> String {
1926    match step {
1927        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
1928        TestStep::Click { target, .. } => format!("[click] {target}"),
1929        TestStep::Type { target, .. } => format!("[type] {target}"),
1930        TestStep::Wait { target, .. } => format!("[wait] {target}"),
1931        TestStep::Assert {
1932            definition,
1933            preset,
1934            prompt,
1935            ..
1936        } => definition.as_ref().map_or_else(
1937            || {
1938                preset.as_ref().map_or_else(
1939                    || {
1940                        prompt.as_ref().map_or_else(
1941                            || "[assert]".to_owned(),
1942                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
1943                        )
1944                    },
1945                    |p| format!("[assert] {p}"),
1946                )
1947            },
1948            |d| format!("[assert] {d}"),
1949        ),
1950        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
1951        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
1952        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
1953    }
1954}
1955
1956/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
1957#[must_use]
1958const fn step_kind_label(step: &TestStep) -> &'static str {
1959    match step {
1960        TestStep::Navigate { .. } => "navigate",
1961        TestStep::Click { .. } => "click",
1962        TestStep::Type { .. } => "type",
1963        TestStep::Wait { .. } => "wait",
1964        TestStep::Assert { .. } => "assert",
1965        TestStep::Screenshot { .. } => "screenshot",
1966        TestStep::Agent { .. } => "agent",
1967        TestStep::Mcp { .. } => "mcp",
1968    }
1969}
1970
1971// ── Support types ──────────────────────────────────────────────────────
1972
1973#[derive(Default)]
1974struct TestRunResult {
1975    passed: u32,
1976    failed: u32,
1977    skipped: u32,
1978    total: u32,
1979    details: Vec<StepResult>,
1980}
1981
1982struct PageContent {
1983    url: String,
1984    title: String,
1985    body_text: String,
1986}