1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::{calculate_llm_cost, UsageTracker};
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, ResolvedEndpoint, TaskType};
14use crate::events::TestEvent;
15use crate::llm_chat_vision_with_usage_chain;
16use crate::llm_chat_with_usage_chain;
17use crate::mcp_client::McpClient;
18use crate::reporting::Reporter;
19use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
20use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
21use crate::truncate;
22use crate::LlmConfig;
23use crate::DOM_EXTRACT_JS;
24
25#[must_use]
29#[allow(clippy::many_single_char_names)]
30fn unix_to_rfc3339(secs: u64) -> String {
31 let secs = i64::try_from(secs).unwrap_or(0);
32 let days = secs.div_euclid(86_400);
33 let rem = secs.rem_euclid(86_400);
34 let (h, m, s) = (rem / 3_600, (rem % 3_600) / 60, rem % 60);
35 let z = days + 719_468;
36 let era = z.div_euclid(146_097);
37 let doe = z.rem_euclid(146_097);
38 let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365;
39 let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
40 let mp = (5 * doy + 2) / 153;
41 let d = doy - (153 * mp + 2) / 5 + 1;
42 let mth = if mp < 10 { mp + 3 } else { mp - 9 };
43 let y = yoe + era * 400 + i64::from(mth <= 2);
44 format!("{y:04}-{mth:02}-{d:02}T{h:02}:{m:02}:{s:02}Z")
45}
46
47#[derive(Debug, serde::Deserialize)]
49struct LayoutIssue {
50 #[serde(rename = "type")]
51 issue_type: String,
52 element: String,
53 detail: String,
54}
55
56const LAYOUT_SCAN_JS: &str = r#"
80(() => {
81 const issues = [];
82 const push = (type, el, detail) => {
83 if (issues.length >= 30) return;
84 let element = el.tagName.toLowerCase();
85 if (el.id) element += '#' + el.id;
86 else if (typeof el.className === 'string' && el.className.trim())
87 element += '.' + el.className.trim().split(/\s+/).join('.');
88 issues.push({ type, element, detail: String(detail).slice(0, 220) });
89 };
90 const vw = document.documentElement.clientWidth || window.innerWidth;
91 const vh = document.documentElement.clientHeight || window.innerHeight;
92 if (!vw || !vh) return JSON.stringify(issues);
93 const de = document.documentElement;
94 // 1. Page-level horizontal overflow.
95 if (maxSW > vw + 2)
96 push('page-overflow-x', de,
97 'page scrollWidth ' + maxSW + ' exceeds viewport width ' + vw);
98 // Class-name prefixes to skip (injected; default = Angular CDK
99 // screen-reader helpers, which are intentionally 1x1 / off-screen).
100 const ignorePrefixes = __IGNORE_CLASSES__;
101 const isIgnored = (el) => {
102 if (typeof el.className !== 'string' || !el.className.trim()) return false;
103 const classes = el.className.trim().split(/\s+/);
104 return classes.some((c) => ignorePrefixes.some((p) => c.startsWith(p)));
105 };
106 const body = document.body;
107 // The document element is not always the scroll container: the app may
108 // scroll via <body> (html{overflow:hidden} + body scroll, e.g. sticky
109 // navs) or via an inner overflow-y:auto panel. Never treat those pages as
110 // "not scrollable" — measure the scrollable content height/width of the
111 // document element AND the body, and take the max.
112 const maxSH = Math.max(de.scrollHeight, body ? body.scrollHeight : 0);
113 const maxSW = Math.max(de.scrollWidth, body ? body.scrollWidth : 0);
114 const vScrollable = maxSH > vh + 2;
115 const hScrollable = maxSW > vw + 2;
116 // True when an ancestor panel (overflow:auto/scroll/overlay) can scroll
117 // this element into view on the given axis — i.e. the element is inside a
118 // scrolling region, so lying beyond the viewport cut is not a defect.
119 const reachableByScroller = (el, axis) => {
120 const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
121 const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
122 const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
123 let p = el.parentElement;
124 while (p && p !== body) {
125 const pcs = getComputedStyle(p);
126 const o = pcs[ovProp];
127 if ((o === 'auto' || o === 'scroll' || o === 'overlay') &&
128 p[dim] > p[dimClient] + 2) return true;
129 p = p.parentElement;
130 }
131 return false;
132 };
133 const selfOverflowing = (el, cs, axis) => {
134 const ov = axis === 'y' ? cs.overflowY : cs.overflowX;
135 const dim = axis === 'y' ? 'scrollHeight' : 'scrollWidth';
136 const dimClient = axis === 'y' ? 'clientHeight' : 'clientWidth';
137 return (ov === 'auto' || ov === 'scroll' || ov === 'overlay') &&
138 el[dim] > el[dimClient] + 2;
139 };
140 // True when an ancestor clips this axis with overflow:hidden/clip — the
141 // element's overhang is not visible, so treat it as reachable.
142 const clippedByAncestor = (el, axis) => {
143 const ovProp = axis === 'y' ? 'overflowY' : 'overflowX';
144 let p = el.parentElement;
145 while (p && p !== body) {
146 const o = getComputedStyle(p)[ovProp];
147 if (o === 'hidden' || o === 'clip') return true;
148 p = p.parentElement;
149 }
150 return false;
151 };
152 const all = Array.prototype.slice.call(document.querySelectorAll('body *'));
153 const visible = (cs) => cs.display !== 'none' && cs.visibility !== 'hidden' && parseFloat(cs.opacity || '1') !== 0;
154 const hasContent = (el) =>
155 ((el.textContent || '').trim().length > 0) ||
156 !!el.querySelector('img,svg,video,canvas,iframe,button,input,textarea,select');
157 // 2. Elements outside the viewport that scrolling cannot reveal.
158 for (const el of all) {
159 const cs = getComputedStyle(el);
160 if (!visible(cs)) continue;
161 if (isIgnored(el)) continue;
162 const r = el.getBoundingClientRect();
163 if (r.width < 2 || r.height < 2) continue;
164 if (!hasContent(el) && el.children.length === 0) continue;
165 if (cs.position === 'fixed') {
166 // Fixed elements never move with the scroll: any edge outside the
167 // viewport is unreachable content and therefore a defect.
168 const overTop = -r.top;
169 const overLeft = -r.left;
170 const overRight = r.right - vw;
171 const overBottom = r.bottom - vh;
172 // Top/left overshoot is never reachable; bottom/right overshoot is
173 // only a defect when the fixed element cannot scroll that content
174 // into view itself (e.g. an opened Material drawer whose inner
175 // container scrolls is not a "cut-off" defect).
176 if (overTop > 2 || overLeft > 2 ||
177 (overRight > 2 && !selfOverflowing(el, cs, 'x')) ||
178 (overBottom > 2 && !selfOverflowing(el, cs, 'y'))) {
179 let where = '';
180 if (overTop > 2 && overLeft > 2) where = 'top+left edges';
181 else if (overTop > 2) where = 'top edge (' + Math.round(r.top) + ' < 0)';
182 else if (overLeft > 2) where = 'left edge (' + Math.round(r.left) + ' < 0)';
183 else if (overRight > 2 && overBottom > 2) where = 'right+bottom edges';
184 else if (overRight > 2) where = 'right edge (' + Math.round(r.right) + ' > ' + vw + ')';
185 else where = 'bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')';
186 push('element-out-of-viewport', el,
187 'fixed element lies ' + Math.max(overTop, overLeft, overRight, overBottom).toFixed(0) + 'px past the ' + where);
188 }
189 continue;
190 }
191 if (cs.position === 'sticky') continue;
192 // Negative left overflow cannot be reached by scrolling (scrollLeft
193 // never goes below 0).
194 if (r.left < -2) {
195 push('element-out-of-viewport', el,
196 'extends ' + Math.round(-r.left).toFixed(0) + 'px past the left edge (' + Math.round(r.left) + ' < 0)');
197 continue;
198 }
199 // Negative top with the page at the top means the element sits above
200 // the document origin — also unreachable.
201 if (r.top < -2 && de.scrollTop <= 2) {
202 push('element-out-of-viewport', el,
203 'extends ' + Math.round(-r.top).toFixed(0) + 'px past the top edge (' + Math.round(r.top) + ' < 0)');
204 continue;
205 }
206 const overRight = r.right - vw;
207 // Right-edge overflow is only a defect when the page cannot scroll
208 // horizontally to reveal it (or the element sticks out past the
209 // scrollable content width itself).
210 if (overRight > 2 && (!hScrollable || r.right > maxSW + 2) && !reachableByScroller(el, 'x') && !clippedByAncestor(el, 'x')) {
211 push('element-out-of-viewport', el,
212 'extends ' + overRight.toFixed(0) + 'px past the right edge (' + Math.round(r.right) + ' > ' + vw + ')');
213 continue;
214 }
215 // Below-the-fold content on a scrollable page is normal (tall landing
216 // pages); only flag bottom overflow the user can never scroll to.
217 const overBottom = r.bottom - vh;
218 if (overBottom > 2 && (!vScrollable || r.bottom > maxSH + 2) && !reachableByScroller(el, 'y') && !clippedByAncestor(el, 'y')) {
219 push('element-out-of-viewport', el,
220 'extends ' + overBottom.toFixed(0) + 'px past the bottom edge (' + Math.round(r.bottom) + ' > ' + vh + ')');
221 }
222 }
223 // 3. Text clipped by overflow:hidden containers.
224 for (const el of all) {
225 if (isIgnored(el)) continue;
226 const cs = getComputedStyle(el);
227 if (cs.overflowX !== 'hidden' && cs.overflowY !== 'hidden') continue;
228 if (el.scrollWidth <= el.clientWidth + 2 && el.scrollHeight <= el.clientHeight + 2) continue;
229 if (!(el.textContent || '').trim()) continue;
230 // Single-line truncation with an ellipsis is an intentional design
231 // pattern (Tailwind .truncate etc.), not a clipping defect.
232 if (cs.textOverflow === 'ellipsis') continue;
233 push('text-clipped', el,
234 'content ' + el.scrollWidth + 'x' + el.scrollHeight +
235 ' clipped to ' + el.clientWidth + 'x' + el.clientHeight);
236 }
237 // 4. Interactive elements covered by a different element.
238 const interactive =
239 'button, a[href], input, textarea, select, [role="button"], [role="link"], [role="menuitem"], label, .mdc-button, .mat-mdc-button, .mat-mdc-icon-button, .mdc-fab';
240 const targets = document.querySelectorAll(interactive);
241 for (const el of targets) {
242 if (isIgnored(el)) continue;
243 const r = el.getBoundingClientRect();
244 if (r.width < 6 || r.height < 6) continue;
245 const cs = getComputedStyle(el);
246 if (!visible(cs)) continue;
247 const cx = r.left + r.width / 2;
248 const cy = r.top + r.height / 2;
249 if (cx < 0 || cy < 0 || cx > vw || cy > vh) continue;
250 const top = document.elementFromPoint(cx, cy);
251 if (!top || top === el || el.contains(top) || top.contains(el)) continue;
252 if (isIgnored(top)) continue;
253 const tcs = getComputedStyle(top);
254 if (!visible(tcs)) continue;
255 if (tcs.pointerEvents === 'none') continue;
256 const tr = top.getBoundingClientRect();
257 if (tr.width * tr.height < r.width * r.height * 0.25) continue;
258 const tname = top.tagName.toLowerCase() + (top.id ? '#' + top.id : '') +
259 (typeof top.className === 'string' && top.className.trim() ? '.' + top.className.trim().split(/\s+/).join('.') : '');
260 push('element-overlap', el,
261 'center point at ' + Math.round(cx) + ',' + Math.round(cy) +
262 ' is covered by <' + tname + '>');
263 }
264 return JSON.stringify(issues);
265})()
266"#;
267
268const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
274
275pub struct ScenarioRunner {
278 config: ScenarioConfig,
279 definitions: HashMap<String, AssertDefinition>,
280 llm: LlmConfig,
281 timeout: Duration,
282 viewport_width: u32,
283 viewport_height: u32,
284 applied_viewport: std::cell::Cell<(u32, u32)>,
287 endpoints: EndpointRegistry,
288 usage: Arc<UsageTracker>,
289 budgets: BudgetTracker,
290 artifacts_dir: PathBuf,
292 reporter: Arc<Reporter>,
294 current_step: std::cell::RefCell<Option<(String, u32)>>,
296 run_started: u64,
299 emit_run_events: bool,
304}
305
306#[derive(Debug, Default)]
308pub struct RunReport {
309 pub tests_passed: u32,
311 pub tests_failed: u32,
313 pub passed: u32,
315 pub failed: u32,
317 pub skipped: u32,
319 pub details: Vec<StepResult>,
321}
322
323#[derive(Debug, Clone)]
325pub struct StepResult {
326 pub name: String,
328 pub status: StepStatus,
330 pub message: String,
332}
333
334pub use crate::events::StepStatus;
336
337struct AssertPreset {
339 name: &'static str,
340 system: &'static str,
341 user_template: &'static str,
342}
343
344#[allow(clippy::literal_string_with_formatting_args)]
346const ASSERTION_PRESETS: &[AssertPreset] = &[
347 AssertPreset {
348 name: "no_error_on_page",
349 system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
350 user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
351 },
352 AssertPreset {
353 name: "text_visible",
354 system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
355 user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
356 },
357 AssertPreset {
358 name: "element_exists",
359 system: "You are a QA tester. Check if a described UI element exists on a web page.",
360 user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
361 },
362 AssertPreset {
363 name: "visual_no_issues",
364 system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
365 user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
366 },
367 AssertPreset {
368 name: "visual_no_overlaps",
369 system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
370 user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
371 },
372 AssertPreset {
373 name: "visual_text_visible",
374 system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
375 user_template: "Inspect the attached screenshot(s) (ordered from the top of the page down) and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
376 },
377 AssertPreset {
378 name: "layout_no_issues",
379 system: "DOM layout scanner: reports horizontal page overflow, elements sticking out of the viewport, clipped text in overflow:hidden containers, and interactive elements covered by other elements. Deterministic in-browser checks — no LLM call.",
380 user_template: "Runs a deterministic DOM layout scan in the browser (no LLM call). Fails with the list of detected issues.",
381 },
382];
383
384fn verdict_is_pass(content: &str) -> bool {
391 content
392 .trim()
393 .trim_start_matches(|c: char| !c.is_alphanumeric())
394 .to_lowercase()
395 .starts_with("pass")
396}
397
398fn verdict_is_fail(content: &str) -> bool {
400 content
401 .trim()
402 .trim_start_matches(|c: char| !c.is_alphanumeric())
403 .to_lowercase()
404 .starts_with("fail")
405}
406
407impl ScenarioRunner {
408 #[must_use]
411 #[allow(clippy::needless_pass_by_value)]
412 pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
413 Self::with_reporter(scenario_config, definitions, Arc::new(Reporter::default()))
414 }
415
416 #[must_use]
418 #[allow(clippy::needless_pass_by_value)]
419 pub fn with_reporter(
420 scenario_config: ScenarioConfig,
421 definitions: Vec<AssertDefinition>,
422 reporter: Arc<Reporter>,
423 ) -> Self {
424 Self::with_reporter_mode(scenario_config, definitions, reporter, true)
425 }
426
427 #[must_use]
431 #[allow(clippy::needless_pass_by_value)]
432 pub fn with_reporter_parallel(
433 scenario_config: ScenarioConfig,
434 definitions: Vec<AssertDefinition>,
435 reporter: Arc<Reporter>,
436 ) -> Self {
437 Self::with_reporter_mode(scenario_config, definitions, reporter, false)
438 }
439
440 #[must_use]
441 #[allow(clippy::needless_pass_by_value)]
442 fn with_reporter_mode(
443 scenario_config: ScenarioConfig,
444 definitions: Vec<AssertDefinition>,
445 reporter: Arc<Reporter>,
446 emit_run_events: bool,
447 ) -> Self {
448 reporter.add_redaction_secrets(crate::redact::collect_secrets_from_scenario_config(
449 &scenario_config,
450 ));
451 let llm = LlmConfig {
452 url: scenario_config
453 .llm_url
454 .clone()
455 .unwrap_or_else(crate::llm_base_url),
456 model: scenario_config
457 .llm_model
458 .clone()
459 .unwrap_or_else(crate::llm_model),
460 api_key: scenario_config
461 .llm_api_key
462 .clone()
463 .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
464 headers: if scenario_config.llm_headers.is_empty() {
465 crate::parse_headers_env()
466 } else {
467 scenario_config.llm_headers.clone()
468 },
469 timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
470 temperature: scenario_config.temperature,
471 thinking: scenario_config.thinking,
472 model_params: scenario_config.model_params.clone(),
473 cache: scenario_config.cache.unwrap_or(true),
474 max_attempts: crate::default_llm_attempts(),
475 provider: crate::scenario::Provider::Openai,
476 deployment: None,
477 api_version: None,
478 auth: crate::scenario::AuthConfig::default(),
479 header_commands: std::collections::HashMap::new(),
480 aws: crate::scenario::AwsConfig::default(),
481 };
482 let endpoints = EndpointRegistry::from_config(
483 &scenario_config.endpoints,
484 Some(&llm),
485 crate::endpoints::EndpointDefaults {
486 cache: scenario_config.cache,
487 cache_pricing: scenario_config.cache_pricing,
488 },
489 );
490 let budgets = BudgetTracker::from_config(&scenario_config.budgets);
491 let defs_map: HashMap<String, AssertDefinition> = definitions
492 .into_iter()
493 .map(|d| (d.name.clone(), d))
494 .collect();
495
496 Self {
497 timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
498 viewport_width: scenario_config.viewport_width.unwrap_or(1280),
499 viewport_height: scenario_config.viewport_height.unwrap_or(720),
500 applied_viewport: std::cell::Cell::new((0, 0)),
501 config: scenario_config.clone(),
502 definitions: defs_map,
503 llm,
504 endpoints,
505 usage: Arc::new(UsageTracker::new()),
506 budgets,
507 artifacts_dir: PathBuf::from(
508 scenario_config
509 .artifacts_dir
510 .unwrap_or_else(|| "artifacts".to_owned()),
511 ),
512 reporter,
513 current_step: std::cell::RefCell::new(None),
514 run_started: SystemTime::now()
515 .duration_since(UNIX_EPOCH)
516 .map_or(0, |d| d.as_secs()),
517 emit_run_events,
518 }
519 }
520
521 fn emit_event(&self, event: &TestEvent) {
524 if let Err(err) = self.reporter.emit(event) {
525 use std::io::Write as _;
526 let mut out = std::io::stderr().lock();
527 let _ = out.write_fmt(format_args!(" ! failed to record event: {err}\n"));
528 }
529 }
530
531 #[must_use]
533 pub fn usage_tracker(&self) -> Arc<UsageTracker> {
534 Arc::clone(&self.usage)
535 }
536
537 #[must_use]
539 pub const fn budget_tracker(&self) -> &BudgetTracker {
540 &self.budgets
541 }
542
543 #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
549 pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
550 let mut report = RunReport::default();
551
552 if tests.is_empty() {
553 self.reporter.warn("No tests defined in scenario.");
554 if self.emit_run_events {
555 self.emit_event(&TestEvent::RunFinished {
556 tests_passed: 0,
557 tests_failed: 0,
558 steps_passed: 0,
559 steps_failed: 0,
560 steps_skipped: 0,
561 total_cost: 0.0,
562 total_tokens: 0,
563 total_input_tokens: 0,
564 total_output_tokens: 0,
565 total_cached_input_tokens: 0,
566 total_cache_creation_input_tokens: 0,
567 models: Vec::new(),
568 total_calls: 0,
569 });
570 }
571 return Ok(report);
572 }
573
574 if self.emit_run_events {
575 self.emit_event(&TestEvent::RunStarted {
576 total_tests: tests.len() as u32,
577 });
578 }
579
580 let browser_headless = self.config.browser_headless.unwrap_or(true);
581
582 let launch_opts = LaunchOptions {
583 headless: browser_headless,
584 window_size: Some((self.viewport_width, self.viewport_height)),
585 sandbox: false,
586 idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
595 ..LaunchOptions::default()
596 };
597
598 let browser = Browser::new(launch_opts).context("failed to launch browser")?;
599 let tab = browser.new_tab().context("failed to open browser tab")?;
600 match (
601 self.config.browser_basic_auth_user.clone(),
602 self.config.browser_basic_auth_password.clone(),
603 ) {
604 (Some(username), Some(password)) if !username.is_empty() && !password.is_empty() => {
605 tab.authenticate(Some(username), Some(password))
606 .context("failed to configure browser HTTP Basic Auth")?;
607 tab.enable_fetch(None, Some(true))
608 .context("failed to enable browser HTTP authentication")?;
609 }
610 (None, None) => {}
611 _ => anyhow::bail!("browser Basic Auth requires both username and password"),
612 }
613 let _ = tab.set_default_timeout(self.timeout);
614
615 #[cfg(feature = "mcp-server")]
617 if let Some(ref mcp_cfg) = self.config.mcp_server {
618 if mcp_cfg.enabled {
619 let port = mcp_cfg.port;
620 std::thread::spawn(move || {
621 let _ = crate::mcp_server::start_mcp_server(port);
622 });
623 }
624 }
625 #[cfg(not(feature = "mcp-server"))]
626 if let Some(mcp_cfg) = &self.config.mcp_server {
627 if mcp_cfg.enabled {
628 self.reporter
629 .warn("MCP server configured but 'mcp-server' feature not enabled");
630 }
631 }
632
633 #[cfg(feature = "a2a-server")]
635 if let Some(ref a2a_cfg) = self.config.a2a_server {
636 if a2a_cfg.enabled {
637 let port = a2a_cfg.port;
638 tokio::spawn(crate::a2a_server::start_a2a_server(port));
639 }
640 }
641 #[cfg(not(feature = "a2a-server"))]
642 if let Some(a2a_cfg) = &self.config.a2a_server {
643 if a2a_cfg.enabled {
644 self.reporter
645 .warn("A2A server configured but 'a2a-server' feature not enabled");
646 }
647 }
648
649 for test in tests {
650 self.emit_event(&TestEvent::TestStarted {
651 test: test.name.clone(),
652 });
653
654 self.usage.reset_per_test();
655
656 let test_started = Instant::now();
657 let test_result = self.run_test(test, &tab);
658 let duration_ms = test_started.elapsed().as_millis() as u64;
659 let usage = self.usage.current_test_snapshot();
660 self.usage.commit_test(&test.name);
661
662 self.emit_event(&TestEvent::TestFinished {
663 test: test.name.clone(),
664 passed: test_result.passed,
665 failed: test_result.failed,
666 skipped: test_result.skipped,
667 duration_ms,
668 cost: usage.total_cost,
669 tokens: usage.total_tokens,
670 input_tokens: usage.total_input_tokens,
671 output_tokens: usage.total_output_tokens,
672 cached_input_tokens: usage.total_cached_input_tokens,
673 cache_creation_input_tokens: usage.total_cache_creation_input_tokens,
674 models: usage.models.clone(),
675 calls: usage.total_calls,
676 });
677
678 if test_result.failed == 0 && test_result.total > 0 {
679 report.tests_passed += 1;
680 } else if test_result.total > 0 {
681 report.tests_failed += 1;
682 }
683
684 report.passed += test_result.passed;
685 report.failed += test_result.failed;
686 report.skipped += test_result.skipped;
687 report.details.extend(test_result.details);
688 }
689
690 let global = self.usage.global_snapshot();
691 if self.emit_run_events {
692 self.emit_event(&TestEvent::RunFinished {
693 tests_passed: report.tests_passed,
694 tests_failed: report.tests_failed,
695 steps_passed: report.passed,
696 steps_failed: report.failed,
697 steps_skipped: report.skipped,
698 total_cost: global.total_cost,
699 total_tokens: global.total_tokens,
700 total_input_tokens: global.total_input_tokens,
701 total_output_tokens: global.total_output_tokens,
702 total_cached_input_tokens: global.total_cached_input_tokens,
703 total_cache_creation_input_tokens: global.total_cache_creation_input_tokens,
704 models: global.models.clone(),
705 total_calls: global.total_calls,
706 });
707 }
708
709 Ok(report)
710 }
711
712 #[allow(clippy::too_many_lines, clippy::cast_possible_truncation)]
713 fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
714 let base_url = test
715 .base_url
716 .clone()
717 .or_else(|| self.config.base_url.clone())
718 .unwrap_or_else(crate::base_url);
719
720 let vw = test.viewport_width.unwrap_or(self.viewport_width);
723 let vh = test.viewport_height.unwrap_or(self.viewport_height);
724 if self.applied_viewport.get() != (vw, vh) {
725 self.apply_viewport(tab, vw, vh);
726 self.applied_viewport.set((vw, vh));
727 }
728
729 let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
733
734 let start_url = test
735 .start_url
736 .clone()
737 .or_else(|| self.config.start_url.clone())
738 .unwrap_or_else(|| "/dashboard".to_owned());
739
740 if auto_navigate {
741 let full_url = resolve_url(&start_url, &base_url);
742 self.reporter.debug(format!("auto-navigate: {full_url}"));
743 let _ = tab.navigate_to(&full_url);
744 let _ = tab.wait_until_navigated();
745 std::thread::sleep(Duration::from_secs(4));
746 }
747
748 let mut result = TestRunResult::default();
749
750 for (step_index, step) in test.steps.iter().enumerate() {
751 result.total += 1;
752
753 let wait_ms = match step {
754 TestStep::Navigate { wait_after_ms, .. }
755 | TestStep::Click { wait_after_ms, .. }
756 | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
757 _ => None,
758 };
759
760 self.current_step
761 .replace(Some((test.name.clone(), step_index as u32)));
762 self.emit_event(&TestEvent::StepStarted {
763 test: test.name.clone(),
764 index: step_index as u32,
765 label: step_label(step),
766 });
767 let step_started = Instant::now();
768
769 let mut step_result = match step {
770 TestStep::Navigate { url, .. } => {
771 let full_url = resolve_url(url, &base_url);
772 run_navigate_step(&full_url, tab)
773 }
774 TestStep::Click {
775 target,
776 selector,
777 endpoint,
778 idempotent,
779 ..
780 } => self.run_click(
781 target,
782 selector.as_deref(),
783 endpoint.as_deref(),
784 test.endpoint.as_deref(),
785 *idempotent,
786 tab,
787 ),
788 TestStep::Type {
789 target,
790 text,
791 selector,
792 endpoint,
793 idempotent,
794 ..
795 } => self.run_type(
796 target,
797 text,
798 selector.as_deref(),
799 endpoint.as_deref(),
800 test.endpoint.as_deref(),
801 *idempotent,
802 tab,
803 ),
804 TestStep::Wait {
805 target,
806 selector,
807 text,
808 timeout_ms,
809 endpoint,
810 idempotent,
811 } => self.run_wait(
812 target,
813 selector.as_deref(),
814 text.as_deref(),
815 *timeout_ms,
816 endpoint.as_deref(),
817 test.endpoint.as_deref(),
818 *idempotent,
819 tab,
820 ),
821 TestStep::Assert {
822 definition,
823 preset,
824 prompt,
825 assert_text,
826 endpoint,
827 screenshot,
828 } => self.run_assert(
829 definition.as_deref(),
830 preset.as_deref(),
831 prompt.as_deref(),
832 assert_text.as_deref(),
833 *screenshot,
834 endpoint.as_deref(),
835 test.endpoint.as_deref(),
836 tab,
837 ),
838 TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
839 TestStep::Agent {
840 agent,
841 task,
842 definition,
843 } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
844 TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
845 };
846
847 let (diagnostics_block, screenshot_path) = if step_result.status == StepStatus::Failed {
851 let state = diagnostics::capture(tab);
852 let screenshot = diagnostics::save_screenshot(
853 tab,
854 &self.artifacts_dir,
855 &test.name,
856 &test.name,
857 step_index,
858 step_kind_label(step),
859 );
860 step_result.message = format!(
861 "{base} — {excerpt}",
862 base = step_result.message,
863 excerpt = diagnostics::inline_excerpt(&state),
864 );
865 (Some(diagnostics::full_context(&state)), screenshot)
866 } else {
867 (None, None)
868 };
869
870 let duration_ms = step_started.elapsed().as_millis() as u64;
871 self.emit_event(&TestEvent::StepFinished {
872 test: test.name.clone(),
873 index: step_index as u32,
874 label: step_result.name.clone(),
875 status: step_result.status,
876 duration_ms,
877 message: step_result.message.clone(),
878 diagnostics: diagnostics_block,
879 screenshot: screenshot_path,
880 });
881 self.current_step.replace(None);
882
883 match step_result.status {
884 StepStatus::Passed => result.passed += 1,
885 StepStatus::Failed => result.failed += 1,
886 StepStatus::Skipped => result.skipped += 1,
887 }
888
889 if step_result.status == StepStatus::Failed
893 && !self.config.continue_on_failure
894 && step_index + 1 < test.steps.len()
895 {
896 self.emit_event(&TestEvent::Warning {
897 message: format!(
898 "failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
899 test.steps.len() - step_index - 1
900 ),
901 });
902 for (offset, skipped) in test.steps[step_index + 1..].iter().enumerate() {
903 let skipped_index = step_index + 1 + offset;
904 let label = step_label(skipped);
905 result.total += 1;
906 result.skipped += 1;
907 self.emit_event(&TestEvent::StepStarted {
908 test: test.name.clone(),
909 index: skipped_index as u32,
910 label: label.clone(),
911 });
912 self.emit_event(&TestEvent::StepFinished {
913 test: test.name.clone(),
914 index: skipped_index as u32,
915 label,
916 status: StepStatus::Skipped,
917 duration_ms: 0,
918 message: "skipped: previous step failed".into(),
919 diagnostics: None,
920 screenshot: None,
921 });
922 result.details.push(StepResult {
923 name: step_label(skipped),
924 status: StepStatus::Skipped,
925 message: "skipped: previous step failed".into(),
926 });
927 }
928 result.details.push(step_result);
929 return result;
930 }
931
932 let test_usage = self.usage.current_test_snapshot();
934 let global_usage = self.usage.global_snapshot();
935 let budget_status = self.budgets.check_all(
936 &test.name,
937 &test_usage,
938 &global_usage,
939 test.budget.as_ref(),
940 );
941 match budget_status {
942 BudgetStatus::HardExceeded { message, .. } => {
943 self.emit_event(&TestEvent::Warning {
944 message: format!("budget exceeded: {message}"),
945 });
946 result.details.push(StepResult {
947 name: "[budget]".into(),
948 status: StepStatus::Failed,
949 message,
950 });
951 result.failed += 1;
952 return result;
953 }
954 BudgetStatus::SoftExceeded { message, .. } => {
955 self.emit_event(&TestEvent::Warning {
956 message: format!("budget warning: {message}"),
957 });
958 }
959 BudgetStatus::Ok => {}
960 }
961
962 if let Some(ms) = wait_ms {
963 std::thread::sleep(Duration::from_millis(ms));
964 }
965
966 result.details.push(step_result);
967 }
968
969 result
970 }
971
972 fn apply_viewport(&self, tab: &Tab, width: u32, height: u32) {
978 use headless_chrome::protocol::cdp::Emulation;
979 let _ = self;
980 let params = Emulation::SetDeviceMetricsOverride {
981 width,
982 height,
983 device_scale_factor: 1.0,
984 mobile: false,
985 scale: None,
986 screen_width: Some(width),
987 screen_height: Some(height),
988 position_x: None,
989 position_y: None,
990 dont_set_visible_size: None,
991 screen_orientation: None,
992 viewport: None,
993 display_feature: None,
994 device_posture: None,
995 };
996 self.reporter.debug(format!("viewport: {width}x{height}"));
997 if let Err(e) = tab.call_method(params) {
998 self.reporter
999 .warn(format!("viewport switch to {width}x{height} failed: {e}"));
1000 }
1001 }
1002
1003 #[must_use]
1010 fn screenshot_height_cap(&self) -> u32 {
1011 let viewport_height = self.current_viewport_height();
1012 let cap = self
1013 .config
1014 .screenshot_max_height
1015 .as_ref()
1016 .map_or(viewport_height * 20, |h| h.to_px(viewport_height));
1017 cap.max(viewport_height)
1018 }
1019
1020 #[must_use]
1023 const fn current_viewport_height(&self) -> u32 {
1024 let (_, height) = self.applied_viewport.get();
1025 if height > 0 {
1026 height
1027 } else {
1028 self.viewport_height
1029 }
1030 }
1031
1032 #[allow(clippy::too_many_lines)]
1035 fn run_click(
1036 &self,
1037 target: &str,
1038 selector_override: Option<&str>,
1039 step_endpoint: Option<&str>,
1040 test_endpoint: Option<&str>,
1041 idempotent: bool,
1042 tab: &Tab,
1043 ) -> StepResult {
1044 let name = format!("[click] {target}");
1045 let selector = match self.resolve_selector(
1046 selector_override,
1047 target,
1048 step_endpoint,
1049 test_endpoint,
1050 tab,
1051 ) {
1052 Ok(s) => s,
1053 Err(msg) => {
1054 if idempotent {
1055 return StepResult {
1056 name,
1057 status: StepStatus::Skipped,
1058 message: format!("skipped (idempotent): no target found — {msg}"),
1059 };
1060 }
1061 return StepResult {
1062 name,
1063 status: StepStatus::Failed,
1064 message: msg,
1065 };
1066 }
1067 };
1068
1069 let probe_secs = if idempotent { 5 } else { 10 };
1074 match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1075 Ok(element) => match element.click() {
1076 Ok(_) => StepResult {
1077 name,
1078 status: StepStatus::Passed,
1079 message: format!("clicked {selector}"),
1080 },
1081 Err(e) => StepResult {
1082 name,
1083 status: StepStatus::Failed,
1084 message: format!("click failed on {selector}: {e}"),
1085 },
1086 },
1087 Err(e) if idempotent => StepResult {
1088 name,
1089 status: StepStatus::Skipped,
1090 message: format!("skipped (idempotent): element {selector} not present — {e}"),
1091 },
1092 Err(e) => StepResult {
1093 name,
1094 status: StepStatus::Failed,
1095 message: format!("element {selector} not found: {e}"),
1096 },
1097 }
1098 }
1099
1100 #[allow(clippy::too_many_arguments)]
1101 fn run_type(
1102 &self,
1103 target: &str,
1104 text: &str,
1105 selector_override: Option<&str>,
1106 step_endpoint: Option<&str>,
1107 test_endpoint: Option<&str>,
1108 idempotent: bool,
1109 tab: &Tab,
1110 ) -> StepResult {
1111 let name = format!("[type] {target}");
1112 let selector = match self.resolve_selector(
1113 selector_override,
1114 target,
1115 step_endpoint,
1116 test_endpoint,
1117 tab,
1118 ) {
1119 Ok(s) => s,
1120 Err(msg) => {
1121 if idempotent {
1122 return StepResult {
1123 name,
1124 status: StepStatus::Skipped,
1125 message: format!("skipped (idempotent): no target found — {msg}"),
1126 };
1127 }
1128 return StepResult {
1129 name,
1130 status: StepStatus::Failed,
1131 message: msg,
1132 };
1133 }
1134 };
1135
1136 let probe_secs = if idempotent { 5 } else { 10 };
1137 match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(probe_secs)) {
1138 Ok(element) => {
1139 if let Err(e) = element.click() {
1140 return StepResult {
1141 name,
1142 status: StepStatus::Failed,
1143 message: format!("click to focus {selector} failed: {e}"),
1144 };
1145 }
1146
1147 let js = format!(
1148 "document.querySelector('{}').value = '';",
1149 selector.replace('\'', "\\'")
1150 );
1151 let _ = tab.evaluate(&js, false);
1152
1153 match element.type_into(text) {
1154 Ok(_) => StepResult {
1155 name,
1156 status: StepStatus::Passed,
1157 message: format!("typed {text:?} into {selector}"),
1158 },
1159 Err(e) => StepResult {
1160 name,
1161 status: StepStatus::Failed,
1162 message: format!("type into {selector} failed: {e}"),
1163 },
1164 }
1165 }
1166 Err(e) if idempotent => StepResult {
1167 name,
1168 status: StepStatus::Skipped,
1169 message: format!("skipped (idempotent): element {selector} not present — {e}"),
1170 },
1171 Err(e) => StepResult {
1172 name,
1173 status: StepStatus::Failed,
1174 message: format!("element {selector} not found: {e}"),
1175 },
1176 }
1177 }
1178
1179 #[allow(clippy::too_many_arguments)]
1180 #[allow(clippy::too_many_lines)]
1181 fn run_wait(
1182 &self,
1183 target: &str,
1184 selector_override: Option<&str>,
1185 text: Option<&str>,
1186 timeout_ms: Option<u64>,
1187 step_endpoint: Option<&str>,
1188 test_endpoint: Option<&str>,
1189 idempotent: bool,
1190 tab: &Tab,
1191 ) -> StepResult {
1192 let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
1193 let step_name = format!("[wait] {target}");
1194
1195 let selector = match selector_override {
1197 Some(s) => Some(s.to_owned()),
1198 None if text.is_some() => None,
1199 None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
1200 Ok(s) => Some(s),
1201 Err(msg) => {
1202 if idempotent {
1203 return StepResult {
1204 name: step_name,
1205 status: StepStatus::Skipped,
1206 message: format!("skipped (idempotent): no target found — {msg}"),
1207 };
1208 }
1209 return StepResult {
1210 name: step_name,
1211 status: StepStatus::Failed,
1212 message: msg,
1213 };
1214 }
1215 },
1216 };
1217
1218 if text.is_some() {
1219 let sel_js = selector
1220 .as_deref()
1221 .map(crate::selectors::selector_matches_js);
1222 let text_js = text.map(|t| {
1223 let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
1224 format!("document.body ? document.body.innerText.includes('{escaped}') : false")
1225 });
1226
1227 let deadline = Instant::now() + timeout;
1228 loop {
1229 let sel_ok = sel_js
1230 .as_ref()
1231 .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1232 let text_ok = text_js
1233 .as_ref()
1234 .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
1235 if sel_ok && text_ok {
1236 let mut what = Vec::new();
1237 if let Some(sel) = &selector {
1238 what.push(format!("found {sel}"));
1239 }
1240 if let Some(t) = text {
1241 what.push(format!("text {t:?} visible"));
1242 }
1243 return StepResult {
1244 name: step_name,
1245 status: StepStatus::Passed,
1246 message: what.join(" and "),
1247 };
1248 }
1249 if Instant::now() >= deadline {
1250 let mut what = Vec::new();
1251 if let Some(sel) = &selector {
1252 what.push(sel.clone());
1253 }
1254 if let Some(t) = text {
1255 what.push(format!("text {t:?}"));
1256 }
1257 let message = format!(
1258 "wait for {} timed out after {}ms: the event waited for never came",
1259 what.join(" / "),
1260 timeout.as_millis(),
1261 );
1262 if idempotent {
1263 return StepResult {
1264 name: step_name,
1265 status: StepStatus::Skipped,
1266 message: format!("skipped (idempotent): {message}"),
1267 };
1268 }
1269 return StepResult {
1270 name: step_name,
1271 status: StepStatus::Failed,
1272 message,
1273 };
1274 }
1275 std::thread::sleep(Duration::from_millis(250));
1276 }
1277 }
1278
1279 match selector.as_deref() {
1280 Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
1281 Ok(_) => StepResult {
1282 name: step_name,
1283 status: StepStatus::Passed,
1284 message: format!("found {sel}"),
1285 },
1286 Err(e) if idempotent => StepResult {
1287 name: step_name,
1288 status: StepStatus::Skipped,
1289 message: format!(
1290 "skipped (idempotent): wait for {sel} timed out after {}ms: {e}",
1291 timeout.as_millis()
1292 ),
1293 },
1294 Err(e) => StepResult {
1295 name: step_name,
1296 status: StepStatus::Failed,
1297 message: format!(
1298 "wait for {sel} timed out after {}ms: {e}",
1299 timeout.as_millis()
1300 ),
1301 },
1302 },
1303 None => StepResult {
1304 name: step_name,
1305 status: StepStatus::Failed,
1306 message: "wait step has neither selector nor text".into(),
1307 },
1308 }
1309 }
1310
1311 #[allow(clippy::too_many_arguments)]
1312 fn run_assert(
1313 &self,
1314 definition: Option<&str>,
1315 preset: Option<&str>,
1316 prompt: Option<&str>,
1317 assert_text: Option<&str>,
1318 screenshot: bool,
1319 step_endpoint: Option<&str>,
1320 test_endpoint: Option<&str>,
1321 tab: &Tab,
1322 ) -> StepResult {
1323 std::thread::sleep(Duration::from_millis(500));
1324
1325 let page_content = get_page_text(tab);
1326
1327 let image: Option<Vec<String>> = if screenshot {
1332 let endpoint = self
1333 .endpoints
1334 .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1335 if !endpoint.vision {
1336 return StepResult {
1337 name: "[assert]".into(),
1338 status: StepStatus::Failed,
1339 message: format!(
1340 "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
1341 name = endpoint.name
1342 ),
1343 };
1344 }
1345 match crate::vision::capture_screenshot_data_urls(
1346 tab,
1347 self.config
1348 .screenshot_max_dimension
1349 .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
1350 self.screenshot_height_cap(),
1351 self.current_viewport_height(),
1352 ) {
1353 Ok(urls) => Some(urls),
1354 Err(e) => {
1355 return StepResult {
1356 name: "[assert]".into(),
1357 status: StepStatus::Failed,
1358 message: format!("screenshot capture failed: {e}"),
1359 };
1360 }
1361 }
1362 } else {
1363 None
1364 };
1365
1366 if let Some(def_name) = definition {
1367 if let Some(def) = self.definitions.get(def_name) {
1368 return self.run_assert_def(
1369 def,
1370 &page_content,
1371 image.as_deref(),
1372 step_endpoint,
1373 test_endpoint,
1374 tab,
1375 );
1376 }
1377 return StepResult {
1378 name: format!("[assert] {def_name}"),
1379 status: StepStatus::Failed,
1380 message: format!("definition '{def_name}' not found"),
1381 };
1382 }
1383
1384 if let Some(preset_name) = preset {
1385 if preset_name == "layout_no_issues" {
1388 return self.run_layout_preset(tab);
1389 }
1390 return self.run_preset(
1391 preset_name,
1392 assert_text,
1393 &page_content,
1394 image.as_deref(),
1395 step_endpoint,
1396 test_endpoint,
1397 );
1398 }
1399
1400 if let Some(prompt_text) = prompt {
1401 return self.run_custom(
1402 prompt_text,
1403 &page_content,
1404 image.as_deref(),
1405 step_endpoint,
1406 test_endpoint,
1407 );
1408 }
1409
1410 StepResult {
1411 name: "[assert]".into(),
1412 status: StepStatus::Skipped,
1413 message: "no definition, preset, or prompt specified".into(),
1414 }
1415 }
1416
1417 fn run_assert_def(
1418 &self,
1419 def: &AssertDefinition,
1420 page_content: &PageContent,
1421 image: Option<&[String]>,
1422 step_endpoint: Option<&str>,
1423 test_endpoint: Option<&str>,
1424 tab: &Tab,
1425 ) -> StepResult {
1426 if let Some(ref agent) = def.agent {
1428 if image.is_some() {
1429 return StepResult {
1430 name: format!("[assert] {}", def.name),
1431 status: StepStatus::Failed,
1432 message: "agent-backed assertions do not support screenshots".into(),
1433 };
1434 }
1435 let task = def
1436 .task_template
1437 .as_deref()
1438 .unwrap_or("Evaluate the assertion")
1439 .replace("{url}", &page_content.url)
1440 .replace("{title}", &page_content.title)
1441 .replace("{content}", &page_content.body_text)
1442 .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
1443
1444 return self.run_agent_step(agent, &task, &def.name);
1445 }
1446
1447 if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
1449 return self.run_custom_preset(
1450 &def.name,
1451 system,
1452 template,
1453 def.assert_text.as_deref(),
1454 page_content,
1455 image,
1456 step_endpoint,
1457 test_endpoint,
1458 );
1459 }
1460
1461 def.preset.as_ref().map_or_else(
1462 || {
1463 def.prompt.as_ref().map_or_else(
1464 || StepResult {
1465 name: format!("[assert] {}", def.name),
1466 status: StepStatus::Failed,
1467 message: "definition has no preset, prompt, or system+user_template".into(),
1468 },
1469 |prompt| {
1470 self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
1471 },
1472 )
1473 },
1474 |preset_name| {
1475 if preset_name == "layout_no_issues" {
1476 return self.run_layout_preset(tab);
1477 }
1478 self.run_preset(
1479 preset_name,
1480 def.assert_text.as_deref(),
1481 page_content,
1482 image,
1483 step_endpoint,
1484 test_endpoint,
1485 )
1486 },
1487 )
1488 }
1489
1490 #[allow(clippy::too_many_arguments)]
1491 fn run_custom_preset(
1492 &self,
1493 name: &str,
1494 system: &str,
1495 template: &str,
1496 assert_text: Option<&str>,
1497 page_content: &PageContent,
1498 image: Option<&[String]>,
1499 step_endpoint: Option<&str>,
1500 test_endpoint: Option<&str>,
1501 ) -> StepResult {
1502 let user_prompt = template
1503 .replace("{url}", &page_content.url)
1504 .replace("{title}", &page_content.title)
1505 .replace("{content}", &page_content.body_text)
1506 .replace("{expected_text}", assert_text.unwrap_or(""))
1507 .replace("{description}", "");
1508
1509 let user_prompt = if template.contains("{content}") {
1514 user_prompt
1515 } else {
1516 format!(
1517 "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1518 url = page_content.url,
1519 title = page_content.title,
1520 content = page_content.body_text,
1521 )
1522 };
1523
1524 self.reporter
1525 .debug(format!("assert: {name} (custom preset)"));
1526
1527 let chain = self
1528 .endpoints
1529 .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1530 let sys = system.to_owned();
1531
1532 let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1533
1534 response.map_or_else(
1535 |e| StepResult {
1536 name: format!("[assert] {name}"),
1537 status: StepStatus::Failed,
1538 message: format!("LLM assertion call failed: {e}"),
1539 },
1540 |(lr, _idx)| {
1541 if verdict_is_pass(&lr.content) {
1542 StepResult {
1543 name: format!("[assert] {name}"),
1544 status: StepStatus::Passed,
1545 message: "PASS".into(),
1546 }
1547 } else {
1548 StepResult {
1549 name: format!("[assert] {name}"),
1550 status: StepStatus::Failed,
1551 message: lr.content,
1552 }
1553 }
1554 },
1555 )
1556 }
1557
1558 fn run_preset(
1559 &self,
1560 preset_name: &str,
1561 assert_text: Option<&str>,
1562 page_content: &PageContent,
1563 image: Option<&[String]>,
1564 step_endpoint: Option<&str>,
1565 test_endpoint: Option<&str>,
1566 ) -> StepResult {
1567 let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
1568 return StepResult {
1569 name: format!("[assert] {preset_name}"),
1570 status: StepStatus::Failed,
1571 message: format!("unknown assertion preset: {preset_name}"),
1572 };
1573 };
1574 if preset_name.starts_with("visual_") && image.is_none() {
1575 return StepResult {
1576 name: format!("[assert] {preset_name}"),
1577 status: StepStatus::Failed,
1578 message: format!(
1579 "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1580 ),
1581 };
1582 }
1583
1584 let user_prompt = preset
1585 .user_template
1586 .replace("{url}", &page_content.url)
1587 .replace("{title}", &page_content.title)
1588 .replace("{content}", &page_content.body_text)
1589 .replace("{expected_text}", assert_text.unwrap_or(""))
1590 .replace("{description}", "");
1591
1592 let user_prompt = if preset.user_template.contains("{content}") {
1595 user_prompt
1596 } else {
1597 format!(
1598 "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1599 url = page_content.url,
1600 title = page_content.title,
1601 content = page_content.body_text,
1602 )
1603 };
1604
1605 self.reporter.debug(format!("assert: {preset_name}"));
1606
1607 let chain = self
1608 .endpoints
1609 .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1610 let sys = preset.system.to_owned();
1611
1612 let response = self.llm_call_chain(&chain, &sys, &user_prompt, image, "assertion");
1613
1614 response.map_or_else(
1615 |e| StepResult {
1616 name: format!("[assert] {preset_name}"),
1617 status: StepStatus::Failed,
1618 message: format!("LLM assertion call failed: {e}"),
1619 },
1620 |(lr, _idx)| {
1621 if verdict_is_pass(&lr.content) {
1622 StepResult {
1623 name: format!("[assert] {preset_name}"),
1624 status: StepStatus::Passed,
1625 message: "PASS".into(),
1626 }
1627 } else {
1628 StepResult {
1629 name: format!("[assert] {preset_name}"),
1630 status: StepStatus::Failed,
1631 message: lr.content,
1632 }
1633 }
1634 },
1635 )
1636 }
1637
1638 fn run_layout_preset(&self, tab: &Tab) -> StepResult {
1647 let name = "[assert] layout_no_issues".to_owned();
1648 self.reporter
1649 .debug("assert: layout_no_issues (DOM layout scan)");
1650 let js = LAYOUT_SCAN_JS.replace(
1651 "__IGNORE_CLASSES__",
1652 &serde_json::to_string(&self.config.layout_ignore_classes)
1653 .unwrap_or_else(|_| "[]".to_owned()),
1654 );
1655 let result = tab.evaluate(&js, false);
1656 let json_str = match result {
1657 Ok(r) => r
1658 .value
1659 .as_ref()
1660 .and_then(|v| v.as_str().map(String::from))
1661 .unwrap_or_else(|| "[]".to_owned()),
1662 Err(e) => {
1663 return StepResult {
1664 name,
1665 status: StepStatus::Failed,
1666 message: format!("layout scan JS failed: {e}"),
1667 };
1668 }
1669 };
1670 let issues: Vec<LayoutIssue> = serde_json::from_str(&json_str).unwrap_or_default();
1671 if issues.is_empty() {
1672 return StepResult {
1673 name,
1674 status: StepStatus::Passed,
1675 message: "PASS — no layout defects detected".into(),
1676 };
1677 }
1678 let mut lines: Vec<String> = issues
1679 .iter()
1680 .take(10)
1681 .map(|i| {
1682 format!(
1683 "- [{type_}] {element}: {detail}",
1684 type_ = i.issue_type,
1685 element = i.element,
1686 detail = i.detail
1687 )
1688 })
1689 .collect();
1690 if issues.len() > 10 {
1691 lines.push(format!("- … and {} more", issues.len() - 10));
1692 }
1693 StepResult {
1694 name,
1695 status: StepStatus::Failed,
1696 message: format!(
1697 "FAIL — {} layout defect(s) detected:\n{}",
1698 issues.len(),
1699 lines.join("\n")
1700 ),
1701 }
1702 }
1703
1704 fn run_custom(
1705 &self,
1706 prompt: &str,
1707 page_content: &PageContent,
1708 image: Option<&[String]>,
1709 step_endpoint: Option<&str>,
1710 test_endpoint: Option<&str>,
1711 ) -> StepResult {
1712 let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1713
1714 let mut user = format!(
1715 "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1716 url = page_content.url,
1717 title = page_content.title,
1718 content = page_content.body_text,
1719 );
1720 if image.is_some() {
1721 user.push_str(
1722 "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1723 );
1724 }
1725
1726 self.reporter.debug("custom assert");
1727
1728 let chain = self
1729 .endpoints
1730 .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Assertion);
1731 let sys = system.to_owned();
1732
1733 let response = self.llm_call_chain(&chain, &sys, &user, image, "assertion");
1734
1735 response.map_or_else(
1736 |e| StepResult {
1737 name: "[assert] custom".into(),
1738 status: StepStatus::Failed,
1739 message: format!("LLM assertion call failed: {e}"),
1740 },
1741 |(lr, _idx)| {
1742 if verdict_is_pass(&lr.content) {
1743 StepResult {
1744 name: "[assert] custom".into(),
1745 status: StepStatus::Passed,
1746 message: "PASS".into(),
1747 }
1748 } else {
1749 StepResult {
1750 name: "[assert] custom".into(),
1751 status: StepStatus::Failed,
1752 message: lr.content,
1753 }
1754 }
1755 },
1756 )
1757 }
1758
1759 fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1760 let path = path.unwrap_or("screenshot.png");
1761
1762 match tab.capture_screenshot(
1763 headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1764 None,
1765 None,
1766 true,
1767 ) {
1768 Ok(data) => {
1769 if let Err(e) = std::fs::write(path, &data) {
1770 return StepResult {
1771 name: format!("[screenshot] {path}"),
1772 status: StepStatus::Failed,
1773 message: format!("failed to write screenshot: {e}"),
1774 };
1775 }
1776 StepResult {
1777 name: format!("[screenshot] {path}"),
1778 status: StepStatus::Passed,
1779 message: format!("saved to {path}"),
1780 }
1781 }
1782 Err(e) => StepResult {
1783 name: format!("[screenshot] {path}"),
1784 status: StepStatus::Failed,
1785 message: format!("screenshot failed: {e}"),
1786 },
1787 }
1788 }
1789
1790 #[allow(clippy::literal_string_with_formatting_args)]
1792 fn run_agent(
1793 &self,
1794 agent_name: &str,
1795 task: &str,
1796 definition: Option<&str>,
1797 _test_endpoint: Option<&str>,
1798 ) -> StepResult {
1799 let resolved_task = if let Some(def_name) = definition {
1801 if let Some(def) = self.definitions.get(def_name) {
1802 let tmpl = def.task_template.as_deref().unwrap_or(task);
1803 tmpl.replace("{task}", task)
1804 } else {
1805 return StepResult {
1806 name: format!("[agent] {def_name}"),
1807 status: StepStatus::Failed,
1808 message: format!("definition '{def_name}' not found"),
1809 };
1810 }
1811 } else {
1812 task.to_owned()
1813 };
1814
1815 self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1816 }
1817
1818 fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1819 let Some(ep) = self.endpoints.get(agent_name) else {
1820 return StepResult {
1821 name: format!("[agent] {display_name}"),
1822 status: StepStatus::Failed,
1823 message: format!("agent endpoint '{agent_name}' not found"),
1824 };
1825 };
1826
1827 if ep.url.is_empty() {
1828 return StepResult {
1829 name: format!("[agent] {display_name}"),
1830 status: StepStatus::Failed,
1831 message: format!("agent endpoint '{agent_name}' has no URL"),
1832 };
1833 }
1834
1835 self.reporter.debug(format!("agent {agent_name}: {task}"));
1836
1837 let url = ep.url.clone();
1838 let client = A2aClient::new(&url, self.timeout);
1839 let task_clone = task.to_owned();
1840
1841 let response = std::thread::spawn(move || {
1842 let rt = tokio::runtime::Builder::new_current_thread()
1843 .enable_all()
1844 .build()
1845 .unwrap();
1846 rt.block_on(client.send_task(&task_clone))
1847 })
1848 .join()
1849 .unwrap();
1850
1851 self.usage.record_flat_call(agent_name, ep);
1853
1854 match response {
1855 Ok(text) => {
1856 let clean = text.trim().to_owned();
1857 if verdict_is_pass(&clean) {
1858 StepResult {
1859 name: format!("[agent] {display_name}"),
1860 status: StepStatus::Passed,
1861 message: format!("PASS: {clean}"),
1862 }
1863 } else if verdict_is_fail(&clean) {
1864 StepResult {
1865 name: format!("[agent] {display_name}"),
1866 status: StepStatus::Failed,
1867 message: clean,
1868 }
1869 } else {
1870 StepResult {
1871 name: format!("[agent] {display_name}"),
1872 status: StepStatus::Passed,
1873 message: format!("response: {clean}"),
1874 }
1875 }
1876 }
1877 Err(e) => StepResult {
1878 name: format!("[agent] {display_name}"),
1879 status: StepStatus::Failed,
1880 message: format!("agent call failed: {e}"),
1881 },
1882 }
1883 }
1884
1885 fn run_mcp(
1887 &self,
1888 server_name: &str,
1889 tool_name: &str,
1890 args: Option<&serde_json::Value>,
1891 ) -> StepResult {
1892 let Some(ep) = self.endpoints.get(server_name) else {
1893 return StepResult {
1894 name: format!("[mcp] {server_name}:{tool_name}"),
1895 status: StepStatus::Failed,
1896 message: format!("MCP server endpoint '{server_name}' not found"),
1897 };
1898 };
1899
1900 let cmd = ep.command.as_deref().unwrap_or("");
1901 if cmd.is_empty() {
1902 return StepResult {
1903 name: format!("[mcp] {server_name}:{tool_name}"),
1904 status: StepStatus::Failed,
1905 message: format!("MCP server '{server_name}' has no command configured"),
1906 };
1907 }
1908
1909 self.reporter
1910 .debug(format!("mcp {server_name} {tool_name}"));
1911
1912 let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1913
1914 let command = cmd.to_owned();
1915 let args_vec = ep.args.clone();
1916 let tool = tool_name.to_owned();
1917
1918 let response = std::thread::spawn(move || {
1919 let mut mcp_client =
1920 McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1921 mcp_client
1922 .call_tool(&tool, &args_val)
1923 .map_err(|e| e.to_string())
1924 })
1925 .join()
1926 .unwrap();
1927
1928 self.usage.record_flat_call(server_name, ep);
1930
1931 match response {
1932 Ok(result) => {
1933 if result.isError {
1934 StepResult {
1935 name: format!("[mcp] {server_name}:{tool_name}"),
1936 status: StepStatus::Failed,
1937 message: result.to_string(),
1938 }
1939 } else {
1940 StepResult {
1941 name: format!("[mcp] {server_name}:{tool_name}"),
1942 status: StepStatus::Passed,
1943 message: result.to_string(),
1944 }
1945 }
1946 }
1947 Err(e) => StepResult {
1948 name: format!("[mcp] {server_name}:{tool_name}"),
1949 status: StepStatus::Failed,
1950 message: format!("MCP call failed: {e}"),
1951 },
1952 }
1953 }
1954
1955 fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1960 LlmConfig {
1961 url: if endpoint.provider == crate::scenario::Provider::Bedrock {
1962 endpoint.url.clone()
1965 } else if endpoint.url.is_empty() {
1966 self.llm.url.clone()
1967 } else {
1968 endpoint.url.clone()
1969 },
1970 model: endpoint
1971 .model
1972 .clone()
1973 .unwrap_or_else(|| self.llm.model.clone()),
1974 api_key: endpoint
1975 .api_key
1976 .clone()
1977 .or_else(|| self.llm.api_key.clone()),
1978 headers: if endpoint.headers.is_empty() {
1979 self.llm.headers.clone()
1980 } else {
1981 endpoint.headers.clone()
1982 },
1983 timeout: self.llm.timeout,
1984 temperature: self.llm.temperature,
1985 thinking: self.llm.thinking,
1986 model_params: self.llm.model_params.clone(),
1987 cache: endpoint.cache_markers,
1988 max_attempts: endpoint.max_attempts.max(1),
1989 provider: endpoint.provider,
1990 deployment: endpoint.deployment.clone(),
1991 api_version: endpoint.api_version.clone(),
1992 auth: endpoint.auth.clone(),
1993 header_commands: endpoint.header_commands.clone(),
1994 aws: endpoint.aws.clone(),
1995 }
1996 }
1997
1998 fn run_context(&self) -> String {
2012 let mut parts = vec![
2013 "RUN CONTEXT (use this to interpret the page, never repeat it back)".into(),
2014 "================================================================".into(),
2015 format!("Run started: {} UTC", unix_to_rfc3339(self.run_started)),
2016 ];
2017 if let Some(base) = self.config.base_url.as_deref() {
2018 parts.push(format!("Target site: {base}"));
2019 }
2020 let now = SystemTime::now()
2021 .duration_since(UNIX_EPOCH)
2022 .map_or(0, |d| d.as_secs());
2023 parts.push(format!("Current time: {} UTC", unix_to_rfc3339(now)));
2024 parts.join("\n")
2025 }
2026
2027 #[allow(clippy::cast_possible_truncation, clippy::too_many_lines)]
2028 fn llm_call_chain(
2029 &self,
2030 chain: &[&ResolvedEndpoint],
2031 system: &str,
2032 user: &str,
2033 image: Option<&[String]>,
2034 purpose: &str,
2035 ) -> Result<(crate::costs::LlmResponse, usize), String> {
2036 if chain.is_empty() {
2037 return Err("empty LLM endpoint chain".into());
2038 }
2039 let primary = self.build_llm_for_endpoint(chain[0]);
2040 let fallbacks: Vec<LlmConfig> = chain[1..]
2041 .iter()
2042 .map(|e| self.build_llm_for_endpoint(e))
2043 .collect();
2044
2045 let (test, index) = self
2046 .current_step
2047 .borrow()
2048 .as_ref()
2049 .map_or_else(|| ("-".to_owned(), 0), |(t, i)| (t.clone(), *i));
2050 let primary_endpoint = chain[0].name.clone();
2051 let primary_model = primary.model.clone();
2052 self.emit_event(&TestEvent::LlmCallStarted {
2053 test: test.clone(),
2054 index,
2055 endpoint: primary_endpoint.clone(),
2056 model: primary_model.clone(),
2057 purpose: purpose.to_owned(),
2058 });
2059
2060 let started = Instant::now();
2061 let sys = system.to_owned();
2062 let context = self.run_context();
2063 let user = if context.is_empty() {
2064 user.to_owned()
2065 } else {
2066 format!("{context}\n\n{user}")
2067 };
2068 let image = image.map(<[String]>::to_vec);
2069
2070 let result = std::thread::spawn(move || {
2071 let rt = tokio::runtime::Builder::new_current_thread()
2072 .enable_all()
2073 .build()
2074 .unwrap();
2075 let call = async {
2076 match image.as_deref() {
2077 Some(img) => {
2078 llm_chat_vision_with_usage_chain(
2079 &primary,
2080 &fallbacks,
2081 &sys,
2082 &user,
2083 Some(img),
2084 )
2085 .await
2086 }
2087 None => llm_chat_with_usage_chain(&primary, &fallbacks, &sys, &user).await,
2088 }
2089 };
2090 rt.block_on(call)
2091 })
2092 .join()
2093 .unwrap();
2094
2095 let duration_ms = started.elapsed().as_millis() as u64;
2096 match result {
2097 Ok((lr, idx)) => {
2098 let cost = calculate_llm_cost(chain[idx], &lr.usage);
2099 let answering = chain[idx].name.clone();
2100 let model = chain[idx]
2101 .model
2102 .clone()
2103 .unwrap_or_else(|| primary_model.clone());
2104 self.usage
2105 .record_llm_call(&answering, chain[idx], &model, &lr.usage);
2106 self.emit_event(&TestEvent::LlmCallFinished {
2107 test,
2108 index,
2109 endpoint: answering,
2110 model,
2111 purpose: purpose.to_owned(),
2112 ok: true,
2113 duration_ms,
2114 input_tokens: lr.usage.prompt_tokens,
2115 output_tokens: lr.usage.completion_tokens,
2116 cached_input_tokens: lr.usage.cached_input_tokens,
2117 cache_creation_input_tokens: lr.usage.cache_creation_input_tokens,
2118 cost,
2119 error: None,
2120 });
2121 Ok((lr, idx))
2122 }
2123 Err(e) => {
2124 self.emit_event(&TestEvent::LlmCallFinished {
2125 test,
2126 index,
2127 endpoint: primary_endpoint,
2128 model: primary_model,
2129 purpose: purpose.to_owned(),
2130 ok: false,
2131 duration_ms,
2132 input_tokens: 0,
2133 output_tokens: 0,
2134 cached_input_tokens: 0,
2135 cache_creation_input_tokens: 0,
2136 cost: 0.0,
2137 error: Some(e.clone()),
2138 });
2139 Err(e)
2140 }
2141 }
2142 }
2143
2144 #[allow(clippy::too_many_lines)]
2153 fn resolve_selector(
2154 &self,
2155 css_override: Option<&str>,
2156 target: &str,
2157 step_endpoint: Option<&str>,
2158 test_endpoint: Option<&str>,
2159 tab: &Tab,
2160 ) -> Result<String, String> {
2161 if let Some(explicit) = css_override {
2162 return Ok(explicit.to_owned());
2163 }
2164
2165 let dom_info = extract_dom_info(tab)?;
2166 let page_content = get_page_text(tab);
2167
2168 let system = concat!(
2169 "You are a browser automation selector generator. ",
2170 "Given a web page's content and interactive elements, ",
2171 "return ONLY the best CSS selector for the described element. ",
2172 "Output nothing except the CSS selector. ",
2173 "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
2174 "[name=\"...\"], tag.class, tag. ",
2175 "Never output explanations, markdown, or extra text."
2176 );
2177
2178 let user = format!(
2179 "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
2180 page_content.url,
2181 page_content.title,
2182 truncate(&page_content.body_text, 4000),
2183 dom_info,
2184 target,
2185 );
2186
2187 let retry_user = format!(
2188 "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
2189 "The selector must match at least one element currently present on the page.",
2190 page_content.url,
2191 page_content.title,
2192 truncate(&page_content.body_text, 4000),
2193 dom_info,
2194 target,
2195 );
2196
2197 self.reporter.debug(format!("LLM targeting: {target}"));
2198
2199 let chain = self
2200 .endpoints
2201 .resolve_chain(step_endpoint.or(test_endpoint), TaskType::Targeting);
2202 let sys = system.to_owned();
2203
2204 let call_llm = |prompt: &str| self.llm_call_chain(&chain, &sys, prompt, None, "targeting");
2205
2206 let first = call_llm(&user);
2207 let (lr, _idx) = match first {
2208 Ok(lr) => lr,
2209 Err(e) => {
2210 return Err(format!("LLM element targeting failed: {e}"));
2211 }
2212 };
2213 let clean = sanitize_selector(&lr.content);
2214 self.reporter.debug(format!("resolved selector: {clean}"));
2215
2216 if selector_is_useless(&clean) {
2217 return Err(format!(
2218 "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
2219 raw = lr.content.trim(),
2220 ));
2221 }
2222 if let Err(reason) = validate_selector(&clean) {
2223 return Err(format!(
2224 "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
2225 raw = lr.content.trim(),
2226 ));
2227 }
2228 if !selector_matches(tab, &clean).unwrap_or(false) {
2229 self.reporter.warn(format!(
2232 "selector {clean} matches nothing — retrying LLM targeting with feedback"
2233 ));
2234 let second = call_llm(&retry_user);
2235 let (lr2, _idx2) = match second {
2236 Ok(lr2) => lr2,
2237 Err(e) => {
2238 return Err(format!(
2239 "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
2240 ));
2241 }
2242 };
2243 let clean2 = sanitize_selector(&lr2.content);
2244 self.reporter
2245 .debug(format!("resolved selector (retry): {clean2}"));
2246 if selector_is_useless(&clean2) {
2247 return Err(format!(
2248 "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
2249 raw = lr2.content.trim(),
2250 excerpt = truncate(&page_content.body_text, 300),
2251 ));
2252 }
2253 if !selector_matches(tab, &clean2).unwrap_or(false) {
2254 return Err(format!(
2255 "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
2256 ));
2257 }
2258 return Ok(clean2);
2259 }
2260
2261 Ok(clean)
2262 }
2263}
2264
2265fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
2267 tab.evaluate(js, false)
2268 .map_err(|e| format!("evaluate failed: {e}"))?
2269 .value
2270 .and_then(|v| v.as_bool())
2271 .ok_or_else(|| "evaluate returned non-boolean".to_owned())
2272}
2273
2274fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
2276 eval_bool(tab, &crate::selectors::selector_matches_js(selector))
2277}
2278
2279fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
2282 let name = format!("[navigate] {full_url}");
2283 match tab.navigate_to(full_url) {
2284 Ok(_) => {
2285 let _ = tab.wait_until_navigated();
2286 StepResult {
2287 name,
2288 status: StepStatus::Passed,
2289 message: format!("navigated to {full_url}"),
2290 }
2291 }
2292 Err(e) => StepResult {
2293 name,
2294 status: StepStatus::Failed,
2295 message: format!("navigation failed: {e}"),
2296 },
2297 }
2298}
2299
2300fn extract_dom_info(tab: &Tab) -> Result<String, String> {
2301 let result = tab
2302 .evaluate(DOM_EXTRACT_JS, false)
2303 .map_err(|e| format!("DOM extraction failed: {e}"))?;
2304
2305 let json_str = result
2306 .value
2307 .as_ref()
2308 .and_then(|v| v.as_str())
2309 .unwrap_or("[]");
2310
2311 let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
2312
2313 if elements.is_empty() {
2314 return Ok("(no interactive elements found)".to_owned());
2315 }
2316
2317 Ok(elements.join("\n"))
2318}
2319
2320fn get_page_text(tab: &Tab) -> PageContent {
2321 let url = tab.get_url();
2322
2323 let title = tab
2324 .evaluate("document.title", false)
2325 .ok()
2326 .and_then(|r| r.value)
2327 .and_then(|v| v.as_str().map(String::from))
2328 .unwrap_or_else(|| "unknown".to_owned());
2329
2330 let body_text = tab
2331 .evaluate(
2332 "document.body ? document.body.innerText : document.documentElement.innerText",
2333 false,
2334 )
2335 .ok()
2336 .and_then(|r| r.value)
2337 .and_then(|v| v.as_str().map(String::from))
2338 .unwrap_or_default();
2339
2340 PageContent {
2341 url,
2342 title,
2343 body_text: truncate(&body_text, 8000),
2344 }
2345}
2346
2347fn resolve_url(url: &str, base_url: &str) -> String {
2348 if url.starts_with("http://") || url.starts_with("https://") {
2349 return url.to_owned();
2350 }
2351 let base = base_url.trim_end_matches('/');
2352 if url.starts_with('/') {
2353 format!("{base}{url}")
2354 } else {
2355 format!("{base}/{url}")
2356 }
2357}
2358
2359fn step_label(step: &TestStep) -> String {
2362 match step {
2363 TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
2364 TestStep::Click { target, .. } => format!("[click] {target}"),
2365 TestStep::Type { target, .. } => format!("[type] {target}"),
2366 TestStep::Wait { target, .. } => format!("[wait] {target}"),
2367 TestStep::Assert {
2368 definition,
2369 preset,
2370 prompt,
2371 ..
2372 } => definition.as_ref().map_or_else(
2373 || {
2374 preset.as_ref().map_or_else(
2375 || {
2376 prompt.as_ref().map_or_else(
2377 || "[assert]".to_owned(),
2378 |pr| format!("[assert] custom ({})", truncate(pr, 60)),
2379 )
2380 },
2381 |p| format!("[assert] {p}"),
2382 )
2383 },
2384 |d| format!("[assert] {d}"),
2385 ),
2386 TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
2387 TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
2388 TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
2389 }
2390}
2391
2392#[must_use]
2394const fn step_kind_label(step: &TestStep) -> &'static str {
2395 match step {
2396 TestStep::Navigate { .. } => "navigate",
2397 TestStep::Click { .. } => "click",
2398 TestStep::Type { .. } => "type",
2399 TestStep::Wait { .. } => "wait",
2400 TestStep::Assert { .. } => "assert",
2401 TestStep::Screenshot { .. } => "screenshot",
2402 TestStep::Agent { .. } => "agent",
2403 TestStep::Mcp { .. } => "mcp",
2404 }
2405}
2406
2407#[derive(Default)]
2410struct TestRunResult {
2411 passed: u32,
2412 failed: u32,
2413 skipped: u32,
2414 total: u32,
2415 details: Vec<StepResult>,
2416}
2417
2418struct PageContent {
2419 url: String,
2420 title: String,
2421 body_text: String,
2422}
2423
2424#[cfg(test)]
2425mod tests {
2426 use super::unix_to_rfc3339;
2427
2428 #[test]
2429 fn rfc3339_epoch_and_reference_dates() {
2430 assert_eq!(unix_to_rfc3339(0), "1970-01-01T00:00:00Z");
2431 assert_eq!(unix_to_rfc3339(1_736_840_000), "2025-01-14T07:33:20Z");
2432 assert_eq!(unix_to_rfc3339(1_784_469_000), "2026-07-19T13:50:00Z");
2433 assert_eq!(unix_to_rfc3339(9_999_999_999), "2286-11-20T17:46:39Z");
2434 }
2435
2436 #[test]
2437 fn rfc3339_handles_leap_years() {
2438 assert_eq!(unix_to_rfc3339(1_582_905_600), "2020-02-28T16:00:00Z");
2439 assert_eq!(unix_to_rfc3339(1_707_408_000), "2024-02-08T16:00:00Z");
2440 }
2441}
2442
2443#[cfg(test)]
2444mod verdict_parse_tests {
2445 use super::{verdict_is_fail, verdict_is_pass};
2446
2447 #[test]
2448 fn tolerates_markdown_punctuation_and_natural_language() {
2449 assert!(verdict_is_pass("PASS"));
2450 assert!(verdict_is_pass("pass"));
2451 assert!(verdict_is_pass("**PASS**\n\nThe page is clean."));
2452 assert!(verdict_is_pass("passes - no explicit error visible"));
2453 assert!(verdict_is_pass(" \"pass\""));
2454 assert!(!verdict_is_pass("FAIL: something broke"));
2455 assert!(!verdict_is_pass("**FAIL** broken"));
2456
2457 assert!(verdict_is_fail("**FAIL** broken"));
2458 assert!(verdict_is_fail("fails - error toast shown"));
2459 assert!(!verdict_is_fail("passes - ok"));
2460 }
2461}