Skip to main content

llm_browser_testkit/
runner.rs

1use std::collections::HashMap;
2use std::path::PathBuf;
3use std::sync::Arc;
4use std::time::{Duration, Instant};
5
6use anyhow::Context;
7use headless_chrome::{Browser, LaunchOptions, Tab};
8
9use crate::a2a::A2aClient;
10use crate::budgets::{BudgetStatus, BudgetTracker};
11use crate::costs::UsageTracker;
12use crate::diagnostics;
13use crate::endpoints::{EndpointRegistry, TaskType};
14use crate::llm_chat_vision_with_usage;
15use crate::llm_chat_with_usage;
16use crate::mcp_client::McpClient;
17use crate::scenario::{AssertDefinition, ScenarioConfig, TestGroup, TestStep};
18use crate::selectors::{sanitize_selector, selector_is_useless, validate_selector};
19use crate::truncate;
20use crate::LlmConfig;
21use crate::DOM_EXTRACT_JS;
22
23/// How long the CDP connection stays open after the browser goes quiet.
24///
25/// `headless_chrome` ships a 30s default and tears down the entire connection
26/// when no traffic arrives for that long; a run must own its connection for
27/// its full duration instead.
28const BROWSER_IDLE_TIMEOUT: Duration = Duration::from_secs(6 * 60 * 60);
29
30/// Executes a [`Scenario`] against a real browser with optional LLM
31/// assistance for element targeting and assertions.
32pub struct ScenarioRunner {
33    config: ScenarioConfig,
34    definitions: HashMap<String, AssertDefinition>,
35    llm: LlmConfig,
36    timeout: Duration,
37    viewport_width: u32,
38    viewport_height: u32,
39    endpoints: EndpointRegistry,
40    usage: Arc<UsageTracker>,
41    budgets: BudgetTracker,
42    /// Directory for failure artifacts (screenshots).
43    artifacts_dir: PathBuf,
44}
45
46/// Aggregated results from a scenario run.
47#[derive(Debug, Default)]
48pub struct RunReport {
49    /// Number of tests that passed.
50    pub tests_passed: u32,
51    /// Number of tests that failed.
52    pub tests_failed: u32,
53    /// Number of steps that passed.
54    pub passed: u32,
55    /// Number of steps that failed.
56    pub failed: u32,
57    /// Number of steps that were skipped.
58    pub skipped: u32,
59    /// Per-step details.
60    pub details: Vec<StepResult>,
61}
62
63/// Result of a single step execution.
64#[derive(Debug)]
65pub struct StepResult {
66    /// The step name.
67    pub name: String,
68    /// Whether the step passed, failed, or was skipped.
69    pub status: StepStatus,
70    /// Human-readable result message.
71    pub message: String,
72}
73
74/// Outcome for a single step.
75#[derive(Debug, PartialEq, Eq)]
76pub enum StepStatus {
77    /// Step executed successfully and all assertions passed.
78    Passed,
79    /// Step execution or assertion failed.
80    Failed,
81    /// Step was skipped.
82    Skipped,
83}
84
85/// Predefined assertion preset definition.
86struct AssertPreset {
87    name: &'static str,
88    system: &'static str,
89    user_template: &'static str,
90}
91
92/// Built-in assertion presets.
93#[allow(clippy::literal_string_with_formatting_args)]
94const ASSERTION_PRESETS: &[AssertPreset] = &[
95    AssertPreset {
96        name: "no_error_on_page",
97        system: "You are a QA tester. Evaluate if a web page contains error messages, stack traces, exception text, HTTP error codes, 'undefined' errors, or any indication of a malfunction. Be strict — even minor rendering glitches count as errors.",
98        user_template: "Check if the following page content contains ANY errors or malfunctions:\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if there are NO errors, or \"FAIL: <reason>\" if there are errors. Only respond with PASS or FAIL.",
99    },
100    AssertPreset {
101        name: "text_visible",
102        system: "You are a QA tester. Your task is to check if specific text is visible in the page content.",
103        user_template: "Check if the following text appears in the page content:\n\nTEXT TO FIND: \"{expected_text}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the text is present (even partial match is OK), or \"FAIL: text not found\" if it is not.",
104    },
105    AssertPreset {
106        name: "element_exists",
107        system: "You are a QA tester. Check if a described UI element exists on a web page.",
108        user_template: "Check if the following element exists on the page:\n\nELEMENT: \"{description}\"\n\nURL: {url}\n\nPage Content:\n{content}\n\nRespond with exactly \"PASS\" if the element exists, or \"FAIL: <reason>\" if it does not.",
109    },
110    AssertPreset {
111        name: "visual_no_issues",
112        system: "You are a visual QA engineer inspecting a website screenshot. Detect clearly visible layout and rendering defects: overlapping elements that hide content or controls, clipped or truncated text, content cut off at the viewport edges, misaligned or broken UI, blank/empty panels where content is expected, broken or missing images, duplicated elements, or rendering glitches. Ignore subjective aesthetics, intentional stacking (dropdowns, tooltips, layered design), and content that is simply not loaded (empty states). Only report defects that a user would actually see or be blocked by.",
113        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre there any clearly visible layout or rendering defects (overlaps, clipping, cut-off content, broken images, blank panels)? Respond with exactly \"PASS\" if the page renders cleanly, or \"FAIL: <describe each defect and where it appears>\" otherwise.",
114    },
115    AssertPreset {
116        name: "visual_no_overlaps",
117        system: "You are a visual QA engineer inspecting a website screenshot for OVERLAPPING elements that harm usability: one element covering another element's text, buttons, links, or input fields (cookie banners, modals, popovers, chat widgets, sticky headers, or mispositioned layers that hide content or intercept clicks). Ignore intentional, non-harmful stacking (dropdowns, tooltips, badges over avatars, layered design where nothing is hidden or unclickable). Only fail on overlaps that visibly hide content or would block a click.",
118        user_template: "Inspect the attached screenshot and the page text below.\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nAre any elements overlapping in a way that hides page content, text, or interactive controls, or that would block clicks? Respond with exactly \"PASS\" if there are no such overlaps, or \"FAIL: <describe the overlapping elements and what they hide>\" otherwise.",
119    },
120    AssertPreset {
121        name: "visual_text_visible",
122        system: "You are a visual QA engineer inspecting a website screenshot. Determine whether a specific text is FULLY visible and readable: present in the viewport, not clipped, not cut off, not covered by another element, and not obscured by overlays or low contrast stacking. A partial word or a covered text counts as FAIL.",
123        user_template: "Inspect the attached screenshot and the page text below.\n\nTEXT TO CHECK: \"{expected_text}\"\n\nURL: {url}\nTitle: {title}\n\nPage Content:\n{content}\n\nIs the text fully visible and readable in the screenshot (not clipped, covered, or hidden)? Respond with exactly \"PASS\" if it is fully visible, or \"FAIL: <explain what hides or clips it>\" otherwise.",
124    },
125];
126
127impl ScenarioRunner {
128    /// Creates a new runner with the given scenario configuration and
129    /// assertion definitions.
130    #[must_use]
131    #[allow(clippy::needless_pass_by_value)]
132    pub fn new(scenario_config: ScenarioConfig, definitions: Vec<AssertDefinition>) -> Self {
133        let llm = LlmConfig {
134            url: scenario_config
135                .llm_url
136                .clone()
137                .unwrap_or_else(crate::llm_base_url),
138            model: scenario_config
139                .llm_model
140                .clone()
141                .unwrap_or_else(crate::llm_model),
142            api_key: scenario_config
143                .llm_api_key
144                .clone()
145                .or_else(|| std::env::var("HARNESS_LLM_API_KEY").ok()),
146            headers: if scenario_config.llm_headers.is_empty() {
147                crate::parse_headers_env()
148            } else {
149                scenario_config.llm_headers.clone()
150            },
151            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
152            temperature: scenario_config.temperature,
153            thinking: scenario_config.thinking,
154            model_params: scenario_config.model_params.clone(),
155        };
156        let endpoints = EndpointRegistry::from_config(&scenario_config.endpoints, Some(&llm));
157        let budgets = BudgetTracker::from_config(&scenario_config.budgets);
158        let defs_map: HashMap<String, AssertDefinition> = definitions
159            .into_iter()
160            .map(|d| (d.name.clone(), d))
161            .collect();
162
163        Self {
164            timeout: Duration::from_secs(scenario_config.timeout_secs.unwrap_or(60)),
165            viewport_width: scenario_config.viewport_width.unwrap_or(1280),
166            viewport_height: scenario_config.viewport_height.unwrap_or(720),
167            config: scenario_config.clone(),
168            definitions: defs_map,
169            llm,
170            endpoints,
171            usage: Arc::new(UsageTracker::new()),
172            budgets,
173            artifacts_dir: PathBuf::from(
174                scenario_config
175                    .artifacts_dir
176                    .unwrap_or_else(|| "artifacts".to_owned()),
177            ),
178        }
179    }
180
181    /// Returns a clone of the [`UsageTracker`] for reporting.
182    #[must_use]
183    pub fn usage_tracker(&self) -> Arc<UsageTracker> {
184        Arc::clone(&self.usage)
185    }
186
187    /// Returns a reference to the [`BudgetTracker`].
188    #[must_use]
189    pub const fn budget_tracker(&self) -> &BudgetTracker {
190        &self.budgets
191    }
192
193    /// Executes all test groups in the scenario and returns a report.
194    ///
195    /// # Errors
196    ///
197    /// Returns an error if the browser fails to launch.
198    #[allow(clippy::too_many_lines)]
199    pub fn run(&self, tests: &[TestGroup]) -> anyhow::Result<RunReport> {
200        let mut report = RunReport::default();
201
202        if tests.is_empty() {
203            eprintln!("No tests defined in scenario.");
204            return Ok(report);
205        }
206
207        let browser_headless = self.config.browser_headless.unwrap_or(true);
208
209        let launch_opts = LaunchOptions {
210            headless: browser_headless,
211            window_size: Some((self.viewport_width, self.viewport_height)),
212            sandbox: false,
213            // headless_chrome defaults this to 30s and shuts down the whole CDP
214            // connection when no messages arrive for that long. A scenario can
215            // easily exceed 30s of browser silence (slow LLM targeting/assertion
216            // calls, page waits, budget checks between steps), after which every
217            // remaining step fails with "Unable to make method calls because
218            // underlying connection is closed" — one quiet gap kills the run.
219            // Open-ended scenarios must own the connection for their full
220            // duration, so keep it alive for 6 hours.
221            idle_browser_timeout: BROWSER_IDLE_TIMEOUT,
222            ..LaunchOptions::default()
223        };
224
225        let browser = Browser::new(launch_opts).context("failed to launch browser")?;
226        let tab = browser.new_tab().context("failed to open browser tab")?;
227        let _ = tab.set_default_timeout(self.timeout);
228
229        // Start MCP server if configured
230        #[cfg(feature = "mcp-server")]
231        if let Some(ref mcp_cfg) = self.config.mcp_server {
232            if mcp_cfg.enabled {
233                let port = mcp_cfg.port;
234                std::thread::spawn(move || {
235                    let _ = crate::mcp_server::start_mcp_server(port);
236                });
237            }
238        }
239        #[cfg(not(feature = "mcp-server"))]
240        if let Some(mcp_cfg) = &self.config.mcp_server {
241            if mcp_cfg.enabled {
242                eprintln!("  ⚠️  MCP server configured but 'mcp-server' feature not enabled");
243            }
244        }
245
246        // Start A2A agent server if configured
247        #[cfg(feature = "a2a-server")]
248        if let Some(ref a2a_cfg) = self.config.a2a_server {
249            if a2a_cfg.enabled {
250                let port = a2a_cfg.port;
251                tokio::spawn(crate::a2a_server::start_a2a_server(port));
252            }
253        }
254        #[cfg(not(feature = "a2a-server"))]
255        if let Some(a2a_cfg) = &self.config.a2a_server {
256            if a2a_cfg.enabled {
257                eprintln!("  ⚠️  A2A server configured but 'a2a-server' feature not enabled");
258            }
259        }
260
261        for test in tests {
262            eprintln!("\n╔══════════════════════════════");
263            eprintln!("║  Test: {}", test.name);
264            eprintln!("╚══════════════════════════════");
265
266            self.usage.reset_per_test();
267
268            let test_result = self.run_test(test, &tab);
269            self.usage.commit_test(&test.name);
270
271            if test_result.failed == 0 && test_result.total > 0 {
272                report.tests_passed += 1;
273                eprintln!("  Test ✅ Passed");
274            } else if test_result.total > 0 {
275                report.tests_failed += 1;
276                eprintln!("  Test ❌ Failed");
277            }
278
279            report.passed += test_result.passed;
280            report.failed += test_result.failed;
281            report.skipped += test_result.skipped;
282            report.details.extend(test_result.details);
283        }
284
285        Ok(report)
286    }
287
288    #[allow(clippy::too_many_lines)]
289    fn run_test(&self, test: &TestGroup, tab: &Tab) -> TestRunResult {
290        let base_url = test
291            .base_url
292            .clone()
293            .or_else(|| self.config.base_url.clone())
294            .unwrap_or_else(crate::base_url);
295
296        let auto_navigate = test.auto_navigate.unwrap_or(self.config.auto_navigate);
297
298        let start_url = test
299            .start_url
300            .clone()
301            .or_else(|| self.config.start_url.clone())
302            .unwrap_or_else(|| "/dashboard".to_owned());
303
304        if auto_navigate {
305            let full_url = resolve_url(&start_url, &base_url);
306            eprintln!("  → auto-navigate: {full_url}");
307            let _ = tab.navigate_to(&full_url);
308            let _ = tab.wait_until_navigated();
309            std::thread::sleep(Duration::from_secs(4));
310        }
311
312        let mut result = TestRunResult::default();
313
314        for (step_index, step) in test.steps.iter().enumerate() {
315            result.total += 1;
316
317            let wait_ms = match step {
318                TestStep::Navigate { wait_after_ms, .. }
319                | TestStep::Click { wait_after_ms, .. }
320                | TestStep::Type { wait_after_ms, .. } => *wait_after_ms,
321                _ => None,
322            };
323
324            let mut step_result = match step {
325                TestStep::Navigate { url, .. } => {
326                    let full_url = resolve_url(url, &base_url);
327                    run_navigate_step(&full_url, tab)
328                }
329                TestStep::Click {
330                    target,
331                    selector,
332                    endpoint,
333                    ..
334                } => self.run_click(
335                    target,
336                    selector.as_deref(),
337                    endpoint.as_deref(),
338                    test.endpoint.as_deref(),
339                    tab,
340                ),
341                TestStep::Type {
342                    target,
343                    text,
344                    selector,
345                    endpoint,
346                    ..
347                } => self.run_type(
348                    target,
349                    text,
350                    selector.as_deref(),
351                    endpoint.as_deref(),
352                    test.endpoint.as_deref(),
353                    tab,
354                ),
355                TestStep::Wait {
356                    target,
357                    selector,
358                    text,
359                    timeout_ms,
360                    endpoint,
361                } => self.run_wait(
362                    target,
363                    selector.as_deref(),
364                    text.as_deref(),
365                    *timeout_ms,
366                    endpoint.as_deref(),
367                    test.endpoint.as_deref(),
368                    tab,
369                ),
370                TestStep::Assert {
371                    definition,
372                    preset,
373                    prompt,
374                    assert_text,
375                    endpoint,
376                    screenshot,
377                } => self.run_assert(
378                    definition.as_deref(),
379                    preset.as_deref(),
380                    prompt.as_deref(),
381                    assert_text.as_deref(),
382                    *screenshot,
383                    endpoint.as_deref(),
384                    test.endpoint.as_deref(),
385                    tab,
386                ),
387                TestStep::Screenshot { path } => Self::run_screenshot(path.as_deref(), tab),
388                TestStep::Agent {
389                    agent,
390                    task,
391                    definition,
392                } => self.run_agent(agent, task, definition.as_deref(), test.endpoint.as_deref()),
393                TestStep::Mcp { server, tool, args } => self.run_mcp(server, tool, args.as_ref()),
394            };
395
396            // Failure diagnostics: capture the page state and a screenshot so
397            // CI logs say WHAT the page looked like when the step failed,
398            // instead of a bare "timed out: The event waited for never came".
399            if step_result.status == StepStatus::Failed {
400                let state = diagnostics::capture(tab);
401                let screenshot = diagnostics::save_screenshot(
402                    tab,
403                    &self.artifacts_dir,
404                    &test.name,
405                    &test.name,
406                    step_index,
407                    step_kind_label(step),
408                );
409                step_result.message = format!(
410                    "{base} — {excerpt}",
411                    base = step_result.message,
412                    excerpt = diagnostics::inline_excerpt(&state),
413                );
414                eprintln!(
415                    "{}",
416                    diagnostics::full_context(&state, screenshot.as_deref())
417                );
418            }
419
420            eprintln!(
421                "    {} {} — {}",
422                if step_result.status == StepStatus::Passed {
423                    "✅"
424                } else if step_result.status == StepStatus::Failed {
425                    "❌"
426                } else {
427                    "⏭️"
428                },
429                step_result.name,
430                step_result.message,
431            );
432
433            match step_result.status {
434                StepStatus::Passed => result.passed += 1,
435                StepStatus::Failed => result.failed += 1,
436                StepStatus::Skipped => result.skipped += 1,
437            }
438
439            // Fail fast: the first failed step ends the test and the
440            // remaining steps are reported as skipped (no LLM budget is
441            // burned asserting against a page that is already known broken).
442            if step_result.status == StepStatus::Failed
443                && !self.config.continue_on_failure
444                && step_index + 1 < test.steps.len()
445            {
446                eprintln!(
447                    "      ⏭️  failing fast: {} remaining step(s) skipped (set continue_on_failure = true in [config] to disable)",
448                    test.steps.len() - step_index - 1
449                );
450                for skipped in &test.steps[step_index + 1..] {
451                    result.total += 1;
452                    result.skipped += 1;
453                    eprintln!(
454                        "    ⏭️  {} — skipped: previous step failed",
455                        step_label(skipped)
456                    );
457                    result.details.push(StepResult {
458                        name: step_label(skipped),
459                        status: StepStatus::Skipped,
460                        message: "skipped: previous step failed".into(),
461                    });
462                }
463                result.details.push(step_result);
464                return result;
465            }
466
467            // Check per-test budget after each step
468            let test_usage = self.usage.current_test_snapshot();
469            let global_usage = self.usage.global_snapshot();
470            let budget_status = self.budgets.check_all(
471                &test.name,
472                &test_usage,
473                &global_usage,
474                test.budget.as_ref(),
475            );
476            match budget_status {
477                BudgetStatus::HardExceeded { message, .. } => {
478                    crate::reporting::print_budget_error(&message);
479                    result.details.push(StepResult {
480                        name: "[budget]".into(),
481                        status: StepStatus::Failed,
482                        message,
483                    });
484                    result.failed += 1;
485                    return result;
486                }
487                BudgetStatus::SoftExceeded { message, .. } => {
488                    crate::reporting::print_budget_warning(&message);
489                }
490                BudgetStatus::Ok => {}
491            }
492
493            if let Some(ms) = wait_ms {
494                std::thread::sleep(Duration::from_millis(ms));
495            }
496
497            result.details.push(step_result);
498        }
499
500        result
501    }
502
503    // ── step handlers ───────────────────────────────────────────────────
504
505    fn run_click(
506        &self,
507        target: &str,
508        selector_override: Option<&str>,
509        step_endpoint: Option<&str>,
510        test_endpoint: Option<&str>,
511        tab: &Tab,
512    ) -> StepResult {
513        let selector = match self.resolve_selector(
514            selector_override,
515            target,
516            step_endpoint,
517            test_endpoint,
518            tab,
519        ) {
520            Ok(s) => s,
521            Err(msg) => {
522                return StepResult {
523                    name: format!("[click] {target}"),
524                    status: StepStatus::Failed,
525                    message: msg,
526                };
527            }
528        };
529
530        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(10)) {
531            Ok(element) => match element.click() {
532                Ok(_) => StepResult {
533                    name: format!("[click] {target}"),
534                    status: StepStatus::Passed,
535                    message: format!("clicked {selector}"),
536                },
537                Err(e) => StepResult {
538                    name: format!("[click] {target}"),
539                    status: StepStatus::Failed,
540                    message: format!("click failed on {selector}: {e}"),
541                },
542            },
543            Err(e) => StepResult {
544                name: format!("[click] {target}"),
545                status: StepStatus::Failed,
546                message: format!("element {selector} not found: {e}"),
547            },
548        }
549    }
550
551    #[allow(clippy::too_many_arguments)]
552    fn run_type(
553        &self,
554        target: &str,
555        text: &str,
556        selector_override: Option<&str>,
557        step_endpoint: Option<&str>,
558        test_endpoint: Option<&str>,
559        tab: &Tab,
560    ) -> StepResult {
561        let selector = match self.resolve_selector(
562            selector_override,
563            target,
564            step_endpoint,
565            test_endpoint,
566            tab,
567        ) {
568            Ok(s) => s,
569            Err(msg) => {
570                return StepResult {
571                    name: format!("[type] {target}"),
572                    status: StepStatus::Failed,
573                    message: msg,
574                };
575            }
576        };
577
578        match tab.wait_for_element_with_custom_timeout(&selector, Duration::from_secs(10)) {
579            Ok(element) => {
580                if let Err(e) = element.click() {
581                    return StepResult {
582                        name: format!("[type] {target}"),
583                        status: StepStatus::Failed,
584                        message: format!("click to focus {selector} failed: {e}"),
585                    };
586                }
587
588                let js = format!(
589                    "document.querySelector('{}').value = '';",
590                    selector.replace('\'', "\\'")
591                );
592                let _ = tab.evaluate(&js, false);
593
594                match element.type_into(text) {
595                    Ok(_) => StepResult {
596                        name: format!("[type] {target}"),
597                        status: StepStatus::Passed,
598                        message: format!("typed {text:?} into {selector}"),
599                    },
600                    Err(e) => StepResult {
601                        name: format!("[type] {target}"),
602                        status: StepStatus::Failed,
603                        message: format!("type into {selector} failed: {e}"),
604                    },
605                }
606            }
607            Err(e) => StepResult {
608                name: format!("[type] {target}"),
609                status: StepStatus::Failed,
610                message: format!("element {selector} not found: {e}"),
611            },
612        }
613    }
614
615    #[allow(clippy::too_many_arguments)]
616    fn run_wait(
617        &self,
618        target: &str,
619        selector_override: Option<&str>,
620        text: Option<&str>,
621        timeout_ms: Option<u64>,
622        step_endpoint: Option<&str>,
623        test_endpoint: Option<&str>,
624        tab: &Tab,
625    ) -> StepResult {
626        let timeout = Duration::from_millis(timeout_ms.unwrap_or(10_000));
627        let step_name = format!("[wait] {target}");
628
629        // Resolve an explicit selector only (text-only waits are LLM-free).
630        let selector = match selector_override {
631            Some(s) => Some(s.to_owned()),
632            None if text.is_some() => None,
633            None => match self.resolve_selector(None, target, step_endpoint, test_endpoint, tab) {
634                Ok(s) => Some(s),
635                Err(msg) => {
636                    return StepResult {
637                        name: step_name,
638                        status: StepStatus::Failed,
639                        message: msg,
640                    };
641                }
642            },
643        };
644
645        if text.is_some() {
646            let sel_js = selector
647                .as_deref()
648                .map(crate::selectors::selector_matches_js);
649            let text_js = text.map(|t| {
650                let escaped = t.replace('\\', "\\\\").replace('\'', "\\'");
651                format!("document.body ? document.body.innerText.includes('{escaped}') : false")
652            });
653
654            let deadline = Instant::now() + timeout;
655            loop {
656                let sel_ok = sel_js
657                    .as_ref()
658                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
659                let text_ok = text_js
660                    .as_ref()
661                    .is_none_or(|js| eval_bool(tab, js).unwrap_or(false));
662                if sel_ok && text_ok {
663                    let mut what = Vec::new();
664                    if let Some(sel) = &selector {
665                        what.push(format!("found {sel}"));
666                    }
667                    if let Some(t) = text {
668                        what.push(format!("text {t:?} visible"));
669                    }
670                    return StepResult {
671                        name: step_name,
672                        status: StepStatus::Passed,
673                        message: what.join(" and "),
674                    };
675                }
676                if Instant::now() >= deadline {
677                    let mut what = Vec::new();
678                    if let Some(sel) = &selector {
679                        what.push(sel.clone());
680                    }
681                    if let Some(t) = text {
682                        what.push(format!("text {t:?}"));
683                    }
684                    return StepResult {
685                        name: step_name,
686                        status: StepStatus::Failed,
687                        message: format!(
688                            "wait for {} timed out after {}ms: the event waited for never came",
689                            what.join(" / "),
690                            timeout.as_millis(),
691                        ),
692                    };
693                }
694                std::thread::sleep(Duration::from_millis(250));
695            }
696        }
697
698        match selector.as_deref() {
699            Some(sel) => match tab.wait_for_element_with_custom_timeout(sel, timeout) {
700                Ok(_) => StepResult {
701                    name: step_name,
702                    status: StepStatus::Passed,
703                    message: format!("found {sel}"),
704                },
705                Err(e) => StepResult {
706                    name: step_name,
707                    status: StepStatus::Failed,
708                    message: format!(
709                        "wait for {sel} timed out after {}ms: {e}",
710                        timeout.as_millis()
711                    ),
712                },
713            },
714            None => StepResult {
715                name: step_name,
716                status: StepStatus::Failed,
717                message: "wait step has neither selector nor text".into(),
718            },
719        }
720    }
721
722    #[allow(clippy::too_many_arguments)]
723    fn run_assert(
724        &self,
725        definition: Option<&str>,
726        preset: Option<&str>,
727        prompt: Option<&str>,
728        assert_text: Option<&str>,
729        screenshot: bool,
730        step_endpoint: Option<&str>,
731        test_endpoint: Option<&str>,
732        tab: &Tab,
733    ) -> StepResult {
734        std::thread::sleep(Duration::from_millis(500));
735
736        let page_content = get_page_text(tab);
737
738        // Vision attach: capture the viewport once per assert step and hand
739        // the JPEG data URL to the preset/prompt evaluation below.
740        let image = if screenshot {
741            let endpoint = self
742                .endpoints
743                .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
744            if !endpoint.vision {
745                return StepResult {
746                    name: "[assert]".into(),
747                    status: StepStatus::Failed,
748                    message: format!(
749                        "screenshot requested but endpoint '{name}' does not declare vision = true (add vision = true to its [config.endpoints] entry)",
750                        name = endpoint.name
751                    ),
752                };
753            }
754            match crate::vision::capture_screenshot_data_url(
755                tab,
756                self.config
757                    .screenshot_max_dimension
758                    .unwrap_or(crate::vision::DEFAULT_MAX_DIMENSION),
759            ) {
760                Ok(data_url) => Some(data_url),
761                Err(e) => {
762                    return StepResult {
763                        name: "[assert]".into(),
764                        status: StepStatus::Failed,
765                        message: format!("screenshot capture failed: {e}"),
766                    };
767                }
768            }
769        } else {
770            None
771        };
772
773        if let Some(def_name) = definition {
774            if let Some(def) = self.definitions.get(def_name) {
775                return self.run_assert_def(
776                    def,
777                    &page_content,
778                    image.as_deref(),
779                    step_endpoint,
780                    test_endpoint,
781                );
782            }
783            return StepResult {
784                name: format!("[assert] {def_name}"),
785                status: StepStatus::Failed,
786                message: format!("definition '{def_name}' not found"),
787            };
788        }
789
790        if let Some(preset_name) = preset {
791            return self.run_preset(
792                preset_name,
793                assert_text,
794                &page_content,
795                image.as_deref(),
796                step_endpoint,
797                test_endpoint,
798            );
799        }
800
801        if let Some(prompt_text) = prompt {
802            return self.run_custom(
803                prompt_text,
804                &page_content,
805                image.as_deref(),
806                step_endpoint,
807                test_endpoint,
808            );
809        }
810
811        StepResult {
812            name: "[assert]".into(),
813            status: StepStatus::Skipped,
814            message: "no definition, preset, or prompt specified".into(),
815        }
816    }
817
818    fn run_assert_def(
819        &self,
820        def: &AssertDefinition,
821        page_content: &PageContent,
822        image: Option<&str>,
823        step_endpoint: Option<&str>,
824        test_endpoint: Option<&str>,
825    ) -> StepResult {
826        // Agent-based definition: delegate to an A2A agent
827        if let Some(ref agent) = def.agent {
828            if image.is_some() {
829                return StepResult {
830                    name: format!("[assert] {}", def.name),
831                    status: StepStatus::Failed,
832                    message: "agent-backed assertions do not support screenshots".into(),
833                };
834            }
835            let task = def
836                .task_template
837                .as_deref()
838                .unwrap_or("Evaluate the assertion")
839                .replace("{url}", &page_content.url)
840                .replace("{title}", &page_content.title)
841                .replace("{content}", &page_content.body_text)
842                .replace("{expected_text}", def.assert_text.as_deref().unwrap_or(""));
843
844            return self.run_agent_step(agent, &task, &def.name);
845        }
846
847        // Custom preset: system + user_template provided in the definition
848        if let (Some(system), Some(template)) = (&def.system, &def.user_template) {
849            return self.run_custom_preset(
850                &def.name,
851                system,
852                template,
853                def.assert_text.as_deref(),
854                page_content,
855                image,
856                step_endpoint,
857                test_endpoint,
858            );
859        }
860
861        def.preset.as_ref().map_or_else(
862            || {
863                def.prompt.as_ref().map_or_else(
864                    || StepResult {
865                        name: format!("[assert] {}", def.name),
866                        status: StepStatus::Failed,
867                        message: "definition has no preset, prompt, or system+user_template".into(),
868                    },
869                    |prompt| {
870                        self.run_custom(prompt, page_content, image, step_endpoint, test_endpoint)
871                    },
872                )
873            },
874            |preset_name| {
875                self.run_preset(
876                    preset_name,
877                    def.assert_text.as_deref(),
878                    page_content,
879                    image,
880                    step_endpoint,
881                    test_endpoint,
882                )
883            },
884        )
885    }
886
887    #[allow(clippy::too_many_arguments)]
888    fn run_custom_preset(
889        &self,
890        name: &str,
891        system: &str,
892        template: &str,
893        assert_text: Option<&str>,
894        page_content: &PageContent,
895        image: Option<&str>,
896        step_endpoint: Option<&str>,
897        test_endpoint: Option<&str>,
898    ) -> StepResult {
899        let user_prompt = template
900            .replace("{url}", &page_content.url)
901            .replace("{title}", &page_content.title)
902            .replace("{content}", &page_content.body_text)
903            .replace("{expected_text}", assert_text.unwrap_or(""))
904            .replace("{description}", "");
905
906        // Custom preset definitions frequently forget the {content}
907        // placeholder — without it the LLM has no page to evaluate and
908        // answers "I can't determine that without seeing the page". Always
909        // append the page context unless the template already references it.
910        let user_prompt = if template.contains("{content}") {
911            user_prompt
912        } else {
913            format!(
914                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
915                url = page_content.url,
916                title = page_content.title,
917                content = page_content.body_text,
918            )
919        };
920
921        eprintln!("      assert: {name} (custom preset)");
922
923        let endpoint = self
924            .endpoints
925            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
926        let llm = self.build_llm_for_endpoint(endpoint);
927        let usage = Arc::clone(&self.usage);
928        let endpoint_name = endpoint.name.clone();
929        let sys = system.to_owned();
930        let image = image.map(str::to_owned);
931
932        let response = std::thread::spawn(move || {
933            let rt = tokio::runtime::Builder::new_current_thread()
934                .enable_all()
935                .build()
936                .unwrap();
937            let call = async {
938                match image.as_deref() {
939                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user_prompt, img).await,
940                    None => llm_chat_with_usage(&llm, &sys, &user_prompt).await,
941                }
942            };
943            rt.block_on(call)
944        })
945        .join()
946        .unwrap();
947
948        response.map_or_else(
949            |e| StepResult {
950                name: format!("[assert] {name}"),
951                status: StepStatus::Failed,
952                message: format!("LLM assertion call failed: {e}"),
953            },
954            |lr| {
955                usage.record_llm_call(
956                    &endpoint_name,
957                    endpoint,
958                    lr.usage.prompt_tokens,
959                    lr.usage.completion_tokens,
960                );
961                let content_lower = lr.content.to_lowercase().trim().to_owned();
962                if content_lower.starts_with("pass") {
963                    StepResult {
964                        name: format!("[assert] {name}"),
965                        status: StepStatus::Passed,
966                        message: "PASS".into(),
967                    }
968                } else {
969                    StepResult {
970                        name: format!("[assert] {name}"),
971                        status: StepStatus::Failed,
972                        message: lr.content,
973                    }
974                }
975            },
976        )
977    }
978
979    fn run_preset(
980        &self,
981        preset_name: &str,
982        assert_text: Option<&str>,
983        page_content: &PageContent,
984        image: Option<&str>,
985        step_endpoint: Option<&str>,
986        test_endpoint: Option<&str>,
987    ) -> StepResult {
988        let Some(preset) = ASSERTION_PRESETS.iter().find(|p| p.name == preset_name) else {
989            return StepResult {
990                name: format!("[assert] {preset_name}"),
991                status: StepStatus::Failed,
992                message: format!("unknown assertion preset: {preset_name}"),
993            };
994        };
995        if preset_name.starts_with("visual_") && image.is_none() {
996            return StepResult {
997                name: format!("[assert] {preset_name}"),
998                status: StepStatus::Failed,
999                message: format!(
1000                    "preset '{preset_name}' evaluates the page screenshot — set screenshot = true on the assert step and point it at a vision endpoint (vision = true)"
1001                ),
1002            };
1003        }
1004
1005        let user_prompt = preset
1006            .user_template
1007            .replace("{url}", &page_content.url)
1008            .replace("{title}", &page_content.title)
1009            .replace("{content}", &page_content.body_text)
1010            .replace("{expected_text}", assert_text.unwrap_or(""))
1011            .replace("{description}", "");
1012
1013        // Same safety net as custom presets: never let the LLM answer with
1014        // no page context at all.
1015        let user_prompt = if preset.user_template.contains("{content}") {
1016            user_prompt
1017        } else {
1018            format!(
1019                "{user_prompt}\n\nPage URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}",
1020                url = page_content.url,
1021                title = page_content.title,
1022                content = page_content.body_text,
1023            )
1024        };
1025
1026        eprintln!("      assert: {preset_name}");
1027
1028        let endpoint = self
1029            .endpoints
1030            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1031        let llm = self.build_llm_for_endpoint(endpoint);
1032        let usage = Arc::clone(&self.usage);
1033        let endpoint_name = endpoint.name.clone();
1034        let sys = preset.system.to_owned();
1035        let image = image.map(str::to_owned);
1036
1037        let response = std::thread::spawn(move || {
1038            let rt = tokio::runtime::Builder::new_current_thread()
1039                .enable_all()
1040                .build()
1041                .unwrap();
1042            let call = async {
1043                match image.as_deref() {
1044                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user_prompt, img).await,
1045                    None => llm_chat_with_usage(&llm, &sys, &user_prompt).await,
1046                }
1047            };
1048            rt.block_on(call)
1049        })
1050        .join()
1051        .unwrap();
1052
1053        response.map_or_else(
1054            |e| StepResult {
1055                name: format!("[assert] {preset_name}"),
1056                status: StepStatus::Failed,
1057                message: format!("LLM assertion call failed: {e}"),
1058            },
1059            |lr| {
1060                usage.record_llm_call(
1061                    &endpoint_name,
1062                    endpoint,
1063                    lr.usage.prompt_tokens,
1064                    lr.usage.completion_tokens,
1065                );
1066                let content_lower = lr.content.to_lowercase().trim().to_owned();
1067                if content_lower.starts_with("pass") {
1068                    StepResult {
1069                        name: format!("[assert] {preset_name}"),
1070                        status: StepStatus::Passed,
1071                        message: "PASS".into(),
1072                    }
1073                } else {
1074                    StepResult {
1075                        name: format!("[assert] {preset_name}"),
1076                        status: StepStatus::Failed,
1077                        message: lr.content,
1078                    }
1079                }
1080            },
1081        )
1082    }
1083
1084    fn run_custom(
1085        &self,
1086        prompt: &str,
1087        page_content: &PageContent,
1088        image: Option<&str>,
1089        step_endpoint: Option<&str>,
1090        test_endpoint: Option<&str>,
1091    ) -> StepResult {
1092        let system = "You are a QA tester evaluating a web page. Respond with exactly \"PASS\" if the assertion holds, or \"FAIL: <reason>\" if it does not.";
1093
1094        let mut user = format!(
1095            "Page URL: {url}\nPage Title: {title}\n\nPage Content:\n{content}\n\nAssertion: {prompt}",
1096            url = page_content.url,
1097            title = page_content.title,
1098            content = page_content.body_text,
1099        );
1100        if image.is_some() {
1101            user.push_str(
1102                "\n\nA screenshot of the page is attached — inspect it for visual evidence when answering.",
1103            );
1104        }
1105
1106        eprintln!("      custom assert");
1107
1108        let endpoint = self
1109            .endpoints
1110            .resolve(step_endpoint.or(test_endpoint), TaskType::Assertion);
1111        let llm = self.build_llm_for_endpoint(endpoint);
1112        let usage = Arc::clone(&self.usage);
1113        let endpoint_name = endpoint.name.clone();
1114        let sys = system.to_owned();
1115        let image = image.map(str::to_owned);
1116
1117        let response = std::thread::spawn(move || {
1118            let rt = tokio::runtime::Builder::new_current_thread()
1119                .enable_all()
1120                .build()
1121                .unwrap();
1122            let call = async {
1123                match image.as_deref() {
1124                    Some(img) => llm_chat_vision_with_usage(&llm, &sys, &user, img).await,
1125                    None => llm_chat_with_usage(&llm, &sys, &user).await,
1126                }
1127            };
1128            rt.block_on(call)
1129        })
1130        .join()
1131        .unwrap();
1132
1133        response.map_or_else(
1134            |e| StepResult {
1135                name: "[assert] custom".into(),
1136                status: StepStatus::Failed,
1137                message: format!("LLM assertion call failed: {e}"),
1138            },
1139            |lr| {
1140                usage.record_llm_call(
1141                    &endpoint_name,
1142                    endpoint,
1143                    lr.usage.prompt_tokens,
1144                    lr.usage.completion_tokens,
1145                );
1146                let content_lower = lr.content.to_lowercase().trim().to_owned();
1147                if content_lower.starts_with("pass") {
1148                    StepResult {
1149                        name: "[assert] custom".into(),
1150                        status: StepStatus::Passed,
1151                        message: "PASS".into(),
1152                    }
1153                } else {
1154                    StepResult {
1155                        name: "[assert] custom".into(),
1156                        status: StepStatus::Failed,
1157                        message: lr.content,
1158                    }
1159                }
1160            },
1161        )
1162    }
1163
1164    fn run_screenshot(path: Option<&str>, tab: &Tab) -> StepResult {
1165        let path = path.unwrap_or("screenshot.png");
1166
1167        match tab.capture_screenshot(
1168            headless_chrome::protocol::cdp::Page::CaptureScreenshotFormatOption::Png,
1169            None,
1170            None,
1171            true,
1172        ) {
1173            Ok(data) => {
1174                if let Err(e) = std::fs::write(path, &data) {
1175                    return StepResult {
1176                        name: format!("[screenshot] {path}"),
1177                        status: StepStatus::Failed,
1178                        message: format!("failed to write screenshot: {e}"),
1179                    };
1180                }
1181                StepResult {
1182                    name: format!("[screenshot] {path}"),
1183                    status: StepStatus::Passed,
1184                    message: format!("saved to {path}"),
1185                }
1186            }
1187            Err(e) => StepResult {
1188                name: format!("[screenshot] {path}"),
1189                status: StepStatus::Failed,
1190                message: format!("screenshot failed: {e}"),
1191            },
1192        }
1193    }
1194
1195    /// Runs an A2A agent step.
1196    #[allow(clippy::literal_string_with_formatting_args)]
1197    fn run_agent(
1198        &self,
1199        agent_name: &str,
1200        task: &str,
1201        definition: Option<&str>,
1202        _test_endpoint: Option<&str>,
1203    ) -> StepResult {
1204        // If a definition is specified, look up the task template
1205        let resolved_task = if let Some(def_name) = definition {
1206            if let Some(def) = self.definitions.get(def_name) {
1207                let tmpl = def.task_template.as_deref().unwrap_or(task);
1208                tmpl.replace("{task}", task)
1209            } else {
1210                return StepResult {
1211                    name: format!("[agent] {def_name}"),
1212                    status: StepStatus::Failed,
1213                    message: format!("definition '{def_name}' not found"),
1214                };
1215            }
1216        } else {
1217            task.to_owned()
1218        };
1219
1220        self.run_agent_step(agent_name, &resolved_task, &format!("agent:{agent_name}"))
1221    }
1222
1223    fn run_agent_step(&self, agent_name: &str, task: &str, display_name: &str) -> StepResult {
1224        let Some(ep) = self.endpoints.get(agent_name) else {
1225            return StepResult {
1226                name: format!("[agent] {display_name}"),
1227                status: StepStatus::Failed,
1228                message: format!("agent endpoint '{agent_name}' not found"),
1229            };
1230        };
1231
1232        if ep.url.is_empty() {
1233            return StepResult {
1234                name: format!("[agent] {display_name}"),
1235                status: StepStatus::Failed,
1236                message: format!("agent endpoint '{agent_name}' has no URL"),
1237            };
1238        }
1239
1240        eprintln!("      → agent {agent_name}: {task}");
1241
1242        let url = ep.url.clone();
1243        let client = A2aClient::new(&url, self.timeout);
1244        let task_clone = task.to_owned();
1245
1246        let response = std::thread::spawn(move || {
1247            let rt = tokio::runtime::Builder::new_current_thread()
1248                .enable_all()
1249                .build()
1250                .unwrap();
1251            rt.block_on(client.send_task(&task_clone))
1252        })
1253        .join()
1254        .unwrap();
1255
1256        // Record the flat-cost call
1257        self.usage.record_flat_call(agent_name, ep);
1258
1259        match response {
1260            Ok(text) => {
1261                let clean = text.trim().to_owned();
1262                let lower = clean.to_lowercase();
1263                if lower.starts_with("pass") {
1264                    StepResult {
1265                        name: format!("[agent] {display_name}"),
1266                        status: StepStatus::Passed,
1267                        message: format!("PASS: {clean}"),
1268                    }
1269                } else if lower.starts_with("fail") {
1270                    StepResult {
1271                        name: format!("[agent] {display_name}"),
1272                        status: StepStatus::Failed,
1273                        message: clean,
1274                    }
1275                } else {
1276                    StepResult {
1277                        name: format!("[agent] {display_name}"),
1278                        status: StepStatus::Passed,
1279                        message: format!("response: {clean}"),
1280                    }
1281                }
1282            }
1283            Err(e) => StepResult {
1284                name: format!("[agent] {display_name}"),
1285                status: StepStatus::Failed,
1286                message: format!("agent call failed: {e}"),
1287            },
1288        }
1289    }
1290
1291    /// Runs an MCP tool call step.
1292    fn run_mcp(
1293        &self,
1294        server_name: &str,
1295        tool_name: &str,
1296        args: Option<&serde_json::Value>,
1297    ) -> StepResult {
1298        let Some(ep) = self.endpoints.get(server_name) else {
1299            return StepResult {
1300                name: format!("[mcp] {server_name}:{tool_name}"),
1301                status: StepStatus::Failed,
1302                message: format!("MCP server endpoint '{server_name}' not found"),
1303            };
1304        };
1305
1306        let cmd = ep.command.as_deref().unwrap_or("");
1307        if cmd.is_empty() {
1308            return StepResult {
1309                name: format!("[mcp] {server_name}:{tool_name}"),
1310                status: StepStatus::Failed,
1311                message: format!("MCP server '{server_name}' has no command configured"),
1312            };
1313        }
1314
1315        eprintln!("      → mcp {server_name} {tool_name}");
1316
1317        let args_val = args.cloned().unwrap_or(serde_json::Value::Null);
1318
1319        let command = cmd.to_owned();
1320        let args_vec = ep.args.clone();
1321        let tool = tool_name.to_owned();
1322
1323        let response = std::thread::spawn(move || {
1324            let mut mcp_client =
1325                McpClient::connect_stdio(&command, &args_vec).map_err(|e| e.to_string())?;
1326            mcp_client
1327                .call_tool(&tool, &args_val)
1328                .map_err(|e| e.to_string())
1329        })
1330        .join()
1331        .unwrap();
1332
1333        // Record the flat-cost call
1334        self.usage.record_flat_call(server_name, ep);
1335
1336        match response {
1337            Ok(result) => {
1338                if result.isError {
1339                    StepResult {
1340                        name: format!("[mcp] {server_name}:{tool_name}"),
1341                        status: StepStatus::Failed,
1342                        message: result.to_string(),
1343                    }
1344                } else {
1345                    StepResult {
1346                        name: format!("[mcp] {server_name}:{tool_name}"),
1347                        status: StepStatus::Passed,
1348                        message: result.to_string(),
1349                    }
1350                }
1351            }
1352            Err(e) => StepResult {
1353                name: format!("[mcp] {server_name}:{tool_name}"),
1354                status: StepStatus::Failed,
1355                message: format!("MCP call failed: {e}"),
1356            },
1357        }
1358    }
1359
1360    // ── helpers ──────────────────────────────────────────────────────────
1361
1362    /// Builds an `LlmConfig` from a resolved endpoint, falling back to
1363    /// the runner's default LLM config for any unset fields.
1364    fn build_llm_for_endpoint(&self, endpoint: &crate::endpoints::ResolvedEndpoint) -> LlmConfig {
1365        LlmConfig {
1366            url: if endpoint.url.is_empty() {
1367                self.llm.url.clone()
1368            } else {
1369                endpoint.url.clone()
1370            },
1371            model: endpoint
1372                .model
1373                .clone()
1374                .unwrap_or_else(|| self.llm.model.clone()),
1375            api_key: endpoint
1376                .api_key
1377                .clone()
1378                .or_else(|| self.llm.api_key.clone()),
1379            headers: if endpoint.headers.is_empty() {
1380                self.llm.headers.clone()
1381            } else {
1382                endpoint.headers.clone()
1383            },
1384            timeout: self.llm.timeout,
1385            temperature: self.llm.temperature,
1386            thinking: self.llm.thinking,
1387            model_params: self.llm.model_params.clone(),
1388        }
1389    }
1390
1391    /// Resolves a CSS selector for the target element. Uses the explicit
1392    /// `selector` if provided, otherwise asks the LLM to find the element
1393    /// from the natural language `target` description and page DOM.
1394    ///
1395    /// LLM responses are sanitized and verified against the live page: a
1396    /// response that is not a selector (empty, `:not(*)`, `null`, …) fails
1397    /// immediately with the raw LLM output, and a selector that matches
1398    /// nothing triggers one retry with feedback before failing.
1399    #[allow(clippy::too_many_lines)]
1400    fn resolve_selector(
1401        &self,
1402        css_override: Option<&str>,
1403        target: &str,
1404        step_endpoint: Option<&str>,
1405        test_endpoint: Option<&str>,
1406        tab: &Tab,
1407    ) -> Result<String, String> {
1408        if let Some(explicit) = css_override {
1409            return Ok(explicit.to_owned());
1410        }
1411
1412        let dom_info = extract_dom_info(tab)?;
1413        let page_content = get_page_text(tab);
1414
1415        let system = concat!(
1416            "You are a browser automation selector generator. ",
1417            "Given a web page's content and interactive elements, ",
1418            "return ONLY the best CSS selector for the described element. ",
1419            "Output nothing except the CSS selector. ",
1420            "Prefer selectors in this order: #id, [data-testid=\"...\"], ",
1421            "[name=\"...\"], tag.class, tag. ",
1422            "Never output explanations, markdown, or extra text."
1423        );
1424
1425        let user = format!(
1426            "Page URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}",
1427            page_content.url,
1428            page_content.title,
1429            truncate(&page_content.body_text, 4000),
1430            dom_info,
1431            target,
1432        );
1433
1434        let retry_user = format!(
1435            "Your previous answer was not usable. {}\n\nPage URL: {}\nPage Title: {}\n\nPage body text (first 4000 chars):\n{}\n\nInteractive elements:\n{}\n\nFind the CSS selector for: {}\nReturn ONLY a single CSS selector that matches an existing element. No explanations.",
1436            "The selector must match at least one element currently present on the page.",
1437            page_content.url,
1438            page_content.title,
1439            truncate(&page_content.body_text, 4000),
1440            dom_info,
1441            target,
1442        );
1443
1444        eprintln!("      LLM targeting: {target}");
1445
1446        let endpoint = self
1447            .endpoints
1448            .resolve(step_endpoint.or(test_endpoint), TaskType::Targeting);
1449        let llm = self.build_llm_for_endpoint(endpoint);
1450        let usage = Arc::clone(&self.usage);
1451        let endpoint_name = endpoint.name.clone();
1452        let endpoint_clone = endpoint.clone();
1453        let sys = system.to_owned();
1454
1455        let call_llm = |prompt: &str| {
1456            let llm = llm.clone();
1457            let sys = sys.clone();
1458            let prompt = prompt.to_owned();
1459            std::thread::spawn(move || {
1460                let rt = tokio::runtime::Builder::new_current_thread()
1461                    .enable_all()
1462                    .build()
1463                    .unwrap();
1464                rt.block_on(llm_chat_with_usage(&llm, &sys, &prompt))
1465            })
1466            .join()
1467            .unwrap()
1468        };
1469
1470        let first = call_llm(&user);
1471        let lr = match first {
1472            Ok(lr) => lr,
1473            Err(e) => {
1474                return Err(format!("LLM element targeting failed: {e}"));
1475            }
1476        };
1477        usage.record_llm_call(
1478            &endpoint_name,
1479            &endpoint_clone,
1480            lr.usage.prompt_tokens,
1481            lr.usage.completion_tokens,
1482        );
1483        let clean = sanitize_selector(&lr.content);
1484        eprintln!("      resolved selector: {clean}");
1485
1486        if selector_is_useless(&clean) {
1487            return Err(format!(
1488                "LLM element targeting failed: the LLM did not return a usable selector for {target:?} (got {raw:?}). Check the page state in the diagnostics above.",
1489                raw = lr.content.trim(),
1490            ));
1491        }
1492        if let Err(reason) = validate_selector(&clean) {
1493            return Err(format!(
1494                "LLM element targeting failed: invalid selector for {target:?}: {reason} (LLM response: {raw:?})",
1495                raw = lr.content.trim(),
1496            ));
1497        }
1498        if !selector_matches(tab, &clean).unwrap_or(false) {
1499            // One retry with feedback: flaky models occasionally invent a
1500            // selector that does not exist on the page.
1501            eprintln!(
1502                "      selector {clean} matches nothing — retrying LLM targeting with feedback"
1503            );
1504            let second = call_llm(&retry_user);
1505            let lr2 = match second {
1506                Ok(lr2) => lr2,
1507                Err(e) => {
1508                    return Err(format!(
1509                        "LLM element targeting failed: first answer {clean:?} matched nothing, retry also failed: {e}"
1510                    ));
1511                }
1512            };
1513            usage.record_llm_call(
1514                &endpoint_name,
1515                &endpoint_clone,
1516                lr2.usage.prompt_tokens,
1517                lr2.usage.completion_tokens,
1518            );
1519            let clean2 = sanitize_selector(&lr2.content);
1520            eprintln!("      resolved selector (retry): {clean2}");
1521            if selector_is_useless(&clean2) {
1522                return Err(format!(
1523                    "LLM element targeting failed: selector {clean:?} matched nothing; the retry returned no usable selector for {target:?} (got {raw:?}). Page excerpt: {excerpt}",
1524                    raw = lr2.content.trim(),
1525                    excerpt = truncate(&page_content.body_text, 300),
1526                ));
1527            }
1528            if !selector_matches(tab, &clean2).unwrap_or(false) {
1529                return Err(format!(
1530                    "LLM element targeting failed: selector {clean2:?} does not match any element on the page for {target:?}. Verify the page state in the diagnostics; the login/SPA may not have rendered."
1531                ));
1532            }
1533            return Ok(clean2);
1534        }
1535
1536        Ok(clean)
1537    }
1538}
1539
1540/// Evaluates a JS expression that is expected to return a boolean.
1541fn eval_bool(tab: &Tab, js: &str) -> Result<bool, String> {
1542    tab.evaluate(js, false)
1543        .map_err(|e| format!("evaluate failed: {e}"))?
1544        .value
1545        .and_then(|v| v.as_bool())
1546        .ok_or_else(|| "evaluate returned non-boolean".to_owned())
1547}
1548
1549/// Checks whether a CSS selector matches at least one current element.
1550fn selector_matches(tab: &Tab, selector: &str) -> Result<bool, String> {
1551    eval_bool(tab, &crate::selectors::selector_matches_js(selector))
1552}
1553
1554// ── Free helper functions ──────────────────────────────────────────────
1555
1556fn run_navigate_step(full_url: &str, tab: &Tab) -> StepResult {
1557    let name = format!("[navigate] {full_url}");
1558    match tab.navigate_to(full_url) {
1559        Ok(_) => {
1560            let _ = tab.wait_until_navigated();
1561            StepResult {
1562                name,
1563                status: StepStatus::Passed,
1564                message: format!("navigated to {full_url}"),
1565            }
1566        }
1567        Err(e) => StepResult {
1568            name,
1569            status: StepStatus::Failed,
1570            message: format!("navigation failed: {e}"),
1571        },
1572    }
1573}
1574
1575fn extract_dom_info(tab: &Tab) -> Result<String, String> {
1576    let result = tab
1577        .evaluate(DOM_EXTRACT_JS, false)
1578        .map_err(|e| format!("DOM extraction failed: {e}"))?;
1579
1580    let json_str = result
1581        .value
1582        .as_ref()
1583        .and_then(|v| v.as_str())
1584        .unwrap_or("[]");
1585
1586    let elements: Vec<String> = serde_json::from_str(json_str).unwrap_or_default();
1587
1588    if elements.is_empty() {
1589        return Ok("(no interactive elements found)".to_owned());
1590    }
1591
1592    Ok(elements.join("\n"))
1593}
1594
1595fn get_page_text(tab: &Tab) -> PageContent {
1596    let url = tab.get_url();
1597
1598    let title = tab
1599        .evaluate("document.title", false)
1600        .ok()
1601        .and_then(|r| r.value)
1602        .and_then(|v| v.as_str().map(String::from))
1603        .unwrap_or_else(|| "unknown".to_owned());
1604
1605    let body_text = tab
1606        .evaluate(
1607            "document.body ? document.body.innerText : document.documentElement.innerText",
1608            false,
1609        )
1610        .ok()
1611        .and_then(|r| r.value)
1612        .and_then(|v| v.as_str().map(String::from))
1613        .unwrap_or_default();
1614
1615    PageContent {
1616        url,
1617        title,
1618        body_text: truncate(&body_text, 8000),
1619    }
1620}
1621
1622fn resolve_url(url: &str, base_url: &str) -> String {
1623    if url.starts_with("http://") || url.starts_with("https://") {
1624        return url.to_owned();
1625    }
1626    let base = base_url.trim_end_matches('/');
1627    if url.starts_with('/') {
1628        format!("{base}{url}")
1629    } else {
1630        format!("{base}/{url}")
1631    }
1632}
1633
1634/// Human-readable label for a step, used when steps are skipped after an
1635/// earlier failure.
1636fn step_label(step: &TestStep) -> String {
1637    match step {
1638        TestStep::Navigate { url, .. } => format!("[navigate] {url}"),
1639        TestStep::Click { target, .. } => format!("[click] {target}"),
1640        TestStep::Type { target, .. } => format!("[type] {target}"),
1641        TestStep::Wait { target, .. } => format!("[wait] {target}"),
1642        TestStep::Assert {
1643            definition,
1644            preset,
1645            prompt,
1646            ..
1647        } => definition.as_ref().map_or_else(
1648            || {
1649                preset.as_ref().map_or_else(
1650                    || {
1651                        prompt.as_ref().map_or_else(
1652                            || "[assert]".to_owned(),
1653                            |pr| format!("[assert] custom ({})", truncate(pr, 60)),
1654                        )
1655                    },
1656                    |p| format!("[assert] {p}"),
1657                )
1658            },
1659            |d| format!("[assert] {d}"),
1660        ),
1661        TestStep::Screenshot { .. } => "[screenshot]".to_owned(),
1662        TestStep::Agent { agent, .. } => format!("[agent] {agent}"),
1663        TestStep::Mcp { server, tool, .. } => format!("[mcp] {server}:{tool}"),
1664    }
1665}
1666
1667/// Short kind word for artifact file names (e.g. `click`, `wait`, `assert`).
1668#[must_use]
1669const fn step_kind_label(step: &TestStep) -> &'static str {
1670    match step {
1671        TestStep::Navigate { .. } => "navigate",
1672        TestStep::Click { .. } => "click",
1673        TestStep::Type { .. } => "type",
1674        TestStep::Wait { .. } => "wait",
1675        TestStep::Assert { .. } => "assert",
1676        TestStep::Screenshot { .. } => "screenshot",
1677        TestStep::Agent { .. } => "agent",
1678        TestStep::Mcp { .. } => "mcp",
1679    }
1680}
1681
1682// ── Support types ──────────────────────────────────────────────────────
1683
1684#[derive(Default)]
1685struct TestRunResult {
1686    passed: u32,
1687    failed: u32,
1688    skipped: u32,
1689    total: u32,
1690    details: Vec<StepResult>,
1691}
1692
1693struct PageContent {
1694    url: String,
1695    title: String,
1696    body_text: String,
1697}