Skip to main content

llm_browser_testkit/
lib.rs

1//! LLM-driven browser test framework.
2//!
3//! Provides reusable building blocks for browser-based test scenarios:
4//! - Browser client via Chrome `DevTools` Protocol (headless)
5//! - LLM client for natural language element targeting and assertions
6//! - A2A agent integration for agent-to-agent communication
7//! - MCP client/server integration for tool-calling
8//! - Cost tracking, token counting, and budget enforcement
9//! - Declarative TOML scenario runner
10//! - `#[browser_test]` macros for `cargo test` integration
11
12#![allow(
13    clippy::expect_used,
14    clippy::unwrap_used,
15    clippy::panic,
16    clippy::missing_panics_doc
17)]
18
19/// A2A agent protocol client.
20pub mod a2a;
21/// Budget tracking and enforcement.
22pub mod budgets;
23/// Cost calculation, usage tracking, and pricing.
24pub mod costs;
25/// Failure diagnostics — page-state capture and artifact writing.
26pub mod diagnostics;
27/// Endpoint registry and routing resolver.
28pub mod endpoints;
29/// Typed run events emitted by the runner.
30pub mod events;
31/// MCP client for connecting to external MCP servers.
32pub mod mcp_client;
33/// MCP server for exposing the framework as an MCP server.
34pub mod mcp_server;
35/// Run reporting: console, NDJSON, JUnit, GitHub and Perfetto sinks.
36pub mod reporting;
37/// Step-by-step scenario executor (navigate, click, type, wait, assert).
38pub mod runner;
39/// Declarative TOML-based test scenario types.
40pub mod scenario;
41/// CSS selector sanitization for LLM-generated selectors.
42pub mod selectors;
43/// Vision support — screenshot capture/downscale/encode for visual asserts.
44pub mod vision;
45
46/// A2A agent server for accepting agent tasks.
47#[cfg(feature = "a2a-server")]
48pub mod a2a_server;
49
50/// `#[browser_test]` macros for `cargo test` integration.
51#[cfg(feature = "macros")]
52pub mod macros;
53
54use std::collections::HashMap;
55use std::time::Duration;
56
57use serde_json::{json, Value};
58
59pub use costs::LlmResponse;
60pub use costs::LlmUsage;
61
62/// Configuration for the LLM client — bundles URL, model, auth, timeouts,
63/// and provider-specific options into a single struct passed everywhere.
64#[derive(Debug, Clone)]
65pub struct LlmConfig {
66    /// OpenAI-compatible API base URL (without trailing `/v1/…`).
67    pub url: String,
68    /// Model name (e.g. `gpt-4o-mini`, `deepseek`).
69    pub model: String,
70    /// API key sent as `Authorization: Bearer <key>`.
71    pub api_key: Option<String>,
72    /// Custom headers appended to every LLM request.
73    pub headers: HashMap<String, String>,
74    /// HTTP timeout.
75    pub timeout: Duration,
76    /// Sampling temperature (0.0–1.0).
77    pub temperature: f64,
78    /// Enable extended thinking / reasoning tokens.
79    /// `None` = don't send any thinking key (provider default).
80    pub thinking: Option<bool>,
81    /// Provider-specific parameters merged into the request body
82    /// (e.g. `effort = "high"` for Anthropic).
83    pub model_params: HashMap<String, Value>,
84    /// How many times a single call to this endpoint is retried on
85    /// transient failures before giving up (or moving to the next fallback
86    /// endpoint). Default 3; override globally with
87    /// `HARNESS_LLM_CALL_ATTEMPTS`.
88    pub max_attempts: u32,
89}
90
91impl LlmConfig {
92    /// Build a config from environment defaults, falling back to safe
93    /// values when no env vars are set.
94    #[must_use]
95    pub fn from_env() -> Self {
96        Self {
97            url: llm_base_url(),
98            model: llm_model(),
99            api_key: std::env::var("HARNESS_LLM_API_KEY").ok(),
100            headers: parse_headers_env(),
101            timeout: Duration::from_secs(60),
102            temperature: 0.0,
103            thinking: None,
104            model_params: HashMap::new(),
105            max_attempts: default_llm_attempts(),
106        }
107    }
108}
109
110/// Reads `HARNESS_LLM_CALL_ATTEMPTS` (default 3) — how many times a single
111/// chat completion is retried before the endpoint is considered failed.
112#[must_use]
113pub fn default_llm_attempts() -> u32 {
114    std::env::var("HARNESS_LLM_CALL_ATTEMPTS")
115        .ok()
116        .and_then(|v| v.parse().ok())
117        .filter(|n| *n >= 1)
118        .unwrap_or(3)
119}
120
121/// Parses `HARNESS_LLM_HEADERS` env var (JSON object) into a header map.
122///
123/// Exposed for use by `endpoints.rs` and tests.
124#[must_use]
125pub fn parse_headers_env() -> HashMap<String, String> {
126    let Ok(raw) = std::env::var("HARNESS_LLM_HEADERS") else {
127        return HashMap::new();
128    };
129    let Ok(json) = serde_json::from_str::<Value>(&raw) else {
130        return HashMap::new();
131    };
132    let Some(obj) = json.as_object() else {
133        return HashMap::new();
134    };
135    obj.iter()
136        .filter_map(|(k, v)| v.as_str().map(|s| (k.clone(), s.to_owned())))
137        .collect()
138}
139
140/// Returns the target base URL from `HARNESS_BROWSER_BASE_URL` env,
141/// defaulting to `http://localhost:4200`.
142#[must_use]
143pub fn base_url() -> String {
144    std::env::var("HARNESS_BROWSER_BASE_URL").unwrap_or_else(|_| "http://localhost:4200".to_owned())
145}
146
147/// Returns the LLM server base URL from `HARNESS_LLM_TEST_URL` env,
148/// defaulting to `http://localhost:8080`.
149#[must_use]
150pub fn llm_base_url() -> String {
151    std::env::var("HARNESS_LLM_TEST_URL")
152        .unwrap_or_else(|_| "http://localhost:8080".to_owned())
153        .trim_end_matches('/')
154        .to_owned()
155}
156
157/// Returns the LLM model name from `HARNESS_LLM_TEST_MODEL` env,
158/// defaulting to `deepseek`.
159#[must_use]
160pub fn llm_model() -> String {
161    std::env::var("HARNESS_LLM_TEST_MODEL").unwrap_or_else(|_| "deepseek".to_owned())
162}
163
164/// Returns whether to run the browser in headless mode from
165/// `HARNESS_BROWSER_HEADLESS` env, defaulting to `true`.
166#[must_use]
167pub fn browser_headless() -> bool {
168    std::env::var("HARNESS_BROWSER_HEADLESS")
169        .map_or(true, |v| v != "0" && v.to_lowercase() != "false")
170}
171
172/// Builds a `reqwest::Client` with the given timeout.
173#[must_use]
174pub fn http_client(timeout: Duration) -> reqwest::Client {
175    reqwest::Client::builder()
176        .timeout(timeout)
177        .build()
178        .expect("build reqwest client")
179}
180
181/// Sends a chat completion request to the LLM.
182///
183/// Returns `Some(content)` on success, `None` on any error.
184///
185/// Prefer `llm_chat_with_usage` if you need token counting.
186#[must_use]
187pub async fn llm_chat(llm: &LlmConfig, system: &str, user: &str) -> Option<String> {
188    llm_chat_with_usage(llm, system, user)
189        .await
190        .map(|r| r.content)
191        .ok()
192}
193
194/// Sends a chat completion request to the LLM and returns both the content
195/// and token usage from the API response.
196///
197/// Retries transient failures (network errors, HTTP 429/5xx, invalid
198/// responses, and HTTP 200 + empty body — a gateway warm-up signature,
199/// retried with a 3s backoff) up to `llm.max_attempts` times, and returns
200/// the last underlying error instead of collapsing everything into a
201/// generic "server down" message. The error text includes the HTTP status
202/// and a truncated response-body snippet, so a gateway that answers with
203/// an HTML error page is identifiable in CI logs instead of surfacing as a
204/// bare JSON decode error. Deterministic client errors (401/403/404) are
205/// not retried.
206///
207/// # Errors
208///
209/// Returns the last underlying error as a human-readable string when every
210/// attempt fails (transport error, non-success HTTP status, response that is
211/// not valid JSON, an empty HTTP 200 body, or a response missing
212/// `choices[0].message.content`).
213pub async fn llm_chat_with_usage(
214    llm: &LlmConfig,
215    system: &str,
216    user: &str,
217) -> Result<LlmResponse, String> {
218    chat_with_retry(llm, system, user, None).await
219}
220
221/// Sends a vision-enabled chat completion request: the user message carries
222/// both the text prompt and a screenshot (JPEG/PNG data URL) as an
223/// OpenAI-compatible `image_url` content part.
224///
225/// Retries and error reporting behave like [`llm_chat_with_usage`].
226///
227/// # Errors
228///
229/// Returns the last underlying error as a human-readable string when every
230/// attempt fails (transport error, non-success HTTP status, response that is
231/// not valid JSON, or a response missing `choices[0].message.content`).
232pub async fn llm_chat_vision_with_usage(
233    llm: &LlmConfig,
234    system: &str,
235    user: &str,
236    image_data_url: &str,
237) -> Result<LlmResponse, String> {
238    chat_with_retry(llm, system, user, Some(image_data_url)).await
239}
240
241/// Calls a chain of endpoints: the primary [`LlmConfig`] first, then each
242/// fallback in order. Every endpoint gets its own `max_attempts` retry
243/// budget; the first endpoint that answers wins.
244///
245/// Returns the response together with the index of the endpoint that
246/// produced it (0 = primary, 1 = first fallback, …) so the caller can
247/// attribute cost/usage to the right endpoint.
248///
249/// # Errors
250///
251/// Returns an error naming every endpoint that failed.
252pub async fn llm_chat_with_usage_chain(
253    primary: &LlmConfig,
254    fallbacks: &[LlmConfig],
255    system: &str,
256    user: &str,
257) -> Result<(LlmResponse, usize), String> {
258    chat_chain_with_retry(primary, fallbacks, system, user, None).await
259}
260
261/// Vision variant of [`llm_chat_with_usage_chain`].
262///
263/// # Errors
264///
265/// Returns an error naming every endpoint that failed.
266pub async fn llm_chat_vision_with_usage_chain(
267    primary: &LlmConfig,
268    fallbacks: &[LlmConfig],
269    system: &str,
270    user: &str,
271    image_data_url: &str,
272) -> Result<(LlmResponse, usize), String> {
273    chat_chain_with_retry(primary, fallbacks, system, user, Some(image_data_url)).await
274}
275
276/// Shared chain loop: try each endpoint (primary then fallbacks) with its
277/// own retry budget; first success wins.
278async fn chat_chain_with_retry(
279    primary: &LlmConfig,
280    fallbacks: &[LlmConfig],
281    system: &str,
282    user: &str,
283    image_data_url: Option<&str>,
284) -> Result<(LlmResponse, usize), String> {
285    let mut failures: Vec<String> = Vec::new();
286    for (i, llm) in std::iter::once(primary).chain(fallbacks.iter()).enumerate() {
287        match chat_with_retry(llm, system, user, image_data_url).await {
288            Ok(resp) => return Ok((resp, i)),
289            Err(e) => failures.push(format!("endpoint '{}' ({:?}): {e}", llm.url, llm.model)),
290        }
291    }
292    let details = failures.iter().fold(String::new(), |mut acc, f| {
293        use std::fmt::Write as _;
294        let _ = writeln!(acc, "  - {f}");
295        acc
296    });
297    Err(format!(
298        "LLM call failed on all {} endpoint(s):\n{details}",
299        failures.len()
300    ))
301}
302
303/// Shared retry loop for text-only and vision chat completions.
304async fn chat_with_retry(
305    llm: &LlmConfig,
306    system: &str,
307    user: &str,
308    image_data_url: Option<&str>,
309) -> Result<LlmResponse, String> {
310    let client = http_client(llm.timeout);
311    let mut last_err = String::from("LLM call failed");
312    let mut attempts: u32 = 0;
313
314    while attempts < llm.max_attempts {
315        attempts += 1;
316        match llm_chat_once(&client, llm, system, user, image_data_url).await {
317            Ok(resp) => return Ok(resp),
318            Err(err) => {
319                // An HTTP 200 with an empty body is a gateway warm-up
320                // signature: give it a longer window to finish booting
321                // instead of hammering it with short retries.
322                let backoff = match &err {
323                    LlmCallError::EmptyBody { .. } => Duration::from_secs(3),
324                    _ => Duration::from_millis(500 * u64::from(attempts)),
325                };
326                last_err = err.to_string();
327                if attempts >= llm.max_attempts || !err.is_retryable() {
328                    break;
329                }
330                tokio::time::sleep(backoff).await;
331            }
332        }
333    }
334
335    Err(format!(
336        "LLM call failed after {attempts} attempt(s) (endpoint {url}): {last_err}",
337        url = llm.url
338    ))
339}
340
341/// Builds the chat messages array. Text-only messages keep the plain
342/// string `content` shape (maximum provider compatibility); vision calls
343/// use the OpenAI-compatible content-part array with a `data:` image URL.
344#[must_use]
345fn build_messages(system: &str, user: &str, image_data_url: Option<&str>) -> Value {
346    let user_content = image_data_url.map_or_else(
347        || Value::String(user.to_owned()),
348        |url| {
349            json!([
350                {"type": "text", "text": user},
351                {"type": "image_url", "image_url": {"url": url}}
352            ])
353        },
354    );
355    json!([
356        {"role": "system", "content": system},
357        {"role": "user", "content": user_content}
358    ])
359}
360
361/// Internal error type for a single LLM request attempt, distinguishing
362/// transient failures (worth retrying) from deterministic configuration
363/// errors (fail immediately).
364enum LlmCallError {
365    /// Transport-level failure (connect, timeout, TLS, …).
366    Transport { message: String },
367    /// Non-success HTTP status with a body snippet.
368    Http { status: u16, body: String },
369    /// Success status but the body is not valid JSON.
370    InvalidJson {
371        status: u16,
372        detail: String,
373        body: String,
374    },
375    /// Success status with an EMPTY body — the classic transient gateway
376    /// warm-up signature (HTTP 200, zero bytes). Retried with a longer
377    /// backoff than other errors.
378    EmptyBody { status: u16 },
379    /// Valid JSON but missing `choices[0].message.content`.
380    MissingContent { json: String },
381}
382
383impl std::fmt::Display for LlmCallError {
384    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
385        match self {
386            Self::Transport { message } => write!(f, "LLM HTTP request failed: {message}"),
387            Self::Http { status, body } => {
388                write!(
389                    f,
390                    "LLM endpoint returned HTTP {status}: {}",
391                    truncate(body, 300)
392                )
393            }
394            Self::InvalidJson {
395                status,
396                detail,
397                body,
398            } => write!(
399                f,
400                "LLM endpoint returned HTTP {status} with non-JSON body ({detail}): {}",
401                truncate(body, 300)
402            ),
403            Self::EmptyBody { status } => write!(
404                f,
405                "LLM endpoint returned HTTP {status} with an empty response (likely gateway warm-up)"
406            ),
407            Self::MissingContent { json } => write!(
408                f,
409                "LLM response missing choices[0].message.content: {}",
410                truncate(json, 300)
411            ),
412        }
413    }
414}
415
416impl LlmCallError {
417    /// Whether another attempt may succeed. Network errors, rate limits,
418    /// server errors, and 200-with-garbage responses can be transient on
419    /// flaky gateways; auth/not-found errors are deterministic.
420    #[must_use]
421    fn is_retryable(&self) -> bool {
422        match self {
423            Self::Transport { .. } | Self::MissingContent { .. } | Self::EmptyBody { .. } => true,
424            Self::Http { status, .. } => {
425                *status == 408 || *status == 429 || (500..600).contains(status)
426            }
427            Self::InvalidJson { status, .. } => {
428                *status == 200 || *status == 408 || *status == 429 || (500..600).contains(status)
429            }
430        }
431    }
432}
433
434/// Single LLM chat request attempt; returns the underlying error as text.
435async fn llm_chat_once(
436    client: &reqwest::Client,
437    llm: &LlmConfig,
438    system: &str,
439    user: &str,
440    image_data_url: Option<&str>,
441) -> Result<LlmResponse, LlmCallError> {
442    let mut payload = serde_json::json!({
443        "model": llm.model,
444        "messages": build_messages(system, user, image_data_url),
445        "max_tokens": 4096,
446        "temperature": llm.temperature
447    });
448    if let Some(think) = llm.thinking {
449        if think {
450            payload["thinking"] = serde_json::json!({"type": "enabled"});
451        } else {
452            payload["thinking"] = serde_json::json!({"type": "disabled"});
453        }
454    }
455    // Merge provider-specific parameters into the request body.
456    if !llm.model_params.is_empty() {
457        if let Value::Object(ref mut map) = payload {
458            for (key, val) in &llm.model_params {
459                map.insert(key.clone(), val.clone());
460            }
461        }
462    }
463
464    let mut req = client
465        .post(format!("{}/v1/chat/completions", llm.url))
466        .header("Content-Type", "application/json");
467
468    if let Some(ref key) = llm.api_key {
469        req = req.header("Authorization", format!("Bearer {key}"));
470    }
471    for (name, value) in &llm.headers {
472        req = req.header(name.as_str(), value.as_str());
473    }
474
475    let resp = req
476        .json(&payload)
477        .send()
478        .await
479        .map_err(|e| LlmCallError::Transport {
480            message: e.to_string(),
481        })?;
482    let status = resp.status();
483    let status_u16 = status.as_u16();
484    let body = resp.text().await.unwrap_or_default();
485    if !status.is_success() {
486        return Err(LlmCallError::Http {
487            status: status_u16,
488            body,
489        });
490    }
491    if body.trim().is_empty() {
492        // HTTP 200 + empty body: a transient gateway hiccup (cold-start /
493        // warm-up), not a client error. Retried with a longer backoff.
494        return Err(LlmCallError::EmptyBody { status: status_u16 });
495    }
496    let json: Value = match serde_json::from_str(&body) {
497        Ok(v) => v,
498        Err(e) => {
499            return Err(LlmCallError::InvalidJson {
500                status: status_u16,
501                detail: e.to_string(),
502                body,
503            });
504        }
505    };
506    let usage = costs::extract_usage(&json);
507    let content = json["choices"][0]["message"]["content"]
508        .as_str()
509        .map(String::from)
510        .ok_or_else(|| LlmCallError::MissingContent {
511            json: json.to_string(),
512        })?;
513
514    Ok(LlmResponse { content, usage })
515}
516
517/// JavaScript to extract interactive elements from the current page.
518/// Returns a JSON array of objects with tag, selector, and label.
519pub const DOM_EXTRACT_JS: &str = r#"
520(() => {
521  const interactive = 'a, button, input, textarea, select, [role="button"], [onclick], [tabindex], [data-testid], [aria-label]';
522  const els = document.querySelectorAll(interactive);
523  const info = [];
524  const seen = new Set();
525  els.forEach((el, i) => {
526    const rect = el.getBoundingClientRect();
527    if (rect.width === 0 || rect.height === 0) return;
528    const tag = el.tagName.toLowerCase();
529    let selector = '';
530    if (el.id) selector = '#' + CSS.escape(el.id);
531    else if (el.getAttribute('data-testid')) selector = '[data-testid="' + el.getAttribute('data-testid') + '"]';
532    else if (el.name) selector = '[name="' + CSS.escape(el.name) + '"]';
533    else if (el.className && typeof el.className === 'string') {
534      const cls = el.className.trim().split(/\\s+/)[0];
535      if (cls) selector = tag + '.' + CSS.escape(cls);
536    }
537    if (!selector) selector = tag;
538    if (seen.has(selector)) return;
539    seen.add(selector);
540
541    let label = '';
542    const aria = el.getAttribute('aria-label');
543    if (aria) {
544      label = aria;
545    } else if (tag === 'input' || tag === 'textarea' || tag === 'select') {
546      label = el.placeholder || el.name || el.getAttribute('aria-label') || '';
547      if (el.type && !label) label = el.type;
548    } else {
549      label = (el.textContent || '').trim().substring(0, 80);
550    }
551
552    info.push(i + ': ' + selector + ' [' + tag + '] "' + label + '"');
553  });
554  return JSON.stringify(info);
555})()
556"#;
557
558/// Truncates a string to the given maximum length, appending a marker with
559/// the number of omitted characters if truncation occurred.
560///
561/// The cut point is always a UTF-8 char boundary, so multi-byte input (umlauts,
562/// emoji, CJK) can never panic the caller.
563#[must_use]
564pub fn truncate(s: &str, max_len: usize) -> String {
565    if s.len() <= max_len {
566        s.to_owned()
567    } else {
568        let cut = floor_char_boundary(s, max_len);
569        let omitted = s[cut..].chars().count();
570        format!("{}...<truncated {omitted} chars>", &s[..cut])
571    }
572}
573
574/// Returns the largest char boundary index in `s` that is `<= index`.
575fn floor_char_boundary(s: &str, index: usize) -> usize {
576    let index = index.min(s.len());
577    let mut i = index;
578    while i > 0 && !s.is_char_boundary(i) {
579        i -= 1;
580    }
581    i
582}
583
584#[cfg(test)]
585mod tests {
586    /// Serializes env-var-mutating tests: they race when the test binary runs
587    /// them in parallel, which intermittently failed CI.
588    static ENV_LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(());
589
590    /// Acquires the env lock for one test.
591    fn env_guard() -> std::sync::MutexGuard<'static, ()> {
592        ENV_LOCK
593            .lock()
594            .unwrap_or_else(std::sync::PoisonError::into_inner)
595    }
596
597    use crate::costs::extract_usage;
598    use crate::truncate;
599    use crate::{default_llm_attempts, llm_base_url, llm_model, parse_headers_env, LlmConfig};
600
601    /// Starts a minimal HTTP server that answers every chat-completions
602    /// request with `status`/`body`. Returns its base URL.
603    fn mock_llm_server(status: u16, body: &'static str) -> String {
604        use std::io::{Read, Write};
605        use std::net::TcpListener;
606        let listener = TcpListener::bind("127.0.0.1:0").unwrap();
607        let addr = listener.local_addr().unwrap();
608        std::thread::spawn(move || {
609            for stream in listener.incoming() {
610                let Ok(mut stream) = stream else { break };
611                let mut buf = [0u8; 4096];
612                let _ = stream.read(&mut buf);
613                let resp = format!(
614                    "HTTP/1.1 {status} {}\r\nContent-Type: application/json\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{}",
615                    if status == 200 { "OK" } else { "ERROR" },
616                    body.len(),
617                    body
618                );
619                let _ = stream.write_all(resp.as_bytes());
620            }
621        });
622        format!("http://{addr}")
623    }
624
625    const PASS_BODY: &str = r#"{"choices":[{"message":{"content":"PASS"}}],"usage":{"prompt_tokens":7,"completion_tokens":2}}"#;
626
627    fn cfg(url: &str, attempts: u32) -> LlmConfig {
628        LlmConfig {
629            url: url.to_owned(),
630            model: "mock".to_owned(),
631            api_key: None,
632            headers: std::collections::HashMap::new(),
633            timeout: std::time::Duration::from_secs(10),
634            temperature: 0.0,
635            thinking: None,
636            model_params: std::collections::HashMap::new(),
637            max_attempts: attempts,
638        }
639    }
640
641    #[tokio::test]
642    async fn test_chain_primary_success_returns_index_zero() {
643        let good = mock_llm_server(200, PASS_BODY);
644        let (resp, idx) = crate::llm_chat_with_usage_chain(&cfg(&good, 2), &[], "s", "u")
645            .await
646            .expect("primary endpoint should answer");
647        assert_eq!(idx, 0);
648        assert_eq!(resp.content, "PASS");
649        assert_eq!(resp.usage.prompt_tokens, 7);
650    }
651
652    #[tokio::test]
653    async fn test_chain_falls_back_on_empty_200() {
654        // Primary always returns HTTP 200 with an EMPTY body (the gateway
655        // warm-up signature) — after `max_attempts` it must hand over to the
656        // fallback, which answers properly.
657        let broken = mock_llm_server(200, "");
658        let good = mock_llm_server(200, PASS_BODY);
659        let (resp, idx) =
660            crate::llm_chat_with_usage_chain(&cfg(&broken, 2), &[cfg(&good, 2)], "s", "u")
661                .await
662                .expect("fallback endpoint should answer");
663        assert_eq!(idx, 1);
664        assert_eq!(resp.content, "PASS");
665    }
666
667    #[tokio::test]
668    async fn test_chain_reports_all_endpoints_on_total_failure() {
669        let broken1 = mock_llm_server(200, "");
670        let broken2 = mock_llm_server(503, "unavailable");
671        let err =
672            crate::llm_chat_with_usage_chain(&cfg(&broken1, 2), &[cfg(&broken2, 2)], "s", "u")
673                .await
674                .expect_err("both endpoints fail");
675        assert!(err.contains("all 2 endpoint(s)"), "got: {err}");
676        assert!(err.contains(&broken1), "primary URL missing: {err}");
677        assert!(err.contains(&broken2), "fallback URL missing: {err}");
678    }
679
680    #[test]
681    fn test_default_llm_attempts_env() {
682        let _env = env_guard();
683        std::env::set_var("HARNESS_LLM_CALL_ATTEMPTS", "7");
684        assert_eq!(default_llm_attempts(), 7);
685        std::env::set_var("HARNESS_LLM_CALL_ATTEMPTS", "0");
686        assert_eq!(default_llm_attempts(), 3, "0 must fall back to default");
687        std::env::set_var("HARNESS_LLM_CALL_ATTEMPTS", "junk");
688        assert_eq!(default_llm_attempts(), 3, "non-numeric must fall back");
689        std::env::remove_var("HARNESS_LLM_CALL_ATTEMPTS");
690        assert_eq!(default_llm_attempts(), 3);
691    }
692
693    #[test]
694    fn test_truncate_short() {
695        assert_eq!(truncate("hello", 10), "hello");
696    }
697
698    #[test]
699    fn test_truncate_long() {
700        let result = truncate("hello world", 5);
701        assert!(result.contains("<truncated 6 chars>"));
702        assert!(result.starts_with("hello"));
703    }
704
705    #[test]
706    fn test_truncate_exact_length() {
707        assert_eq!(truncate("abcde", 5), "abcde");
708    }
709
710    #[test]
711    fn test_truncate_empty() {
712        assert_eq!(truncate("", 5), "");
713    }
714
715    #[test]
716    fn test_parse_headers_env_empty() {
717        let _env = env_guard();
718        std::env::remove_var("HARNESS_LLM_HEADERS");
719        let h = parse_headers_env();
720        assert!(h.is_empty());
721    }
722
723    #[test]
724    fn test_parse_headers_env_valid() {
725        let _env = env_guard();
726        std::env::set_var("HARNESS_LLM_HEADERS", r#"{"X-Org":"acme","X-Version":"1"}"#);
727        let h = parse_headers_env();
728        assert_eq!(h.get("X-Org").map(String::as_str), Some("acme"));
729        assert_eq!(h.get("X-Version").map(String::as_str), Some("1"));
730        std::env::remove_var("HARNESS_LLM_HEADERS");
731    }
732
733    #[test]
734    fn test_parse_headers_env_invalid_json() {
735        let _env = env_guard();
736        std::env::set_var("HARNESS_LLM_HEADERS", "not-json");
737        let h = parse_headers_env();
738        assert!(h.is_empty());
739        std::env::remove_var("HARNESS_LLM_HEADERS");
740    }
741
742    #[test]
743    fn test_llm_config_from_env_defaults() {
744        let _env = env_guard();
745        #[allow(clippy::float_cmp)]
746        {
747            let config = LlmConfig::from_env();
748            assert_eq!(config.temperature, 0.0);
749            assert!(config.thinking.is_none());
750            assert!(config.model_params.is_empty());
751        }
752    }
753
754    #[test]
755    fn test_extract_usage_full() {
756        let json = serde_json::json!({
757            "usage": {
758                "prompt_tokens": 100,
759                "completion_tokens": 200,
760                "total_tokens": 300
761            }
762        });
763        let usage = extract_usage(&json);
764        assert_eq!(usage.prompt_tokens, 100);
765        assert_eq!(usage.completion_tokens, 200);
766        assert_eq!(usage.total_tokens, 300);
767    }
768
769    #[test]
770    fn test_extract_usage_empty() {
771        let json = serde_json::json!({});
772        let usage = extract_usage(&json);
773        assert_eq!(usage.prompt_tokens, 0);
774        assert_eq!(usage.completion_tokens, 0);
775        assert_eq!(usage.total_tokens, 0);
776    }
777
778    #[test]
779    fn test_truncate_unicode() {
780        // max_len is a byte budget; the cut lands on a char boundary.
781        // 'é' is 2 bytes, so index 3 captures "hé" (h=0, é=bytes 1-2)
782        assert_eq!(truncate("héllo", 3), "hé...<truncated 3 chars>");
783        // Length 5 captures full string (5 bytes)
784        assert_eq!(truncate("hello", 5), "hello");
785    }
786
787    #[test]
788    fn test_truncate_utf8_boundary_mid_char_does_not_panic() {
789        // Previously: &s[..2] panicked with "not a char boundary" because the
790        // cut landed inside the 2-byte 'é'. The runner hit this on real pages
791        // full of umlauts/emoji — inside the diagnostics path that was
792        // supposed to save the run.
793        // floor_char_boundary(2) lands BEFORE 'é' (byte 1), keeping "h".
794        let result = truncate("héllo", 2);
795        assert_eq!(result, "h...<truncated 4 chars>");
796
797        // 4-byte emoji: max_len=4 lands exactly on the first 🎉 boundary? No —
798        // boundary 0 is the largest <= 4 only if 4 is a boundary; it is, so
799        // cut=4 keeps "🎉". Check with 5 instead: cut back to 4.
800        let cut_inside = truncate("🎉🎉🎉 boom", 5);
801        assert_eq!(cut_inside, "🎉...<truncated 7 chars>");
802        assert!(is_valid_utf8(&cut_inside), "result must stay valid UTF-8");
803    }
804
805    #[test]
806    fn test_truncate_utf8_exact_omitted_count() {
807        // 5 ASCII chars cut at 10 bytes → 5 omitted
808        assert_eq!(truncate("abcdefghij", 5), "abcde...<truncated 5 chars>");
809        // 3 multibyte chars cut at exactly their boundary → 0... but the
810        // guard `len <= max_len` returns the raw string first.
811        assert_eq!(truncate("ééé", 6), "ééé");
812        assert_eq!(truncate("ééé", 5), "éé...<truncated 1 chars>");
813    }
814
815    fn is_valid_utf8(s: &str) -> bool {
816        std::str::from_utf8(s.as_bytes()).is_ok()
817    }
818
819    #[test]
820    fn test_parse_headers_env_non_object() {
821        let _env = env_guard();
822        std::env::set_var("HARNESS_LLM_HEADERS", "[1, 2, 3]");
823        let h = parse_headers_env();
824        assert!(h.is_empty());
825        std::env::remove_var("HARNESS_LLM_HEADERS");
826    }
827
828    #[test]
829    fn test_parse_headers_env_nested_values_filtered() {
830        let _env = env_guard();
831        std::env::set_var(
832            "HARNESS_LLM_HEADERS",
833            r#"{"str":"val","num":42,"bool":true}"#,
834        );
835        let h = parse_headers_env();
836        assert_eq!(h.get("str").map(String::as_str), Some("val"));
837        assert!(!h.contains_key("num"));
838        assert!(!h.contains_key("bool"));
839        std::env::remove_var("HARNESS_LLM_HEADERS");
840    }
841
842    #[test]
843    fn test_llm_config_has_default_model() {
844        let _env = env_guard();
845        let config = LlmConfig::from_env();
846        assert!(!config.model.is_empty());
847    }
848
849    #[test]
850    fn test_llm_base_url_default() {
851        let _env = env_guard();
852        std::env::remove_var("HARNESS_LLM_TEST_URL");
853        let url = llm_base_url();
854        assert_eq!(url, "http://localhost:8080");
855    }
856
857    #[test]
858    fn test_llm_base_url_custom() {
859        let _env = env_guard();
860        std::env::set_var("HARNESS_LLM_TEST_URL", "https://custom.api.com/v1");
861        let url = llm_base_url();
862        assert_eq!(url, "https://custom.api.com/v1");
863        std::env::remove_var("HARNESS_LLM_TEST_URL");
864    }
865
866    #[test]
867    fn test_llm_base_url_trailing_slash() {
868        let _env = env_guard();
869        std::env::set_var("HARNESS_LLM_TEST_URL", "https://api.com/");
870        let url = llm_base_url();
871        assert_eq!(url, "https://api.com");
872        std::env::remove_var("HARNESS_LLM_TEST_URL");
873    }
874
875    #[test]
876    fn test_llm_model_default() {
877        let _env = env_guard();
878        std::env::remove_var("HARNESS_LLM_TEST_MODEL");
879        assert_eq!(llm_model(), "deepseek");
880    }
881
882    #[test]
883    fn test_llm_model_custom() {
884        let _env = env_guard();
885        std::env::set_var("HARNESS_LLM_TEST_MODEL", "gpt-4o");
886        assert_eq!(llm_model(), "gpt-4o");
887        std::env::remove_var("HARNESS_LLM_TEST_MODEL");
888    }
889
890    #[test]
891    fn test_extract_usage_partial() {
892        let _env = env_guard();
893        let json = serde_json::json!({
894            "usage": {
895                "prompt_tokens": 50
896            }
897        });
898        let usage = extract_usage(&json);
899        assert_eq!(usage.prompt_tokens, 50);
900        assert_eq!(usage.completion_tokens, 0);
901        assert_eq!(usage.total_tokens, 0);
902    }
903
904    #[test]
905    fn test_browser_headless_default() {
906        let _env = env_guard();
907        std::env::remove_var("HARNESS_BROWSER_HEADLESS");
908        assert!(crate::browser_headless());
909    }
910}