Skip to main content

everruns_contracts/
error.rs

1// Error types for the agent loop
2//
3// StoreResultExt: extension trait to replace repeated .map_err(|e| AgentLoopError::store(...))? patterns
4// json_val / from_json: helpers to replace repeated serde_json::to_value/from_value boilerplate
5
6use crate::typed_id::{AgentId, HarnessId, SessionId};
7use crate::user_facing_error::{
8    AttestationRequirement, UserFacingError, UserFacingErrorContext,
9    classify_runtime_error_message, codes as user_facing_error_codes,
10    parse_attestation_requirement,
11};
12use serde::{Serialize, de::DeserializeOwned};
13use thiserror::Error;
14
15/// Result type alias for agent loop operations
16pub type Result<T> = std::result::Result<T, AgentLoopError>;
17
18// Stable provider errors remain re-exported here for compatibility; internal
19// capability markers stay private to keep this surface focused.
20pub use crate::llm_error::{BillingPressureReason, LlmError, LlmErrorKind};
21use crate::llm_error::{RejectedProviderCapability, provider_error_code_in};
22/// Errors that can occur during agent loop execution
23#[derive(Debug, Error)]
24pub enum AgentLoopError {
25    /// LLM provider error
26    #[error("LLM error: {0}")]
27    Llm(LlmError),
28
29    /// Request too large error (context length exceeded, token limits, etc.)
30    /// Contains the original error message for logging
31    #[error("Request too large: {0}")]
32    RequestTooLarge(String),
33
34    /// Model not available (404, model not found, access denied for model)
35    /// Contains the model_id string that was requested
36    #[error("Model not available: {0}")]
37    ModelNotAvailable(String),
38
39    /// No explicit, snapshot, or system-default model could be resolved.
40    #[error("Model not configured")]
41    ModelNotConfigured,
42
43    /// Tool execution error
44    #[error("Tool execution error: {0}")]
45    ToolExecution(String),
46
47    /// Message store error
48    #[error("Message store error: {0}")]
49    MessageStore(String),
50
51    /// Event emission error
52    #[error("Event emission error: {0}")]
53    EventEmission(String),
54
55    /// Configuration error
56    #[error("Configuration error: {0}")]
57    Configuration(String),
58
59    /// Loop terminated due to max iterations
60    #[error("Max iterations ({0}) reached")]
61    MaxIterationsReached(usize),
62
63    /// Loop was cancelled
64    #[error("Loop cancelled")]
65    Cancelled,
66
67    /// No messages to process
68    #[error("No messages to process")]
69    NoMessages,
70
71    /// Agent not found
72    #[error("Agent not found: {0}")]
73    AgentNotFound(AgentId),
74
75    /// Harness not found
76    #[error("Harness not found: {0}")]
77    HarnessNotFound(HarnessId),
78
79    /// Session not found
80    #[error("Session not found: {0}")]
81    SessionNotFound(SessionId),
82
83    /// Internal error
84    #[error("Internal error: {0}")]
85    Internal(#[from] anyhow::Error),
86
87    /// Driver not registered for provider type
88    #[error(
89        "No driver registered for provider type '{0}'. Make sure the driver is registered at startup."
90    )]
91    DriverNotRegistered(String),
92}
93
94impl AgentLoopError {
95    /// Prefix provider-bound messages without changing structured identifiers.
96    pub fn with_provider(mut self, provider: &str) -> Self {
97        let prefix = format!("provider '{provider}': ");
98        match &mut self {
99            AgentLoopError::Llm(error) if !error.message.starts_with(&prefix) => {
100                error.message.insert_str(0, &prefix)
101            }
102            AgentLoopError::RequestTooLarge(message) | AgentLoopError::Configuration(message)
103                if !message.starts_with(&prefix) =>
104            {
105                message.insert_str(0, &prefix)
106            }
107            // ModelNotAvailable stores a model ID, not a free-form message.
108            _ => {}
109        }
110        self
111    }
112
113    /// Create an LLM error with no semantic kind (falls back to string
114    /// classification downstream).
115    pub fn llm(msg: impl Into<String>) -> Self {
116        AgentLoopError::Llm(LlmError::new(LlmErrorKind::Other, msg))
117    }
118
119    /// Create an LLM error with a semantic kind assigned at the driver boundary.
120    pub fn llm_kind(kind: LlmErrorKind, msg: impl Into<String>) -> Self {
121        AgentLoopError::Llm(LlmError::new(kind, msg))
122    }
123
124    /// Create an LLM error from an HTTP failure, classifying it and keeping
125    /// the status.
126    ///
127    /// This is the constructor HTTP drivers reach for: it is the one place
128    /// that both classifies (via
129    /// [`LlmErrorKind::from_provider_status`]) and preserves the status a
130    /// consumer would otherwise have to parse back out of the message.
131    pub fn llm_http(status: u16, body: &str, msg: impl Into<String>) -> Self {
132        Self::llm_http_kind(
133            LlmErrorKind::from_provider_status(status, body),
134            status,
135            body,
136            msg,
137        )
138    }
139
140    /// Create an LLM error from an HTTP failure a driver has already
141    /// classified, keeping the status and the provider's error code.
142    ///
143    /// The counterpart to [`llm_http`](Self::llm_http) for drivers whose
144    /// protocol extension classifies more precisely than status and body
145    /// alone allow.
146    pub fn llm_http_kind(
147        kind: LlmErrorKind,
148        status: u16,
149        body: &str,
150        msg: impl Into<String>,
151    ) -> Self {
152        let mut error = LlmError::new(kind, msg).with_status(status);
153        if let Some(code) = provider_error_code_in(body) {
154            error = error.with_code(code);
155        }
156        AgentLoopError::Llm(error)
157    }
158
159    /// Create a structured pre-stream provider capability rejection.
160    pub fn provider_capability_rejected(
161        capability: RejectedProviderCapability,
162        status: u16,
163        body: &str,
164        msg: impl Into<String>,
165    ) -> Self {
166        let mut error = LlmError::new(LlmErrorKind::InvalidRequest, msg)
167            .with_status(status)
168            .with_rejected_capability(capability);
169        if let Some(code) = provider_error_code_in(body) {
170            error = error.with_code(code);
171        }
172        AgentLoopError::Llm(error)
173    }
174
175    /// Whether this error rejects the specified provider capability.
176    pub fn rejected_provider_capability(&self, capability: RejectedProviderCapability) -> bool {
177        matches!(
178            self,
179            AgentLoopError::Llm(error)
180                if error.rejected_capability == Some(capability)
181        )
182    }
183
184    /// Attach the HTTP status to an LLM error that was classified elsewhere.
185    ///
186    /// No-op on non-LLM variants, whose status is implied by the variant and
187    /// reported by [`http_status`](Self::http_status).
188    #[must_use]
189    pub fn with_status(mut self, status: u16) -> Self {
190        if let AgentLoopError::Llm(error) = &mut self {
191            error.status = Some(status);
192        }
193        self
194    }
195
196    /// Attach the delay the provider asked for before another attempt.
197    #[must_use]
198    pub fn with_retry_after_secs(mut self, secs: u64) -> Self {
199        if let AgentLoopError::Llm(error) = &mut self {
200            error.retry_after_secs = Some(secs);
201        }
202        self
203    }
204
205    /// The HTTP status this failure corresponds to, if any.
206    ///
207    /// Reports the recorded status for [`Llm`](Self::Llm), and the status the
208    /// variant itself stands for where Everruns classified the failure
209    /// semantically instead of carrying one:
210    /// [`ModelNotAvailable`](Self::ModelNotAvailable) is `404` and
211    /// [`RequestTooLarge`](Self::RequestTooLarge) is `413`. Everything else is
212    /// `None` — a caller putting an API in front of Everruns picks its own
213    /// status rather than being handed a guess.
214    pub fn http_status(&self) -> Option<u16> {
215        match self {
216            AgentLoopError::Llm(error) => error.status,
217            AgentLoopError::ModelNotAvailable(_) => Some(404),
218            AgentLoopError::RequestTooLarge(_) => Some(413),
219            _ => None,
220        }
221    }
222
223    /// The provider's own machine-readable error code, if one was preserved.
224    pub fn provider_error_code(&self) -> Option<&str> {
225        match self {
226            AgentLoopError::Llm(error) => error.code.as_deref(),
227            _ => None,
228        }
229    }
230
231    /// The delay the provider asked for before another attempt, in seconds.
232    ///
233    /// Reads the recorded `Retry-After` first, then the delay
234    /// [`LlmErrorKind::BillingPressure`] carries.
235    pub fn retry_after_secs(&self) -> Option<u64> {
236        match self {
237            AgentLoopError::Llm(error) => error.retry_after_secs.or(match error.kind {
238                LlmErrorKind::BillingPressure {
239                    retry_after_secs, ..
240                } => retry_after_secs,
241                _ => None,
242            }),
243            _ => None,
244        }
245    }
246
247    /// Attach retries already consumed by a lower provider layer. The reason
248    /// loop uses this to avoid multiplying attempt budgets across layers.
249    pub fn with_retry_metadata(mut self, metadata: &crate::llm_retry::RetryMetadata) -> Self {
250        if let AgentLoopError::Llm(error) = &mut self {
251            error.retry_attempts = metadata.attempts;
252            error.retry_wait_ms = metadata.total_retry_wait.as_millis() as u64;
253            error.retry_handled = true;
254        }
255        self
256    }
257
258    /// Number of lower-layer retries already consumed by this failure.
259    pub fn llm_retry_attempts(&self) -> u32 {
260        match self {
261            AgentLoopError::Llm(error) => error.retry_attempts,
262            _ => 0,
263        }
264    }
265
266    /// Whether a lower provider layer already exhausted or rejected recovery.
267    pub fn llm_retry_handled(&self) -> bool {
268        matches!(self, AgentLoopError::Llm(error) if error.retry_handled)
269    }
270
271    /// Get the semantic LLM error kind, if this is an LLM error.
272    pub fn llm_error_kind(&self) -> Option<LlmErrorKind> {
273        match self {
274            AgentLoopError::Llm(err) => Some(err.kind),
275            _ => None,
276        }
277    }
278
279    /// Create a tool execution error
280    pub fn tool(msg: impl Into<String>) -> Self {
281        AgentLoopError::ToolExecution(msg.into())
282    }
283
284    /// Create a message store error
285    pub fn store(msg: impl Into<String>) -> Self {
286        AgentLoopError::MessageStore(msg.into())
287    }
288
289    /// Create an event emission error
290    pub fn event(msg: impl Into<String>) -> Self {
291        AgentLoopError::EventEmission(msg.into())
292    }
293
294    /// Create a configuration error
295    pub fn config(msg: impl Into<String>) -> Self {
296        AgentLoopError::Configuration(msg.into())
297    }
298
299    /// Create an agent not found error
300    pub fn agent_not_found(agent_id: AgentId) -> Self {
301        AgentLoopError::AgentNotFound(agent_id)
302    }
303
304    /// Create a harness not found error
305    pub fn harness_not_found(harness_id: HarnessId) -> Self {
306        AgentLoopError::HarnessNotFound(harness_id)
307    }
308
309    /// Create a session not found error
310    pub fn session_not_found(session_id: SessionId) -> Self {
311        AgentLoopError::SessionNotFound(session_id)
312    }
313
314    /// Create a driver not registered error
315    pub fn driver_not_registered(provider_type: impl Into<String>) -> Self {
316        AgentLoopError::DriverNotRegistered(provider_type.into())
317    }
318
319    /// Create a request too large error
320    pub fn request_too_large(msg: impl Into<String>) -> Self {
321        AgentLoopError::RequestTooLarge(msg.into())
322    }
323
324    /// Create a model not available error
325    pub fn model_not_available(model_id: impl Into<String>) -> Self {
326        AgentLoopError::ModelNotAvailable(model_id.into())
327    }
328
329    /// Create a missing-model configuration error.
330    pub fn model_not_configured() -> Self {
331        AgentLoopError::ModelNotConfigured
332    }
333
334    /// Check if this is a request-too-large error
335    pub fn is_request_too_large(&self) -> bool {
336        matches!(self, AgentLoopError::RequestTooLarge(_))
337    }
338
339    /// Check if this is a model-not-available error
340    pub fn is_model_not_available(&self) -> bool {
341        matches!(self, AgentLoopError::ModelNotAvailable(_))
342    }
343
344    /// Get the model ID if this is a model-not-available error
345    pub fn model_not_available_id(&self) -> Option<&str> {
346        match self {
347            AgentLoopError::ModelNotAvailable(id) => Some(id),
348            _ => None,
349        }
350    }
351
352    /// Check if this is a rate-limit error (semantic kind, or HTTP 429 /
353    /// rate-limit keywords for untyped errors)
354    pub fn is_rate_limited(&self) -> bool {
355        match self {
356            AgentLoopError::Llm(err) => match err.kind {
357                LlmErrorKind::RateLimited => true,
358                // A recorded status is the provider's own answer; the string
359                // scan below is the fallback for failures that carry none.
360                LlmErrorKind::Other => match err.status {
361                    Some(status) => status == 429,
362                    None => {
363                        let msg_lower = err.message.to_ascii_lowercase();
364                        msg_lower.contains("(429)")
365                            || msg_lower.contains("rate limit")
366                            || msg_lower.contains("too many requests")
367                    }
368                },
369                _ => false,
370            },
371            _ => false,
372        }
373    }
374
375    /// Check if this is an authentication/authorization error (HTTP 401/403)
376    pub fn is_auth_error(&self) -> bool {
377        match self {
378            AgentLoopError::Llm(err) => match err.kind {
379                LlmErrorKind::Authentication => true,
380                LlmErrorKind::Other => match err.status {
381                    Some(status) => status == 401 || status == 403,
382                    None => err.message.contains("(401)") || err.message.contains("(403)"),
383                },
384                _ => false,
385            },
386            _ => false,
387        }
388    }
389
390    /// Check if this is a server error (HTTP 5xx or transient provider issue)
391    pub fn is_server_error(&self) -> bool {
392        match self {
393            AgentLoopError::Llm(err) => match err.kind {
394                LlmErrorKind::Unavailable => true,
395                LlmErrorKind::Other => match err.status {
396                    Some(status) => status >= 500,
397                    None => {
398                        let msg = &err.message;
399                        msg.contains("(500)")
400                            || msg.contains("(502)")
401                            || msg.contains("(503)")
402                            || msg.contains("(504)")
403                            || msg.contains("(529)")
404                    }
405                },
406                _ => false,
407            },
408            _ => false,
409        }
410    }
411
412    /// Check whether an LLM failure is safe to retry.
413    ///
414    /// Semantic driver classification is authoritative. Untyped legacy errors
415    /// retain the message-based fallback until all drivers preserve structure.
416    pub fn is_transient_llm_error(&self) -> bool {
417        match self {
418            AgentLoopError::Llm(err) => match err.kind {
419                LlmErrorKind::RateLimited | LlmErrorKind::Unavailable => true,
420                LlmErrorKind::Authentication
421                | LlmErrorKind::QuotaExhausted
422                | LlmErrorKind::BillingPressure { .. }
423                | LlmErrorKind::AttestationRequired
424                | LlmErrorKind::MalformedResponse
425                | LlmErrorKind::InvalidRequest => false,
426                LlmErrorKind::Other => crate::llm_retry::is_transient_error_message(&err.message),
427            },
428            _ => false,
429        }
430    }
431
432    /// Check if this error is deterministic and should never be retried.
433    ///
434    /// Non-retryable errors reference data that is permanently gone (e.g. a
435    /// deleted message, a missing agent). Retrying will never succeed and only
436    /// burns attempts while keeping the workflow stuck.
437    ///
438    /// Note: the durable worker currently uses string-matching via
439    /// `is_non_retryable_task_error` because task errors arrive as strings.
440    /// This method provides the typed equivalent for callers that have access
441    /// to a structured `AgentLoopError`.
442    pub fn is_non_retryable(&self) -> bool {
443        match self {
444            // Missing data is permanent — the entity was deleted.
445            AgentLoopError::AgentNotFound(_)
446            | AgentLoopError::HarnessNotFound(_)
447            | AgentLoopError::SessionNotFound(_)
448            | AgentLoopError::NoMessages
449            | AgentLoopError::ModelNotConfigured => true,
450
451            // Config/driver errors won't self-heal within retries.
452            AgentLoopError::Configuration(_) | AgentLoopError::DriverNotRegistered(_) => true,
453
454            // MessageStore "not found" errors (deleted messages).
455            AgentLoopError::MessageStore(msg) => msg.to_ascii_lowercase().contains("not found"),
456
457            // Everything else is potentially transient.
458            _ => false,
459        }
460    }
461
462    /// Get user-facing error message based on error classification
463    pub fn user_facing_message(&self) -> String {
464        self.user_facing_error(UserFacingErrorContext::default())
465            .fallback_message()
466    }
467
468    /// Get structured user-facing error metadata based on error classification.
469    pub fn user_facing_error(&self, context: UserFacingErrorContext) -> UserFacingError {
470        match self {
471            AgentLoopError::ModelNotConfigured => {
472                UserFacingError::new(user_facing_error_codes::MODEL_NOT_CONFIGURED)
473            }
474            AgentLoopError::ModelNotAvailable(model_id) => {
475                UserFacingError::new(user_facing_error_codes::MODEL_UNAVAILABLE)
476                    .with_field("model_id", model_id)
477                    .with_optional_field("provider", context.provider)
478            }
479            AgentLoopError::RequestTooLarge(_) => {
480                UserFacingError::new(user_facing_error_codes::REQUEST_TOO_LARGE)
481                    .with_optional_field("provider", context.provider)
482                    .with_optional_field("model_id", context.model_id)
483            }
484            AgentLoopError::MaxIterationsReached(max_iterations) => {
485                UserFacingError::new(user_facing_error_codes::MAX_ITERATIONS)
486                    .with_field("max_iterations", max_iterations)
487            }
488            AgentLoopError::Llm(err) => {
489                // Prefer the semantic kind the driver assigned at the provider
490                // boundary; fall back to string classification for untyped
491                // errors so legacy paths keep working.
492                let code = match err.kind {
493                    LlmErrorKind::Authentication => {
494                        Some(user_facing_error_codes::PROVIDER_MISCONFIGURED)
495                    }
496                    LlmErrorKind::QuotaExhausted => {
497                        Some(user_facing_error_codes::PROVIDER_QUOTA_EXHAUSTED)
498                    }
499                    LlmErrorKind::BillingPressure { reason, .. } => Some(match reason {
500                        BillingPressureReason::InFlightBudgetExhausted => {
501                            user_facing_error_codes::PROVIDER_RATE_LIMITED
502                        }
503                        BillingPressureReason::InsufficientCredits => {
504                            user_facing_error_codes::PROVIDER_QUOTA_EXHAUSTED
505                        }
506                    }),
507                    LlmErrorKind::RateLimited => {
508                        Some(user_facing_error_codes::PROVIDER_RATE_LIMITED)
509                    }
510                    LlmErrorKind::Unavailable => {
511                        Some(user_facing_error_codes::PROVIDER_UNAVAILABLE)
512                    }
513                    LlmErrorKind::AttestationRequired => {
514                        Some(user_facing_error_codes::PROVIDER_ATTESTATION_REQUIRED)
515                    }
516                    LlmErrorKind::InvalidRequest
517                    | LlmErrorKind::MalformedResponse
518                    | LlmErrorKind::Other => None,
519                };
520                match code {
521                    Some(code) => {
522                        let error = UserFacingError::new(code)
523                            .with_optional_field("provider", context.provider)
524                            .with_optional_field("model_id", context.model_id);
525                        if code == user_facing_error_codes::PROVIDER_RATE_LIMITED {
526                            let retry_after = match err.kind {
527                                LlmErrorKind::BillingPressure {
528                                    retry_after_secs, ..
529                                } => retry_after_secs,
530                                _ => context.retry_after,
531                            };
532                            error.with_optional_field("retry_after", retry_after)
533                        } else if code == user_facing_error_codes::PROVIDER_ATTESTATION_REQUIRED {
534                            // The attestation variant carries no payload, so
535                            // the confirmations and URL are read from the raw
536                            // body the driver retained in `message`.
537                            parse_attestation_requirement(&err.message)
538                                .unwrap_or_else(AttestationRequirement::fallback)
539                                .apply_fields(error)
540                        } else {
541                            error
542                        }
543                    }
544                    None => classify_runtime_error_message(&err.message, &context),
545                }
546            }
547            _ => UserFacingError::new(user_facing_error_codes::PROCESSING_ERROR)
548                .with_optional_field("provider", context.provider)
549                .with_optional_field("model_id", context.model_id),
550        }
551    }
552}
553
554// ============================================================================
555// Store Result Extension Trait
556// ============================================================================
557
558/// Extension trait that converts any `Result<T, E: Display>` into `Result<T, AgentLoopError>`
559/// via `AgentLoopError::store(e.to_string())`.
560///
561/// Replaces the boilerplate pattern:
562/// ```ignore
563/// .map_err(|e| AgentLoopError::store(e.to_string()))?
564/// ```
565/// with:
566/// ```ignore
567/// .store_err()?
568/// ```
569pub trait StoreResultExt<T> {
570    fn store_err(self) -> Result<T>;
571}
572
573impl<T, E: std::fmt::Display> StoreResultExt<T> for std::result::Result<T, E> {
574    fn store_err(self) -> Result<T> {
575        self.map_err(|e| AgentLoopError::store(e.to_string()))
576    }
577}
578
579// ============================================================================
580// SessionFileSystem error classification (EVE-645)
581// ============================================================================
582
583/// Typed classification of a `SessionFileSystem` failure.
584///
585/// The file-system tools (`integrations/filesystem/src/lib.rs`) decide
586/// whether a failure is a *tool error* (surfaced to the agent verbatim — bad
587/// input it can correct) or an *internal error* (logged, generic copy). They
588/// previously made that call with `msg.contains("readonly")` / `"is a
589/// directory"` / `"not found"` style sniffs against the stringified error.
590///
591/// The `SessionFileSystem` trait returns `anyhow::Result<T>` and has 10+
592/// implementors across crates, so widening the trait's error type is out of
593/// scope. Instead, [`classify_fs_error`] gives a single typed seam: it
594/// downcasts to [`FileSystemError`] when an implementor opts in, and otherwise
595/// falls back to the legacy substring heuristics in one place. Implementors can
596/// migrate to returning `FileSystemError` (via `anyhow::Error::new`)
597/// incrementally without changing behavior.
598#[derive(Debug, Clone, Copy, PartialEq, Eq)]
599pub enum FileSystemErrorClass {
600    /// The target (or a path component) does not exist.
601    NotFound,
602    /// The target is read-only and cannot be written or deleted.
603    ReadOnly,
604    /// Expected a file but the path is a directory.
605    IsADirectory,
606    /// Expected a directory but the path is not one.
607    NotADirectory,
608    /// A non-recursive delete refused a non-empty directory.
609    NotEmpty,
610    /// No recognized client-correctable condition; treat as internal.
611    Other,
612}
613
614/// Typed `SessionFileSystem` error. Implementors may return this (wrapped in
615/// `anyhow::Error`) so [`classify_fs_error`] resolves the class without string
616/// matching. Each variant carries the human-facing message so the file tools
617/// can keep surfacing the same text to the agent.
618#[derive(Debug, Error)]
619pub enum FileSystemError {
620    #[error("{0}")]
621    NotFound(String),
622    #[error("{0}")]
623    ReadOnly(String),
624    #[error("{0}")]
625    IsADirectory(String),
626    #[error("{0}")]
627    NotADirectory(String),
628    #[error("{0}")]
629    NotEmpty(String),
630}
631
632impl FileSystemError {
633    fn class(&self) -> FileSystemErrorClass {
634        match self {
635            FileSystemError::NotFound(_) => FileSystemErrorClass::NotFound,
636            FileSystemError::ReadOnly(_) => FileSystemErrorClass::ReadOnly,
637            FileSystemError::IsADirectory(_) => FileSystemErrorClass::IsADirectory,
638            FileSystemError::NotADirectory(_) => FileSystemErrorClass::NotADirectory,
639            FileSystemError::NotEmpty(_) => FileSystemErrorClass::NotEmpty,
640        }
641    }
642}
643
644/// Classify a `SessionFileSystem` failure into a [`FileSystemErrorClass`].
645///
646/// Prefers a typed [`FileSystemError`] in the error chain; falls back to the
647/// legacy substring heuristics (the single remaining place they live) so
648/// untyped implementors keep their current routing. Behavior is identical to
649/// the previous inline `msg.contains(...)` checks in `file_system.rs`:
650/// "readonly" and "is a directory" mark client-correctable write failures,
651/// "not found" / "not a directory" mark client-correctable read failures, and
652/// "not empty" / "recursive" mark client-correctable delete failures.
653pub fn classify_fs_error<E>(err: &E) -> FileSystemErrorClass
654where
655    E: std::error::Error + 'static,
656{
657    // Prefer a typed FileSystemError anywhere in the source chain so an
658    // implementor that opts in is classified without string matching. Works
659    // whether the error is a bare FileSystemError or wrapped (e.g. inside
660    // `AgentLoopError::Internal(anyhow!(FileSystemError::..))`).
661    let mut source: Option<&(dyn std::error::Error + 'static)> = Some(err);
662    while let Some(current) = source {
663        if let Some(typed) = current.downcast_ref::<FileSystemError>() {
664            return typed.class();
665        }
666        source = current.source();
667    }
668
669    let msg = err.to_string();
670    // Note: real-disk backends emit "read-only" (hyphenated); the legacy check
671    // only matched "readonly", so we preserve that exact behavior rather than
672    // silently widening it.
673    if msg.contains("readonly") {
674        FileSystemErrorClass::ReadOnly
675    } else if msg.contains("is a directory") {
676        FileSystemErrorClass::IsADirectory
677    } else if msg.contains("not a directory") {
678        FileSystemErrorClass::NotADirectory
679    } else if msg.contains("not empty") || msg.contains("recursive") {
680        FileSystemErrorClass::NotEmpty
681    } else if msg.contains("not found") {
682        FileSystemErrorClass::NotFound
683    } else {
684        FileSystemErrorClass::Other
685    }
686}
687
688// ============================================================================
689// JSON Helpers
690// ============================================================================
691
692/// Convert a serializable value to `serde_json::Value`, falling back to `Value::Null` on error.
693///
694/// Replaces the boilerplate pattern:
695/// ```ignore
696/// serde_json::to_value(&x).unwrap_or_default()
697/// ```
698pub fn json_val<T: Serialize>(value: &T) -> serde_json::Value {
699    serde_json::to_value(value).unwrap_or_default()
700}
701
702/// Deserialize a `serde_json::Value` into `T`, falling back to `T::default()` on error.
703///
704/// Replaces the boilerplate pattern:
705/// ```ignore
706/// serde_json::from_value(v).unwrap_or_default()
707/// ```
708pub fn from_json<T: DeserializeOwned + Default>(value: serde_json::Value) -> T {
709    serde_json::from_value(value).unwrap_or_default()
710}
711
712#[cfg(test)]
713mod tests {
714    use super::*;
715    use serde_json::json;
716
717    #[test]
718    fn filesystem_typed_errors_win_over_conflicting_messages_and_wrappers() {
719        for (error, expected) in [
720            (
721                FileSystemError::NotFound("readonly".into()),
722                FileSystemErrorClass::NotFound,
723            ),
724            (
725                FileSystemError::ReadOnly("not found".into()),
726                FileSystemErrorClass::ReadOnly,
727            ),
728            (
729                FileSystemError::IsADirectory("not empty".into()),
730                FileSystemErrorClass::IsADirectory,
731            ),
732            (
733                FileSystemError::NotADirectory("is a directory".into()),
734                FileSystemErrorClass::NotADirectory,
735            ),
736            (
737                FileSystemError::NotEmpty("not found".into()),
738                FileSystemErrorClass::NotEmpty,
739            ),
740        ] {
741            assert_eq!(classify_fs_error(&error), expected);
742            let wrapped = AgentLoopError::Internal(
743                anyhow::Error::new(error).context("readonly outer failure"),
744            );
745            assert_eq!(classify_fs_error(&wrapped), expected);
746        }
747    }
748
749    #[test]
750    fn filesystem_legacy_messages_preserve_routing_and_case_boundaries() {
751        for (message, expected) in [
752            (
753                "Cannot modify readonly file: /a",
754                FileSystemErrorClass::ReadOnly,
755            ),
756            (
757                "Cannot delete readonly file: /a",
758                FileSystemErrorClass::ReadOnly,
759            ),
760            (
761                "write target is a directory: /a",
762                FileSystemErrorClass::IsADirectory,
763            ),
764            (
765                "Path is not a directory: /a",
766                FileSystemErrorClass::NotADirectory,
767            ),
768            (
769                "workspace root is not a directory: /a",
770                FileSystemErrorClass::NotADirectory,
771            ),
772            ("Directory not found: /a", FileSystemErrorClass::NotFound),
773            (
774                "Directory is not empty. Use recursive=true to delete",
775                FileSystemErrorClass::NotEmpty,
776            ),
777            (
778                "Cannot delete root directory without recursive flag",
779                FileSystemErrorClass::NotEmpty,
780            ),
781            (
782                "recursive delete failed for /a: io",
783                FileSystemErrorClass::NotEmpty,
784            ),
785            ("readonly file not found", FileSystemErrorClass::ReadOnly),
786            ("file is read-only: /a", FileSystemErrorClass::Other),
787            ("NOT FOUND", FileSystemErrorClass::Other),
788            ("disk full", FileSystemErrorClass::Other),
789        ] {
790            assert_eq!(
791                classify_fs_error(&AgentLoopError::store(message)),
792                expected,
793                "{message}"
794            );
795        }
796    }
797
798    #[test]
799    fn typed_request_and_model_errors_preserve_identity_and_safe_user_payload() {
800        let context = || {
801            UserFacingErrorContext::default()
802                .with_provider("provider")
803                .with_model_id("context-model")
804                .with_retry_after(9)
805        };
806        let request = AgentLoopError::request_too_large("private payload");
807        assert_eq!(request.to_string(), "Request too large: private payload");
808        assert!(request.is_request_too_large());
809        assert!(!request.is_model_not_available());
810        assert_eq!(request.model_not_available_id(), None);
811        assert_eq!(
812            serde_json::to_value(request.user_facing_error(context())).unwrap(),
813            json!({"code":"request_too_large","fields":{"provider":"provider","model_id":"context-model"}})
814        );
815        assert_eq!(
816            request.user_facing_message(),
817            "The conversation has become too long for the model to process. Please start a new session or reduce the context size."
818        );
819        let model = AgentLoopError::model_not_available("gpt-99")
820            .with_provider("custom")
821            .with_provider("custom");
822        assert!(!model.is_request_too_large());
823        assert!(model.is_model_not_available());
824        assert_eq!(model.model_not_available_id(), Some("gpt-99"));
825        assert_eq!(model.to_string(), "Model not available: gpt-99");
826        assert_eq!(
827            model.user_facing_message(),
828            "The model `gpt-99` is not available. It may have been removed, renamed, or your API key may not have access to it. Please select a different model."
829        );
830        assert_eq!(
831            serde_json::to_value(model.user_facing_error(context())).unwrap(),
832            json!({"code":"model_unavailable","fields":{"provider":"provider","model_id":"gpt-99"}})
833        );
834        for other in [
835            AgentLoopError::llm("Request too large: Model not available: gpt-99"),
836            AgentLoopError::tool("failed"),
837            AgentLoopError::Cancelled,
838        ] {
839            assert!(!other.is_request_too_large());
840            assert!(!other.is_model_not_available());
841            assert_eq!(other.model_not_available_id(), None);
842        }
843    }
844
845    #[test]
846    fn semantic_kinds_override_conflicting_text_for_predicates_and_payloads() {
847        for (kind, message, predicates, code) in [
848            (
849                LlmErrorKind::Authentication,
850                "(429) rate limit (503)",
851                (false, true, false, false),
852                "provider_misconfigured",
853            ),
854            (
855                LlmErrorKind::QuotaExhausted,
856                "(401) (429) rate limit (503)",
857                (false, false, false, false),
858                "provider_quota_exhausted",
859            ),
860            (
861                LlmErrorKind::RateLimited,
862                "(401) (503) insufficient_quota",
863                (true, false, false, true),
864                "provider_rate_limited",
865            ),
866            (
867                LlmErrorKind::Unavailable,
868                "(401) (429) insufficient_quota",
869                (false, false, true, true),
870                "provider_unavailable",
871            ),
872            (
873                LlmErrorKind::InvalidRequest,
874                "opaque private failure",
875                (false, false, false, false),
876                "processing_error",
877            ),
878        ] {
879            let error = AgentLoopError::llm_kind(kind, message);
880            assert_eq!(error.llm_error_kind(), Some(kind));
881            assert_eq!(
882                (
883                    error.is_rate_limited(),
884                    error.is_auth_error(),
885                    error.is_server_error(),
886                    error.is_transient_llm_error()
887                ),
888                predicates,
889                "{kind:?}"
890            );
891            let mut fields = json!({"provider":"provider","model_id":"model"});
892            if kind == LlmErrorKind::RateLimited {
893                fields["retry_after"] = json!(12);
894            }
895            assert_eq!(
896                serde_json::to_value(
897                    error.user_facing_error(
898                        UserFacingErrorContext::default()
899                            .with_provider("provider")
900                            .with_model_id("model")
901                            .with_retry_after(12)
902                    )
903                )
904                .unwrap(),
905                json!({"code":code,"fields":fields})
906            );
907        }
908    }
909
910    #[test]
911    fn legacy_predicates_and_user_copy_use_independent_literal_cases() {
912        for (message, expected, copy) in [
913            (
914                "Anthropic API error (429): rate limit exceeded",
915                (true, false, false),
916                "Rate limited by the AI provider. Please wait a moment.",
917            ),
918            (
919                "Rate limit exceeded (after 2 retries)",
920                (true, false, false),
921                "Rate limited by the AI provider. Please wait a moment.",
922            ),
923            (
924                "too many requests",
925                (true, false, false),
926                "Rate limited by the AI provider. Please wait a moment.",
927            ),
928            (
929                "Anthropic API error (401): invalid api key",
930                (false, true, false),
931                "There is a misconfiguration with the AI provider. Please contact support.",
932            ),
933            (
934                "OpenAI API error (403): forbidden",
935                (false, true, false),
936                "There is a misconfiguration with the AI provider. Please contact support.",
937            ),
938            (
939                "Anthropic API error (500): internal server error",
940                (false, false, true),
941                "The AI provider is experiencing issues. Please try again shortly.",
942            ),
943            (
944                "OpenAI API error (503): service unavailable",
945                (false, false, true),
946                "The AI provider is experiencing issues. Please try again shortly.",
947            ),
948            (
949                "Failed to send request: connection refused",
950                (false, false, false),
951                "I encountered an error while processing your request. Please try again later.",
952            ),
953        ] {
954            let error = AgentLoopError::llm(message);
955            assert_eq!(
956                (
957                    error.is_rate_limited(),
958                    error.is_auth_error(),
959                    error.is_server_error()
960                ),
961                expected,
962                "{message}"
963            );
964            assert_eq!(error.user_facing_message(), copy, "{message}");
965        }
966        for status in [502, 504, 529] {
967            assert!(AgentLoopError::llm(format!("error ({status})")).is_server_error());
968        }
969        let non_llm = AgentLoopError::tool("(401) (429) (503) rate limit");
970        assert_eq!(
971            (
972                non_llm.is_rate_limited(),
973                non_llm.is_auth_error(),
974                non_llm.is_server_error(),
975                non_llm.is_transient_llm_error()
976            ),
977            (false, false, false, false)
978        );
979    }
980
981    #[test]
982    fn provider_status_classification_covers_boundaries_and_quota_precedence() {
983        for (status, expected) in [
984            (200, LlmErrorKind::Other),
985            (399, LlmErrorKind::Other),
986            (400, LlmErrorKind::InvalidRequest),
987            (401, LlmErrorKind::Authentication),
988            (403, LlmErrorKind::Authentication),
989            (404, LlmErrorKind::InvalidRequest),
990            (408, LlmErrorKind::Unavailable),
991            (409, LlmErrorKind::Unavailable),
992            (429, LlmErrorKind::RateLimited),
993            (499, LlmErrorKind::InvalidRequest),
994            (500, LlmErrorKind::Unavailable),
995            (501, LlmErrorKind::Other),
996            (502, LlmErrorKind::Unavailable),
997            (503, LlmErrorKind::Unavailable),
998            (529, LlmErrorKind::Unavailable),
999            (599, LlmErrorKind::Unavailable),
1000            (600, LlmErrorKind::Other),
1001        ] {
1002            assert_eq!(
1003                LlmErrorKind::from_provider_status(status, "opaque"),
1004                expected,
1005                "{status}"
1006            );
1007        }
1008        for message in [
1009            r#"{"error":{"type":"insufficient_quota"}}"#,
1010            r#"{"error":{"code":"credit_balance_exhausted"}}"#,
1011            r#"{"error":{"type":"usage_limit_reached"}}"#,
1012            "Your credit balance is too low to access the Anthropic API.",
1013        ] {
1014            for status in [400, 401, 429, 503] {
1015                assert_eq!(
1016                    LlmErrorKind::from_provider_status(status, message),
1017                    LlmErrorKind::QuotaExhausted,
1018                    "{status}: {message}"
1019                );
1020            }
1021        }
1022    }
1023
1024    /// The canonical OpenRouter refusal from EVE-952, verbatim off the wire.
1025    const ATTESTATION_BODY: &str = r#"{"error":{"message":"This model requires you to complete the following before use: 18+ age confirmation. Confirm at https://openrouter.ai/settings/preferences.","code":403,"metadata":{"missing_attestation_types":["age_18plus"],"routing_funnel":[{"step":"Initial Endpoints","endpoint_count":1}],"failed_routing_step":"Gate Endpoints with Attestations"}}}"#;
1026
1027    #[test]
1028    fn attestation_gate_is_classified_apart_from_other_403s() {
1029        assert_eq!(
1030            LlmErrorKind::from_provider_status(403, ATTESTATION_BODY),
1031            LlmErrorKind::AttestationRequired
1032        );
1033        // Status is not the signal: the same body under another status still
1034        // names the gate, and a 403 without one stays an auth failure.
1035        assert_eq!(
1036            LlmErrorKind::from_provider_status(429, ATTESTATION_BODY),
1037            LlmErrorKind::AttestationRequired
1038        );
1039        for body in [
1040            r#"{"error":{"message":"Invalid credentials","code":403}}"#,
1041            r#"{"error":{"message":"Insufficient credits","code":403,"metadata":{"routing_funnel":[]}}}"#,
1042            "opaque",
1043        ] {
1044            assert_ne!(
1045                LlmErrorKind::from_provider_status(403, body),
1046                LlmErrorKind::AttestationRequired,
1047                "{body}"
1048            );
1049        }
1050        // Exhausted billing keeps precedence over the gate check.
1051        assert_eq!(
1052            LlmErrorKind::from_provider_status(
1053                403,
1054                r#"{"error":{"message":"insufficient_quota; requires you to complete the following before use"}}"#
1055            ),
1056            LlmErrorKind::QuotaExhausted
1057        );
1058    }
1059
1060    #[test]
1061    fn attestation_gate_reaches_the_reader_with_the_types_and_the_confirm_url() {
1062        let error = AgentLoopError::llm_kind(
1063            LlmErrorKind::AttestationRequired,
1064            format!("OpenAI Responses API error (403): {ATTESTATION_BODY}"),
1065        )
1066        .with_provider("openrouter");
1067        // Not a credential problem and never worth retrying.
1068        assert!(!error.is_auth_error());
1069        assert!(!error.is_transient_llm_error());
1070        assert_eq!(
1071            serde_json::to_value(
1072                error.user_facing_error(
1073                    UserFacingErrorContext::default()
1074                        .with_provider("openrouter")
1075                        .with_model_id("meta/muse-spark-1.3-contributor")
1076                )
1077            )
1078            .unwrap(),
1079            json!({
1080                "code": "provider_attestation_required",
1081                "fields": {
1082                    "provider": "openrouter",
1083                    "model_id": "meta/muse-spark-1.3-contributor",
1084                    "missing_types": ["age_18plus"],
1085                    "confirm_url": "https://openrouter.ai/settings/preferences",
1086                }
1087            })
1088        );
1089        assert_eq!(
1090            error.user_facing_message(),
1091            "The AI provider account has not completed a confirmation this model requires (age_18plus). Complete it at https://openrouter.ai/settings/preferences, then try again."
1092        );
1093    }
1094
1095    #[test]
1096    fn untyped_attestation_bodies_still_route_off_the_403_misconfiguration_copy() {
1097        // Legacy/untyped errors reach the string classifier instead; it must
1098        // reach the same code rather than "contact support".
1099        let error = AgentLoopError::llm(format!(
1100            "provider 'openrouter': OpenAI Responses API error (403): {ATTESTATION_BODY}"
1101        ));
1102        assert_eq!(
1103            error
1104                .user_facing_error(UserFacingErrorContext::default())
1105                .code,
1106            "provider_attestation_required"
1107        );
1108    }
1109
1110    #[test]
1111    fn attestation_parsing_covers_multiple_types_escaped_bodies_and_a_missing_url() {
1112        let requirement = |body: &str| {
1113            parse_attestation_requirement(body).unwrap_or_else(|| panic!("no gate in {body}"))
1114        };
1115
1116        // Multiple gates, in payload order.
1117        let multiple = requirement(
1118            r#"{"error":{"message":"This model requires you to complete the following before use: 18+ age confirmation and identity verification. Confirm at https://openrouter.ai/settings/preferences.","metadata":{"missing_attestation_types":["age_18plus","identity_verified"]}}}"#,
1119        );
1120        assert_eq!(multiple.missing_types, ["age_18plus", "identity_verified"]);
1121        assert_eq!(
1122            multiple.confirm_url,
1123            "https://openrouter.ai/settings/preferences"
1124        );
1125
1126        // JSON-escaped body (a provider error nested in another envelope).
1127        let escaped = requirement(
1128            r#"{"detail":"{\"error\":{\"message\":\"This model requires you to complete the following before use: 18+ age confirmation. Confirm at https:\/\/openrouter.ai\/settings\/gates.\",\"metadata\":{\"missing_attestation_types\":[\"age_18plus\"]}}}"}"#,
1129        );
1130        assert_eq!(escaped.missing_types, ["age_18plus"]);
1131        assert_eq!(escaped.confirm_url, "https://openrouter.ai/settings/gates");
1132
1133        // No URL in the message: fall back rather than leave the reader with
1134        // nowhere to go.
1135        let no_url = requirement(
1136            r#"{"error":{"message":"This model requires you to complete the following before use: 18+ age confirmation.","metadata":{"missing_attestation_types":["age_18plus"]}}}"#,
1137        );
1138        assert_eq!(
1139            no_url.confirm_url,
1140            "https://openrouter.ai/settings/preferences"
1141        );
1142
1143        // The gate sentence alone is enough; the metadata block is optional.
1144        let sentence_only = requirement(
1145            "This model requires you to complete the following before use: 18+ age confirmation. Confirm at https://openrouter.ai/settings/preferences",
1146        );
1147        assert!(sentence_only.missing_types.is_empty());
1148        assert_eq!(
1149            sentence_only.confirm_url,
1150            "https://openrouter.ai/settings/preferences"
1151        );
1152        // With nothing parsed, the message drops the list rather than
1153        // rendering an empty one.
1154        assert_eq!(
1155            AgentLoopError::llm_kind(
1156                LlmErrorKind::AttestationRequired,
1157                "This model requires you to complete the following before use: a confirmation."
1158            )
1159            .user_facing_message(),
1160            "The AI provider account has not completed a confirmation this model requires. Complete it at https://openrouter.ai/settings/preferences, then try again."
1161        );
1162
1163        // A URL in the driver's own prefix is not mistaken for the gate page.
1164        assert_eq!(
1165            requirement(&format!(
1166                "POST https://openrouter.ai/api/v1/responses failed: {ATTESTATION_BODY}"
1167            ))
1168            .confirm_url,
1169            "https://openrouter.ai/settings/preferences"
1170        );
1171
1172        for body in [
1173            r#"{"error":{"message":"Invalid credentials"}}"#,
1174            r#"{"error":{"metadata":{"missing_attestation_types":[]}}}"#,
1175            "",
1176        ] {
1177            assert!(parse_attestation_requirement(body).is_none(), "{body}");
1178        }
1179    }
1180
1181    #[test]
1182    fn a_hostile_attestation_payload_cannot_choose_how_much_reaches_the_viewer() {
1183        let types = (0..40)
1184            .map(|index| format!(r#""gate_{index}""#))
1185            .collect::<Vec<_>>()
1186            .join(",");
1187        let long_type = "x".repeat(65);
1188        let long_url = format!("https://evil.example/{}", "a".repeat(400));
1189        let requirement = parse_attestation_requirement(&format!(
1190            r#"{{"error":{{"message":"This model requires you to complete the following before use: gates. Confirm at {long_url}","metadata":{{"missing_attestation_types":["{long_type}",{types}]}}}}}}"#
1191        ))
1192        .expect("gate recognized");
1193
1194        // Over-long entries are dropped, not truncated, and the list is capped.
1195        assert_eq!(requirement.missing_types.len(), 8);
1196        assert_eq!(requirement.missing_types[0], "gate_0");
1197        // An over-long URL falls back rather than shipping a 400-char link.
1198        assert_eq!(
1199            requirement.confirm_url,
1200            "https://openrouter.ai/settings/preferences"
1201        );
1202
1203        // Non-http(s) schemes never become the confirmation link.
1204        for scheme in [
1205            "javascript:alert(1)",
1206            "data:text/html,<script>",
1207            "file:///etc/passwd",
1208        ] {
1209            assert_eq!(
1210                parse_attestation_requirement(&format!(
1211                    "This model requires you to complete the following before use: a gate. Confirm at {scheme}"
1212                ))
1213                .expect("gate recognized")
1214                .confirm_url,
1215                "https://openrouter.ai/settings/preferences",
1216                "{scheme}"
1217            );
1218        }
1219    }
1220
1221    #[test]
1222    fn provider_text_classification_uses_independent_keywords_and_precedence() {
1223        for (message, expected) in [
1224            ("ThrottlingException", LlmErrorKind::RateLimited),
1225            ("TooManyRequestsException", LlmErrorKind::RateLimited),
1226            ("RATE LIMIT", LlmErrorKind::RateLimited),
1227            ("too many requests", LlmErrorKind::RateLimited),
1228            ("AccessDeniedException", LlmErrorKind::Authentication),
1229            ("UnrecognizedClientException", LlmErrorKind::Authentication),
1230            ("ExpiredTokenException", LlmErrorKind::Authentication),
1231            ("InvalidSignatureException", LlmErrorKind::Authentication),
1232            ("unauthorized", LlmErrorKind::Authentication),
1233            ("ServiceUnavailableException", LlmErrorKind::Unavailable),
1234            ("service unavailable", LlmErrorKind::Unavailable),
1235            ("InternalServerException", LlmErrorKind::Unavailable),
1236            ("ModelNotReadyException", LlmErrorKind::Unavailable),
1237            (
1238                "usage_limit_reached; resets_at=1783767823; throttlingexception",
1239                LlmErrorKind::QuotaExhausted,
1240            ),
1241            ("something else entirely", LlmErrorKind::Other),
1242        ] {
1243            assert_eq!(
1244                LlmErrorKind::from_error_text(message),
1245                expected,
1246                "{message}"
1247            );
1248        }
1249    }
1250
1251    #[test]
1252    fn provider_prefix_preserves_kind_and_retry_metadata_without_duplication() {
1253        let metadata = crate::llm_retry::RetryMetadata {
1254            attempts: 2,
1255            total_retry_wait: std::time::Duration::from_millis(1234),
1256            ..Default::default()
1257        };
1258        let error = AgentLoopError::llm_kind(LlmErrorKind::Unavailable, "network failure")
1259            .with_retry_metadata(&metadata)
1260            .with_provider("custom")
1261            .with_provider("custom");
1262        assert_eq!(error.llm_retry_attempts(), 2);
1263        assert!(error.llm_retry_handled());
1264        let AgentLoopError::Llm(error) = error else {
1265            panic!("lost LLM variant")
1266        };
1267        assert_eq!(
1268            serde_json::to_value(error).unwrap(),
1269            json!({"kind":"unavailable","message":"provider 'custom': network failure","retry_attempts":2,"retry_wait_ms":1234,"retry_handled":true})
1270        );
1271        let legacy: LlmError =
1272            serde_json::from_value(json!({"kind":"other","message":"legacy"})).unwrap();
1273        assert_eq!(
1274            serde_json::to_value(legacy).unwrap(),
1275            json!({"kind":"other","message":"legacy","retry_attempts":0,"retry_wait_ms":0,"retry_handled":false})
1276        );
1277        let non_llm = AgentLoopError::Cancelled
1278            .with_retry_metadata(&metadata)
1279            .with_provider("custom");
1280        assert!(matches!(non_llm, AgentLoopError::Cancelled));
1281        assert_eq!(non_llm.llm_retry_attempts(), 0);
1282        assert!(!non_llm.llm_retry_handled());
1283    }
1284
1285    #[test]
1286    fn billing_pressure_preserves_typed_payload_and_safe_user_fields() {
1287        let kind = LlmErrorKind::BillingPressure {
1288            reason: BillingPressureReason::InFlightBudgetExhausted,
1289            retry_after_secs: Some(120),
1290        };
1291        let error = AgentLoopError::llm_kind(kind, "private provider body");
1292        assert_eq!(error.llm_error_kind(), Some(kind));
1293        assert!(!error.is_transient_llm_error());
1294        assert_eq!(
1295            serde_json::to_value(
1296                error.user_facing_error(
1297                    UserFacingErrorContext::default()
1298                        .with_provider("openrouter")
1299                        .with_model_id("vendor/model")
1300                )
1301            )
1302            .unwrap(),
1303            json!({
1304                "code": "provider_rate_limited",
1305                "fields": {
1306                    "provider": "openrouter",
1307                    "model_id": "vendor/model",
1308                    "retry_after": 120,
1309                }
1310            })
1311        );
1312        let AgentLoopError::Llm(error) = error else {
1313            panic!("lost LLM variant")
1314        };
1315        assert_eq!(
1316            serde_json::to_value(error.kind).unwrap(),
1317            json!({
1318                "billing_pressure": {
1319                    "reason": "in_flight_budget_exhausted",
1320                    "retry_after_secs": 120,
1321                }
1322            })
1323        );
1324
1325        let exhausted = AgentLoopError::llm_kind(
1326            LlmErrorKind::BillingPressure {
1327                reason: BillingPressureReason::InsufficientCredits,
1328                retry_after_secs: None,
1329            },
1330            "private provider body",
1331        );
1332        assert_eq!(
1333            exhausted
1334                .user_facing_error(UserFacingErrorContext::default())
1335                .code,
1336            "provider_quota_exhausted"
1337        );
1338    }
1339
1340    #[test]
1341    fn missing_model_and_iteration_limits_have_complete_safe_payloads() {
1342        let missing = AgentLoopError::model_not_configured();
1343        assert!(missing.is_non_retryable());
1344        assert_eq!(
1345            missing.user_facing_message(),
1346            "No model is configured for this chat. Choose a model or configure a default model, then try again."
1347        );
1348        assert_eq!(
1349            serde_json::to_value(missing.user_facing_error(UserFacingErrorContext::default()))
1350                .unwrap(),
1351            json!({"code":"model_not_configured"})
1352        );
1353        assert_eq!(
1354            serde_json::to_value(
1355                AgentLoopError::MaxIterationsReached(7)
1356                    .user_facing_error(UserFacingErrorContext::default())
1357            )
1358            .unwrap(),
1359            json!({"code":"max_iterations","fields":{"max_iterations":7}})
1360        );
1361    }
1362
1363    #[test]
1364    fn store_adapter_preserves_success_and_exact_error_variant_and_message() {
1365        let success: std::result::Result<Vec<String>, String> =
1366            Ok(vec!["first".into(), "second".into()]);
1367        assert_eq!(success.store_err().unwrap(), ["first", "second"]);
1368        let failure: std::result::Result<(), std::io::Error> =
1369            Err(std::io::Error::other("db unavailable"));
1370        let error = failure.store_err().unwrap_err();
1371        assert_eq!(error.to_string(), "Message store error: db unavailable");
1372        assert!(matches!(error,AgentLoopError::MessageStore(message) if message=="db unavailable"));
1373    }
1374
1375    #[test]
1376    fn json_helpers_preserve_structures_and_apply_documented_error_defaults() {
1377        assert_eq!(json_val(&vec![1, 2, 3]), json!([1, 2, 3]));
1378        assert_eq!(from_json::<Vec<String>>(json!(["a", "b"])), ["a", "b"]);
1379        assert_eq!(from_json::<i32>(json!("not a number")), 0);
1380        struct Fails;
1381        impl Serialize for Fails {
1382            fn serialize<S: serde::Serializer>(
1383                &self,
1384                _: S,
1385            ) -> std::result::Result<S::Ok, S::Error> {
1386                Err(serde::ser::Error::custom("synthetic serialization failure"))
1387            }
1388        }
1389        assert_eq!(json_val(&Fails), serde_json::Value::Null);
1390    }
1391
1392    #[test]
1393    fn llm_http_records_the_status_and_provider_code() {
1394        let body =
1395            r#"{"error":{"message":"no","code":"model_not_found","type":"invalid_request_error"}}"#;
1396        let error = AgentLoopError::llm_http(404, body, "OpenAI API error (404)");
1397        assert_eq!(error.http_status(), Some(404));
1398        assert_eq!(error.provider_error_code(), Some("model_not_found"));
1399    }
1400
1401    #[test]
1402    fn llm_http_falls_back_to_the_anthropic_error_type() {
1403        let body = r#"{"error":{"type":"overloaded_error","message":"busy"}}"#;
1404        let error = AgentLoopError::llm_http(529, body, "Anthropic API error (529)");
1405        assert_eq!(error.provider_error_code(), Some("overloaded_error"));
1406        assert_eq!(error.llm_error_kind(), Some(LlmErrorKind::Unavailable));
1407    }
1408
1409    #[test]
1410    fn a_provider_sized_error_body_is_not_parsed_and_a_long_code_is_not_kept() {
1411        // TM-DOS-038: the body is provider-controlled and nothing upstream
1412        // bounds it, so neither the parse nor the retained string may be
1413        // sized by the provider.
1414        let padding = "x".repeat(70 * 1024);
1415        let huge = format!(r#"{{"error":{{"code":"rate_limit","pad":"{padding}"}}}}"#);
1416        let error = AgentLoopError::llm_http(429, &huge, "provider refused");
1417        assert_eq!(error.provider_error_code(), None);
1418        // The status still classifies it, so nothing is lost but the code.
1419        assert!(error.is_rate_limited());
1420
1421        let long_code = "c".repeat(200);
1422        let body = format!(r#"{{"error":{{"code":"{long_code}"}}}}"#);
1423        assert_eq!(
1424            AgentLoopError::llm_http(400, &body, "boom").provider_error_code(),
1425            None,
1426            "a code longer than any real one is not a code"
1427        );
1428    }
1429
1430    #[test]
1431    fn a_non_json_body_yields_no_provider_code() {
1432        let error = AgentLoopError::llm_http(503, "upstream is down", "boom");
1433        assert_eq!(error.http_status(), Some(503));
1434        assert_eq!(error.provider_error_code(), None);
1435    }
1436
1437    #[test]
1438    fn semantic_variants_report_the_status_they_stand_for() {
1439        assert_eq!(
1440            AgentLoopError::model_not_available("gpt-5").http_status(),
1441            Some(404)
1442        );
1443        assert_eq!(
1444            AgentLoopError::request_too_large("too big").http_status(),
1445            Some(413)
1446        );
1447        assert_eq!(AgentLoopError::Cancelled.http_status(), None);
1448    }
1449
1450    #[test]
1451    fn a_recorded_status_classifies_an_untyped_failure() {
1452        // The message carries no "(429)" marker, so only the recorded status
1453        // can answer this — the case that forced embedders to scrape strings.
1454        let error = AgentLoopError::llm_http(429, "slow down", "provider refused the request");
1455        assert!(error.is_rate_limited());
1456        assert_eq!(error.retry_after_secs(), None);
1457        assert!(!AgentLoopError::llm("provider refused the request").is_rate_limited());
1458    }
1459
1460    #[test]
1461    fn retry_after_prefers_the_recorded_delay_over_the_billing_hint() {
1462        let error = AgentLoopError::llm_kind(
1463            LlmErrorKind::BillingPressure {
1464                reason: BillingPressureReason::InFlightBudgetExhausted,
1465                retry_after_secs: Some(30),
1466            },
1467            "budget",
1468        );
1469        assert_eq!(error.retry_after_secs(), Some(30));
1470        assert_eq!(error.with_retry_after_secs(5).retry_after_secs(), Some(5));
1471    }
1472
1473    #[test]
1474    fn transport_fields_survive_a_serde_round_trip_and_default_when_absent() {
1475        let error = LlmError::new(LlmErrorKind::RateLimited, "slow down")
1476            .with_status(429)
1477            .with_code("rate_limit_exceeded")
1478            .with_retry_after_secs(12);
1479        let json = serde_json::to_value(&error).unwrap();
1480        let back: LlmError = serde_json::from_value(json).unwrap();
1481        assert_eq!(back.status, Some(429));
1482        assert_eq!(back.code.as_deref(), Some("rate_limit_exceeded"));
1483        assert_eq!(back.retry_after_secs, Some(12));
1484
1485        let legacy: LlmError =
1486            serde_json::from_str(r#"{"kind":"other","message":"legacy"}"#).unwrap();
1487        assert_eq!(legacy.status, None);
1488        assert_eq!(legacy.code, None);
1489    }
1490
1491    #[test]
1492    fn a_malformed_response_is_classified_and_never_retried() {
1493        let error = AgentLoopError::llm_kind(
1494            LlmErrorKind::MalformedResponse,
1495            "stream ended before its terminal event",
1496        );
1497        assert!(!error.is_transient_llm_error());
1498        assert!(!error.is_rate_limited());
1499    }
1500}