Skip to main content

everruns_contracts/
error.rs

1// Error types for the agent loop
2//
3// StoreResultExt: extension trait to replace repeated .map_err(|e| AgentLoopError::store(...))? patterns
4// json_val / from_json: helpers to replace repeated serde_json::to_value/from_value boilerplate
5
6use crate::typed_id::{AgentId, HarnessId, SessionId};
7use crate::user_facing_error::{
8    AttestationRequirement, UserFacingError, UserFacingErrorContext,
9    classify_runtime_error_message, codes as user_facing_error_codes,
10    parse_attestation_requirement,
11};
12use serde::{Serialize, de::DeserializeOwned};
13use thiserror::Error;
14
15/// Result type alias for agent loop operations
16pub type Result<T> = std::result::Result<T, AgentLoopError>;
17
18// Stable provider errors remain re-exported here for compatibility; internal
19// capability markers stay private to keep this surface focused.
20pub use crate::llm_error::{BillingPressureReason, LlmError, LlmErrorKind};
21use crate::llm_error::{RejectedProviderCapability, provider_error_code_in};
22/// Errors that can occur during agent loop execution
23#[derive(Debug, Error)]
24pub enum AgentLoopError {
25    /// LLM provider error
26    #[error("LLM error: {0}")]
27    Llm(LlmError),
28
29    /// Request too large error (context length exceeded, token limits, etc.)
30    /// Contains the original error message for logging
31    #[error("Request too large: {0}")]
32    RequestTooLarge(String),
33
34    /// Model not available (404, model not found, access denied for model)
35    /// Contains the model_id string that was requested
36    #[error("Model not available: {0}")]
37    ModelNotAvailable(String),
38
39    /// No explicit, snapshot, or system-default model could be resolved.
40    #[error("Model not configured")]
41    ModelNotConfigured,
42
43    /// Tool execution error
44    #[error("Tool execution error: {0}")]
45    ToolExecution(String),
46
47    /// Message store error
48    #[error("Message store error: {0}")]
49    MessageStore(String),
50
51    /// Event emission error
52    #[error("Event emission error: {0}")]
53    EventEmission(String),
54
55    /// Configuration error
56    #[error("Configuration error: {0}")]
57    Configuration(String),
58
59    /// Loop terminated due to max iterations
60    #[error("Max iterations ({0}) reached")]
61    MaxIterationsReached(usize),
62
63    /// Loop was cancelled
64    #[error("Loop cancelled")]
65    Cancelled,
66
67    /// No messages to process
68    #[error("No messages to process")]
69    NoMessages,
70
71    /// Agent not found
72    #[error("Agent not found: {0}")]
73    AgentNotFound(AgentId),
74
75    /// Harness not found
76    #[error("Harness not found: {0}")]
77    HarnessNotFound(HarnessId),
78
79    /// Session not found
80    #[error("Session not found: {0}")]
81    SessionNotFound(SessionId),
82
83    /// Internal error
84    #[error("Internal error: {0}")]
85    Internal(#[from] anyhow::Error),
86
87    /// Driver not registered for provider type
88    #[error(
89        "No driver registered for provider type '{0}'. Make sure the driver is registered at startup."
90    )]
91    DriverNotRegistered(String),
92}
93
94impl AgentLoopError {
95    /// Prefix provider-bound messages without changing structured identifiers.
96    pub fn with_provider(mut self, provider: &str) -> Self {
97        let prefix = format!("provider '{provider}': ");
98        match &mut self {
99            AgentLoopError::Llm(error) if !error.message.starts_with(&prefix) => {
100                error.message.insert_str(0, &prefix)
101            }
102            AgentLoopError::RequestTooLarge(message) | AgentLoopError::Configuration(message)
103                if !message.starts_with(&prefix) =>
104            {
105                message.insert_str(0, &prefix)
106            }
107            // ModelNotAvailable stores a model ID, not a free-form message.
108            _ => {}
109        }
110        self
111    }
112
113    /// Create an LLM error with no semantic kind (falls back to string
114    /// classification downstream).
115    pub fn llm(msg: impl Into<String>) -> Self {
116        AgentLoopError::Llm(LlmError::new(LlmErrorKind::Other, msg))
117    }
118
119    /// Create an LLM error with a semantic kind assigned at the driver boundary.
120    pub fn llm_kind(kind: LlmErrorKind, msg: impl Into<String>) -> Self {
121        AgentLoopError::Llm(LlmError::new(kind, msg))
122    }
123
124    /// Create an LLM error from an HTTP failure, classifying it and keeping
125    /// the status.
126    ///
127    /// This is the constructor HTTP drivers reach for: it is the one place
128    /// that both classifies (via
129    /// [`LlmErrorKind::from_provider_status`]) and preserves the status a
130    /// consumer would otherwise have to parse back out of the message.
131    pub fn llm_http(status: u16, body: &str, msg: impl Into<String>) -> Self {
132        Self::llm_http_kind(
133            LlmErrorKind::from_provider_status(status, body),
134            status,
135            body,
136            msg,
137        )
138    }
139
140    /// Create an LLM error from an HTTP failure a driver has already
141    /// classified, keeping the status and the provider's error code.
142    ///
143    /// The counterpart to [`llm_http`](Self::llm_http) for drivers whose
144    /// protocol extension classifies more precisely than status and body
145    /// alone allow.
146    pub fn llm_http_kind(
147        kind: LlmErrorKind,
148        status: u16,
149        body: &str,
150        msg: impl Into<String>,
151    ) -> Self {
152        let mut error = LlmError::new(kind, msg).with_status(status);
153        if let Some(code) = provider_error_code_in(body) {
154            error = error.with_code(code);
155        }
156        AgentLoopError::Llm(error)
157    }
158
159    /// Create a structured pre-stream provider capability rejection.
160    pub fn provider_capability_rejected(
161        capability: RejectedProviderCapability,
162        status: u16,
163        body: &str,
164        msg: impl Into<String>,
165    ) -> Self {
166        let mut error = LlmError::new(LlmErrorKind::InvalidRequest, msg)
167            .with_status(status)
168            .with_rejected_capability(capability);
169        if let Some(code) = provider_error_code_in(body) {
170            error = error.with_code(code);
171        }
172        AgentLoopError::Llm(error)
173    }
174
175    /// Whether this error rejects the specified provider capability.
176    pub fn rejected_provider_capability(&self, capability: RejectedProviderCapability) -> bool {
177        matches!(
178            self,
179            AgentLoopError::Llm(error)
180                if error.rejected_capability == Some(capability)
181        )
182    }
183
184    /// Attach the HTTP status to an LLM error that was classified elsewhere.
185    ///
186    /// No-op on non-LLM variants, whose status is implied by the variant and
187    /// reported by [`http_status`](Self::http_status).
188    #[must_use]
189    pub fn with_status(mut self, status: u16) -> Self {
190        if let AgentLoopError::Llm(error) = &mut self {
191            error.status = Some(status);
192        }
193        self
194    }
195
196    /// Attach the delay the provider asked for before another attempt.
197    #[must_use]
198    pub fn with_retry_after_secs(mut self, secs: u64) -> Self {
199        if let AgentLoopError::Llm(error) = &mut self {
200            error.retry_after_secs = Some(secs);
201        }
202        self
203    }
204
205    /// The HTTP status this failure corresponds to, if any.
206    ///
207    /// Reports the recorded status for [`Llm`](Self::Llm), and the status the
208    /// variant itself stands for where Everruns classified the failure
209    /// semantically instead of carrying one:
210    /// [`ModelNotAvailable`](Self::ModelNotAvailable) is `404` and
211    /// [`RequestTooLarge`](Self::RequestTooLarge) is `413`. Everything else is
212    /// `None` — a caller putting an API in front of Everruns picks its own
213    /// status rather than being handed a guess.
214    pub fn http_status(&self) -> Option<u16> {
215        match self {
216            AgentLoopError::Llm(error) => error.status,
217            AgentLoopError::ModelNotAvailable(_) => Some(404),
218            AgentLoopError::RequestTooLarge(_) => Some(413),
219            _ => None,
220        }
221    }
222
223    /// The provider's own machine-readable error code, if one was preserved.
224    pub fn provider_error_code(&self) -> Option<&str> {
225        match self {
226            AgentLoopError::Llm(error) => error.code.as_deref(),
227            _ => None,
228        }
229    }
230
231    /// The delay the provider asked for before another attempt, in seconds.
232    ///
233    /// Reads the recorded `Retry-After` first, then the delay
234    /// [`LlmErrorKind::BillingPressure`] carries.
235    pub fn retry_after_secs(&self) -> Option<u64> {
236        match self {
237            AgentLoopError::Llm(error) => error.retry_after_secs.or(match error.kind {
238                LlmErrorKind::BillingPressure {
239                    retry_after_secs, ..
240                } => retry_after_secs,
241                _ => None,
242            }),
243            _ => None,
244        }
245    }
246
247    /// Attach retries already consumed by a lower provider layer. The reason
248    /// loop uses this to avoid multiplying attempt budgets across layers.
249    pub fn with_retry_metadata(mut self, metadata: &crate::llm_retry::RetryMetadata) -> Self {
250        if let AgentLoopError::Llm(error) = &mut self {
251            error.retry_attempts = metadata.attempts;
252            error.retry_wait_ms = metadata.total_retry_wait.as_millis() as u64;
253            error.retry_handled = true;
254        }
255        self
256    }
257
258    /// Number of lower-layer retries already consumed by this failure.
259    pub fn llm_retry_attempts(&self) -> u32 {
260        match self {
261            AgentLoopError::Llm(error) => error.retry_attempts,
262            _ => 0,
263        }
264    }
265
266    /// Whether a lower provider layer already exhausted or rejected recovery.
267    pub fn llm_retry_handled(&self) -> bool {
268        matches!(self, AgentLoopError::Llm(error) if error.retry_handled)
269    }
270
271    /// Get the semantic LLM error kind, if this is an LLM error.
272    pub fn llm_error_kind(&self) -> Option<LlmErrorKind> {
273        match self {
274            AgentLoopError::Llm(err) => Some(err.kind),
275            _ => None,
276        }
277    }
278
279    /// Create a tool execution error
280    pub fn tool(msg: impl Into<String>) -> Self {
281        AgentLoopError::ToolExecution(msg.into())
282    }
283
284    /// Create a message store error
285    pub fn store(msg: impl Into<String>) -> Self {
286        AgentLoopError::MessageStore(msg.into())
287    }
288
289    /// Create an event emission error
290    pub fn event(msg: impl Into<String>) -> Self {
291        AgentLoopError::EventEmission(msg.into())
292    }
293
294    /// Create a configuration error
295    pub fn config(msg: impl Into<String>) -> Self {
296        AgentLoopError::Configuration(msg.into())
297    }
298
299    /// Create an agent not found error
300    pub fn agent_not_found(agent_id: AgentId) -> Self {
301        AgentLoopError::AgentNotFound(agent_id)
302    }
303
304    /// Create a harness not found error
305    pub fn harness_not_found(harness_id: HarnessId) -> Self {
306        AgentLoopError::HarnessNotFound(harness_id)
307    }
308
309    /// Create a session not found error
310    pub fn session_not_found(session_id: SessionId) -> Self {
311        AgentLoopError::SessionNotFound(session_id)
312    }
313
314    /// Create a driver not registered error
315    pub fn driver_not_registered(provider_type: impl Into<String>) -> Self {
316        AgentLoopError::DriverNotRegistered(provider_type.into())
317    }
318
319    /// Create a request too large error
320    pub fn request_too_large(msg: impl Into<String>) -> Self {
321        AgentLoopError::RequestTooLarge(msg.into())
322    }
323
324    /// Create a model not available error
325    pub fn model_not_available(model_id: impl Into<String>) -> Self {
326        AgentLoopError::ModelNotAvailable(model_id.into())
327    }
328
329    /// Create a missing-model configuration error.
330    pub fn model_not_configured() -> Self {
331        AgentLoopError::ModelNotConfigured
332    }
333
334    /// Check if this is a request-too-large error
335    pub fn is_request_too_large(&self) -> bool {
336        matches!(self, AgentLoopError::RequestTooLarge(_))
337    }
338
339    /// Check if this is a model-not-available error
340    pub fn is_model_not_available(&self) -> bool {
341        matches!(self, AgentLoopError::ModelNotAvailable(_))
342    }
343
344    /// Get the model ID if this is a model-not-available error
345    pub fn model_not_available_id(&self) -> Option<&str> {
346        match self {
347            AgentLoopError::ModelNotAvailable(id) => Some(id),
348            _ => None,
349        }
350    }
351
352    /// Check if this is a rate-limit error (semantic kind, or HTTP 429 /
353    /// rate-limit keywords for untyped errors)
354    pub fn is_rate_limited(&self) -> bool {
355        match self {
356            AgentLoopError::Llm(err) => match err.kind {
357                LlmErrorKind::RateLimited => true,
358                // A recorded status is the provider's own answer; the string
359                // scan below is the fallback for failures that carry none.
360                LlmErrorKind::Other => match err.status {
361                    Some(status) => status == 429,
362                    None => {
363                        let msg_lower = err.message.to_ascii_lowercase();
364                        msg_lower.contains("(429)")
365                            || msg_lower.contains("rate limit")
366                            || msg_lower.contains("too many requests")
367                    }
368                },
369                _ => false,
370            },
371            _ => false,
372        }
373    }
374
375    /// Check if this is an authentication/authorization error (HTTP 401/403)
376    pub fn is_auth_error(&self) -> bool {
377        match self {
378            AgentLoopError::Llm(err) => match err.kind {
379                LlmErrorKind::Authentication => true,
380                LlmErrorKind::Other => match err.status {
381                    Some(status) => status == 401 || status == 403,
382                    None => err.message.contains("(401)") || err.message.contains("(403)"),
383                },
384                _ => false,
385            },
386            _ => false,
387        }
388    }
389
390    /// Check if this is a server error (HTTP 5xx or transient provider issue)
391    pub fn is_server_error(&self) -> bool {
392        match self {
393            AgentLoopError::Llm(err) => match err.kind {
394                LlmErrorKind::Unavailable => true,
395                LlmErrorKind::Other => match err.status {
396                    Some(status) => status >= 500,
397                    None => {
398                        let msg = &err.message;
399                        msg.contains("(500)")
400                            || msg.contains("(502)")
401                            || msg.contains("(503)")
402                            || msg.contains("(504)")
403                            || msg.contains("(529)")
404                    }
405                },
406                _ => false,
407            },
408            _ => false,
409        }
410    }
411
412    /// Check whether an LLM failure is safe to retry.
413    ///
414    /// Semantic driver classification is authoritative. Untyped legacy errors
415    /// retain the message-based fallback until all drivers preserve structure.
416    pub fn is_transient_llm_error(&self) -> bool {
417        match self {
418            AgentLoopError::Llm(err) => match err.kind {
419                LlmErrorKind::RateLimited | LlmErrorKind::Unavailable => true,
420                LlmErrorKind::Authentication
421                | LlmErrorKind::QuotaExhausted
422                | LlmErrorKind::BillingPressure { .. }
423                | LlmErrorKind::AttestationRequired
424                | LlmErrorKind::MalformedResponse
425                | LlmErrorKind::InvalidRequest => false,
426                LlmErrorKind::Other => crate::llm_retry::is_transient_error_message(&err.message),
427            },
428            _ => false,
429        }
430    }
431
432    /// Check if this error is deterministic and should never be retried.
433    ///
434    /// Non-retryable errors reference data that is permanently gone (e.g. a
435    /// deleted message, a missing agent). Retrying will never succeed and only
436    /// burns attempts while keeping the workflow stuck.
437    ///
438    /// Note: the durable worker currently uses string-matching via
439    /// `is_non_retryable_task_error` because task errors arrive as strings.
440    /// This method provides the typed equivalent for callers that have access
441    /// to a structured `AgentLoopError`.
442    pub fn is_non_retryable(&self) -> bool {
443        match self {
444            // Missing data is permanent — the entity was deleted.
445            AgentLoopError::AgentNotFound(_)
446            | AgentLoopError::HarnessNotFound(_)
447            | AgentLoopError::SessionNotFound(_)
448            | AgentLoopError::NoMessages
449            | AgentLoopError::ModelNotConfigured => true,
450
451            // Config/driver errors won't self-heal within retries.
452            AgentLoopError::Configuration(_) | AgentLoopError::DriverNotRegistered(_) => true,
453
454            // MessageStore "not found" errors (deleted messages).
455            AgentLoopError::MessageStore(msg) => msg.to_ascii_lowercase().contains("not found"),
456
457            // Everything else is potentially transient.
458            _ => false,
459        }
460    }
461
462    /// Get user-facing error message based on error classification
463    pub fn user_facing_message(&self) -> String {
464        self.user_facing_error(UserFacingErrorContext::default())
465            .fallback_message()
466    }
467
468    /// Get structured user-facing error metadata based on error classification.
469    pub fn user_facing_error(&self, context: UserFacingErrorContext) -> UserFacingError {
470        match self {
471            AgentLoopError::ModelNotConfigured => {
472                UserFacingError::new(user_facing_error_codes::MODEL_NOT_CONFIGURED)
473            }
474            AgentLoopError::ModelNotAvailable(model_id) => {
475                UserFacingError::new(user_facing_error_codes::MODEL_UNAVAILABLE)
476                    .with_field("model_id", model_id)
477                    .with_optional_field("provider", context.provider)
478            }
479            AgentLoopError::RequestTooLarge(_) => {
480                UserFacingError::new(user_facing_error_codes::REQUEST_TOO_LARGE)
481                    .with_optional_field("provider", context.provider)
482                    .with_optional_field("model_id", context.model_id)
483            }
484            AgentLoopError::MaxIterationsReached(max_iterations) => {
485                UserFacingError::new(user_facing_error_codes::MAX_ITERATIONS)
486                    .with_field("max_iterations", max_iterations)
487            }
488            AgentLoopError::Llm(err) => {
489                // Prefer the semantic kind the driver assigned at the provider
490                // boundary; fall back to string classification for untyped
491                // errors so legacy paths keep working.
492                let code = match err.kind {
493                    LlmErrorKind::Authentication => {
494                        Some(user_facing_error_codes::PROVIDER_MISCONFIGURED)
495                    }
496                    LlmErrorKind::QuotaExhausted => {
497                        Some(user_facing_error_codes::PROVIDER_QUOTA_EXHAUSTED)
498                    }
499                    LlmErrorKind::BillingPressure { reason, .. } => Some(match reason {
500                        BillingPressureReason::InFlightBudgetExhausted => {
501                            user_facing_error_codes::PROVIDER_RATE_LIMITED
502                        }
503                        BillingPressureReason::InsufficientCredits => {
504                            user_facing_error_codes::PROVIDER_QUOTA_EXHAUSTED
505                        }
506                    }),
507                    LlmErrorKind::RateLimited => {
508                        Some(user_facing_error_codes::PROVIDER_RATE_LIMITED)
509                    }
510                    LlmErrorKind::Unavailable => {
511                        Some(user_facing_error_codes::PROVIDER_UNAVAILABLE)
512                    }
513                    LlmErrorKind::AttestationRequired => {
514                        Some(user_facing_error_codes::PROVIDER_ATTESTATION_REQUIRED)
515                    }
516                    LlmErrorKind::InvalidRequest
517                    | LlmErrorKind::MalformedResponse
518                    | LlmErrorKind::Other => None,
519                };
520                match code {
521                    Some(code) => {
522                        let error = UserFacingError::new(code)
523                            .with_optional_field("provider", context.provider)
524                            .with_optional_field("model_id", context.model_id);
525                        if code == user_facing_error_codes::PROVIDER_RATE_LIMITED {
526                            let retry_after = match err.kind {
527                                LlmErrorKind::BillingPressure {
528                                    retry_after_secs, ..
529                                } => retry_after_secs,
530                                _ => context.retry_after,
531                            };
532                            error.with_optional_field("retry_after", retry_after)
533                        } else if code == user_facing_error_codes::PROVIDER_ATTESTATION_REQUIRED {
534                            // The attestation variant carries no payload, so
535                            // the confirmations and URL are read from the raw
536                            // body the driver retained in `message`.
537                            parse_attestation_requirement(&err.message)
538                                .unwrap_or_else(AttestationRequirement::fallback)
539                                .apply_fields(error)
540                        } else {
541                            error
542                        }
543                    }
544                    None => classify_runtime_error_message(&err.message, &context),
545                }
546            }
547            _ => UserFacingError::new(user_facing_error_codes::PROCESSING_ERROR)
548                .with_optional_field("provider", context.provider)
549                .with_optional_field("model_id", context.model_id),
550        }
551    }
552}
553
554// ============================================================================
555// Store Result Extension Trait
556// ============================================================================
557
558/// Extension trait that converts any `Result<T, E: Display>` into `Result<T, AgentLoopError>`
559/// via `AgentLoopError::store(e.to_string())`.
560///
561/// Replaces the boilerplate pattern:
562/// ```ignore
563/// .map_err(|e| AgentLoopError::store(e.to_string()))?
564/// ```
565/// with:
566/// ```ignore
567/// .store_err()?
568/// ```
569pub trait StoreResultExt<T> {
570    fn store_err(self) -> Result<T>;
571}
572
573impl<T, E: std::fmt::Display> StoreResultExt<T> for std::result::Result<T, E> {
574    fn store_err(self) -> Result<T> {
575        self.map_err(|e| AgentLoopError::store(e.to_string()))
576    }
577}
578
579// ============================================================================
580// SessionFileSystem error classification (EVE-645)
581// ============================================================================
582
583/// Typed classification of a `SessionFileSystem` failure.
584///
585/// The file-system tools decide
586/// whether a failure is a *tool error* (surfaced to the agent verbatim — bad
587/// input it can correct) or an *internal error* (logged, generic copy). They
588/// previously made that call with `msg.contains("readonly")` / `"is a
589/// directory"` / `"not found"` style sniffs against the stringified error.
590///
591/// The `SessionFileSystem` trait returns `anyhow::Result<T>` and has 10+
592/// implementors across crates, so widening the trait's error type is out of
593/// scope. Instead, [`classify_fs_error`] gives a single typed seam: it
594/// downcasts to [`FileSystemError`] when an implementor opts in, and otherwise
595/// falls back to the legacy substring heuristics in one place. Implementors can
596/// migrate to returning `FileSystemError` (via `anyhow::Error::new`)
597/// incrementally without changing behavior.
598#[derive(Debug, Clone, Copy, PartialEq, Eq)]
599pub enum FileSystemErrorClass {
600    /// The target (or a path component) does not exist.
601    NotFound,
602    /// The target is read-only and cannot be written or deleted.
603    ReadOnly,
604    /// Expected a file but the path is a directory.
605    IsADirectory,
606    /// Expected a directory but the path is not one.
607    NotADirectory,
608    /// A non-recursive delete refused a non-empty directory.
609    NotEmpty,
610    /// No recognized client-correctable condition; treat as internal.
611    Other,
612}
613
614/// Typed `SessionFileSystem` error. Implementors may return this (wrapped in
615/// `anyhow::Error`) so [`classify_fs_error`] resolves the class without string
616/// matching. Each variant carries the human-facing message so the file tools
617/// can keep surfacing the same text to the agent.
618#[derive(Debug, Error)]
619pub enum FileSystemError {
620    #[error("{0}")]
621    NotFound(String),
622    #[error("{0}")]
623    ReadOnly(String),
624    #[error("{0}")]
625    IsADirectory(String),
626    #[error("{0}")]
627    NotADirectory(String),
628    #[error("{0}")]
629    NotEmpty(String),
630}
631
632impl FileSystemError {
633    fn class(&self) -> FileSystemErrorClass {
634        match self {
635            FileSystemError::NotFound(_) => FileSystemErrorClass::NotFound,
636            FileSystemError::ReadOnly(_) => FileSystemErrorClass::ReadOnly,
637            FileSystemError::IsADirectory(_) => FileSystemErrorClass::IsADirectory,
638            FileSystemError::NotADirectory(_) => FileSystemErrorClass::NotADirectory,
639            FileSystemError::NotEmpty(_) => FileSystemErrorClass::NotEmpty,
640        }
641    }
642}
643
644/// Classify a `SessionFileSystem` failure into a [`FileSystemErrorClass`].
645///
646/// Prefers a typed [`FileSystemError`] in the error chain; falls back to the
647/// legacy substring heuristics (the single remaining place they live) so
648/// untyped implementors keep their current routing.
649/// "readonly" and "is a directory" mark client-correctable write failures,
650/// "not found" / "not a directory" mark client-correctable read failures, and
651/// "not empty" / "recursive" mark client-correctable delete failures.
652pub fn classify_fs_error<E>(err: &E) -> FileSystemErrorClass
653where
654    E: std::error::Error + 'static,
655{
656    // Prefer a typed FileSystemError anywhere in the source chain so an
657    // implementor that opts in is classified without string matching. Works
658    // whether the error is a bare FileSystemError or wrapped (e.g. inside
659    // `AgentLoopError::Internal(anyhow!(FileSystemError::..))`).
660    let mut source: Option<&(dyn std::error::Error + 'static)> = Some(err);
661    while let Some(current) = source {
662        if let Some(typed) = current.downcast_ref::<FileSystemError>() {
663            return typed.class();
664        }
665        source = current.source();
666    }
667
668    let msg = err.to_string();
669    // Note: real-disk backends emit "read-only" (hyphenated); the legacy check
670    // only matched "readonly", so we preserve that exact behavior rather than
671    // silently widening it.
672    if msg.contains("readonly") {
673        FileSystemErrorClass::ReadOnly
674    } else if msg.contains("is a directory") {
675        FileSystemErrorClass::IsADirectory
676    } else if msg.contains("not a directory") {
677        FileSystemErrorClass::NotADirectory
678    } else if msg.contains("not empty") || msg.contains("recursive") {
679        FileSystemErrorClass::NotEmpty
680    } else if msg.contains("not found") {
681        FileSystemErrorClass::NotFound
682    } else {
683        FileSystemErrorClass::Other
684    }
685}
686
687// ============================================================================
688// JSON Helpers
689// ============================================================================
690
691/// Convert a serializable value to `serde_json::Value`, falling back to `Value::Null` on error.
692///
693/// Replaces the boilerplate pattern:
694/// ```ignore
695/// serde_json::to_value(&x).unwrap_or_default()
696/// ```
697pub fn json_val<T: Serialize>(value: &T) -> serde_json::Value {
698    serde_json::to_value(value).unwrap_or_default()
699}
700
701/// Deserialize a `serde_json::Value` into `T`, falling back to `T::default()` on error.
702///
703/// Replaces the boilerplate pattern:
704/// ```ignore
705/// serde_json::from_value(v).unwrap_or_default()
706/// ```
707pub fn from_json<T: DeserializeOwned + Default>(value: serde_json::Value) -> T {
708    serde_json::from_value(value).unwrap_or_default()
709}
710
711#[cfg(test)]
712mod tests {
713    use super::*;
714    use serde_json::json;
715
716    #[test]
717    fn filesystem_typed_errors_win_over_conflicting_messages_and_wrappers() {
718        for (error, expected) in [
719            (
720                FileSystemError::NotFound("readonly".into()),
721                FileSystemErrorClass::NotFound,
722            ),
723            (
724                FileSystemError::ReadOnly("not found".into()),
725                FileSystemErrorClass::ReadOnly,
726            ),
727            (
728                FileSystemError::IsADirectory("not empty".into()),
729                FileSystemErrorClass::IsADirectory,
730            ),
731            (
732                FileSystemError::NotADirectory("is a directory".into()),
733                FileSystemErrorClass::NotADirectory,
734            ),
735            (
736                FileSystemError::NotEmpty("not found".into()),
737                FileSystemErrorClass::NotEmpty,
738            ),
739        ] {
740            assert_eq!(classify_fs_error(&error), expected);
741            let wrapped = AgentLoopError::Internal(
742                anyhow::Error::new(error).context("readonly outer failure"),
743            );
744            assert_eq!(classify_fs_error(&wrapped), expected);
745        }
746    }
747
748    #[test]
749    fn filesystem_legacy_messages_preserve_routing_and_case_boundaries() {
750        for (message, expected) in [
751            (
752                "Cannot modify readonly file: /a",
753                FileSystemErrorClass::ReadOnly,
754            ),
755            (
756                "Cannot delete readonly file: /a",
757                FileSystemErrorClass::ReadOnly,
758            ),
759            (
760                "write target is a directory: /a",
761                FileSystemErrorClass::IsADirectory,
762            ),
763            (
764                "Path is not a directory: /a",
765                FileSystemErrorClass::NotADirectory,
766            ),
767            (
768                "workspace root is not a directory: /a",
769                FileSystemErrorClass::NotADirectory,
770            ),
771            ("Directory not found: /a", FileSystemErrorClass::NotFound),
772            (
773                "Directory is not empty. Use recursive=true to delete",
774                FileSystemErrorClass::NotEmpty,
775            ),
776            (
777                "Cannot delete root directory without recursive flag",
778                FileSystemErrorClass::NotEmpty,
779            ),
780            (
781                "recursive delete failed for /a: io",
782                FileSystemErrorClass::NotEmpty,
783            ),
784            ("readonly file not found", FileSystemErrorClass::ReadOnly),
785            ("file is read-only: /a", FileSystemErrorClass::Other),
786            ("NOT FOUND", FileSystemErrorClass::Other),
787            ("disk full", FileSystemErrorClass::Other),
788        ] {
789            assert_eq!(
790                classify_fs_error(&AgentLoopError::store(message)),
791                expected,
792                "{message}"
793            );
794        }
795    }
796
797    #[test]
798    fn typed_request_and_model_errors_preserve_identity_and_safe_user_payload() {
799        let context = || {
800            UserFacingErrorContext::default()
801                .with_provider("provider")
802                .with_model_id("context-model")
803                .with_retry_after(9)
804        };
805        let request = AgentLoopError::request_too_large("private payload");
806        assert_eq!(request.to_string(), "Request too large: private payload");
807        assert!(request.is_request_too_large());
808        assert!(!request.is_model_not_available());
809        assert_eq!(request.model_not_available_id(), None);
810        assert_eq!(
811            serde_json::to_value(request.user_facing_error(context())).unwrap(),
812            json!({"code":"request_too_large","fields":{"provider":"provider","model_id":"context-model"}})
813        );
814        assert_eq!(
815            request.user_facing_message(),
816            "The conversation has become too long for the model to process. Please start a new session or reduce the context size."
817        );
818        let model = AgentLoopError::model_not_available("gpt-99")
819            .with_provider("custom")
820            .with_provider("custom");
821        assert!(!model.is_request_too_large());
822        assert!(model.is_model_not_available());
823        assert_eq!(model.model_not_available_id(), Some("gpt-99"));
824        assert_eq!(model.to_string(), "Model not available: gpt-99");
825        assert_eq!(
826            model.user_facing_message(),
827            "The model `gpt-99` is not available. It may have been removed, renamed, or your API key may not have access to it. Please select a different model."
828        );
829        assert_eq!(
830            serde_json::to_value(model.user_facing_error(context())).unwrap(),
831            json!({"code":"model_unavailable","fields":{"provider":"provider","model_id":"gpt-99"}})
832        );
833        for other in [
834            AgentLoopError::llm("Request too large: Model not available: gpt-99"),
835            AgentLoopError::tool("failed"),
836            AgentLoopError::Cancelled,
837        ] {
838            assert!(!other.is_request_too_large());
839            assert!(!other.is_model_not_available());
840            assert_eq!(other.model_not_available_id(), None);
841        }
842    }
843
844    #[test]
845    fn semantic_kinds_override_conflicting_text_for_predicates_and_payloads() {
846        for (kind, message, predicates, code) in [
847            (
848                LlmErrorKind::Authentication,
849                "(429) rate limit (503)",
850                (false, true, false, false),
851                "provider_misconfigured",
852            ),
853            (
854                LlmErrorKind::QuotaExhausted,
855                "(401) (429) rate limit (503)",
856                (false, false, false, false),
857                "provider_quota_exhausted",
858            ),
859            (
860                LlmErrorKind::RateLimited,
861                "(401) (503) insufficient_quota",
862                (true, false, false, true),
863                "provider_rate_limited",
864            ),
865            (
866                LlmErrorKind::Unavailable,
867                "(401) (429) insufficient_quota",
868                (false, false, true, true),
869                "provider_unavailable",
870            ),
871            (
872                LlmErrorKind::InvalidRequest,
873                "opaque private failure",
874                (false, false, false, false),
875                "processing_error",
876            ),
877        ] {
878            let error = AgentLoopError::llm_kind(kind, message);
879            assert_eq!(error.llm_error_kind(), Some(kind));
880            assert_eq!(
881                (
882                    error.is_rate_limited(),
883                    error.is_auth_error(),
884                    error.is_server_error(),
885                    error.is_transient_llm_error()
886                ),
887                predicates,
888                "{kind:?}"
889            );
890            let mut fields = json!({"provider":"provider","model_id":"model"});
891            if kind == LlmErrorKind::RateLimited {
892                fields["retry_after"] = json!(12);
893            }
894            assert_eq!(
895                serde_json::to_value(
896                    error.user_facing_error(
897                        UserFacingErrorContext::default()
898                            .with_provider("provider")
899                            .with_model_id("model")
900                            .with_retry_after(12)
901                    )
902                )
903                .unwrap(),
904                json!({"code":code,"fields":fields})
905            );
906        }
907    }
908
909    #[test]
910    fn legacy_predicates_and_user_copy_use_independent_literal_cases() {
911        for (message, expected, copy) in [
912            (
913                "Anthropic API error (429): rate limit exceeded",
914                (true, false, false),
915                "Rate limited by the AI provider. Please wait a moment.",
916            ),
917            (
918                "Rate limit exceeded (after 2 retries)",
919                (true, false, false),
920                "Rate limited by the AI provider. Please wait a moment.",
921            ),
922            (
923                "too many requests",
924                (true, false, false),
925                "Rate limited by the AI provider. Please wait a moment.",
926            ),
927            (
928                "Anthropic API error (401): invalid api key",
929                (false, true, false),
930                "There is a misconfiguration with the AI provider. Please contact support.",
931            ),
932            (
933                "OpenAI API error (403): forbidden",
934                (false, true, false),
935                "There is a misconfiguration with the AI provider. Please contact support.",
936            ),
937            (
938                "Anthropic API error (500): internal server error",
939                (false, false, true),
940                "The AI provider is experiencing issues. Please try again shortly.",
941            ),
942            (
943                "OpenAI API error (503): service unavailable",
944                (false, false, true),
945                "The AI provider is experiencing issues. Please try again shortly.",
946            ),
947            (
948                "Failed to send request: connection refused",
949                (false, false, false),
950                "I encountered an error while processing your request. Please try again later.",
951            ),
952        ] {
953            let error = AgentLoopError::llm(message);
954            assert_eq!(
955                (
956                    error.is_rate_limited(),
957                    error.is_auth_error(),
958                    error.is_server_error()
959                ),
960                expected,
961                "{message}"
962            );
963            assert_eq!(error.user_facing_message(), copy, "{message}");
964        }
965        for status in [502, 504, 529] {
966            assert!(AgentLoopError::llm(format!("error ({status})")).is_server_error());
967        }
968        let non_llm = AgentLoopError::tool("(401) (429) (503) rate limit");
969        assert_eq!(
970            (
971                non_llm.is_rate_limited(),
972                non_llm.is_auth_error(),
973                non_llm.is_server_error(),
974                non_llm.is_transient_llm_error()
975            ),
976            (false, false, false, false)
977        );
978    }
979
980    #[test]
981    fn provider_status_classification_covers_boundaries_and_quota_precedence() {
982        for (status, expected) in [
983            (200, LlmErrorKind::Other),
984            (399, LlmErrorKind::Other),
985            (400, LlmErrorKind::InvalidRequest),
986            (401, LlmErrorKind::Authentication),
987            (403, LlmErrorKind::Authentication),
988            (404, LlmErrorKind::InvalidRequest),
989            (408, LlmErrorKind::Unavailable),
990            (409, LlmErrorKind::Unavailable),
991            (429, LlmErrorKind::RateLimited),
992            (499, LlmErrorKind::InvalidRequest),
993            (500, LlmErrorKind::Unavailable),
994            (501, LlmErrorKind::Other),
995            (502, LlmErrorKind::Unavailable),
996            (503, LlmErrorKind::Unavailable),
997            (529, LlmErrorKind::Unavailable),
998            (599, LlmErrorKind::Unavailable),
999            (600, LlmErrorKind::Other),
1000        ] {
1001            assert_eq!(
1002                LlmErrorKind::from_provider_status(status, "opaque"),
1003                expected,
1004                "{status}"
1005            );
1006        }
1007        for message in [
1008            r#"{"error":{"type":"insufficient_quota"}}"#,
1009            r#"{"error":{"code":"credit_balance_exhausted"}}"#,
1010            r#"{"error":{"type":"usage_limit_reached"}}"#,
1011            "Your credit balance is too low to access the Anthropic API.",
1012        ] {
1013            for status in [400, 401, 429, 503] {
1014                assert_eq!(
1015                    LlmErrorKind::from_provider_status(status, message),
1016                    LlmErrorKind::QuotaExhausted,
1017                    "{status}: {message}"
1018                );
1019            }
1020        }
1021    }
1022
1023    /// The canonical OpenRouter refusal from EVE-952, verbatim off the wire.
1024    const ATTESTATION_BODY: &str = r#"{"error":{"message":"This model requires you to complete the following before use: 18+ age confirmation. Confirm at https://openrouter.ai/settings/preferences.","code":403,"metadata":{"missing_attestation_types":["age_18plus"],"routing_funnel":[{"step":"Initial Endpoints","endpoint_count":1}],"failed_routing_step":"Gate Endpoints with Attestations"}}}"#;
1025
1026    #[test]
1027    fn attestation_gate_is_classified_apart_from_other_403s() {
1028        assert_eq!(
1029            LlmErrorKind::from_provider_status(403, ATTESTATION_BODY),
1030            LlmErrorKind::AttestationRequired
1031        );
1032        // Status is not the signal: the same body under another status still
1033        // names the gate, and a 403 without one stays an auth failure.
1034        assert_eq!(
1035            LlmErrorKind::from_provider_status(429, ATTESTATION_BODY),
1036            LlmErrorKind::AttestationRequired
1037        );
1038        for body in [
1039            r#"{"error":{"message":"Invalid credentials","code":403}}"#,
1040            r#"{"error":{"message":"Insufficient credits","code":403,"metadata":{"routing_funnel":[]}}}"#,
1041            "opaque",
1042        ] {
1043            assert_ne!(
1044                LlmErrorKind::from_provider_status(403, body),
1045                LlmErrorKind::AttestationRequired,
1046                "{body}"
1047            );
1048        }
1049        // Exhausted billing keeps precedence over the gate check.
1050        assert_eq!(
1051            LlmErrorKind::from_provider_status(
1052                403,
1053                r#"{"error":{"message":"insufficient_quota; requires you to complete the following before use"}}"#
1054            ),
1055            LlmErrorKind::QuotaExhausted
1056        );
1057    }
1058
1059    #[test]
1060    fn attestation_gate_reaches_the_reader_with_the_types_and_the_confirm_url() {
1061        let error = AgentLoopError::llm_kind(
1062            LlmErrorKind::AttestationRequired,
1063            format!("OpenAI Responses API error (403): {ATTESTATION_BODY}"),
1064        )
1065        .with_provider("openrouter");
1066        // Not a credential problem and never worth retrying.
1067        assert!(!error.is_auth_error());
1068        assert!(!error.is_transient_llm_error());
1069        assert_eq!(
1070            serde_json::to_value(
1071                error.user_facing_error(
1072                    UserFacingErrorContext::default()
1073                        .with_provider("openrouter")
1074                        .with_model_id("meta/muse-spark-1.3-contributor")
1075                )
1076            )
1077            .unwrap(),
1078            json!({
1079                "code": "provider_attestation_required",
1080                "fields": {
1081                    "provider": "openrouter",
1082                    "model_id": "meta/muse-spark-1.3-contributor",
1083                    "missing_types": ["age_18plus"],
1084                    "confirm_url": "https://openrouter.ai/settings/preferences",
1085                }
1086            })
1087        );
1088        assert_eq!(
1089            error.user_facing_message(),
1090            "The AI provider account has not completed a confirmation this model requires (age_18plus). Complete it at https://openrouter.ai/settings/preferences, then try again."
1091        );
1092    }
1093
1094    #[test]
1095    fn untyped_attestation_bodies_still_route_off_the_403_misconfiguration_copy() {
1096        // Legacy/untyped errors reach the string classifier instead; it must
1097        // reach the same code rather than "contact support".
1098        let error = AgentLoopError::llm(format!(
1099            "provider 'openrouter': OpenAI Responses API error (403): {ATTESTATION_BODY}"
1100        ));
1101        assert_eq!(
1102            error
1103                .user_facing_error(UserFacingErrorContext::default())
1104                .code,
1105            "provider_attestation_required"
1106        );
1107    }
1108
1109    #[test]
1110    fn attestation_parsing_covers_multiple_types_escaped_bodies_and_a_missing_url() {
1111        let requirement = |body: &str| {
1112            parse_attestation_requirement(body).unwrap_or_else(|| panic!("no gate in {body}"))
1113        };
1114
1115        // Multiple gates, in payload order.
1116        let multiple = requirement(
1117            r#"{"error":{"message":"This model requires you to complete the following before use: 18+ age confirmation and identity verification. Confirm at https://openrouter.ai/settings/preferences.","metadata":{"missing_attestation_types":["age_18plus","identity_verified"]}}}"#,
1118        );
1119        assert_eq!(multiple.missing_types, ["age_18plus", "identity_verified"]);
1120        assert_eq!(
1121            multiple.confirm_url,
1122            "https://openrouter.ai/settings/preferences"
1123        );
1124
1125        // JSON-escaped body (a provider error nested in another envelope).
1126        let escaped = requirement(
1127            r#"{"detail":"{\"error\":{\"message\":\"This model requires you to complete the following before use: 18+ age confirmation. Confirm at https:\/\/openrouter.ai\/settings\/gates.\",\"metadata\":{\"missing_attestation_types\":[\"age_18plus\"]}}}"}"#,
1128        );
1129        assert_eq!(escaped.missing_types, ["age_18plus"]);
1130        assert_eq!(escaped.confirm_url, "https://openrouter.ai/settings/gates");
1131
1132        // No URL in the message: fall back rather than leave the reader with
1133        // nowhere to go.
1134        let no_url = requirement(
1135            r#"{"error":{"message":"This model requires you to complete the following before use: 18+ age confirmation.","metadata":{"missing_attestation_types":["age_18plus"]}}}"#,
1136        );
1137        assert_eq!(
1138            no_url.confirm_url,
1139            "https://openrouter.ai/settings/preferences"
1140        );
1141
1142        // The gate sentence alone is enough; the metadata block is optional.
1143        let sentence_only = requirement(
1144            "This model requires you to complete the following before use: 18+ age confirmation. Confirm at https://openrouter.ai/settings/preferences",
1145        );
1146        assert!(sentence_only.missing_types.is_empty());
1147        assert_eq!(
1148            sentence_only.confirm_url,
1149            "https://openrouter.ai/settings/preferences"
1150        );
1151        // With nothing parsed, the message drops the list rather than
1152        // rendering an empty one.
1153        assert_eq!(
1154            AgentLoopError::llm_kind(
1155                LlmErrorKind::AttestationRequired,
1156                "This model requires you to complete the following before use: a confirmation."
1157            )
1158            .user_facing_message(),
1159            "The AI provider account has not completed a confirmation this model requires. Complete it at https://openrouter.ai/settings/preferences, then try again."
1160        );
1161
1162        // A URL in the driver's own prefix is not mistaken for the gate page.
1163        assert_eq!(
1164            requirement(&format!(
1165                "POST https://openrouter.ai/api/v1/responses failed: {ATTESTATION_BODY}"
1166            ))
1167            .confirm_url,
1168            "https://openrouter.ai/settings/preferences"
1169        );
1170
1171        for body in [
1172            r#"{"error":{"message":"Invalid credentials"}}"#,
1173            r#"{"error":{"metadata":{"missing_attestation_types":[]}}}"#,
1174            "",
1175        ] {
1176            assert!(parse_attestation_requirement(body).is_none(), "{body}");
1177        }
1178    }
1179
1180    #[test]
1181    fn a_hostile_attestation_payload_cannot_choose_how_much_reaches_the_viewer() {
1182        let types = (0..40)
1183            .map(|index| format!(r#""gate_{index}""#))
1184            .collect::<Vec<_>>()
1185            .join(",");
1186        let long_type = "x".repeat(65);
1187        let long_url = format!("https://evil.example/{}", "a".repeat(400));
1188        let requirement = parse_attestation_requirement(&format!(
1189            r#"{{"error":{{"message":"This model requires you to complete the following before use: gates. Confirm at {long_url}","metadata":{{"missing_attestation_types":["{long_type}",{types}]}}}}}}"#
1190        ))
1191        .expect("gate recognized");
1192
1193        // Over-long entries are dropped, not truncated, and the list is capped.
1194        assert_eq!(requirement.missing_types.len(), 8);
1195        assert_eq!(requirement.missing_types[0], "gate_0");
1196        // An over-long URL falls back rather than shipping a 400-char link.
1197        assert_eq!(
1198            requirement.confirm_url,
1199            "https://openrouter.ai/settings/preferences"
1200        );
1201
1202        // Non-http(s) schemes never become the confirmation link.
1203        for scheme in [
1204            "javascript:alert(1)",
1205            "data:text/html,<script>",
1206            "file:///etc/passwd",
1207        ] {
1208            assert_eq!(
1209                parse_attestation_requirement(&format!(
1210                    "This model requires you to complete the following before use: a gate. Confirm at {scheme}"
1211                ))
1212                .expect("gate recognized")
1213                .confirm_url,
1214                "https://openrouter.ai/settings/preferences",
1215                "{scheme}"
1216            );
1217        }
1218    }
1219
1220    #[test]
1221    fn provider_text_classification_uses_independent_keywords_and_precedence() {
1222        for (message, expected) in [
1223            ("ThrottlingException", LlmErrorKind::RateLimited),
1224            ("TooManyRequestsException", LlmErrorKind::RateLimited),
1225            ("RATE LIMIT", LlmErrorKind::RateLimited),
1226            ("too many requests", LlmErrorKind::RateLimited),
1227            ("AccessDeniedException", LlmErrorKind::Authentication),
1228            ("UnrecognizedClientException", LlmErrorKind::Authentication),
1229            ("ExpiredTokenException", LlmErrorKind::Authentication),
1230            ("InvalidSignatureException", LlmErrorKind::Authentication),
1231            ("unauthorized", LlmErrorKind::Authentication),
1232            ("ServiceUnavailableException", LlmErrorKind::Unavailable),
1233            ("service unavailable", LlmErrorKind::Unavailable),
1234            ("InternalServerException", LlmErrorKind::Unavailable),
1235            ("ModelNotReadyException", LlmErrorKind::Unavailable),
1236            (
1237                "usage_limit_reached; resets_at=1783767823; throttlingexception",
1238                LlmErrorKind::QuotaExhausted,
1239            ),
1240            ("something else entirely", LlmErrorKind::Other),
1241        ] {
1242            assert_eq!(
1243                LlmErrorKind::from_error_text(message),
1244                expected,
1245                "{message}"
1246            );
1247        }
1248    }
1249
1250    #[test]
1251    fn provider_prefix_preserves_kind_and_retry_metadata_without_duplication() {
1252        let metadata = crate::llm_retry::RetryMetadata {
1253            attempts: 2,
1254            total_retry_wait: std::time::Duration::from_millis(1234),
1255            ..Default::default()
1256        };
1257        let error = AgentLoopError::llm_kind(LlmErrorKind::Unavailable, "network failure")
1258            .with_retry_metadata(&metadata)
1259            .with_provider("custom")
1260            .with_provider("custom");
1261        assert_eq!(error.llm_retry_attempts(), 2);
1262        assert!(error.llm_retry_handled());
1263        let AgentLoopError::Llm(error) = error else {
1264            panic!("lost LLM variant")
1265        };
1266        assert_eq!(
1267            serde_json::to_value(error).unwrap(),
1268            json!({"kind":"unavailable","message":"provider 'custom': network failure","retry_attempts":2,"retry_wait_ms":1234,"retry_handled":true})
1269        );
1270        let legacy: LlmError =
1271            serde_json::from_value(json!({"kind":"other","message":"legacy"})).unwrap();
1272        assert_eq!(
1273            serde_json::to_value(legacy).unwrap(),
1274            json!({"kind":"other","message":"legacy","retry_attempts":0,"retry_wait_ms":0,"retry_handled":false})
1275        );
1276        let non_llm = AgentLoopError::Cancelled
1277            .with_retry_metadata(&metadata)
1278            .with_provider("custom");
1279        assert!(matches!(non_llm, AgentLoopError::Cancelled));
1280        assert_eq!(non_llm.llm_retry_attempts(), 0);
1281        assert!(!non_llm.llm_retry_handled());
1282    }
1283
1284    #[test]
1285    fn billing_pressure_preserves_typed_payload_and_safe_user_fields() {
1286        let kind = LlmErrorKind::BillingPressure {
1287            reason: BillingPressureReason::InFlightBudgetExhausted,
1288            retry_after_secs: Some(120),
1289        };
1290        let error = AgentLoopError::llm_kind(kind, "private provider body");
1291        assert_eq!(error.llm_error_kind(), Some(kind));
1292        assert!(!error.is_transient_llm_error());
1293        assert_eq!(
1294            serde_json::to_value(
1295                error.user_facing_error(
1296                    UserFacingErrorContext::default()
1297                        .with_provider("openrouter")
1298                        .with_model_id("vendor/model")
1299                )
1300            )
1301            .unwrap(),
1302            json!({
1303                "code": "provider_rate_limited",
1304                "fields": {
1305                    "provider": "openrouter",
1306                    "model_id": "vendor/model",
1307                    "retry_after": 120,
1308                }
1309            })
1310        );
1311        let AgentLoopError::Llm(error) = error else {
1312            panic!("lost LLM variant")
1313        };
1314        assert_eq!(
1315            serde_json::to_value(error.kind).unwrap(),
1316            json!({
1317                "billing_pressure": {
1318                    "reason": "in_flight_budget_exhausted",
1319                    "retry_after_secs": 120,
1320                }
1321            })
1322        );
1323
1324        let exhausted = AgentLoopError::llm_kind(
1325            LlmErrorKind::BillingPressure {
1326                reason: BillingPressureReason::InsufficientCredits,
1327                retry_after_secs: None,
1328            },
1329            "private provider body",
1330        );
1331        assert_eq!(
1332            exhausted
1333                .user_facing_error(UserFacingErrorContext::default())
1334                .code,
1335            "provider_quota_exhausted"
1336        );
1337    }
1338
1339    #[test]
1340    fn missing_model_and_iteration_limits_have_complete_safe_payloads() {
1341        let missing = AgentLoopError::model_not_configured();
1342        assert!(missing.is_non_retryable());
1343        assert_eq!(
1344            missing.user_facing_message(),
1345            "No model is configured for this chat. Choose a model or configure a default model, then try again."
1346        );
1347        assert_eq!(
1348            serde_json::to_value(missing.user_facing_error(UserFacingErrorContext::default()))
1349                .unwrap(),
1350            json!({"code":"model_not_configured"})
1351        );
1352        assert_eq!(
1353            serde_json::to_value(
1354                AgentLoopError::MaxIterationsReached(7)
1355                    .user_facing_error(UserFacingErrorContext::default())
1356            )
1357            .unwrap(),
1358            json!({"code":"max_iterations","fields":{"max_iterations":7}})
1359        );
1360    }
1361
1362    #[test]
1363    fn store_adapter_preserves_success_and_exact_error_variant_and_message() {
1364        let success: std::result::Result<Vec<String>, String> =
1365            Ok(vec!["first".into(), "second".into()]);
1366        assert_eq!(success.store_err().unwrap(), ["first", "second"]);
1367        let failure: std::result::Result<(), std::io::Error> =
1368            Err(std::io::Error::other("db unavailable"));
1369        let error = failure.store_err().unwrap_err();
1370        assert_eq!(error.to_string(), "Message store error: db unavailable");
1371        assert!(matches!(error,AgentLoopError::MessageStore(message) if message=="db unavailable"));
1372    }
1373
1374    #[test]
1375    fn json_helpers_preserve_structures_and_apply_documented_error_defaults() {
1376        assert_eq!(json_val(&vec![1, 2, 3]), json!([1, 2, 3]));
1377        assert_eq!(from_json::<Vec<String>>(json!(["a", "b"])), ["a", "b"]);
1378        assert_eq!(from_json::<i32>(json!("not a number")), 0);
1379        struct Fails;
1380        impl Serialize for Fails {
1381            fn serialize<S: serde::Serializer>(
1382                &self,
1383                _: S,
1384            ) -> std::result::Result<S::Ok, S::Error> {
1385                Err(serde::ser::Error::custom("synthetic serialization failure"))
1386            }
1387        }
1388        assert_eq!(json_val(&Fails), serde_json::Value::Null);
1389    }
1390
1391    #[test]
1392    fn llm_http_records_the_status_and_provider_code() {
1393        let body =
1394            r#"{"error":{"message":"no","code":"model_not_found","type":"invalid_request_error"}}"#;
1395        let error = AgentLoopError::llm_http(404, body, "OpenAI API error (404)");
1396        assert_eq!(error.http_status(), Some(404));
1397        assert_eq!(error.provider_error_code(), Some("model_not_found"));
1398    }
1399
1400    #[test]
1401    fn llm_http_falls_back_to_the_anthropic_error_type() {
1402        let body = r#"{"error":{"type":"overloaded_error","message":"busy"}}"#;
1403        let error = AgentLoopError::llm_http(529, body, "Anthropic API error (529)");
1404        assert_eq!(error.provider_error_code(), Some("overloaded_error"));
1405        assert_eq!(error.llm_error_kind(), Some(LlmErrorKind::Unavailable));
1406    }
1407
1408    #[test]
1409    fn a_provider_sized_error_body_is_not_parsed_and_a_long_code_is_not_kept() {
1410        // TM-DOS-038: the body is provider-controlled and nothing upstream
1411        // bounds it, so neither the parse nor the retained string may be
1412        // sized by the provider.
1413        let padding = "x".repeat(70 * 1024);
1414        let huge = format!(r#"{{"error":{{"code":"rate_limit","pad":"{padding}"}}}}"#);
1415        let error = AgentLoopError::llm_http(429, &huge, "provider refused");
1416        assert_eq!(error.provider_error_code(), None);
1417        // The status still classifies it, so nothing is lost but the code.
1418        assert!(error.is_rate_limited());
1419
1420        let long_code = "c".repeat(200);
1421        let body = format!(r#"{{"error":{{"code":"{long_code}"}}}}"#);
1422        assert_eq!(
1423            AgentLoopError::llm_http(400, &body, "boom").provider_error_code(),
1424            None,
1425            "a code longer than any real one is not a code"
1426        );
1427    }
1428
1429    #[test]
1430    fn a_non_json_body_yields_no_provider_code() {
1431        let error = AgentLoopError::llm_http(503, "upstream is down", "boom");
1432        assert_eq!(error.http_status(), Some(503));
1433        assert_eq!(error.provider_error_code(), None);
1434    }
1435
1436    #[test]
1437    fn semantic_variants_report_the_status_they_stand_for() {
1438        assert_eq!(
1439            AgentLoopError::model_not_available("gpt-5").http_status(),
1440            Some(404)
1441        );
1442        assert_eq!(
1443            AgentLoopError::request_too_large("too big").http_status(),
1444            Some(413)
1445        );
1446        assert_eq!(AgentLoopError::Cancelled.http_status(), None);
1447    }
1448
1449    #[test]
1450    fn a_recorded_status_classifies_an_untyped_failure() {
1451        // The message carries no "(429)" marker, so only the recorded status
1452        // can answer this — the case that forced embedders to scrape strings.
1453        let error = AgentLoopError::llm_http(429, "slow down", "provider refused the request");
1454        assert!(error.is_rate_limited());
1455        assert_eq!(error.retry_after_secs(), None);
1456        assert!(!AgentLoopError::llm("provider refused the request").is_rate_limited());
1457    }
1458
1459    #[test]
1460    fn retry_after_prefers_the_recorded_delay_over_the_billing_hint() {
1461        let error = AgentLoopError::llm_kind(
1462            LlmErrorKind::BillingPressure {
1463                reason: BillingPressureReason::InFlightBudgetExhausted,
1464                retry_after_secs: Some(30),
1465            },
1466            "budget",
1467        );
1468        assert_eq!(error.retry_after_secs(), Some(30));
1469        assert_eq!(error.with_retry_after_secs(5).retry_after_secs(), Some(5));
1470    }
1471
1472    #[test]
1473    fn transport_fields_survive_a_serde_round_trip_and_default_when_absent() {
1474        let error = LlmError::new(LlmErrorKind::RateLimited, "slow down")
1475            .with_status(429)
1476            .with_code("rate_limit_exceeded")
1477            .with_retry_after_secs(12);
1478        let json = serde_json::to_value(&error).unwrap();
1479        let back: LlmError = serde_json::from_value(json).unwrap();
1480        assert_eq!(back.status, Some(429));
1481        assert_eq!(back.code.as_deref(), Some("rate_limit_exceeded"));
1482        assert_eq!(back.retry_after_secs, Some(12));
1483
1484        let legacy: LlmError =
1485            serde_json::from_str(r#"{"kind":"other","message":"legacy"}"#).unwrap();
1486        assert_eq!(legacy.status, None);
1487        assert_eq!(legacy.code, None);
1488    }
1489
1490    #[test]
1491    fn a_malformed_response_is_classified_and_never_retried() {
1492        let error = AgentLoopError::llm_kind(
1493            LlmErrorKind::MalformedResponse,
1494            "stream ended before its terminal event",
1495        );
1496        assert!(!error.is_transient_llm_error());
1497        assert!(!error.is_rate_limited());
1498    }
1499}