everruns_contracts/llm_error.rs
1//! How a provider failure is classified, and what is preserved of it.
2//!
3//! [`LlmErrorKind`] is Everruns' taxonomy — what runtime policy retries, what
4//! it surfaces, what it refuses. [`LlmError`] carries it alongside the
5//! provider's own answer (status, error code, requested retry delay), recorded
6//! at the boundary where the HTTP response was still structured. Without those,
7//! anything downstream that has to re-express the failure is left scraping the
8//! display string, which is not a contract.
9//!
10//! Split out of [`error`](crate::error) when that file outgrew what anyone can
11//! hold in their head; everything here is re-exported from there, so existing
12//! paths keep working.
13
14use serde::{Deserialize, Serialize};
15
16use crate::user_facing_error::{
17 is_attestation_required_message, is_provider_quota_message, is_usage_limit_message,
18};
19
20/// Machine-readable reason for provider billing pressure.
21#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
22#[serde(rename_all = "snake_case")]
23pub enum BillingPressureReason {
24 /// Existing requests temporarily consume the account's available budget.
25 InFlightBudgetExhausted,
26 /// The provider account does not have enough credits for the request.
27 InsufficientCredits,
28}
29
30/// Provider feature rejected before a response stream began.
31#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
32#[serde(rename_all = "snake_case")]
33pub enum RejectedProviderCapability {
34 /// Anthropic threshold server-side compaction.
35 AnthropicServerCompaction,
36}
37
38/// Semantic classification of an LLM provider error, assigned by the driver
39/// at the provider boundary where the HTTP status and response body are still
40/// available. Downstream consumers prefer this over re-parsing error strings;
41/// `LlmErrorKind::Other` falls back to string classification
42/// (`classify_runtime_error_message`) so untyped errors keep working.
43#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
44#[serde(rename_all = "snake_case")]
45#[non_exhaustive]
46pub enum LlmErrorKind {
47 /// Invalid or missing credentials, or access denied (401/403, bad API key).
48 Authentication,
49 /// Provider account is out of credits/quota (billing). Non-transient:
50 /// needs operator action, unlike a regular rate limit.
51 QuotaExhausted,
52 /// Provider billing pressure with the stable machine-readable reason and
53 /// the provider's suggested retry delay. This is non-transient at the
54 /// runtime layer: consumers decide whether to wait or add credits.
55 BillingPressure {
56 /// Stable provider reason for the billing refusal.
57 reason: BillingPressureReason,
58 /// Provider-requested delay before another attempt, in seconds.
59 retry_after_secs: Option<u64>,
60 },
61 /// Transient rate limit (429).
62 RateLimited,
63 /// Provider outage or unreachable (5xx, 529, network failure).
64 Unavailable,
65 /// Provider account has not completed a confirmation the model requires
66 /// (OpenRouter's 18+ age gate). Non-transient and not a credential
67 /// problem: it clears when the account holder completes the confirmation,
68 /// so it is kept apart from `Authentication` even though it arrives as a
69 /// 403.
70 AttestationRequired,
71 /// Provider rejected the request shape (4xx that is not auth/quota/429).
72 InvalidRequest,
73 /// The provider answered, but the answer could not be used as a turn: the
74 /// stream ended before its terminal event, or it exceeded a limit the
75 /// caller set.
76 ///
77 /// Distinct from [`Unavailable`](Self::Unavailable) because retrying does
78 /// not help by itself, and from [`Other`](Self::Other) because the
79 /// classification is certain — the failure was decided on Everruns' side
80 /// of the wire, not guessed from provider prose.
81 MalformedResponse,
82 /// Unclassified; downstream falls back to string classification.
83 Other,
84}
85
86impl LlmErrorKind {
87 /// Classify a provider's stable machine-readable error code.
88 pub fn from_provider_code(code: &str) -> Option<Self> {
89 let code = code.trim().to_ascii_lowercase();
90 match code.as_str() {
91 "subscription_sharing_usage_limit_exceeded"
92 | "insufficient_quota"
93 | "billing_hard_limit_reached"
94 | "credit_balance_too_low"
95 | "credit_balance_exhausted" => Some(Self::QuotaExhausted),
96 "subscription_sharing_user_not_eligible"
97 | "subscription_sharing_invalid_user"
98 | "chatpass_v2_scope_not_authorized"
99 | "chatpass_v2_invalid_authorization_context"
100 | "authentication_error"
101 | "invalid_api_key"
102 | "permission_denied" => Some(Self::Authentication),
103 "rate_limit_exceeded" | "rate_limit_error" | "overloaded_error" => {
104 Some(Self::RateLimited)
105 }
106 "subscription_sharing_usage_unavailable"
107 | "subscription_sharing_user_unavailable"
108 | "server_error"
109 | "internal_error"
110 | "processing_error"
111 | "service_unavailable"
112 | "timeout" => Some(Self::Unavailable),
113 "subscription_sharing_unsupported_capability"
114 | "subscription_sharing_route_not_supported"
115 | "invalid_request_error"
116 | "model_not_found" => Some(Self::InvalidRequest),
117 "malformed_response" => Some(Self::MalformedResponse),
118 _ => None,
119 }
120 }
121
122 /// Classify a provider HTTP error from status code + response body.
123 ///
124 /// Quota/billing patterns are checked before the status code because
125 /// providers surface exhausted billing under different statuses
126 /// (OpenAI: 429 `insufficient_quota`, Anthropic: 400 "credit balance is
127 /// too low").
128 pub fn from_provider_status(status: u16, body: &str) -> Self {
129 if is_provider_quota_message(body) || is_usage_limit_message(body) {
130 return LlmErrorKind::QuotaExhausted;
131 }
132 // Body-driven for the same reason as quota: the 403 this arrives under
133 // is indistinguishable from a bad-key 403 by status alone, and the
134 // gate is worth naming only when the body actually reports one.
135 if is_attestation_required_message(body) {
136 return LlmErrorKind::AttestationRequired;
137 }
138 match status {
139 401 | 403 => LlmErrorKind::Authentication,
140 429 => LlmErrorKind::RateLimited,
141 408 | 409 => LlmErrorKind::Unavailable,
142 501 => LlmErrorKind::Other,
143 500..=599 => LlmErrorKind::Unavailable,
144 400..=499 => LlmErrorKind::InvalidRequest,
145 _ => LlmErrorKind::Other,
146 }
147 }
148
149 /// Keyword-based classification for drivers without an HTTP status at the
150 /// error site (e.g. Bedrock SDK errors).
151 pub fn from_error_text(text: &str) -> Self {
152 if is_provider_quota_message(text) || is_usage_limit_message(text) {
153 return LlmErrorKind::QuotaExhausted;
154 }
155 let lower = text.to_ascii_lowercase();
156 if lower.contains("throttlingexception")
157 || lower.contains("toomanyrequestsexception")
158 || lower.contains("rate limit")
159 || lower.contains("too many requests")
160 {
161 return LlmErrorKind::RateLimited;
162 }
163 if lower.contains("accessdeniedexception")
164 || lower.contains("unrecognizedclientexception")
165 || lower.contains("expiredtokenexception")
166 || lower.contains("invalidsignatureexception")
167 || lower.contains("unauthorized")
168 {
169 return LlmErrorKind::Authentication;
170 }
171 if lower.contains("serviceunavailable")
172 || lower.contains("service unavailable")
173 || lower.contains("internalserverexception")
174 || lower.contains("modelnotreadyexception")
175 {
176 return LlmErrorKind::Unavailable;
177 }
178 LlmErrorKind::Other
179 }
180}
181
182/// LLM provider error with a semantic kind attached by the driver.
183///
184/// [`kind`](Self::kind) is what runtime policy keys on. The transport fields
185/// beside it — [`status`](Self::status), [`code`](Self::code),
186/// [`retry_after_secs`](Self::retry_after_secs) — are the provider's own
187/// answer, recorded at the boundary where it was still structured. Without
188/// them an embedder that has to re-express a failure (an HTTP API in front of
189/// Everruns, a retry budget of its own) can only scrape them back out of
190/// [`message`](Self::message), which is display text and not a contract.
191///
192/// `#[non_exhaustive]`: the boundary keeps learning to preserve more, and a
193/// field addition must not break construction downstream. Build with
194/// [`LlmError::new`] and the `with_*` setters.
195#[derive(Debug, Clone, Serialize, Deserialize)]
196#[non_exhaustive]
197pub struct LlmError {
198 pub kind: LlmErrorKind,
199 pub message: String,
200 /// HTTP status the provider answered with, when the failure arrived as an
201 /// HTTP response. `None` for SDK, transport, and protocol failures that
202 /// never carried one.
203 #[serde(default, skip_serializing_if = "Option::is_none")]
204 pub status: Option<u16>,
205 /// The provider's own machine-readable error code, verbatim (OpenAI
206 /// `error.code`, Anthropic `error.type`). Kept beside `kind` rather than
207 /// folded into it: `kind` is Everruns' taxonomy, this is the provider's.
208 #[serde(default, skip_serializing_if = "Option::is_none")]
209 pub code: Option<String>,
210 /// Everruns capability that the provider rejected before a response stream
211 /// started. Runtime fallback policy consumes this marker without parsing
212 /// provider error prose.
213 #[serde(default, skip_serializing_if = "Option::is_none")]
214 pub rejected_capability: Option<RejectedProviderCapability>,
215 /// Delay the provider asked for before another attempt, in seconds
216 /// (`Retry-After` or an equivalent rate-limit header).
217 #[serde(default, skip_serializing_if = "Option::is_none")]
218 pub retry_after_secs: Option<u64>,
219 /// Retries already consumed below the turn loop.
220 #[serde(default)]
221 pub retry_attempts: u32,
222 /// Backoff time already consumed below the turn loop.
223 #[serde(default)]
224 pub retry_wait_ms: u64,
225 /// Whether a lower provider layer already made the terminal retry decision.
226 #[serde(default)]
227 pub retry_handled: bool,
228}
229
230impl LlmError {
231 /// A provider failure with its semantic kind and nothing else known.
232 pub fn new(kind: LlmErrorKind, message: impl Into<String>) -> Self {
233 LlmError {
234 kind,
235 message: message.into(),
236 status: None,
237 code: None,
238 rejected_capability: None,
239 retry_after_secs: None,
240 retry_attempts: 0,
241 retry_wait_ms: 0,
242 retry_handled: false,
243 }
244 }
245
246 /// Record the HTTP status the provider answered with.
247 #[must_use]
248 pub fn with_status(mut self, status: u16) -> Self {
249 self.status = Some(status);
250 self
251 }
252
253 /// Record the provider's machine-readable error code.
254 #[must_use]
255 pub fn with_code(mut self, code: impl Into<String>) -> Self {
256 self.code = Some(code.into());
257 self
258 }
259
260 /// Mark a provider capability as rejected before streaming began.
261 #[must_use]
262 pub fn with_rejected_capability(mut self, capability: RejectedProviderCapability) -> Self {
263 self.rejected_capability = Some(capability);
264 self
265 }
266
267 /// Record the delay the provider asked for before another attempt.
268 #[must_use]
269 pub fn with_retry_after_secs(mut self, secs: u64) -> Self {
270 self.retry_after_secs = Some(secs);
271 self
272 }
273}
274
275impl std::fmt::Display for LlmError {
276 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
277 f.write_str(&self.message)
278 }
279}
280
281/// Read the provider's machine-readable error code out of a JSON error body.
282///
283/// Both shapes in circulation are accepted: OpenAI-style `error.code` and
284/// Anthropic-style `error.type`. A body that is not JSON, or carries neither,
285/// yields `None` — this never guesses a code out of prose.
286pub(crate) fn provider_error_code_in(body: &str) -> Option<String> {
287 // THREAT[TM-DOS-038]: the body is provider-controlled and arrives from a
288 // failed request, where nothing upstream has bounded it. A body larger
289 // than any real error payload is not parsed at all, and the code taken
290 // out of one that is parsed is length-capped, so neither the parse nor
291 // the retained string is sized by the provider.
292 let body = body.trim();
293 if !body.starts_with('{') || body.len() > MAX_ERROR_BODY_PARSE_BYTES {
294 return None;
295 }
296 let parsed: serde_json::Value = serde_json::from_str(body).ok()?;
297 let error = parsed.get("error")?;
298 let code = error
299 .get("code")
300 .and_then(|value| value.as_str())
301 .or_else(|| error.get("type").and_then(|value| value.as_str()))?;
302 let code = code.trim();
303 if code.is_empty() || code.len() > MAX_ERROR_CODE_BYTES {
304 return None;
305 }
306 Some(code.to_owned())
307}
308
309/// Largest error body parsed to recover a provider error code.
310///
311/// Real provider error payloads are a few hundred bytes; 64 KiB is far above
312/// any of them and far below a body worth spending a JSON parse on.
313const MAX_ERROR_BODY_PARSE_BYTES: usize = 64 * 1024;
314
315/// Largest provider error code retained. Real codes are short identifiers
316/// (`insufficient_quota`, `overloaded_error`); anything longer is not one.
317const MAX_ERROR_CODE_BYTES: usize = 128;