1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
//! How a provider failure is classified, and what is preserved of it.
//!
//! [`LlmErrorKind`] is Everruns' taxonomy — what runtime policy retries, what
//! it surfaces, what it refuses. [`LlmError`] carries it alongside the
//! provider's own answer (status, error code, requested retry delay), recorded
//! at the boundary where the HTTP response was still structured. Without those,
//! anything downstream that has to re-express the failure is left scraping the
//! display string, which is not a contract.
//!
//! Split out of [`error`](crate::error) when that file outgrew what anyone can
//! hold in their head; everything here is re-exported from there, so existing
//! paths keep working.
use serde::{Deserialize, Serialize};
use crate::user_facing_error::{
is_attestation_required_message, is_provider_quota_message, is_usage_limit_message,
};
/// Machine-readable reason for provider billing pressure.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum BillingPressureReason {
/// Existing requests temporarily consume the account's available budget.
InFlightBudgetExhausted,
/// The provider account does not have enough credits for the request.
InsufficientCredits,
}
/// Provider feature rejected before a response stream began.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum RejectedProviderCapability {
/// Anthropic threshold server-side compaction.
AnthropicServerCompaction,
}
/// Semantic classification of an LLM provider error, assigned by the driver
/// at the provider boundary where the HTTP status and response body are still
/// available. Downstream consumers prefer this over re-parsing error strings;
/// `LlmErrorKind::Other` falls back to string classification
/// (`classify_runtime_error_message`) so untyped errors keep working.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
#[non_exhaustive]
pub enum LlmErrorKind {
/// Invalid or missing credentials, or access denied (401/403, bad API key).
Authentication,
/// Provider account is out of credits/quota (billing). Non-transient:
/// needs operator action, unlike a regular rate limit.
QuotaExhausted,
/// Provider billing pressure with the stable machine-readable reason and
/// the provider's suggested retry delay. This is non-transient at the
/// runtime layer: consumers decide whether to wait or add credits.
BillingPressure {
/// Stable provider reason for the billing refusal.
reason: BillingPressureReason,
/// Provider-requested delay before another attempt, in seconds.
retry_after_secs: Option<u64>,
},
/// Transient rate limit (429).
RateLimited,
/// Provider outage or unreachable (5xx, 529, network failure).
Unavailable,
/// Provider account has not completed a confirmation the model requires
/// (OpenRouter's 18+ age gate). Non-transient and not a credential
/// problem: it clears when the account holder completes the confirmation,
/// so it is kept apart from `Authentication` even though it arrives as a
/// 403.
AttestationRequired,
/// Provider rejected the request shape (4xx that is not auth/quota/429).
InvalidRequest,
/// The provider answered, but the answer could not be used as a turn: the
/// stream ended before its terminal event, or it exceeded a limit the
/// caller set.
///
/// Distinct from [`Unavailable`](Self::Unavailable) because retrying does
/// not help by itself, and from [`Other`](Self::Other) because the
/// classification is certain — the failure was decided on Everruns' side
/// of the wire, not guessed from provider prose.
MalformedResponse,
/// Unclassified; downstream falls back to string classification.
Other,
}
impl LlmErrorKind {
/// Classify a provider's stable machine-readable error code.
pub fn from_provider_code(code: &str) -> Option<Self> {
let code = code.trim().to_ascii_lowercase();
match code.as_str() {
"subscription_sharing_usage_limit_exceeded"
| "insufficient_quota"
| "billing_hard_limit_reached"
| "credit_balance_too_low"
| "credit_balance_exhausted" => Some(Self::QuotaExhausted),
"subscription_sharing_user_not_eligible"
| "subscription_sharing_invalid_user"
| "chatpass_v2_scope_not_authorized"
| "chatpass_v2_invalid_authorization_context"
| "authentication_error"
| "invalid_api_key"
| "permission_denied" => Some(Self::Authentication),
"rate_limit_exceeded" | "rate_limit_error" | "overloaded_error" => {
Some(Self::RateLimited)
}
"subscription_sharing_usage_unavailable"
| "subscription_sharing_user_unavailable"
| "server_error"
| "internal_error"
| "processing_error"
| "service_unavailable"
| "timeout" => Some(Self::Unavailable),
"subscription_sharing_unsupported_capability"
| "subscription_sharing_route_not_supported"
| "invalid_request_error"
| "model_not_found" => Some(Self::InvalidRequest),
"malformed_response" => Some(Self::MalformedResponse),
_ => None,
}
}
/// Classify a provider HTTP error from status code + response body.
///
/// Quota/billing patterns are checked before the status code because
/// providers surface exhausted billing under different statuses
/// (OpenAI: 429 `insufficient_quota`, Anthropic: 400 "credit balance is
/// too low").
pub fn from_provider_status(status: u16, body: &str) -> Self {
if is_provider_quota_message(body) || is_usage_limit_message(body) {
return LlmErrorKind::QuotaExhausted;
}
// Body-driven for the same reason as quota: the 403 this arrives under
// is indistinguishable from a bad-key 403 by status alone, and the
// gate is worth naming only when the body actually reports one.
if is_attestation_required_message(body) {
return LlmErrorKind::AttestationRequired;
}
match status {
401 | 403 => LlmErrorKind::Authentication,
429 => LlmErrorKind::RateLimited,
408 | 409 => LlmErrorKind::Unavailable,
501 => LlmErrorKind::Other,
500..=599 => LlmErrorKind::Unavailable,
400..=499 => LlmErrorKind::InvalidRequest,
_ => LlmErrorKind::Other,
}
}
/// Keyword-based classification for drivers without an HTTP status at the
/// error site (e.g. Bedrock SDK errors).
pub fn from_error_text(text: &str) -> Self {
if is_provider_quota_message(text) || is_usage_limit_message(text) {
return LlmErrorKind::QuotaExhausted;
}
let lower = text.to_ascii_lowercase();
if lower.contains("throttlingexception")
|| lower.contains("toomanyrequestsexception")
|| lower.contains("rate limit")
|| lower.contains("too many requests")
{
return LlmErrorKind::RateLimited;
}
if lower.contains("accessdeniedexception")
|| lower.contains("unrecognizedclientexception")
|| lower.contains("expiredtokenexception")
|| lower.contains("invalidsignatureexception")
|| lower.contains("unauthorized")
{
return LlmErrorKind::Authentication;
}
if lower.contains("serviceunavailable")
|| lower.contains("service unavailable")
|| lower.contains("internalserverexception")
|| lower.contains("modelnotreadyexception")
{
return LlmErrorKind::Unavailable;
}
LlmErrorKind::Other
}
}
/// LLM provider error with a semantic kind attached by the driver.
///
/// [`kind`](Self::kind) is what runtime policy keys on. The transport fields
/// beside it — [`status`](Self::status), [`code`](Self::code),
/// [`retry_after_secs`](Self::retry_after_secs) — are the provider's own
/// answer, recorded at the boundary where it was still structured. Without
/// them an embedder that has to re-express a failure (an HTTP API in front of
/// Everruns, a retry budget of its own) can only scrape them back out of
/// [`message`](Self::message), which is display text and not a contract.
///
/// `#[non_exhaustive]`: the boundary keeps learning to preserve more, and a
/// field addition must not break construction downstream. Build with
/// [`LlmError::new`] and the `with_*` setters.
#[derive(Debug, Clone, Serialize, Deserialize)]
#[non_exhaustive]
pub struct LlmError {
pub kind: LlmErrorKind,
pub message: String,
/// HTTP status the provider answered with, when the failure arrived as an
/// HTTP response. `None` for SDK, transport, and protocol failures that
/// never carried one.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub status: Option<u16>,
/// The provider's own machine-readable error code, verbatim (OpenAI
/// `error.code`, Anthropic `error.type`). Kept beside `kind` rather than
/// folded into it: `kind` is Everruns' taxonomy, this is the provider's.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub code: Option<String>,
/// Everruns capability that the provider rejected before a response stream
/// started. Runtime fallback policy consumes this marker without parsing
/// provider error prose.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub rejected_capability: Option<RejectedProviderCapability>,
/// Delay the provider asked for before another attempt, in seconds
/// (`Retry-After` or an equivalent rate-limit header).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub retry_after_secs: Option<u64>,
/// Retries already consumed below the turn loop.
#[serde(default)]
pub retry_attempts: u32,
/// Backoff time already consumed below the turn loop.
#[serde(default)]
pub retry_wait_ms: u64,
/// Whether a lower provider layer already made the terminal retry decision.
#[serde(default)]
pub retry_handled: bool,
}
impl LlmError {
/// A provider failure with its semantic kind and nothing else known.
pub fn new(kind: LlmErrorKind, message: impl Into<String>) -> Self {
LlmError {
kind,
message: message.into(),
status: None,
code: None,
rejected_capability: None,
retry_after_secs: None,
retry_attempts: 0,
retry_wait_ms: 0,
retry_handled: false,
}
}
/// Record the HTTP status the provider answered with.
#[must_use]
pub fn with_status(mut self, status: u16) -> Self {
self.status = Some(status);
self
}
/// Record the provider's machine-readable error code.
#[must_use]
pub fn with_code(mut self, code: impl Into<String>) -> Self {
self.code = Some(code.into());
self
}
/// Mark a provider capability as rejected before streaming began.
#[must_use]
pub fn with_rejected_capability(mut self, capability: RejectedProviderCapability) -> Self {
self.rejected_capability = Some(capability);
self
}
/// Record the delay the provider asked for before another attempt.
#[must_use]
pub fn with_retry_after_secs(mut self, secs: u64) -> Self {
self.retry_after_secs = Some(secs);
self
}
}
impl std::fmt::Display for LlmError {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
f.write_str(&self.message)
}
}
/// Read the provider's machine-readable error code out of a JSON error body.
///
/// Both shapes in circulation are accepted: OpenAI-style `error.code` and
/// Anthropic-style `error.type`. A body that is not JSON, or carries neither,
/// yields `None` — this never guesses a code out of prose.
pub(crate) fn provider_error_code_in(body: &str) -> Option<String> {
// THREAT[TM-DOS-038]: the body is provider-controlled and arrives from a
// failed request, where nothing upstream has bounded it. A body larger
// than any real error payload is not parsed at all, and the code taken
// out of one that is parsed is length-capped, so neither the parse nor
// the retained string is sized by the provider.
let body = body.trim();
if !body.starts_with('{') || body.len() > MAX_ERROR_BODY_PARSE_BYTES {
return None;
}
let parsed: serde_json::Value = serde_json::from_str(body).ok()?;
let error = parsed.get("error")?;
let code = error
.get("code")
.and_then(|value| value.as_str())
.or_else(|| error.get("type").and_then(|value| value.as_str()))?;
let code = code.trim();
if code.is_empty() || code.len() > MAX_ERROR_CODE_BYTES {
return None;
}
Some(code.to_owned())
}
/// Largest error body parsed to recover a provider error code.
///
/// Real provider error payloads are a few hundred bytes; 64 KiB is far above
/// any of them and far below a body worth spending a JSON parse on.
const MAX_ERROR_BODY_PARSE_BYTES: usize = 64 * 1024;
/// Largest provider error code retained. Real codes are short identifiers
/// (`insufficient_quota`, `overloaded_error`); anything longer is not one.
const MAX_ERROR_CODE_BYTES: usize = 128;