use crate::error::AgentLoopError;
const OVERFLOW_WORDING: &[&str] = &[
"context length",
"context_length",
"context window",
"maximum context",
"prompt is too long",
"prompt too long",
"input is too long",
"too many tokens",
"token limit",
"input tokens exceed",
"request too large",
"reduce the length",
];
pub fn looks_like_context_overflow(text: &str) -> bool {
let lower = text.to_ascii_lowercase();
OVERFLOW_WORDING.iter().any(|needle| lower.contains(needle))
}
pub fn provider_error_code(text: &str) -> Option<String> {
const POINTERS: &[&str] = &[
"/error/code",
"/error/status",
"/error/type",
"/code",
"/type",
];
if text.len() > 64 * 1024 {
return None;
}
text.match_indices('{').take(4).find_map(|(start, _)| {
let value: serde_json::Value = serde_json::Deserializer::from_str(&text[start..])
.into_iter()
.next()?
.ok()?;
POINTERS.iter().find_map(|pointer| {
value
.pointer(pointer)
.and_then(serde_json::Value::as_str)
.filter(|code| !code.is_empty() && code.len() <= 128 && *code != "error")
.map(str::to_owned)
})
})
}
pub fn http_status_in(text: &str) -> Option<u16> {
text.match_indices('(').find_map(|(start, _)| {
let rest = text.get(start + 1..)?;
let digits = rest.get(..3)?;
let after = rest[3..].chars().next();
(digits.bytes().all(|b| b.is_ascii_digit()) && matches!(after, Some(' ' | ')')))
.then(|| digits.parse().ok())
.flatten()
.filter(|status| (100..600).contains(status))
})
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum OverflowObservation {
Classified,
SuspectedUnclassified,
None,
}
impl OverflowObservation {
pub fn of(error: &AgentLoopError) -> Self {
if error.is_request_too_large() {
Self::Classified
} else if looks_like_context_overflow(&error.to_string()) {
Self::SuspectedUnclassified
} else {
Self::None
}
}
}
pub fn observe_request_error(
provider: &str,
model: Option<&str>,
http_status: Option<u16>,
error: &AgentLoopError,
) -> OverflowObservation {
let observation = OverflowObservation::of(error);
if observation == OverflowObservation::None {
return observation;
}
let structured = match error {
AgentLoopError::Llm(llm) => Some(llm),
_ => None,
};
let message = error.to_string();
let http_status = http_status
.or(structured.and_then(|llm| llm.status))
.or_else(|| http_status_in(&message));
let code = structured
.and_then(|llm| llm.code.clone())
.or_else(|| provider_error_code(&message));
match observation {
OverflowObservation::Classified => tracing::warn!(
target: "everruns::llm_telemetry",
provider,
model,
http_status,
provider_error_code = code.as_deref(),
classified = true,
"LLM request rejected as too large (context overflow)"
),
OverflowObservation::SuspectedUnclassified => tracing::warn!(
target: "everruns::llm_telemetry",
provider,
model,
http_status,
provider_error_code = code.as_deref(),
classified = false,
overflow_suspected_unclassified = true,
"LLM error reads like a context overflow but was not classified as one"
),
OverflowObservation::None => {}
}
observation
}
pub fn warn_tool_calls_dropped(provider: &str, model: &str, count: u32, stop_reason: &str) {
if count == 0 {
return;
}
tracing::warn!(
target: "everruns::llm_telemetry",
provider,
model,
tool_calls_dropped = count,
stop_reason,
"LLM response ended without completing its tool calls; calls discarded"
);
}
pub fn warn_tool_calls_truncated_executed(
provider: &str,
model: &str,
count: u32,
stop_reason: &str,
) {
if count == 0 {
return;
}
tracing::warn!(
target: "everruns::llm_telemetry",
provider,
model,
tool_calls_truncated_executed = count,
stop_reason,
"LLM tool calls from a truncated response will run; the response was cut off after them"
);
}
pub fn warn_truncation_gate(
provider: &str,
model: &str,
action: &str,
policy: &str,
tool_calls_lost: u32,
consecutive: u32,
finish_reason: &str,
) {
tracing::warn!(
target: "everruns::llm_telemetry",
provider,
model,
truncation_gate = action,
policy,
tool_calls_lost,
consecutive,
finish_reason,
"LLM response lost tool calls; output-truncation gate acted"
);
}
pub fn warn_retry_exhausted(provider: &str, attempts: u32, max_retries: u32, budget: &str) {
tracing::warn!(
target: "everruns::llm_telemetry",
provider,
attempts,
max_retries,
retry_budget = budget,
retry_exhausted = true,
"LLM retry budget exhausted"
);
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn overflow_wording_flags_known_phrases_and_ignores_unrelated_limits() {
for text in [
"prompt is too long: 210000 tokens > 200000 maximum",
"This model's maximum context length is 128000 tokens",
"Input tokens exceed the configured limit of 272000 tokens",
"The input token count exceeds the context window",
] {
assert!(looks_like_context_overflow(text), "{text}");
}
for text in [
"rate limit exceeded",
"Number of tools exceeds the maximum of 128",
"invalid api key",
] {
assert!(!looks_like_context_overflow(text), "{text}");
}
}
#[test]
fn provider_error_code_reads_embedded_json_bodies() {
assert_eq!(
provider_error_code(
r#"OpenAI API error (400): {"error":{"code":"context_length_exceeded","message":"x"}}"#
)
.as_deref(),
Some("context_length_exceeded")
);
assert_eq!(
provider_error_code(
r#"{"type":"error","error":{"type":"invalid_request_error","message":"prompt is too long"}}"#
)
.as_deref(),
Some("invalid_request_error")
);
assert_eq!(
provider_error_code(r#"{"error":{"code":400,"status":"INVALID_ARGUMENT"}}"#).as_deref(),
Some("INVALID_ARGUMENT")
);
assert_eq!(provider_error_code("plain text, no body"), None);
assert_eq!(provider_error_code("{not json"), None);
}
#[test]
fn http_status_is_read_from_driver_error_text() {
assert_eq!(
http_status_in("Anthropic API error (400 Bad Request): {}"),
Some(400)
);
assert_eq!(http_status_in("Gemini API error (413): too big"), Some(413));
assert_eq!(http_status_in("limit (12345 tokens) exceeded"), None);
assert_eq!(http_status_in("(999) nope, (é)"), None);
assert_eq!(http_status_in("no status here"), None);
}
#[test]
fn observation_separates_classified_suspected_and_unrelated_errors() {
assert_eq!(
observe_request_error(
"anthropic",
Some("m"),
Some(400),
&AgentLoopError::request_too_large("prompt is too long")
),
OverflowObservation::Classified
);
assert_eq!(
observe_request_error(
"openai",
None,
Some(400),
&AgentLoopError::llm("upstream: input tokens exceed the context window")
),
OverflowObservation::SuspectedUnclassified
);
assert_eq!(
observe_request_error("openai", None, Some(500), &AgentLoopError::llm("boom")),
OverflowObservation::None
);
}
}