use crate::ProbeError;
use crate::client::{ProbeClient, ProbeFinish, ProbeRequest};
use crate::types::{ProbeResult, classify};
use super::{refuse_truncated_incomplete, user_text};
pub async fn probe_max_tokens_compliance<C: ProbeClient>(
llm: &C,
) -> Result<ProbeResult, ProbeError> {
let request = ProbeRequest {
messages: vec![user_text("List the numbers 1 through 200, one per line.")],
tools: vec![],
model: llm.model_id().to_string(),
temperature: Some(0.0),
max_tokens: Some(40),
};
let response = llm.chat(request).await?;
if response.finish == ProbeFinish::Length && response.text.trim().is_empty() {
refuse_truncated_incomplete(response.finish, 0.0)?;
}
let char_count = response.text.len();
let (score, details) = if response.text.trim().is_empty() {
(
0.0,
format!("Response {char_count} chars with max_tokens=40 (empty)"),
)
} else if char_count <= 400 {
(
1.0,
format!("Response {char_count} chars with max_tokens=40 (compliant)"),
)
} else if char_count <= 800 {
(
0.5,
format!("Response {char_count} chars with max_tokens=40 (borderline)"),
)
} else {
(
0.0,
format!("Response {char_count} chars with max_tokens=40 (limit ignored)"),
)
};
Ok(ProbeResult {
name: "max_tokens_compliance".to_string(),
score,
max_score: 1.0,
level: classify(score),
details,
})
}
#[cfg(test)]
mod tests {
use super::*;
use crate::probes::test_support::*;
use crate::types::CapabilityLevel;
#[tokio::test]
async fn length_empty_is_transient() {
let llm = MockLlm {
response: length_text_response(""),
};
let err = probe_max_tokens_compliance(&llm)
.await
.expect_err("Length plus empty is not a 30-day Weak card");
assert!(matches!(&err, ProbeError::Transient(_)), "{err:?}");
}
#[tokio::test]
async fn empty_response_is_weak() {
let llm = MockLlm {
response: text_response(""),
};
let result = probe_max_tokens_compliance(&llm).await.unwrap();
assert_eq!(result.level, CapabilityLevel::Weak);
assert_eq!(result.score, 0.0);
}
#[tokio::test]
async fn whitespace_response_is_weak() {
let llm = MockLlm {
response: text_response(" \n"),
};
let result = probe_max_tokens_compliance(&llm).await.unwrap();
assert_eq!(result.level, CapabilityLevel::Weak);
}
#[tokio::test]
async fn compliant_short_response() {
let llm = MockLlm {
response: text_response("1\n2\n3\n4\n5\n6\n7\n8"),
};
let result = probe_max_tokens_compliance(&llm).await.unwrap();
assert_eq!(result.level, CapabilityLevel::Strong);
}
#[tokio::test]
async fn non_compliant_huge_response() {
let long = (1..=500)
.map(|i| i.to_string())
.collect::<Vec<_>>()
.join("\n");
let llm = MockLlm {
response: text_response(&long),
};
let result = probe_max_tokens_compliance(&llm).await.unwrap();
assert_eq!(result.level, CapabilityLevel::Weak);
}
}