turnframe_runtime/budget.rs
1//! The resource budget of the sandboxed autonomous mode (spec §11.1).
2//!
3//! [`OrchestrationMode::SandboxedAutonomous`](crate::config::OrchestrationMode::SandboxedAutonomous)
4//! is the one mode where the model drives writes nobody reviewed, and the
5//! [`ResourceBudget`] it carries is what keeps that from being open-ended. This
6//! module is the enforcement: it measures what a turn has spent and answers one
7//! question — *may the turn keep going?* — with the name of the bound that ran
8//! out when the answer is no.
9//!
10//! # Three bounds, one rule
11//!
12//! | Bound | Measured as | Exhausted when |
13//! | --- | --- | --- |
14//! | [`max_model_calls`](ResourceBudget::max_model_calls) | one [`ProviderAttempt`] per call the turn made, retries and fallbacks included | `spent >= max` |
15//! | [`max_prompt_tokens`](ResourceBudget::max_prompt_tokens) | the prompt tokens the provider reported for every successful attempt | `spent >= max` |
16//! | [`max_wall_clock`](ResourceBudget::max_wall_clock) | the runtime clock, from the moment the turn was accepted | `elapsed >= max` |
17//!
18//! The comparison is `>=`, not `>`, and that is the fail-closed reading: a
19//! budget of eight calls means a turn may make eight, so a turn that has made
20//! eight has nothing left to spend and stops before asking for a ninth. An
21//! attempt that failed still counts — it cost the same call.
22//!
23//! # Where it is checked, and what happens
24//!
25//! [`Orchestrator::handle_turn`](crate::orchestrator::Orchestrator::handle_turn)
26//! checks the budget after interpretation and again before anything executes.
27//! Both are **before** any effect, so an exhausted budget fails the turn
28//! outright: [`BudgetLimit::into_error`] names the bound in a typed
29//! [`PolicyError::BudgetExhausted`], nothing was journaled and nothing can have
30//! happened.
31//!
32//! After the commit the answer is different, and deliberately so. The effects
33//! are real by then, and §23.1 says a turn that committed keeps its effects and
34//! regenerates its wording; failing it to save a token budget would throw away
35//! the only truthful account of what just happened. So a budget that runs out
36//! post-commit stops the *model calls* — narration and answers — and the turn
37//! is delivered with its receipts, its notices and an explicit
38//! [`budget_exhausted`](crate::compose::notice::BUDGET_EXHAUSTED) notice.
39//!
40//! ```
41//! use std::time::Duration;
42//! use turnframe_runtime::budget::{BudgetLimit, BudgetSpend, TurnBudget};
43//! use turnframe_runtime::config::ResourceBudget;
44//!
45//! let budget = ResourceBudget::conservative().with_max_model_calls(2);
46//! let started = chrono::DateTime::UNIX_EPOCH;
47//! let turn = TurnBudget::new(budget, started);
48//!
49//! let spent = BudgetSpend::none().with_model_calls(2);
50//! assert_eq!(turn.exhausted(spent, started), Some(BudgetLimit::ModelCalls));
51//! assert_eq!(turn.exhausted(BudgetSpend::none(), started), None);
52//! ```
53
54use chrono::{DateTime, Utc};
55use turnframe_core::error::{OrchestratorError, PolicyError};
56use turnframe_provider::fallback::{AttemptOutcome, ProviderAttempt};
57
58use crate::config::ResourceBudget;
59
60/// Which bound of a [`ResourceBudget`] ran out.
61#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
62#[non_exhaustive]
63pub enum BudgetLimit {
64 /// The turn made as many model calls as it was allowed.
65 ModelCalls,
66 /// The turn sent as many prompt tokens as it was allowed.
67 PromptTokens,
68 /// The turn took as long as it was allowed.
69 WallClock,
70}
71
72impl BudgetLimit {
73 /// Every bound, in the order they are checked.
74 pub const ALL: [Self; 3] = [Self::ModelCalls, Self::PromptTokens, Self::WallClock];
75
76 /// Stable snake-case name, safe on the wire and in a metric label.
77 #[must_use]
78 pub const fn as_str(self) -> &'static str {
79 match self {
80 Self::ModelCalls => "model_calls",
81 Self::PromptTokens => "prompt_tokens",
82 Self::WallClock => "wall_clock",
83 }
84 }
85
86 /// The typed failure a turn stopped by this bound returns.
87 ///
88 /// It is a [`PolicyError`] rather than an internal failure because that is
89 /// what it is: a configured limit refused to let the turn continue, and the
90 /// caller is entitled to know which one.
91 #[must_use]
92 pub fn into_error(self) -> OrchestratorError {
93 OrchestratorError::Policy(PolicyError::BudgetExhausted {
94 limit: self.as_str().to_owned(),
95 })
96 }
97}
98
99impl std::fmt::Display for BudgetLimit {
100 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
101 f.write_str(self.as_str())
102 }
103}
104
105/// What a turn has spent so far.
106#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
107#[non_exhaustive]
108pub struct BudgetSpend {
109 /// Model calls made, retries and fallbacks included.
110 pub model_calls: u64,
111 /// Prompt tokens the providers reported.
112 pub prompt_tokens: u64,
113}
114
115impl BudgetSpend {
116 /// Nothing spent yet.
117 #[must_use]
118 pub const fn none() -> Self {
119 Self {
120 model_calls: 0,
121 prompt_tokens: 0,
122 }
123 }
124
125 /// Returns a copy with another call count.
126 #[must_use]
127 pub const fn with_model_calls(mut self, model_calls: u64) -> Self {
128 self.model_calls = model_calls;
129 self
130 }
131
132 /// Returns a copy with another token count.
133 #[must_use]
134 pub const fn with_prompt_tokens(mut self, prompt_tokens: u64) -> Self {
135 self.prompt_tokens = prompt_tokens;
136 self
137 }
138
139 /// Measures a turn's attempt trail.
140 ///
141 /// Every attempt is one model call, whatever became of it: a retry after a
142 /// timeout cost the same call as the answer that followed it. Tokens are
143 /// counted only where a provider reported them, because a provider that
144 /// reports nothing has told us nothing, and inventing an estimate would
145 /// make the bound a guess.
146 ///
147 /// A [cancelled](AttemptOutcome::Cancelled) attempt is not counted: the
148 /// caller went away before the provider was asked to do anything.
149 #[must_use]
150 pub fn of(attempts: &[ProviderAttempt]) -> Self {
151 let mut spend = Self::none();
152 for attempt in attempts {
153 if attempt.outcome == AttemptOutcome::Cancelled {
154 continue;
155 }
156 spend.model_calls = spend.model_calls.saturating_add(1);
157 spend.prompt_tokens = spend
158 .prompt_tokens
159 .saturating_add(attempt.input_tokens.unwrap_or(0));
160 }
161 spend
162 }
163}
164
165/// One turn's budget, measured against the clock it started on.
166#[derive(Debug, Clone, Copy, PartialEq, Eq)]
167pub struct TurnBudget {
168 budget: ResourceBudget,
169 started_at: DateTime<Utc>,
170}
171
172impl TurnBudget {
173 /// Binds `budget` to the instant the turn was accepted.
174 #[must_use]
175 pub const fn new(budget: ResourceBudget, started_at: DateTime<Utc>) -> Self {
176 Self { budget, started_at }
177 }
178
179 /// The budget in force.
180 #[must_use]
181 pub const fn budget(&self) -> &ResourceBudget {
182 &self.budget
183 }
184
185 /// When the turn started.
186 #[must_use]
187 pub const fn started_at(&self) -> DateTime<Utc> {
188 self.started_at
189 }
190
191 /// The first bound that is exhausted at `now`, if any.
192 ///
193 /// The order is fixed — calls, then tokens, then the clock — so the same
194 /// turn always reports the same bound, and a replay of it says the same
195 /// thing.
196 #[must_use]
197 pub fn exhausted(&self, spent: BudgetSpend, now: DateTime<Utc>) -> Option<BudgetLimit> {
198 if spent.model_calls >= u64::from(self.budget.max_model_calls) {
199 return Some(BudgetLimit::ModelCalls);
200 }
201 if spent.prompt_tokens >= self.budget.max_prompt_tokens {
202 return Some(BudgetLimit::PromptTokens);
203 }
204 let elapsed = (now - self.started_at).to_std().unwrap_or_default();
205 if elapsed >= self.budget.max_wall_clock {
206 return Some(BudgetLimit::WallClock);
207 }
208 None
209 }
210}
211
212#[cfg(test)]
213mod tests {
214 use std::time::Duration;
215
216 use turnframe_provider::fallback::FallbackStage;
217 use turnframe_provider::ids::{AttemptNumber, ModelRef, RequestId};
218 use turnframe_provider::purpose::ModelPurpose;
219
220 use super::*;
221
222 fn attempt(outcome: AttemptOutcome, input_tokens: Option<u64>) -> ProviderAttempt {
223 ProviderAttempt {
224 attempt: AttemptNumber::FIRST,
225 request_id: RequestId::nil(),
226 purpose: ModelPurpose::Extract,
227 stage: FallbackStage::PreCommit,
228 model: ModelRef::new("p", "m"),
229 outcome,
230 class: None,
231 latency: Duration::ZERO,
232 input_tokens,
233 output_tokens: None,
234 temperature: None,
235 finish_reasons: Vec::new(),
236 }
237 }
238
239 fn started() -> DateTime<Utc> {
240 DateTime::from_timestamp(1_700_000_000, 0).expect("a valid fixed instant")
241 }
242
243 #[test]
244 fn every_attempt_is_a_call_and_only_reported_tokens_count() {
245 let spend = BudgetSpend::of(&[
246 attempt(
247 AttemptOutcome::Retried {
248 code: "timeout".to_owned(),
249 },
250 None,
251 ),
252 attempt(AttemptOutcome::Succeeded, Some(120)),
253 attempt(AttemptOutcome::Cancelled, Some(999)),
254 ]);
255 assert_eq!(
256 spend.model_calls, 2,
257 "a retry cost a call; a cancel did not"
258 );
259 assert_eq!(spend.prompt_tokens, 120);
260 assert_eq!(BudgetSpend::of(&[]), BudgetSpend::none());
261 }
262
263 #[test]
264 fn each_bound_reports_itself_and_the_order_is_fixed() {
265 let budget = ResourceBudget::conservative()
266 .with_max_model_calls(4)
267 .with_max_prompt_tokens(100)
268 .with_max_wall_clock(Duration::from_secs(30));
269 let turn = TurnBudget::new(budget, started());
270 assert_eq!(turn.budget(), &budget);
271 assert_eq!(turn.started_at(), started());
272
273 assert_eq!(turn.exhausted(BudgetSpend::none(), started()), None);
274 assert_eq!(
275 turn.exhausted(BudgetSpend::none().with_model_calls(4), started()),
276 Some(BudgetLimit::ModelCalls)
277 );
278 assert_eq!(
279 turn.exhausted(BudgetSpend::none().with_prompt_tokens(100), started()),
280 Some(BudgetLimit::PromptTokens)
281 );
282 assert_eq!(
283 turn.exhausted(
284 BudgetSpend::none(),
285 started() + chrono::Duration::seconds(30)
286 ),
287 Some(BudgetLimit::WallClock)
288 );
289 // Everything at once still names the first bound in the fixed order.
290 assert_eq!(
291 turn.exhausted(
292 BudgetSpend::none()
293 .with_model_calls(9)
294 .with_prompt_tokens(999),
295 started() + chrono::Duration::seconds(99)
296 ),
297 Some(BudgetLimit::ModelCalls)
298 );
299 // A clock that went backwards is not an exhausted budget.
300 assert_eq!(
301 turn.exhausted(
302 BudgetSpend::none(),
303 started() - chrono::Duration::seconds(5)
304 ),
305 None
306 );
307 }
308
309 #[test]
310 fn a_limit_names_itself_in_the_error_it_raises() {
311 for limit in BudgetLimit::ALL {
312 let error = limit.into_error();
313 assert!(
314 matches!(
315 &error,
316 OrchestratorError::Policy(PolicyError::BudgetExhausted { limit: name })
317 if name == limit.as_str()
318 ),
319 "{limit} did not name itself: {error}"
320 );
321 assert_eq!(limit.to_string(), limit.as_str());
322 }
323 }
324}