1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
//! BP-7 (catalog §4a "Turn/budget caps" — the *spend* cap; "Per-turn
//! cost/usage accounting" — the *cost* half): token→dollars on the request
//! path.
//!
//! Until BP-7 the only pricing in the workspace was
//! [`crate::pricing_ref`], whose own module doc says it is "not used
//! anywhere on the request path" — which is exactly why the ledger's
//! `turn-budget-caps` row read "no spend/budget cap at all — no pricing
//! exists on the request path". This module is that missing piece, and
//! nothing more: a per-million-token price pair for a model id, resolved
//! from (1) the config's explicit override
//! ([`crate::Config::price_input_per_mtok`] /
//! [`crate::Config::price_output_per_mtok`]) or (2) a small built-in table
//! of published list prices for the model families the parity presets pin.
//!
//! **Deliberately fails closed.** [`resolve`] returns `None` for a model it
//! cannot price rather than guessing, and [`crate::Agent::new`] refuses to
//! build an agent that arms `core.max_budget_usd` against an unpriceable
//! model. A spend cap that silently never bites is worse than no cap: the
//! caller believes they are protected.
/// Per-million-token prices for one model, in US dollars.
#[derive(Debug, Clone, Copy, PartialEq)]
pub struct ModelPrice {
/// Input (prompt) tokens, dollars per million.
pub input_per_mtok: f64,
/// Output (completion) tokens, dollars per million.
pub output_per_mtok: f64,
/// Prompt tokens read from the cache (a cache hit or refresh), dollars per million.
pub cache_read_per_mtok: f64,
/// Prompt tokens written to the 5-minute cache, dollars per million.
pub cache_write_5m_per_mtok: f64,
/// Prompt tokens written to the 1-hour cache, dollars per million.
pub cache_write_1h_per_mtok: f64,
/// What fast mode (`usage.speed: "fast"`) multiplies every rate by, cache rates included; 1 where the model has
/// no fast mode (a request asking for it runs, and is billed, at standard speed).
pub fast_multiplier: f64,
/// What US-only inference (`inference_geo: "us"`) multiplies every rate by: 1.1 on Claude 4.6 and later models,
/// 1 where the model has no such premium.
pub us_only_multiplier: f64,
}
impl ModelPrice {
/// A model priced with the provider's standard cache multipliers: a cache read at 0.1× input, a 5-minute write
/// at 1.25×, a 1-hour write at 2×, and no fast mode.
pub const fn standard(input_per_mtok: f64, output_per_mtok: f64) -> Self {
ModelPrice {
input_per_mtok,
output_per_mtok,
cache_read_per_mtok: input_per_mtok * 0.1,
cache_write_5m_per_mtok: input_per_mtok * 1.25,
cache_write_1h_per_mtok: input_per_mtok * 2.0,
fast_multiplier: 1.0,
us_only_multiplier: 1.0,
}
}
/// Dollar cost of one round-trip's token counts.
///
/// Cached prompt tokens are billed at the full input rate here: the
/// provider-reported discount varies per provider and per cache tier,
/// and over-reporting cost is the safe direction for a *cap* (a budget
/// that stops slightly early never overspends). Named rather than
/// silently assumed — see the `per-turn-cost-usage-accounting` ledger
/// row's note. A recorded session's spend is [`Self::recorded_cost_usd`].
pub fn cost_usd(&self, prompt_tokens: u64, completion_tokens: u64) -> f64 {
(prompt_tokens as f64 / 1_000_000.0) * self.input_per_mtok
+ (completion_tokens as f64 / 1_000_000.0) * self.output_per_mtok
}
/// What a harness's recorded use cost at these list rates: uncached input, output, cache reads and each cache
/// write at its own rate (the counts are disjoint, as Claude's `usage` reports them), all multiplied in fast mode
/// and for US-only inference (the multipliers stack).
pub fn recorded_cost_usd(&self, tokens: &RecordedTokens, fast: bool, us_only: bool) -> f64 {
let rate = |count: u64, per_mtok: f64| count as f64 / 1_000_000.0 * per_mtok;
let standard = rate(tokens.input, self.input_per_mtok)
+ rate(tokens.output, self.output_per_mtok)
+ rate(tokens.cache_read, self.cache_read_per_mtok)
+ rate(tokens.cache_write_5m, self.cache_write_5m_per_mtok)
+ rate(tokens.cache_write_1h, self.cache_write_1h_per_mtok);
let fast = if fast { self.fast_multiplier } else { 1.0 };
let us_only = if us_only {
self.us_only_multiplier
} else {
1.0
};
standard * fast * us_only
}
}
/// Token counts a harness recorded, each kind disjoint from the others.
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
pub struct RecordedTokens {
/// Prompt tokens neither read from nor written to the cache.
pub input: u64,
/// Output tokens (thinking included).
pub output: u64,
/// Prompt tokens read from the cache.
pub cache_read: u64,
/// Prompt tokens written to the 5-minute cache.
pub cache_write_5m: u64,
/// Prompt tokens written to the 1-hour cache.
pub cache_write_1h: u64,
}
impl RecordedTokens {
/// Every prompt and output token, cached or not.
pub fn total(&self) -> u64 {
self.input + self.output + self.cache_read + self.cache_write_5m + self.cache_write_1h
}
/// Add `other`'s counts to these.
pub fn add(&mut self, other: &RecordedTokens) {
self.input += other.input;
self.output += other.output;
self.cache_read += other.cache_read;
self.cache_write_5m += other.cache_write_5m;
self.cache_write_1h += other.cache_write_1h;
}
}
/// One row of the published list: `input`/`output` and the cache rates in dollars per million tokens.
const fn listed(
input: f64,
output: f64,
cache_read: f64,
cache_write_5m: f64,
cache_write_1h: f64,
fast_multiplier: f64,
us_only_multiplier: f64,
) -> ModelPrice {
ModelPrice {
input_per_mtok: input,
output_per_mtok: output,
cache_read_per_mtok: cache_read,
cache_write_5m_per_mtok: cache_write_5m,
cache_write_1h_per_mtok: cache_write_1h,
fast_multiplier,
us_only_multiplier,
}
}
/// Anthropic's published list prices (https://platform.claude.com/docs/en/about-claude/pricing, read 2026-10-03), the
/// last two numbers the fast-mode and US-only multipliers:
/// base input, output, cache hits and refreshes, and 5-minute and 1-hour cache writes. Cache reads are 0.1× input
/// except on Claude Opus 5.5 (0.05×) and Claude Fable 5.1 / Mythos 5.1 (0.025×). Fast mode (Opus 5.5, Opus 5, Opus
/// 4.8) is 2× every rate, with the cache multipliers on top; US-only inference (`inference_geo: "us"`) is 1.1× every
/// rate on Claude 4.6 and later models. Models from 4.6 on bill the whole 1M context at these rates; older models'
/// long-context premium is not modelled. An id this table does not list (`claude-3-7-sonnet`, `claude-3-5-sonnet`,
/// `claude-3-opus`, a `*-latest` alias) is unpriced.
///
/// Each pattern names one model, matched inside a provider's spelling of it (`anthropic/claude-opus-4-8`,
/// `us.anthropic.claude-opus-4-8-v1:0`, `claude-opus-4-5-20251101`, `claude-opus-5-5[1m]`) by [`names_model`]: a
/// pattern never matches a later version of its family (`claude-opus-5` is not `claude-opus-5-5`). A model not listed
/// is unpriced, never priced at a neighbour's rate.
const BUILT_IN_PRICES: &[(&str, ModelPrice)] = &[
("claude-fable-5-1", listed(10.00, 50.00, 0.25, 12.50, 20.00, 1.0, 1.1)),
("claude-mythos-5-1", listed(10.00, 50.00, 0.25, 12.50, 20.00, 1.0, 1.1)),
("claude-fable-5", listed(10.00, 50.00, 1.00, 12.50, 20.00, 1.0, 1.1)),
("claude-mythos-5", listed(10.00, 50.00, 1.00, 12.50, 20.00, 1.0, 1.1)),
("claude-opus-5-5", listed(4.00, 20.00, 0.20, 5.00, 8.00, 2.0, 1.1)),
("claude-opus-5", listed(5.00, 25.00, 0.50, 6.25, 10.00, 2.0, 1.1)),
("claude-opus-4-8", listed(5.00, 25.00, 0.50, 6.25, 10.00, 2.0, 1.1)),
("claude-opus-4-7", listed(5.00, 25.00, 0.50, 6.25, 10.00, 1.0, 1.1)),
("claude-opus-4-6", listed(5.00, 25.00, 0.50, 6.25, 10.00, 1.0, 1.1)),
("claude-opus-4-5", listed(5.00, 25.00, 0.50, 6.25, 10.00, 1.0, 1.0)),
("claude-opus-4-1", listed(15.00, 75.00, 1.50, 18.75, 30.00, 1.0, 1.0)),
("claude-opus-4-0", listed(15.00, 75.00, 1.50, 18.75, 30.00, 1.0, 1.0)),
("claude-opus-4", listed(15.00, 75.00, 1.50, 18.75, 30.00, 1.0, 1.0)),
("claude-sonnet-5-5", listed(2.00, 10.00, 0.20, 2.50, 4.00, 1.0, 1.1)),
("claude-sonnet-5", listed(2.00, 10.00, 0.20, 2.50, 4.00, 1.0, 1.1)),
("claude-sonnet-4-6", listed(3.00, 15.00, 0.30, 3.75, 6.00, 1.0, 1.1)),
("claude-sonnet-4-5", listed(3.00, 15.00, 0.30, 3.75, 6.00, 1.0, 1.0)),
("claude-sonnet-4-0", listed(3.00, 15.00, 0.30, 3.75, 6.00, 1.0, 1.0)),
("claude-sonnet-4", listed(3.00, 15.00, 0.30, 3.75, 6.00, 1.0, 1.0)),
("claude-haiku-4-5", listed(1.00, 5.00, 0.10, 1.25, 2.00, 1.0, 1.0)),
("claude-3-5-haiku", listed(0.80, 4.00, 0.08, 1.00, 1.60, 1.0, 1.0)),
];
/// Whether `model` names the model `pattern` does: the pattern appears in it, and is not followed by a further
/// version segment (`-5`, `-12`) that would make it a later model. A dated snapshot (`-20251101`), a provider suffix
/// (`-v1:0`, `@20251001`) or a context tag (`[1m]`) still names it. A dot separates versions as a dash does
/// (`claude-opus-4.6` is `claude-opus-4-6`): [`built_in`] reads the model with its dots as dashes.
fn names_model(model: &str, pattern: &str) -> bool {
model.match_indices(pattern).any(|(at, _)| {
let rest = &model[at + pattern.len()..];
let Some(after_dash) = rest.strip_prefix('-') else {
return true;
};
let digits = after_dash.chars().take_while(char::is_ascii_digit).count();
!(1..=2).contains(&digits)
})
}
/// The built-in list price for `model`, if this build knows one.
pub fn built_in(model: &str) -> Option<ModelPrice> {
let model = model.replace('.', "-");
BUILT_IN_PRICES
.iter()
.find(|(pattern, _)| names_model(&model, pattern))
.map(|(_, price)| *price)
}
/// Resolve the price to bill `model` at: the explicit config override when
/// BOTH halves are set, else the built-in table, else `None`.
///
/// Both override halves are required together on purpose — an input price
/// with no output price would silently bill completions at zero.
pub fn resolve(
model: &str,
price_input_per_mtok: Option<f64>,
price_output_per_mtok: Option<f64>,
) -> Option<ModelPrice> {
match (price_input_per_mtok, price_output_per_mtok) {
(Some(input_per_mtok), Some(output_per_mtok)) => {
Some(ModelPrice::standard(input_per_mtok, output_per_mtok))
}
_ => built_in(model),
}
}