supercode_harness/pricing.rs
1//! BP-7 (catalog §4a "Turn/budget caps" — the *spend* cap; "Per-turn
2//! cost/usage accounting" — the *cost* half): token→dollars on the request
3//! path.
4//!
5//! Until BP-7 the only pricing in the workspace was
6//! [`crate::pricing_ref`], whose own module doc says it is "not used
7//! anywhere on the request path" — which is exactly why the ledger's
8//! `turn-budget-caps` row read "no spend/budget cap at all — no pricing
9//! exists on the request path". This module is that missing piece, and
10//! nothing more: a per-million-token price pair for a model id, resolved
11//! from (1) the config's explicit override
12//! ([`crate::Config::price_input_per_mtok`] /
13//! [`crate::Config::price_output_per_mtok`]) or (2) a small built-in table
14//! of published list prices for the model families the parity presets pin.
15//!
16//! **Deliberately fails closed.** [`resolve`] returns `None` for a model it
17//! cannot price rather than guessing, and [`crate::Agent::new`] refuses to
18//! build an agent that arms `core.max_budget_usd` against an unpriceable
19//! model. A spend cap that silently never bites is worse than no cap: the
20//! caller believes they are protected.
21
22/// Per-million-token prices for one model, in US dollars.
23#[derive(Debug, Clone, Copy, PartialEq)]
24pub struct ModelPrice {
25 /// Input (prompt) tokens, dollars per million.
26 pub input_per_mtok: f64,
27 /// Output (completion) tokens, dollars per million.
28 pub output_per_mtok: f64,
29 /// Prompt tokens read from the cache (a cache hit or refresh), dollars per million.
30 pub cache_read_per_mtok: f64,
31 /// Prompt tokens written to the 5-minute cache, dollars per million.
32 pub cache_write_5m_per_mtok: f64,
33 /// Prompt tokens written to the 1-hour cache, dollars per million.
34 pub cache_write_1h_per_mtok: f64,
35 /// What fast mode (`usage.speed: "fast"`) multiplies every rate by, cache rates included; 1 where the model has
36 /// no fast mode (a request asking for it runs, and is billed, at standard speed).
37 pub fast_multiplier: f64,
38 /// What US-only inference (`inference_geo: "us"`) multiplies every rate by: 1.1 on Claude 4.6 and later models,
39 /// 1 where the model has no such premium.
40 pub us_only_multiplier: f64,
41}
42
43impl ModelPrice {
44 /// A model priced with the provider's standard cache multipliers: a cache read at 0.1× input, a 5-minute write
45 /// at 1.25×, a 1-hour write at 2×, and no fast mode.
46 pub const fn standard(input_per_mtok: f64, output_per_mtok: f64) -> Self {
47 ModelPrice {
48 input_per_mtok,
49 output_per_mtok,
50 cache_read_per_mtok: input_per_mtok * 0.1,
51 cache_write_5m_per_mtok: input_per_mtok * 1.25,
52 cache_write_1h_per_mtok: input_per_mtok * 2.0,
53 fast_multiplier: 1.0,
54 us_only_multiplier: 1.0,
55 }
56 }
57
58 /// Dollar cost of one round-trip's token counts.
59 ///
60 /// Cached prompt tokens are billed at the full input rate here: the
61 /// provider-reported discount varies per provider and per cache tier,
62 /// and over-reporting cost is the safe direction for a *cap* (a budget
63 /// that stops slightly early never overspends). Named rather than
64 /// silently assumed — see the `per-turn-cost-usage-accounting` ledger
65 /// row's note. A recorded session's spend is [`Self::recorded_cost_usd`].
66 pub fn cost_usd(&self, prompt_tokens: u64, completion_tokens: u64) -> f64 {
67 (prompt_tokens as f64 / 1_000_000.0) * self.input_per_mtok
68 + (completion_tokens as f64 / 1_000_000.0) * self.output_per_mtok
69 }
70
71 /// What a harness's recorded use cost at these list rates: uncached input, output, cache reads and each cache
72 /// write at its own rate (the counts are disjoint, as Claude's `usage` reports them), all multiplied in fast mode
73 /// and for US-only inference (the multipliers stack).
74 pub fn recorded_cost_usd(&self, tokens: &RecordedTokens, fast: bool, us_only: bool) -> f64 {
75 let rate = |count: u64, per_mtok: f64| count as f64 / 1_000_000.0 * per_mtok;
76 let standard = rate(tokens.input, self.input_per_mtok)
77 + rate(tokens.output, self.output_per_mtok)
78 + rate(tokens.cache_read, self.cache_read_per_mtok)
79 + rate(tokens.cache_write_5m, self.cache_write_5m_per_mtok)
80 + rate(tokens.cache_write_1h, self.cache_write_1h_per_mtok);
81 let fast = if fast { self.fast_multiplier } else { 1.0 };
82 let us_only = if us_only {
83 self.us_only_multiplier
84 } else {
85 1.0
86 };
87 standard * fast * us_only
88 }
89}
90
91/// Token counts a harness recorded, each kind disjoint from the others.
92#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
93pub struct RecordedTokens {
94 /// Prompt tokens neither read from nor written to the cache.
95 pub input: u64,
96 /// Output tokens (thinking included).
97 pub output: u64,
98 /// Prompt tokens read from the cache.
99 pub cache_read: u64,
100 /// Prompt tokens written to the 5-minute cache.
101 pub cache_write_5m: u64,
102 /// Prompt tokens written to the 1-hour cache.
103 pub cache_write_1h: u64,
104}
105
106impl RecordedTokens {
107 /// Every prompt and output token, cached or not.
108 pub fn total(&self) -> u64 {
109 self.input + self.output + self.cache_read + self.cache_write_5m + self.cache_write_1h
110 }
111
112 /// Add `other`'s counts to these.
113 pub fn add(&mut self, other: &RecordedTokens) {
114 self.input += other.input;
115 self.output += other.output;
116 self.cache_read += other.cache_read;
117 self.cache_write_5m += other.cache_write_5m;
118 self.cache_write_1h += other.cache_write_1h;
119 }
120}
121
122/// One row of the published list: `input`/`output` and the cache rates in dollars per million tokens.
123const fn listed(
124 input: f64,
125 output: f64,
126 cache_read: f64,
127 cache_write_5m: f64,
128 cache_write_1h: f64,
129 fast_multiplier: f64,
130 us_only_multiplier: f64,
131) -> ModelPrice {
132 ModelPrice {
133 input_per_mtok: input,
134 output_per_mtok: output,
135 cache_read_per_mtok: cache_read,
136 cache_write_5m_per_mtok: cache_write_5m,
137 cache_write_1h_per_mtok: cache_write_1h,
138 fast_multiplier,
139 us_only_multiplier,
140 }
141}
142
143/// Anthropic's published list prices (https://platform.claude.com/docs/en/about-claude/pricing, read 2026-10-03), the
144/// last two numbers the fast-mode and US-only multipliers:
145/// base input, output, cache hits and refreshes, and 5-minute and 1-hour cache writes. Cache reads are 0.1× input
146/// except on Claude Opus 5.5 (0.05×) and Claude Fable 5.1 / Mythos 5.1 (0.025×). Fast mode (Opus 5.5, Opus 5, Opus
147/// 4.8) is 2× every rate, with the cache multipliers on top; US-only inference (`inference_geo: "us"`) is 1.1× every
148/// rate on Claude 4.6 and later models. Models from 4.6 on bill the whole 1M context at these rates; older models'
149/// long-context premium is not modelled. An id this table does not list (`claude-3-7-sonnet`, `claude-3-5-sonnet`,
150/// `claude-3-opus`, a `*-latest` alias) is unpriced.
151///
152/// Each pattern names one model, matched inside a provider's spelling of it (`anthropic/claude-opus-4-8`,
153/// `us.anthropic.claude-opus-4-8-v1:0`, `claude-opus-4-5-20251101`, `claude-opus-5-5[1m]`) by [`names_model`]: a
154/// pattern never matches a later version of its family (`claude-opus-5` is not `claude-opus-5-5`). A model not listed
155/// is unpriced, never priced at a neighbour's rate.
156const BUILT_IN_PRICES: &[(&str, ModelPrice)] = &[
157 ("claude-fable-5-1", listed(10.00, 50.00, 0.25, 12.50, 20.00, 1.0, 1.1)),
158 ("claude-mythos-5-1", listed(10.00, 50.00, 0.25, 12.50, 20.00, 1.0, 1.1)),
159 ("claude-fable-5", listed(10.00, 50.00, 1.00, 12.50, 20.00, 1.0, 1.1)),
160 ("claude-mythos-5", listed(10.00, 50.00, 1.00, 12.50, 20.00, 1.0, 1.1)),
161 ("claude-opus-5-5", listed(4.00, 20.00, 0.20, 5.00, 8.00, 2.0, 1.1)),
162 ("claude-opus-5", listed(5.00, 25.00, 0.50, 6.25, 10.00, 2.0, 1.1)),
163 ("claude-opus-4-8", listed(5.00, 25.00, 0.50, 6.25, 10.00, 2.0, 1.1)),
164 ("claude-opus-4-7", listed(5.00, 25.00, 0.50, 6.25, 10.00, 1.0, 1.1)),
165 ("claude-opus-4-6", listed(5.00, 25.00, 0.50, 6.25, 10.00, 1.0, 1.1)),
166 ("claude-opus-4-5", listed(5.00, 25.00, 0.50, 6.25, 10.00, 1.0, 1.0)),
167 ("claude-opus-4-1", listed(15.00, 75.00, 1.50, 18.75, 30.00, 1.0, 1.0)),
168 ("claude-opus-4-0", listed(15.00, 75.00, 1.50, 18.75, 30.00, 1.0, 1.0)),
169 ("claude-opus-4", listed(15.00, 75.00, 1.50, 18.75, 30.00, 1.0, 1.0)),
170 ("claude-sonnet-5-5", listed(2.00, 10.00, 0.20, 2.50, 4.00, 1.0, 1.1)),
171 ("claude-sonnet-5", listed(2.00, 10.00, 0.20, 2.50, 4.00, 1.0, 1.1)),
172 ("claude-sonnet-4-6", listed(3.00, 15.00, 0.30, 3.75, 6.00, 1.0, 1.1)),
173 ("claude-sonnet-4-5", listed(3.00, 15.00, 0.30, 3.75, 6.00, 1.0, 1.0)),
174 ("claude-sonnet-4-0", listed(3.00, 15.00, 0.30, 3.75, 6.00, 1.0, 1.0)),
175 ("claude-sonnet-4", listed(3.00, 15.00, 0.30, 3.75, 6.00, 1.0, 1.0)),
176 ("claude-haiku-4-5", listed(1.00, 5.00, 0.10, 1.25, 2.00, 1.0, 1.0)),
177 ("claude-3-5-haiku", listed(0.80, 4.00, 0.08, 1.00, 1.60, 1.0, 1.0)),
178];
179
180/// Whether `model` names the model `pattern` does: the pattern appears in it, and is not followed by a further
181/// version segment (`-5`, `-12`) that would make it a later model. A dated snapshot (`-20251101`), a provider suffix
182/// (`-v1:0`, `@20251001`) or a context tag (`[1m]`) still names it. A dot separates versions as a dash does
183/// (`claude-opus-4.6` is `claude-opus-4-6`): [`built_in`] reads the model with its dots as dashes.
184fn names_model(model: &str, pattern: &str) -> bool {
185 model.match_indices(pattern).any(|(at, _)| {
186 let rest = &model[at + pattern.len()..];
187 let Some(after_dash) = rest.strip_prefix('-') else {
188 return true;
189 };
190 let digits = after_dash.chars().take_while(char::is_ascii_digit).count();
191 !(1..=2).contains(&digits)
192 })
193}
194
195/// The built-in list price for `model`, if this build knows one.
196pub fn built_in(model: &str) -> Option<ModelPrice> {
197 let model = model.replace('.', "-");
198 BUILT_IN_PRICES
199 .iter()
200 .find(|(pattern, _)| names_model(&model, pattern))
201 .map(|(_, price)| *price)
202}
203
204/// Resolve the price to bill `model` at: the explicit config override when
205/// BOTH halves are set, else the built-in table, else `None`.
206///
207/// Both override halves are required together on purpose — an input price
208/// with no output price would silently bill completions at zero.
209pub fn resolve(
210 model: &str,
211 price_input_per_mtok: Option<f64>,
212 price_output_per_mtok: Option<f64>,
213) -> Option<ModelPrice> {
214 match (price_input_per_mtok, price_output_per_mtok) {
215 (Some(input_per_mtok), Some(output_per_mtok)) => {
216 Some(ModelPrice::standard(input_per_mtok, output_per_mtok))
217 }
218 _ => built_in(model),
219 }
220}