Skip to main content

supercode_harness/
pricing.rs

1//! BP-7 (catalog §4a "Turn/budget caps" — the *spend* cap; "Per-turn
2//! cost/usage accounting" — the *cost* half): token→dollars on the request
3//! path.
4//!
5//! Until BP-7 the only pricing in the workspace was
6//! [`crate::pricing_ref`], whose own module doc says it is "not used
7//! anywhere on the request path" — which is exactly why the ledger's
8//! `turn-budget-caps` row read "no spend/budget cap at all — no pricing
9//! exists on the request path". This module is that missing piece, and
10//! nothing more: a per-million-token price pair for a model id, resolved
11//! from (1) the config's explicit override
12//! ([`crate::Config::price_input_per_mtok`] /
13//! [`crate::Config::price_output_per_mtok`]) or (2) a small built-in table
14//! of published list prices for the model families the parity presets pin.
15//!
16//! **Deliberately fails closed.** [`resolve`] returns `None` for a model it
17//! cannot price rather than guessing, and [`crate::Agent::new`] refuses to
18//! build an agent that arms `core.max_budget_usd` against an unpriceable
19//! model. A spend cap that silently never bites is worse than no cap: the
20//! caller believes they are protected.
21
22/// Per-million-token prices for one model, in US dollars.
23#[derive(Debug, Clone, Copy, PartialEq)]
24pub struct ModelPrice {
25    /// Input (prompt) tokens, dollars per million.
26    pub input_per_mtok: f64,
27    /// Output (completion) tokens, dollars per million.
28    pub output_per_mtok: f64,
29    /// Prompt tokens read from the cache (a cache hit or refresh), dollars per million.
30    pub cache_read_per_mtok: f64,
31    /// Prompt tokens written to the 5-minute cache, dollars per million.
32    pub cache_write_5m_per_mtok: f64,
33    /// Prompt tokens written to the 1-hour cache, dollars per million.
34    pub cache_write_1h_per_mtok: f64,
35    /// What fast mode (`usage.speed: "fast"`) multiplies every rate by, cache rates included; 1 where the model has
36    /// no fast mode (a request asking for it runs, and is billed, at standard speed).
37    pub fast_multiplier: f64,
38    /// What US-only inference (`inference_geo: "us"`) multiplies every rate by: 1.1 on Claude 4.6 and later models,
39    /// 1 where the model has no such premium.
40    pub us_only_multiplier: f64,
41}
42
43impl ModelPrice {
44    /// A model priced with the provider's standard cache multipliers: a cache read at 0.1× input, a 5-minute write
45    /// at 1.25×, a 1-hour write at 2×, and no fast mode.
46    pub const fn standard(input_per_mtok: f64, output_per_mtok: f64) -> Self {
47        ModelPrice {
48            input_per_mtok,
49            output_per_mtok,
50            cache_read_per_mtok: input_per_mtok * 0.1,
51            cache_write_5m_per_mtok: input_per_mtok * 1.25,
52            cache_write_1h_per_mtok: input_per_mtok * 2.0,
53            fast_multiplier: 1.0,
54            us_only_multiplier: 1.0,
55        }
56    }
57
58    /// Dollar cost of one round-trip's token counts.
59    ///
60    /// Cached prompt tokens are billed at the full input rate here: the
61    /// provider-reported discount varies per provider and per cache tier,
62    /// and over-reporting cost is the safe direction for a *cap* (a budget
63    /// that stops slightly early never overspends). Named rather than
64    /// silently assumed — see the `per-turn-cost-usage-accounting` ledger
65    /// row's note. A recorded session's spend is [`Self::recorded_cost_usd`].
66    pub fn cost_usd(&self, prompt_tokens: u64, completion_tokens: u64) -> f64 {
67        (prompt_tokens as f64 / 1_000_000.0) * self.input_per_mtok
68            + (completion_tokens as f64 / 1_000_000.0) * self.output_per_mtok
69    }
70
71    /// What a harness's recorded use cost at these list rates: uncached input, output, cache reads and each cache
72    /// write at its own rate (the counts are disjoint, as Claude's `usage` reports them), all multiplied in fast mode
73    /// and for US-only inference (the multipliers stack).
74    pub fn recorded_cost_usd(&self, tokens: &RecordedTokens, fast: bool, us_only: bool) -> f64 {
75        let rate = |count: u64, per_mtok: f64| count as f64 / 1_000_000.0 * per_mtok;
76        let standard = rate(tokens.input, self.input_per_mtok)
77            + rate(tokens.output, self.output_per_mtok)
78            + rate(tokens.cache_read, self.cache_read_per_mtok)
79            + rate(tokens.cache_write_5m, self.cache_write_5m_per_mtok)
80            + rate(tokens.cache_write_1h, self.cache_write_1h_per_mtok);
81        let fast = if fast { self.fast_multiplier } else { 1.0 };
82        let us_only = if us_only {
83            self.us_only_multiplier
84        } else {
85            1.0
86        };
87        standard * fast * us_only
88    }
89}
90
91/// Token counts a harness recorded, each kind disjoint from the others.
92#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
93pub struct RecordedTokens {
94    /// Prompt tokens neither read from nor written to the cache.
95    pub input: u64,
96    /// Output tokens (thinking included).
97    pub output: u64,
98    /// Prompt tokens read from the cache.
99    pub cache_read: u64,
100    /// Prompt tokens written to the 5-minute cache.
101    pub cache_write_5m: u64,
102    /// Prompt tokens written to the 1-hour cache.
103    pub cache_write_1h: u64,
104}
105
106impl RecordedTokens {
107    /// Every prompt and output token, cached or not.
108    pub fn total(&self) -> u64 {
109        self.input + self.output + self.cache_read + self.cache_write_5m + self.cache_write_1h
110    }
111
112    /// Add `other`'s counts to these.
113    pub fn add(&mut self, other: &RecordedTokens) {
114        self.input += other.input;
115        self.output += other.output;
116        self.cache_read += other.cache_read;
117        self.cache_write_5m += other.cache_write_5m;
118        self.cache_write_1h += other.cache_write_1h;
119    }
120}
121
122/// One row of the published list: `input`/`output` and the cache rates in dollars per million tokens.
123const fn listed(
124    input: f64,
125    output: f64,
126    cache_read: f64,
127    cache_write_5m: f64,
128    cache_write_1h: f64,
129    fast_multiplier: f64,
130    us_only_multiplier: f64,
131) -> ModelPrice {
132    ModelPrice {
133        input_per_mtok: input,
134        output_per_mtok: output,
135        cache_read_per_mtok: cache_read,
136        cache_write_5m_per_mtok: cache_write_5m,
137        cache_write_1h_per_mtok: cache_write_1h,
138        fast_multiplier,
139        us_only_multiplier,
140    }
141}
142
143/// Anthropic's published list prices (https://platform.claude.com/docs/en/about-claude/pricing, read 2026-10-03), the
144/// last two numbers the fast-mode and US-only multipliers:
145/// base input, output, cache hits and refreshes, and 5-minute and 1-hour cache writes. Cache reads are 0.1× input
146/// except on Claude Opus 5.5 (0.05×) and Claude Fable 5.1 / Mythos 5.1 (0.025×). Fast mode (Opus 5.5, Opus 5, Opus
147/// 4.8) is 2× every rate, with the cache multipliers on top; US-only inference (`inference_geo: "us"`) is 1.1× every
148/// rate on Claude 4.6 and later models. Models from 4.6 on bill the whole 1M context at these rates; older models'
149/// long-context premium is not modelled. An id this table does not list (`claude-3-7-sonnet`, `claude-3-5-sonnet`,
150/// `claude-3-opus`, a `*-latest` alias) is unpriced.
151///
152/// Each pattern names one model, matched inside a provider's spelling of it (`anthropic/claude-opus-4-8`,
153/// `us.anthropic.claude-opus-4-8-v1:0`, `claude-opus-4-5-20251101`, `claude-opus-5-5[1m]`) by [`names_model`]: a
154/// pattern never matches a later version of its family (`claude-opus-5` is not `claude-opus-5-5`). A model not listed
155/// is unpriced, never priced at a neighbour's rate.
156const BUILT_IN_PRICES: &[(&str, ModelPrice)] = &[
157    ("claude-fable-5-1", listed(10.00, 50.00, 0.25, 12.50, 20.00, 1.0, 1.1)),
158    ("claude-mythos-5-1", listed(10.00, 50.00, 0.25, 12.50, 20.00, 1.0, 1.1)),
159    ("claude-fable-5", listed(10.00, 50.00, 1.00, 12.50, 20.00, 1.0, 1.1)),
160    ("claude-mythos-5", listed(10.00, 50.00, 1.00, 12.50, 20.00, 1.0, 1.1)),
161    ("claude-opus-5-5", listed(4.00, 20.00, 0.20, 5.00, 8.00, 2.0, 1.1)),
162    ("claude-opus-5", listed(5.00, 25.00, 0.50, 6.25, 10.00, 2.0, 1.1)),
163    ("claude-opus-4-8", listed(5.00, 25.00, 0.50, 6.25, 10.00, 2.0, 1.1)),
164    ("claude-opus-4-7", listed(5.00, 25.00, 0.50, 6.25, 10.00, 1.0, 1.1)),
165    ("claude-opus-4-6", listed(5.00, 25.00, 0.50, 6.25, 10.00, 1.0, 1.1)),
166    ("claude-opus-4-5", listed(5.00, 25.00, 0.50, 6.25, 10.00, 1.0, 1.0)),
167    ("claude-opus-4-1", listed(15.00, 75.00, 1.50, 18.75, 30.00, 1.0, 1.0)),
168    ("claude-opus-4-0", listed(15.00, 75.00, 1.50, 18.75, 30.00, 1.0, 1.0)),
169    ("claude-opus-4", listed(15.00, 75.00, 1.50, 18.75, 30.00, 1.0, 1.0)),
170    ("claude-sonnet-5-5", listed(2.00, 10.00, 0.20, 2.50, 4.00, 1.0, 1.1)),
171    ("claude-sonnet-5", listed(2.00, 10.00, 0.20, 2.50, 4.00, 1.0, 1.1)),
172    ("claude-sonnet-4-6", listed(3.00, 15.00, 0.30, 3.75, 6.00, 1.0, 1.1)),
173    ("claude-sonnet-4-5", listed(3.00, 15.00, 0.30, 3.75, 6.00, 1.0, 1.0)),
174    ("claude-sonnet-4-0", listed(3.00, 15.00, 0.30, 3.75, 6.00, 1.0, 1.0)),
175    ("claude-sonnet-4", listed(3.00, 15.00, 0.30, 3.75, 6.00, 1.0, 1.0)),
176    ("claude-haiku-4-5", listed(1.00, 5.00, 0.10, 1.25, 2.00, 1.0, 1.0)),
177    ("claude-3-5-haiku", listed(0.80, 4.00, 0.08, 1.00, 1.60, 1.0, 1.0)),
178];
179
180/// Whether `model` names the model `pattern` does: the pattern appears in it, and is not followed by a further
181/// version segment (`-5`, `-12`) that would make it a later model. A dated snapshot (`-20251101`), a provider suffix
182/// (`-v1:0`, `@20251001`) or a context tag (`[1m]`) still names it. A dot separates versions as a dash does
183/// (`claude-opus-4.6` is `claude-opus-4-6`): [`built_in`] reads the model with its dots as dashes.
184fn names_model(model: &str, pattern: &str) -> bool {
185    model.match_indices(pattern).any(|(at, _)| {
186        let rest = &model[at + pattern.len()..];
187        let Some(after_dash) = rest.strip_prefix('-') else {
188            return true;
189        };
190        let digits = after_dash.chars().take_while(char::is_ascii_digit).count();
191        !(1..=2).contains(&digits)
192    })
193}
194
195/// The built-in list price for `model`, if this build knows one.
196pub fn built_in(model: &str) -> Option<ModelPrice> {
197    let model = model.replace('.', "-");
198    BUILT_IN_PRICES
199        .iter()
200        .find(|(pattern, _)| names_model(&model, pattern))
201        .map(|(_, price)| *price)
202}
203
204/// Resolve the price to bill `model` at: the explicit config override when
205/// BOTH halves are set, else the built-in table, else `None`.
206///
207/// Both override halves are required together on purpose — an input price
208/// with no output price would silently bill completions at zero.
209pub fn resolve(
210    model: &str,
211    price_input_per_mtok: Option<f64>,
212    price_output_per_mtok: Option<f64>,
213) -> Option<ModelPrice> {
214    match (price_input_per_mtok, price_output_per_mtok) {
215        (Some(input_per_mtok), Some(output_per_mtok)) => {
216            Some(ModelPrice::standard(input_per_mtok, output_per_mtok))
217        }
218        _ => built_in(model),
219    }
220}