systemprompt_models/wire/canonical/usage.rs
1//! The canonical token-usage types and their conventions.
2//!
3//! # Reasoning tokens
4//!
5//! `CanonicalUsage::reasoning_tokens` is a **breakdown of** `output_tokens`,
6//! never an addition to it. Providers disagree on the wire, so every adapter
7//! normalises to that one rule before the count reaches billing:
8//!
9//! * `OpenAI` (chat and responses) already folds
10//! `*_tokens_details.reasoning_tokens` into `completion_tokens` /
11//! `output_tokens`, so the adapter copies it across untouched.
12//! * Gemini reports `thoughtsTokenCount` *beside* `candidatesTokenCount` (and
13//! inside `totalTokenCount`), so its adapter adds it into `output_tokens` on
14//! the way in.
15//! * Anthropic bills thinking as ordinary output tokens and, for adaptive
16//! thinking on Claude 5 models, reports the share as
17//! `usage.output_tokens_details.thinking_tokens`; the adapter copies it
18//! across untouched. Models that report no details yield 0.
19//!
20//! Holding that invariant here is what makes reasoning billable: cost is
21//! computed from `output_tokens`, so a reasoning-only turn is charged at the
22//! output rate with no per-provider arithmetic downstream, and no count is
23//! charged twice. It is also why `reasoning_tokens` is absent from the
24//! `total_tokens` sum in [`CanonicalUsageUpdate::apply_to`].
25//!
26//! Third-party `OpenAI`-compatible upstreams (Cerebras, Moonshot, Qwen) are not
27//! probed, so `CanonicalUsage::normalise_reasoning` enforces the rule at
28//! runtime rather than trusting it: a breakdown cannot exceed its parent, and a
29//! wire `total_tokens` that overshoots `input + output` by exactly the
30//! reasoning count is the same signal: because `input_tokens` excludes cache
31//! reads, an additive provider's wire total is exactly `billable_total() +
32//! reasoning_tokens`, while a conforming one states `billable_total()` alone.
33//! Either signal means the provider reported reasoning *additionally*, so the
34//! count is folded into `output_tokens` and warned about. The total-based half
35//! fires on both paths: [`CanonicalUsageUpdate`] carries the wire's own
36//! `total_tokens` when a frame states one, and
37//! [`CanonicalUsageUpdate::apply_to`] recomputes only when it does not — and a
38//! recomputed total is `billable_total()`, which is never the additive shape.
39//!
40//! # Cache tokens
41//!
42//! `input_tokens` is **exclusive** of `cache_read_tokens` on every wire.
43//! Anthropic reports the two disjointly; `OpenAI`, Gemini and the
44//! `OpenAI`-compatible upstreams report the cached count as a *subset* of the
45//! prompt count, so their adapters subtract it before it reaches this type.
46//! Billing therefore charges each token exactly once, at exactly one rate, and
47//! `CanonicalUsage::billable_total` is the only definition of `tokens_used`.
48//!
49//! Copyright (c) systemprompt.io — Business Source License 1.1.
50//! See <https://systemprompt.io> for licensing details.
51
52#[derive(Debug, Clone, Copy, Default)]
53#[expect(
54 clippy::struct_field_names,
55 reason = "every field is a token count; the `_tokens` suffix is the domain vocabulary shared \
56 with the provider usage wire formats"
57)]
58pub struct CanonicalUsage {
59 pub input_tokens: u32,
60
61 pub output_tokens: u32,
62 pub cache_read_tokens: u32,
63 pub cache_creation_tokens: u32,
64
65 pub reasoning_tokens: u32,
66
67 pub total_tokens: u32,
68}
69
70impl CanonicalUsage {
71 #[must_use]
72 pub const fn billable_total(&self) -> u32 {
73 self.input_tokens
74 .saturating_add(self.output_tokens)
75 .saturating_add(self.cache_read_tokens)
76 .saturating_add(self.cache_creation_tokens)
77 }
78
79 pub fn normalise_reasoning(&mut self, provider: &str) -> bool {
80 let additive = self.reasoning_tokens > self.output_tokens
81 || (self.reasoning_tokens > 0
82 && self.total_tokens
83 == self.billable_total().saturating_add(self.reasoning_tokens));
84 if !additive {
85 return false;
86 }
87 let folded = self.output_tokens.saturating_add(self.reasoning_tokens);
88 tracing::warn!(
89 provider,
90 reasoning_tokens = self.reasoning_tokens,
91 reported_output_tokens = self.output_tokens,
92 folded_output_tokens = folded,
93 "provider reports reasoning tokens in addition to output tokens; folding them in so \
94 the thinking share is billed"
95 );
96 self.output_tokens = folded;
97 self.total_tokens = self.billable_total();
98 true
99 }
100}
101
102/// A streaming usage report, carrying only the counts its frame actually
103/// stated.
104///
105/// [`CanonicalUsage`] cannot express this: an unreported count and a reported
106/// zero are both `0`. Providers differ in what a mid-stream usage frame
107/// includes — an Anthropic `message_delta` may carry `output_tokens` alone —
108/// so folding one in as though it were complete zeroes the input and cache
109/// counts an earlier frame established, and billing loses them.
110#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
111#[expect(
112 clippy::struct_field_names,
113 reason = "every field is a token count; the `_tokens` suffix is the domain vocabulary shared \
114 with the provider usage wire formats"
115)]
116pub struct CanonicalUsageUpdate {
117 pub input_tokens: Option<u32>,
118 pub output_tokens: Option<u32>,
119 pub cache_read_tokens: Option<u32>,
120 pub cache_creation_tokens: Option<u32>,
121 pub reasoning_tokens: Option<u32>,
122
123 pub total_tokens: Option<u32>,
124}
125
126impl CanonicalUsageUpdate {
127 #[must_use]
128 pub const fn is_empty(&self) -> bool {
129 self.input_tokens.is_none()
130 && self.output_tokens.is_none()
131 && self.cache_read_tokens.is_none()
132 && self.cache_creation_tokens.is_none()
133 && self.reasoning_tokens.is_none()
134 && self.total_tokens.is_none()
135 }
136
137 pub const fn apply_to(&self, usage: &mut CanonicalUsage) {
138 if let Some(v) = self.input_tokens {
139 usage.input_tokens = v;
140 }
141 if let Some(v) = self.output_tokens {
142 usage.output_tokens = v;
143 }
144 if let Some(v) = self.cache_read_tokens {
145 usage.cache_read_tokens = v;
146 }
147 if let Some(v) = self.cache_creation_tokens {
148 usage.cache_creation_tokens = v;
149 }
150 if let Some(v) = self.reasoning_tokens {
151 usage.reasoning_tokens = v;
152 }
153 usage.total_tokens = match self.total_tokens {
154 Some(v) => v,
155 None => usage.billable_total(),
156 };
157 }
158}