rig_core/catalog/spec.rs
1//! What the catalog knows about one model, and the checks it makes against
2//! [`GenerationOptions`].
3
4use std::ops::RangeInclusive;
5
6use serde::{Deserialize, Serialize};
7
8use crate::completion::{
9 CacheRetention, Cost, Effort, GenerationOptions, Reasoning, UnsupportedOption, Usage,
10};
11use crate::providers::registry::ProviderId;
12
13/// One model's facts: limits, input modalities, the reasoning and caching it
14/// takes, and its prices. Built by [`Catalog`](super::Catalog) from its data.
15#[non_exhaustive]
16#[derive(Clone, Debug, PartialEq)]
17pub struct ModelSpec {
18 /// The provider's own model id.
19 pub id: String,
20 /// The provider serving the model under [`Self::id`].
21 pub provider: ProviderId,
22 /// A human-readable name.
23 pub display_name: String,
24 /// The most tokens the model reads and writes in one request, if known.
25 pub context_window: Option<u32>,
26 /// The most tokens the model writes in one reply, if known.
27 pub max_output_tokens: Option<u32>,
28 /// What the model reads.
29 pub input: Modalities,
30 /// The reasoning the model takes.
31 pub reasoning: ReasoningSupport,
32 /// The cache retentions the model honours.
33 pub caching: CacheSupport,
34 /// Whether the model calls tools.
35 pub tools: bool,
36 /// Whether the model constrains its output to a JSON schema.
37 pub structured_output: bool,
38 /// Prices in USD per million tokens, if known.
39 pub pricing: Option<Pricing>,
40 /// Whether the provider has deprecated the model.
41 pub deprecated: bool,
42 /// When the model takes sampling parameters (`temperature`, `top_p`,
43 /// `top_logprobs`, `logprobs`), or `None` when unknown.
44 pub sampling: Option<Sampling>,
45 /// Facts the encoders read that no portable field holds.
46 ///
47 /// For rig's own crates; not covered by semver.
48 #[doc(hidden)]
49 pub compat: Compat,
50}
51
52/// What a model reads.
53#[non_exhaustive]
54#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Hash)]
55pub struct Modalities {
56 /// Text.
57 pub text: bool,
58 /// Images.
59 pub image: bool,
60 /// Audio.
61 pub audio: bool,
62 /// Video.
63 pub video: bool,
64 /// PDF documents.
65 pub pdf: bool,
66}
67
68/// The reasoning a model takes. A row that marks a model as reasoning but
69/// lists no reasoning options (193 rows of the built-in catalog, most of
70/// them gateway rows on HuggingFace, OpenRouter, Venice and Bedrock) has no
71/// levels, no budget and cannot disable, so [`ModelSpec::validate`] refuses
72/// every effort, every budget and `Off` on it; unlike [`CacheSupport`],
73/// empty here does not mean unknown.
74#[non_exhaustive]
75#[derive(Clone, Debug, Default, PartialEq, Eq)]
76pub struct ReasoningSupport {
77 /// Whether the model reasons at all.
78 pub supported: bool,
79 /// The effort levels it takes, empty when it takes none.
80 pub levels: Vec<Effort>,
81 /// The reasoning-token budgets it takes, if it takes one.
82 pub budget: Option<RangeInclusive<u32>>,
83 /// Whether a request can turn its reasoning off.
84 pub can_disable: bool,
85 /// The effort it uses when a request names none, if documented.
86 pub default: Option<Effort>,
87}
88
89/// The cache retentions a model honours.
90#[non_exhaustive]
91#[derive(Clone, Debug, Default, PartialEq, Eq)]
92pub struct CacheSupport {
93 /// The [`CacheRetention`] values the model honours. Empty when the
94 /// catalog does not know, in which case nothing is refused.
95 pub retention: Vec<CacheRetention>,
96}
97
98/// Prices in USD per million tokens.
99#[non_exhaustive]
100#[derive(Clone, Copy, Debug, PartialEq)]
101pub struct Pricing {
102 /// Uncached input.
103 pub input: f64,
104 /// Output, reasoning included.
105 pub output: f64,
106 /// Input read from the cache, or `None` when unknown.
107 pub cache_read: Option<f64>,
108 /// Input written to the cache, or `None` when unknown.
109 pub cache_write: Option<f64>,
110}
111
112impl Pricing {
113 /// What `usage` costs at these prices, or `None` unless it reports both
114 /// its input and output tokens. Uncached input is the input tokens less
115 /// those read from and written to the cache; cache reads and writes
116 /// with no price of their own are charged at [`Self::input`]. Every
117 /// cache write is charged at one price, so a provider that bills longer
118 /// retention higher costs more than this says. It is the standard-tier
119 /// list price of the tokens: the service tier, long-context price tiers
120 /// and hosted-tool fees (web search, code execution) are not in it.
121 pub fn cost(&self, usage: &Usage) -> Option<Cost> {
122 let input = usage.input_tokens?;
123 let output = usage.output_tokens?;
124 let read = usage.cached_input_tokens.unwrap_or(0);
125 let written = usage.cache_creation_input_tokens.unwrap_or(0);
126 let uncached = input.saturating_sub(read).saturating_sub(written);
127 // Token counts stay far below 2^53, so the conversion is exact.
128 let price = |tokens: u64, per_million: f64| tokens as f64 * per_million / 1_000_000.0;
129 Some(Cost::from_parts(
130 price(uncached, self.input),
131 price(output, self.output),
132 price(read, self.cache_read.unwrap_or(self.input)),
133 price(written, self.cache_write.unwrap_or(self.input)),
134 ))
135 }
136}
137
138/// When a model takes sampling parameters.
139#[non_exhaustive]
140#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, Serialize, Deserialize)]
141#[serde(rename_all = "snake_case")]
142pub enum Sampling {
143 /// Always.
144 Any,
145 /// Only while its reasoning is off.
146 ReasoningOff,
147 /// Never.
148 Never,
149}
150
151/// Model facts the encoders read that no portable field holds. Each defaults
152/// to `false` or `None`, which is what a model the field does not concern
153/// has.
154///
155/// For rig's own crates; not covered by semver.
156#[doc(hidden)]
157#[non_exhaustive]
158#[derive(Clone, Debug, Default, PartialEq, Eq)]
159pub struct Compat {
160 /// The field an OpenAI Chat Completions assistant message must carry
161 /// its reasoning under, such as `reasoning_content`.
162 pub reasoning_field: Option<String>,
163 /// Anthropic: an effort level goes with `thinking: {"type": "adaptive"}`.
164 pub adaptive_thinking: bool,
165 /// Anthropic: the `thinking.type` that turns reasoning off, where it is
166 /// not `disabled`.
167 pub thinking_off: Option<String>,
168 /// Anthropic: the model takes `role: "system"` inside `messages`.
169 pub mid_conversation_system: bool,
170 /// Anthropic: the model answers a forced `tool_choice` with an error.
171 pub rejects_forced_tool_choice: bool,
172 /// Anthropic: the model binds its thinking to the request's tools and
173 /// system prompt.
174 pub binds_context: bool,
175 /// OpenAI: the prompt cache takes `prompt_cache_options`, not
176 /// `prompt_cache_retention`.
177 pub prompt_cache_options: bool,
178 /// OpenAI: Chat Completions takes the model's tools only while its
179 /// reasoning is off.
180 pub chat_tools_need_reasoning_off: bool,
181}
182
183impl ModelSpec {
184 /// Checks `options` against what the model takes: `reasoning` against
185 /// [`Self::reasoning`] and `cache` against [`Self::caching`]. The error
186 /// names the option, the provider's vendor and this model. A reasoning
187 /// row with no reasoning options refuses every `reasoning` value (see
188 /// [`ReasoningSupport`]).
189 pub fn validate(&self, options: &GenerationOptions) -> Result<(), UnsupportedOption> {
190 let refuse = |option: &'static str, reason: String| {
191 UnsupportedOption::new(option, self.provider.vendor(), &self.id, reason)
192 };
193 if let Some(reason) = options
194 .reasoning
195 .as_ref()
196 .and_then(|reasoning| self.reasoning.refusal(reasoning))
197 {
198 return Err(refuse("reasoning", reason));
199 }
200 if let Some(reason) = options
201 .cache
202 .as_ref()
203 .and_then(|cache| self.caching.refusal(cache))
204 {
205 return Err(refuse("cache", reason));
206 }
207 Ok(())
208 }
209}
210
211impl ReasoningSupport {
212 /// Why the model cannot take `reasoning`, or `None` when it can.
213 pub fn refusal(&self, reasoning: &Reasoning) -> Option<String> {
214 match reasoning {
215 Reasoning::Off if !self.supported || self.can_disable => None,
216 Reasoning::Off => Some("reasoning cannot be turned off on this model".to_owned()),
217 _ if !self.supported => Some("the model does not reason".to_owned()),
218 Reasoning::Effort(effort) if self.levels.contains(effort) => None,
219 Reasoning::Effort(_) if self.levels.is_empty() && self.budget.is_some() => {
220 Some("the model takes a reasoning budget, not an effort level".to_owned())
221 }
222 Reasoning::Effort(_) if self.levels.is_empty() => {
223 Some("the model takes no effort level".to_owned())
224 }
225 Reasoning::Effort(effort) => Some(format!(
226 "the model has no `{}` effort level",
227 effort.as_str()
228 )),
229 Reasoning::Budget { tokens } => match &self.budget {
230 Some(range) if range.contains(tokens) => None,
231 Some(range) => Some(format!(
232 "the model takes a reasoning budget from {} to {} tokens",
233 range.start(),
234 range.end()
235 )),
236 None if self.levels.is_empty() => {
237 Some("the model takes no reasoning budget".to_owned())
238 }
239 None => Some("the model takes an effort level, not a reasoning budget".to_owned()),
240 },
241 }
242 }
243}
244
245impl CacheSupport {
246 /// Why the model cannot honour `cache`, or `None` when it can or the
247 /// catalog does not know.
248 pub fn refusal(&self, cache: &CacheRetention) -> Option<String> {
249 (!self.retention.is_empty() && !self.retention.contains(cache)).then(|| {
250 format!(
251 "the model does not honour `{}` cache retention",
252 retention_word(cache)
253 )
254 })
255 }
256}
257
258/// The lower-case word `cache` serializes as.
259pub(super) fn retention_word(cache: &CacheRetention) -> &'static str {
260 match cache {
261 CacheRetention::None => "none",
262 CacheRetention::Short => "short",
263 CacheRetention::Long => "long",
264 }
265}