car_inference/schema.rs
1//! Model schema — declarative metadata for models, analogous to ToolSchema for tools.
2//!
3//! Every model (local GGUF, remote API, Ollama) is described by a `ModelSchema`
4//! that declares identity, capabilities, constraints, cost, and source.
5//! The router uses this schema for initial routing; observed outcomes refine it.
6
7use serde::{Deserialize, Serialize};
8
9/// What a model can do.
10#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, schemars::JsonSchema)]
11#[serde(rename_all = "snake_case")]
12pub enum ModelCapability {
13 /// Text completion / chat generation
14 Generate,
15 /// Vector embeddings
16 Embed,
17 /// Cross-encoder relevance scoring (query + document → relevance
18 /// score). Qwen3-Reranker is the canonical local implementation.
19 Rerank,
20 /// Label assignment / classification
21 Classify,
22 /// Code generation, repair, refactoring
23 Code,
24 /// Chain-of-thought, planning, analysis
25 Reasoning,
26 /// Text condensation
27 Summarize,
28 /// Function/tool calling
29 ToolUse,
30 /// Multiple tool calls in a single response (parallel tool execution)
31 MultiToolCall,
32 /// Vision / image understanding
33 Vision,
34 /// Video understanding (multi-frame sampling + temporal tokens).
35 /// Distinct from `Vision` so routing can prefer video-trained
36 /// models when the caller attaches a video content block.
37 VideoUnderstanding,
38 /// Audio understanding (speech + non-speech audio as an input to
39 /// a chat/reasoning model). Distinct from `SpeechToText` which is
40 /// the transcription-only task. Gemma 4 E2B/E4B and Gemini do
41 /// this; Qwen2.5-VL does not.
42 AudioUnderstanding,
43 /// Visual grounding — structured object-localization output
44 /// (bounding boxes keyed to object labels) in addition to text.
45 Grounding,
46 /// Speech recognition / transcription
47 SpeechToText,
48 /// Speech synthesis / text-to-speech
49 TextToSpeech,
50 /// Image generation
51 ImageGeneration,
52 /// Video generation
53 VideoGeneration,
54}
55
56/// How much the project vouches for a model. Gates automatic upgrades and
57/// is surfaced in recommendation rationale. Closed enum — a new tier is a
58/// deliberate FFI-visible change, never a silent string fallback.
59#[derive(
60 Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, Default, schemars::JsonSchema,
61)]
62#[serde(rename_all = "snake_case")]
63pub enum TrustTier {
64 /// Vetted by the project — the built-in catalog and verified upgrades.
65 /// Eligible for background auto-apply when the user opts in.
66 #[default]
67 Curated,
68 /// User-registered or upstream-discovered, not project-vetted. Always
69 /// notify-only; never auto-applied regardless of update policy.
70 Community,
71}
72
73/// How to access the model.
74#[derive(Debug, Clone, Serialize, Deserialize, schemars::JsonSchema)]
75#[serde(tag = "type", rename_all = "snake_case")]
76pub enum ModelSource {
77 /// Local GGUF file via Candle backend.
78 Local {
79 hf_repo: String,
80 hf_filename: String,
81 tokenizer_repo: String,
82 },
83 /// Remote API endpoint (OpenAI-compatible, Anthropic, etc.)
84 RemoteApi {
85 endpoint: String,
86 /// Environment variable name containing the API key (never the key itself).
87 /// The env var value may contain comma-separated keys for load balancing.
88 api_key_env: String,
89 /// Additional environment variable names for load balancing across multiple keys.
90 /// Each env var may also contain comma-separated keys.
91 #[serde(default)]
92 api_key_envs: Vec<String>,
93 #[serde(default)]
94 api_version: Option<String>,
95 protocol: ApiProtocol,
96 },
97 /// One text-generation turn through the locally installed OpenAI Codex CLI.
98 ///
99 /// Codex remains the sole holder of its ChatGPT-subscription credential:
100 /// CAR neither reads that credential nor accepts an API key for this source.
101 /// An optional reasoning suffix is part of the model string (for example,
102 /// `gpt-5.6-sol:high`) and maps to Codex's `model_reasoning_effort` config.
103 CodexCli { model: String },
104 /// Ollama local server.
105 Ollama {
106 model_tag: String,
107 #[serde(default = "default_ollama_host")]
108 host: String,
109 },
110 /// Local MLX model via mlx-rs backend (Apple Silicon, safetensors format).
111 /// Models from mlx-community on HuggingFace.
112 Mlx {
113 /// HuggingFace repo (e.g., "mlx-community/Qwen3-4B-4bit").
114 hf_repo: String,
115 /// Optional specific weight filename. If None, auto-discovers safetensors files.
116 #[serde(default)]
117 hf_weight_file: Option<String>,
118 },
119 /// Local whisper.cpp speech-to-text model — a ggml `.bin` from the
120 /// `ggerganov/whisper.cpp` HF repo, run in-process via the shared
121 /// `car-whisper` crate. Cross-platform (Windows/Linux/macOS): this is the
122 /// on-device STT path where MLX isn't available. Cached at
123 /// `~/.tokhn/whisper/ggml-<model>.bin`.
124 WhisperCpp {
125 /// whisper.cpp model id — the suffix of `ggml-<model>.bin`
126 /// (e.g. `"large-v3-turbo-q5_0"`).
127 model: String,
128 },
129 /// Windows OS text-to-speech via `Windows.Media.SpeechSynthesis` (WinRT),
130 /// run in-process. The catalog-side analog of the `car-voice`
131 /// `TtsProvider::WindowsSpeech` live path and the parity counterpart of
132 /// Apple's OS synthesizer — free, on-device, no model download, no MLX.
133 /// Windows-only; availability is `false` on every other target (like
134 /// `AppleFoundationModels`).
135 WindowsSpeech {},
136 /// Local vLLM-MLX server (Apple Silicon, OpenAI-compatible API).
137 /// Routes through RemoteBackend with OpenAI protocol handler.
138 VllmMlx {
139 /// Server endpoint (e.g., "http://localhost:8000").
140 endpoint: String,
141 /// The model name as known to vLLM-MLX (e.g., "mlx-community/Qwen3-4B-4bit").
142 model_name: String,
143 },
144 /// CAR-owned supervised vLLM-MLX process backed by a managed HuggingFace
145 /// artifact. Unlike `VllmMlx`, CAR downloads, admits, spawns, reaps, and
146 /// accounts this allocation. Dispatch rewrites a clone to `VllmMlx` only
147 /// after the child reports healthy.
148 ManagedVllmMlx {
149 hf_repo: String,
150 #[serde(default)]
151 hf_weight_file: Option<String>,
152 },
153 /// Apple's on-device system model via the FoundationModels framework
154 /// (macOS 26+, Apple Silicon). Inference happens in-process through a
155 /// Swift shim — there is no HTTP, no API key, and no model file: the
156 /// OS owns the weights. Availability is checked at runtime via
157 /// `@available(macOS 26.0, *)`; on older macOS or non-Apple-Silicon
158 /// hosts the backend reports `UnsupportedMode` and the router falls
159 /// through to the next candidate.
160 AppleFoundationModels {
161 /// Optional Apple use-case hint passed through to
162 /// `LanguageModelSession`. Apple's framework tunes its prompt and
163 /// safety scaffolding per use case (e.g. "general", "summarize").
164 /// `None` uses the default.
165 #[serde(default)]
166 use_case: Option<String>,
167 },
168 /// Proprietary provider with custom auth and protocol.
169 ///
170 /// For vendor-specific APIs that aren't generic OpenAI-compatible endpoints.
171 /// Parslee is the first proprietary provider — custom auth (OAuth2),
172 /// custom response format, multi-provider routing built into the API.
173 Proprietary {
174 /// Provider identifier (e.g., "parslee").
175 provider: String,
176 /// Base URL for the API.
177 endpoint: String,
178 /// Auth configuration.
179 auth: ProprietaryAuth,
180 /// Custom protocol details.
181 protocol: ProprietaryProtocol,
182 },
183 /// Inference is delegated to a host-registered runner. CAR does
184 /// not own the wire format — the runner (typically a JS / Python
185 /// host) translates the `GenerateRequest` to its provider's API,
186 /// streams chunks back through the runner's event callback, and
187 /// returns the final aggregated result.
188 ///
189 /// Closes Parslee-ai/car-releases#24. Use this when the host
190 /// already has an SDK relationship with a provider (Anthropic,
191 /// OpenAI, GitHub Models, Vercel AI SDK) and wants CAR to sit in
192 /// the lifecycle / policy / replay path without learning every
193 /// provider's wire format.
194 ///
195 /// Routing requires that a runner has been registered via
196 /// [`crate::set_inference_runner`] (or its FFI equivalent —
197 /// `registerInferenceRunner` on JS, `register_inference_runner`
198 /// on Python, the `InferenceRunner` foreign trait on UniFFI,
199 /// `inference.register_runner` on the WebSocket protocol).
200 /// Without a runner, dispatch fails with `InferenceFailed`.
201 Delegated {
202 /// Opaque hint passed through to the runner — typically the
203 /// provider id (`"anthropic"`, `"openai"`, `"vercel-ai-sdk"`)
204 /// so a multi-provider runner can dispatch internally. CAR
205 /// does not interpret this string.
206 #[serde(default)]
207 hint: Option<String>,
208 },
209}
210
211/// Provider id for ChatGPT subscription inference backed by the Codex credential record.
212pub const OPENAI_CODEX_PROVIDER: &str = "openai-codex";
213
214/// What a signed-out openai-codex row tells a person to do.
215pub const OPENAI_CODEX_SIGN_IN_HINT: &str = "sign in with `car auth login --provider openai-codex`";
216
217/// Authentication method for proprietary providers.
218#[derive(Debug, Clone, Serialize, Deserialize, schemars::JsonSchema)]
219#[serde(tag = "type", rename_all = "snake_case")]
220pub enum ProprietaryAuth {
221 /// OAuth2 PKCE flow (e.g., Azure AD for Parslee).
222 #[serde(rename = "oauth2_pkce", alias = "o_auth2_pkce")]
223 OAuth2Pkce {
224 authority: String,
225 client_id: String,
226 scopes: Vec<String>,
227 },
228 /// Static API key from environment variable.
229 ApiKeyEnv { env_var: String },
230 /// Bearer token from environment variable.
231 BearerTokenEnv { env_var: String },
232 /// ChatGPT subscription OAuth read from the car-auth Codex record; admitted only for provider
233 /// `openai-codex`.
234 #[serde(rename = "chatgpt_subscription")]
235 ChatGptSubscription {},
236}
237
238/// Protocol configuration for proprietary providers.
239#[derive(Debug, Clone, Serialize, Deserialize, schemars::JsonSchema)]
240pub struct ProprietaryProtocol {
241 /// Chat/completion endpoint path (appended to base URL).
242 #[serde(default = "default_chat_path")]
243 pub chat_path: String,
244 /// Content type for requests.
245 #[serde(default = "default_content_type")]
246 pub content_type: String,
247 /// Whether the API streams responses via SSE.
248 #[serde(default)]
249 pub streaming: bool,
250 /// Custom headers to include in every request.
251 #[serde(default)]
252 pub extra_headers: std::collections::HashMap<String, String>,
253 /// The wire format the endpoint speaks. A row names it rather than CAR
254 /// inferring it from the path, so the model decides the backend.
255 #[serde(default)]
256 pub wire: ProprietaryWire,
257}
258
259/// What a proprietary endpoint speaks.
260#[derive(
261 Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize, schemars::JsonSchema,
262)]
263#[serde(rename_all = "snake_case")]
264pub enum ProprietaryWire {
265 /// OpenAI Responses-style chat (Parslee's `/inference/responses`).
266 #[default]
267 Responses,
268 /// TypeSafe System One (Parslee's `/inference/typesafe/v1/systemone`):
269 /// typed decisions with a probability per option, no text generation.
270 /// Such a row serves `classify` only.
271 SystemOne,
272}
273
274impl Default for ProprietaryProtocol {
275 fn default() -> Self {
276 Self {
277 chat_path: default_chat_path(),
278 content_type: default_content_type(),
279 streaming: false,
280 extra_headers: std::collections::HashMap::new(),
281 wire: ProprietaryWire::default(),
282 }
283 }
284}
285
286fn default_chat_path() -> String {
287 "/chat".to_string()
288}
289
290fn default_content_type() -> String {
291 "application/json".to_string()
292}
293
294fn default_ollama_host() -> String {
295 "http://localhost:11434".to_string()
296}
297
298#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, schemars::JsonSchema)]
299#[serde(rename_all = "snake_case")]
300pub enum ApiProtocol {
301 OpenAiCompat,
302 /// OpenRouter's OpenAI-compatible Chat Completions surface. Distinct so
303 /// credential precedence and error translation stay provider-specific.
304 OpenRouter,
305 /// OpenAI Responses API (/v1/responses) — works with all OpenAI models including codex.
306 OpenAiResponses,
307 Anthropic,
308 Google,
309 /// Azure OpenAI — uses api-key header and deployment-based URLs.
310 /// Endpoint format: {base}/openai/deployments/{model}/chat/completions?api-version={version}
311 AzureOpenAi,
312 /// Google Vertex AI — the enterprise Gemini surface. Same request/response
313 /// shape as the AI-Studio `Google` protocol, but a project/location URL and
314 /// OAuth Bearer auth (a GCP access token from `gcloud auth print-access-token`
315 /// or a service account) instead of an `?key=` query param. Endpoint format:
316 /// `{base}/publishers/google/models/{model}:generateContent`, where `base`
317 /// is `https://{loc}-aiplatform.googleapis.com/v1/projects/{proj}/locations/{loc}`.
318 VertexAi,
319 /// AWS Bedrock — the **Converse** API (`bedrock-runtime`), a unified
320 /// messages surface across Bedrock-hosted models (Claude, Llama, Mistral,
321 /// Titan, …). Auth is **SigV4** request signing (not a bearer token), with
322 /// credentials from the standard AWS env vars; the model's `endpoint` is the
323 /// region (e.g. `us-east-1`) and `name` is the Bedrock model id. Non-stream
324 /// only for now (Converse streaming uses a separate binary event-stream).
325 Bedrock,
326}
327
328impl ApiProtocol {
329 /// Prompt-cache economics for this provider, relative to its base input
330 /// rate — used by the cost scoreboard to price cached tokens correctly.
331 /// Anthropic uses explicit breakpoints (deep read discount + write
332 /// premium); OpenAI/Azure cache automatically (~0.5× read, no write
333 /// charge). Providers whose cache tokens CAR does not parse (Google/Vertex/
334 /// Bedrock) report zero cache tokens, so their rates are inert.
335 pub fn cache_rates(&self) -> crate::outcome::CacheRates {
336 use crate::outcome::CacheRates;
337 match self {
338 ApiProtocol::Anthropic => CacheRates::ANTHROPIC,
339 ApiProtocol::OpenAiCompat | ApiProtocol::OpenAiResponses | ApiProtocol::AzureOpenAi => {
340 CacheRates::OPENAI
341 }
342 // OpenRouter rates differ per upstream model and are carried by
343 // ModelSchema::cost. A blanket OpenAI-shaped discount is false.
344 ApiProtocol::OpenRouter => CacheRates::NONE,
345 ApiProtocol::Google | ApiProtocol::VertexAi | ApiProtocol::Bedrock => CacheRates::NONE,
346 }
347 }
348
349 /// This protocol as an OpenTelemetry GenAI `gen_ai.provider.name` value.
350 ///
351 /// The protocol is the right source for this attribute and
352 /// [`ModelSchema::provider`] is not: that field carries the model's
353 /// *publisher* (`qwen`, `mlx-community`, `apple`, `ggerganov`), while
354 /// `gen_ai.provider.name` names the API surface the request is spoken to.
355 /// A Qwen checkpoint served over Bedrock is `aws.bedrock`, not `qwen`.
356 ///
357 /// Both OpenAI protocols map to `openai` on purpose. The convention
358 /// defines this attribute as a discriminator for the *telemetry format
359 /// flavor*, not the company serving the weights — so every
360 /// OpenAI-REST-shaped surface reports `openai`, and Chat Completions vs
361 /// Responses is not a distinction it draws. `openrouter` has no
362 /// well-known value; the convention's fallback is a provider-specific
363 /// string, which is what this returns.
364 ///
365 /// Exhaustive by intent: a new protocol must decide its own value rather
366 /// than inherit a wildcard's guess.
367 pub fn genai_provider_name(&self) -> &'static str {
368 match self {
369 ApiProtocol::OpenAiCompat | ApiProtocol::OpenAiResponses => "openai",
370 ApiProtocol::OpenRouter => "openrouter",
371 ApiProtocol::Anthropic => "anthropic",
372 ApiProtocol::Google => "gcp.gemini",
373 ApiProtocol::VertexAi => "gcp.vertex_ai",
374 ApiProtocol::AzureOpenAi => "azure.ai.openai",
375 ApiProtocol::Bedrock => "aws.bedrock",
376 }
377 }
378}
379
380/// The numeric format a checkpoint's weights are stored in.
381///
382/// Deliberately says nothing about *which engine* serves the file —
383/// [`ModelSource`] already carries that, and keying this on the container
384/// instead of the format produces falsehoods: whisper.cpp's `q5_0` ggml
385/// checkpoints use the same round-to-nearest block format as llama.cpp's, and
386/// a `Gguf*` variant would assert a GGUF text path that cannot load them.
387///
388/// What it does express is the axis neither the bit width nor the source can:
389/// an MLX 4-bit affine checkpoint and a `Q4_K_M` are both "4-bit", were
390/// produced by different algorithms, and do not have the same quality.
391#[derive(
392 Debug, Clone, Copy, PartialEq, Eq, Hash, Default, Serialize, Deserialize, schemars::JsonSchema,
393)]
394#[serde(rename_all = "snake_case")]
395pub enum QuantScheme {
396 /// Integer weights with a scale and bias shared across `group_size`
397 /// elements (`4bit`, `6bit`). MLX's default and only affine format.
398 AffineGroupInt,
399 /// Block-scaled float — MLX's `mxfp4`, `mxfp8` and NVIDIA's `nvfp4`. A
400 /// genuinely different loader path from [`QuantScheme::AffineGroupInt`] at
401 /// the same nominal width, which is why the width alone cannot classify a
402 /// checkpoint; see `backend::local::quantization_is_decodable`.
403 BlockScaledFloat,
404 /// Mixed per-tensor bit allocation, optionally importance-weighted
405 /// (`Q4_K_M`, `Q5_K_S`, `IQ4_XS`, `TQ1_0`).
406 KQuantMixed,
407 /// Uniform round-to-nearest blocks, no per-tensor mixing and no importance
408 /// weighting (`Q8_0`, `q5_0`, `Q4_0_4_4`).
409 RtnBlock,
410 /// Full-precision weights (`bf16`, `f16`, `f32`). Distinct from `None` at
411 /// the [`ModelSchema::quantization`] level: a positive claim that the
412 /// checkpoint is unquantized, not an absence of information.
413 Unquantized,
414 /// A format this parser could not identify. The label is still preserved
415 /// verbatim; only the classification is missing.
416 #[default]
417 Unknown,
418}
419
420/// Structured quantization descriptor.
421///
422/// This replaces a free-text string that mixed two vocabularies — MLX's
423/// `4bit`/`6bit` and GGUF's `Q4_K_M`/`Q8_0` — under one field, so nothing could
424/// tell a group-quantized integer checkpoint from a mixed k-quant one. It also
425/// threw away what the ingest paths already had in hand: MLX `config.json`
426/// declares `{bits, group_size, mode}` and only `bits` survived.
427///
428/// Deserializes from **either** the legacy bare string or the structured
429/// object, so catalogs, user `models.json` files, and `models.register`
430/// payloads written before this change keep loading unchanged. The object form
431/// requires `label` and rejects unknown fields: a misspelled key is a producer
432/// bug, and accepting it silently is how the field it replaced lost
433/// information in the first place.
434///
435/// **Writes the bare label back whenever that is lossless**, and the object
436/// only when it carries something parsing cannot recover — a `group_size`, or
437/// a `scheme` that disambiguates a label the parser reads as `Unknown`. Two
438/// reasons, both about blast radius rather than taste. This value is inside
439/// `catalog_identity::row_digest`, so an unconditional shape change would move
440/// the digest of every quantized row and hard-fail any client that pinned
441/// `expected_catalog_revision`. And `registry::load_user_config` fails a
442/// `models.json` **whole**, with its only production caller discarding the
443/// error — so an older daemon meeting an object it cannot parse boots with
444/// zero registered models and says nothing. Emitting the object only where it
445/// adds information keeps both costs proportional to what actually changed.
446#[allow(dead_code)]
447#[derive(schemars::JsonSchema)]
448#[serde(untagged)]
449enum QuantizationWireSchema {
450 Label(String),
451 Structured(QuantizationObjectWireSchema),
452}
453
454#[allow(dead_code)]
455#[derive(schemars::JsonSchema)]
456#[serde(deny_unknown_fields)]
457struct QuantizationObjectWireSchema {
458 bits: Option<u8>,
459 scheme: QuantScheme,
460 group_size: Option<u32>,
461 label: String,
462}
463
464#[derive(Debug, Clone, PartialEq, Eq)]
465pub struct Quantization {
466 /// Weight bit width. `None` when the label names none.
467 pub bits: Option<u8>,
468 /// Which quantization family this is.
469 pub scheme: QuantScheme,
470 /// Elements sharing one scale/zero point (MLX: 32, 64, 128). `None` when
471 /// the scheme has no group concept, or the source declared none.
472 pub group_size: Option<u32>,
473 /// The label exactly as published. Never synthesized: it is what the user
474 /// sees, what Hugging Face repos are named after, and the only thing that
475 /// survives a scheme CAR does not recognize yet.
476 pub label: String,
477}
478
479/// Widths affine group quantization actually uses. A `16bit` or `32bit`
480/// label is full precision that happens to be spelled like a quant.
481const AFFINE_GROUP_WIDTHS: std::ops::RangeInclusive<u8> = 2..=8;
482
483impl Quantization {
484 /// Best-effort structure from a published label.
485 ///
486 /// Never fails and never invents: an unrecognized label yields
487 /// [`QuantScheme::Unknown`] with the label intact.
488 pub fn parse(label: &str) -> Self {
489 let label = label.trim();
490 let lower = label.to_ascii_lowercase();
491 let build = |bits: Option<u8>, scheme: QuantScheme| Self {
492 bits,
493 scheme,
494 group_size: None,
495 label: label.to_string(),
496 };
497
498 // Full precision. `fp8` is deliberately NOT here — an 8-bit float is a
499 // quantized weight format, just not one this parser can attribute.
500 match lower.as_str() {
501 "bf16" | "f16" | "fp16" | "float16" | "half" => {
502 return build(Some(16), QuantScheme::Unquantized)
503 }
504 "f32" | "fp32" | "float32" | "full" => {
505 return build(Some(32), QuantScheme::Unquantized)
506 }
507 "none" => return build(None, QuantScheme::Unquantized),
508 "f8" | "fp8" | "float8" => return build(Some(8), QuantScheme::Unknown),
509 // Empty means nobody said, which is not a claim of full precision.
510 "" => return build(None, QuantScheme::Unknown),
511 _ => {}
512 }
513
514 // MLX block-scaled floats: `mxfp4`, `mxfp8`. A bare `mxfp` names no
515 // width, so it identifies nothing.
516 if let Some(rest) = lower.strip_prefix("mxfp") {
517 return match leading_number(rest) {
518 Some((bits, _)) => build(Some(bits), QuantScheme::BlockScaledFloat),
519 None => build(None, QuantScheme::Unknown),
520 };
521 }
522
523 // MLX affine group quant: `4bit`, `4-bit`, `6bit`.
524 if let Some(width) = lower.strip_suffix("bit").map(|w| w.trim_end_matches('-')) {
525 if let Some((bits, consumed)) = leading_number(width) {
526 if consumed == width.len() && AFFINE_GROUP_WIDTHS.contains(&bits) {
527 return build(Some(bits), QuantScheme::AffineGroupInt);
528 }
529 // `16bit`/`32bit` are full precision spelled as a width.
530 if consumed == width.len() && (bits == 16 || bits == 32) {
531 return build(Some(bits), QuantScheme::Unquantized);
532 }
533 }
534 }
535
536 // GGUF. `iq`/`tq` are k-quant families; a bare `q` needs its suffix
537 // read to tell k-quant from legacy round-to-nearest.
538 let gguf = lower
539 .strip_prefix("iq")
540 .or_else(|| lower.strip_prefix("tq"))
541 .map(|rest| (rest, true))
542 .or_else(|| lower.strip_prefix('q').map(|rest| (rest, false)));
543 if let Some((rest, k_family)) = gguf {
544 if let Some((bits, consumed)) = leading_number(rest) {
545 if bits == 0 {
546 return build(None, QuantScheme::Unknown);
547 }
548 // Slice by digits consumed, not by the width's decimal length —
549 // `q08_0` has a two-character prefix for a one-character number.
550 let suffix = &rest[consumed..];
551 let scheme = if k_family || suffix.contains("_k") {
552 QuantScheme::KQuantMixed
553 } else if suffix.starts_with("_0") || suffix.starts_with("_1") {
554 // `starts_with`, not equality: the aarch64 repack quants
555 // are `Q4_0_4_4`, `Q4_0_4_8`, `Q4_0_8_8`.
556 QuantScheme::RtnBlock
557 } else {
558 // A bare `Q4` names a width and no producer — llama.cpp
559 // has no such format, so this came from somewhere else.
560 QuantScheme::Unknown
561 };
562 return build(Some(bits), scheme);
563 }
564 }
565
566 build(None, QuantScheme::Unknown)
567 }
568
569 /// Recover the quantization a GGUF file names in its own filename
570 /// (`Qwen3-8B-Q4_K_M.gguf`, `ggml-large-v3-turbo-q5_0.gguf`).
571 ///
572 /// Scans the hyphen-separated segments from the right and takes the first
573 /// that classifies, so a model whose *name* contains something quant-shaped
574 /// does not outrank the real suffix. Returns `None` rather than guessing
575 /// when nothing in the name is recognizable.
576 pub fn from_gguf_filename(filename: &str) -> Option<Self> {
577 let stem = filename
578 .rsplit_once('.')
579 .map(|(stem, _)| stem)
580 .unwrap_or(filename);
581 stem.rsplit('-')
582 .map(Self::parse)
583 .find(|q| q.scheme != QuantScheme::Unknown)
584 }
585
586 /// Descriptor for an MLX checkpoint, from the `quantization` block of its
587 /// `config.json`. `mode` is MLX's own name for the format (`affine`,
588 /// `mxfp4`, `mxfp8`); absent means affine, which is MLX's default.
589 ///
590 /// Returns `None` when the block carried nothing usable — declaring
591 /// `AffineGroupInt` on the strength of an empty object would assert exactly the
592 /// thing this type exists to establish.
593 pub fn from_mlx_config(
594 bits: Option<u8>,
595 group_size: Option<u32>,
596 mode: Option<&str>,
597 ) -> Option<Self> {
598 if bits.is_none() && group_size.is_none() && mode.is_none() {
599 return None;
600 }
601 let normalized = mode.map(str::to_ascii_lowercase);
602 let scheme = match normalized.as_deref() {
603 // MLX's `QuantizationMode` is exactly these four. `nvfp4` used to
604 // fall through to `Unknown`, which was a refusal dressed as a parse
605 // failure — it is block-scaled float like the MX pair and the
606 // linked loader reads it.
607 Some(m) if m.starts_with("mxfp") || m == "nvfp4" => QuantScheme::BlockScaledFloat,
608 Some("affine") | None => QuantScheme::AffineGroupInt,
609 Some(_) => QuantScheme::Unknown,
610 };
611 let label = match (normalized.as_deref(), bits) {
612 (Some(m), _) if m != "affine" => m.to_string(),
613 (_, Some(b)) => format!("{b}bit"),
614 // No width and no distinguishing mode: MLX said "affine" and
615 // nothing else. Say that rather than inventing a width.
616 (_, None) => "affine".to_string(),
617 };
618 Some(Self {
619 bits,
620 scheme,
621 group_size,
622 label,
623 })
624 }
625}
626
627/// Leading run of ASCII digits as a bit width, with the number of bytes it
628/// occupied. The byte count is returned because it is not recoverable from the
629/// value — `08` and `8` parse the same and slice differently.
630fn leading_number(s: &str) -> Option<(u8, usize)> {
631 let digits: String = s.chars().take_while(char::is_ascii_digit).collect();
632 if digits.is_empty() {
633 return None;
634 }
635 // Overflow (`Q256_K`) is a parse failure, not a silent truncation.
636 digits.parse().ok().map(|n| (n, digits.len()))
637}
638
639impl std::fmt::Display for Quantization {
640 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
641 f.write_str(&self.label)
642 }
643}
644
645impl Serialize for Quantization {
646 fn serialize<S: serde::Serializer>(&self, serializer: S) -> Result<S::Ok, S::Error> {
647 use serde::ser::SerializeStruct;
648
649 // Round-tripping through `parse` is the exact test for "the label
650 // carries everything": if it reproduces this value, the object form
651 // would be a longer spelling of the same information.
652 if Self::parse(&self.label) == *self {
653 return serializer.serialize_str(&self.label);
654 }
655
656 let len = 2 + usize::from(self.bits.is_some()) + usize::from(self.group_size.is_some());
657 let mut row = serializer.serialize_struct("Quantization", len)?;
658 if let Some(bits) = self.bits {
659 row.serialize_field("bits", &bits)?;
660 }
661 row.serialize_field("scheme", &self.scheme)?;
662 if let Some(group_size) = self.group_size {
663 row.serialize_field("group_size", &group_size)?;
664 }
665 row.serialize_field("label", &self.label)?;
666 row.end()
667 }
668}
669
670impl<'de> Deserialize<'de> for Quantization {
671 fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
672 /// The object form. `label` is required and unknown fields are
673 /// rejected, so a typo'd or double-nested payload is an error rather
674 /// than a row that silently claims `Unknown`.
675 #[derive(Deserialize)]
676 #[serde(deny_unknown_fields)]
677 struct Structured {
678 #[serde(default)]
679 bits: Option<u8>,
680 #[serde(default)]
681 scheme: Option<QuantScheme>,
682 #[serde(default)]
683 group_size: Option<u32>,
684 label: String,
685 }
686
687 struct QuantizationVisitor;
688
689 impl<'de> serde::de::Visitor<'de> for QuantizationVisitor {
690 type Value = Quantization;
691
692 fn expecting(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
693 f.write_str("a quantization label string, or an object with a `label` field")
694 }
695
696 fn visit_str<E: serde::de::Error>(self, label: &str) -> Result<Self::Value, E> {
697 Ok(Quantization::parse(label))
698 }
699
700 fn visit_map<M: serde::de::MapAccess<'de>>(
701 self,
702 map: M,
703 ) -> Result<Self::Value, M::Error> {
704 // Dispatched by hand rather than via `#[serde(untagged)]`,
705 // which buffers the input and reports only "data did not match
706 // any variant" — for a `models.json` holding dozens of models
707 // that error names neither the field nor the row.
708 let s = Structured::deserialize(serde::de::value::MapAccessDeserializer::new(map))?;
709 // A row that omits `scheme` or `bits` recovers what its label
710 // implies; an explicit value always wins.
711 let inferred = Quantization::parse(&s.label);
712 Ok(Quantization {
713 bits: s.bits.or(inferred.bits),
714 scheme: s.scheme.unwrap_or(inferred.scheme),
715 group_size: s.group_size,
716 label: s.label,
717 })
718 }
719 }
720
721 d.deserialize_any(QuantizationVisitor)
722 }
723}
724
725/// Declared performance expectations. Overridden by observed data once available.
726#[derive(Debug, Clone, Default, Serialize, Deserialize, schemars::JsonSchema)]
727pub struct PerformanceEnvelope {
728 /// Median latency in milliseconds (declared/estimated).
729 #[serde(default)]
730 pub latency_p50_ms: Option<u64>,
731 /// 99th percentile latency in milliseconds.
732 #[serde(default)]
733 pub latency_p99_ms: Option<u64>,
734 /// Tokens per second throughput.
735 #[serde(default)]
736 pub tokens_per_second: Option<f64>,
737}
738
739/// Cost model for routing optimization.
740/// Generation parameters that a model may or may not support.
741/// Models declare which params they accept. The inference layer
742/// strips unsupported params before sending to the API.
743#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, schemars::JsonSchema)]
744#[serde(rename_all = "snake_case")]
745pub enum GenerateParam {
746 Temperature,
747 TopP,
748 TopK,
749 MaxTokens,
750 StopSequences,
751 FrequencyPenalty,
752 PresencePenalty,
753 Seed,
754 ResponseFormat,
755 /// Extended thinking / internal reasoning before responding.
756 ExtendedThinking,
757}
758
759/// Standard parameter set for most models.
760pub fn standard_params() -> Vec<GenerateParam> {
761 vec![
762 GenerateParam::Temperature,
763 GenerateParam::TopP,
764 GenerateParam::MaxTokens,
765 GenerateParam::StopSequences,
766 GenerateParam::FrequencyPenalty,
767 GenerateParam::PresencePenalty,
768 GenerateParam::Seed,
769 ]
770}
771
772/// Parameter set for reasoning models (no temperature, no top_p).
773pub fn reasoning_params() -> Vec<GenerateParam> {
774 vec![GenerateParam::MaxTokens, GenerateParam::StopSequences]
775}
776
777#[derive(Debug, Clone, Copy, Default, PartialEq, Serialize, Deserialize, schemars::JsonSchema)]
778pub struct TokenPrices {
779 /// USD per 1M uncached input tokens.
780 #[serde(default)]
781 pub input_per_mtok: Option<f64>,
782 /// USD per 1M output tokens.
783 #[serde(default)]
784 pub output_per_mtok: Option<f64>,
785 /// USD per 1M cache-read input tokens.
786 #[serde(default)]
787 pub cache_read_input_per_mtok: Option<f64>,
788 /// USD per 1M cache-write input tokens.
789 #[serde(default)]
790 pub cache_write_input_per_mtok: Option<f64>,
791}
792
793#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize, schemars::JsonSchema)]
794pub struct TokenPricingTier {
795 /// Inclusive prompt-token threshold at which this tier applies.
796 pub min_prompt_tokens: usize,
797 #[serde(flatten)]
798 pub prices: TokenPrices,
799}
800
801#[derive(Debug, Clone, Default, Serialize, Deserialize, schemars::JsonSchema)]
802pub struct CostModel {
803 /// USD per 1M input tokens (remote models).
804 #[serde(default)]
805 pub input_per_mtok: Option<f64>,
806 /// USD per 1M output tokens (remote models).
807 #[serde(default)]
808 pub output_per_mtok: Option<f64>,
809 /// USD per 1M cache-read input tokens. Unlike protocol-wide cache
810 /// multipliers, this is model-specific and comes from the provider's
811 /// published catalog.
812 #[serde(default)]
813 pub cache_read_input_per_mtok: Option<f64>,
814 /// USD per 1M cache-write input tokens, when the provider charges one.
815 #[serde(default)]
816 pub cache_write_input_per_mtok: Option<f64>,
817 /// Prompt-size pricing overrides, sorted by increasing threshold.
818 /// The highest threshold not greater than the prompt size wins.
819 #[serde(default)]
820 pub pricing_tiers: Vec<TokenPricingTier>,
821 /// On-disk size in MB (local models).
822 #[serde(default)]
823 pub size_mb: Option<u64>,
824 /// RAM required during inference in MB.
825 #[serde(default)]
826 pub ram_mb: Option<u64>,
827}
828
829impl CostModel {
830 pub fn prices_for(&self, prompt_tokens: usize) -> TokenPrices {
831 let mut prices = TokenPrices {
832 input_per_mtok: self.input_per_mtok,
833 output_per_mtok: self.output_per_mtok,
834 cache_read_input_per_mtok: self.cache_read_input_per_mtok,
835 cache_write_input_per_mtok: self.cache_write_input_per_mtok,
836 };
837 for tier in self
838 .pricing_tiers
839 .iter()
840 .filter(|tier| tier.min_prompt_tokens <= prompt_tokens)
841 {
842 if tier.prices.input_per_mtok.is_some() {
843 prices.input_per_mtok = tier.prices.input_per_mtok;
844 }
845 if tier.prices.output_per_mtok.is_some() {
846 prices.output_per_mtok = tier.prices.output_per_mtok;
847 }
848 if tier.prices.cache_read_input_per_mtok.is_some() {
849 prices.cache_read_input_per_mtok = tier.prices.cache_read_input_per_mtok;
850 }
851 if tier.prices.cache_write_input_per_mtok.is_some() {
852 prices.cache_write_input_per_mtok = tier.prices.cache_write_input_per_mtok;
853 }
854 }
855 prices
856 }
857
858 /// Estimated request cost from the provider's declared token prices.
859 /// Unknown price components contribute zero; callers that need to
860 /// distinguish unknown pricing should inspect `prices_for` first.
861 ///
862 /// **This is the routing-score input, not a display or billing figure.**
863 /// The zero-fill is load-bearing here — `adaptive_router` normalizes the
864 /// result into a 0..1 cost score, and it separately neutralizes models
865 /// with no pricing at all, so changing the fill would change routing.
866 /// For anything a human reads, use
867 /// [`estimated_usd_bounded`](Self::estimated_usd_bounded), which refuses
868 /// to bill an unrated bucket at zero and says which way it can be wrong.
869 pub fn estimated_usd(
870 &self,
871 prompt_tokens: usize,
872 output_tokens: usize,
873 cache_read_tokens: usize,
874 cache_write_tokens: usize,
875 ) -> f64 {
876 let prices = self.prices_for(prompt_tokens);
877 let uncached_input = prompt_tokens
878 .saturating_sub(cache_read_tokens)
879 .saturating_sub(cache_write_tokens);
880 (uncached_input as f64 * prices.input_per_mtok.unwrap_or(0.0)
881 + output_tokens as f64 * prices.output_per_mtok.unwrap_or(0.0)
882 + cache_read_tokens as f64 * prices.cache_read_input_per_mtok.unwrap_or(0.0)
883 + cache_write_tokens as f64 * prices.cache_write_input_per_mtok.unwrap_or(0.0))
884 / 1_000_000.0
885 }
886
887 /// Effective per-bucket rates under one resolved price sheet, in the
888 /// bucket order `[uncached input, output, cache read, cache write]`, each
889 /// paired with which way a *substituted* rate can be wrong:
890 /// `(rate, may_overstate, may_understate)`.
891 ///
892 /// The two cache buckets substitute the uncached-input rate but are **not
893 /// governed by one rule**, because the providers in this catalog do not
894 /// price them the same way relative to input:
895 ///
896 /// | bucket | observed vs input, curated table |
897 /// |---|---|
898 /// | cache read | `0.10x`–`0.64x` — always a discount |
899 /// | cache write | `0.1875x` (Google) … `1.25x` (Anthropic) — **both sides** |
900 ///
901 /// So substituting input for a missing **cache-read** rate can only be too
902 /// high (`may_overstate`), while for a missing **cache-write** rate it can
903 /// land either side and gets both flags, rendering `~` rather than a `≤`
904 /// the figure cannot honour. Treating the two alike is how `claude-opus-4.8`
905 /// — `input 5.0`, `cache_write 6.25` — would have worn a `≤$5.00` ceiling
906 /// over a true cost of `$6.25`, or `$10.00` at OpenRouter's 1h-TTL rate.
907 ///
908 /// **Output** substitutes nothing and refuses instead: output runs
909 /// `1.5x`–`8x` input across this same table, which is not a ballpark
910 /// estimate in either direction, merely a wrong number wearing a marker.
911 /// The line is whether a substitute is *within range and merely of unknown
912 /// sign* (estimate, flag it) or *out of range entirely* (refuse) — see
913 /// [`estimated_usd_bounded`](Self::estimated_usd_bounded).
914 fn bucket_rates(prices: &TokenPrices) -> [(Option<f64>, bool, bool); 4] {
915 // A cached READ is always discounted relative to uncached input — the
916 // entire point of the cache — so the input rate is a true ceiling.
917 let cache_read = match prices.cache_read_input_per_mtok {
918 Some(rate) => (Some(rate), false, false),
919 None => {
920 let substituted = prices.input_per_mtok.is_some();
921 (prices.input_per_mtok, substituted, false)
922 }
923 };
924 // A cache WRITE may be a surcharge (Anthropic 1.25x, OpenRouter's 1h
925 // TTL 2x) or a discount (Google 0.1875x). Unknown sign, so no bound.
926 let cache_write = match prices.cache_write_input_per_mtok {
927 Some(rate) => (Some(rate), false, false),
928 None => {
929 let substituted = prices.input_per_mtok.is_some();
930 (prices.input_per_mtok, substituted, substituted)
931 }
932 };
933 [
934 (prices.input_per_mtok, false, false),
935 (prices.output_per_mtok, false, false),
936 cache_read,
937 cache_write,
938 ]
939 }
940
941 /// Cost estimate for a figure a person will read, carrying which way it
942 /// can be wrong.
943 ///
944 /// Differs from [`estimated_usd`](Self::estimated_usd) in refusing to
945 /// invent numbers. A token bucket the provider charges for but whose rate
946 /// this catalog does not declare is **never billed at zero**, and a figure
947 /// that might be wrong never presents itself as exact.
948 ///
949 /// `tier_prompt_tokens` selects the prompt-size pricing tier and is the
950 /// parameter callers most often get wrong:
951 ///
952 /// - `Some(n)` — the prompt size of **one request**. Tiers resolve exactly.
953 /// - `None` — the caller cannot say (a lifetime accumulator has summed
954 /// many requests and lost their boundaries). Base rates are used and the
955 /// result is flagged in whichever direction the model's own tiers run.
956 ///
957 /// Passing a *summed* token count as `Some` is the bug this signature
958 /// exists to prevent: thirty 10K-token requests sum to 300K, which crosses
959 /// a 272K threshold that no individual request came near, and every token
960 /// ever sent gets priced at the high-context rate — roughly double, stated
961 /// with total confidence.
962 ///
963 /// Two rejected alternatives, for the next person who wants tiers on an
964 /// aggregate. **Pricing lifetime totals at the tier their sum lands in** is
965 /// the bug above. **Pricing everything at the highest declared tier** does
966 /// yield a true ceiling, but a useless one — it doubles the figure for a
967 /// user whose prompts never approached the threshold, which is the same
968 /// confident wrongness in the other direction. Resolving tiers properly
969 /// needs per-request prompt sizes, which means [`crate::ModelProfile`]
970 /// would have to accumulate per-tier token buckets at record time; that is
971 /// a real feature with a persisted-schema change, not something to fake
972 /// here from data that has already been summed away.
973 ///
974 /// Returns `None` when no defensible number exists — either the model
975 /// declares no rate card at all, or a bucket carrying tokens has no rate
976 /// and no usable substitute. Output deliberately has no input-rate
977 /// fallback: it runs 1.5x–8x input across this catalog, far enough out of
978 /// range that no marker could rescue the number. Which buckets substitute,
979 /// and which way each substitution can be wrong, is decided in
980 /// [`bucket_rates`](Self::bucket_rates) from observed provider pricing —
981 /// notably cache *reads* and cache *writes* do not share a direction.
982 /// `None` means unpriced, and a caller must render it as such, not as free.
983 pub fn estimated_usd_bounded(
984 &self,
985 tier_prompt_tokens: Option<usize>,
986 uncached_input_tokens: usize,
987 output_tokens: usize,
988 cache_read_tokens: usize,
989 cache_write_tokens: usize,
990 ) -> Option<ApproxCost> {
991 // `prices_for(0)` is the base sheet: no tier threshold is <= 0 in a
992 // catalog whose thresholds are positive, so nothing overrides.
993 let prices = self.prices_for(tier_prompt_tokens.unwrap_or(0));
994 if prices.input_per_mtok.is_none() && prices.output_per_mtok.is_none() {
995 // No rate card. Not free — unknown.
996 return None;
997 }
998
999 let tokens = [
1000 uncached_input_tokens,
1001 output_tokens,
1002 cache_read_tokens,
1003 cache_write_tokens,
1004 ];
1005 let rates = Self::bucket_rates(&prices);
1006 let mut usd = 0.0;
1007 let mut may_overstate = false;
1008 let mut may_understate = false;
1009 for (count, (rate, substitute_high, substitute_low)) in tokens.into_iter().zip(rates) {
1010 if count == 0 {
1011 continue;
1012 }
1013 // Charged, rate unknown, no usable substitute — admit we can't.
1014 let rate = rate?;
1015 // Direction comes from the bucket, not from one blanket rule: a
1016 // substituted cache-READ rate can only be high, a substituted
1017 // cache-WRITE rate can land either side.
1018 may_overstate |= substitute_high;
1019 may_understate |= substitute_low;
1020 usd += count as f64 * rate;
1021 }
1022
1023 // Unresolvable tiers: say which way the base sheet can be wrong rather
1024 // than assuming tiers always cost more. Compare the rate each bucket
1025 // would actually pay at every declared threshold against the base.
1026 if tier_prompt_tokens.is_none() {
1027 for tier in &self.pricing_tiers {
1028 let at_tier = Self::bucket_rates(&self.prices_for(tier.min_prompt_tokens));
1029 for (count, ((base_rate, _, _), (tier_rate, _, _))) in
1030 tokens.into_iter().zip(rates.iter().zip(at_tier))
1031 {
1032 if count == 0 {
1033 continue;
1034 }
1035 if let (Some(base), Some(tiered)) = (base_rate, tier_rate) {
1036 may_understate |= tiered > *base;
1037 may_overstate |= tiered < *base;
1038 }
1039 }
1040 }
1041 }
1042
1043 Some(ApproxCost {
1044 usd: usd / 1_000_000.0,
1045 may_overstate,
1046 may_understate,
1047 })
1048 }
1049}
1050
1051/// A cost figure plus which way it can be wrong.
1052///
1053/// Presenting an estimate as an exact price is the same failure as billing an
1054/// unrated bucket at zero, one step later — so the direction travels with the
1055/// number instead of being re-derived (or forgotten) at each display site.
1056#[derive(Debug, Clone, Copy, PartialEq)]
1057pub struct ApproxCost {
1058 /// USD.
1059 pub usd: f64,
1060 /// The true cost may be **lower** — a bucket was priced at a substitute
1061 /// rate that can only be too high (a cache read at the uncached-input
1062 /// rate), or a pricing tier is cheaper than the base sheet used.
1063 pub may_overstate: bool,
1064 /// The true cost may be **higher** — a pricing tier dearer than the base
1065 /// sheet could not be resolved from the tokens the caller had, or a cache
1066 /// *write* was priced at the uncached-input rate and the provider charges
1067 /// a surcharge for it (Anthropic 1.25x, OpenRouter's 1h TTL 2x).
1068 pub may_understate: bool,
1069}
1070
1071impl ApproxCost {
1072 /// Every rate applied exactly; the figure is the price.
1073 pub fn is_exact(&self) -> bool {
1074 !self.may_overstate && !self.may_understate
1075 }
1076
1077 /// Prefix for the figure: `≤` a ceiling, `≥` a floor, `~` neither bound
1078 /// holds, empty when exact. Rendering the number without this is the
1079 /// defect the type exists to prevent.
1080 pub fn marker(&self) -> &'static str {
1081 match (self.may_overstate, self.may_understate) {
1082 (false, false) => "",
1083 (true, false) => "≤",
1084 (false, true) => "≥",
1085 (true, true) => "~",
1086 }
1087 }
1088}
1089
1090/// A score on a public benchmark from a published source (model card,
1091/// paper, leaderboard). The schema is deliberately permissive — no enum
1092/// of benchmark names — so the catalog can carry whichever benchmarks
1093/// the upstream provider chose to publish, and new ones can be added
1094/// without a code change. Scores are stored on a 0.0–1.0 scale (e.g.
1095/// 73.5% accuracy → 0.735) so they compare cleanly across benchmarks
1096/// and so `routing_ext::apply_benchmark_priors` can consume them
1097/// directly when wired in later.
1098#[derive(Debug, Clone, Serialize, Deserialize, schemars::JsonSchema)]
1099pub struct BenchmarkScore {
1100 /// Benchmark name as published (e.g., "MMLU-Pro", "GPQA-Diamond",
1101 /// "SWE-bench-Verified", "HumanEval", "MATH").
1102 pub name: String,
1103 /// Score on a 0.0–1.0 scale.
1104 pub score: f64,
1105 /// Evaluation harness or setup label (e.g., "5-shot", "0-shot CoT",
1106 /// "agentic", "pass@1"). Optional but strongly recommended — the
1107 /// same benchmark name can mean different things under different
1108 /// harnesses.
1109 #[serde(default)]
1110 pub harness: Option<String>,
1111 /// Where the score came from (model card URL, paper, leaderboard
1112 /// snapshot). Empty when the source is the upstream provider's
1113 /// announcement and a stable URL is not yet known.
1114 #[serde(default)]
1115 pub source_url: Option<String>,
1116 /// ISO 8601 date of the score snapshot (e.g., "2025-08-12"). Lets
1117 /// downstream code judge how stale a number is.
1118 #[serde(default)]
1119 pub measured_at: Option<String>,
1120 /// How many independent runs the score pools, when CAR measured it.
1121 /// `None` for a published score, whose sample is not ours to know.
1122 #[serde(default, skip_serializing_if = "Option::is_none")]
1123 pub runs: Option<u32>,
1124 /// Highest minus lowest score across those runs; present only with two
1125 /// or more. A difference smaller than this is not a measured difference.
1126 #[serde(default, skip_serializing_if = "Option::is_none")]
1127 pub spread: Option<f64>,
1128}
1129
1130/// The full declarative schema for a model.
1131///
1132/// Analogous to `ToolSchema` — describes what a model is, what it can do,
1133/// and how to access it. The router uses this for constraint-based filtering
1134/// and cold-start scoring before observed performance data is available.
1135#[derive(Debug, Clone, Serialize, Deserialize, schemars::JsonSchema)]
1136pub struct ModelSchema {
1137 /// Unique identifier: "provider/model-name:variant" (e.g., "qwen/qwen3-4b:q4_k_m").
1138 pub id: String,
1139 /// Human-readable display name.
1140 pub name: String,
1141 /// Provider (qwen, openai, anthropic, google, meta, ollama, custom).
1142 pub provider: String,
1143 /// Model family for grouping (qwen3, gpt-4, claude-4, llama-3).
1144 pub family: String,
1145 /// Semantic version or checkpoint label.
1146 #[serde(default)]
1147 pub version: String,
1148 /// What this model can do — ordered by primary capability first.
1149 pub capabilities: Vec<ModelCapability>,
1150 /// Context window in tokens.
1151 pub context_length: usize,
1152 /// Per-model maximum OUTPUT tokens the provider will return in one
1153 /// response. None = unknown; callers fall back to
1154 /// effective_max_output() which derives a fraction of context_length.
1155 #[serde(default)]
1156 pub max_output_tokens: Option<usize>,
1157 /// Parameter count as human-readable string (e.g., "4B", "30B (3B active)").
1158 #[serde(default)]
1159 pub param_count: String,
1160 /// How the weights are quantized, if at all. `None` for remote models and
1161 /// for local ones whose source declared nothing. See [`Quantization`] —
1162 /// this accepts the legacy bare-string form on the wire.
1163 #[serde(default)]
1164 #[schemars(with = "Option<QuantizationWireSchema>")]
1165 pub quantization: Option<Quantization>,
1166 /// Declared performance envelope (initial estimate, overridden by observed data).
1167 #[serde(default)]
1168 pub performance: PerformanceEnvelope,
1169 /// Cost structure.
1170 #[serde(default)]
1171 pub cost: CostModel,
1172 /// How to access this model.
1173 pub source: ModelSource,
1174 /// Free-form tags for filtering (e.g., "fast", "multilingual", "moe").
1175 #[serde(default)]
1176 pub tags: Vec<String>,
1177 /// Supported generation parameters. The inference layer strips any parameter
1178 /// not in this set before sending to the API. Empty = all supported.
1179 #[serde(default)]
1180 pub supported_params: Vec<GenerateParam>,
1181 /// Public benchmark scores as published by the model provider or
1182 /// reproduced on a public leaderboard (MMLU-Pro, GPQA-Diamond,
1183 /// SWE-bench, HumanEval, etc.). The built-in catalog ships this
1184 /// empty — population is a curation step, not a code change. See
1185 /// `BenchmarkScore` for the field shape and the 0.0–1.0 scoring
1186 /// convention.
1187 #[serde(default)]
1188 pub public_benchmarks: Vec<BenchmarkScore>,
1189 /// How much the project vouches for this model. The built-in catalog is
1190 /// `Curated`. Deserialization retains the legacy `Curated` default, so
1191 /// every user-controlled ingestion boundary must call
1192 /// [`Self::mark_user_registered`] before persistence or registration.
1193 /// Gates auto-apply (task #8) and this is surfaced in recommendation
1194 /// rationale.
1195 #[serde(default)]
1196 pub trust_tier: TrustTier,
1197 /// Superseded models stay listed if installed but are excluded from
1198 /// fresh recommendations. `#[serde(default)]` → not deprecated.
1199 #[serde(default)]
1200 pub deprecated: bool,
1201 /// Whether this model is currently available (downloaded / reachable).
1202 /// Not serialized — computed at runtime.
1203 #[serde(skip)]
1204 pub available: bool,
1205 /// Whether this model can be used **right now, without a download**.
1206 ///
1207 /// Deliberately narrower than [`Self::available`], which for a local MLX
1208 /// model is true as soon as an `hf_repo` is declared — `ensure_local()`
1209 /// lazy-downloads on first use, so a declared repo is "functionally
1210 /// available" (see #164). That is the right default for open-ended work and
1211 /// wrong for work on a deadline: a step with a bounded budget that picks a
1212 /// model it must first fetch spends the whole budget downloading and fails.
1213 /// That is exactly how `car code`'s 120s contract derivation became
1214 /// unusable on a machine with no local weights (Parslee-ai/car#638).
1215 ///
1216 /// Callers express the requirement with [`crate::IntentHint::require_ready`];
1217 /// this is the per-candidate fact that hint filters on. Recomputed on every
1218 /// registration, so a cached schema can't carry a stale value.
1219 #[serde(skip)]
1220 pub weights_ready: bool,
1221}
1222
1223impl ModelSchema {
1224 /// Mark a schema as user-controlled rather than project-vetted.
1225 ///
1226 /// This is intentionally separate from serde's legacy default: old built-in
1227 /// and test fixtures omit `trust_tier` and must continue to deserialize,
1228 /// while `models.json`, CLI imports, and daemon `models.register` must never
1229 /// inherit `Curated` merely because a caller omitted the field or supplied
1230 /// a forged value.
1231 pub fn mark_user_registered(&mut self) {
1232 self.trust_tier = TrustTier::Community;
1233 }
1234
1235 /// Check if this model has a given capability.
1236 pub fn has_capability(&self, cap: ModelCapability) -> bool {
1237 self.capabilities.contains(&cap)
1238 }
1239
1240 /// Whether usage is covered by the person's ChatGPT subscription.
1241 /// Callers must render `subscription` and never a dollar figure.
1242 pub fn is_subscription_billed(&self) -> bool {
1243 matches!(
1244 &self.source,
1245 ModelSource::Proprietary {
1246 auth: ProprietaryAuth::ChatGptSubscription {},
1247 ..
1248 }
1249 )
1250 }
1251
1252 /// Live availability for credential-backed providers. The catalog field is
1253 /// a startup snapshot; Settings/OAuth changes must affect the next list and
1254 /// route without a daemon restart.
1255 /// The credential this row is waiting on, when a missing credential is the
1256 /// only thing standing between the user and using it.
1257 ///
1258 /// Deliberately narrow. It answers `Some` only for a plain remote API row
1259 /// that declares one credential variable and is currently unavailable —
1260 /// the case where "add this key" is complete and correct advice. Every
1261 /// other shape answers `None`, because for them the advice would be wrong:
1262 ///
1263 /// - available rows need nothing;
1264 /// - local rows have no credential;
1265 /// - `OpenRouter` rows resolve through a credential *source* (pasted key
1266 /// or OAuth) rather than a named variable a person can type;
1267 /// - `Proprietary` OAuth rows (`parslee/*`) are a sign-in, not a key, and
1268 /// their availability additionally depends on the gateway having an
1269 /// upstream — credential presence alone has already been shipped as a
1270 /// false "available" for those (car#786);
1271 /// - `CodexCli` owns its own auth;
1272 /// - a deprecated row will not become usable by adding anything.
1273 ///
1274 /// A row declaring several alternative variables answers `None` as well:
1275 /// telling a person to set one of four names is not actionable, and
1276 /// picking one for them would be a guess about which account they have.
1277 pub fn credential_required(&self) -> Option<String> {
1278 if self.available_now() || self.deprecated || self.is_local() {
1279 return None;
1280 }
1281 match &self.source {
1282 ModelSource::RemoteApi {
1283 protocol: ApiProtocol::OpenRouter,
1284 ..
1285 } => None,
1286 ModelSource::RemoteApi {
1287 api_key_env,
1288 api_key_envs,
1289 ..
1290 } if api_key_envs.is_empty() && !api_key_env.trim().is_empty() => {
1291 Some(api_key_env.clone())
1292 }
1293 _ => None,
1294 }
1295 }
1296
1297 pub fn available_now(&self) -> bool {
1298 match &self.source {
1299 ModelSource::RemoteApi {
1300 protocol: ApiProtocol::OpenRouter,
1301 ..
1302 } => self.available && crate::openrouter::credential_source().is_some(),
1303 ModelSource::CodexCli { .. } => crate::backend::codex_cli::is_available(),
1304 // Live (cached a few seconds): Apple Intelligence can be switched
1305 // on or off, or finish provisioning, while CAR runs, and the
1306 // catalog snapshot would not see it until the next refresh.
1307 ModelSource::AppleFoundationModels { .. } => {
1308 #[cfg(any(
1309 all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)),
1310 all(target_os = "ios", target_arch = "aarch64")
1311 ))]
1312 {
1313 crate::backend::foundation_models::is_available()
1314 }
1315 #[cfg(not(any(
1316 all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)),
1317 all(target_os = "ios", target_arch = "aarch64")
1318 )))]
1319 {
1320 false
1321 }
1322 }
1323 _ => self.available,
1324 }
1325 }
1326
1327 /// Prompt-cache economics for this model, derived from its remote
1328 /// protocol. Local / non-remote models have no remote prompt cache, so
1329 /// their cache rates are inert ([`CacheRates::NONE`](crate::outcome::CacheRates::NONE)).
1330 pub fn cache_rates(&self) -> crate::outcome::CacheRates {
1331 match &self.source {
1332 ModelSource::RemoteApi {
1333 protocol: ApiProtocol::OpenRouter,
1334 ..
1335 } => {
1336 let input = self.cost.input_per_mtok.unwrap_or(0.0);
1337 if input > 0.0 {
1338 crate::outcome::CacheRates {
1339 read_mult: self.cost.cache_read_input_per_mtok.unwrap_or(0.0) / input,
1340 write_mult: self.cost.cache_write_input_per_mtok.unwrap_or(0.0) / input,
1341 }
1342 } else {
1343 crate::outcome::CacheRates::NONE
1344 }
1345 }
1346 ModelSource::RemoteApi { protocol, .. } => protocol.cache_rates(),
1347 _ => crate::outcome::CacheRates::NONE,
1348 }
1349 }
1350
1351 /// The organization whose model this is, when that can be honestly known.
1352 ///
1353 /// Answers one question — could two models be expected to fail the same
1354 /// way? — so it is deliberately conservative. `None` means UNKNOWABLE, not
1355 /// "none", and callers must treat it as "cannot tell" rather than folding
1356 /// it into a count of distinct vendors.
1357 ///
1358 /// NOT [`Self::provider`], which means four different things depending on
1359 /// which path built the schema:
1360 ///
1361 /// * curated remote rows — the real vendor (`openai`, `anthropic`);
1362 /// * OpenRouter and the Parslee gateway — the AGGREGATOR, so three vendors
1363 /// behind one gateway all report `openrouter`/`parslee` and one vendor
1364 /// reached two ways reports as two;
1365 /// * a HuggingFace-derived row — the repo ORG that uploaded it, so
1366 /// `unsloth/Qwen3` and `mlx-community/Qwen3` are the same weights under
1367 /// two "vendors";
1368 /// * a discovered local-server row — a guess from the model name.
1369 ///
1370 /// Only the first is a vendor, so only the first is reported. NOT `family`
1371 /// either — that is the model line, so `claude-4.6` and `claude-4.8` read as
1372 /// different and are both Anthropic.
1373 pub fn vendor(&self) -> Option<&str> {
1374 // The aggregator case: the curated table knows which upstream a gateway
1375 // alias resolves to, which is the only place that survives an id
1376 // carrying no trace of its vendor.
1377 if self.provider.eq_ignore_ascii_case("openrouter")
1378 || self.provider.eq_ignore_ascii_case("parslee")
1379 {
1380 return crate::openrouter::curated_vendor(&self.id);
1381 }
1382 // A local model is served by the operator's own machine. Whoever
1383 // uploaded the weights is not an organization that could fail
1384 // independently of the process running beside it.
1385 if self.is_local() {
1386 return None;
1387 }
1388 // Only a project-vetted row's `provider` was assigned deliberately. On a
1389 // community row it is the uploader or a guess, and claiming it here is
1390 // how two repacks of one checkpoint would pass as two vendors.
1391 if self.trust_tier != TrustTier::Curated {
1392 return None;
1393 }
1394 (!self.provider.is_empty()).then_some(self.provider.as_str())
1395 }
1396
1397 /// Check if this model is local (runs on-device).
1398 pub fn is_local(&self) -> bool {
1399 matches!(
1400 self.source,
1401 ModelSource::Local { .. }
1402 | ModelSource::Mlx { .. }
1403 | ModelSource::WhisperCpp { .. }
1404 | ModelSource::WindowsSpeech { .. }
1405 | ModelSource::ManagedVllmMlx { .. }
1406 | ModelSource::AppleFoundationModels { .. }
1407 )
1408 }
1409
1410 /// Whether this model has weights CAR fetches to disk before it can be
1411 /// used — i.e. whether "is it installed?" is a question with an answer.
1412 ///
1413 /// Three predicates in this area are easy to conflate, and conflating them
1414 /// is what Parslee-ai/car#894 was about:
1415 ///
1416 /// - [`is_local`](Self::is_local) — *owned on this machine*. True for
1417 /// `WindowsSpeech` (the OS owns the voices), `AppleFoundationModels`
1418 /// (the OS owns the weights), and CAR-managed sources. An external
1419 /// `VllmMlx` endpoint is remote even when its URL happens to be loopback.
1420 /// - [`weights_ready`](Self::weights_ready) — *the weights are on disk
1421 /// now*. Only meaningful when this predicate is true; for everything
1422 /// else the registry sets it to `true` as a "nothing blocks an attempt"
1423 /// sentinel, which reads as "installed" if taken literally.
1424 /// - `downloads_weights` (this one) — *there is something to install at
1425 /// all*. Use it to decide whether an install/download status should be
1426 /// reported, then use `weights_ready` for the status itself.
1427 ///
1428 /// Rendering `weights_ready` without this gate is what made
1429 /// `windows/speech-synthesis:os` claim `INSTALLED yes` while `car doctor`
1430 /// said `Models: none installed`, and made `apple/foundation:default` and
1431 /// the `vllm-mlx/*` rows claim `INSTALLED no` for models that install
1432 /// nothing.
1433 ///
1434 /// Written as an exhaustive `match` rather than `matches!` so that adding
1435 /// a `ModelSource` variant is a compile error here instead of a silently
1436 /// wrong answer in the CLI.
1437 pub fn downloads_weights(&self) -> bool {
1438 match self.source {
1439 // CAR fetches these to disk itself: a GGUF file, an MLX
1440 // safetensors repo, a whisper.cpp ggml `.bin`.
1441 ModelSource::Local { .. }
1442 | ModelSource::Mlx { .. }
1443 | ModelSource::WhisperCpp { .. }
1444 | ModelSource::ManagedVllmMlx { .. } => true,
1445 // The OS owns the voices / the weights — nothing to download.
1446 ModelSource::WindowsSpeech {} | ModelSource::AppleFoundationModels { .. } => false,
1447 // Someone else holds the weights: a local server (vLLM-MLX,
1448 // Ollama), a remote API, or a host-registered runner.
1449 ModelSource::VllmMlx { .. }
1450 | ModelSource::Ollama { .. }
1451 | ModelSource::RemoteApi { .. }
1452 | ModelSource::CodexCli { .. }
1453 | ModelSource::Proprietary { .. }
1454 | ModelSource::Delegated { .. } => false,
1455 }
1456 }
1457
1458 /// Whether a CAR-downloadable artifact is physically present now.
1459 /// General runtime availability and OS/server-owned weights are not an
1460 /// installation claim.
1461 pub fn has_installed_weights(&self) -> bool {
1462 self.downloads_weights() && self.weights_ready
1463 }
1464
1465 /// Whether CAR decodes this model **in its own process**, token by token,
1466 /// through the shared decode loop.
1467 ///
1468 /// Narrower than [`is_local`](Self::is_local) on purpose. `is_local` also
1469 /// covers CAR-managed vLLM-MLX and the speech backends. Those do not spend
1470 /// an output-token budget as this process's wall clock, so they must keep
1471 /// the remote treatment. Getting that distinction wrong takes the
1472 /// anti-truncation budget away from vLLM-MLX, which is the documented way
1473 /// to get structured tool calls out of a local model. (car#851)
1474 pub fn decodes_in_process(&self) -> bool {
1475 matches!(
1476 self.source,
1477 ModelSource::Local { .. } | ModelSource::Mlx { .. }
1478 )
1479 }
1480
1481 /// Check if this model delegates inference to a host-registered
1482 /// runner (closes Parslee-ai/car-releases#24).
1483 pub fn is_delegated(&self) -> bool {
1484 matches!(self.source, ModelSource::Delegated { .. })
1485 }
1486
1487 /// Check if this model runs one tool-free turn through `codex exec`.
1488 pub fn is_codex_cli(&self) -> bool {
1489 matches!(self.source, ModelSource::CodexCli { .. })
1490 }
1491
1492 /// Check if this model uses the MLX backend.
1493 pub fn is_mlx(&self) -> bool {
1494 matches!(self.source, ModelSource::Mlx { .. })
1495 }
1496
1497 /// Check if this model routes to Apple's on-device FoundationModels
1498 /// framework. True only for `ModelSource::AppleFoundationModels`;
1499 /// callers must still verify runtime availability before dispatch
1500 /// (the schema can describe the model on any host, but execution
1501 /// requires macOS 26+ on Apple Silicon).
1502 pub fn is_foundation_models(&self) -> bool {
1503 matches!(self.source, ModelSource::AppleFoundationModels { .. })
1504 }
1505
1506 /// Whether the operating system owns the implementation and its memory.
1507 /// These rows have no CAR-downloadable artifact and no local-model memory
1508 /// figure to compare with the admission budget.
1509 pub fn is_os_provided(&self) -> bool {
1510 matches!(
1511 self.source,
1512 ModelSource::WindowsSpeech {} | ModelSource::AppleFoundationModels { .. }
1513 )
1514 }
1515
1516 /// Check if this model uses vLLM-MLX backend.
1517 pub fn is_vllm_mlx(&self) -> bool {
1518 matches!(
1519 self.source,
1520 ModelSource::VllmMlx { .. } | ModelSource::ManagedVllmMlx { .. }
1521 )
1522 }
1523
1524 /// Whether CAR, rather than an independently managed endpoint, owns the
1525 /// vLLM-MLX child process and its physical weight allocation. The existing
1526 /// HTTP `VllmMlx` contract remains external/server-owned. A supervised
1527 /// source must opt in explicitly and provide an already-installed local
1528 /// model path in `model_name`; it is then replaced with a loopback HTTP
1529 /// endpoint only after admission and a successful readiness ACK.
1530 pub fn is_car_managed_vllm_mlx(&self) -> bool {
1531 matches!(self.source, ModelSource::ManagedVllmMlx { .. })
1532 }
1533
1534 /// Whether a CAR/OS-owned model can only run on Apple Silicon (Metal).
1535 /// External vLLM-MLX endpoints own their hardware and are not constrained
1536 /// by the client machine's accelerator.
1537 pub fn requires_apple_silicon(&self) -> bool {
1538 self.is_mlx() || self.is_car_managed_vllm_mlx() || self.is_foundation_models()
1539 }
1540
1541 /// Check if this model is remote (requires API call).
1542 pub fn is_remote(&self) -> bool {
1543 matches!(
1544 self.source,
1545 ModelSource::RemoteApi { .. }
1546 | ModelSource::CodexCli { .. }
1547 | ModelSource::Proprietary { .. }
1548 | ModelSource::VllmMlx { .. }
1549 )
1550 }
1551
1552 /// Collect all API key env var names for this model (primary + extras).
1553 /// Returns empty vec for non-remote models.
1554 pub fn all_api_key_envs(&self) -> Vec<String> {
1555 match &self.source {
1556 ModelSource::RemoteApi {
1557 api_key_env,
1558 api_key_envs,
1559 ..
1560 } => {
1561 let mut all = vec![api_key_env.clone()];
1562 all.extend(api_key_envs.iter().cloned());
1563 all
1564 }
1565 ModelSource::Proprietary {
1566 auth: ProprietaryAuth::ApiKeyEnv { env_var },
1567 ..
1568 }
1569 | ModelSource::Proprietary {
1570 auth: ProprietaryAuth::BearerTokenEnv { env_var },
1571 ..
1572 } => vec![env_var.clone()],
1573 ModelSource::Proprietary {
1574 auth: ProprietaryAuth::OAuth2Pkce { .. } | ProprietaryAuth::ChatGptSubscription {},
1575 ..
1576 } => vec![],
1577 _ => vec![],
1578 }
1579 }
1580
1581 /// Get the size in MB (from cost model or 0 if unknown).
1582 pub fn size_mb(&self) -> u64 {
1583 self.cost.size_mb.unwrap_or(0)
1584 }
1585
1586 /// Get the RAM requirement in MB (from cost model, falls back to size_mb).
1587 pub fn ram_mb(&self) -> u64 {
1588 self.cost.ram_mb.unwrap_or_else(|| self.size_mb())
1589 }
1590
1591 /// Estimated cost per 1K output tokens in USD. Returns 0.0 for local models.
1592 pub fn cost_per_1k_output(&self) -> f64 {
1593 self.cost.output_per_mtok.map(|c| c / 1000.0).unwrap_or(0.0)
1594 }
1595
1596 /// The per-turn output-token ceiling to use when the caller didn't
1597 /// specify one. Prefers the registry-declared `max_output_tokens`;
1598 /// otherwise derives a quarter of the context window, clamped to a
1599 /// sane [4096, 32768] band so a 1M-context model doesn't request a
1600 /// 250K-token response the API rejects and a tiny 8K model doesn't
1601 /// get an absurdly small ceiling. (Registry value first, computed
1602 /// fallback second — mirrors a provider lookup with a derived default.)
1603 pub fn effective_max_output(&self) -> usize {
1604 self.max_output_tokens
1605 .unwrap_or_else(|| (self.context_length / 4).clamp(4096, 32_768))
1606 }
1607}
1608
1609#[cfg(test)]
1610mod tests {
1611 use super::*;
1612
1613 fn sample_local() -> ModelSchema {
1614 ModelSchema {
1615 id: "qwen/qwen3-4b:q4_k_m".into(),
1616 name: "Qwen3-4B".into(),
1617 provider: "qwen".into(),
1618 family: "qwen3".into(),
1619 version: "1.0".into(),
1620 capabilities: vec![ModelCapability::Generate, ModelCapability::Code],
1621 context_length: 32768,
1622 max_output_tokens: None,
1623 param_count: "4B".into(),
1624 quantization: Some(Quantization::parse("Q4_K_M")),
1625 performance: PerformanceEnvelope {
1626 tokens_per_second: Some(45.0),
1627 ..Default::default()
1628 },
1629 cost: CostModel {
1630 size_mb: Some(2500),
1631 ram_mb: Some(2500),
1632 ..Default::default()
1633 },
1634 source: ModelSource::Local {
1635 hf_repo: "Qwen/Qwen3-4B-GGUF".into(),
1636 hf_filename: "Qwen3-4B-Q4_K_M.gguf".into(),
1637 tokenizer_repo: "Qwen/Qwen3-4B".into(),
1638 },
1639 tags: vec!["code".into(), "fast".into()],
1640 supported_params: vec![],
1641 public_benchmarks: vec![],
1642 trust_tier: TrustTier::Curated,
1643 deprecated: false,
1644 available: false,
1645 weights_ready: false,
1646 }
1647 }
1648
1649 fn sample_remote() -> ModelSchema {
1650 ModelSchema {
1651 id: "anthropic/claude-sonnet-4-6:latest".into(),
1652 name: "Claude Sonnet 4.6".into(),
1653 provider: "anthropic".into(),
1654 family: "claude-4".into(),
1655 version: "latest".into(),
1656 capabilities: vec![
1657 ModelCapability::Generate,
1658 ModelCapability::Code,
1659 ModelCapability::Reasoning,
1660 ModelCapability::ToolUse,
1661 ModelCapability::Vision,
1662 ],
1663 context_length: 200000,
1664 max_output_tokens: None,
1665 param_count: String::new(),
1666 quantization: None,
1667 performance: PerformanceEnvelope {
1668 latency_p50_ms: Some(2000),
1669 latency_p99_ms: Some(8000),
1670 tokens_per_second: Some(80.0),
1671 },
1672 cost: CostModel {
1673 input_per_mtok: Some(3.0),
1674 output_per_mtok: Some(15.0),
1675 ..Default::default()
1676 },
1677 source: ModelSource::RemoteApi {
1678 endpoint: "https://api.anthropic.com/v1/messages".into(),
1679 api_key_env: "ANTHROPIC_API_KEY".into(),
1680 api_key_envs: vec![],
1681 api_version: Some("2023-06-01".into()),
1682 protocol: ApiProtocol::Anthropic,
1683 },
1684 tags: vec!["reasoning".into(), "tool_use".into()],
1685 supported_params: vec![],
1686 public_benchmarks: vec![],
1687 trust_tier: TrustTier::Curated,
1688 deprecated: false,
1689 available: false,
1690 weights_ready: false,
1691 }
1692 }
1693
1694 #[test]
1695 fn capabilities() {
1696 let m = sample_local();
1697 assert!(m.has_capability(ModelCapability::Code));
1698 assert!(!m.has_capability(ModelCapability::Vision));
1699 }
1700
1701 #[test]
1702 fn local_vs_remote() {
1703 assert!(sample_local().is_local());
1704 assert!(!sample_local().is_remote());
1705 assert!(sample_remote().is_remote());
1706 assert!(!sample_remote().is_local());
1707 let codex = ModelSchema {
1708 source: ModelSource::CodexCli {
1709 model: "gpt-5.6-sol:high".into(),
1710 },
1711 ..sample_local()
1712 };
1713 assert!(codex.is_remote());
1714 assert!(!codex.is_local());
1715 }
1716
1717 #[test]
1718 fn vllm_ownership_drives_local_remote_and_apple_predicates() {
1719 let external = ModelSchema {
1720 source: ModelSource::VllmMlx {
1721 endpoint: "https://gpu-owner.example/v1".into(),
1722 model_name: "owner/runtime-model".into(),
1723 },
1724 ..sample_local()
1725 };
1726 assert!(!external.is_local());
1727 assert!(external.is_remote());
1728 assert!(!external.requires_apple_silicon());
1729
1730 let managed = ModelSchema {
1731 source: ModelSource::ManagedVllmMlx {
1732 hf_repo: "mlx-community/Qwen3-4B-4bit".into(),
1733 hf_weight_file: None,
1734 },
1735 ..sample_local()
1736 };
1737 assert!(managed.is_local());
1738 assert!(!managed.is_remote());
1739 assert!(managed.requires_apple_silicon());
1740 }
1741
1742 #[test]
1743 fn cost() {
1744 let local = sample_local();
1745 assert_eq!(local.cost_per_1k_output(), 0.0);
1746
1747 let remote = sample_remote();
1748 assert!(remote.cost_per_1k_output() > 0.0);
1749 }
1750
1751 #[test]
1752 fn serde_roundtrip() {
1753 let local = sample_local();
1754 let json = serde_json::to_string(&local).unwrap();
1755 let parsed: ModelSchema = serde_json::from_str(&json).unwrap();
1756 assert_eq!(parsed.id, local.id);
1757 assert_eq!(parsed.capabilities, local.capabilities);
1758
1759 let remote = sample_remote();
1760 let json = serde_json::to_string(&remote).unwrap();
1761 let parsed: ModelSchema = serde_json::from_str(&json).unwrap();
1762 assert_eq!(parsed.id, remote.id);
1763 // available is skip-serialized, defaults to false
1764 assert!(!parsed.available);
1765 }
1766
1767 #[test]
1768 fn managed_vllm_source_is_versioned_without_reinterpreting_legacy_vllm_json() {
1769 let legacy: ModelSource = serde_json::from_str(
1770 r#"{"type":"vllm_mlx","endpoint":"http://localhost:8000","model_name":"legacy"}"#,
1771 )
1772 .unwrap();
1773 assert!(matches!(
1774 legacy,
1775 ModelSource::VllmMlx {
1776 endpoint,
1777 model_name
1778 } if endpoint == "http://localhost:8000" && model_name == "legacy"
1779 ));
1780
1781 let managed = ModelSource::ManagedVllmMlx {
1782 hf_repo: "mlx-community/Qwen3-4B-4bit".into(),
1783 hf_weight_file: Some("model.safetensors".into()),
1784 };
1785 let encoded = serde_json::to_string(&managed).unwrap();
1786 assert!(encoded.contains(r#""type":"managed_vllm_mlx""#));
1787 assert!(matches!(
1788 serde_json::from_str::<ModelSource>(&encoded).unwrap(),
1789 ModelSource::ManagedVllmMlx { .. }
1790 ));
1791 }
1792
1793 #[test]
1794 fn vendor_refuses_to_claim_a_repackager_or_a_guess() {
1795 // A HuggingFace-derived row's `provider` is the repo ORG that uploaded
1796 // it, so `unsloth/Qwen3` and `mlx-community/Qwen3` — one checkpoint,
1797 // two repacks — would otherwise read as two independent vendors. That
1798 // is a false independence claim produced silently, which is the exact
1799 // failure a vendor check exists to prevent.
1800 let mut m = sample_remote();
1801 m.provider = "mlx-community".into();
1802 m.trust_tier = TrustTier::Community;
1803 assert_eq!(m.vendor(), None);
1804
1805 // A project-vetted remote row IS authoritative.
1806 m.provider = "openai".into();
1807 m.trust_tier = TrustTier::Curated;
1808 assert_eq!(m.vendor(), Some("openai"));
1809
1810 // A local model is served by the operator's own machine; whoever
1811 // uploaded the weights is not an independently-failing organization.
1812 assert_eq!(sample_local().vendor(), None);
1813 }
1814
1815 #[test]
1816 fn trust_tier_and_deprecated_default_when_absent() {
1817 // Pre-existing ~/.car/models.json configs omit the new fields.
1818 // They must deserialize to Curated / not-deprecated, not error.
1819 let json = serde_json::to_string(&sample_local()).unwrap();
1820 let stripped = json
1821 .replace(",\"trust_tier\":\"curated\"", "")
1822 .replace(",\"deprecated\":false", "");
1823 let parsed: ModelSchema = serde_json::from_str(&stripped).unwrap();
1824 assert_eq!(parsed.trust_tier, TrustTier::Curated);
1825 assert!(!parsed.deprecated);
1826 }
1827
1828 #[test]
1829 fn trust_tier_serializes_snake_case() {
1830 assert_eq!(
1831 serde_json::to_string(&TrustTier::Community).unwrap(),
1832 "\"community\""
1833 );
1834 assert_eq!(TrustTier::default(), TrustTier::Curated);
1835 }
1836
1837 #[test]
1838 fn requires_apple_silicon_only_for_metal_backends() {
1839 // GGUF/Candle local and remote models run anywhere CAR builds for.
1840 assert!(!sample_local().requires_apple_silicon());
1841 assert!(!sample_remote().requires_apple_silicon());
1842
1843 let mlx = ModelSchema {
1844 source: ModelSource::Mlx {
1845 hf_repo: "mlx-community/Qwen3-4B-4bit".into(),
1846 hf_weight_file: None,
1847 },
1848 ..sample_local()
1849 };
1850 assert!(mlx.requires_apple_silicon());
1851
1852 // Only CAR-managed vLLM-MLX and Apple FoundationModels are Metal-bound.
1853 // A raw endpoint is external and its owner's hardware is opaque to CAR.
1854 let managed_vllm = ModelSchema {
1855 source: ModelSource::ManagedVllmMlx {
1856 hf_repo: "mlx-community/Qwen3-4B-4bit".into(),
1857 hf_weight_file: None,
1858 },
1859 ..sample_local()
1860 };
1861 assert!(managed_vllm.requires_apple_silicon());
1862
1863 let external_vllm = ModelSchema {
1864 source: ModelSource::VllmMlx {
1865 endpoint: "https://gpu-owner.example/v1".into(),
1866 model_name: "mlx-community/Qwen3-4B-4bit".into(),
1867 },
1868 ..sample_local()
1869 };
1870 assert!(!external_vllm.requires_apple_silicon());
1871
1872 let foundation = ModelSchema {
1873 source: ModelSource::AppleFoundationModels { use_case: None },
1874 ..sample_local()
1875 };
1876 assert!(foundation.requires_apple_silicon());
1877 }
1878
1879 fn priced(
1880 input: Option<f64>,
1881 output: Option<f64>,
1882 cache_read: Option<f64>,
1883 cache_write: Option<f64>,
1884 ) -> CostModel {
1885 CostModel {
1886 input_per_mtok: input,
1887 output_per_mtok: output,
1888 cache_read_input_per_mtok: cache_read,
1889 cache_write_input_per_mtok: cache_write,
1890 ..Default::default()
1891 }
1892 }
1893
1894 #[test]
1895 fn an_undeclared_cache_rate_is_bounded_by_the_input_rate_not_billed_at_zero() {
1896 // The real shape this exists for: qwen3.5-plus declares input and
1897 // output but no cache-read rate. 1M cache-read tokens must not be free.
1898 let cost = priced(Some(0.26), Some(1.56), None, None);
1899 let bounded = cost
1900 .estimated_usd_bounded(Some(1_000_000), 0, 0, 1_000_000, 0)
1901 .expect("a model with input+output rates is priceable");
1902 assert!(bounded.may_overstate, "substituted rate can only be high");
1903 assert!(!bounded.may_understate);
1904 assert_eq!(bounded.marker(), "≤");
1905 assert!(
1906 (bounded.usd - 0.26).abs() < 1e-9,
1907 "cache reads fall back to the $0.26/MTok input rate, got {}",
1908 bounded.usd
1909 );
1910
1911 // The routing-score path still zero-fills, deliberately and untouched.
1912 assert_eq!(cost.estimated_usd(1_000_000, 0, 1_000_000, 0), 0.0);
1913 }
1914
1915 #[test]
1916 fn an_undeclared_cache_write_rate_claims_no_bound_it_cannot_keep() {
1917 // The shipped `claude-opus-4.8` shape with its cache-write rate
1918 // omitted. Anthropic's cache write is a 1.25x SURCHARGE (6.25 against
1919 // 5.0 input), and OpenRouter's 1h-TTL rate is 2x — so pricing the
1920 // bucket at the input rate is a FLOOR, and the `≤` a blanket
1921 // "cache is always cheaper" rule would have produced is a false
1922 // ceiling over a true cost of $6.25, or $10.00 at the 1h rate.
1923 let cost = priced(Some(5.0), Some(25.0), Some(0.5), None);
1924 let bounded = cost
1925 .estimated_usd_bounded(Some(1_000_000), 0, 0, 0, 1_000_000)
1926 .expect("input and output rates are published");
1927 assert!((bounded.usd - 5.0).abs() < 1e-9, "got {}", bounded.usd);
1928 assert!(
1929 bounded.may_understate,
1930 "a cache-write surcharge can exceed the input rate"
1931 );
1932 assert_ne!(bounded.marker(), "≤", "must not claim a ceiling it fails");
1933 assert_eq!(bounded.marker(), "~", "sign is unknown, so neither bound");
1934 // Both real cache-write rates this shape could carry sit ABOVE the
1935 // substituted figure — which is exactly why `≤` would have been a
1936 // false ceiling. Assert the tighter of the two; it implies the looser,
1937 // and spelling out both comparisons is a `redundant_comparisons` lint.
1938 let at_anthropics_1_25x: f64 = 1_000_000.0 * 6.25 / 1e6;
1939 let at_openrouters_1h_2x: f64 = 1_000_000.0 * 10.0 / 1e6;
1940 assert!(bounded.usd < at_anthropics_1_25x.min(at_openrouters_1h_2x));
1941
1942 // Declaring the rate makes it exact — the flag tracks substitution,
1943 // not the mere presence of cache-write tokens.
1944 let declared = priced(Some(5.0), Some(25.0), Some(0.5), Some(6.25))
1945 .estimated_usd_bounded(Some(1_000_000), 0, 0, 0, 1_000_000)
1946 .unwrap();
1947 assert!(declared.is_exact());
1948 assert!((declared.usd - 6.25).abs() < 1e-9);
1949
1950 // The same omission on the cache READ side DOES keep its ceiling —
1951 // the two directions are not shared. Note the rate must be missing for
1952 // a substitution to happen at all.
1953 let read = priced(Some(5.0), Some(25.0), None, Some(6.25))
1954 .estimated_usd_bounded(Some(1_000_000), 0, 0, 1_000_000, 0)
1955 .unwrap();
1956 assert_eq!(read.marker(), "≤");
1957 assert!((read.usd - 5.0).abs() < 1e-9);
1958 // A real cache-read rate is a discount, so the ceiling holds.
1959 assert!(read.usd > 1_000_000.0 * 0.5 / 1e6);
1960 }
1961
1962 #[test]
1963 fn a_declared_cache_rate_is_exact_and_never_flagged() {
1964 let cost = priced(Some(5.0), Some(25.0), Some(0.5), Some(6.25));
1965 let bounded = cost
1966 .estimated_usd_bounded(Some(1_600_000), 1_000_000, 200_000, 500_000, 100_000)
1967 .expect("fully rated");
1968 assert!(bounded.is_exact(), "nothing was substituted or unresolved");
1969 assert_eq!(bounded.marker(), "");
1970 // 1M uncached x 5 + 200k x 25 + 500k x 0.5 + 100k x 6.25, per MTok.
1971 assert!((bounded.usd - 10.875).abs() < 1e-9, "got {}", bounded.usd);
1972 // Identical to the router's figure when every rate is published.
1973 assert!(
1974 (bounded.usd - cost.estimated_usd(1_600_000, 200_000, 500_000, 100_000)).abs() < 1e-9
1975 );
1976 }
1977
1978 #[test]
1979 fn an_unbounded_bucket_refuses_rather_than_understating() {
1980 // Output has no safe substitute — it is normally the dearer side, so
1981 // pricing it at the input rate would UNDER-state. Refuse instead.
1982 let cost = priced(Some(1.0), None, None, None);
1983 assert_eq!(
1984 cost.estimated_usd_bounded(Some(1_000), 1_000, 1_000, 0, 0),
1985 None
1986 );
1987 // With no output tokens the same model is priceable and exact.
1988 let bounded = cost
1989 .estimated_usd_bounded(Some(1_000), 1_000, 0, 0, 0)
1990 .unwrap();
1991 assert!(bounded.is_exact());
1992 }
1993
1994 #[test]
1995 fn no_rate_card_is_unpriced_rather_than_free() {
1996 let cost = CostModel::default();
1997 assert_eq!(
1998 cost.estimated_usd_bounded(Some(1_000), 1_000, 1_000, 0, 0),
1999 None
2000 );
2001 // And a zero-usage priced model is genuinely free, not unpriced.
2002 let free = priced(Some(0.0), Some(0.0), None, None)
2003 .estimated_usd_bounded(Some(0), 0, 0, 0, 0)
2004 .expect("a declared zero rate card is priced");
2005 assert_eq!(free.usd, 0.0);
2006 assert!(free.is_exact());
2007 }
2008
2009 fn tiered() -> CostModel {
2010 CostModel {
2011 input_per_mtok: Some(2.5),
2012 output_per_mtok: Some(15.0),
2013 cache_read_input_per_mtok: Some(0.25),
2014 pricing_tiers: vec![TokenPricingTier {
2015 min_prompt_tokens: 272_000,
2016 prices: TokenPrices {
2017 input_per_mtok: Some(5.0),
2018 output_per_mtok: Some(22.5),
2019 cache_read_input_per_mtok: Some(0.5),
2020 cache_write_input_per_mtok: None,
2021 },
2022 }],
2023 ..Default::default()
2024 }
2025 }
2026
2027 #[test]
2028 fn a_known_prompt_size_resolves_the_tier_exactly() {
2029 let cost = tiered();
2030 let below = cost
2031 .estimated_usd_bounded(Some(271_999), 271_999, 0, 0, 0)
2032 .unwrap();
2033 assert!(below.is_exact());
2034 assert!((below.usd - 271_999.0 * 2.5 / 1e6).abs() < 1e-9);
2035
2036 let above = cost
2037 .estimated_usd_bounded(Some(272_000), 272_000, 0, 0, 0)
2038 .unwrap();
2039 assert!(above.is_exact());
2040 assert!((above.usd - 272_000.0 * 5.0 / 1e6).abs() < 1e-9);
2041 }
2042
2043 #[test]
2044 fn a_lifetime_aggregate_uses_base_rates_and_admits_it_may_be_low() {
2045 // Thirty 10K-token requests. Their SUM crosses the 272K threshold that
2046 // no single request came near — the double-charging bug. `None` says
2047 // "boundaries lost", so base rates apply and the figure is marked.
2048 let cost = tiered();
2049 let aggregate = cost.estimated_usd_bounded(None, 300_000, 0, 0, 0).unwrap();
2050 assert!(
2051 (aggregate.usd - 300_000.0 * 2.5 / 1e6).abs() < 1e-9,
2052 "must use the $2.50 base rate, got {}",
2053 aggregate.usd
2054 );
2055 assert!(aggregate.may_understate, "a dearer tier may apply");
2056 assert!(!aggregate.may_overstate);
2057 assert_eq!(aggregate.marker(), "≥");
2058
2059 // What the bug looked like: summed tokens passed as a real prompt size
2060 // price at the high-context rate — exactly double, and unmarked.
2061 let bug = cost
2062 .estimated_usd_bounded(Some(300_000), 300_000, 0, 0, 0)
2063 .unwrap();
2064 assert!((bug.usd - 2.0 * aggregate.usd).abs() < 1e-9);
2065 assert!(bug.is_exact(), "and it would have claimed to be exact");
2066 }
2067
2068 #[test]
2069 fn a_cheaper_tier_flags_the_aggregate_as_possibly_high_instead() {
2070 // Direction is derived from the tiers, not assumed. A volume DISCOUNT
2071 // makes the base-rate figure too high, not too low.
2072 let cost = CostModel {
2073 input_per_mtok: Some(2.0),
2074 output_per_mtok: Some(10.0),
2075 pricing_tiers: vec![TokenPricingTier {
2076 min_prompt_tokens: 100_000,
2077 prices: TokenPrices {
2078 input_per_mtok: Some(1.0),
2079 ..Default::default()
2080 },
2081 }],
2082 ..Default::default()
2083 };
2084 let aggregate = cost.estimated_usd_bounded(None, 500_000, 0, 0, 0).unwrap();
2085 assert!(aggregate.may_overstate);
2086 assert!(!aggregate.may_understate);
2087 assert_eq!(aggregate.marker(), "≤");
2088 }
2089
2090 #[test]
2091 fn both_directions_at_once_claims_neither_bound() {
2092 // Unrated cache bucket (can be high) plus an unresolved dearer tier
2093 // (can be low). Neither bound survives, so the figure is just an
2094 // estimate and must not wear a `≤` it cannot honour.
2095 let cost = CostModel {
2096 input_per_mtok: Some(2.0),
2097 output_per_mtok: Some(10.0),
2098 pricing_tiers: vec![TokenPricingTier {
2099 min_prompt_tokens: 100_000,
2100 prices: TokenPrices {
2101 input_per_mtok: Some(4.0),
2102 ..Default::default()
2103 },
2104 }],
2105 ..Default::default()
2106 };
2107 let aggregate = cost.estimated_usd_bounded(None, 0, 0, 500_000, 0).unwrap();
2108 assert!(aggregate.may_overstate && aggregate.may_understate);
2109 assert_eq!(aggregate.marker(), "~");
2110 }
2111
2112 /// Parslee-ai/car#894 (follow-up): "runs here" and "has weights to fetch"
2113 /// are different questions, and `is_local` answers only the first. Three
2114 /// `is_local` sources download nothing — the CLI must not offer an
2115 /// install status for them.
2116 #[test]
2117 fn downloads_weights_is_true_only_for_sources_car_fetches() {
2118 let mut schema = sample_local();
2119
2120 // CAR downloads these itself.
2121 schema.source = ModelSource::Local {
2122 hf_repo: "Qwen/Qwen3-4B-GGUF".into(),
2123 hf_filename: "Qwen3-4B-Q4_K_M.gguf".into(),
2124 tokenizer_repo: "Qwen/Qwen3-4B".into(),
2125 };
2126 assert!(schema.downloads_weights(), "GGUF weights are downloaded");
2127
2128 schema.source = ModelSource::Mlx {
2129 hf_repo: "mlx-community/Qwen3-4B-4bit".into(),
2130 hf_weight_file: None,
2131 };
2132 assert!(schema.downloads_weights(), "MLX weights are downloaded");
2133
2134 schema.source = ModelSource::WhisperCpp {
2135 model: "large-v3-turbo-q5_0".into(),
2136 };
2137 assert!(
2138 schema.downloads_weights(),
2139 "the whisper.cpp ggml bin is downloaded"
2140 );
2141
2142 // The OS owns these — there is nothing to install. `WindowsSpeech` is
2143 // the row that rendered `INSTALLED yes` against `car doctor`'s
2144 // `Models: none installed`.
2145 schema.source = ModelSource::WindowsSpeech {};
2146 assert!(
2147 !schema.downloads_weights(),
2148 "WinRT speech synthesis has no weights to download"
2149 );
2150
2151 schema.source = ModelSource::AppleFoundationModels { use_case: None };
2152 assert!(
2153 !schema.downloads_weights(),
2154 "Apple FoundationModels weights belong to the OS"
2155 );
2156
2157 // Someone else holds the weights.
2158 schema.source = ModelSource::VllmMlx {
2159 endpoint: "http://localhost:8000".into(),
2160 model_name: "mlx-community/Qwen3-4B-4bit".into(),
2161 };
2162 assert!(
2163 !schema.downloads_weights(),
2164 "the vLLM-MLX server owns its own weights"
2165 );
2166
2167 schema.source = ModelSource::ManagedVllmMlx {
2168 hf_repo: "mlx-community/Qwen3-4B-4bit".into(),
2169 hf_weight_file: None,
2170 };
2171 assert!(
2172 schema.downloads_weights(),
2173 "the explicitly managed vLLM-MLX contract is CAR-owned"
2174 );
2175
2176 schema.source = ModelSource::Ollama {
2177 model_tag: "qwen3:4b".into(),
2178 host: default_ollama_host(),
2179 };
2180 assert!(!schema.downloads_weights(), "Ollama owns its own weights");
2181
2182 schema.source = sample_remote().source;
2183 assert!(
2184 !schema.downloads_weights(),
2185 "a remote API has no weights on this machine"
2186 );
2187
2188 schema.source = ModelSource::Delegated { hint: None };
2189 assert!(
2190 !schema.downloads_weights(),
2191 "a host-registered runner owns its own weights"
2192 );
2193 }
2194
2195 /// `is_local` is the predicate the CLI used to reach for. These OS-owned
2196 /// sources are where the two answers diverge, which is why a
2197 /// separate predicate exists rather than a reuse of `is_local`.
2198 #[test]
2199 fn downloads_weights_differs_from_is_local_on_os_owned_sources() {
2200 let mut schema = sample_local();
2201 for source in [
2202 ModelSource::WindowsSpeech {},
2203 ModelSource::AppleFoundationModels { use_case: None },
2204 ] {
2205 schema.source = source;
2206 assert!(
2207 schema.is_local(),
2208 "this source runs on-device: {:?}",
2209 schema.source
2210 );
2211 assert!(
2212 !schema.downloads_weights(),
2213 "…but CAR downloads nothing for it: {:?}",
2214 schema.source
2215 );
2216 }
2217 }
2218}
2219
2220#[cfg(test)]
2221mod quantization_tests {
2222 use super::*;
2223
2224 /// Every distinct label the built-in catalog ships, the hyphenated form
2225 /// `registry.rs` writes into auto-discovered entries, and the formats that
2226 /// review found misclassified.
2227 #[test]
2228 fn classifies_known_labels() {
2229 let cases: &[(&str, Option<u8>, QuantScheme)] = &[
2230 ("4bit", Some(4), QuantScheme::AffineGroupInt),
2231 ("6bit", Some(6), QuantScheme::AffineGroupInt),
2232 ("3bit", Some(3), QuantScheme::AffineGroupInt),
2233 ("5bit", Some(5), QuantScheme::AffineGroupInt),
2234 // registry.rs emits the hyphen; users already have it on disk.
2235 ("4-bit", Some(4), QuantScheme::AffineGroupInt),
2236 ("mxfp8", Some(8), QuantScheme::BlockScaledFloat),
2237 ("mxfp4", Some(4), QuantScheme::BlockScaledFloat),
2238 ("Q4_K_M", Some(4), QuantScheme::KQuantMixed),
2239 ("Q5_K_S", Some(5), QuantScheme::KQuantMixed),
2240 ("IQ4_XS", Some(4), QuantScheme::KQuantMixed),
2241 ("TQ1_0", Some(1), QuantScheme::KQuantMixed),
2242 ("Q8_0", Some(8), QuantScheme::RtnBlock),
2243 ("q5_0", Some(5), QuantScheme::RtnBlock),
2244 // aarch64 repack quants: the suffix is a prefix match, not equality.
2245 ("Q4_0_4_4", Some(4), QuantScheme::RtnBlock),
2246 ("Q4_0_8_8", Some(4), QuantScheme::RtnBlock),
2247 // A zero-padded width must slice by digits consumed, not by the
2248 // decimal length of the parsed number.
2249 ("q08_0", Some(8), QuantScheme::RtnBlock),
2250 ("bf16", Some(16), QuantScheme::Unquantized),
2251 ("F16", Some(16), QuantScheme::Unquantized),
2252 ("f32", Some(32), QuantScheme::Unquantized),
2253 // Spelled like a quant, but affine group quant has no such width.
2254 ("16bit", Some(16), QuantScheme::Unquantized),
2255 ("32bit", Some(32), QuantScheme::Unquantized),
2256 // An 8-bit float IS quantized — just not attributable from this
2257 // label alone. Calling it full precision was the bug.
2258 ("fp8", Some(8), QuantScheme::Unknown),
2259 // Names a width and no producer.
2260 ("Q4", Some(4), QuantScheme::Unknown),
2261 ];
2262 for (label, bits, scheme) in cases {
2263 let q = Quantization::parse(label);
2264 assert_eq!(q.bits, *bits, "bits for {label}");
2265 assert_eq!(q.scheme, *scheme, "scheme for {label}");
2266 assert_eq!(q.label, *label, "label must survive verbatim");
2267 }
2268 }
2269
2270 /// Labels that identify nothing must say so rather than guess.
2271 #[test]
2272 fn refuses_to_guess() {
2273 for label in [
2274 "", // nobody said; not a claim of full precision
2275 "mxfp", // MX family, no width
2276 "Q256_K", // width overflows u8
2277 "q0_0", // a zero-bit quantization is not a thing
2278 "awq-marlin-w4a16", // real format, unknown to this parser
2279 ] {
2280 let q = Quantization::parse(label);
2281 assert_eq!(q.scheme, QuantScheme::Unknown, "scheme for {label:?}");
2282 assert_eq!(q.bits, None, "bits for {label:?}");
2283 assert_eq!(q.label, label, "label for {label:?}");
2284 }
2285 }
2286
2287 /// The distinction the free-text field could not express: same width,
2288 /// different algorithm, different loader.
2289 #[test]
2290 fn same_width_different_scheme() {
2291 let affine = Quantization::parse("4bit");
2292 let kquant = Quantization::parse("Q4_K_M");
2293 let mx = Quantization::parse("mxfp4");
2294 assert_eq!(affine.bits, kquant.bits);
2295 assert_eq!(affine.bits, mx.bits);
2296 assert_ne!(affine.scheme, kquant.scheme);
2297 assert_ne!(affine.scheme, mx.scheme);
2298 // And within GGUF, k-quant is not round-to-nearest.
2299 assert_ne!(
2300 Quantization::parse("Q5_K_S").scheme,
2301 Quantization::parse("q5_0").scheme
2302 );
2303 }
2304
2305 /// The scheme names a numeric format, never a container or an engine.
2306 /// whisper.cpp ships `q5_0` ggml checkpoints that no GGUF text path can
2307 /// load; a container-named variant would assert otherwise.
2308 #[test]
2309 fn scheme_does_not_imply_an_engine() {
2310 let whisper = Quantization::parse("q5_0");
2311 let llama = Quantization::parse("Q5_0");
2312 assert_eq!(whisper.scheme, llama.scheme);
2313 assert_eq!(whisper.scheme, QuantScheme::RtnBlock);
2314 }
2315
2316 #[test]
2317 fn mlx_config_block_survives_ingest() {
2318 let q = Quantization::from_mlx_config(Some(8), Some(32), Some("mxfp8")).unwrap();
2319 assert_eq!(q.bits, Some(8));
2320 assert_eq!(q.group_size, Some(32));
2321 assert_eq!(q.scheme, QuantScheme::BlockScaledFloat);
2322
2323 // Absent `mode` means affine — MLX's default, not unknown.
2324 let affine = Quantization::from_mlx_config(Some(4), Some(64), None).unwrap();
2325 assert_eq!(affine.scheme, QuantScheme::AffineGroupInt);
2326 assert_eq!(affine.group_size, Some(64));
2327 assert_eq!(affine.label, "4bit");
2328 }
2329
2330 /// An empty or absent block must not become a claim of affine group quant.
2331 /// The field it replaced returned nothing here, and asserting a scheme is
2332 /// exactly the error a router acting on `scheme` would inherit.
2333 #[test]
2334 fn empty_mlx_block_asserts_nothing() {
2335 assert!(Quantization::from_mlx_config(None, None, None).is_none());
2336 }
2337
2338 #[test]
2339 fn deserializes_legacy_bare_string() {
2340 let q: Quantization = serde_json::from_str(r#""Q4_K_M""#).unwrap();
2341 assert_eq!(q.scheme, QuantScheme::KQuantMixed);
2342 assert_eq!(q.bits, Some(4));
2343 assert_eq!(q.label, "Q4_K_M");
2344 }
2345
2346 #[test]
2347 fn deserializes_structured_object() {
2348 let q: Quantization = serde_json::from_str(
2349 r#"{"bits":4,"scheme":"affine_group_int","group_size":64,"label":"4bit"}"#,
2350 )
2351 .unwrap();
2352 assert_eq!(q.group_size, Some(64));
2353 assert_eq!(q.scheme, QuantScheme::AffineGroupInt);
2354 }
2355
2356 /// A partial object recovers from its label rather than defaulting to
2357 /// Unknown — the same reason the bare string is still accepted.
2358 #[test]
2359 fn partial_object_recovers_from_label() {
2360 let q: Quantization = serde_json::from_str(r#"{"label":"Q8_0"}"#).unwrap();
2361 assert_eq!(q.scheme, QuantScheme::RtnBlock);
2362 assert_eq!(q.bits, Some(8));
2363 }
2364
2365 /// A malformed object must be an error, not a row that silently claims
2366 /// `Unknown`. Under `#[serde(untagged)]` every one of these deserialized
2367 /// successfully into a fabricated descriptor.
2368 #[test]
2369 fn malformed_objects_are_rejected() {
2370 for bad in [
2371 r#"{"btis":4,"scehme":"k_quant_mixed","labl":"Q4_K_M"}"#, // typos
2372 r#"{"quantization":{"bits":4}}"#, // double-nested
2373 r#"{}"#, // nothing at all
2374 r#"{"group_size":64}"#, // no label
2375 r#"{"bits":4,"scheme":"k_quant_mixed"}"#, // no label
2376 ] {
2377 let parsed: Result<Quantization, _> = serde_json::from_str(bad);
2378 assert!(parsed.is_err(), "should have rejected {bad}");
2379 }
2380 }
2381
2382 /// The error must name what was wrong. `#[serde(untagged)]` reported only
2383 /// "data did not match any variant", for a `models.json` that fails whole.
2384 #[test]
2385 fn rejection_names_the_offending_field() {
2386 let err = serde_json::from_str::<Quantization>(
2387 r#"{"bits":4,"scheme":"affine_grp_int","label":"4bit"}"#,
2388 )
2389 .unwrap_err()
2390 .to_string();
2391 assert!(
2392 err.contains("affine_grp_int") || err.contains("scheme"),
2393 "unhelpful error: {err}"
2394 );
2395 }
2396
2397 #[test]
2398 fn round_trips_through_the_object_form() {
2399 for label in ["Q4_K_M", "4bit", "mxfp8", "bf16", "weird-vendor-format"] {
2400 let q = Quantization::parse(label);
2401 let round: Quantization =
2402 serde_json::from_str(&serde_json::to_string(&q).unwrap()).unwrap();
2403 assert_eq!(q, round, "round trip for {label}");
2404 }
2405 }
2406
2407 /// The label alone is the wire form whenever it is lossless. This is what
2408 /// keeps `row_digest` stable for rows that gained no new information.
2409 #[test]
2410 fn serializes_as_a_bare_label_when_lossless() {
2411 for label in [
2412 "Q4_K_M",
2413 "4bit",
2414 "mxfp8",
2415 "bf16",
2416 "q5_0",
2417 "weird-vendor-format",
2418 ] {
2419 let json = serde_json::to_string(&Quantization::parse(label)).unwrap();
2420 assert_eq!(json, format!("\"{label}\""), "should stay a bare string");
2421 }
2422 }
2423
2424 /// ...and the object form appears exactly where the label would lose
2425 /// something: a group size, or a scheme the label cannot express.
2426 #[test]
2427 fn serializes_as_an_object_only_when_it_adds_information() {
2428 let with_group = Quantization::from_mlx_config(Some(4), Some(64), None).unwrap();
2429 let json = serde_json::to_value(&with_group).unwrap();
2430 assert_eq!(json["group_size"], 64, "group_size must survive the write");
2431 assert_eq!(json["label"], "4bit");
2432
2433 // A bare `Q4` parses as Unknown, so an explicit scheme is real
2434 // information and has to be written out.
2435 let disambiguated = Quantization {
2436 bits: Some(4),
2437 scheme: QuantScheme::AffineGroupInt,
2438 group_size: None,
2439 label: "Q4".into(),
2440 };
2441 let json = serde_json::to_value(&disambiguated).unwrap();
2442 assert_eq!(json["scheme"], "affine_group_int");
2443 assert!(json.get("group_size").is_none(), "no null padding");
2444 }
2445
2446 /// Both write forms must read back identically, or the minimal-write rule
2447 /// would trade a digest change for silent data loss.
2448 #[test]
2449 fn every_write_form_round_trips() {
2450 let cases = [
2451 Quantization::parse("Q4_K_M"),
2452 Quantization::parse("bf16"),
2453 Quantization::parse("unattributable-format"),
2454 Quantization::from_mlx_config(Some(8), Some(32), Some("mxfp8")).unwrap(),
2455 Quantization::from_mlx_config(Some(4), Some(64), None).unwrap(),
2456 Quantization {
2457 bits: Some(4),
2458 scheme: QuantScheme::AffineGroupInt,
2459 group_size: None,
2460 label: "Q4".into(),
2461 },
2462 ];
2463 for q in cases {
2464 let round: Quantization =
2465 serde_json::from_str(&serde_json::to_string(&q).unwrap()).unwrap();
2466 assert_eq!(q, round, "round trip for {q:?}");
2467 }
2468 }
2469
2470 /// The digest guard. `catalog_identity::row_digest` is a SHA-256 over the
2471 /// serialized schema and clients pin it through
2472 /// `expected_catalog_revision`, so a row whose quantization gained no new
2473 /// information must re-serialize to the byte-identical JSON it was read
2474 /// from. Only rows that genuinely changed may move.
2475 #[test]
2476 fn catalog_quantizations_reserialize_unchanged() {
2477 let raw: Vec<serde_json::Value> =
2478 serde_json::from_str(include_str!("builtin_catalog.json")).unwrap();
2479 let parsed: Vec<ModelSchema> =
2480 serde_json::from_str(include_str!("builtin_catalog.json")).unwrap();
2481 let mut moved = Vec::new();
2482 for (raw_row, model) in raw.iter().zip(&parsed) {
2483 let before = raw_row
2484 .get("quantization")
2485 .cloned()
2486 .unwrap_or(serde_json::Value::Null);
2487 let after = serde_json::to_value(&model.quantization).unwrap();
2488 if before != after {
2489 moved.push(format!("{}: {before} -> {after}", model.id));
2490 }
2491 }
2492 assert!(
2493 moved.is_empty(),
2494 "these rows would change catalog digest: {moved:#?}"
2495 );
2496 }
2497
2498 /// Every quantization the built-in catalog ships must classify. An entry
2499 /// may carry an explicit `scheme` only where its label is genuinely
2500 /// ambiguous — the guard is that an explicit scheme never *contradicts* a
2501 /// label the parser can already read, which is how catalog data stops
2502 /// being a place to make a failing test pass.
2503 #[test]
2504 fn builtin_catalog_quantizations_are_coherent() {
2505 let catalog: Vec<ModelSchema> =
2506 serde_json::from_str(include_str!("builtin_catalog.json")).unwrap();
2507 let mut problems = Vec::new();
2508 for model in &catalog {
2509 let Some(q) = &model.quantization else {
2510 continue;
2511 };
2512 if q.scheme == QuantScheme::Unknown {
2513 problems.push(format!("{}: unclassified label {:?}", model.id, q.label));
2514 continue;
2515 }
2516 let from_label = Quantization::parse(&q.label).scheme;
2517 if from_label != QuantScheme::Unknown && from_label != q.scheme {
2518 problems.push(format!(
2519 "{}: label {:?} parses as {:?} but the row claims {:?}",
2520 model.id, q.label, from_label, q.scheme
2521 ));
2522 }
2523 }
2524 assert!(
2525 problems.is_empty(),
2526 "incoherent catalog rows: {problems:#?}"
2527 );
2528 }
2529}