Skip to main content

car_inference/
schema.rs

1//! Model schema — declarative metadata for models, analogous to ToolSchema for tools.
2//!
3//! Every model (local GGUF, remote API, Ollama) is described by a `ModelSchema`
4//! that declares identity, capabilities, constraints, cost, and source.
5//! The router uses this schema for initial routing; observed outcomes refine it.
6
7use serde::{Deserialize, Serialize};
8
9/// What a model can do.
10#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, schemars::JsonSchema)]
11#[serde(rename_all = "snake_case")]
12pub enum ModelCapability {
13    /// Text completion / chat generation
14    Generate,
15    /// Vector embeddings
16    Embed,
17    /// Cross-encoder relevance scoring (query + document → relevance
18    /// score). Qwen3-Reranker is the canonical local implementation.
19    Rerank,
20    /// Label assignment / classification
21    Classify,
22    /// Code generation, repair, refactoring
23    Code,
24    /// Chain-of-thought, planning, analysis
25    Reasoning,
26    /// Text condensation
27    Summarize,
28    /// Function/tool calling
29    ToolUse,
30    /// Multiple tool calls in a single response (parallel tool execution)
31    MultiToolCall,
32    /// Vision / image understanding
33    Vision,
34    /// Video understanding (multi-frame sampling + temporal tokens).
35    /// Distinct from `Vision` so routing can prefer video-trained
36    /// models when the caller attaches a video content block.
37    VideoUnderstanding,
38    /// Audio understanding (speech + non-speech audio as an input to
39    /// a chat/reasoning model). Distinct from `SpeechToText` which is
40    /// the transcription-only task. Gemma 4 E2B/E4B and Gemini do
41    /// this; Qwen2.5-VL does not.
42    AudioUnderstanding,
43    /// Visual grounding — structured object-localization output
44    /// (bounding boxes keyed to object labels) in addition to text.
45    Grounding,
46    /// Speech recognition / transcription
47    SpeechToText,
48    /// Speech synthesis / text-to-speech
49    TextToSpeech,
50    /// Image generation
51    ImageGeneration,
52    /// Video generation
53    VideoGeneration,
54}
55
56/// How much the project vouches for a model. Gates automatic upgrades and
57/// is surfaced in recommendation rationale. Closed enum — a new tier is a
58/// deliberate FFI-visible change, never a silent string fallback.
59#[derive(
60    Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, Default, schemars::JsonSchema,
61)]
62#[serde(rename_all = "snake_case")]
63pub enum TrustTier {
64    /// Vetted by the project — the built-in catalog and verified upgrades.
65    /// Eligible for background auto-apply when the user opts in.
66    #[default]
67    Curated,
68    /// User-registered or upstream-discovered, not project-vetted. Always
69    /// notify-only; never auto-applied regardless of update policy.
70    Community,
71}
72
73/// How to access the model.
74#[derive(Debug, Clone, Serialize, Deserialize, schemars::JsonSchema)]
75#[serde(tag = "type", rename_all = "snake_case")]
76pub enum ModelSource {
77    /// Local GGUF file via Candle backend.
78    Local {
79        hf_repo: String,
80        hf_filename: String,
81        tokenizer_repo: String,
82    },
83    /// Remote API endpoint (OpenAI-compatible, Anthropic, etc.)
84    RemoteApi {
85        endpoint: String,
86        /// Environment variable name containing the API key (never the key itself).
87        /// The env var value may contain comma-separated keys for load balancing.
88        api_key_env: String,
89        /// Additional environment variable names for load balancing across multiple keys.
90        /// Each env var may also contain comma-separated keys.
91        #[serde(default)]
92        api_key_envs: Vec<String>,
93        #[serde(default)]
94        api_version: Option<String>,
95        protocol: ApiProtocol,
96    },
97    /// One text-generation turn through the locally installed OpenAI Codex CLI.
98    ///
99    /// Codex remains the sole holder of its ChatGPT-subscription credential:
100    /// CAR neither reads that credential nor accepts an API key for this source.
101    /// An optional reasoning suffix is part of the model string (for example,
102    /// `gpt-5.6-sol:high`) and maps to Codex's `model_reasoning_effort` config.
103    CodexCli { model: String },
104    /// Ollama local server.
105    Ollama {
106        model_tag: String,
107        #[serde(default = "default_ollama_host")]
108        host: String,
109    },
110    /// Local MLX model via mlx-rs backend (Apple Silicon, safetensors format).
111    /// Models from mlx-community on HuggingFace.
112    Mlx {
113        /// HuggingFace repo (e.g., "mlx-community/Qwen3-4B-4bit").
114        hf_repo: String,
115        /// Optional specific weight filename. If None, auto-discovers safetensors files.
116        #[serde(default)]
117        hf_weight_file: Option<String>,
118    },
119    /// Local whisper.cpp speech-to-text model — a ggml `.bin` from the
120    /// `ggerganov/whisper.cpp` HF repo, run in-process via the shared
121    /// `car-whisper` crate. Cross-platform (Windows/Linux/macOS): this is the
122    /// on-device STT path where MLX isn't available. Cached at
123    /// `~/.tokhn/whisper/ggml-<model>.bin`.
124    WhisperCpp {
125        /// whisper.cpp model id — the suffix of `ggml-<model>.bin`
126        /// (e.g. `"large-v3-turbo-q5_0"`).
127        model: String,
128    },
129    /// Windows OS text-to-speech via `Windows.Media.SpeechSynthesis` (WinRT),
130    /// run in-process. The catalog-side analog of the `car-voice`
131    /// `TtsProvider::WindowsSpeech` live path and the parity counterpart of
132    /// Apple's OS synthesizer — free, on-device, no model download, no MLX.
133    /// Windows-only; availability is `false` on every other target (like
134    /// `AppleFoundationModels`).
135    WindowsSpeech {},
136    /// Local vLLM-MLX server (Apple Silicon, OpenAI-compatible API).
137    /// Routes through RemoteBackend with OpenAI protocol handler.
138    VllmMlx {
139        /// Server endpoint (e.g., "http://localhost:8000").
140        endpoint: String,
141        /// The model name as known to vLLM-MLX (e.g., "mlx-community/Qwen3-4B-4bit").
142        model_name: String,
143    },
144    /// CAR-owned supervised vLLM-MLX process backed by a managed HuggingFace
145    /// artifact. Unlike `VllmMlx`, CAR downloads, admits, spawns, reaps, and
146    /// accounts this allocation. Dispatch rewrites a clone to `VllmMlx` only
147    /// after the child reports healthy.
148    ManagedVllmMlx {
149        hf_repo: String,
150        #[serde(default)]
151        hf_weight_file: Option<String>,
152    },
153    /// Apple's on-device system model via the FoundationModels framework
154    /// (macOS 26+, Apple Silicon). Inference happens in-process through a
155    /// Swift shim — there is no HTTP, no API key, and no model file: the
156    /// OS owns the weights. Availability is checked at runtime via
157    /// `@available(macOS 26.0, *)`; on older macOS or non-Apple-Silicon
158    /// hosts the backend reports `UnsupportedMode` and the router falls
159    /// through to the next candidate.
160    AppleFoundationModels {
161        /// Optional Apple use-case hint passed through to
162        /// `LanguageModelSession`. Apple's framework tunes its prompt and
163        /// safety scaffolding per use case (e.g. "general", "summarize").
164        /// `None` uses the default.
165        #[serde(default)]
166        use_case: Option<String>,
167    },
168    /// Proprietary provider with custom auth and protocol.
169    ///
170    /// For vendor-specific APIs that aren't generic OpenAI-compatible endpoints.
171    /// Parslee is the first proprietary provider — custom auth (OAuth2),
172    /// custom response format, multi-provider routing built into the API.
173    Proprietary {
174        /// Provider identifier (e.g., "parslee").
175        provider: String,
176        /// Base URL for the API.
177        endpoint: String,
178        /// Auth configuration.
179        auth: ProprietaryAuth,
180        /// Custom protocol details.
181        protocol: ProprietaryProtocol,
182    },
183    /// Inference is delegated to a host-registered runner. CAR does
184    /// not own the wire format — the runner (typically a JS / Python
185    /// host) translates the `GenerateRequest` to its provider's API,
186    /// streams chunks back through the runner's event callback, and
187    /// returns the final aggregated result.
188    ///
189    /// Closes Parslee-ai/car-releases#24. Use this when the host
190    /// already has an SDK relationship with a provider (Anthropic,
191    /// OpenAI, GitHub Models, Vercel AI SDK) and wants CAR to sit in
192    /// the lifecycle / policy / replay path without learning every
193    /// provider's wire format.
194    ///
195    /// Routing requires that a runner has been registered via
196    /// [`crate::set_inference_runner`] (or its FFI equivalent —
197    /// `registerInferenceRunner` on JS, `register_inference_runner`
198    /// on Python, the `InferenceRunner` foreign trait on UniFFI,
199    /// `inference.register_runner` on the WebSocket protocol).
200    /// Without a runner, dispatch fails with `InferenceFailed`.
201    Delegated {
202        /// Opaque hint passed through to the runner — typically the
203        /// provider id (`"anthropic"`, `"openai"`, `"vercel-ai-sdk"`)
204        /// so a multi-provider runner can dispatch internally. CAR
205        /// does not interpret this string.
206        #[serde(default)]
207        hint: Option<String>,
208    },
209}
210
211/// Provider id for ChatGPT subscription inference backed by the Codex credential record.
212pub const OPENAI_CODEX_PROVIDER: &str = "openai-codex";
213
214/// What a signed-out openai-codex row tells a person to do.
215pub const OPENAI_CODEX_SIGN_IN_HINT: &str = "sign in with `car auth login --provider openai-codex`";
216
217/// Authentication method for proprietary providers.
218#[derive(Debug, Clone, Serialize, Deserialize, schemars::JsonSchema)]
219#[serde(tag = "type", rename_all = "snake_case")]
220pub enum ProprietaryAuth {
221    /// OAuth2 PKCE flow (e.g., Azure AD for Parslee).
222    #[serde(rename = "oauth2_pkce", alias = "o_auth2_pkce")]
223    OAuth2Pkce {
224        authority: String,
225        client_id: String,
226        scopes: Vec<String>,
227    },
228    /// Static API key from environment variable.
229    ApiKeyEnv { env_var: String },
230    /// Bearer token from environment variable.
231    BearerTokenEnv { env_var: String },
232    /// ChatGPT subscription OAuth read from the car-auth Codex record; admitted only for provider
233    /// `openai-codex`.
234    #[serde(rename = "chatgpt_subscription")]
235    ChatGptSubscription {},
236}
237
238/// Protocol configuration for proprietary providers.
239#[derive(Debug, Clone, Serialize, Deserialize, schemars::JsonSchema)]
240pub struct ProprietaryProtocol {
241    /// Chat/completion endpoint path (appended to base URL).
242    #[serde(default = "default_chat_path")]
243    pub chat_path: String,
244    /// Content type for requests.
245    #[serde(default = "default_content_type")]
246    pub content_type: String,
247    /// Whether the API streams responses via SSE.
248    #[serde(default)]
249    pub streaming: bool,
250    /// Custom headers to include in every request.
251    #[serde(default)]
252    pub extra_headers: std::collections::HashMap<String, String>,
253    /// The wire format the endpoint speaks. A row names it rather than CAR
254    /// inferring it from the path, so the model decides the backend.
255    #[serde(default)]
256    pub wire: ProprietaryWire,
257}
258
259/// What a proprietary endpoint speaks.
260#[derive(
261    Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize, schemars::JsonSchema,
262)]
263#[serde(rename_all = "snake_case")]
264pub enum ProprietaryWire {
265    /// OpenAI Responses-style chat (Parslee's `/inference/responses`).
266    #[default]
267    Responses,
268    /// TypeSafe System One (Parslee's `/inference/typesafe/v1/systemone`):
269    /// typed decisions with a probability per option, no text generation.
270    /// Such a row serves `classify` only.
271    SystemOne,
272}
273
274impl Default for ProprietaryProtocol {
275    fn default() -> Self {
276        Self {
277            chat_path: default_chat_path(),
278            content_type: default_content_type(),
279            streaming: false,
280            extra_headers: std::collections::HashMap::new(),
281            wire: ProprietaryWire::default(),
282        }
283    }
284}
285
286fn default_chat_path() -> String {
287    "/chat".to_string()
288}
289
290fn default_content_type() -> String {
291    "application/json".to_string()
292}
293
294fn default_ollama_host() -> String {
295    "http://localhost:11434".to_string()
296}
297
298#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, schemars::JsonSchema)]
299#[serde(rename_all = "snake_case")]
300pub enum ApiProtocol {
301    OpenAiCompat,
302    /// OpenRouter's OpenAI-compatible Chat Completions surface. Distinct so
303    /// credential precedence and error translation stay provider-specific.
304    OpenRouter,
305    /// OpenAI Responses API (/v1/responses) — works with all OpenAI models including codex.
306    OpenAiResponses,
307    Anthropic,
308    Google,
309    /// Azure OpenAI — uses api-key header and deployment-based URLs.
310    /// Endpoint format: {base}/openai/deployments/{model}/chat/completions?api-version={version}
311    AzureOpenAi,
312    /// Google Vertex AI — the enterprise Gemini surface. Same request/response
313    /// shape as the AI-Studio `Google` protocol, but a project/location URL and
314    /// OAuth Bearer auth (a GCP access token from `gcloud auth print-access-token`
315    /// or a service account) instead of an `?key=` query param. Endpoint format:
316    /// `{base}/publishers/google/models/{model}:generateContent`, where `base`
317    /// is `https://{loc}-aiplatform.googleapis.com/v1/projects/{proj}/locations/{loc}`.
318    VertexAi,
319    /// AWS Bedrock — the **Converse** API (`bedrock-runtime`), a unified
320    /// messages surface across Bedrock-hosted models (Claude, Llama, Mistral,
321    /// Titan, …). Auth is **SigV4** request signing (not a bearer token), with
322    /// credentials from the standard AWS env vars; the model's `endpoint` is the
323    /// region (e.g. `us-east-1`) and `name` is the Bedrock model id. Non-stream
324    /// only for now (Converse streaming uses a separate binary event-stream).
325    Bedrock,
326}
327
328impl ApiProtocol {
329    /// Prompt-cache economics for this provider, relative to its base input
330    /// rate — used by the cost scoreboard to price cached tokens correctly.
331    /// Anthropic uses explicit breakpoints (deep read discount + write
332    /// premium); OpenAI/Azure cache automatically (~0.5× read, no write
333    /// charge). Providers whose cache tokens CAR does not parse (Google/Vertex/
334    /// Bedrock) report zero cache tokens, so their rates are inert.
335    pub fn cache_rates(&self) -> crate::outcome::CacheRates {
336        use crate::outcome::CacheRates;
337        match self {
338            ApiProtocol::Anthropic => CacheRates::ANTHROPIC,
339            ApiProtocol::OpenAiCompat | ApiProtocol::OpenAiResponses | ApiProtocol::AzureOpenAi => {
340                CacheRates::OPENAI
341            }
342            // OpenRouter rates differ per upstream model and are carried by
343            // ModelSchema::cost. A blanket OpenAI-shaped discount is false.
344            ApiProtocol::OpenRouter => CacheRates::NONE,
345            ApiProtocol::Google | ApiProtocol::VertexAi | ApiProtocol::Bedrock => CacheRates::NONE,
346        }
347    }
348
349    /// This protocol as an OpenTelemetry GenAI `gen_ai.provider.name` value.
350    ///
351    /// The protocol is the right source for this attribute and
352    /// [`ModelSchema::provider`] is not: that field carries the model's
353    /// *publisher* (`qwen`, `mlx-community`, `apple`, `ggerganov`), while
354    /// `gen_ai.provider.name` names the API surface the request is spoken to.
355    /// A Qwen checkpoint served over Bedrock is `aws.bedrock`, not `qwen`.
356    ///
357    /// Both OpenAI protocols map to `openai` on purpose. The convention
358    /// defines this attribute as a discriminator for the *telemetry format
359    /// flavor*, not the company serving the weights — so every
360    /// OpenAI-REST-shaped surface reports `openai`, and Chat Completions vs
361    /// Responses is not a distinction it draws. `openrouter` has no
362    /// well-known value; the convention's fallback is a provider-specific
363    /// string, which is what this returns.
364    ///
365    /// Exhaustive by intent: a new protocol must decide its own value rather
366    /// than inherit a wildcard's guess.
367    pub fn genai_provider_name(&self) -> &'static str {
368        match self {
369            ApiProtocol::OpenAiCompat | ApiProtocol::OpenAiResponses => "openai",
370            ApiProtocol::OpenRouter => "openrouter",
371            ApiProtocol::Anthropic => "anthropic",
372            ApiProtocol::Google => "gcp.gemini",
373            ApiProtocol::VertexAi => "gcp.vertex_ai",
374            ApiProtocol::AzureOpenAi => "azure.ai.openai",
375            ApiProtocol::Bedrock => "aws.bedrock",
376        }
377    }
378}
379
380/// The numeric format a checkpoint's weights are stored in.
381///
382/// Deliberately says nothing about *which engine* serves the file —
383/// [`ModelSource`] already carries that, and keying this on the container
384/// instead of the format produces falsehoods: whisper.cpp's `q5_0` ggml
385/// checkpoints use the same round-to-nearest block format as llama.cpp's, and
386/// a `Gguf*` variant would assert a GGUF text path that cannot load them.
387///
388/// What it does express is the axis neither the bit width nor the source can:
389/// an MLX 4-bit affine checkpoint and a `Q4_K_M` are both "4-bit", were
390/// produced by different algorithms, and do not have the same quality.
391#[derive(
392    Debug, Clone, Copy, PartialEq, Eq, Hash, Default, Serialize, Deserialize, schemars::JsonSchema,
393)]
394#[serde(rename_all = "snake_case")]
395pub enum QuantScheme {
396    /// Integer weights with a scale and bias shared across `group_size`
397    /// elements (`4bit`, `6bit`). MLX's default and only affine format.
398    AffineGroupInt,
399    /// Block-scaled float — MLX's `mxfp4`, `mxfp8` and NVIDIA's `nvfp4`. A
400    /// genuinely different loader path from [`QuantScheme::AffineGroupInt`] at
401    /// the same nominal width, which is why the width alone cannot classify a
402    /// checkpoint; see `backend::local::quantization_is_decodable`.
403    BlockScaledFloat,
404    /// Mixed per-tensor bit allocation, optionally importance-weighted
405    /// (`Q4_K_M`, `Q5_K_S`, `IQ4_XS`, `TQ1_0`).
406    KQuantMixed,
407    /// Uniform round-to-nearest blocks, no per-tensor mixing and no importance
408    /// weighting (`Q8_0`, `q5_0`, `Q4_0_4_4`).
409    RtnBlock,
410    /// Full-precision weights (`bf16`, `f16`, `f32`). Distinct from `None` at
411    /// the [`ModelSchema::quantization`] level: a positive claim that the
412    /// checkpoint is unquantized, not an absence of information.
413    Unquantized,
414    /// A format this parser could not identify. The label is still preserved
415    /// verbatim; only the classification is missing.
416    #[default]
417    Unknown,
418}
419
420/// Structured quantization descriptor.
421///
422/// This replaces a free-text string that mixed two vocabularies — MLX's
423/// `4bit`/`6bit` and GGUF's `Q4_K_M`/`Q8_0` — under one field, so nothing could
424/// tell a group-quantized integer checkpoint from a mixed k-quant one. It also
425/// threw away what the ingest paths already had in hand: MLX `config.json`
426/// declares `{bits, group_size, mode}` and only `bits` survived.
427///
428/// Deserializes from **either** the legacy bare string or the structured
429/// object, so catalogs, user `models.json` files, and `models.register`
430/// payloads written before this change keep loading unchanged. The object form
431/// requires `label` and rejects unknown fields: a misspelled key is a producer
432/// bug, and accepting it silently is how the field it replaced lost
433/// information in the first place.
434///
435/// **Writes the bare label back whenever that is lossless**, and the object
436/// only when it carries something parsing cannot recover — a `group_size`, or
437/// a `scheme` that disambiguates a label the parser reads as `Unknown`. Two
438/// reasons, both about blast radius rather than taste. This value is inside
439/// `catalog_identity::row_digest`, so an unconditional shape change would move
440/// the digest of every quantized row and hard-fail any client that pinned
441/// `expected_catalog_revision`. And `registry::load_user_config` fails a
442/// `models.json` **whole**, with its only production caller discarding the
443/// error — so an older daemon meeting an object it cannot parse boots with
444/// zero registered models and says nothing. Emitting the object only where it
445/// adds information keeps both costs proportional to what actually changed.
446#[allow(dead_code)]
447#[derive(schemars::JsonSchema)]
448#[serde(untagged)]
449enum QuantizationWireSchema {
450    Label(String),
451    Structured(QuantizationObjectWireSchema),
452}
453
454#[allow(dead_code)]
455#[derive(schemars::JsonSchema)]
456#[serde(deny_unknown_fields)]
457struct QuantizationObjectWireSchema {
458    bits: Option<u8>,
459    scheme: QuantScheme,
460    group_size: Option<u32>,
461    label: String,
462}
463
464#[derive(Debug, Clone, PartialEq, Eq)]
465pub struct Quantization {
466    /// Weight bit width. `None` when the label names none.
467    pub bits: Option<u8>,
468    /// Which quantization family this is.
469    pub scheme: QuantScheme,
470    /// Elements sharing one scale/zero point (MLX: 32, 64, 128). `None` when
471    /// the scheme has no group concept, or the source declared none.
472    pub group_size: Option<u32>,
473    /// The label exactly as published. Never synthesized: it is what the user
474    /// sees, what Hugging Face repos are named after, and the only thing that
475    /// survives a scheme CAR does not recognize yet.
476    pub label: String,
477}
478
479/// Widths affine group quantization actually uses. A `16bit` or `32bit`
480/// label is full precision that happens to be spelled like a quant.
481const AFFINE_GROUP_WIDTHS: std::ops::RangeInclusive<u8> = 2..=8;
482
483impl Quantization {
484    /// Best-effort structure from a published label.
485    ///
486    /// Never fails and never invents: an unrecognized label yields
487    /// [`QuantScheme::Unknown`] with the label intact.
488    pub fn parse(label: &str) -> Self {
489        let label = label.trim();
490        let lower = label.to_ascii_lowercase();
491        let build = |bits: Option<u8>, scheme: QuantScheme| Self {
492            bits,
493            scheme,
494            group_size: None,
495            label: label.to_string(),
496        };
497
498        // Full precision. `fp8` is deliberately NOT here — an 8-bit float is a
499        // quantized weight format, just not one this parser can attribute.
500        match lower.as_str() {
501            "bf16" | "f16" | "fp16" | "float16" | "half" => {
502                return build(Some(16), QuantScheme::Unquantized)
503            }
504            "f32" | "fp32" | "float32" | "full" => {
505                return build(Some(32), QuantScheme::Unquantized)
506            }
507            "none" => return build(None, QuantScheme::Unquantized),
508            "f8" | "fp8" | "float8" => return build(Some(8), QuantScheme::Unknown),
509            // Empty means nobody said, which is not a claim of full precision.
510            "" => return build(None, QuantScheme::Unknown),
511            _ => {}
512        }
513
514        // MLX block-scaled floats: `mxfp4`, `mxfp8`. A bare `mxfp` names no
515        // width, so it identifies nothing.
516        if let Some(rest) = lower.strip_prefix("mxfp") {
517            return match leading_number(rest) {
518                Some((bits, _)) => build(Some(bits), QuantScheme::BlockScaledFloat),
519                None => build(None, QuantScheme::Unknown),
520            };
521        }
522
523        // MLX affine group quant: `4bit`, `4-bit`, `6bit`.
524        if let Some(width) = lower.strip_suffix("bit").map(|w| w.trim_end_matches('-')) {
525            if let Some((bits, consumed)) = leading_number(width) {
526                if consumed == width.len() && AFFINE_GROUP_WIDTHS.contains(&bits) {
527                    return build(Some(bits), QuantScheme::AffineGroupInt);
528                }
529                // `16bit`/`32bit` are full precision spelled as a width.
530                if consumed == width.len() && (bits == 16 || bits == 32) {
531                    return build(Some(bits), QuantScheme::Unquantized);
532                }
533            }
534        }
535
536        // GGUF. `iq`/`tq` are k-quant families; a bare `q` needs its suffix
537        // read to tell k-quant from legacy round-to-nearest.
538        let gguf = lower
539            .strip_prefix("iq")
540            .or_else(|| lower.strip_prefix("tq"))
541            .map(|rest| (rest, true))
542            .or_else(|| lower.strip_prefix('q').map(|rest| (rest, false)));
543        if let Some((rest, k_family)) = gguf {
544            if let Some((bits, consumed)) = leading_number(rest) {
545                if bits == 0 {
546                    return build(None, QuantScheme::Unknown);
547                }
548                // Slice by digits consumed, not by the width's decimal length —
549                // `q08_0` has a two-character prefix for a one-character number.
550                let suffix = &rest[consumed..];
551                let scheme = if k_family || suffix.contains("_k") {
552                    QuantScheme::KQuantMixed
553                } else if suffix.starts_with("_0") || suffix.starts_with("_1") {
554                    // `starts_with`, not equality: the aarch64 repack quants
555                    // are `Q4_0_4_4`, `Q4_0_4_8`, `Q4_0_8_8`.
556                    QuantScheme::RtnBlock
557                } else {
558                    // A bare `Q4` names a width and no producer — llama.cpp
559                    // has no such format, so this came from somewhere else.
560                    QuantScheme::Unknown
561                };
562                return build(Some(bits), scheme);
563            }
564        }
565
566        build(None, QuantScheme::Unknown)
567    }
568
569    /// Recover the quantization a GGUF file names in its own filename
570    /// (`Qwen3-8B-Q4_K_M.gguf`, `ggml-large-v3-turbo-q5_0.gguf`).
571    ///
572    /// Scans the hyphen-separated segments from the right and takes the first
573    /// that classifies, so a model whose *name* contains something quant-shaped
574    /// does not outrank the real suffix. Returns `None` rather than guessing
575    /// when nothing in the name is recognizable.
576    pub fn from_gguf_filename(filename: &str) -> Option<Self> {
577        let stem = filename
578            .rsplit_once('.')
579            .map(|(stem, _)| stem)
580            .unwrap_or(filename);
581        stem.rsplit('-')
582            .map(Self::parse)
583            .find(|q| q.scheme != QuantScheme::Unknown)
584    }
585
586    /// Descriptor for an MLX checkpoint, from the `quantization` block of its
587    /// `config.json`. `mode` is MLX's own name for the format (`affine`,
588    /// `mxfp4`, `mxfp8`); absent means affine, which is MLX's default.
589    ///
590    /// Returns `None` when the block carried nothing usable — declaring
591    /// `AffineGroupInt` on the strength of an empty object would assert exactly the
592    /// thing this type exists to establish.
593    pub fn from_mlx_config(
594        bits: Option<u8>,
595        group_size: Option<u32>,
596        mode: Option<&str>,
597    ) -> Option<Self> {
598        if bits.is_none() && group_size.is_none() && mode.is_none() {
599            return None;
600        }
601        let normalized = mode.map(str::to_ascii_lowercase);
602        let scheme = match normalized.as_deref() {
603            // MLX's `QuantizationMode` is exactly these four. `nvfp4` used to
604            // fall through to `Unknown`, which was a refusal dressed as a parse
605            // failure — it is block-scaled float like the MX pair and the
606            // linked loader reads it.
607            Some(m) if m.starts_with("mxfp") || m == "nvfp4" => QuantScheme::BlockScaledFloat,
608            Some("affine") | None => QuantScheme::AffineGroupInt,
609            Some(_) => QuantScheme::Unknown,
610        };
611        let label = match (normalized.as_deref(), bits) {
612            (Some(m), _) if m != "affine" => m.to_string(),
613            (_, Some(b)) => format!("{b}bit"),
614            // No width and no distinguishing mode: MLX said "affine" and
615            // nothing else. Say that rather than inventing a width.
616            (_, None) => "affine".to_string(),
617        };
618        Some(Self {
619            bits,
620            scheme,
621            group_size,
622            label,
623        })
624    }
625}
626
627/// Leading run of ASCII digits as a bit width, with the number of bytes it
628/// occupied. The byte count is returned because it is not recoverable from the
629/// value — `08` and `8` parse the same and slice differently.
630fn leading_number(s: &str) -> Option<(u8, usize)> {
631    let digits: String = s.chars().take_while(char::is_ascii_digit).collect();
632    if digits.is_empty() {
633        return None;
634    }
635    // Overflow (`Q256_K`) is a parse failure, not a silent truncation.
636    digits.parse().ok().map(|n| (n, digits.len()))
637}
638
639impl std::fmt::Display for Quantization {
640    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
641        f.write_str(&self.label)
642    }
643}
644
645impl Serialize for Quantization {
646    fn serialize<S: serde::Serializer>(&self, serializer: S) -> Result<S::Ok, S::Error> {
647        use serde::ser::SerializeStruct;
648
649        // Round-tripping through `parse` is the exact test for "the label
650        // carries everything": if it reproduces this value, the object form
651        // would be a longer spelling of the same information.
652        if Self::parse(&self.label) == *self {
653            return serializer.serialize_str(&self.label);
654        }
655
656        let len = 2 + usize::from(self.bits.is_some()) + usize::from(self.group_size.is_some());
657        let mut row = serializer.serialize_struct("Quantization", len)?;
658        if let Some(bits) = self.bits {
659            row.serialize_field("bits", &bits)?;
660        }
661        row.serialize_field("scheme", &self.scheme)?;
662        if let Some(group_size) = self.group_size {
663            row.serialize_field("group_size", &group_size)?;
664        }
665        row.serialize_field("label", &self.label)?;
666        row.end()
667    }
668}
669
670impl<'de> Deserialize<'de> for Quantization {
671    fn deserialize<D: serde::Deserializer<'de>>(d: D) -> Result<Self, D::Error> {
672        /// The object form. `label` is required and unknown fields are
673        /// rejected, so a typo'd or double-nested payload is an error rather
674        /// than a row that silently claims `Unknown`.
675        #[derive(Deserialize)]
676        #[serde(deny_unknown_fields)]
677        struct Structured {
678            #[serde(default)]
679            bits: Option<u8>,
680            #[serde(default)]
681            scheme: Option<QuantScheme>,
682            #[serde(default)]
683            group_size: Option<u32>,
684            label: String,
685        }
686
687        struct QuantizationVisitor;
688
689        impl<'de> serde::de::Visitor<'de> for QuantizationVisitor {
690            type Value = Quantization;
691
692            fn expecting(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
693                f.write_str("a quantization label string, or an object with a `label` field")
694            }
695
696            fn visit_str<E: serde::de::Error>(self, label: &str) -> Result<Self::Value, E> {
697                Ok(Quantization::parse(label))
698            }
699
700            fn visit_map<M: serde::de::MapAccess<'de>>(
701                self,
702                map: M,
703            ) -> Result<Self::Value, M::Error> {
704                // Dispatched by hand rather than via `#[serde(untagged)]`,
705                // which buffers the input and reports only "data did not match
706                // any variant" — for a `models.json` holding dozens of models
707                // that error names neither the field nor the row.
708                let s = Structured::deserialize(serde::de::value::MapAccessDeserializer::new(map))?;
709                // A row that omits `scheme` or `bits` recovers what its label
710                // implies; an explicit value always wins.
711                let inferred = Quantization::parse(&s.label);
712                Ok(Quantization {
713                    bits: s.bits.or(inferred.bits),
714                    scheme: s.scheme.unwrap_or(inferred.scheme),
715                    group_size: s.group_size,
716                    label: s.label,
717                })
718            }
719        }
720
721        d.deserialize_any(QuantizationVisitor)
722    }
723}
724
725/// Declared performance expectations. Overridden by observed data once available.
726#[derive(Debug, Clone, Default, Serialize, Deserialize, schemars::JsonSchema)]
727pub struct PerformanceEnvelope {
728    /// Median latency in milliseconds (declared/estimated).
729    #[serde(default)]
730    pub latency_p50_ms: Option<u64>,
731    /// 99th percentile latency in milliseconds.
732    #[serde(default)]
733    pub latency_p99_ms: Option<u64>,
734    /// Tokens per second throughput.
735    #[serde(default)]
736    pub tokens_per_second: Option<f64>,
737}
738
739/// Cost model for routing optimization.
740/// Generation parameters that a model may or may not support.
741/// Models declare which params they accept. The inference layer
742/// strips unsupported params before sending to the API.
743#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, schemars::JsonSchema)]
744#[serde(rename_all = "snake_case")]
745pub enum GenerateParam {
746    Temperature,
747    TopP,
748    TopK,
749    MaxTokens,
750    StopSequences,
751    FrequencyPenalty,
752    PresencePenalty,
753    Seed,
754    ResponseFormat,
755    /// Extended thinking / internal reasoning before responding.
756    ExtendedThinking,
757}
758
759/// Standard parameter set for most models.
760pub fn standard_params() -> Vec<GenerateParam> {
761    vec![
762        GenerateParam::Temperature,
763        GenerateParam::TopP,
764        GenerateParam::MaxTokens,
765        GenerateParam::StopSequences,
766        GenerateParam::FrequencyPenalty,
767        GenerateParam::PresencePenalty,
768        GenerateParam::Seed,
769    ]
770}
771
772/// Parameter set for reasoning models (no temperature, no top_p).
773pub fn reasoning_params() -> Vec<GenerateParam> {
774    vec![GenerateParam::MaxTokens, GenerateParam::StopSequences]
775}
776
777#[derive(Debug, Clone, Copy, Default, PartialEq, Serialize, Deserialize, schemars::JsonSchema)]
778pub struct TokenPrices {
779    /// USD per 1M uncached input tokens.
780    #[serde(default)]
781    pub input_per_mtok: Option<f64>,
782    /// USD per 1M output tokens.
783    #[serde(default)]
784    pub output_per_mtok: Option<f64>,
785    /// USD per 1M cache-read input tokens.
786    #[serde(default)]
787    pub cache_read_input_per_mtok: Option<f64>,
788    /// USD per 1M cache-write input tokens.
789    #[serde(default)]
790    pub cache_write_input_per_mtok: Option<f64>,
791}
792
793#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize, schemars::JsonSchema)]
794pub struct TokenPricingTier {
795    /// Inclusive prompt-token threshold at which this tier applies.
796    pub min_prompt_tokens: usize,
797    #[serde(flatten)]
798    pub prices: TokenPrices,
799}
800
801#[derive(Debug, Clone, Default, Serialize, Deserialize, schemars::JsonSchema)]
802pub struct CostModel {
803    /// USD per 1M input tokens (remote models).
804    #[serde(default)]
805    pub input_per_mtok: Option<f64>,
806    /// USD per 1M output tokens (remote models).
807    #[serde(default)]
808    pub output_per_mtok: Option<f64>,
809    /// USD per 1M cache-read input tokens. Unlike protocol-wide cache
810    /// multipliers, this is model-specific and comes from the provider's
811    /// published catalog.
812    #[serde(default)]
813    pub cache_read_input_per_mtok: Option<f64>,
814    /// USD per 1M cache-write input tokens, when the provider charges one.
815    #[serde(default)]
816    pub cache_write_input_per_mtok: Option<f64>,
817    /// Prompt-size pricing overrides, sorted by increasing threshold.
818    /// The highest threshold not greater than the prompt size wins.
819    #[serde(default)]
820    pub pricing_tiers: Vec<TokenPricingTier>,
821    /// On-disk size in MB (local models).
822    #[serde(default)]
823    pub size_mb: Option<u64>,
824    /// RAM required during inference in MB.
825    #[serde(default)]
826    pub ram_mb: Option<u64>,
827}
828
829impl CostModel {
830    pub fn prices_for(&self, prompt_tokens: usize) -> TokenPrices {
831        let mut prices = TokenPrices {
832            input_per_mtok: self.input_per_mtok,
833            output_per_mtok: self.output_per_mtok,
834            cache_read_input_per_mtok: self.cache_read_input_per_mtok,
835            cache_write_input_per_mtok: self.cache_write_input_per_mtok,
836        };
837        for tier in self
838            .pricing_tiers
839            .iter()
840            .filter(|tier| tier.min_prompt_tokens <= prompt_tokens)
841        {
842            if tier.prices.input_per_mtok.is_some() {
843                prices.input_per_mtok = tier.prices.input_per_mtok;
844            }
845            if tier.prices.output_per_mtok.is_some() {
846                prices.output_per_mtok = tier.prices.output_per_mtok;
847            }
848            if tier.prices.cache_read_input_per_mtok.is_some() {
849                prices.cache_read_input_per_mtok = tier.prices.cache_read_input_per_mtok;
850            }
851            if tier.prices.cache_write_input_per_mtok.is_some() {
852                prices.cache_write_input_per_mtok = tier.prices.cache_write_input_per_mtok;
853            }
854        }
855        prices
856    }
857
858    /// Estimated request cost from the provider's declared token prices.
859    /// Unknown price components contribute zero; callers that need to
860    /// distinguish unknown pricing should inspect `prices_for` first.
861    ///
862    /// **This is the routing-score input, not a display or billing figure.**
863    /// The zero-fill is load-bearing here — `adaptive_router` normalizes the
864    /// result into a 0..1 cost score, and it separately neutralizes models
865    /// with no pricing at all, so changing the fill would change routing.
866    /// For anything a human reads, use
867    /// [`estimated_usd_bounded`](Self::estimated_usd_bounded), which refuses
868    /// to bill an unrated bucket at zero and says which way it can be wrong.
869    pub fn estimated_usd(
870        &self,
871        prompt_tokens: usize,
872        output_tokens: usize,
873        cache_read_tokens: usize,
874        cache_write_tokens: usize,
875    ) -> f64 {
876        let prices = self.prices_for(prompt_tokens);
877        let uncached_input = prompt_tokens
878            .saturating_sub(cache_read_tokens)
879            .saturating_sub(cache_write_tokens);
880        (uncached_input as f64 * prices.input_per_mtok.unwrap_or(0.0)
881            + output_tokens as f64 * prices.output_per_mtok.unwrap_or(0.0)
882            + cache_read_tokens as f64 * prices.cache_read_input_per_mtok.unwrap_or(0.0)
883            + cache_write_tokens as f64 * prices.cache_write_input_per_mtok.unwrap_or(0.0))
884            / 1_000_000.0
885    }
886
887    /// Effective per-bucket rates under one resolved price sheet, in the
888    /// bucket order `[uncached input, output, cache read, cache write]`, each
889    /// paired with which way a *substituted* rate can be wrong:
890    /// `(rate, may_overstate, may_understate)`.
891    ///
892    /// The two cache buckets substitute the uncached-input rate but are **not
893    /// governed by one rule**, because the providers in this catalog do not
894    /// price them the same way relative to input:
895    ///
896    /// | bucket | observed vs input, curated table |
897    /// |---|---|
898    /// | cache read | `0.10x`–`0.64x` — always a discount |
899    /// | cache write | `0.1875x` (Google) … `1.25x` (Anthropic) — **both sides** |
900    ///
901    /// So substituting input for a missing **cache-read** rate can only be too
902    /// high (`may_overstate`), while for a missing **cache-write** rate it can
903    /// land either side and gets both flags, rendering `~` rather than a `≤`
904    /// the figure cannot honour. Treating the two alike is how `claude-opus-4.8`
905    /// — `input 5.0`, `cache_write 6.25` — would have worn a `≤$5.00` ceiling
906    /// over a true cost of `$6.25`, or `$10.00` at OpenRouter's 1h-TTL rate.
907    ///
908    /// **Output** substitutes nothing and refuses instead: output runs
909    /// `1.5x`–`8x` input across this same table, which is not a ballpark
910    /// estimate in either direction, merely a wrong number wearing a marker.
911    /// The line is whether a substitute is *within range and merely of unknown
912    /// sign* (estimate, flag it) or *out of range entirely* (refuse) — see
913    /// [`estimated_usd_bounded`](Self::estimated_usd_bounded).
914    fn bucket_rates(prices: &TokenPrices) -> [(Option<f64>, bool, bool); 4] {
915        // A cached READ is always discounted relative to uncached input — the
916        // entire point of the cache — so the input rate is a true ceiling.
917        let cache_read = match prices.cache_read_input_per_mtok {
918            Some(rate) => (Some(rate), false, false),
919            None => {
920                let substituted = prices.input_per_mtok.is_some();
921                (prices.input_per_mtok, substituted, false)
922            }
923        };
924        // A cache WRITE may be a surcharge (Anthropic 1.25x, OpenRouter's 1h
925        // TTL 2x) or a discount (Google 0.1875x). Unknown sign, so no bound.
926        let cache_write = match prices.cache_write_input_per_mtok {
927            Some(rate) => (Some(rate), false, false),
928            None => {
929                let substituted = prices.input_per_mtok.is_some();
930                (prices.input_per_mtok, substituted, substituted)
931            }
932        };
933        [
934            (prices.input_per_mtok, false, false),
935            (prices.output_per_mtok, false, false),
936            cache_read,
937            cache_write,
938        ]
939    }
940
941    /// Cost estimate for a figure a person will read, carrying which way it
942    /// can be wrong.
943    ///
944    /// Differs from [`estimated_usd`](Self::estimated_usd) in refusing to
945    /// invent numbers. A token bucket the provider charges for but whose rate
946    /// this catalog does not declare is **never billed at zero**, and a figure
947    /// that might be wrong never presents itself as exact.
948    ///
949    /// `tier_prompt_tokens` selects the prompt-size pricing tier and is the
950    /// parameter callers most often get wrong:
951    ///
952    /// - `Some(n)` — the prompt size of **one request**. Tiers resolve exactly.
953    /// - `None` — the caller cannot say (a lifetime accumulator has summed
954    ///   many requests and lost their boundaries). Base rates are used and the
955    ///   result is flagged in whichever direction the model's own tiers run.
956    ///
957    /// Passing a *summed* token count as `Some` is the bug this signature
958    /// exists to prevent: thirty 10K-token requests sum to 300K, which crosses
959    /// a 272K threshold that no individual request came near, and every token
960    /// ever sent gets priced at the high-context rate — roughly double, stated
961    /// with total confidence.
962    ///
963    /// Two rejected alternatives, for the next person who wants tiers on an
964    /// aggregate. **Pricing lifetime totals at the tier their sum lands in** is
965    /// the bug above. **Pricing everything at the highest declared tier** does
966    /// yield a true ceiling, but a useless one — it doubles the figure for a
967    /// user whose prompts never approached the threshold, which is the same
968    /// confident wrongness in the other direction. Resolving tiers properly
969    /// needs per-request prompt sizes, which means [`crate::ModelProfile`]
970    /// would have to accumulate per-tier token buckets at record time; that is
971    /// a real feature with a persisted-schema change, not something to fake
972    /// here from data that has already been summed away.
973    ///
974    /// Returns `None` when no defensible number exists — either the model
975    /// declares no rate card at all, or a bucket carrying tokens has no rate
976    /// and no usable substitute. Output deliberately has no input-rate
977    /// fallback: it runs 1.5x–8x input across this catalog, far enough out of
978    /// range that no marker could rescue the number. Which buckets substitute,
979    /// and which way each substitution can be wrong, is decided in
980    /// [`bucket_rates`](Self::bucket_rates) from observed provider pricing —
981    /// notably cache *reads* and cache *writes* do not share a direction.
982    /// `None` means unpriced, and a caller must render it as such, not as free.
983    pub fn estimated_usd_bounded(
984        &self,
985        tier_prompt_tokens: Option<usize>,
986        uncached_input_tokens: usize,
987        output_tokens: usize,
988        cache_read_tokens: usize,
989        cache_write_tokens: usize,
990    ) -> Option<ApproxCost> {
991        // `prices_for(0)` is the base sheet: no tier threshold is <= 0 in a
992        // catalog whose thresholds are positive, so nothing overrides.
993        let prices = self.prices_for(tier_prompt_tokens.unwrap_or(0));
994        if prices.input_per_mtok.is_none() && prices.output_per_mtok.is_none() {
995            // No rate card. Not free — unknown.
996            return None;
997        }
998
999        let tokens = [
1000            uncached_input_tokens,
1001            output_tokens,
1002            cache_read_tokens,
1003            cache_write_tokens,
1004        ];
1005        let rates = Self::bucket_rates(&prices);
1006        let mut usd = 0.0;
1007        let mut may_overstate = false;
1008        let mut may_understate = false;
1009        for (count, (rate, substitute_high, substitute_low)) in tokens.into_iter().zip(rates) {
1010            if count == 0 {
1011                continue;
1012            }
1013            // Charged, rate unknown, no usable substitute — admit we can't.
1014            let rate = rate?;
1015            // Direction comes from the bucket, not from one blanket rule: a
1016            // substituted cache-READ rate can only be high, a substituted
1017            // cache-WRITE rate can land either side.
1018            may_overstate |= substitute_high;
1019            may_understate |= substitute_low;
1020            usd += count as f64 * rate;
1021        }
1022
1023        // Unresolvable tiers: say which way the base sheet can be wrong rather
1024        // than assuming tiers always cost more. Compare the rate each bucket
1025        // would actually pay at every declared threshold against the base.
1026        if tier_prompt_tokens.is_none() {
1027            for tier in &self.pricing_tiers {
1028                let at_tier = Self::bucket_rates(&self.prices_for(tier.min_prompt_tokens));
1029                for (count, ((base_rate, _, _), (tier_rate, _, _))) in
1030                    tokens.into_iter().zip(rates.iter().zip(at_tier))
1031                {
1032                    if count == 0 {
1033                        continue;
1034                    }
1035                    if let (Some(base), Some(tiered)) = (base_rate, tier_rate) {
1036                        may_understate |= tiered > *base;
1037                        may_overstate |= tiered < *base;
1038                    }
1039                }
1040            }
1041        }
1042
1043        Some(ApproxCost {
1044            usd: usd / 1_000_000.0,
1045            may_overstate,
1046            may_understate,
1047        })
1048    }
1049}
1050
1051/// A cost figure plus which way it can be wrong.
1052///
1053/// Presenting an estimate as an exact price is the same failure as billing an
1054/// unrated bucket at zero, one step later — so the direction travels with the
1055/// number instead of being re-derived (or forgotten) at each display site.
1056#[derive(Debug, Clone, Copy, PartialEq)]
1057pub struct ApproxCost {
1058    /// USD.
1059    pub usd: f64,
1060    /// The true cost may be **lower** — a bucket was priced at a substitute
1061    /// rate that can only be too high (a cache read at the uncached-input
1062    /// rate), or a pricing tier is cheaper than the base sheet used.
1063    pub may_overstate: bool,
1064    /// The true cost may be **higher** — a pricing tier dearer than the base
1065    /// sheet could not be resolved from the tokens the caller had, or a cache
1066    /// *write* was priced at the uncached-input rate and the provider charges
1067    /// a surcharge for it (Anthropic 1.25x, OpenRouter's 1h TTL 2x).
1068    pub may_understate: bool,
1069}
1070
1071impl ApproxCost {
1072    /// Every rate applied exactly; the figure is the price.
1073    pub fn is_exact(&self) -> bool {
1074        !self.may_overstate && !self.may_understate
1075    }
1076
1077    /// Prefix for the figure: `≤` a ceiling, `≥` a floor, `~` neither bound
1078    /// holds, empty when exact. Rendering the number without this is the
1079    /// defect the type exists to prevent.
1080    pub fn marker(&self) -> &'static str {
1081        match (self.may_overstate, self.may_understate) {
1082            (false, false) => "",
1083            (true, false) => "≤",
1084            (false, true) => "≥",
1085            (true, true) => "~",
1086        }
1087    }
1088}
1089
1090/// A score on a public benchmark from a published source (model card,
1091/// paper, leaderboard). The schema is deliberately permissive — no enum
1092/// of benchmark names — so the catalog can carry whichever benchmarks
1093/// the upstream provider chose to publish, and new ones can be added
1094/// without a code change. Scores are stored on a 0.0–1.0 scale (e.g.
1095/// 73.5% accuracy → 0.735) so they compare cleanly across benchmarks
1096/// and so `routing_ext::apply_benchmark_priors` can consume them
1097/// directly when wired in later.
1098#[derive(Debug, Clone, Serialize, Deserialize, schemars::JsonSchema)]
1099pub struct BenchmarkScore {
1100    /// Benchmark name as published (e.g., "MMLU-Pro", "GPQA-Diamond",
1101    /// "SWE-bench-Verified", "HumanEval", "MATH").
1102    pub name: String,
1103    /// Score on a 0.0–1.0 scale.
1104    pub score: f64,
1105    /// Evaluation harness or setup label (e.g., "5-shot", "0-shot CoT",
1106    /// "agentic", "pass@1"). Optional but strongly recommended — the
1107    /// same benchmark name can mean different things under different
1108    /// harnesses.
1109    #[serde(default)]
1110    pub harness: Option<String>,
1111    /// Where the score came from (model card URL, paper, leaderboard
1112    /// snapshot). Empty when the source is the upstream provider's
1113    /// announcement and a stable URL is not yet known.
1114    #[serde(default)]
1115    pub source_url: Option<String>,
1116    /// ISO 8601 date of the score snapshot (e.g., "2025-08-12"). Lets
1117    /// downstream code judge how stale a number is.
1118    #[serde(default)]
1119    pub measured_at: Option<String>,
1120    /// How many independent runs the score pools, when CAR measured it.
1121    /// `None` for a published score, whose sample is not ours to know.
1122    #[serde(default, skip_serializing_if = "Option::is_none")]
1123    pub runs: Option<u32>,
1124    /// Highest minus lowest score across those runs; present only with two
1125    /// or more. A difference smaller than this is not a measured difference.
1126    #[serde(default, skip_serializing_if = "Option::is_none")]
1127    pub spread: Option<f64>,
1128}
1129
1130/// The full declarative schema for a model.
1131///
1132/// Analogous to `ToolSchema` — describes what a model is, what it can do,
1133/// and how to access it. The router uses this for constraint-based filtering
1134/// and cold-start scoring before observed performance data is available.
1135#[derive(Debug, Clone, Serialize, Deserialize, schemars::JsonSchema)]
1136pub struct ModelSchema {
1137    /// Unique identifier: "provider/model-name:variant" (e.g., "qwen/qwen3-4b:q4_k_m").
1138    pub id: String,
1139    /// Human-readable display name.
1140    pub name: String,
1141    /// Provider (qwen, openai, anthropic, google, meta, ollama, custom).
1142    pub provider: String,
1143    /// Model family for grouping (qwen3, gpt-4, claude-4, llama-3).
1144    pub family: String,
1145    /// Semantic version or checkpoint label.
1146    #[serde(default)]
1147    pub version: String,
1148    /// What this model can do — ordered by primary capability first.
1149    pub capabilities: Vec<ModelCapability>,
1150    /// Context window in tokens.
1151    pub context_length: usize,
1152    /// Per-model maximum OUTPUT tokens the provider will return in one
1153    /// response. None = unknown; callers fall back to
1154    /// effective_max_output() which derives a fraction of context_length.
1155    #[serde(default)]
1156    pub max_output_tokens: Option<usize>,
1157    /// Parameter count as human-readable string (e.g., "4B", "30B (3B active)").
1158    #[serde(default)]
1159    pub param_count: String,
1160    /// How the weights are quantized, if at all. `None` for remote models and
1161    /// for local ones whose source declared nothing. See [`Quantization`] —
1162    /// this accepts the legacy bare-string form on the wire.
1163    #[serde(default)]
1164    #[schemars(with = "Option<QuantizationWireSchema>")]
1165    pub quantization: Option<Quantization>,
1166    /// Declared performance envelope (initial estimate, overridden by observed data).
1167    #[serde(default)]
1168    pub performance: PerformanceEnvelope,
1169    /// Cost structure.
1170    #[serde(default)]
1171    pub cost: CostModel,
1172    /// How to access this model.
1173    pub source: ModelSource,
1174    /// Free-form tags for filtering (e.g., "fast", "multilingual", "moe").
1175    #[serde(default)]
1176    pub tags: Vec<String>,
1177    /// Supported generation parameters. The inference layer strips any parameter
1178    /// not in this set before sending to the API. Empty = all supported.
1179    #[serde(default)]
1180    pub supported_params: Vec<GenerateParam>,
1181    /// Public benchmark scores as published by the model provider or
1182    /// reproduced on a public leaderboard (MMLU-Pro, GPQA-Diamond,
1183    /// SWE-bench, HumanEval, etc.). The built-in catalog ships this
1184    /// empty — population is a curation step, not a code change. See
1185    /// `BenchmarkScore` for the field shape and the 0.0–1.0 scoring
1186    /// convention.
1187    #[serde(default)]
1188    pub public_benchmarks: Vec<BenchmarkScore>,
1189    /// How much the project vouches for this model. The built-in catalog is
1190    /// `Curated`. Deserialization retains the legacy `Curated` default, so
1191    /// every user-controlled ingestion boundary must call
1192    /// [`Self::mark_user_registered`] before persistence or registration.
1193    /// Gates auto-apply (task #8) and this is surfaced in recommendation
1194    /// rationale.
1195    #[serde(default)]
1196    pub trust_tier: TrustTier,
1197    /// Superseded models stay listed if installed but are excluded from
1198    /// fresh recommendations. `#[serde(default)]` → not deprecated.
1199    #[serde(default)]
1200    pub deprecated: bool,
1201    /// Whether this model is currently available (downloaded / reachable).
1202    /// Not serialized — computed at runtime.
1203    #[serde(skip)]
1204    pub available: bool,
1205    /// Whether this model can be used **right now, without a download**.
1206    ///
1207    /// Deliberately narrower than [`Self::available`], which for a local MLX
1208    /// model is true as soon as an `hf_repo` is declared — `ensure_local()`
1209    /// lazy-downloads on first use, so a declared repo is "functionally
1210    /// available" (see #164). That is the right default for open-ended work and
1211    /// wrong for work on a deadline: a step with a bounded budget that picks a
1212    /// model it must first fetch spends the whole budget downloading and fails.
1213    /// That is exactly how `car code`'s 120s contract derivation became
1214    /// unusable on a machine with no local weights (Parslee-ai/car#638).
1215    ///
1216    /// Callers express the requirement with [`crate::IntentHint::require_ready`];
1217    /// this is the per-candidate fact that hint filters on. Recomputed on every
1218    /// registration, so a cached schema can't carry a stale value.
1219    #[serde(skip)]
1220    pub weights_ready: bool,
1221}
1222
1223impl ModelSchema {
1224    /// Mark a schema as user-controlled rather than project-vetted.
1225    ///
1226    /// This is intentionally separate from serde's legacy default: old built-in
1227    /// and test fixtures omit `trust_tier` and must continue to deserialize,
1228    /// while `models.json`, CLI imports, and daemon `models.register` must never
1229    /// inherit `Curated` merely because a caller omitted the field or supplied
1230    /// a forged value.
1231    pub fn mark_user_registered(&mut self) {
1232        self.trust_tier = TrustTier::Community;
1233    }
1234
1235    /// Check if this model has a given capability.
1236    pub fn has_capability(&self, cap: ModelCapability) -> bool {
1237        self.capabilities.contains(&cap)
1238    }
1239
1240    /// Whether usage is covered by the person's ChatGPT subscription.
1241    /// Callers must render `subscription` and never a dollar figure.
1242    pub fn is_subscription_billed(&self) -> bool {
1243        matches!(
1244            &self.source,
1245            ModelSource::Proprietary {
1246                auth: ProprietaryAuth::ChatGptSubscription {},
1247                ..
1248            }
1249        )
1250    }
1251
1252    /// Live availability for credential-backed providers. The catalog field is
1253    /// a startup snapshot; Settings/OAuth changes must affect the next list and
1254    /// route without a daemon restart.
1255    /// The credential this row is waiting on, when a missing credential is the
1256    /// only thing standing between the user and using it.
1257    ///
1258    /// Deliberately narrow. It answers `Some` only for a plain remote API row
1259    /// that declares one credential variable and is currently unavailable —
1260    /// the case where "add this key" is complete and correct advice. Every
1261    /// other shape answers `None`, because for them the advice would be wrong:
1262    ///
1263    /// - available rows need nothing;
1264    /// - local rows have no credential;
1265    /// - `OpenRouter` rows resolve through a credential *source* (pasted key
1266    ///   or OAuth) rather than a named variable a person can type;
1267    /// - `Proprietary` OAuth rows (`parslee/*`) are a sign-in, not a key, and
1268    ///   their availability additionally depends on the gateway having an
1269    ///   upstream — credential presence alone has already been shipped as a
1270    ///   false "available" for those (car#786);
1271    /// - `CodexCli` owns its own auth;
1272    /// - a deprecated row will not become usable by adding anything.
1273    ///
1274    /// A row declaring several alternative variables answers `None` as well:
1275    /// telling a person to set one of four names is not actionable, and
1276    /// picking one for them would be a guess about which account they have.
1277    pub fn credential_required(&self) -> Option<String> {
1278        if self.available_now() || self.deprecated || self.is_local() {
1279            return None;
1280        }
1281        match &self.source {
1282            ModelSource::RemoteApi {
1283                protocol: ApiProtocol::OpenRouter,
1284                ..
1285            } => None,
1286            ModelSource::RemoteApi {
1287                api_key_env,
1288                api_key_envs,
1289                ..
1290            } if api_key_envs.is_empty() && !api_key_env.trim().is_empty() => {
1291                Some(api_key_env.clone())
1292            }
1293            _ => None,
1294        }
1295    }
1296
1297    pub fn available_now(&self) -> bool {
1298        match &self.source {
1299            ModelSource::RemoteApi {
1300                protocol: ApiProtocol::OpenRouter,
1301                ..
1302            } => self.available && crate::openrouter::credential_source().is_some(),
1303            ModelSource::CodexCli { .. } => crate::backend::codex_cli::is_available(),
1304            // Live (cached a few seconds): Apple Intelligence can be switched
1305            // on or off, or finish provisioning, while CAR runs, and the
1306            // catalog snapshot would not see it until the next refresh.
1307            ModelSource::AppleFoundationModels { .. } => {
1308                #[cfg(any(
1309                    all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)),
1310                    all(target_os = "ios", target_arch = "aarch64")
1311                ))]
1312                {
1313                    crate::backend::foundation_models::is_available()
1314                }
1315                #[cfg(not(any(
1316                    all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)),
1317                    all(target_os = "ios", target_arch = "aarch64")
1318                )))]
1319                {
1320                    false
1321                }
1322            }
1323            _ => self.available,
1324        }
1325    }
1326
1327    /// Prompt-cache economics for this model, derived from its remote
1328    /// protocol. Local / non-remote models have no remote prompt cache, so
1329    /// their cache rates are inert ([`CacheRates::NONE`](crate::outcome::CacheRates::NONE)).
1330    pub fn cache_rates(&self) -> crate::outcome::CacheRates {
1331        match &self.source {
1332            ModelSource::RemoteApi {
1333                protocol: ApiProtocol::OpenRouter,
1334                ..
1335            } => {
1336                let input = self.cost.input_per_mtok.unwrap_or(0.0);
1337                if input > 0.0 {
1338                    crate::outcome::CacheRates {
1339                        read_mult: self.cost.cache_read_input_per_mtok.unwrap_or(0.0) / input,
1340                        write_mult: self.cost.cache_write_input_per_mtok.unwrap_or(0.0) / input,
1341                    }
1342                } else {
1343                    crate::outcome::CacheRates::NONE
1344                }
1345            }
1346            ModelSource::RemoteApi { protocol, .. } => protocol.cache_rates(),
1347            _ => crate::outcome::CacheRates::NONE,
1348        }
1349    }
1350
1351    /// The organization whose model this is, when that can be honestly known.
1352    ///
1353    /// Answers one question — could two models be expected to fail the same
1354    /// way? — so it is deliberately conservative. `None` means UNKNOWABLE, not
1355    /// "none", and callers must treat it as "cannot tell" rather than folding
1356    /// it into a count of distinct vendors.
1357    ///
1358    /// NOT [`Self::provider`], which means four different things depending on
1359    /// which path built the schema:
1360    ///
1361    /// * curated remote rows — the real vendor (`openai`, `anthropic`);
1362    /// * OpenRouter and the Parslee gateway — the AGGREGATOR, so three vendors
1363    ///   behind one gateway all report `openrouter`/`parslee` and one vendor
1364    ///   reached two ways reports as two;
1365    /// * a HuggingFace-derived row — the repo ORG that uploaded it, so
1366    ///   `unsloth/Qwen3` and `mlx-community/Qwen3` are the same weights under
1367    ///   two "vendors";
1368    /// * a discovered local-server row — a guess from the model name.
1369    ///
1370    /// Only the first is a vendor, so only the first is reported. NOT `family`
1371    /// either — that is the model line, so `claude-4.6` and `claude-4.8` read as
1372    /// different and are both Anthropic.
1373    pub fn vendor(&self) -> Option<&str> {
1374        // The aggregator case: the curated table knows which upstream a gateway
1375        // alias resolves to, which is the only place that survives an id
1376        // carrying no trace of its vendor.
1377        if self.provider.eq_ignore_ascii_case("openrouter")
1378            || self.provider.eq_ignore_ascii_case("parslee")
1379        {
1380            return crate::openrouter::curated_vendor(&self.id);
1381        }
1382        // A local model is served by the operator's own machine. Whoever
1383        // uploaded the weights is not an organization that could fail
1384        // independently of the process running beside it.
1385        if self.is_local() {
1386            return None;
1387        }
1388        // Only a project-vetted row's `provider` was assigned deliberately. On a
1389        // community row it is the uploader or a guess, and claiming it here is
1390        // how two repacks of one checkpoint would pass as two vendors.
1391        if self.trust_tier != TrustTier::Curated {
1392            return None;
1393        }
1394        (!self.provider.is_empty()).then_some(self.provider.as_str())
1395    }
1396
1397    /// Check if this model is local (runs on-device).
1398    pub fn is_local(&self) -> bool {
1399        matches!(
1400            self.source,
1401            ModelSource::Local { .. }
1402                | ModelSource::Mlx { .. }
1403                | ModelSource::WhisperCpp { .. }
1404                | ModelSource::WindowsSpeech { .. }
1405                | ModelSource::ManagedVllmMlx { .. }
1406                | ModelSource::AppleFoundationModels { .. }
1407        )
1408    }
1409
1410    /// Whether this model has weights CAR fetches to disk before it can be
1411    /// used — i.e. whether "is it installed?" is a question with an answer.
1412    ///
1413    /// Three predicates in this area are easy to conflate, and conflating them
1414    /// is what Parslee-ai/car#894 was about:
1415    ///
1416    /// - [`is_local`](Self::is_local) — *owned on this machine*. True for
1417    ///   `WindowsSpeech` (the OS owns the voices), `AppleFoundationModels`
1418    ///   (the OS owns the weights), and CAR-managed sources. An external
1419    ///   `VllmMlx` endpoint is remote even when its URL happens to be loopback.
1420    /// - [`weights_ready`](Self::weights_ready) — *the weights are on disk
1421    ///   now*. Only meaningful when this predicate is true; for everything
1422    ///   else the registry sets it to `true` as a "nothing blocks an attempt"
1423    ///   sentinel, which reads as "installed" if taken literally.
1424    /// - `downloads_weights` (this one) — *there is something to install at
1425    ///   all*. Use it to decide whether an install/download status should be
1426    ///   reported, then use `weights_ready` for the status itself.
1427    ///
1428    /// Rendering `weights_ready` without this gate is what made
1429    /// `windows/speech-synthesis:os` claim `INSTALLED yes` while `car doctor`
1430    /// said `Models: none installed`, and made `apple/foundation:default` and
1431    /// the `vllm-mlx/*` rows claim `INSTALLED no` for models that install
1432    /// nothing.
1433    ///
1434    /// Written as an exhaustive `match` rather than `matches!` so that adding
1435    /// a `ModelSource` variant is a compile error here instead of a silently
1436    /// wrong answer in the CLI.
1437    pub fn downloads_weights(&self) -> bool {
1438        match self.source {
1439            // CAR fetches these to disk itself: a GGUF file, an MLX
1440            // safetensors repo, a whisper.cpp ggml `.bin`.
1441            ModelSource::Local { .. }
1442            | ModelSource::Mlx { .. }
1443            | ModelSource::WhisperCpp { .. }
1444            | ModelSource::ManagedVllmMlx { .. } => true,
1445            // The OS owns the voices / the weights — nothing to download.
1446            ModelSource::WindowsSpeech {} | ModelSource::AppleFoundationModels { .. } => false,
1447            // Someone else holds the weights: a local server (vLLM-MLX,
1448            // Ollama), a remote API, or a host-registered runner.
1449            ModelSource::VllmMlx { .. }
1450            | ModelSource::Ollama { .. }
1451            | ModelSource::RemoteApi { .. }
1452            | ModelSource::CodexCli { .. }
1453            | ModelSource::Proprietary { .. }
1454            | ModelSource::Delegated { .. } => false,
1455        }
1456    }
1457
1458    /// Whether a CAR-downloadable artifact is physically present now.
1459    /// General runtime availability and OS/server-owned weights are not an
1460    /// installation claim.
1461    pub fn has_installed_weights(&self) -> bool {
1462        self.downloads_weights() && self.weights_ready
1463    }
1464
1465    /// Whether CAR decodes this model **in its own process**, token by token,
1466    /// through the shared decode loop.
1467    ///
1468    /// Narrower than [`is_local`](Self::is_local) on purpose. `is_local` also
1469    /// covers CAR-managed vLLM-MLX and the speech backends. Those do not spend
1470    /// an output-token budget as this process's wall clock, so they must keep
1471    /// the remote treatment. Getting that distinction wrong takes the
1472    /// anti-truncation budget away from vLLM-MLX, which is the documented way
1473    /// to get structured tool calls out of a local model. (car#851)
1474    pub fn decodes_in_process(&self) -> bool {
1475        matches!(
1476            self.source,
1477            ModelSource::Local { .. } | ModelSource::Mlx { .. }
1478        )
1479    }
1480
1481    /// Check if this model delegates inference to a host-registered
1482    /// runner (closes Parslee-ai/car-releases#24).
1483    pub fn is_delegated(&self) -> bool {
1484        matches!(self.source, ModelSource::Delegated { .. })
1485    }
1486
1487    /// Check if this model runs one tool-free turn through `codex exec`.
1488    pub fn is_codex_cli(&self) -> bool {
1489        matches!(self.source, ModelSource::CodexCli { .. })
1490    }
1491
1492    /// Check if this model uses the MLX backend.
1493    pub fn is_mlx(&self) -> bool {
1494        matches!(self.source, ModelSource::Mlx { .. })
1495    }
1496
1497    /// Check if this model routes to Apple's on-device FoundationModels
1498    /// framework. True only for `ModelSource::AppleFoundationModels`;
1499    /// callers must still verify runtime availability before dispatch
1500    /// (the schema can describe the model on any host, but execution
1501    /// requires macOS 26+ on Apple Silicon).
1502    pub fn is_foundation_models(&self) -> bool {
1503        matches!(self.source, ModelSource::AppleFoundationModels { .. })
1504    }
1505
1506    /// Whether the operating system owns the implementation and its memory.
1507    /// These rows have no CAR-downloadable artifact and no local-model memory
1508    /// figure to compare with the admission budget.
1509    pub fn is_os_provided(&self) -> bool {
1510        matches!(
1511            self.source,
1512            ModelSource::WindowsSpeech {} | ModelSource::AppleFoundationModels { .. }
1513        )
1514    }
1515
1516    /// Check if this model uses vLLM-MLX backend.
1517    pub fn is_vllm_mlx(&self) -> bool {
1518        matches!(
1519            self.source,
1520            ModelSource::VllmMlx { .. } | ModelSource::ManagedVllmMlx { .. }
1521        )
1522    }
1523
1524    /// Whether CAR, rather than an independently managed endpoint, owns the
1525    /// vLLM-MLX child process and its physical weight allocation. The existing
1526    /// HTTP `VllmMlx` contract remains external/server-owned. A supervised
1527    /// source must opt in explicitly and provide an already-installed local
1528    /// model path in `model_name`; it is then replaced with a loopback HTTP
1529    /// endpoint only after admission and a successful readiness ACK.
1530    pub fn is_car_managed_vllm_mlx(&self) -> bool {
1531        matches!(self.source, ModelSource::ManagedVllmMlx { .. })
1532    }
1533
1534    /// Whether a CAR/OS-owned model can only run on Apple Silicon (Metal).
1535    /// External vLLM-MLX endpoints own their hardware and are not constrained
1536    /// by the client machine's accelerator.
1537    pub fn requires_apple_silicon(&self) -> bool {
1538        self.is_mlx() || self.is_car_managed_vllm_mlx() || self.is_foundation_models()
1539    }
1540
1541    /// Check if this model is remote (requires API call).
1542    pub fn is_remote(&self) -> bool {
1543        matches!(
1544            self.source,
1545            ModelSource::RemoteApi { .. }
1546                | ModelSource::CodexCli { .. }
1547                | ModelSource::Proprietary { .. }
1548                | ModelSource::VllmMlx { .. }
1549        )
1550    }
1551
1552    /// Collect all API key env var names for this model (primary + extras).
1553    /// Returns empty vec for non-remote models.
1554    pub fn all_api_key_envs(&self) -> Vec<String> {
1555        match &self.source {
1556            ModelSource::RemoteApi {
1557                api_key_env,
1558                api_key_envs,
1559                ..
1560            } => {
1561                let mut all = vec![api_key_env.clone()];
1562                all.extend(api_key_envs.iter().cloned());
1563                all
1564            }
1565            ModelSource::Proprietary {
1566                auth: ProprietaryAuth::ApiKeyEnv { env_var },
1567                ..
1568            }
1569            | ModelSource::Proprietary {
1570                auth: ProprietaryAuth::BearerTokenEnv { env_var },
1571                ..
1572            } => vec![env_var.clone()],
1573            ModelSource::Proprietary {
1574                auth: ProprietaryAuth::OAuth2Pkce { .. } | ProprietaryAuth::ChatGptSubscription {},
1575                ..
1576            } => vec![],
1577            _ => vec![],
1578        }
1579    }
1580
1581    /// Get the size in MB (from cost model or 0 if unknown).
1582    pub fn size_mb(&self) -> u64 {
1583        self.cost.size_mb.unwrap_or(0)
1584    }
1585
1586    /// Get the RAM requirement in MB (from cost model, falls back to size_mb).
1587    pub fn ram_mb(&self) -> u64 {
1588        self.cost.ram_mb.unwrap_or_else(|| self.size_mb())
1589    }
1590
1591    /// Estimated cost per 1K output tokens in USD. Returns 0.0 for local models.
1592    pub fn cost_per_1k_output(&self) -> f64 {
1593        self.cost.output_per_mtok.map(|c| c / 1000.0).unwrap_or(0.0)
1594    }
1595
1596    /// The per-turn output-token ceiling to use when the caller didn't
1597    /// specify one. Prefers the registry-declared `max_output_tokens`;
1598    /// otherwise derives a quarter of the context window, clamped to a
1599    /// sane [4096, 32768] band so a 1M-context model doesn't request a
1600    /// 250K-token response the API rejects and a tiny 8K model doesn't
1601    /// get an absurdly small ceiling. (Registry value first, computed
1602    /// fallback second — mirrors a provider lookup with a derived default.)
1603    pub fn effective_max_output(&self) -> usize {
1604        self.max_output_tokens
1605            .unwrap_or_else(|| (self.context_length / 4).clamp(4096, 32_768))
1606    }
1607}
1608
1609#[cfg(test)]
1610mod tests {
1611    use super::*;
1612
1613    fn sample_local() -> ModelSchema {
1614        ModelSchema {
1615            id: "qwen/qwen3-4b:q4_k_m".into(),
1616            name: "Qwen3-4B".into(),
1617            provider: "qwen".into(),
1618            family: "qwen3".into(),
1619            version: "1.0".into(),
1620            capabilities: vec![ModelCapability::Generate, ModelCapability::Code],
1621            context_length: 32768,
1622            max_output_tokens: None,
1623            param_count: "4B".into(),
1624            quantization: Some(Quantization::parse("Q4_K_M")),
1625            performance: PerformanceEnvelope {
1626                tokens_per_second: Some(45.0),
1627                ..Default::default()
1628            },
1629            cost: CostModel {
1630                size_mb: Some(2500),
1631                ram_mb: Some(2500),
1632                ..Default::default()
1633            },
1634            source: ModelSource::Local {
1635                hf_repo: "Qwen/Qwen3-4B-GGUF".into(),
1636                hf_filename: "Qwen3-4B-Q4_K_M.gguf".into(),
1637                tokenizer_repo: "Qwen/Qwen3-4B".into(),
1638            },
1639            tags: vec!["code".into(), "fast".into()],
1640            supported_params: vec![],
1641            public_benchmarks: vec![],
1642            trust_tier: TrustTier::Curated,
1643            deprecated: false,
1644            available: false,
1645            weights_ready: false,
1646        }
1647    }
1648
1649    fn sample_remote() -> ModelSchema {
1650        ModelSchema {
1651            id: "anthropic/claude-sonnet-4-6:latest".into(),
1652            name: "Claude Sonnet 4.6".into(),
1653            provider: "anthropic".into(),
1654            family: "claude-4".into(),
1655            version: "latest".into(),
1656            capabilities: vec![
1657                ModelCapability::Generate,
1658                ModelCapability::Code,
1659                ModelCapability::Reasoning,
1660                ModelCapability::ToolUse,
1661                ModelCapability::Vision,
1662            ],
1663            context_length: 200000,
1664            max_output_tokens: None,
1665            param_count: String::new(),
1666            quantization: None,
1667            performance: PerformanceEnvelope {
1668                latency_p50_ms: Some(2000),
1669                latency_p99_ms: Some(8000),
1670                tokens_per_second: Some(80.0),
1671            },
1672            cost: CostModel {
1673                input_per_mtok: Some(3.0),
1674                output_per_mtok: Some(15.0),
1675                ..Default::default()
1676            },
1677            source: ModelSource::RemoteApi {
1678                endpoint: "https://api.anthropic.com/v1/messages".into(),
1679                api_key_env: "ANTHROPIC_API_KEY".into(),
1680                api_key_envs: vec![],
1681                api_version: Some("2023-06-01".into()),
1682                protocol: ApiProtocol::Anthropic,
1683            },
1684            tags: vec!["reasoning".into(), "tool_use".into()],
1685            supported_params: vec![],
1686            public_benchmarks: vec![],
1687            trust_tier: TrustTier::Curated,
1688            deprecated: false,
1689            available: false,
1690            weights_ready: false,
1691        }
1692    }
1693
1694    #[test]
1695    fn capabilities() {
1696        let m = sample_local();
1697        assert!(m.has_capability(ModelCapability::Code));
1698        assert!(!m.has_capability(ModelCapability::Vision));
1699    }
1700
1701    #[test]
1702    fn local_vs_remote() {
1703        assert!(sample_local().is_local());
1704        assert!(!sample_local().is_remote());
1705        assert!(sample_remote().is_remote());
1706        assert!(!sample_remote().is_local());
1707        let codex = ModelSchema {
1708            source: ModelSource::CodexCli {
1709                model: "gpt-5.6-sol:high".into(),
1710            },
1711            ..sample_local()
1712        };
1713        assert!(codex.is_remote());
1714        assert!(!codex.is_local());
1715    }
1716
1717    #[test]
1718    fn vllm_ownership_drives_local_remote_and_apple_predicates() {
1719        let external = ModelSchema {
1720            source: ModelSource::VllmMlx {
1721                endpoint: "https://gpu-owner.example/v1".into(),
1722                model_name: "owner/runtime-model".into(),
1723            },
1724            ..sample_local()
1725        };
1726        assert!(!external.is_local());
1727        assert!(external.is_remote());
1728        assert!(!external.requires_apple_silicon());
1729
1730        let managed = ModelSchema {
1731            source: ModelSource::ManagedVllmMlx {
1732                hf_repo: "mlx-community/Qwen3-4B-4bit".into(),
1733                hf_weight_file: None,
1734            },
1735            ..sample_local()
1736        };
1737        assert!(managed.is_local());
1738        assert!(!managed.is_remote());
1739        assert!(managed.requires_apple_silicon());
1740    }
1741
1742    #[test]
1743    fn cost() {
1744        let local = sample_local();
1745        assert_eq!(local.cost_per_1k_output(), 0.0);
1746
1747        let remote = sample_remote();
1748        assert!(remote.cost_per_1k_output() > 0.0);
1749    }
1750
1751    #[test]
1752    fn serde_roundtrip() {
1753        let local = sample_local();
1754        let json = serde_json::to_string(&local).unwrap();
1755        let parsed: ModelSchema = serde_json::from_str(&json).unwrap();
1756        assert_eq!(parsed.id, local.id);
1757        assert_eq!(parsed.capabilities, local.capabilities);
1758
1759        let remote = sample_remote();
1760        let json = serde_json::to_string(&remote).unwrap();
1761        let parsed: ModelSchema = serde_json::from_str(&json).unwrap();
1762        assert_eq!(parsed.id, remote.id);
1763        // available is skip-serialized, defaults to false
1764        assert!(!parsed.available);
1765    }
1766
1767    #[test]
1768    fn managed_vllm_source_is_versioned_without_reinterpreting_legacy_vllm_json() {
1769        let legacy: ModelSource = serde_json::from_str(
1770            r#"{"type":"vllm_mlx","endpoint":"http://localhost:8000","model_name":"legacy"}"#,
1771        )
1772        .unwrap();
1773        assert!(matches!(
1774            legacy,
1775            ModelSource::VllmMlx {
1776                endpoint,
1777                model_name
1778            } if endpoint == "http://localhost:8000" && model_name == "legacy"
1779        ));
1780
1781        let managed = ModelSource::ManagedVllmMlx {
1782            hf_repo: "mlx-community/Qwen3-4B-4bit".into(),
1783            hf_weight_file: Some("model.safetensors".into()),
1784        };
1785        let encoded = serde_json::to_string(&managed).unwrap();
1786        assert!(encoded.contains(r#""type":"managed_vllm_mlx""#));
1787        assert!(matches!(
1788            serde_json::from_str::<ModelSource>(&encoded).unwrap(),
1789            ModelSource::ManagedVllmMlx { .. }
1790        ));
1791    }
1792
1793    #[test]
1794    fn vendor_refuses_to_claim_a_repackager_or_a_guess() {
1795        // A HuggingFace-derived row's `provider` is the repo ORG that uploaded
1796        // it, so `unsloth/Qwen3` and `mlx-community/Qwen3` — one checkpoint,
1797        // two repacks — would otherwise read as two independent vendors. That
1798        // is a false independence claim produced silently, which is the exact
1799        // failure a vendor check exists to prevent.
1800        let mut m = sample_remote();
1801        m.provider = "mlx-community".into();
1802        m.trust_tier = TrustTier::Community;
1803        assert_eq!(m.vendor(), None);
1804
1805        // A project-vetted remote row IS authoritative.
1806        m.provider = "openai".into();
1807        m.trust_tier = TrustTier::Curated;
1808        assert_eq!(m.vendor(), Some("openai"));
1809
1810        // A local model is served by the operator's own machine; whoever
1811        // uploaded the weights is not an independently-failing organization.
1812        assert_eq!(sample_local().vendor(), None);
1813    }
1814
1815    #[test]
1816    fn trust_tier_and_deprecated_default_when_absent() {
1817        // Pre-existing ~/.car/models.json configs omit the new fields.
1818        // They must deserialize to Curated / not-deprecated, not error.
1819        let json = serde_json::to_string(&sample_local()).unwrap();
1820        let stripped = json
1821            .replace(",\"trust_tier\":\"curated\"", "")
1822            .replace(",\"deprecated\":false", "");
1823        let parsed: ModelSchema = serde_json::from_str(&stripped).unwrap();
1824        assert_eq!(parsed.trust_tier, TrustTier::Curated);
1825        assert!(!parsed.deprecated);
1826    }
1827
1828    #[test]
1829    fn trust_tier_serializes_snake_case() {
1830        assert_eq!(
1831            serde_json::to_string(&TrustTier::Community).unwrap(),
1832            "\"community\""
1833        );
1834        assert_eq!(TrustTier::default(), TrustTier::Curated);
1835    }
1836
1837    #[test]
1838    fn requires_apple_silicon_only_for_metal_backends() {
1839        // GGUF/Candle local and remote models run anywhere CAR builds for.
1840        assert!(!sample_local().requires_apple_silicon());
1841        assert!(!sample_remote().requires_apple_silicon());
1842
1843        let mlx = ModelSchema {
1844            source: ModelSource::Mlx {
1845                hf_repo: "mlx-community/Qwen3-4B-4bit".into(),
1846                hf_weight_file: None,
1847            },
1848            ..sample_local()
1849        };
1850        assert!(mlx.requires_apple_silicon());
1851
1852        // Only CAR-managed vLLM-MLX and Apple FoundationModels are Metal-bound.
1853        // A raw endpoint is external and its owner's hardware is opaque to CAR.
1854        let managed_vllm = ModelSchema {
1855            source: ModelSource::ManagedVllmMlx {
1856                hf_repo: "mlx-community/Qwen3-4B-4bit".into(),
1857                hf_weight_file: None,
1858            },
1859            ..sample_local()
1860        };
1861        assert!(managed_vllm.requires_apple_silicon());
1862
1863        let external_vllm = ModelSchema {
1864            source: ModelSource::VllmMlx {
1865                endpoint: "https://gpu-owner.example/v1".into(),
1866                model_name: "mlx-community/Qwen3-4B-4bit".into(),
1867            },
1868            ..sample_local()
1869        };
1870        assert!(!external_vllm.requires_apple_silicon());
1871
1872        let foundation = ModelSchema {
1873            source: ModelSource::AppleFoundationModels { use_case: None },
1874            ..sample_local()
1875        };
1876        assert!(foundation.requires_apple_silicon());
1877    }
1878
1879    fn priced(
1880        input: Option<f64>,
1881        output: Option<f64>,
1882        cache_read: Option<f64>,
1883        cache_write: Option<f64>,
1884    ) -> CostModel {
1885        CostModel {
1886            input_per_mtok: input,
1887            output_per_mtok: output,
1888            cache_read_input_per_mtok: cache_read,
1889            cache_write_input_per_mtok: cache_write,
1890            ..Default::default()
1891        }
1892    }
1893
1894    #[test]
1895    fn an_undeclared_cache_rate_is_bounded_by_the_input_rate_not_billed_at_zero() {
1896        // The real shape this exists for: qwen3.5-plus declares input and
1897        // output but no cache-read rate. 1M cache-read tokens must not be free.
1898        let cost = priced(Some(0.26), Some(1.56), None, None);
1899        let bounded = cost
1900            .estimated_usd_bounded(Some(1_000_000), 0, 0, 1_000_000, 0)
1901            .expect("a model with input+output rates is priceable");
1902        assert!(bounded.may_overstate, "substituted rate can only be high");
1903        assert!(!bounded.may_understate);
1904        assert_eq!(bounded.marker(), "≤");
1905        assert!(
1906            (bounded.usd - 0.26).abs() < 1e-9,
1907            "cache reads fall back to the $0.26/MTok input rate, got {}",
1908            bounded.usd
1909        );
1910
1911        // The routing-score path still zero-fills, deliberately and untouched.
1912        assert_eq!(cost.estimated_usd(1_000_000, 0, 1_000_000, 0), 0.0);
1913    }
1914
1915    #[test]
1916    fn an_undeclared_cache_write_rate_claims_no_bound_it_cannot_keep() {
1917        // The shipped `claude-opus-4.8` shape with its cache-write rate
1918        // omitted. Anthropic's cache write is a 1.25x SURCHARGE (6.25 against
1919        // 5.0 input), and OpenRouter's 1h-TTL rate is 2x — so pricing the
1920        // bucket at the input rate is a FLOOR, and the `≤` a blanket
1921        // "cache is always cheaper" rule would have produced is a false
1922        // ceiling over a true cost of $6.25, or $10.00 at the 1h rate.
1923        let cost = priced(Some(5.0), Some(25.0), Some(0.5), None);
1924        let bounded = cost
1925            .estimated_usd_bounded(Some(1_000_000), 0, 0, 0, 1_000_000)
1926            .expect("input and output rates are published");
1927        assert!((bounded.usd - 5.0).abs() < 1e-9, "got {}", bounded.usd);
1928        assert!(
1929            bounded.may_understate,
1930            "a cache-write surcharge can exceed the input rate"
1931        );
1932        assert_ne!(bounded.marker(), "≤", "must not claim a ceiling it fails");
1933        assert_eq!(bounded.marker(), "~", "sign is unknown, so neither bound");
1934        // Both real cache-write rates this shape could carry sit ABOVE the
1935        // substituted figure — which is exactly why `≤` would have been a
1936        // false ceiling. Assert the tighter of the two; it implies the looser,
1937        // and spelling out both comparisons is a `redundant_comparisons` lint.
1938        let at_anthropics_1_25x: f64 = 1_000_000.0 * 6.25 / 1e6;
1939        let at_openrouters_1h_2x: f64 = 1_000_000.0 * 10.0 / 1e6;
1940        assert!(bounded.usd < at_anthropics_1_25x.min(at_openrouters_1h_2x));
1941
1942        // Declaring the rate makes it exact — the flag tracks substitution,
1943        // not the mere presence of cache-write tokens.
1944        let declared = priced(Some(5.0), Some(25.0), Some(0.5), Some(6.25))
1945            .estimated_usd_bounded(Some(1_000_000), 0, 0, 0, 1_000_000)
1946            .unwrap();
1947        assert!(declared.is_exact());
1948        assert!((declared.usd - 6.25).abs() < 1e-9);
1949
1950        // The same omission on the cache READ side DOES keep its ceiling —
1951        // the two directions are not shared. Note the rate must be missing for
1952        // a substitution to happen at all.
1953        let read = priced(Some(5.0), Some(25.0), None, Some(6.25))
1954            .estimated_usd_bounded(Some(1_000_000), 0, 0, 1_000_000, 0)
1955            .unwrap();
1956        assert_eq!(read.marker(), "≤");
1957        assert!((read.usd - 5.0).abs() < 1e-9);
1958        // A real cache-read rate is a discount, so the ceiling holds.
1959        assert!(read.usd > 1_000_000.0 * 0.5 / 1e6);
1960    }
1961
1962    #[test]
1963    fn a_declared_cache_rate_is_exact_and_never_flagged() {
1964        let cost = priced(Some(5.0), Some(25.0), Some(0.5), Some(6.25));
1965        let bounded = cost
1966            .estimated_usd_bounded(Some(1_600_000), 1_000_000, 200_000, 500_000, 100_000)
1967            .expect("fully rated");
1968        assert!(bounded.is_exact(), "nothing was substituted or unresolved");
1969        assert_eq!(bounded.marker(), "");
1970        // 1M uncached x 5 + 200k x 25 + 500k x 0.5 + 100k x 6.25, per MTok.
1971        assert!((bounded.usd - 10.875).abs() < 1e-9, "got {}", bounded.usd);
1972        // Identical to the router's figure when every rate is published.
1973        assert!(
1974            (bounded.usd - cost.estimated_usd(1_600_000, 200_000, 500_000, 100_000)).abs() < 1e-9
1975        );
1976    }
1977
1978    #[test]
1979    fn an_unbounded_bucket_refuses_rather_than_understating() {
1980        // Output has no safe substitute — it is normally the dearer side, so
1981        // pricing it at the input rate would UNDER-state. Refuse instead.
1982        let cost = priced(Some(1.0), None, None, None);
1983        assert_eq!(
1984            cost.estimated_usd_bounded(Some(1_000), 1_000, 1_000, 0, 0),
1985            None
1986        );
1987        // With no output tokens the same model is priceable and exact.
1988        let bounded = cost
1989            .estimated_usd_bounded(Some(1_000), 1_000, 0, 0, 0)
1990            .unwrap();
1991        assert!(bounded.is_exact());
1992    }
1993
1994    #[test]
1995    fn no_rate_card_is_unpriced_rather_than_free() {
1996        let cost = CostModel::default();
1997        assert_eq!(
1998            cost.estimated_usd_bounded(Some(1_000), 1_000, 1_000, 0, 0),
1999            None
2000        );
2001        // And a zero-usage priced model is genuinely free, not unpriced.
2002        let free = priced(Some(0.0), Some(0.0), None, None)
2003            .estimated_usd_bounded(Some(0), 0, 0, 0, 0)
2004            .expect("a declared zero rate card is priced");
2005        assert_eq!(free.usd, 0.0);
2006        assert!(free.is_exact());
2007    }
2008
2009    fn tiered() -> CostModel {
2010        CostModel {
2011            input_per_mtok: Some(2.5),
2012            output_per_mtok: Some(15.0),
2013            cache_read_input_per_mtok: Some(0.25),
2014            pricing_tiers: vec![TokenPricingTier {
2015                min_prompt_tokens: 272_000,
2016                prices: TokenPrices {
2017                    input_per_mtok: Some(5.0),
2018                    output_per_mtok: Some(22.5),
2019                    cache_read_input_per_mtok: Some(0.5),
2020                    cache_write_input_per_mtok: None,
2021                },
2022            }],
2023            ..Default::default()
2024        }
2025    }
2026
2027    #[test]
2028    fn a_known_prompt_size_resolves_the_tier_exactly() {
2029        let cost = tiered();
2030        let below = cost
2031            .estimated_usd_bounded(Some(271_999), 271_999, 0, 0, 0)
2032            .unwrap();
2033        assert!(below.is_exact());
2034        assert!((below.usd - 271_999.0 * 2.5 / 1e6).abs() < 1e-9);
2035
2036        let above = cost
2037            .estimated_usd_bounded(Some(272_000), 272_000, 0, 0, 0)
2038            .unwrap();
2039        assert!(above.is_exact());
2040        assert!((above.usd - 272_000.0 * 5.0 / 1e6).abs() < 1e-9);
2041    }
2042
2043    #[test]
2044    fn a_lifetime_aggregate_uses_base_rates_and_admits_it_may_be_low() {
2045        // Thirty 10K-token requests. Their SUM crosses the 272K threshold that
2046        // no single request came near — the double-charging bug. `None` says
2047        // "boundaries lost", so base rates apply and the figure is marked.
2048        let cost = tiered();
2049        let aggregate = cost.estimated_usd_bounded(None, 300_000, 0, 0, 0).unwrap();
2050        assert!(
2051            (aggregate.usd - 300_000.0 * 2.5 / 1e6).abs() < 1e-9,
2052            "must use the $2.50 base rate, got {}",
2053            aggregate.usd
2054        );
2055        assert!(aggregate.may_understate, "a dearer tier may apply");
2056        assert!(!aggregate.may_overstate);
2057        assert_eq!(aggregate.marker(), "≥");
2058
2059        // What the bug looked like: summed tokens passed as a real prompt size
2060        // price at the high-context rate — exactly double, and unmarked.
2061        let bug = cost
2062            .estimated_usd_bounded(Some(300_000), 300_000, 0, 0, 0)
2063            .unwrap();
2064        assert!((bug.usd - 2.0 * aggregate.usd).abs() < 1e-9);
2065        assert!(bug.is_exact(), "and it would have claimed to be exact");
2066    }
2067
2068    #[test]
2069    fn a_cheaper_tier_flags_the_aggregate_as_possibly_high_instead() {
2070        // Direction is derived from the tiers, not assumed. A volume DISCOUNT
2071        // makes the base-rate figure too high, not too low.
2072        let cost = CostModel {
2073            input_per_mtok: Some(2.0),
2074            output_per_mtok: Some(10.0),
2075            pricing_tiers: vec![TokenPricingTier {
2076                min_prompt_tokens: 100_000,
2077                prices: TokenPrices {
2078                    input_per_mtok: Some(1.0),
2079                    ..Default::default()
2080                },
2081            }],
2082            ..Default::default()
2083        };
2084        let aggregate = cost.estimated_usd_bounded(None, 500_000, 0, 0, 0).unwrap();
2085        assert!(aggregate.may_overstate);
2086        assert!(!aggregate.may_understate);
2087        assert_eq!(aggregate.marker(), "≤");
2088    }
2089
2090    #[test]
2091    fn both_directions_at_once_claims_neither_bound() {
2092        // Unrated cache bucket (can be high) plus an unresolved dearer tier
2093        // (can be low). Neither bound survives, so the figure is just an
2094        // estimate and must not wear a `≤` it cannot honour.
2095        let cost = CostModel {
2096            input_per_mtok: Some(2.0),
2097            output_per_mtok: Some(10.0),
2098            pricing_tiers: vec![TokenPricingTier {
2099                min_prompt_tokens: 100_000,
2100                prices: TokenPrices {
2101                    input_per_mtok: Some(4.0),
2102                    ..Default::default()
2103                },
2104            }],
2105            ..Default::default()
2106        };
2107        let aggregate = cost.estimated_usd_bounded(None, 0, 0, 500_000, 0).unwrap();
2108        assert!(aggregate.may_overstate && aggregate.may_understate);
2109        assert_eq!(aggregate.marker(), "~");
2110    }
2111
2112    /// Parslee-ai/car#894 (follow-up): "runs here" and "has weights to fetch"
2113    /// are different questions, and `is_local` answers only the first. Three
2114    /// `is_local` sources download nothing — the CLI must not offer an
2115    /// install status for them.
2116    #[test]
2117    fn downloads_weights_is_true_only_for_sources_car_fetches() {
2118        let mut schema = sample_local();
2119
2120        // CAR downloads these itself.
2121        schema.source = ModelSource::Local {
2122            hf_repo: "Qwen/Qwen3-4B-GGUF".into(),
2123            hf_filename: "Qwen3-4B-Q4_K_M.gguf".into(),
2124            tokenizer_repo: "Qwen/Qwen3-4B".into(),
2125        };
2126        assert!(schema.downloads_weights(), "GGUF weights are downloaded");
2127
2128        schema.source = ModelSource::Mlx {
2129            hf_repo: "mlx-community/Qwen3-4B-4bit".into(),
2130            hf_weight_file: None,
2131        };
2132        assert!(schema.downloads_weights(), "MLX weights are downloaded");
2133
2134        schema.source = ModelSource::WhisperCpp {
2135            model: "large-v3-turbo-q5_0".into(),
2136        };
2137        assert!(
2138            schema.downloads_weights(),
2139            "the whisper.cpp ggml bin is downloaded"
2140        );
2141
2142        // The OS owns these — there is nothing to install. `WindowsSpeech` is
2143        // the row that rendered `INSTALLED yes` against `car doctor`'s
2144        // `Models: none installed`.
2145        schema.source = ModelSource::WindowsSpeech {};
2146        assert!(
2147            !schema.downloads_weights(),
2148            "WinRT speech synthesis has no weights to download"
2149        );
2150
2151        schema.source = ModelSource::AppleFoundationModels { use_case: None };
2152        assert!(
2153            !schema.downloads_weights(),
2154            "Apple FoundationModels weights belong to the OS"
2155        );
2156
2157        // Someone else holds the weights.
2158        schema.source = ModelSource::VllmMlx {
2159            endpoint: "http://localhost:8000".into(),
2160            model_name: "mlx-community/Qwen3-4B-4bit".into(),
2161        };
2162        assert!(
2163            !schema.downloads_weights(),
2164            "the vLLM-MLX server owns its own weights"
2165        );
2166
2167        schema.source = ModelSource::ManagedVllmMlx {
2168            hf_repo: "mlx-community/Qwen3-4B-4bit".into(),
2169            hf_weight_file: None,
2170        };
2171        assert!(
2172            schema.downloads_weights(),
2173            "the explicitly managed vLLM-MLX contract is CAR-owned"
2174        );
2175
2176        schema.source = ModelSource::Ollama {
2177            model_tag: "qwen3:4b".into(),
2178            host: default_ollama_host(),
2179        };
2180        assert!(!schema.downloads_weights(), "Ollama owns its own weights");
2181
2182        schema.source = sample_remote().source;
2183        assert!(
2184            !schema.downloads_weights(),
2185            "a remote API has no weights on this machine"
2186        );
2187
2188        schema.source = ModelSource::Delegated { hint: None };
2189        assert!(
2190            !schema.downloads_weights(),
2191            "a host-registered runner owns its own weights"
2192        );
2193    }
2194
2195    /// `is_local` is the predicate the CLI used to reach for. These OS-owned
2196    /// sources are where the two answers diverge, which is why a
2197    /// separate predicate exists rather than a reuse of `is_local`.
2198    #[test]
2199    fn downloads_weights_differs_from_is_local_on_os_owned_sources() {
2200        let mut schema = sample_local();
2201        for source in [
2202            ModelSource::WindowsSpeech {},
2203            ModelSource::AppleFoundationModels { use_case: None },
2204        ] {
2205            schema.source = source;
2206            assert!(
2207                schema.is_local(),
2208                "this source runs on-device: {:?}",
2209                schema.source
2210            );
2211            assert!(
2212                !schema.downloads_weights(),
2213                "…but CAR downloads nothing for it: {:?}",
2214                schema.source
2215            );
2216        }
2217    }
2218}
2219
2220#[cfg(test)]
2221mod quantization_tests {
2222    use super::*;
2223
2224    /// Every distinct label the built-in catalog ships, the hyphenated form
2225    /// `registry.rs` writes into auto-discovered entries, and the formats that
2226    /// review found misclassified.
2227    #[test]
2228    fn classifies_known_labels() {
2229        let cases: &[(&str, Option<u8>, QuantScheme)] = &[
2230            ("4bit", Some(4), QuantScheme::AffineGroupInt),
2231            ("6bit", Some(6), QuantScheme::AffineGroupInt),
2232            ("3bit", Some(3), QuantScheme::AffineGroupInt),
2233            ("5bit", Some(5), QuantScheme::AffineGroupInt),
2234            // registry.rs emits the hyphen; users already have it on disk.
2235            ("4-bit", Some(4), QuantScheme::AffineGroupInt),
2236            ("mxfp8", Some(8), QuantScheme::BlockScaledFloat),
2237            ("mxfp4", Some(4), QuantScheme::BlockScaledFloat),
2238            ("Q4_K_M", Some(4), QuantScheme::KQuantMixed),
2239            ("Q5_K_S", Some(5), QuantScheme::KQuantMixed),
2240            ("IQ4_XS", Some(4), QuantScheme::KQuantMixed),
2241            ("TQ1_0", Some(1), QuantScheme::KQuantMixed),
2242            ("Q8_0", Some(8), QuantScheme::RtnBlock),
2243            ("q5_0", Some(5), QuantScheme::RtnBlock),
2244            // aarch64 repack quants: the suffix is a prefix match, not equality.
2245            ("Q4_0_4_4", Some(4), QuantScheme::RtnBlock),
2246            ("Q4_0_8_8", Some(4), QuantScheme::RtnBlock),
2247            // A zero-padded width must slice by digits consumed, not by the
2248            // decimal length of the parsed number.
2249            ("q08_0", Some(8), QuantScheme::RtnBlock),
2250            ("bf16", Some(16), QuantScheme::Unquantized),
2251            ("F16", Some(16), QuantScheme::Unquantized),
2252            ("f32", Some(32), QuantScheme::Unquantized),
2253            // Spelled like a quant, but affine group quant has no such width.
2254            ("16bit", Some(16), QuantScheme::Unquantized),
2255            ("32bit", Some(32), QuantScheme::Unquantized),
2256            // An 8-bit float IS quantized — just not attributable from this
2257            // label alone. Calling it full precision was the bug.
2258            ("fp8", Some(8), QuantScheme::Unknown),
2259            // Names a width and no producer.
2260            ("Q4", Some(4), QuantScheme::Unknown),
2261        ];
2262        for (label, bits, scheme) in cases {
2263            let q = Quantization::parse(label);
2264            assert_eq!(q.bits, *bits, "bits for {label}");
2265            assert_eq!(q.scheme, *scheme, "scheme for {label}");
2266            assert_eq!(q.label, *label, "label must survive verbatim");
2267        }
2268    }
2269
2270    /// Labels that identify nothing must say so rather than guess.
2271    #[test]
2272    fn refuses_to_guess() {
2273        for label in [
2274            "",                 // nobody said; not a claim of full precision
2275            "mxfp",             // MX family, no width
2276            "Q256_K",           // width overflows u8
2277            "q0_0",             // a zero-bit quantization is not a thing
2278            "awq-marlin-w4a16", // real format, unknown to this parser
2279        ] {
2280            let q = Quantization::parse(label);
2281            assert_eq!(q.scheme, QuantScheme::Unknown, "scheme for {label:?}");
2282            assert_eq!(q.bits, None, "bits for {label:?}");
2283            assert_eq!(q.label, label, "label for {label:?}");
2284        }
2285    }
2286
2287    /// The distinction the free-text field could not express: same width,
2288    /// different algorithm, different loader.
2289    #[test]
2290    fn same_width_different_scheme() {
2291        let affine = Quantization::parse("4bit");
2292        let kquant = Quantization::parse("Q4_K_M");
2293        let mx = Quantization::parse("mxfp4");
2294        assert_eq!(affine.bits, kquant.bits);
2295        assert_eq!(affine.bits, mx.bits);
2296        assert_ne!(affine.scheme, kquant.scheme);
2297        assert_ne!(affine.scheme, mx.scheme);
2298        // And within GGUF, k-quant is not round-to-nearest.
2299        assert_ne!(
2300            Quantization::parse("Q5_K_S").scheme,
2301            Quantization::parse("q5_0").scheme
2302        );
2303    }
2304
2305    /// The scheme names a numeric format, never a container or an engine.
2306    /// whisper.cpp ships `q5_0` ggml checkpoints that no GGUF text path can
2307    /// load; a container-named variant would assert otherwise.
2308    #[test]
2309    fn scheme_does_not_imply_an_engine() {
2310        let whisper = Quantization::parse("q5_0");
2311        let llama = Quantization::parse("Q5_0");
2312        assert_eq!(whisper.scheme, llama.scheme);
2313        assert_eq!(whisper.scheme, QuantScheme::RtnBlock);
2314    }
2315
2316    #[test]
2317    fn mlx_config_block_survives_ingest() {
2318        let q = Quantization::from_mlx_config(Some(8), Some(32), Some("mxfp8")).unwrap();
2319        assert_eq!(q.bits, Some(8));
2320        assert_eq!(q.group_size, Some(32));
2321        assert_eq!(q.scheme, QuantScheme::BlockScaledFloat);
2322
2323        // Absent `mode` means affine — MLX's default, not unknown.
2324        let affine = Quantization::from_mlx_config(Some(4), Some(64), None).unwrap();
2325        assert_eq!(affine.scheme, QuantScheme::AffineGroupInt);
2326        assert_eq!(affine.group_size, Some(64));
2327        assert_eq!(affine.label, "4bit");
2328    }
2329
2330    /// An empty or absent block must not become a claim of affine group quant.
2331    /// The field it replaced returned nothing here, and asserting a scheme is
2332    /// exactly the error a router acting on `scheme` would inherit.
2333    #[test]
2334    fn empty_mlx_block_asserts_nothing() {
2335        assert!(Quantization::from_mlx_config(None, None, None).is_none());
2336    }
2337
2338    #[test]
2339    fn deserializes_legacy_bare_string() {
2340        let q: Quantization = serde_json::from_str(r#""Q4_K_M""#).unwrap();
2341        assert_eq!(q.scheme, QuantScheme::KQuantMixed);
2342        assert_eq!(q.bits, Some(4));
2343        assert_eq!(q.label, "Q4_K_M");
2344    }
2345
2346    #[test]
2347    fn deserializes_structured_object() {
2348        let q: Quantization = serde_json::from_str(
2349            r#"{"bits":4,"scheme":"affine_group_int","group_size":64,"label":"4bit"}"#,
2350        )
2351        .unwrap();
2352        assert_eq!(q.group_size, Some(64));
2353        assert_eq!(q.scheme, QuantScheme::AffineGroupInt);
2354    }
2355
2356    /// A partial object recovers from its label rather than defaulting to
2357    /// Unknown — the same reason the bare string is still accepted.
2358    #[test]
2359    fn partial_object_recovers_from_label() {
2360        let q: Quantization = serde_json::from_str(r#"{"label":"Q8_0"}"#).unwrap();
2361        assert_eq!(q.scheme, QuantScheme::RtnBlock);
2362        assert_eq!(q.bits, Some(8));
2363    }
2364
2365    /// A malformed object must be an error, not a row that silently claims
2366    /// `Unknown`. Under `#[serde(untagged)]` every one of these deserialized
2367    /// successfully into a fabricated descriptor.
2368    #[test]
2369    fn malformed_objects_are_rejected() {
2370        for bad in [
2371            r#"{"btis":4,"scehme":"k_quant_mixed","labl":"Q4_K_M"}"#, // typos
2372            r#"{"quantization":{"bits":4}}"#,                         // double-nested
2373            r#"{}"#,                                                  // nothing at all
2374            r#"{"group_size":64}"#,                                   // no label
2375            r#"{"bits":4,"scheme":"k_quant_mixed"}"#,                 // no label
2376        ] {
2377            let parsed: Result<Quantization, _> = serde_json::from_str(bad);
2378            assert!(parsed.is_err(), "should have rejected {bad}");
2379        }
2380    }
2381
2382    /// The error must name what was wrong. `#[serde(untagged)]` reported only
2383    /// "data did not match any variant", for a `models.json` that fails whole.
2384    #[test]
2385    fn rejection_names_the_offending_field() {
2386        let err = serde_json::from_str::<Quantization>(
2387            r#"{"bits":4,"scheme":"affine_grp_int","label":"4bit"}"#,
2388        )
2389        .unwrap_err()
2390        .to_string();
2391        assert!(
2392            err.contains("affine_grp_int") || err.contains("scheme"),
2393            "unhelpful error: {err}"
2394        );
2395    }
2396
2397    #[test]
2398    fn round_trips_through_the_object_form() {
2399        for label in ["Q4_K_M", "4bit", "mxfp8", "bf16", "weird-vendor-format"] {
2400            let q = Quantization::parse(label);
2401            let round: Quantization =
2402                serde_json::from_str(&serde_json::to_string(&q).unwrap()).unwrap();
2403            assert_eq!(q, round, "round trip for {label}");
2404        }
2405    }
2406
2407    /// The label alone is the wire form whenever it is lossless. This is what
2408    /// keeps `row_digest` stable for rows that gained no new information.
2409    #[test]
2410    fn serializes_as_a_bare_label_when_lossless() {
2411        for label in [
2412            "Q4_K_M",
2413            "4bit",
2414            "mxfp8",
2415            "bf16",
2416            "q5_0",
2417            "weird-vendor-format",
2418        ] {
2419            let json = serde_json::to_string(&Quantization::parse(label)).unwrap();
2420            assert_eq!(json, format!("\"{label}\""), "should stay a bare string");
2421        }
2422    }
2423
2424    /// ...and the object form appears exactly where the label would lose
2425    /// something: a group size, or a scheme the label cannot express.
2426    #[test]
2427    fn serializes_as_an_object_only_when_it_adds_information() {
2428        let with_group = Quantization::from_mlx_config(Some(4), Some(64), None).unwrap();
2429        let json = serde_json::to_value(&with_group).unwrap();
2430        assert_eq!(json["group_size"], 64, "group_size must survive the write");
2431        assert_eq!(json["label"], "4bit");
2432
2433        // A bare `Q4` parses as Unknown, so an explicit scheme is real
2434        // information and has to be written out.
2435        let disambiguated = Quantization {
2436            bits: Some(4),
2437            scheme: QuantScheme::AffineGroupInt,
2438            group_size: None,
2439            label: "Q4".into(),
2440        };
2441        let json = serde_json::to_value(&disambiguated).unwrap();
2442        assert_eq!(json["scheme"], "affine_group_int");
2443        assert!(json.get("group_size").is_none(), "no null padding");
2444    }
2445
2446    /// Both write forms must read back identically, or the minimal-write rule
2447    /// would trade a digest change for silent data loss.
2448    #[test]
2449    fn every_write_form_round_trips() {
2450        let cases = [
2451            Quantization::parse("Q4_K_M"),
2452            Quantization::parse("bf16"),
2453            Quantization::parse("unattributable-format"),
2454            Quantization::from_mlx_config(Some(8), Some(32), Some("mxfp8")).unwrap(),
2455            Quantization::from_mlx_config(Some(4), Some(64), None).unwrap(),
2456            Quantization {
2457                bits: Some(4),
2458                scheme: QuantScheme::AffineGroupInt,
2459                group_size: None,
2460                label: "Q4".into(),
2461            },
2462        ];
2463        for q in cases {
2464            let round: Quantization =
2465                serde_json::from_str(&serde_json::to_string(&q).unwrap()).unwrap();
2466            assert_eq!(q, round, "round trip for {q:?}");
2467        }
2468    }
2469
2470    /// The digest guard. `catalog_identity::row_digest` is a SHA-256 over the
2471    /// serialized schema and clients pin it through
2472    /// `expected_catalog_revision`, so a row whose quantization gained no new
2473    /// information must re-serialize to the byte-identical JSON it was read
2474    /// from. Only rows that genuinely changed may move.
2475    #[test]
2476    fn catalog_quantizations_reserialize_unchanged() {
2477        let raw: Vec<serde_json::Value> =
2478            serde_json::from_str(include_str!("builtin_catalog.json")).unwrap();
2479        let parsed: Vec<ModelSchema> =
2480            serde_json::from_str(include_str!("builtin_catalog.json")).unwrap();
2481        let mut moved = Vec::new();
2482        for (raw_row, model) in raw.iter().zip(&parsed) {
2483            let before = raw_row
2484                .get("quantization")
2485                .cloned()
2486                .unwrap_or(serde_json::Value::Null);
2487            let after = serde_json::to_value(&model.quantization).unwrap();
2488            if before != after {
2489                moved.push(format!("{}: {before} -> {after}", model.id));
2490            }
2491        }
2492        assert!(
2493            moved.is_empty(),
2494            "these rows would change catalog digest: {moved:#?}"
2495        );
2496    }
2497
2498    /// Every quantization the built-in catalog ships must classify. An entry
2499    /// may carry an explicit `scheme` only where its label is genuinely
2500    /// ambiguous — the guard is that an explicit scheme never *contradicts* a
2501    /// label the parser can already read, which is how catalog data stops
2502    /// being a place to make a failing test pass.
2503    #[test]
2504    fn builtin_catalog_quantizations_are_coherent() {
2505        let catalog: Vec<ModelSchema> =
2506            serde_json::from_str(include_str!("builtin_catalog.json")).unwrap();
2507        let mut problems = Vec::new();
2508        for model in &catalog {
2509            let Some(q) = &model.quantization else {
2510                continue;
2511            };
2512            if q.scheme == QuantScheme::Unknown {
2513                problems.push(format!("{}: unclassified label {:?}", model.id, q.label));
2514                continue;
2515            }
2516            let from_label = Quantization::parse(&q.label).scheme;
2517            if from_label != QuantScheme::Unknown && from_label != q.scheme {
2518                problems.push(format!(
2519                    "{}: label {:?} parses as {:?} but the row claims {:?}",
2520                    model.id, q.label, from_label, q.scheme
2521                ));
2522            }
2523        }
2524        assert!(
2525            problems.is_empty(),
2526            "incoherent catalog rows: {problems:#?}"
2527        );
2528    }
2529}