car-inference 0.56.1

Local model inference for CAR — Candle backend with Qwen3 models
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
//! Local in-process inference backends — the local mirror of the remote
//! [`ProtocolHandler`](crate::protocol::ProtocolHandler) abstraction.
//!
//! Remote models dispatch cleanly through `ProtocolHandler` + `handler_for`.
//! Local (in-process) models historically did not: the engine hand-dispatched
//! a single hardcoded `Qwen3Model` via ad-hoc `cfg` + tag matching. This module
//! introduces the missing seam so each architecture is one isolated backend
//! impl instead of another arm in `generate_tracked_inner`.
//!
//! Two layers, on purpose:
//!
//! - [`TextDecoder`] — the token-level primitives (encode / forward / decode /
//!   eos / context / cache). The engine's shared decode loop (`drive_generation`)
//!   runs sampling, stop detection, TTFT timing, and the Metal panic-catch +
//!   cache-invalidation over `&mut dyn TextDecoder`, so that engine-state-coupled
//!   logic lives in ONE place and every backend reuses it.
//! - [`LocalInferenceBackend`] — the higher-level surface (capability contract,
//!   prompt rendering, tool-call parsing). A `LocalInferenceBackend` IS a
//!   `TextDecoder` (supertrait); the engine upcasts to drive the loop.
//!
//! Adding an architecture = implement both traits for a new backend struct and
//! implement the trait. No engine-dispatch surgery.

use crate::schema::{ModelCapability, QuantScheme, Quantization};
use crate::tasks::generate::{parse_tool_calls, render_chat_prompt, GenerateRequest, ToolCall};
use crate::InferenceError;

/// Token-level primitives every in-process text backend exposes.
///
/// `forward` returns the next-token logits as `Vec<f32>` (the unifying shape the
/// shared MLX sampler `sample_from_logits` consumes). Backends whose native
/// forward returns something else (Candle returns a `Tensor`) adapt here.
pub trait TextDecoder: Send {
    /// Encode text to token ids (with the tokenizer's special tokens).
    fn encode(&self, text: &str) -> Result<Vec<u32>, InferenceError>;

    /// Decode token ids back to text.
    fn decode(&self, tokens: &[u32]) -> Result<String, InferenceError>;

    /// One prefill/decode pass; returns the final position's logits.
    fn forward(&mut self, tokens: &[u32], pos: usize) -> Result<Vec<f32>, InferenceError>;

    /// Every token id that terminates generation — the model's `eos` plus any
    /// chat-turn-ending control tokens. The shared loop stops on any of these,
    /// so it never needs to know a specific architecture's eos convention.
    fn eos_ids(&self) -> Vec<u32>;

    /// Context window in tokens (for the prompt-truncation guard).
    fn context_length(&self) -> usize;

    /// Reset the KV cache between independent generations.
    fn clear_kv_cache(&mut self);

    /// Prepare the KV cache to prefill `prompt_tokens`, reusing any already-cached
    /// matching prefix (prompt/prefix caching). Returns the offset to begin
    /// prefilling from: `prompt_tokens[..offset]` are already in the cache (their
    /// KV is identical because KV for a fixed token prefix at fixed positions is
    /// deterministic), so the caller only prefills `prompt_tokens[offset..]` at
    /// position `offset`.
    ///
    /// The default clears the cache and returns 0 — a full re-prefill, no reuse.
    /// Backends opt into cross-call reuse by overriding this (and tracking which
    /// tokens their cache represents). This is the big win for multi-turn agent
    /// loops, where each turn re-sends the whole growing conversation.
    fn begin_prompt(&mut self, prompt_tokens: &[u32]) -> usize {
        let _ = prompt_tokens;
        self.clear_kv_cache();
        0
    }
}

/// Result of one engine-driven decode pass.
pub struct LocalGeneration {
    pub text: String,
    pub ttft_ms: Option<u64>,
    /// `"stop"` (hit an eos id or a stop sequence), `"length"` (hit max_tokens),
    /// or `"local_decode_timeout"` (hit the wall-clock decode ceiling, car#851
    /// — spelled out rather than a bare `"timeout"`, which could collide with a
    /// remote provider's raw `finish_reason`). A ceiling-stopped pass still
    /// carries whatever text it managed to generate.
    pub stop_reason: Option<String>,
    /// Prompt tokens fed to the model (post context-window truncation), so the
    /// in-process path reports real `TokenUsage` like the remote providers do —
    /// without it, decode-throughput (tokens/sec) is unmeasurable (it was always
    /// 0 for local models).
    pub prompt_tokens: usize,
    /// Tokens the model generated this pass.
    pub completion_tokens: usize,
}

/// Outcome of the shared decode loop. Distinguishes a normal failure (the
/// backend is still usable) from one where a panic crossed the compute/FFI
/// boundary and left the backend in an indeterminate state — the caller, which
/// owns the cache lock, must then evict it.
pub enum DriveError {
    /// Normal failure (encode/decode/sample). Backend remains usable.
    Recoverable(InferenceError),
    /// A panic was caught mid-forward. The caller MUST drop its guard and
    /// invalidate the backend from the cache before returning.
    BackendCorrupted(InferenceError),
}

impl DriveError {
    pub fn into_inner(self) -> InferenceError {
        match self {
            DriveError::Recoverable(e) | DriveError::BackendCorrupted(e) => e,
        }
    }
}

/// Whether an in-process backend can load `model_type` **on this build**.
///
/// There is no list here, and that is the design. A `NATIVE_MLX_MODEL_TYPES`
/// constant used to stand in this spot, naming the architectures CAR had
/// hand-written MLX loaders for. Those loaders are gone —
/// `backend::swift_lm` links Apple's `mlx-swift-lm`, which maintains
/// sixty-odd — and the constant survived them naming four, so admission refused
/// every architecture the linked library would in fact have served. A list that
/// has to be edited to stay true will eventually be false; asking the library
/// cannot drift, and a package bump widens support with no Rust edit.
///
/// Platform-gated on purpose: the backend is MLX, so off Apple Silicon the
/// honest answer is always "no" regardless of the architecture, and a caller
/// choosing between the native and external source must not be told otherwise.
/// It also answers "no" when the Swift package was not built into this binary
/// (`CAR_BUILD_SWIFT_LM=0`, `car_skip_mlx`, or no Swift toolchain), because the
/// alternative is admitting a model to a stub.
/// Whether an in-process loader exists in this build at all, independent of any
/// architecture.
///
/// The cfg alone does not answer this: a macOS aarch64 build with
/// `CAR_BUILD_SWIFT_LM=0`, or built without a Swift toolchain, has the target
/// and no loader. Exposed here rather than leaving callers to reach for
/// `swift_lm::is_available`, because that module does not exist off Apple
/// Silicon — a caller outside a matching `#[cfg]` does not compile, which is
/// how a `car_skip_mlx` build broke on a test that asked the honest question
/// the dishonest way.
pub fn in_process_loader_available() -> bool {
    #[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
    {
        crate::backend::swift_lm::is_available()
    }
    #[cfg(not(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx))))]
    {
        false
    }
}

pub fn has_native_backend(model_type: &str) -> bool {
    #[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
    {
        crate::backend::swift_lm::supports_model_type(model_type)
    }
    #[cfg(not(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx))))]
    {
        let _ = model_type;
        false
    }
}

/// `config.json` `model_type` values the GGUF path implements.
///
/// Short on purpose. GGUF is not how a machine runs the best model it can: the
/// answer is MLX on Apple Silicon and CUDA elsewhere, both reading safetensors.
/// This path exists for the latency-critical hot set — see the local-models
/// rule in CLAUDE.md.
pub const GGUF_MODEL_TYPES: &[&str] = &["qwen3", "qwen3_moe"];

/// Whether a GGUF checkpoint of this architecture has a backend **on this
/// build**.
///
/// False on Apple Silicon for everything: `backend::candle` is compiled only
/// when the MLX path is absent (see `backend/mod.rs`), so a Mac has no GGUF
/// loader at all and a GGUF row there names a backend that does not exist.
/// Elsewhere it is the Candle GGUF loader, on CUDA, for the architectures it
/// implements.
pub fn gguf_backend_serves(model_type: &str) -> bool {
    #[cfg(not(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx))))]
    {
        GGUF_MODEL_TYPES.contains(&model_type)
    }
    #[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
    {
        let _ = model_type;
        false
    }
}

/// Whether the native MLX loader can decode this weight layout.
///
/// Platform-independent on purpose: it is a statement about the *format*, so it
/// is testable on every runner rather than only where `has_native_backend` can
/// return true. [`native_backend_serves`] combines it with the architecture.
///
/// The vocabulary is MLX's own, and the arms below follow MLX's own
/// `QuantizationMode`: `affine`, `mxfp4`, `mxfp8` and `nvfp4` are the four it
/// defines. They are all accepted.
///
/// This used to mirror the deleted `backend::mlx`'s `build_qlinear`, which had
/// a decoder for affine and `mxfp8` only and refused `mxfp4` at load. Carrying
/// that restriction past the loader it described refuses checkpoints the
/// linked MLX can serve — the `mlx-community` MXFP4 conversions are the live
/// case.
///
/// **Two limits, stated because the widening was argued from MLX's enum.** The
/// enum is the API surface, not the kernel set: MLX instantiates exactly three
/// block-scaled triples (`nvfp4` 16/4, `mxfp8` 32/8, `mxfp4` 32/4 — see
/// `fp_quantized.metal`), and an explicit `group_size`/`bits` passes through
/// unvalidated, so a non-canonical triple fails at kernel lookup *after* the
/// download this gate exists to avoid. Real `mlx-community` checkpoints use
/// the canonical triples, so that is latent rather than live. And `nvfp4` is
/// accepted from reading upstream source, not from loading one end to end;
/// `mxfp4` and `mxfp8` are reachable through `mlx_lm.convert --quant-mode` on
/// the Qwen3-0.6B checkpoint the parity fixture already uses, if this needs
/// falsifying for real.
pub fn quantization_is_decodable(quantization: Option<&Quantization>) -> bool {
    let Some(quantization) = quantization else {
        // Nothing declared: a full-precision checkpoint on the dense path.
        return true;
    };
    match quantization.scheme {
        // The affine `quantized_matmul` path, and unquantized weights.
        QuantScheme::AffineGroupInt | QuantScheme::Unquantized => true,
        // `mxfp4`, `mxfp8`, `nvfp4` — MLX's block-scaled float modes.
        QuantScheme::BlockScaledFloat => true,
        // GGUF layouts are not safetensors and never reach this loader.
        QuantScheme::KQuantMixed | QuantScheme::RtnBlock => false,
        // A layout that named itself and was not recognized — in practice the
        // HuggingFace `quant_method` families (awq, gptq, bitsandbytes). These
        // are the ones that matter: the loader has no path for them and does
        // not refuse them either, it builds a dense layer from a packed weight
        // and emits garbage (car-releases#61). Refusing here is what makes that
        // loud, before a multi-gigabyte download rather than after.
        QuantScheme::Unknown => false,
    }
}

/// Whether the in-process Rust backend can serve a checkpoint of this
/// `model_type` **in this quantization**, on this build.
///
/// [`has_native_backend`] answers only the architecture half. The loader also
/// has to be able to decode the weight layout — see
/// [`quantization_is_decodable`].
///
/// The refusal it mirrors happens while reading weights, which is *after* the
/// download. Consulting the quantization at admission turns a wasted
/// multi-gigabyte fetch and a load-time error into a routing decision: the
/// checkpoint goes to the CAR-managed external runtime, which does serve it.
/// Same rule and same reason as
/// `external_flux::native_backend_serves` on
/// the image path — the model decides the backend.
pub fn native_backend_serves(model_type: &str, quantization: Option<&Quantization>) -> bool {
    has_native_backend(model_type) && quantization_is_decodable(quantization)
}

/// What a backend claims *without* loading weights — consulted by the registry
/// gate (does a backend exist for this `model_type`?) and dispatch.
///
/// The local-directory scan in `registry::synthesize_local_schema` keeps its own
/// `KNOWN_LLM_TYPES` list alongside this, and that is not a duplicate: it runs
/// on every platform and answers whether a directory holds a causal-LM
/// checkpoint at all, which a macOS-only loader cannot be asked. It consults
/// [`has_native_backend`] as well, so the two widen together.
pub struct BackendDescriptor {
    pub backend_name: &'static str,
    /// `config.json` `model_type` strings this backend services.
    pub model_types: &'static [&'static str],
}

/// The local in-process mirror of [`ProtocolHandler`](crate::protocol::ProtocolHandler).
/// One impl per architecture family; loaded lazily and cached in the engine.
pub trait LocalInferenceBackend: TextDecoder {
    /// Stable id for tracing / unsupported-mode messages (e.g. `"native-mlx-qwen3"`).
    fn backend_name(&self) -> &'static str;

    /// The *execution* contract: what THIS loaded checkpoint can actually
    /// service, independent of the registry's routing claim. Mirrors today's
    /// `MlxBackend::supports_capability`.
    fn supports_capability(&self, cap: ModelCapability) -> bool;

    /// Render a request to the model's wire prompt string. The default is the
    /// hardcoded Qwen3 chat format; backends with their own template (gemma) or
    /// the data-driven `ChatTemplate` path override this.
    fn render_prompt(&self, req: &GenerateRequest) -> Result<String, InferenceError> {
        Ok(render_chat_prompt(req))
    }

    /// Extract tool calls from generated text. The default understands the Qwen
    /// Hermes `<tool_call>{json}</tool_call>` convention; architectures with a
    /// different convention (gemma's `<|tool_call>…<tool_call|>`) override.
    fn parse_tool_calls(&self, text: &str) -> (String, Vec<ToolCall>) {
        parse_tool_calls(text)
    }
}

// Non-macOS (Candle) deliberately does NOT implement these traits: the Candle
// backend is already a single generic GGUF loader, so it has no multi-arch
// dispatch problem to solve. This abstraction targets the macOS MLX path, where
// each architecture (Qwen3, Gemma 4, …) is a distinct hand-written backend that
// would otherwise each need its own arm in the engine's dispatch.

// ── Dispatch (macOS MLX) ─────────────────────────────────────────────────────

/// Read a local model's `config.json` `model_type` (lowercased). This is the
/// authoritative architecture signal — the same field the registry gates on.
#[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
fn read_model_type(model_dir: &std::path::Path) -> Result<String, InferenceError> {
    let cfg_path = model_dir.join("config.json");
    let raw = std::fs::read_to_string(&cfg_path).map_err(|e| {
        InferenceError::InferenceFailed(format!("read {}: {e}", cfg_path.display()))
    })?;
    let cfg: serde_json::Value = serde_json::from_str(&raw).map_err(|e| {
        InferenceError::InferenceFailed(format!("parse {}: {e}", cfg_path.display()))
    })?;
    Ok(cfg
        .get("model_type")
        .and_then(|v| v.as_str())
        .unwrap_or("")
        .to_ascii_lowercase())
}

/// Describe a model directory's architecture for an error message.
///
/// There used to be a `local_backend_for` here: a dispatch from `model_type` to
/// one of several hand-written backends. `mlx-swift-lm` serves every
/// architecture through one loader, so the dispatch collapsed to a single
/// construction and the indirection became a pass-through. It is gone, and
/// `SwiftLmBackend::load` is called from the engine's admitted loader closure
/// instead — which is also the only place a reservation exists, so the
/// allocation and its admission now sit in one function rather than two files
/// (`scripts/check-local-model-admission.sh` requires exactly that).
#[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
pub fn describe_model_type(model_dir: &std::path::Path) -> String {
    read_model_type(model_dir).unwrap_or_else(|_| "unknown".into())
}

#[cfg(test)]
mod native_admission_tests {
    use super::*;

    /// The gate asks the linked library, so it must admit architectures no
    /// Rust code ever named.
    ///
    /// `glm4_moe_lite` is the regression case with teeth: `mlx-swift-lm`
    /// registers a loader for it, and the `NATIVE_MLX_MODEL_TYPES` constant
    /// this replaced did not list it, so admission sent it to vLLM-MLX while a
    /// working in-process backend sat right there. CLAUDE.md's local-models
    /// rule even used it as the example of a family that "reaches users through
    /// the supervised vllm-mlx runtime".
    ///
    /// The nonsense `model_type` is the positive control. Without it this test
    /// would also pass against a predicate that returned `true` unconditionally
    /// — which is exactly the shape of a gate that admits a model to a backend
    /// that cannot load it.
    #[cfg(car_mlxlm_swift_built)]
    #[test]
    fn admission_follows_the_linked_loader_not_a_constant() {
        // ONE name, not an inventory. A list of seven reproduces in the test
        // suite exactly the drift this change removed from production: an
        // upstream rename would turn CI red for a reason that has nothing to
        // do with CAR. `glm4_moe_lite` earns its place because it is the
        // actual regression — the constant refused it while the linked library
        // implemented it — and because CLAUDE.md used it as the example of a
        // family that must go out to vLLM-MLX.
        assert!(
            has_native_backend("glm4_moe_lite"),
            "glm4_moe_lite: the linked loader registers it, so admission must allow it"
        );
        assert!(
            !has_native_backend("not_a_real_architecture_9f3c"),
            "the gate must be capable of refusing, or it is not a gate"
        );
        // `config.json` spells these lowercase; callers do not all normalize.
        assert!(has_native_backend("Qwen3"), "the gate is case-insensitive");

        // The shape, rather than another name: the gate must agree with
        // whether a loader is linked at all. This is what actually fails if
        // someone reintroduces a hardcoded list, and it cannot rot.
        assert_eq!(
            has_native_backend("glm4_moe_lite"),
            in_process_loader_available(),
            "admission must follow the linked loader, not a list"
        );
    }

    /// Off the Swift build there is no in-process loader, so the honest answer
    /// is "no" for everything — including the architectures the deleted Rust
    /// backends used to serve.
    #[cfg(not(car_mlxlm_swift_built))]
    #[test]
    fn admission_refuses_everything_without_the_swift_stack() {
        for t in ["qwen3", "gemma4_unified", "glm4_moe_lite", "nonsense_xyz"] {
            assert!(
                !has_native_backend(t),
                "{t}: no Swift stack in this build, so nothing loads in-process"
            );
        }
    }

    /// The layout half of the gate, following MLX's own `QuantizationMode`.
    /// Platform-independent so it actually executes on CI — an earlier version
    /// of this test was gated to Apple Silicon, where no CI runner evaluates
    /// it, and its ungated companion asserted only `false`s that
    /// `has_native_backend` already produced off-Mac.
    #[test]
    fn decodable_layouts_mirror_the_loader() {
        let decodable = [
            // Affine group quant → `quantized_matmul`.
            Quantization::from_mlx_config(Some(4), Some(64), None).unwrap(),
            Quantization::from_mlx_config(Some(4), Some(64), Some("affine")).unwrap(),
            // MLX's three block-scaled float modes. `mlx-swift-lm` quantizes
            // on load with whichever the checkpoint declares, so the width is
            // not a second gate — `mxfp4` in particular is how every `gpt_oss`
            // checkpoint ships.
            Quantization::from_mlx_config(Some(8), Some(32), Some("mxfp8")).unwrap(),
            Quantization::from_mlx_config(Some(4), Some(32), Some("mxfp4")).unwrap(),
            Quantization::from_mlx_config(Some(4), Some(16), Some("nvfp4")).unwrap(),
            // Full precision → the dense path.
            Quantization::parse("bf16"),
        ];
        for q in &decodable {
            assert!(
                quantization_is_decodable(Some(q)),
                "loader decodes {q:?}, gate must admit it"
            );
        }

        let refused = [
            // GGUF layouts never reach the safetensors loader.
            Quantization::parse("Q4_K_M"),
            Quantization::parse("Q8_0"),
            // A `quant_method` family the loader has no path for, and does not
            // refuse either — it builds a dense layer from a packed weight.
            Quantization::parse("awq"),
        ];
        for q in &refused {
            assert!(
                !quantization_is_decodable(Some(q)),
                "loader cannot decode {q:?}, gate must refuse it"
            );
        }

        // No block at all is a full-precision checkpoint, not an unknown one.
        assert!(quantization_is_decodable(None));
    }

    /// Block-scaled float is admitted on its mode, not on its width.
    ///
    /// This test used to assert the opposite — that only 8-bit microscaling was
    /// decodable — which mirrored the deleted Rust loader, where `mxfp8` was
    /// dequantized to dense and `mxfp4` refused. MLX itself defines `affine`,
    /// `mxfp4`, `mxfp8` and `nvfp4`, and the linked `mlx-swift-lm` quantizes on
    /// load with whichever the config names.
    #[test]
    fn block_scaled_float_is_decodable_at_any_width() {
        for mode in ["mxfp4", "mxfp8", "nvfp4"] {
            for bits in [4u8, 6, 8, 16] {
                let q = Quantization::from_mlx_config(Some(bits), Some(32), Some(mode)).unwrap();
                assert_eq!(q.scheme, QuantScheme::BlockScaledFloat, "{mode}/{bits}");
                assert!(quantization_is_decodable(Some(&q)), "{mode}/{bits}");
            }
        }
        // Positive control: the predicate can still say no, so the assertions
        // above are not passing against an unconditional `true`.
        assert!(!quantization_is_decodable(Some(&Quantization::parse(
            "gptq"
        ))));
    }

    /// The gate is the conjunction: neither half can rescue the other.
    #[test]
    fn architecture_and_layout_must_both_pass() {
        let affine = Quantization::from_mlx_config(Some(4), Some(64), None).unwrap();
        // An `awq` block, not `mxfp4`: MLX's own block-scaled float modes are
        // all decodable now, so the layout half is only falsifiable with a
        // `quant_method` family the loader has no path for.
        let undecodable = Quantization::parse("awq");

        // An unsupported architecture is refused whatever the layout — and on
        // a non-Apple build, every architecture is unsupported.
        //
        // This used to name `qwen3_5_moe` as the unsupported one, which was
        // true only while admission read a four-entry constant; `mlx-swift-lm`
        // registers a loader for it, so it is now supported and the assertion
        // was testing the limitation rather than the rule. A `model_type` no
        // registry can ever hold keeps the rule testable on every platform.
        assert!(!native_backend_serves(
            "not_a_real_architecture_9f3c",
            Some(&affine)
        ));
        assert!(!native_backend_serves(
            "some_architecture_from_next_month",
            None
        ));

        // A supported architecture still needs a decodable layout. Off Apple
        // Silicon `has_native_backend` is false for everything, so assert
        // against it rather than hardcoding a platform answer.
        let supported = has_native_backend("qwen3");
        assert_eq!(native_backend_serves("qwen3", Some(&affine)), supported);
        assert_eq!(native_backend_serves("qwen3", None), supported);
        assert!(
            !native_backend_serves("qwen3", Some(&undecodable)),
            "an undecodable layout is refused on every platform"
        );
    }
}