Skip to main content

car_inference/
registry.rs

1//! Unified model registry — local and remote models under one schema.
2//!
3//! Replaces the hardcoded `ModelRegistry` from `models.rs` with a schema-driven
4//! registry that treats all models as first-class typed resources. Users can
5//! register custom models (fine-tuned endpoints, private APIs) alongside the
6//! built-in catalog.
7
8use std::collections::{HashMap, HashSet};
9use std::path::{Path, PathBuf};
10use std::time::SystemTime;
11
12use serde::{Deserialize, Serialize};
13use tracing::{debug, info, warn};
14
15use crate::download::{DownloadEvent, ProgressSink};
16use crate::schema::*;
17use crate::InferenceError;
18
19/// Filter for querying the registry.
20#[derive(Debug, Clone, Default)]
21pub struct ModelFilter {
22    /// Required capabilities (model must have ALL of these).
23    pub capabilities: Vec<ModelCapability>,
24    /// Maximum on-disk / RAM size in MB.
25    pub max_size_mb: Option<u64>,
26    /// Maximum expected latency in ms (from declared envelope).
27    pub max_latency_ms: Option<u64>,
28    /// Maximum cost per 1M output tokens in USD.
29    pub max_cost_per_mtok: Option<f64>,
30    /// Required tags (model must have ALL of these).
31    pub tags: Vec<String>,
32    /// Filter by provider.
33    pub provider: Option<String>,
34    /// Only local models.
35    pub local_only: bool,
36    /// Only models that are currently available.
37    pub available_only: bool,
38}
39
40/// A curated replacement for a local model that is installed but no longer
41/// the preferred model in its line.
42#[derive(Debug, Clone, Serialize, Deserialize)]
43pub struct ModelUpgrade {
44    pub from_id: String,
45    pub from_name: String,
46    pub to_id: String,
47    pub to_name: String,
48    pub reason: String,
49    pub target_runtime: Option<String>,
50    pub target_runtime_requirement: Option<String>,
51    pub minimum_runtimes: Vec<ModelRuntimeRequirement>,
52    pub target_available: bool,
53    pub target_pullable: bool,
54    pub remove_old_supported: bool,
55}
56
57#[derive(Debug, Clone, Serialize, Deserialize)]
58pub struct ModelRuntimeRequirement {
59    pub name: String,
60    pub minimum_version: String,
61}
62
63/// How a registry answers "is a Parslee session signed in?", and whether it may
64/// act on a "no" by discarding session-scoped evidence.
65///
66/// Both call sites used to read `car_auth::access_token_is_available()`
67/// directly, which quietly made *constructing* a registry a process-global
68/// write: `new_with_catalog_public_key` ends in a `refresh_availability`, and on
69/// a machine with no session that refresh clears the learned gateway and
70/// credential observations. #989 took the crate's serial lock in the two tests
71/// that call `refresh_availability` themselves, but roughly twenty others merely
72/// build a registry and reached the same clear through construction — so a test
73/// that had just recorded an observation could have it erased under it
74/// (Parslee-ai/car#986). This probe is the seam that separates *what the answer
75/// is* from *whether acting on it is allowed*.
76#[derive(Debug, Clone, Copy)]
77pub(crate) enum SessionProbe {
78    /// Production. Ask `car_auth` on every refresh and ACT on the answer — a
79    /// signed-out answer discards the session-scoped gateway/credential
80    /// evidence (Parslee-ai/car#786, Parslee-ai/car#887).
81    // Never constructed in a test build — `Inert` is the `cfg(test)` default,
82    // which is the whole point of the seam rather than dead code.
83    #[cfg_attr(test, allow(dead_code))]
84    Live,
85    /// A caller-supplied answer, acted on exactly as `Live` acts on it:
86    /// `Fixed(false)` discards, `Fixed(true)` does not. A test that means to
87    /// exercise the discard asks for it here rather than depending on the
88    /// ambient auth state of the machine it runs on.
89    #[cfg(test)]
90    Fixed(bool),
91    /// The real answer, asked exactly as `Live` asks it, and never acted on.
92    /// Availability is therefore identical to `Live`, and the only difference
93    /// is that nothing outside the registry is written. Used by read-only
94    /// diagnostics and as the default under `cfg(test)`.
95    Inert,
96}
97
98impl SessionProbe {
99    /// Whether a Parslee session is signed in, using only the prompt-free
100    /// authority hint. Registry construction and catalog refresh must not turn
101    /// into a physical Keychain read.
102    fn available(&self) -> bool {
103        match self {
104            Self::Live => passive_parslee_oauth_available(),
105            Self::Inert => passive_parslee_oauth_available(),
106            #[cfg(test)]
107            Self::Fixed(available) => *available,
108        }
109    }
110
111    /// Whether the authority is explicitly signed out rather than merely
112    /// unknown or temporarily unreadable.
113    fn signed_out(&self) -> bool {
114        match self {
115            Self::Live => matches!(
116                car_auth::credential_authority_hint().state,
117                car_auth::CredentialAuthorityState::SignedOut
118            ),
119            Self::Inert => matches!(
120                car_auth::credential_authority_hint().state,
121                car_auth::CredentialAuthorityState::SignedOut
122            ),
123            #[cfg(test)]
124            Self::Fixed(available) => !available,
125        }
126    }
127
128    /// Whether a "not signed in" answer may discard the session-scoped
129    /// gateway/credential observations. False only for [`Self::Inert`], which
130    /// exists precisely so that building a registry has no effect outside it.
131    fn may_forget_session_evidence(&self) -> bool {
132        match self {
133            Self::Live => true,
134            #[cfg(test)]
135            Self::Fixed(_) => true,
136            Self::Inert => false,
137        }
138    }
139}
140
141/// Unified registry of all known models.
142#[derive(Clone)]
143pub struct UnifiedRegistry {
144    models_dir: PathBuf,
145    /// CAR state root — `~/.car`, or `$CAR_HOME` when the operator relocated
146    /// this daemon. Every *state* path the registry touches (`models.json`, the
147    /// signed-catalog cache, the discovery cache) hangs off this, never off
148    /// `models_dir`, which is the machine-shared weights cache and deliberately
149    /// does not move.
150    state_root: PathBuf,
151    /// All registered models, keyed by id.
152    models: HashMap<String, ModelSchema>,
153    /// Credential variables whose state the LAST authoritative refresh could
154    /// not determine -- the secret store errored rather than answering
155    /// "absent". Per-refresh runtime state, deliberately here rather than on
156    /// `ModelSchema`, which is catalog content. Read when projecting a row's
157    /// `credential_required`: a store failure is not evidence a key is
158    /// missing, and rendering "Add key…" for one the person may already hold
159    /// is worse than saying nothing (car#1920).
160    credential_state_unknown: HashSet<String>,
161    /// Exact ids owned by the compiled or authenticated project catalog.
162    /// Public/user-controlled rows are additive and may never replace these.
163    project_model_ids: HashSet<String>,
164    /// The signed catalog each signed row came from, so a row kept after its
165    /// catalog moved on can be re-verified from those exact bytes at the
166    /// next startup.
167    signed_origin: HashMap<String, std::sync::Arc<crate::catalog::CatalogCacheEnvelope>>,
168    /// Signed rows a newer catalog withdrew but something still depends on;
169    /// persisted by the engine so a restart keeps them too.
170    retained_ids: HashSet<String>,
171    /// Ids the current signed catalog revokes. Kept so startup never restores
172    /// a retained row the publisher has since revoked.
173    revoked_ids: HashSet<String>,
174    /// Compiled builtins a revocation marked deprecated; the build's own value
175    /// comes back when a later catalog stops revoking them.
176    revoked_builtin_ids: HashSet<String>,
177    /// Compiled builtin ids whose metadata a signed row currently overlays
178    /// (see [`Self::overlay_builtin`]); the next catalog that omits one
179    /// restores the compiled values.
180    signed_overlay_ids: HashSet<String>,
181    /// Exact ids whose current rows came from the signed catalog, so the next
182    /// catalog replaces them rather than accreting: a row the publisher
183    /// dropped must leave the registry, not linger until restart.
184    signed_catalog_ids: HashSet<String>,
185    /// The strongest reserved namespace: exact ids compiled into this build.
186    /// Even an authenticated refresh is additive and cannot change them.
187    builtin_model_ids: HashSet<String>,
188    /// IDs whose current rows came from the user-controlled `models.json`
189    /// boundary or an explicit user registration.
190    ///
191    /// Tags cannot carry this provenance: a user can add `builtin`, while a
192    /// valid signed catalog row does not have to. Keeping the boundary private
193    /// prevents generic/builtin/catalog/discovery registration from being
194    /// serialized back into user config and losing its trust tier on restart.
195    user_config_ids: HashSet<String>,
196    /// IDs synthesized solely from directories found under `models_dir`.
197    ///
198    /// These rows exist only while their directories do. Keep provenance in a
199    /// private set rather than trusting the public `auto-discovered` tag: a
200    /// user-configured schema may carry any tag and must not disappear merely
201    /// because no same-named local directory exists.
202    on_disk_discovered_ids: HashSet<String>,
203    /// User-added model config file path — `models.json` resolved against
204    /// `state_root`, i.e. the same answer the free [`user_config_path`] gives.
205    user_config_path: PathBuf,
206    /// Where to report downloads that nobody explicitly asked for.
207    ///
208    /// [`ensure_local`](Self::ensure_local) is the *implicit* acquisition path:
209    /// it runs inside `generate`, when the router picks a model whose weights
210    /// aren't on disk yet. It used to hard-code `ProgressSink::none()`, so a
211    /// multi-gigabyte fetch triggered by a plain `car infer "hi"` produced no
212    /// output on any surface — the command simply appeared to hang, and a
213    /// user on a metered connection had no signal at all (Parslee-ai/car#620).
214    ///
215    /// Explicit pulls (`car models pull`) always passed a real sink and were
216    /// never affected; this closes the gap for the path users actually hit
217    /// first. Defaults to [`ProgressSink::none`], so an embedder that doesn't
218    /// opt in behaves exactly as before.
219    ambient_progress: ProgressSink,
220    /// How this registry learns whether a Parslee session is signed in, and
221    /// whether it may act on a "no". See [`SessionProbe`]: normal production
222    /// registries carry `Live`; read-only diagnostics and test registries use
223    /// `Inert` so construction cannot erase an existing observation.
224    session: SessionProbe,
225}
226
227#[derive(Debug, Clone, Deserialize)]
228struct ModelUpgradeRule {
229    from_ids: Vec<String>,
230    to_id: String,
231    reason: String,
232    target_runtime: Option<String>,
233    target_runtime_requirement: Option<String>,
234    #[serde(default)]
235    minimum_runtimes: Vec<ModelRuntimeRequirement>,
236    #[serde(default = "default_remove_old_after_available")]
237    remove_old_after_available: bool,
238}
239
240fn default_remove_old_after_available() -> bool {
241    true
242}
243
244fn environment_credential_available(env_var: &str) -> bool {
245    std::env::var(env_var).is_ok_and(|value| !value.trim().is_empty())
246}
247
248/// Does a credential exist, in the environment or the OS secret store —
249/// **without decrypting it**.
250///
251/// The distinction is enforced, not stylistic. `car-cli`'s
252/// `models_list_explicit_status_refreshes_openrouter_without_value_read` and
253/// its `doctor` twin assert that a status surface raises
254/// `secret_store_activity().status_attempts` and leaves `get_attempts`
255/// unchanged: answering "can I use this model?" is an existence question, and
256/// a surface that answers it by reading secret values hands every catalog
257/// render a reason to decrypt every provider key on the machine.
258///
259/// `resolve_env_or_keychain` is the wrong tool here for exactly that reason —
260/// it returns the value, and using it made this refresh perform four value
261/// reads per render. `status_via_operator_broker_or_keychain` takes the same
262/// broker decision and reduces the answer to a boolean before any caller sees
263/// it.
264///
265/// An error — a refused read, a store that will not open — is **not**
266/// evidence of absence, but it is also not evidence the model is usable, so it
267/// reports `false` the same way a missing key does. That is the conservative
268/// direction for a runnability claim: it can under-promise, never over-promise.
269pub(crate) fn credential_present_without_reading_it(env_var: &str) -> bool {
270    matches!(credential_presence(env_var), CredentialPresence::Present)
271}
272
273/// What the store could tell us about a credential, without decrypting it.
274///
275/// The third answer is the point. Collapsing a store FAILURE into "absent"
276/// made a temporarily unreadable key indistinguishable from one that was
277/// never set, and the model screen then told the person to add a credential
278/// they may already have (car#1920).
279#[derive(Debug, Clone, Copy, PartialEq, Eq)]
280pub(crate) enum CredentialPresence {
281    Present,
282    Absent,
283    /// The store errored. Not evidence either way.
284    Unknown,
285}
286
287pub(crate) fn credential_presence(env_var: &str) -> CredentialPresence {
288    if environment_credential_available(env_var) {
289        return CredentialPresence::Present;
290    }
291    match car_secrets::status_via_operator_broker_or_keychain(&car_secrets::SecretRef::new(
292        car_secrets::DEFAULT_SERVICE,
293        env_var,
294    )) {
295        Ok(status) if status.exists => CredentialPresence::Present,
296        Ok(_) => CredentialPresence::Absent,
297        Err(_) => CredentialPresence::Unknown,
298    }
299}
300
301/// Process-local hint that an openai-codex sign-in exists. Refreshed by
302/// explicit status reads and by each openai-codex request credential lookup,
303/// so passive catalog reads never touch the secret store. Relaxed ordering is
304/// enough: it is an advisory availability hint, not a synchronisation point.
305static CODEX_SIGNED_IN_HINT: std::sync::atomic::AtomicBool =
306    std::sync::atomic::AtomicBool::new(false);
307
308pub(crate) fn remember_codex_sign_in(present: bool) {
309    CODEX_SIGNED_IN_HINT.store(present, std::sync::atomic::Ordering::Relaxed);
310}
311
312/// Whether an openai-codex sign-in is known from the process-local hint,
313/// without touching the secret store.
314pub(crate) fn passive_codex_oauth_available() -> bool {
315    CODEX_SIGNED_IN_HINT.load(std::sync::atomic::Ordering::Relaxed)
316}
317
318fn passive_parslee_oauth_available() -> bool {
319    matches!(
320        car_auth::credential_authority_hint().state,
321        car_auth::CredentialAuthorityState::Configured
322    )
323}
324
325/// `resolved` is the per-refresh credential memo. Passive catalog refreshes
326/// populate it from environment presence only; request-time refreshes may
327/// populate it authoritatively from the secret store. Keeping the source of
328/// truth outside this helper prevents a single-model registration from
329/// silently becoming a physical Keychain read.
330fn proprietary_auth_available(
331    model_id: &str,
332    schema_provider: &str,
333    source_provider: &str,
334    auth: &ProprietaryAuth,
335    parslee_oauth_available: bool,
336    resolved: &std::collections::HashMap<String, bool>,
337) -> bool {
338    // Authentication is necessary but NOT sufficient for a PROXIED namespace.
339    // The OAuth2Pkce arm below resolves to exactly "is a Parslee session signed
340    // in", which is true regardless of what the gateway can actually serve —
341    // so ten `parslee/openrouter/*` aliases advertised `available` while every
342    // one of them 503'd with `openrouter_not_configured`, costing a benchmark
343    // sweep that picked one on the strength of that claim (Parslee-ai/car#786).
344    //
345    // There is no discovery endpoint to consult instead (`parslee.capabilities`
346    // enumerates product entitlements, not inference upstreams), so the gateway's
347    // own answer to a real request is the only truthful signal there is. Once it
348    // has told us, stop advertising — the catalog is then optimistic once and
349    // self-correcting, rather than permanently wrong.
350    //
351    // Checked here rather than at the call sites so `register` and
352    // `refresh_availability` cannot drift apart.
353    if crate::openrouter::is_curated_managed_gateway_alias(model_id)
354        && crate::openrouter::gateway_unconfigured()
355    {
356        return false;
357    }
358    // The neighbouring half of the same mistake, one layer earlier: a signed-in
359    // session is not a WORKING one. `access_token_is_available` is documented
360    // existence-only — it reports that a V2 record has an active slot, never
361    // that the token is still accepted — so a machine with a stale sign-in
362    // offered every managed alias as a top-quality candidate and 401'd on every
363    // one, per call, forever (Parslee-ai/car#887).
364    //
365    // Unlike the gateway check above this is NOT scoped to the curated
366    // OpenRouter aliases: a dead credential kills every `parslee/*` row, so the
367    // suppression belongs on the OAuth2Pkce arm as a whole.
368    //
369    // Checked here rather than at the call sites for the same reason as the
370    // gateway verdict — so `register` and `refresh_availability` cannot drift.
371    if crate::parslee_credential::credential_rejected()
372        && matches!(auth, ProprietaryAuth::OAuth2Pkce { .. })
373    {
374        return false;
375    }
376    match auth {
377        ProprietaryAuth::ApiKeyEnv { env_var } | ProprietaryAuth::BearerTokenEnv { env_var } => {
378            resolved.get(env_var).copied().unwrap_or(false)
379        }
380        ProprietaryAuth::OAuth2Pkce { .. } => {
381            schema_provider.eq_ignore_ascii_case("parslee")
382                && source_provider.eq_ignore_ascii_case("parslee")
383                && parslee_oauth_available
384        }
385        ProprietaryAuth::ChatGptSubscription {} => {
386            schema_provider.eq_ignore_ascii_case(OPENAI_CODEX_PROVIDER)
387                && source_provider.eq_ignore_ascii_case(OPENAI_CODEX_PROVIDER)
388                && passive_codex_oauth_available()
389        }
390    }
391}
392
393fn model_upgrade_rules() -> Vec<ModelUpgradeRule> {
394    serde_json::from_str(include_str!("../assets/model-upgrades.json"))
395        .expect("built-in model-upgrades.json should parse")
396}
397
398/// File name of the user-registered model config under the CAR state root.
399pub const USER_MODELS_FILE: &str = "models.json";
400
401/// The one resolver for `models.json`: [`USER_MODELS_FILE`] under the CAR state
402/// root (`~/.car/models.json` unless `CAR_HOME` moves the root).
403///
404/// Every reader and writer in the tree goes through this — the registry's own
405/// load/save, and the daemon's `models.register` / `models.unregister`
406/// handlers. They have to agree: when the write side moved to `CAR_HOME` and
407/// the read side stayed derived from the (deliberately unmoved) weights dir, a
408/// relocated daemon persisted a registration to one file and read a different
409/// one on the next boot, so the registration silently never took effect.
410///
411/// `None` only when there is no `CAR_HOME`, `HOME` or `USERPROFILE` — the same
412/// condition every other state path treats as "cannot resolve". The registry
413/// itself joins [`USER_MODELS_FILE`] onto the state root it was constructed
414/// with, which for [`UnifiedRegistry::new`] is the infallible
415/// `car_home::root_or_relative()`; same root, same file name, differing only in
416/// what they do when nothing resolves at all (error here, relative `.car`
417/// there — the pre-existing behaviour of each caller).
418pub fn user_config_path() -> Option<PathBuf> {
419    car_home::root().map(|root| root.join(USER_MODELS_FILE))
420}
421
422impl UnifiedRegistry {
423    /// Registry for the install that is actually running: state files resolve
424    /// under the CAR state root (`$CAR_HOME`, else `~/.car`), weights under
425    /// `models_dir`.
426    pub fn new(models_dir: PathBuf) -> Self {
427        Self::new_with_state_root(car_home::root_or_relative(), models_dir)
428    }
429
430    /// [`new`](Self::new) with the state root supplied rather than resolved
431    /// from the environment. The daemon passes its own resolved root; tests and
432    /// embedders that want every file inside one directory pass that directory.
433    pub fn new_with_state_root(state_root: PathBuf, models_dir: PathBuf) -> Self {
434        let catalog_public_key = crate::catalog::catalog_public_key();
435        Self::new_with_catalog_public_key(state_root, models_dir, Some(&catalog_public_key))
436    }
437
438    fn new_with_catalog_public_key(
439        state_root: PathBuf,
440        models_dir: PathBuf,
441        catalog_public_key: Option<&str>,
442    ) -> Self {
443        // Production asks `car_auth` live and acts on the answer. Under
444        // `cfg(test)` the same question is asked the same way but held INERT, so
445        // building a registry — which every registry test does, and most of them
446        // without taking the crate's serial lock — reports identical
447        // availability while clearing nothing outside itself
448        // (Parslee-ai/car#986). A test that means to exercise the clear asks for
449        // it explicitly with `new_with_session(.., SessionProbe::Fixed(false))`.
450        #[cfg(not(test))]
451        let session = SessionProbe::Live;
452        #[cfg(test)]
453        let session = SessionProbe::Inert;
454        Self::new_with_session(state_root, models_dir, catalog_public_key, session)
455    }
456
457    /// [`new_with_catalog_public_key`](Self::new_with_catalog_public_key) with
458    /// the session answer supplied rather than defaulted. See [`SessionProbe`].
459    pub(crate) fn new_with_session(
460        state_root: PathBuf,
461        models_dir: PathBuf,
462        catalog_public_key: Option<&str>,
463        session: SessionProbe,
464    ) -> Self {
465        let user_config_path = state_root.join(USER_MODELS_FILE);
466
467        let mut registry = Self {
468            models_dir,
469            state_root,
470            models: HashMap::new(),
471            credential_state_unknown: HashSet::new(),
472            project_model_ids: HashSet::new(),
473            signed_catalog_ids: HashSet::new(),
474            signed_overlay_ids: HashSet::new(),
475            signed_origin: HashMap::new(),
476            retained_ids: HashSet::new(),
477            revoked_ids: HashSet::new(),
478            revoked_builtin_ids: HashSet::new(),
479            builtin_model_ids: HashSet::new(),
480            user_config_ids: HashSet::new(),
481            on_disk_discovered_ids: HashSet::new(),
482            user_config_path,
483            ambient_progress: ProgressSink::none(),
484            session,
485        };
486        registry.load_builtin_catalog();
487        // Refreshable signed catalog (E1): a prior `refresh_catalog` stores the
488        // exact authenticated body + signature envelope. Startup re-verifies
489        // that envelope with the currently configured key before loading any
490        // model additively alongside the built-ins.
491        // Then rows an earlier catalog had that something here still depends
492        // on, each re-verified from the catalog it came from.
493        let cached = crate::catalog::load_verified(
494            &crate::catalog::cache_path(&registry.state_root),
495            catalog_public_key,
496        );
497        if let Some(cached) = cached {
498            let origin = std::sync::Arc::new(cached.envelope());
499            let revoked: HashSet<String> = cached.revoked().iter().cloned().collect();
500            registry.replace_signed_catalog(
501                cached.into_models(),
502                Some(origin),
503                &HashSet::new(),
504                &revoked,
505            );
506        }
507        registry.restore_retained(crate::catalog::load_retained(
508            &crate::catalog::retained_path(&registry.state_root),
509            catalog_public_key,
510        ));
511        // Auto-discovered models (E2): Community-tier entries cached by a prior
512        // discovery pass (provider /v1/models). Loaded on top of built-ins +
513        // signed catalog, but never *overwriting* a curated/signed entry of the
514        // same id — discovery only ever ADDS models the catalog doesn't have.
515        for schema in crate::discovery::load_cache(&crate::discovery::cache_path(
516            &registry.state_models_dir(),
517        )) {
518            if !registry.models.contains_key(&schema.id) {
519                registry.register(schema);
520            }
521        }
522        registry.refresh_availability();
523        // Load user config on top (silently ignore if missing)
524        let _ = registry.load_user_config();
525        // Surface models the user pulled/placed under ~/.car/models/ that no
526        // catalog or user-config entry covers, so a locally-present model is
527        // visible and routable instead of invisible (car-releases#62).
528        registry.discover_on_disk_models();
529        registry
530    }
531
532    fn empty_with_state_root(state_root: PathBuf, models_dir: PathBuf) -> Self {
533        let user_config_path = state_root.join(USER_MODELS_FILE);
534        Self {
535            models_dir,
536            state_root,
537            models: HashMap::new(),
538            credential_state_unknown: HashSet::new(),
539            project_model_ids: HashSet::new(),
540            signed_catalog_ids: HashSet::new(),
541            signed_overlay_ids: HashSet::new(),
542            signed_origin: HashMap::new(),
543            retained_ids: HashSet::new(),
544            revoked_ids: HashSet::new(),
545            revoked_builtin_ids: HashSet::new(),
546            builtin_model_ids: HashSet::new(),
547            user_config_ids: HashSet::new(),
548            on_disk_discovered_ids: HashSet::new(),
549            user_config_path,
550            ambient_progress: ProgressSink::none(),
551            // No refresh runs during construction, so this probe is never
552            // consulted. Keep it inert in case an internal caller refreshes.
553            session: SessionProbe::Inert,
554        }
555    }
556
557    /// Test-only: a registry with NO builtin catalog, signed cache, discovery
558    /// cache, or user config — only the models a test explicitly `register`s.
559    /// Routing unit tests that assert a synthetic model wins MUST use this:
560    /// `new()` loads the real builtin frontier models, which become routable
561    /// candidates the moment a test sets their `api_key_env` (e.g.
562    /// `OPENAI_API_KEY`) — and once those models carry real benchmark scores
563    /// (e.g. car-judged), they legitimately outrank the synthetic stand-ins and
564    /// silently break the test's assumption. An empty registry keeps the test
565    /// hermetic.
566    ///
567    /// The state root is taken to be `models_dir`'s parent, so a test that
568    /// passes `<tmp>/models` gets `<tmp>` and every state file stays inside the
569    /// temp dir. This constructor never reads or writes anything on its own, so
570    /// the derivation only matters if the test then calls `save_user_config`.
571    #[cfg(test)]
572    pub fn new_empty(models_dir: PathBuf) -> Self {
573        let state_root = models_dir.parent().unwrap_or(&models_dir).to_path_buf();
574        Self::empty_with_state_root(state_root, models_dir)
575    }
576
577    /// Registry for a deterministic offline diagnosis fixture.
578    ///
579    /// It reads only the explicitly named models directory: no builtin,
580    /// signed, discovered, or user catalog is loaded, and no credential or
581    /// session availability probe runs. Local model directories are inferred
582    /// with the same conservative on-disk discovery used by production.
583    pub(crate) fn new_isolated_for_diagnosis(state_root: PathBuf, models_dir: PathBuf) -> Self {
584        let mut registry = Self::empty_with_state_root(state_root, models_dir);
585        registry.discover_on_disk_models();
586        registry
587    }
588
589    /// Where the registry's own state that has always sat beside the weights
590    /// lives: `models/` under the state root. Identical to `models_dir`
591    /// whenever `CAR_HOME` is unset, which is why no existing install's files
592    /// move; different — and per-daemon — once it is set.
593    fn state_models_dir(&self) -> PathBuf {
594        self.state_root.join("models")
595    }
596
597    /// Register on-disk models under `models_dir` that nothing else already
598    /// covers. Curated catalog entries already light up via the
599    /// availability check (their `name` maps to `<models_dir>/<name>`), so
600    /// this only fills the gap for *uncatalogued* local models — e.g. a
601    /// model a user fetched by hand.
602    ///
603    /// Capability guessing from a bare directory is unreliable, and
604    /// mis-tagging a speech/vision/video checkpoint as `Generate` would
605    /// poison routing. So this is deliberately conservative: it only
606    /// registers text LLMs it can positively identify by a recognized
607    /// `model_type` in `config.json` (MLX) or by a GGUF weight file, infers
608    /// embed/rerank from the directory name, and skips everything else.
609    /// car-releases#62.
610    fn discover_on_disk_models(&mut self) {
611        let entries = match std::fs::read_dir(&self.models_dir) {
612            Ok(e) => e,
613            Err(_) => return,
614        };
615        // Names already registered (case-insensitive) — never shadow a
616        // curated/user/discovered entry with an inferred one — and names of
617        // models a signed catalog revoked, which must not come back as
618        // uncatalogued local ones.
619        let known: std::collections::HashSet<String> = self
620            .models
621            .values()
622            .map(|m| m.name.to_ascii_lowercase())
623            .chain(
624                crate::catalog::load_revoked_names(&crate::catalog::revoked_names_path(
625                    &self.state_root,
626                ))
627                .into_keys()
628                .map(|name| name.to_ascii_lowercase()),
629            )
630            .collect();
631
632        for entry in entries.flatten() {
633            let path = entry.path();
634            if !path.is_dir() {
635                continue;
636            }
637            let Some(name) = path
638                .file_name()
639                .and_then(|n| n.to_str())
640                .map(str::to_string)
641            else {
642                continue;
643            };
644            if known.contains(&name.to_ascii_lowercase()) {
645                continue;
646            }
647
648            let Some(schema) = synthesize_local_schema(&name, &path) else {
649                continue;
650            };
651            tracing::info!(
652                id = %schema.id,
653                name = %name,
654                "auto-discovered uncatalogued local model under models_dir (car-releases#62)"
655            );
656            let id = schema.id.clone();
657            self.register(schema);
658            if self.models.contains_key(&id) {
659                self.on_disk_discovered_ids.insert(id);
660            }
661        }
662    }
663
664    /// Remove inferred rows whose source directories no longer exist.
665    ///
666    /// The daemon keeps its registry in memory, so an on-disk model removed by
667    /// another process otherwise remains visible until restart. Activity and
668    /// mutation lock records are durable coordination sentinels, not registry
669    /// entries, and deliberately have no bearing on this reconciliation.
670    pub(crate) fn prune_missing_on_disk_models(&mut self) {
671        let missing = self
672            .on_disk_discovered_ids
673            .iter()
674            .filter(|id| {
675                self.models
676                    .get(id.as_str())
677                    .is_none_or(|schema| !self.models_dir.join(&schema.name).is_dir())
678            })
679            .cloned()
680            .collect::<Vec<_>>();
681        for id in missing {
682            tracing::debug!(
683                model_id = %id,
684                "dropping vanished auto-discovered model; coordination records do not register models"
685            );
686            self.on_disk_discovered_ids.remove(&id);
687            self.models.remove(&id);
688        }
689    }
690
691    /// Register a model at a public runtime boundary.
692    ///
693    /// Callers cannot confer project curation: even a legacy schema whose
694    /// omitted `trust_tier` deserializes as `Curated` is normalized to
695    /// `Community` here. Project-owned builtins and signature-verified catalogs
696    /// use the crate-private [`register_project_model`](Self::register_project_model)
697    /// boundary instead.
698    pub fn register(&mut self, mut schema: ModelSchema) {
699        if crate::openrouter::is_curated_managed_gateway_alias(&schema.id) {
700            warn!(id = %schema.id, "ignoring user registration for reserved Parslee-managed alias");
701            return;
702        }
703        if self.project_model_ids.contains(&schema.id) {
704            warn!(id = %schema.id, "ignoring public registration for project-owned exact id");
705            return;
706        }
707        schema.mark_user_registered();
708        let id = schema.id.clone();
709        if self.register_preserving_trust(schema) {
710            self.on_disk_discovered_ids.remove(&id);
711        }
712    }
713
714    /// Register a model whose provenance was established by CAR itself.
715    ///
716    /// This trust-preserving path is intentionally crate-private. Production
717    /// callers are limited to compiled builtins and signature-verified catalog
718    /// rows; tests may use it to construct project-curated fixtures.
719    pub(crate) fn register_project_model(&mut self, schema: ModelSchema) -> bool {
720        let id = schema.id.clone();
721        if !self.register_preserving_trust(schema) {
722            return false;
723        }
724        self.on_disk_discovered_ids.remove(&id);
725        self.project_model_ids.insert(id);
726        true
727    }
728
729    /// Make `rows` the signed catalog, under the one rule both startup and a
730    /// live refresh use. Returns how many rows it loaded or overlaid.
731    ///
732    /// - A row for a compiled builtin id overlays only the builtin's
733    ///   [`OVERLAY_FIELDS`](Self::overlay_builtin) — benchmarks, cost,
734    ///   performance, deprecation — and only when it names the same weights;
735    ///   which weights an id loads stays fixed by the build.
736    /// - Any other row is additive: never over another project-owned id, and
737    ///   never over a row the user registered, whose definition wins while
738    ///   the daemon runs and is what `save_user_config` keeps writing.
739    /// - A previous signed row the new catalog drops leaves — unless `keep`
740    ///   names it (installed, pinned, a lane default). Then it stays
741    ///   resolvable, marked `deprecated` so nothing suggests it, until nothing
742    ///   depends on it. Removing it would orphan the weights and blank the
743    ///   lane that names it.
744    pub(crate) fn replace_signed_catalog(
745        &mut self,
746        rows: Vec<ModelSchema>,
747        origin: Option<std::sync::Arc<crate::catalog::CatalogCacheEnvelope>>,
748        keep: &HashSet<String>,
749        revoked: &HashSet<String>,
750    ) -> usize {
751        let incoming: HashSet<&str> = rows.iter().map(|m| m.id.as_str()).collect();
752        for id in std::mem::take(&mut self.signed_catalog_ids) {
753            // A revoked row goes whatever depends on it: revocation is the
754            // publisher saying the weights are bad, not merely superseded.
755            if keep.contains(&id) && !incoming.contains(id.as_str()) && !revoked.contains(&id) {
756                if let Some(row) = self.models.get_mut(&id) {
757                    row.deprecated = true;
758                }
759                self.retained_ids.insert(id.clone());
760                self.signed_catalog_ids.insert(id);
761                continue;
762            }
763            self.models.remove(&id);
764            self.project_model_ids.remove(&id);
765            self.credential_state_unknown.remove(&id);
766            self.signed_origin.remove(&id);
767            self.retained_ids.remove(&id);
768        }
769        let overlaid = std::mem::take(&mut self.signed_overlay_ids);
770        let unrevoked = std::mem::take(&mut self.revoked_builtin_ids);
771        let restore: HashSet<String> = overlaid.union(&unrevoked).cloned().collect();
772        let compiled: HashMap<String, ModelSchema> = if restore.is_empty() {
773            HashMap::new()
774        } else {
775            builtin_catalog()
776                .into_iter()
777                .filter(|m| restore.contains(&m.id))
778                .map(|m| (m.id.clone(), m))
779                .collect()
780        };
781        for id in &restore {
782            if let (Some(row), Some(original)) = (self.models.get_mut(id), compiled.get(id)) {
783                Self::copy_overlay_fields(original, row);
784            }
785        }
786        self.revoked_ids = revoked.clone();
787        let mut loaded = 0;
788        for schema in rows {
789            let id = schema.id.clone();
790            if revoked.contains(&id) {
791                warn!(%id, "ignoring a signed catalog row its own catalog revokes");
792                continue;
793            }
794            if self.register_signed_catalog_model(schema) {
795                loaded += 1;
796                if let Some(origin) = &origin {
797                    if self.signed_catalog_ids.contains(&id) {
798                        self.signed_origin.insert(id, origin.clone());
799                    }
800                }
801            }
802        }
803        // A revoked builtin cannot leave the build; it stops being suggested
804        // or applied.
805        for id in revoked {
806            if self.builtin_model_ids.contains(id) {
807                if let Some(row) = self.models.get_mut(id) {
808                    row.deprecated = true;
809                    self.revoked_builtin_ids.insert(id.clone());
810                }
811            }
812        }
813        loaded
814    }
815
816    /// Whether the current signed catalog revokes `id`.
817    pub fn is_revoked(&self, id: &str) -> bool {
818        self.revoked_ids.contains(id)
819    }
820
821    /// Ids the current signed catalog revokes.
822    pub fn revoked_ids(&self) -> Vec<String> {
823        self.revoked_ids.iter().cloned().collect()
824    }
825
826    /// Whether `id` is a row compiled into this build (signed overlays
827    /// included), as opposed to one only a signed catalog or the user added.
828    pub fn is_builtin(&self, id: &str) -> bool {
829        self.builtin_model_ids.contains(id)
830    }
831
832    /// Signed rows kept after their catalog withdrew them, with the catalog
833    /// each came from — what the engine persists.
834    pub(crate) fn retained_rows(
835        &self,
836    ) -> Vec<(String, std::sync::Arc<crate::catalog::CatalogCacheEnvelope>)> {
837        let mut rows: Vec<_> = self
838            .retained_ids
839            .iter()
840            .filter_map(|id| Some((id.clone(), self.signed_origin.get(id)?.clone())))
841            .collect();
842        rows.sort_by(|a, b| a.0.cmp(&b.0));
843        rows
844    }
845
846    /// The signed catalog a row came from, if it is a signed row.
847    pub(crate) fn signed_origin_of(
848        &self,
849        id: &str,
850    ) -> Option<std::sync::Arc<crate::catalog::CatalogCacheEnvelope>> {
851        self.signed_origin.get(id).cloned()
852    }
853
854    /// Bring back verified rows as retained — deprecated, resolvable, never
855    /// suggested — unless the id is already present. Startup restores what
856    /// the last run kept; a pull that outlived its row's catalog keeps it.
857    pub(crate) fn restore_retained(
858        &mut self,
859        rows: Vec<(
860            ModelSchema,
861            std::sync::Arc<crate::catalog::CatalogCacheEnvelope>,
862        )>,
863    ) {
864        for (mut schema, origin) in rows {
865            if self.models.contains_key(&schema.id) || self.revoked_ids.contains(&schema.id) {
866                continue;
867            }
868            schema.deprecated = true;
869            let id = schema.id.clone();
870            if self.register_signed_catalog_model(schema) && self.signed_catalog_ids.contains(&id) {
871                self.signed_origin.insert(id.clone(), origin);
872                self.retained_ids.insert(id);
873            }
874        }
875    }
876
877    /// The fields a signed row may change on a compiled builtin: what CAR
878    /// measured or priced about a model, never which weights it loads.
879    fn copy_overlay_fields(from: &ModelSchema, to: &mut ModelSchema) {
880        to.public_benchmarks = from.public_benchmarks.clone();
881        to.cost = from.cost.clone();
882        to.performance = from.performance.clone();
883        to.deprecated = from.deprecated;
884    }
885
886    /// Apply a signed row's measurements to the compiled builtin it names.
887    /// Refused when the row names different weights: a publisher may update
888    /// what CAR knows about a builtin, not repoint it.
889    fn overlay_builtin(&mut self, schema: &ModelSchema) -> bool {
890        let Some(row) = self.models.get_mut(&schema.id) else {
891            return false;
892        };
893        let same_weights =
894            serde_json::to_value(&row.source).ok() == serde_json::to_value(&schema.source).ok();
895        if !same_weights {
896            warn!(id = %schema.id, "ignoring signed overlay that names different weights for a compiled builtin");
897            return false;
898        }
899        Self::copy_overlay_fields(schema, row);
900        self.signed_overlay_ids.insert(schema.id.clone());
901        true
902    }
903
904    /// Add one signature-verified catalog row without allowing the remote
905    /// publisher to redefine an exact id already owned by this build, a user
906    /// registration, or an earlier authenticated row in the same document.
907    fn register_signed_catalog_model(&mut self, schema: ModelSchema) -> bool {
908        if self.builtin_model_ids.contains(&schema.id) {
909            if self.signed_overlay_ids.contains(&schema.id) {
910                warn!(id = %schema.id, "ignoring duplicate signed overlay for compiled builtin");
911                return false;
912            }
913            return self.overlay_builtin(&schema);
914        }
915        if self.user_config_ids.contains(&schema.id) {
916            warn!(id = %schema.id, "ignoring signed catalog row for a user-registered id");
917            return false;
918        }
919        if self.project_model_ids.contains(&schema.id) {
920            warn!(id = %schema.id, "ignoring duplicate signed catalog row for project-owned exact id");
921            return false;
922        }
923        let id = schema.id.clone();
924        let registered = self.register_project_model(schema);
925        if registered {
926            self.signed_catalog_ids.insert(id);
927        }
928        registered
929    }
930
931    fn register_preserving_trust(&mut self, mut schema: ModelSchema) -> bool {
932        if let Err(error) = crate::catalog_identity::row_digest(&schema) {
933            warn!(id = %schema.id, %error, "rejecting model without canonical catalog identity");
934            return false;
935        }
936        // Check availability for local models
937        if schema.is_mlx() || schema.is_car_managed_vllm_mlx() {
938            // MLX requires Apple Silicon Metal. On any other build target —
939            // Intel Mac, Linux, Windows, or `car_skip_mlx` — the backend
940            // is not compiled in and any execution attempt would fail at
941            // dispatch with a "model not found" or backend-missing error.
942            // Mirror the AppleFoundationModels cfg-gating below: the
943            // registry must reflect that the model is *not* runnable here
944            // so the adaptive router doesn't add it to fallback chains
945            // (Parslee-ai/car#231 — §7.1, the "fresh Windows install has
946            // no usable inference path" finding). Same shape as the
947            // `refresh_availability` MLX branch.
948            #[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
949            {
950                schema.available = if schema.tags.contains(&"speech".to_string()) {
951                    speech_mlx_available()
952                } else if let ModelSource::Mlx { ref hf_repo, .. }
953                | ModelSource::ManagedVllmMlx { ref hf_repo, .. } = schema.source
954                {
955                    // A row the in-process loader runs needs the Swift MLX
956                    // stack linked into THIS binary; the target cfg above does
957                    // not imply it (`CAR_BUILD_SWIFT_LM=0`, or a Swift build
958                    // that failed and warned). A row marked available without
959                    // it is one the router puts in a fallback chain and then
960                    // fails on. Everything else under `ModelSource::Mlx` is a
961                    // subprocess and does not care — see
962                    // `mlx_row_needs_in_process_loader`.
963                    //
964                    // Given its executor, available if cached locally OR it
965                    // declares an hf_repo — ensure_local() lazy-downloads on
966                    // first use, so that is "functionally available" the same
967                    // way Ollama/RemoteApi entries are (#164).
968                    let loader_ok = !mlx_row_needs_in_process_loader(&schema)
969                        || crate::backend::local::in_process_loader_available();
970                    let mlx_dir = self.models_dir.join(&schema.name);
971                    loader_ok && (mlx_dir_has_weights(&mlx_dir) || !hf_repo.is_empty())
972                } else {
973                    let mlx_dir = self.models_dir.join(&schema.name);
974                    mlx_dir_has_weights(&mlx_dir)
975                };
976            }
977            #[cfg(not(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx))))]
978            {
979                schema.available = false;
980            }
981        } else if schema.is_vllm_mlx() {
982            // vLLM-MLX: available if endpoint env var set or was manually marked available
983            schema.available = std::env::var("VLLM_MLX_ENDPOINT").is_ok() || schema.available;
984        } else if matches!(schema.source, ModelSource::WhisperCpp { .. }) {
985            // whisper.cpp compiles on every platform and lazy-downloads its ggml
986            // model on first use, so it's functionally available everywhere
987            // (mirrors refresh_availability). Checked before the is_local() GGUF
988            // branch below — whisper is_local() too but has no model.gguf.
989            schema.available = true;
990        } else if matches!(schema.source, ModelSource::WindowsSpeech {}) {
991            // OS-provided WinRT synthesizer — available on Windows only.
992            schema.available = cfg!(target_os = "windows");
993        } else if schema.is_codex_cli() {
994            // The CLI owns auth; availability only answers whether there is a
995            // binary to attempt. Login failures surface from the child without
996            // CAR reading Codex's credential store.
997            schema.available = crate::backend::codex_cli::is_available();
998        } else if schema.is_local() {
999            let local_path = self.models_dir.join(&schema.name).join("model.gguf");
1000            // Mirrors the `ModelSource::Local` arm in `refresh_availability`:
1001            // where the Candle GGUF backend exists, a declared `hf_repo` is
1002            // functionally available because `ensure_local` fetches from it.
1003            // Registration and refresh must agree, or a model's availability
1004            // flips depending on which ran last.
1005            let lazily_fetchable = matches!(
1006                schema.source,
1007                ModelSource::Local { ref hf_repo, .. } if !hf_repo.is_empty()
1008            ) && !cfg!(all(
1009                target_os = "macos",
1010                target_arch = "aarch64",
1011                not(car_skip_mlx)
1012            ));
1013            schema.available = local_path.exists() || lazily_fetchable;
1014        } else if schema.is_remote() {
1015            // Registration is catalog work, so it may use environment presence
1016            // and non-secret authority hints but must never query a secret
1017            // backend. Request-time snapshots refresh this authoritatively.
1018            schema.available = match schema.source {
1019                ModelSource::RemoteApi {
1020                    protocol: crate::schema::ApiProtocol::OpenRouter,
1021                    ..
1022                } => crate::openrouter::credential_source().is_some(),
1023                ModelSource::RemoteApi {
1024                    ref api_key_env, ..
1025                } => environment_credential_available(api_key_env),
1026                ModelSource::Proprietary {
1027                    ref provider,
1028                    ref auth,
1029                    ..
1030                } => {
1031                    let resolved = match auth {
1032                        ProprietaryAuth::ApiKeyEnv { env_var }
1033                        | ProprietaryAuth::BearerTokenEnv { env_var } => {
1034                            std::collections::HashMap::from([(
1035                                env_var.clone(),
1036                                environment_credential_available(env_var),
1037                            )])
1038                        }
1039                        ProprietaryAuth::OAuth2Pkce { .. }
1040                        | ProprietaryAuth::ChatGptSubscription {} => Default::default(),
1041                    };
1042                    proprietary_auth_available(
1043                        &schema.id,
1044                        &schema.provider,
1045                        provider,
1046                        auth,
1047                        self.session.available(),
1048                        &resolved,
1049                    )
1050                }
1051                _ => schema.available,
1052            };
1053        }
1054        // Ready = usable without a download. For anything local that means
1055        // weights resolvable on disk *now*; `available` alone can't answer this,
1056        // since a declared `hf_repo` makes an MLX model "available" before a
1057        // byte is fetched (#164). Remote models never need a weights download.
1058        // Their live credential state belongs in `available`; coupling it to
1059        // `weights_ready` permanently excludes models registered before a key
1060        // is connected from `require_ready` routes.
1061        // See `ModelSchema::weights_ready` (Parslee-ai/car#638).
1062        // NOTE: `is_mlx()` must be tested before `is_local()` — `is_local()`
1063        // returns true for `ModelSource::Mlx` too, so the GGUF branch would
1064        // otherwise swallow every MLX model and look for a `model.gguf` that
1065        // never exists.
1066        schema.weights_ready = physical_weights_ready(&schema, &self.models_dir);
1067        // `debug`, not `info`: this fires once per catalog entry, and the
1068        // catalog is ~87 rows. At the default filter that put 87 lines of
1069        // bookkeeping in front of whatever the person actually asked for —
1070        // `car do "What is 2 plus 2?"`, `car setup`, `car doctor`.
1071        //
1072        // car#1788 quieted it per-command, which needs a new entry in
1073        // `should_quiet_tracing_for_machine_output` for every command that
1074        // builds a registry; `do` and `setup` were missed and reported as
1075        // car#1873. Registering one known model is not an event worth an
1076        // `info` line, so lowering it fixes every command at once and cannot
1077        // be missed by a new one. `RUST_LOG=debug` still shows the full set.
1078        debug!(
1079            id = %schema.id,
1080            name = %schema.name,
1081            available = schema.available,
1082            weights_ready = schema.weights_ready,
1083            "registered model"
1084        );
1085        self.models.insert(schema.id.clone(), schema);
1086        true
1087    }
1088
1089    /// Register a schema that crossed the user-controlled configuration
1090    /// boundary.
1091    ///
1092    /// This is deliberately separate from [`register`](Self::register):
1093    /// builtin, signed-catalog, and discovery rows must never become
1094    /// `models.json` rows merely because they share the same in-memory map.
1095    pub fn register_user_model(&mut self, mut schema: ModelSchema) {
1096        if crate::openrouter::is_curated_managed_gateway_alias(&schema.id) {
1097            warn!(id = %schema.id, "ignoring persisted user model for reserved Parslee-managed alias");
1098            return;
1099        }
1100        if self.project_model_ids.contains(&schema.id) {
1101            warn!(id = %schema.id, "ignoring persisted user model for project-owned exact id");
1102            return;
1103        }
1104        schema.mark_user_registered();
1105        let id = schema.id.clone();
1106        if self.register_preserving_trust(schema) {
1107            self.on_disk_discovered_ids.remove(&id);
1108            self.user_config_ids.insert(id);
1109        }
1110    }
1111
1112    /// Unregister a model by id. Returns the removed schema if found.
1113    pub fn unregister(&mut self, id: &str) -> Option<ModelSchema> {
1114        self.on_disk_discovered_ids.remove(id);
1115        let removed = self.models.remove(id);
1116        if let Some(ref m) = removed {
1117            info!(id = %m.id, "unregistered model");
1118        }
1119        removed
1120    }
1121
1122    /// Unregister an explicitly user-configured model.
1123    ///
1124    /// An untracked builtin, signed-catalog, or discovery row is not removable
1125    /// through this persistence boundary.
1126    pub fn unregister_user_model(&mut self, id: &str) -> Option<ModelSchema> {
1127        if !self.user_config_ids.remove(id) {
1128            return None;
1129        }
1130        self.unregister(id)
1131    }
1132
1133    /// List all models.
1134    pub fn list(&self) -> Vec<&ModelSchema> {
1135        let mut models: Vec<&ModelSchema> = self.models.values().collect();
1136        models.sort_by(|a, b| a.id.cmp(&b.id));
1137        models
1138    }
1139
1140    /// Query models matching a filter.
1141    pub fn query(&self, filter: &ModelFilter) -> Vec<&ModelSchema> {
1142        self.models
1143            .values()
1144            .filter(|m| {
1145                // Capability check: model must have ALL required capabilities
1146                if !filter.capabilities.iter().all(|c| m.has_capability(*c)) {
1147                    return false;
1148                }
1149                // Size check
1150                if let Some(max) = filter.max_size_mb {
1151                    if m.size_mb() > max && m.is_local() {
1152                        return false;
1153                    }
1154                }
1155                // Latency check (declared envelope)
1156                if let Some(max) = filter.max_latency_ms {
1157                    if let Some(p50) = m.performance.latency_p50_ms {
1158                        if p50 > max {
1159                            return false;
1160                        }
1161                    }
1162                }
1163                // Cost check
1164                if let Some(max) = filter.max_cost_per_mtok {
1165                    if let Some(cost) = m.cost.output_per_mtok {
1166                        if cost > max {
1167                            return false;
1168                        }
1169                    }
1170                }
1171                // Tag check
1172                if !filter.tags.iter().all(|t| m.tags.contains(t)) {
1173                    return false;
1174                }
1175                // Provider check
1176                if let Some(ref p) = filter.provider {
1177                    if &m.provider != p {
1178                        return false;
1179                    }
1180                }
1181                // Local only
1182                if filter.local_only && !m.is_local() {
1183                    return false;
1184                }
1185                // Available only
1186                if filter.available_only && !m.available_now() {
1187                    return false;
1188                }
1189                true
1190            })
1191            .collect()
1192    }
1193
1194    /// Query models by a single capability.
1195    pub fn query_by_capability(&self, cap: ModelCapability) -> Vec<&ModelSchema> {
1196        self.query(&ModelFilter {
1197            capabilities: vec![cap],
1198            ..Default::default()
1199        })
1200    }
1201
1202    /// Report installed local models with curated newer replacements.
1203    pub fn available_upgrades(&self) -> Vec<ModelUpgrade> {
1204        let mut upgrades = Vec::new();
1205        for rule in model_upgrade_rules() {
1206            let Some(from) = rule
1207                .from_ids
1208                .iter()
1209                .find_map(|id| self.models.get(id.as_str()))
1210                .filter(|schema| schema.available)
1211            else {
1212                continue;
1213            };
1214            let Some(to) = self.models.get(rule.to_id.as_str()) else {
1215                continue;
1216            };
1217            upgrades.push(ModelUpgrade {
1218                from_id: from.id.clone(),
1219                from_name: from.name.clone(),
1220                to_id: to.id.clone(),
1221                to_name: to.name.clone(),
1222                reason: rule.reason.clone(),
1223                target_runtime: rule.target_runtime.clone(),
1224                target_runtime_requirement: rule.target_runtime_requirement.clone(),
1225                minimum_runtimes: rule.minimum_runtimes.clone(),
1226                target_available: to.available,
1227                target_pullable: matches!(
1228                    to.source,
1229                    ModelSource::Local { .. } | ModelSource::Mlx { .. }
1230                ),
1231                remove_old_supported: matches!(
1232                    from.source,
1233                    ModelSource::Local { .. } | ModelSource::Mlx { .. }
1234                ) && rule.remove_old_after_available,
1235            });
1236        }
1237        upgrades.sort_by(|a, b| a.from_id.cmp(&b.from_id).then(a.to_id.cmp(&b.to_id)));
1238        upgrades.dedup_by(|a, b| a.from_id == b.from_id && a.to_id == b.to_id);
1239        upgrades
1240    }
1241
1242    /// Get a specific model by id.
1243    pub fn get(&self, id: &str) -> Option<&ModelSchema> {
1244        self.models.get(id)
1245    }
1246
1247    /// Return the currently registered schema without refreshing availability.
1248    ///
1249    /// Callers that need runtime credential/file state must explicitly use a
1250    /// refreshing snapshot; identity/provenance checks should use this exact
1251    /// stored row instead.
1252    pub fn registered_schema(&self, id: &str) -> Option<&ModelSchema> {
1253        self.get(id)
1254    }
1255
1256    /// Iterate all registered model schemas (built-in + signed + discovered).
1257    pub fn all(&self) -> impl Iterator<Item = &ModelSchema> {
1258        self.models.values()
1259    }
1260
1261    /// Find a model by name (case-insensitive). For backward compatibility
1262    /// with the old registry that used short names like "Qwen3-4B".
1263    pub fn find_by_name(&self, name: &str) -> Option<&ModelSchema> {
1264        #[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
1265        if !name.to_ascii_lowercase().ends_with("-mlx") {
1266            if let Some(mlx_variant) = self
1267                .models
1268                .values()
1269                .find(|m| m.name.eq_ignore_ascii_case(&format!("{name}-MLX")))
1270            {
1271                return Some(mlx_variant);
1272            }
1273        }
1274
1275        self.models
1276            .values()
1277            .find(|m| m.name.eq_ignore_ascii_case(name))
1278    }
1279
1280    /// On Apple Silicon, resolve a GGUF/Candle model to its MLX equivalent.
1281    /// Returns the MLX model schema if one exists with the same family and
1282    /// matching capabilities; otherwise returns None.
1283    #[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
1284    pub fn resolve_mlx_equivalent(&self, schema: &ModelSchema) -> Option<&ModelSchema> {
1285        // Already MLX — no redirect needed.
1286        if schema.is_mlx() || schema.is_vllm_mlx() {
1287            return None;
1288        }
1289        // Only redirect local GGUF models.
1290        if !matches!(schema.source, ModelSource::Local { .. }) {
1291            return None;
1292        }
1293        // Find the MLX twin: same family, SAME PARAMETER COUNT, and at least
1294        // the same primary capability. Family alone is too coarse — every
1295        // Qwen3 size shares family "qwen3", so a family-only `.find()` could
1296        // map an 8B GGUF to a 4B MLX model (whichever the HashMap yielded
1297        // first), silently swapping model size at execution and routing.
1298        // Keying on param_count makes the resolution deterministic and 1:1.
1299        let primary_cap = schema.capabilities.first()?;
1300        self.models.values().find(|m| {
1301            m.is_mlx()
1302                && m.family == schema.family
1303                && m.param_count == schema.param_count
1304                && m.capabilities.contains(primary_cap)
1305        })
1306    }
1307
1308    /// Ensure a local model is downloaded, returning its local directory path.
1309    pub async fn ensure_local(&self, id: &str) -> Result<PathBuf, InferenceError> {
1310        // Report to the ambient sink rather than dropping the events on the
1311        // floor — see [`Self::ambient_progress`] (Parslee-ai/car#620). Still a
1312        // no-op sink unless an embedder opted in, so this is not a behaviour
1313        // change for anyone who hasn't.
1314        let sink = self.ambient_progress.clone();
1315        self.ensure_local_with_progress(id, &sink).await
1316    }
1317
1318    /// Route implicit-download progress to `sink`.
1319    ///
1320    /// Applies to acquisitions nobody explicitly requested — the ones
1321    /// [`ensure_local`](Self::ensure_local) performs mid-`generate` when the
1322    /// router lands on a model that isn't on disk. Explicit pulls take their
1323    /// sink as an argument and ignore this.
1324    pub fn set_ambient_progress(&mut self, sink: ProgressSink) {
1325        self.ambient_progress = sink;
1326    }
1327
1328    /// Like [`ensure_local`](Self::ensure_local) but drives a progress sink and
1329    /// enforces the acquisition lifecycle: a per-model lock (so this can't race
1330    /// a concurrent pull / remove / upgrade of the same model), a disk-space
1331    /// preflight, and `Started`/`Completed`/`Failed` events around the work.
1332    /// When every required local file is already complete, it returns that path
1333    /// under the lock without a preflight or acquisition events.
1334    pub async fn ensure_local_with_progress(
1335        &self,
1336        id: &str,
1337        sink: &ProgressSink,
1338    ) -> Result<PathBuf, InferenceError> {
1339        self.acquire_and_ensure(id, sink, false, None).await
1340    }
1341
1342    /// Download into a caller-created private staging directory. The caller
1343    /// atomically publishes that directory only after the complete acquisition
1344    /// succeeds, so a pre-existing same-name directory can never be promoted
1345    /// into CAR ownership by a pull.
1346    pub(crate) async fn ensure_local_with_progress_staged(
1347        &self,
1348        id: &str,
1349        sink: &ProgressSink,
1350        staging_dir: &Path,
1351    ) -> Result<PathBuf, InferenceError> {
1352        self.acquire_and_ensure(id, sink, false, Some(staging_dir))
1353            .await
1354    }
1355
1356    /// Force a complete re-acquisition: skip the "reuse what's already on disk"
1357    /// short-circuits so every expected file is re-checked and any missing one
1358    /// (e.g. a blob the self-heal path purged as provably corrupt) is
1359    /// re-downloaded. The per-file checks still skip intact files, so this
1360    /// re-fetches only what's actually gone — not the whole model.
1361    pub async fn redownload_local(&self, id: &str) -> Result<PathBuf, InferenceError> {
1362        self.acquire_and_ensure(id, &ProgressSink::none(), true, None)
1363            .await
1364    }
1365
1366    async fn acquire_and_ensure(
1367        &self,
1368        id: &str,
1369        sink: &ProgressSink,
1370        force: bool,
1371        managed_dir_override: Option<&Path>,
1372    ) -> Result<PathBuf, InferenceError> {
1373        let schema = self
1374            .get(id)
1375            .or_else(|| self.find_by_name(id))
1376            .ok_or_else(|| InferenceError::ModelNotFound(id.to_string()))?;
1377        let model_name = schema.name.clone();
1378        let model_id = schema.id.clone();
1379        let needed_mb = schema.size_mb();
1380        let model_dir = self.models_dir.join(&schema.name);
1381
1382        // Serialize acquisition of this model id against other pull/remove tasks.
1383        let _guard = crate::download::acquire_model_lock(&model_id).await;
1384
1385        // Reusing complete bytes touches neither disk nor the network, so it
1386        // must precede the size-plus-reserve preflight and acquisition events.
1387        // Forced repairs and staged pulls deliberately keep the full lifecycle.
1388        if !force && managed_dir_override.is_none() {
1389            if let Some(path) = self.try_reuse_local(schema).await? {
1390                return Ok(path);
1391            }
1392        }
1393
1394        // Preflight: fail fast if there isn't room, before touching disk.
1395        if let Err(e) = crate::download::check_disk_space(&model_dir, needed_mb) {
1396            sink.emit(DownloadEvent::Failed { error: e.clone() });
1397            return Err(InferenceError::DownloadFailed(e));
1398        }
1399
1400        sink.emit(DownloadEvent::Started {
1401            model: model_name.clone(),
1402            total_files: 0,
1403            total_mb: needed_mb,
1404        });
1405        // Record, don't infer: only the files this acquisition actually
1406        // fetched — never a snapshot it reused, which may be the user's own.
1407        let (result, fetched) = crate::retire::collecting_fetches(self.ensure_local_inner(
1408            id,
1409            sink,
1410            force,
1411            managed_dir_override,
1412        ))
1413        .await;
1414        crate::retire::record_download(
1415            &self.state_root,
1416            &crate::hf_cache::hub_dir(),
1417            &model_id,
1418            &fetched,
1419        );
1420        match &result {
1421            Ok(_) => sink.emit(DownloadEvent::Completed { model: model_name }),
1422            Err(e) => sink.emit(DownloadEvent::Failed {
1423                error: e.to_string(),
1424            }),
1425        }
1426        result
1427    }
1428
1429    async fn try_reuse_local(
1430        &self,
1431        schema: &ModelSchema,
1432    ) -> Result<Option<PathBuf>, InferenceError> {
1433        match &schema.source {
1434            ModelSource::Local { .. } => {
1435                let model_dir = self.models_dir.join(&schema.name);
1436                let model_path = model_dir.join("model.gguf");
1437                let tokenizer_path = model_dir.join("tokenizer.json");
1438
1439                if crate::download::cache_file_usable(&model_path)
1440                    && crate::download::cache_file_usable(&tokenizer_path)
1441                {
1442                    return Ok(Some(model_dir));
1443                }
1444            }
1445            ModelSource::Mlx { hf_repo, .. } | ModelSource::ManagedVllmMlx { hf_repo, .. } => {
1446                let model_dir = self.models_dir.join(&schema.name);
1447                let is_diffusers = mlx_schema_is_diffusers(schema);
1448
1449                if mlx_layout_complete(is_diffusers, &model_dir) {
1450                    if auxiliary_mlx_files_missing(&schema.name, hf_repo, &model_dir) {
1451                        // Preserve managed-directory precedence: the inner path
1452                        // repairs this directory before considering a shared HF
1453                        // snapshot, just as it did before reuse moved earlier.
1454                        return Ok(None);
1455                    }
1456                    info!(model = %schema.name, path = %model_dir.display(), "using managed local MLX model");
1457                    return Ok(Some(model_dir));
1458                }
1459
1460                if let Some(snapshot_dir) = latest_huggingface_repo_snapshot(hf_repo)
1461                    .filter(|d| mlx_layout_complete(is_diffusers, d))
1462                {
1463                    if !auxiliary_mlx_files_missing(&schema.name, hf_repo, &snapshot_dir) {
1464                        info!(model = %schema.name, path = %snapshot_dir.display(), "using cached MLX snapshot");
1465                        return Ok(Some(snapshot_dir));
1466                    }
1467                }
1468            }
1469            _ => {}
1470        }
1471
1472        Ok(None)
1473    }
1474
1475    async fn ensure_local_inner(
1476        &self,
1477        id: &str,
1478        sink: &ProgressSink,
1479        force: bool,
1480        managed_dir_override: Option<&Path>,
1481    ) -> Result<PathBuf, InferenceError> {
1482        let schema = self
1483            .get(id)
1484            .or_else(|| self.find_by_name(id))
1485            .ok_or_else(|| InferenceError::ModelNotFound(id.to_string()))?;
1486
1487        match &schema.source {
1488            ModelSource::Local {
1489                hf_repo,
1490                hf_filename,
1491                tokenizer_repo,
1492            } => {
1493                let model_dir = managed_dir_override
1494                    .map(Path::to_path_buf)
1495                    .unwrap_or_else(|| self.models_dir.join(&schema.name));
1496                let model_path = model_dir.join("model.gguf");
1497                let tokenizer_path = model_dir.join("tokenizer.json");
1498
1499                if !force
1500                    && crate::download::cache_file_usable(&model_path)
1501                    && crate::download::cache_file_usable(&tokenizer_path)
1502                {
1503                    return Ok(model_dir);
1504                }
1505
1506                // A row with no `hf_repo` came from the local-directory scan:
1507                // its weights are whatever the user put on disk, and there is
1508                // nowhere to fetch the missing pieces from. Downloading with an
1509                // empty repo asks HuggingFace for `https://huggingface.co//…`
1510                // and reports whatever that returns, which explains nothing.
1511                // Say what CAR actually needs instead.
1512                if hf_repo.is_empty() {
1513                    let missing = [
1514                        ("model.gguf", &model_path),
1515                        ("tokenizer.json", &tokenizer_path),
1516                    ]
1517                    .into_iter()
1518                    .filter(|(_, path)| !crate::download::cache_file_usable(path))
1519                    .map(|(name, _)| name)
1520                    .collect::<Vec<_>>()
1521                    .join(" and ");
1522                    return Err(InferenceError::InferenceFailed(format!(
1523                        "{}: discovered on disk at {} but not loadable — the GGUF \
1524                         backend reads `model.gguf` and `tokenizer.json` from the \
1525                         model directory, and this one is missing {missing}. Rename \
1526                         the weight file to `model.gguf` and add the tokenizer, or \
1527                         register the model against its HuggingFace repo so CAR can \
1528                         fetch both.",
1529                        schema.name,
1530                        model_dir.display()
1531                    )));
1532                }
1533
1534                std::fs::create_dir_all(&model_dir)?;
1535
1536                if !crate::download::cache_file_usable(&model_path) {
1537                    info!(model = %schema.name, repo = %hf_repo, "downloading model weights");
1538                    sink.emit(DownloadEvent::FileStarted {
1539                        filename: "model weights".into(),
1540                        index: 1,
1541                        total_files: 2,
1542                        size_mb: schema.size_mb(),
1543                    });
1544                    download_file(hf_repo, hf_filename, &model_path).await?;
1545                    sink.emit(DownloadEvent::FileCompleted {
1546                        filename: "model weights".into(),
1547                    });
1548                }
1549                if !crate::download::cache_file_usable(&tokenizer_path) {
1550                    info!(model = %schema.name, repo = %tokenizer_repo, "downloading tokenizer");
1551                    sink.emit(DownloadEvent::FileStarted {
1552                        filename: "tokenizer".into(),
1553                        index: 2,
1554                        total_files: 2,
1555                        size_mb: 0,
1556                    });
1557                    download_file(tokenizer_repo, "tokenizer.json", &tokenizer_path).await?;
1558                    sink.emit(DownloadEvent::FileCompleted {
1559                        filename: "tokenizer".into(),
1560                    });
1561                }
1562
1563                Ok(model_dir)
1564            }
1565            ModelSource::Mlx {
1566                hf_repo,
1567                hf_weight_file,
1568            }
1569            | ModelSource::ManagedVllmMlx {
1570                hf_repo,
1571                hf_weight_file,
1572            } => {
1573                let model_dir = managed_dir_override
1574                    .map(Path::to_path_buf)
1575                    .unwrap_or_else(|| self.models_dir.join(&schema.name));
1576                let config_path = model_dir.join("config.json");
1577
1578                // Diffusers-layout models (Flux image-gen, LTX video) keep their
1579                // weights in component subdirs (transformer/, vae/, text_encoder/,
1580                // tokenizer/) and ship NO root config.json / top-level weight
1581                // index. The standard MLX flow below hard-fetches config.json
1582                // (404 for these repos) and a weight index (absent), so they need
1583                // the whole-repo snapshot path. Detect by capability so DOWNLOAD,
1584                // REUSE, and (via mlx_dir_has_weights) LOAD all agree on the
1585                // layout — the recurring image-gen 404 was fixing only the load
1586                // path (recurse subdirs) while download + reuse still demanded a
1587                // config.json these repos don't have.
1588                let is_diffusers = mlx_schema_is_diffusers(schema);
1589
1590                // Reuse the managed dir ONLY if it actually holds weights — a
1591                // config-only stub (or dangling-symlink install) must fall
1592                // through to the download below, not no-op (car-releases#391).
1593                // `force` (self-heal re-pull) skips reuse so a purged shard is
1594                // actually re-fetched rather than masked by the surviving shards.
1595                // Standard MLX also gates on a non-empty root config.json, the
1596                // same bar `download_file` applies per file, so a zero-byte or
1597                // dangling config is re-fetched below instead of reused;
1598                // diffusers has none, so recursive weights-present is its
1599                // completeness signal.
1600                if !force && mlx_layout_complete(is_diffusers, &model_dir) {
1601                    ensure_auxiliary_mlx_files(&schema.name, hf_repo, &model_dir).await?;
1602                    info!(model = %schema.name, path = %model_dir.display(), "using managed local MLX model");
1603                    return Ok(model_dir);
1604                }
1605
1606                // Same guard for a cached HF snapshot: a partial/pruned snapshot
1607                // whose config or weights don't resolve isn't usable —
1608                // re-download instead.
1609                if !force {
1610                    if let Some(snapshot_dir) = latest_huggingface_repo_snapshot(hf_repo)
1611                        .filter(|d| mlx_layout_complete(is_diffusers, d))
1612                    {
1613                        ensure_auxiliary_mlx_files(&schema.name, hf_repo, &snapshot_dir).await?;
1614                        info!(model = %schema.name, path = %snapshot_dir.display(), "using cached MLX snapshot");
1615                        return Ok(snapshot_dir);
1616                    }
1617                }
1618
1619                std::fs::create_dir_all(&model_dir)?;
1620
1621                info!(model = %schema.name, repo = %hf_repo, "downloading MLX model");
1622
1623                // Diffusers-layout: mirror the whole repo (component subdirs
1624                // preserved). There is no root config.json or weight index to
1625                // drive the file-by-file flow below, and demanding one is the
1626                // 404 that broke image/video generation. Verify component weights
1627                // actually landed so a half-finished pull can't masquerade as
1628                // installed (the false-availability the caller relies on).
1629                if is_diffusers {
1630                    download_repo_snapshot(hf_repo, &model_dir, sink).await?;
1631                    ensure_auxiliary_mlx_files(&schema.name, hf_repo, &model_dir).await?;
1632                    if !mlx_dir_has_weights(&model_dir) {
1633                        return Err(InferenceError::DownloadFailed(format!(
1634                            "{hf_repo}: snapshot fetched but no component weights found"
1635                        )));
1636                    }
1637                    info!(model = %schema.name, path = %model_dir.display(), "downloaded diffusers model");
1638                    return Ok(model_dir);
1639                }
1640
1641                // Download config, tokenizer, and weight files. total_files is
1642                // not known up front for MLX (single vs sharded), so events
1643                // carry total_files = 0 ("unknown") and the UI shows names.
1644                emit_file(sink, "config", 0, schema.size_mb());
1645                download_file(hf_repo, "config.json", &config_path).await?;
1646                download_tokenizer_assets(hf_repo, &model_dir, sink).await;
1647                let tok_config_path = model_dir.join("tokenizer_config.json");
1648                if !crate::download::cache_file_usable(&tok_config_path) {
1649                    let _ = download_file(hf_repo, "tokenizer_config.json", &tok_config_path).await;
1650                }
1651
1652                // Download weight files
1653                if let Some(ref wf) = hf_weight_file {
1654                    let wf_path = model_dir.join(wf);
1655                    if !crate::download::cache_file_usable(&wf_path) {
1656                        emit_file(sink, "model weights", 0, schema.size_mb());
1657                        download_file(hf_repo, wf, &wf_path).await?;
1658                    }
1659                } else {
1660                    // Try single file first, then sharded
1661                    let single = model_dir.join("model.safetensors");
1662                    if !crate::download::cache_file_usable(&single) {
1663                        emit_file(sink, "model weights", 0, schema.size_mb());
1664                        match download_file(hf_repo, "model.safetensors", &single).await {
1665                            Ok(()) => {}
1666                            Err(_) => {
1667                                // Sharded: download index and then each shard
1668                                let index_path = model_dir.join("model.safetensors.index.json");
1669                                download_file(hf_repo, "model.safetensors.index.json", &index_path)
1670                                    .await?;
1671
1672                                let index_json: serde_json::Value =
1673                                    serde_json::from_str(&std::fs::read_to_string(&index_path)?)
1674                                        .map_err(|e| {
1675                                            InferenceError::InferenceFailed(format!(
1676                                                "parse index: {e}"
1677                                            ))
1678                                        })?;
1679
1680                                if let Some(weight_map) =
1681                                    index_json.get("weight_map").and_then(|m| m.as_object())
1682                                {
1683                                    let mut files: std::collections::HashSet<String> =
1684                                        std::collections::HashSet::new();
1685                                    for filename in weight_map.values() {
1686                                        if let Some(f) = filename.as_str() {
1687                                            files.insert(f.to_string());
1688                                        }
1689                                    }
1690                                    let shard_total = files.len() as u32;
1691                                    for (i, file) in files.iter().enumerate() {
1692                                        let dest = model_dir.join(file);
1693                                        if !crate::download::cache_file_usable(&dest) {
1694                                            info!(file = %file, "downloading weight shard");
1695                                            sink.emit(DownloadEvent::FileStarted {
1696                                                filename: format!("weights part {}", i + 1),
1697                                                index: (i + 1) as u32,
1698                                                total_files: shard_total,
1699                                                size_mb: 0,
1700                                            });
1701                                            download_file(hf_repo, file, &dest).await?;
1702                                            sink.emit(DownloadEvent::FileCompleted {
1703                                                filename: format!("weights part {}", i + 1),
1704                                            });
1705                                        }
1706                                    }
1707                                }
1708                            }
1709                        }
1710                    }
1711                }
1712
1713                ensure_auxiliary_mlx_files(&schema.name, hf_repo, &model_dir).await?;
1714
1715                // Verify what we actually landed, exactly as the diffusers branch
1716                // above does (#808). Without this, a pull whose shard fetch was
1717                // interrupted returned `Ok(model_dir)` — indistinguishable from
1718                // success — and the model stayed broken until inference failed
1719                // with `load model-00001-of-00002.safetensors: Path must point to
1720                // a local file`, inside whatever job was running.
1721                //
1722                // A caller reaching for the two obvious mechanisms — check
1723                // availability, then pull — got "yes" and "ok" from both while
1724                // the model was unusable. Report the missing shards by name so
1725                // the failure is actionable at pull time.
1726                let missing = missing_weight_shards(&model_dir);
1727                if !missing.is_empty() {
1728                    return Err(InferenceError::DownloadFailed(format!(
1729                        "{}: pull finished but {} weight shard(s) are still missing: {}. \
1730                         The download was interrupted; re-run the pull to resume it.",
1731                        schema.name,
1732                        missing.len(),
1733                        missing.join(", ")
1734                    )));
1735                }
1736                if !mlx_dir_has_weights(&model_dir) {
1737                    return Err(InferenceError::DownloadFailed(format!(
1738                        "{}: pull finished but no usable weights are present under {}",
1739                        schema.name,
1740                        model_dir.display()
1741                    )));
1742                }
1743                Ok(model_dir)
1744            }
1745            _ => Err(InferenceError::InferenceFailed(format!(
1746                "model {} is not local",
1747                id
1748            ))),
1749        }
1750    }
1751
1752    /// Legacy registry-local deletion is intentionally disabled because it
1753    /// recursively removed shared Hugging Face caches without receipts or
1754    /// runtime/lease coordination. Use `InferenceEngine::remove_model_from_car`.
1755    #[deprecated(note = "use InferenceEngine::remove_model_from_car")]
1756    pub fn remove_local(&mut self, id: &str) -> Result<(), InferenceError> {
1757        Err(InferenceError::InferenceFailed(format!(
1758            "legacy registry removal for {id} is disabled; use receipt-backed model management"
1759        )))
1760    }
1761
1762    /// Refresh passive catalog availability flags for all models.
1763    ///
1764    /// Physical local/runtime probes remain live, but credential-backed rows
1765    /// use only environment presence, the non-secret Parslee authority hint,
1766    /// and the process-local Codex sign-in hint. This method is safe for
1767    /// startup, catalog, setup, and health surfaces.
1768    pub fn refresh_availability(&mut self) {
1769        let parslee_oauth_available = self.session.available();
1770        self.refresh_availability_with(
1771            parslee_oauth_available,
1772            self.session.signed_out() && self.session.may_forget_session_evidence(),
1773            false,
1774        );
1775    }
1776
1777    /// Refresh availability for an **explicit status request**, resolving
1778    /// credentials authoritatively (environment *or* OS keychain).
1779    ///
1780    /// [`Self::refresh_availability`] is deliberately passive — environment
1781    /// presence only — because it runs on startup and catalog paths where a
1782    /// physical secret-store read per provider would be wrong. That is right
1783    /// for those callers and wrong for a user who typed `car models list`:
1784    /// with a key held in `car secrets` rather than exported, the passive
1785    /// verdict reports a perfectly usable model as `RUNNABLE no`. Measured on
1786    /// a machine with `OPENAI_API_KEY` in the keychain and nothing in the
1787    /// environment, the embedded list path reported `openai/gpt-5.5` as not
1788    /// runnable while inference through the same binary worked.
1789    ///
1790    /// Use this only for surfaces a person explicitly asked for a status
1791    /// from, never on a startup or catalog path. It performs one secret-store
1792    /// read per DISTINCT credential name (deduplicated in
1793    /// `refresh_availability_with`), not one per model.
1794    pub fn refresh_availability_for_explicit_status(&mut self) {
1795        let parslee_oauth_available = self.session.available();
1796        self.refresh_availability_with(
1797            parslee_oauth_available,
1798            self.session.signed_out() && self.session.may_forget_session_evidence(),
1799            true,
1800        );
1801    }
1802
1803    /// Refresh availability for an explicit request-time routing/use path.
1804    /// `parslee_api_base` comes from Task 2's coordinator-owned credential
1805    /// snapshot; `parslee_signed_out` is true only for an authoritative absence,
1806    /// never for denial/cooldown/unreadable failures.
1807    pub(crate) fn refresh_routing_availability(
1808        &mut self,
1809        parslee_api_base: Option<&str>,
1810        parslee_signed_out: bool,
1811    ) {
1812        if let Some(api_base) = parslee_api_base {
1813            let api_base = api_base.trim_end_matches('/');
1814            for schema in self.models.values_mut() {
1815                if schema.provider.eq_ignore_ascii_case("parslee") {
1816                    if let ModelSource::Proprietary {
1817                        provider, endpoint, ..
1818                    } = &mut schema.source
1819                    {
1820                        if provider.eq_ignore_ascii_case("parslee") {
1821                            *endpoint = api_base.to_string();
1822                        }
1823                    }
1824                }
1825            }
1826        }
1827        self.refresh_availability_with(parslee_api_base.is_some(), parslee_signed_out, true);
1828    }
1829
1830    /// Request-local OAuth availability must not inherit a different organization's
1831    /// credential or gateway verdict. Other providers retain their usual checks.
1832    pub(crate) fn refresh_work_context_availability(&mut self, api_base: &str, available: bool) {
1833        self.refresh_availability();
1834        for schema in self.models.values_mut() {
1835            if schema.provider.eq_ignore_ascii_case("parslee") {
1836                if let ModelSource::Proprietary {
1837                    provider,
1838                    endpoint,
1839                    auth: ProprietaryAuth::OAuth2Pkce { .. },
1840                    ..
1841                } = &mut schema.source
1842                {
1843                    if provider.eq_ignore_ascii_case("parslee") {
1844                        *endpoint = api_base.trim_end_matches('/').to_string();
1845                        schema.available = available;
1846                    }
1847                }
1848            }
1849        }
1850    }
1851
1852    /// Whether the last authoritative refresh failed to determine this
1853    /// credential variable's state, as opposed to finding it absent.
1854    pub fn credential_state_is_unknown(&self, env_var: &str) -> bool {
1855        self.credential_state_unknown.contains(env_var)
1856    }
1857
1858    fn refresh_availability_with(
1859        &mut self,
1860        parslee_oauth_available: bool,
1861        clear_parslee_observations: bool,
1862        authoritative_credentials: bool,
1863    ) {
1864        // `models_dir` is consumed only inside the MLX and Local arms.
1865        // On non-MLX targets the MLX arm is a cfg-gated `available = false`
1866        // and never touches `models_dir`; the Local arm still needs it.
1867        let models_dir = self.models_dir.clone();
1868        // mlx-vlm CLI is the same probe call no matter which model
1869        // requires it; do it once per refresh, not per-model. On
1870        // non-MLX targets the variable is unused (the consuming arm
1871        // is cfg-gated out), so suppress the unused-variable warning.
1872        #[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
1873        let mlx_vlm_cli_present = crate::backend::mlx_vlm_cli::is_available();
1874        #[cfg(not(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx))))]
1875        #[allow(unused_variables)]
1876        let mlx_vlm_cli_present = false;
1877        // A learned "gateway has no OpenRouter upstream" is about the
1878        // environment behind THIS session. When the session is gone, so is the
1879        // evidence — otherwise a sign-in to a properly-configured org would
1880        // inherit the previous one's suppression (Parslee-ai/car#786).
1881        if clear_parslee_observations {
1882            crate::openrouter::clear_gateway_unconfigured();
1883            // Same reasoning for the credential verdict: it was learned about
1884            // the credential that is now gone, and inheriting it would suppress
1885            // the managed lanes through the next sign-in (Parslee-ai/car#887).
1886            crate::parslee_credential::clear_credential_rejected();
1887        }
1888
1889        // Resolve each DISTINCT credential once, not once per model
1890        // (car-releases#75).
1891        //
1892        // Every `RemoteApi` and `Proprietary` row used to call
1893        // `resolve_env_or_keychain` inside the loop below, and on macOS that is
1894        // a keychain query — ~14 ms each. A stock catalog is ~72 models over
1895        // roughly a dozen providers, so a refresh spent ~1 s asking the keychain
1896        // the same handful of questions dozens of times. `routing_registry_
1897        // snapshot` runs a refresh per request AND `estimated_tokens` runs
1898        // another, so a single delegated call paid it twice: ~2 s before the
1899        // request reached the runner, which is the reported latency.
1900        //
1901        // Deduplicating changes no semantics. Within ONE refresh the same env
1902        // var must yield the same answer — reading it 30 times cannot be more
1903        // correct than reading it once. Credentials are still re-read on every
1904        // refresh, so the "pasted/OAuth key changes take effect without a
1905        // restart" property above is untouched.
1906        let mut credential_envs: std::collections::BTreeSet<String> = Default::default();
1907        let mut needs_openrouter = false;
1908        let mut needs_codex = false;
1909        for m in self.models.values() {
1910            match &m.source {
1911                ModelSource::RemoteApi {
1912                    protocol: crate::schema::ApiProtocol::OpenRouter,
1913                    ..
1914                } => needs_openrouter = true,
1915                ModelSource::RemoteApi { api_key_env, .. } => {
1916                    credential_envs.insert(api_key_env.clone());
1917                }
1918                ModelSource::Proprietary { auth, .. } => match auth {
1919                    ProprietaryAuth::ApiKeyEnv { env_var }
1920                    | ProprietaryAuth::BearerTokenEnv { env_var } => {
1921                        credential_envs.insert(env_var.clone());
1922                    }
1923                    ProprietaryAuth::ChatGptSubscription {} => needs_codex = true,
1924                    ProprietaryAuth::OAuth2Pkce { .. } => {}
1925                },
1926                _ => {}
1927            }
1928        }
1929        let mut credential_unknown: std::collections::HashSet<String> =
1930            std::collections::HashSet::new();
1931        // Carried on the registry rather than on each row: `ModelSchema` is
1932        // catalog content and this is per-refresh runtime state.  See
1933        // `credential_state_unknown` for what reads it.
1934        // The Codex record is not an environment credential. Explicit status
1935        // reads check its existence once through car-auth and refresh the
1936        // process-local hint; passive catalog reads use only that hint.
1937        if authoritative_credentials && needs_codex {
1938            remember_codex_sign_in(car_auth::codex_credential_present_without_reading_it());
1939        }
1940        let credential_available: std::collections::HashMap<String, bool> = credential_envs
1941            .into_iter()
1942            .map(|env| {
1943                let available = if authoritative_credentials {
1944                    // A store failure is not absence. The row still reads
1945                    // unavailable -- conservative for a runnability claim --
1946                    // but it must not go on to advise adding a key (car#1920).
1947                    match credential_presence(&env) {
1948                        CredentialPresence::Present => true,
1949                        CredentialPresence::Absent => false,
1950                        CredentialPresence::Unknown => {
1951                            credential_unknown.insert(env.clone());
1952                            false
1953                        }
1954                    }
1955                } else {
1956                    environment_credential_available(&env)
1957                };
1958                (env, available)
1959            })
1960            .collect();
1961        self.credential_state_unknown = credential_unknown;
1962        // Only probe OpenRouter when a row actually needs it — an unnecessary
1963        // probe is the same class of waste this block exists to remove.
1964        let openrouter_available = needs_openrouter
1965            && if authoritative_credentials {
1966                crate::openrouter::refresh_credential_source().is_some()
1967            } else {
1968                crate::openrouter::credential_source().is_some()
1969            };
1970
1971        for m in self.models.values_mut() {
1972            match &m.source {
1973                ModelSource::Mlx { hf_repo, .. } | ModelSource::ManagedVllmMlx { hf_repo, .. } => {
1974                    // MLX requires Apple Silicon Metal. On any other
1975                    // build target — Intel Mac, Linux, Windows, or
1976                    // `car_skip_mlx` — the backend is not compiled in
1977                    // and the adaptive router must not select these
1978                    // models. Cfg-gate identical to the
1979                    // AppleFoundationModels branch below and the
1980                    // `register()` MLX branch above. Closes the §7.1
1981                    // arm of Parslee-ai/car#231.
1982                    #[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
1983                    {
1984                        // Models tagged `requires-mlx-vlm` shell out to the
1985                        // mlx_vlm Python CLI for image inference (#115).
1986                        // If the CLI isn't on PATH, the runtime reaches it
1987                        // anyway and bails — the registry MUST reflect that
1988                        // by marking such entries unavailable until the
1989                        // user installs `uv tool install mlx-vlm`. #137.
1990                        let needs_mlx_vlm = m.tags.iter().any(|t| t == "requires-mlx-vlm");
1991
1992                        // Only a row the in-process loader runs needs the
1993                        // Swift MLX stack linked into THIS binary; the target
1994                        // cfg does not imply it (`CAR_BUILD_SWIFT_LM=0`, or a
1995                        // Swift build that failed and warned). Both availability
1996                        // paths ask the same helper, because they are required
1997                        // to agree — a model whose `available` flips depending
1998                        // on which ran last is worse than either answer.
1999                        let in_process_loader_present = !mlx_row_needs_in_process_loader(m)
2000                            || crate::backend::local::in_process_loader_available();
2001
2002                        m.available = if needs_mlx_vlm {
2003                            mlx_vlm_cli_present
2004                        } else if m.tags.contains(&"speech".to_string()) {
2005                            speech_mlx_available()
2006                        } else {
2007                            // Available if cached locally OR has an hf_repo —
2008                            // ensure_local() lazy-downloads on first use, so a
2009                            // declared hf_repo is "functionally available" the
2010                            // same way Ollama and RemoteApi entries are
2011                            // ("should work in principle" not "physically
2012                            // cached"). Closes #164: mlx/ltx-2.3:q4 reported
2013                            // unavailable even though `car video` would just
2014                            // download-and-run successfully.
2015                            let mlx_dir = models_dir.join(&m.name);
2016                            in_process_loader_present
2017                                && (mlx_dir_has_weights(&mlx_dir) || !hf_repo.is_empty())
2018                        };
2019                    }
2020                    #[cfg(not(all(
2021                        target_os = "macos",
2022                        target_arch = "aarch64",
2023                        not(car_skip_mlx)
2024                    )))]
2025                    {
2026                        let _ = hf_repo; // unused on non-MLX targets
2027                        m.available = false;
2028                    }
2029                }
2030                ModelSource::Local {
2031                    hf_repo: local_repo,
2032                    ..
2033                } => {
2034                    let local_path = models_dir.join(&m.name).join("model.gguf");
2035                    // Where the Candle GGUF backend exists — every target that
2036                    // is not Apple-Silicon-with-MLX, i.e. the CUDA machines
2037                    // where GGUF *is* the local path — a declared `hf_repo` is
2038                    // functionally available, because `ensure_local` fetches
2039                    // `model.gguf` and `tokenizer.json` from it on first use.
2040                    // Same rule #164 gave the MLX path and whisper.cpp already
2041                    // has; this branch was left behind, so a fresh CUDA install
2042                    // reported every local model unavailable, the router
2043                    // dropped them from fallback chains, and the user was sent
2044                    // to cloud with a GPU sitting idle one download away.
2045                    //
2046                    // An empty `hf_repo` means a row from the local-directory
2047                    // scan, which has nowhere to fetch from and stays gated on
2048                    // the file actually being there.
2049                    #[cfg(not(all(
2050                        target_os = "macos",
2051                        target_arch = "aarch64",
2052                        not(car_skip_mlx)
2053                    )))]
2054                    {
2055                        m.available = local_path.exists() || !local_repo.is_empty();
2056                    }
2057                    // On Apple Silicon `backend::candle` is not compiled at
2058                    // all: a GGUF row runs only by `resolve_mlx_equivalent`
2059                    // substitution, so it stays gated on physical presence
2060                    // rather than claiming a lazy download that has no loader.
2061                    #[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
2062                    {
2063                        let _ = local_repo;
2064                        m.available = local_path.exists();
2065                    }
2066                }
2067                ModelSource::WhisperCpp { .. } => {
2068                    // whisper.cpp compiles on every platform and lazy-downloads
2069                    // its ggml model on first use (car-whisper), so it is
2070                    // functionally available everywhere — the cross-platform
2071                    // on-device STT the MLX speech models can't be off Apple.
2072                    m.available = true;
2073                }
2074                ModelSource::WindowsSpeech {} => {
2075                    // OS-provided WinRT synthesizer — available on Windows only
2076                    // (like AppleFoundationModels is Apple-only).
2077                    #[cfg(target_os = "windows")]
2078                    {
2079                        m.available = true;
2080                    }
2081                    #[cfg(not(target_os = "windows"))]
2082                    {
2083                        m.available = false;
2084                    }
2085                }
2086                ModelSource::RemoteApi {
2087                    protocol: crate::schema::ApiProtocol::OpenRouter,
2088                    ..
2089                } => {
2090                    m.available = openrouter_available;
2091                }
2092                ModelSource::RemoteApi { api_key_env, .. } => {
2093                    // env OR keychain — see `register`. Resolved once per
2094                    // refresh above, not once per model (car-releases#75).
2095                    m.available = credential_available
2096                        .get(api_key_env)
2097                        .copied()
2098                        .unwrap_or(false);
2099                }
2100                ModelSource::CodexCli { .. } => {
2101                    m.available = crate::backend::codex_cli::is_available();
2102                }
2103                ModelSource::Ollama { .. } => {
2104                    // Assume available; health check is async and done lazily
2105                    m.available = true;
2106                }
2107                ModelSource::VllmMlx { .. } => {
2108                    // External endpoints remain server-provided even on
2109                    // loopback. CAR's managed runtime says nothing about the
2110                    // health or ownership of this separately registered server.
2111                    m.available = std::env::var("VLLM_MLX_ENDPOINT").is_ok() || m.available;
2112                }
2113                ModelSource::Proprietary { provider, auth, .. } => {
2114                    m.available = proprietary_auth_available(
2115                        &m.id,
2116                        &m.provider,
2117                        provider,
2118                        auth,
2119                        parslee_oauth_available,
2120                        &credential_available,
2121                    );
2122                }
2123                ModelSource::AppleFoundationModels { .. } => {
2124                    // Apple Silicon macOS 26+ AND iOS 26+ both expose
2125                    // the FoundationModels framework. The shim's
2126                    // runtime probe handles per-device availability
2127                    // (Apple Intelligence may be off, the device may
2128                    // be pre-A17, etc.); cfg-gating here just hides
2129                    // the call on targets where the shim isn't built.
2130                    #[cfg(any(
2131                        all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)),
2132                        all(target_os = "ios", target_arch = "aarch64")
2133                    ))]
2134                    {
2135                        m.available = crate::backend::foundation_models::is_available();
2136                        // Parallel tool calls are a HOST property here, not a
2137                        // catalog one: the same shim yields one captured call
2138                        // per turn on macOS 26 and every proposed call on 27,
2139                        // because the framework's post-throw batch behaviour
2140                        // differs between them. The catalog literal stays
2141                        // conservative and the row is upgraded only where the
2142                        // behaviour was measured, so a host that cannot do it
2143                        // never advertises it.
2144                        //
2145                        // Precedent: `available` on the line above is already
2146                        // host-dependent at this same hook, so a capability
2147                        // that varies by host is not a new shape for this row.
2148                        if crate::backend::foundation_models::supports_parallel_tool_calls()
2149                            && !m.capabilities.contains(&ModelCapability::MultiToolCall)
2150                        {
2151                            m.capabilities.push(ModelCapability::MultiToolCall);
2152                        }
2153                        // Vision, same shape, one extra constraint: the claim is
2154                        // only made where the request can actually be SERVED.
2155                        // `supports_vision()` reads the model's own
2156                        // `LanguageModelCapabilities`, and the only path that
2157                        // consumes images — `generate_with_images` — is gated on
2158                        // the same probe. Claiming vision without that path is
2159                        // the 2b bug again: a route the router picks and the
2160                        // backend then fails at generation time.
2161                        if crate::backend::foundation_models::supports_vision()
2162                            && !m.capabilities.contains(&ModelCapability::Vision)
2163                        {
2164                            m.capabilities.push(ModelCapability::Vision);
2165                        }
2166                        // Prefer the framework's own answer over the catalog
2167                        // literal. `contextSize` exists from 26.4 on, so this
2168                        // is None on 26.0-26.3 and the declared value stands.
2169                        // The literal over-claiming the window is not cosmetic:
2170                        // `build_context_for_model` sizes the assembly budget
2171                        // from it, so a wrong number fails at generation time
2172                        // with the assembly already paid for.
2173                        if let Some(window) = crate::backend::foundation_models::context_size() {
2174                            m.context_length = window as usize;
2175                        }
2176                    }
2177                    #[cfg(not(any(
2178                        all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)),
2179                        all(target_os = "ios", target_arch = "aarch64")
2180                    )))]
2181                    {
2182                        m.available = false;
2183                    }
2184                }
2185                ModelSource::Delegated { .. } => {
2186                    // Availability tracks whether a runner is registered.
2187                    // Hosts call `registerInferenceRunner` (or its
2188                    // language equivalent) at startup; until then the
2189                    // model is unavailable.
2190                    m.available = crate::runner::current_inference_runner().is_some();
2191                }
2192            }
2193            m.weights_ready = physical_weights_ready(m, &models_dir);
2194        }
2195    }
2196
2197    /// Persist only explicitly user-registered models to disk.
2198    pub fn save_user_config(&self) -> Result<(), InferenceError> {
2199        let mut user_models: Vec<ModelSchema> = self
2200            .user_config_ids
2201            .iter()
2202            .filter_map(|id| self.models.get(id))
2203            .cloned()
2204            .map(|mut model| {
2205                // `models.json` is a user-controlled trust boundary regardless
2206                // of how the in-memory row was originally registered.
2207                model.mark_user_registered();
2208                model
2209            })
2210            .collect();
2211        user_models.sort_by(|a, b| a.id.cmp(&b.id));
2212
2213        for model in &user_models {
2214            crate::catalog_identity::row_digest(model).map_err(|error| {
2215                InferenceError::InferenceFailed(format!(
2216                    "refuse to persist model without canonical catalog identity: {error}"
2217                ))
2218            })?;
2219        }
2220
2221        let json = serde_json::to_string_pretty(&user_models)
2222            .map_err(|e| InferenceError::InferenceFailed(format!("serialize: {e}")))?;
2223        std::fs::write(&self.user_config_path, json)?;
2224        Ok(())
2225    }
2226
2227    /// Load user-registered models from disk.
2228    pub fn load_user_config(&mut self) -> Result<(), InferenceError> {
2229        if !self.user_config_path.exists() {
2230            return Ok(());
2231        }
2232
2233        let json = std::fs::read_to_string(&self.user_config_path)?;
2234        let models: Vec<ModelSchema> = serde_json::from_str(&json)
2235            .map_err(|e| InferenceError::InferenceFailed(format!("parse models.json: {e}")))?;
2236
2237        for m in models {
2238            // Legacy rows omit `trust_tier` and serde must keep accepting that
2239            // shape, but a manual config can never confer project curation.
2240            self.register_user_model(m);
2241        }
2242        Ok(())
2243    }
2244
2245    /// Get the models directory path.
2246    pub fn models_dir(&self) -> &Path {
2247        &self.models_dir
2248    }
2249
2250    /// Whether invoking this model can start without CAR-managed model setup.
2251    ///
2252    /// `available` deliberately includes lazy-downloadable local models so
2253    /// normal CLI inference can pull on first use. Interactive host actions
2254    /// need the stricter answer: avoid starting a large download from a
2255    /// ready-to-use assistant button and point the user at Models setup
2256    /// instead.
2257    pub fn ready_without_download(&self, id: &str) -> Option<bool> {
2258        let schema = self.get(id).or_else(|| self.find_by_name(id))?;
2259        Some(match &schema.source {
2260            ModelSource::Local { .. } => {
2261                let model_dir = self.models_dir.join(&schema.name);
2262                crate::download::cache_file_usable(&model_dir.join("model.gguf"))
2263                    && crate::download::cache_file_usable(&model_dir.join("tokenizer.json"))
2264            }
2265            ModelSource::Mlx { hf_repo, .. } | ModelSource::ManagedVllmMlx { hf_repo, .. } => {
2266                let managed_dir = self.models_dir.join(&schema.name);
2267                let managed_ready = mlx_snapshot_complete(schema, &managed_dir)
2268                    && crate::download::cache_file_usable(&managed_dir.join("tokenizer.json"));
2269                let snapshot_ready =
2270                    latest_huggingface_repo_snapshot(hf_repo).is_some_and(|snapshot| {
2271                        mlx_snapshot_complete(schema, &snapshot)
2272                            && crate::download::cache_file_usable(&snapshot.join("tokenizer.json"))
2273                    });
2274                managed_ready || snapshot_ready
2275            }
2276            ModelSource::WindowsSpeech {} => true, // OS-provided; nothing to download
2277            ModelSource::WhisperCpp { model } => {
2278                // "ready without download" = the ggml file is physically cached
2279                // at ~/.tokhn/whisper/. (Availability is looser — whisper always
2280                // lazy-downloads — but this is the strict host-action answer.)
2281                car_whisper::model_cached(model)
2282            }
2283            ModelSource::RemoteApi { .. }
2284            | ModelSource::CodexCli { .. }
2285            | ModelSource::Ollama { .. }
2286            | ModelSource::VllmMlx { .. }
2287            | ModelSource::AppleFoundationModels { .. }
2288            | ModelSource::Proprietary { .. }
2289            | ModelSource::Delegated { .. } => true,
2290        })
2291    }
2292
2293    /// Resolve a usable local artifact without downloading anything. This is
2294    /// the only source path accepted by explicit model adoption; callers do
2295    /// not get to supply a filesystem path.
2296    pub fn existing_local_artifact(&self, id: &str) -> Option<PathBuf> {
2297        let schema = self.get(id).or_else(|| self.find_by_name(id))?;
2298        let managed = self.models_dir.join(&schema.name);
2299        if std::fs::symlink_metadata(&managed).is_ok()
2300            && self.ready_without_download(&schema.id) == Some(true)
2301        {
2302            return Some(managed);
2303        }
2304        match &schema.source {
2305            ModelSource::Mlx { hf_repo, .. } | ModelSource::ManagedVllmMlx { hf_repo, .. } => {
2306                latest_huggingface_repo_snapshot(hf_repo).filter(|snapshot| {
2307                    mlx_snapshot_complete(schema, snapshot)
2308                        && crate::download::cache_file_usable(&snapshot.join("tokenizer.json"))
2309                })
2310            }
2311            _ => None,
2312        }
2313    }
2314
2315    /// Load the built-in Qwen3 catalog as ModelSchema objects.
2316    fn load_builtin_catalog(&mut self) {
2317        for schema in builtin_catalog() {
2318            let id = schema.id.clone();
2319            if self.register_project_model(schema) {
2320                self.builtin_model_ids.insert(id);
2321            }
2322        }
2323    }
2324}
2325
2326/// Build a `ModelSchema` for an uncatalogued local model directory, or
2327/// `None` if the directory isn't a recognized text-LLM checkpoint.
2328///
2329/// Recognizes two layouts: MLX (a `config.json` with a known causal-LM
2330/// `model_type` plus safetensors weights) and GGUF (a `*.gguf` weight
2331/// file). Capabilities are inferred conservatively from the directory
2332/// name — anything not positively identified as a text LLM is skipped so
2333/// speech/vision/image/video checkpoints are never mistaken for
2334/// generators. car-releases#62.
2335fn synthesize_local_schema(name: &str, dir: &Path) -> Option<ModelSchema> {
2336    let lower = name.to_ascii_lowercase();
2337
2338    // Hard skip directory names that are unambiguously non-text models —
2339    // these would be catastrophic to route as `Generate`.
2340    const NON_TEXT_HINTS: &[&str] = &[
2341        "vad",
2342        "whisper",
2343        "parakeet",
2344        "kokoro",
2345        "tts",
2346        "stt",
2347        "flux",
2348        "ltx",
2349        "yume",
2350        "sd-",
2351        "stable-diffusion",
2352        "wan",
2353        "mochi",
2354        "sana",
2355        "diffusion",
2356    ];
2357    if NON_TEXT_HINTS.iter().any(|h| lower.contains(h)) {
2358        return None;
2359    }
2360
2361    // Capabilities by name. Embedding / reranker checkpoints share the same
2362    // causal-LM architecture as generators, so the name is the only signal.
2363    let capabilities: Vec<ModelCapability> =
2364        if lower.contains("embedding") || lower.contains("embed") {
2365            vec![ModelCapability::Embed]
2366        } else if lower.contains("reranker") || lower.contains("rerank") {
2367            vec![ModelCapability::Rerank]
2368        } else {
2369            vec![
2370                ModelCapability::Generate,
2371                ModelCapability::Code,
2372                ModelCapability::Reasoning,
2373            ]
2374        };
2375
2376    // Locate weights and (for MLX) validate the architecture.
2377    let config_path = dir.join("config.json");
2378    let has_safetensors =
2379        dir.join("model.safetensors").exists() || dir.join("model.safetensors.index.json").exists();
2380
2381    let (source, context_length, quantization) = if config_path.exists() && has_safetensors {
2382        // MLX layout. Only accept a recognized causal-LM model_type.
2383        let cfg: serde_json::Value = std::fs::read_to_string(&config_path)
2384            .ok()
2385            .and_then(|s| serde_json::from_str(&s).ok())?;
2386        let model_type = cfg
2387            .get("model_type")
2388            .and_then(|v| v.as_str())
2389            .unwrap_or("")
2390            .to_ascii_lowercase();
2391        // A floor, not the gate. This answers "does this directory hold a
2392        // causal-LM checkpoint?" — a question about the *layout*, asked while
2393        // scanning a local directory on every platform, so it cannot be
2394        // delegated to a macOS-only loader. The loader's own answer is added
2395        // below, which is what lets an architecture `mlx-swift-lm` knows about
2396        // be recognized here without anyone editing this list.
2397        const KNOWN_LLM_TYPES: &[&str] = &[
2398            "qwen",
2399            "qwen2",
2400            "qwen3",
2401            "qwen3_moe",
2402            "llama",
2403            "mistral",
2404            "mixtral",
2405            "gemma",
2406            "gemma2",
2407            "gemma3",
2408            "gemma4_unified",
2409            "gemma4_unified_text",
2410            "phi",
2411            "phi3",
2412            "phimoe",
2413            "starcoder2",
2414            "deepseek",
2415            "deepseek_v2",
2416            "internlm2",
2417            "cohere",
2418            "olmo",
2419        ];
2420        let recognized = KNOWN_LLM_TYPES.iter().any(|t| model_type == *t)
2421            || crate::backend::local::has_native_backend(&model_type);
2422        if !recognized {
2423            return None;
2424        }
2425        let ctx = cfg
2426            .get("max_position_embeddings")
2427            .and_then(|v| v.as_u64())
2428            .unwrap_or(32_768) as usize;
2429        // Same block `hf_schema::quantization_of` reads, and for the same
2430        // reason: `bits` alone cannot tell an affine group quant from an MX
2431        // float, and the group size is what a kernel has to match. This used
2432        // to emit `"4-bit"` while the other ingest path emitted `"4bit"` —
2433        // two spellings of one thing, in a field nothing could compare.
2434        let quant = cfg
2435            .get("quantization")
2436            .filter(|q| q.is_object())
2437            .and_then(|q| {
2438                let bits = q
2439                    .get("bits")
2440                    .and_then(|b| b.as_u64())
2441                    .and_then(|b| u8::try_from(b).ok());
2442                let group_size = q
2443                    .get("group_size")
2444                    .and_then(|g| g.as_u64())
2445                    .and_then(|g| u32::try_from(g).ok());
2446                let mode = q.get("mode").and_then(|m| m.as_str());
2447                crate::schema::Quantization::from_mlx_config(bits, group_size, mode)
2448            });
2449        (
2450            serde_json::json!({ "type": "mlx", "hf_repo": "" }),
2451            ctx,
2452            quant,
2453        )
2454    } else {
2455        let gguf = std::fs::read_dir(dir).ok().and_then(|rd| {
2456            rd.flatten().map(|e| e.path()).find(|p| {
2457                p.extension()
2458                    .and_then(|x| x.to_str())
2459                    .is_some_and(|x| x.eq_ignore_ascii_case("gguf"))
2460            })
2461        })?;
2462        // GGUF layout (Candle). file name carried for the Local source.
2463        let filename = gguf
2464            .file_name()
2465            .and_then(|n| n.to_str())
2466            .unwrap_or("model.gguf")
2467            .to_string();
2468        // A GGUF file names its own quantization and this path used to throw
2469        // that away, so a user's own local checkpoints were the one ingest
2470        // route that recorded nothing at all.
2471        let quant = crate::schema::Quantization::from_gguf_filename(&filename);
2472        (
2473            serde_json::json!({
2474                "type": "local",
2475                "hf_repo": "",
2476                "hf_filename": filename,
2477                "tokenizer_repo": "",
2478            }),
2479            4_096,
2480            quant,
2481        )
2482    };
2483
2484    let id = format!("local/{}", lower.replace(['/', ' '], "-"));
2485    serde_json::from_value(serde_json::json!({
2486        "id": id,
2487        "name": name,
2488        "provider": "local",
2489        "family": "local",
2490        "capabilities": capabilities,
2491        "context_length": context_length,
2492        "quantization": quantization,
2493        "source": source,
2494        "tags": ["auto-discovered"],
2495        "trust_tier": "community",
2496    }))
2497    .ok()
2498}
2499
2500/// Whether an MLX-sourced speech model can actually run: the managed
2501/// `mlx-audio` runtime has to be provisioned.
2502///
2503/// Platform-independent, and it used to answer an unconditional `true` on Apple
2504/// Silicon on the grounds that "speech uses native MLX backends — no Python CLI
2505/// needed". Those backends (`mlx_kokoro`, `mlx_parakeet`) were deleted with the
2506/// Swift migration, so that branch promised a path that no longer exists —
2507/// which is the same false `available: true` car#649 was filed about, reached
2508/// from the other side. The managed runtime is now the only way an
2509/// `mlx/kokoro-*`, `mlx/parakeet-*` or `mlx/qwen3-tts-*` entry runs anywhere.
2510///
2511/// Only `ModelSource::Mlx` / `ManagedVllmMlx` rows reach this; the cross-
2512/// platform `WhisperCpp` STT entry, Windows OS TTS and the remote ElevenLabs
2513/// rows are filtered out by the caller's `is_mlx()` guard and keep their own
2514/// availability.
2515// Both call sites sit inside `cfg(macos + aarch64 + not(car_skip_mlx))` blocks;
2516// the else-arm marks every MLX row unavailable without asking, because there is
2517// no MLX there to ask about. So this is genuinely dead off Apple Silicon.
2518#[allow(dead_code)]
2519fn speech_mlx_available() -> bool {
2520    // Via `venv_program`, so this agrees with `SpeechRuntime` about where a
2521    // provisioned runtime keeps its console scripts — `Scripts\*.exe` on
2522    // Windows, `bin/*` elsewhere.
2523    let runtime_root = speech_runtime_root();
2524    crate::managed_venv::venv_program(&runtime_root, "mlx_audio.stt.generate").exists()
2525        || crate::managed_venv::venv_program(&runtime_root, "mlx_audio.tts.generate").exists()
2526}
2527
2528// Dead off Apple Silicon for the same reason as its only caller above.
2529#[allow(dead_code)]
2530fn speech_runtime_root() -> PathBuf {
2531    if let Ok(path) = std::env::var("CAR_SPEECH_RUNTIME_DIR") {
2532        if !path.trim().is_empty() {
2533            return PathBuf::from(path);
2534        }
2535    }
2536    std::env::var_os("HOME")
2537        .or_else(|| std::env::var_os("USERPROFILE"))
2538        .map(PathBuf::from)
2539        .unwrap_or_else(|| PathBuf::from("."))
2540        .join(".car")
2541        .join("speech-runtime")
2542}
2543
2544/// Backward-compatible ModelInfo for listing (used by CLI and old callers).
2545#[derive(Debug, Clone, Serialize, Deserialize)]
2546pub struct ModelInfo {
2547    pub id: String,
2548    pub name: String,
2549    pub provider: String,
2550    pub capabilities: Vec<ModelCapability>,
2551    pub param_count: String,
2552    pub size_mb: u64,
2553    pub context_length: usize,
2554    pub available: bool,
2555    pub is_local: bool,
2556    /// True only for a raw `VllmMlx` source whose runtime is operated outside
2557    /// CAR. The endpoint may be loopback or remote; its address is not
2558    /// ownership evidence. CAR-owned `ManagedVllmMlx` remains a local row.
2559    #[serde(default)]
2560    pub operator_managed_external_runtime: bool,
2561    /// Weights resolvable on disk *now* — i.e. usable without a download.
2562    /// Distinct from `available`, which for a lazy-downloading MLX entry is
2563    /// true before a byte is fetched (#164). Remote models need no weights
2564    /// and report `true`. See `ModelSchema::weights_ready` (#638).
2565    ///
2566    /// `#[serde(default)]` so a client built against this version can still
2567    /// parse a catalog from a daemon that predates the field.
2568    #[serde(default)]
2569    pub weights_ready: bool,
2570    /// Whether CAR fetches weights to disk for this entry at all — i.e.
2571    /// whether `weights_ready` is a question with a meaningful answer.
2572    ///
2573    /// `false` for OS-provided models (`windows/speech-synthesis:os`,
2574    /// `apple/foundation:default`), for server-backed local models
2575    /// (`vllm-mlx/*`, Ollama) and for every remote entry. `is_local` is NOT a
2576    /// substitute: the first four of those are local and still download
2577    /// nothing. Consumers should render no install status when this is
2578    /// `false`. See `ModelSchema::downloads_weights` (#894).
2579    ///
2580    /// `#[serde(default)]` so a client built against this version can still
2581    /// parse a catalog from a daemon that predates the field.
2582    #[serde(default)]
2583    pub downloads_weights: bool,
2584    /// Per-model maximum OUTPUT tokens (registry-declared). `None` when the
2585    /// catalog entry omits it; callers fall back to a fraction of
2586    /// `context_length` via `ModelSchema::effective_max_output()`.
2587    #[serde(default)]
2588    pub max_output_tokens: Option<usize>,
2589    /// Public benchmark scores carried straight through from `ModelSchema`.
2590    /// The built-in catalog ships this empty; populating it is a curation
2591    /// step (see `BenchmarkScore` in the schema for shape and conventions).
2592    #[serde(default)]
2593    pub public_benchmarks: Vec<crate::schema::BenchmarkScore>,
2594    /// Declared per-token prices carried straight through from
2595    /// `ModelSchema::cost` — USD per 1M tokens for uncached input, output,
2596    /// cache reads and cache writes, plus any prompt-size pricing tiers the
2597    /// catalog declares.
2598    ///
2599    /// Every price component is `Option`: a local model that declares no
2600    /// prices reports them as `null`, which is deliberately **not** the same
2601    /// as a declared `0.0`. Consumers that price a request must distinguish
2602    /// "free" from "unpriced" — collapsing the two is how a caller ends up
2603    /// publishing a fabricated $0.00.
2604    ///
2605    /// `#[serde(default)]` so a newly built client can parse a catalog from
2606    /// an older daemon that predates this field: the row deserializes with an
2607    /// all-`None` cost rather than failing the whole response.
2608    #[serde(default)]
2609    pub cost: crate::schema::CostModel,
2610    #[serde(default = "default_true")]
2611    pub car_enabled: bool,
2612    #[serde(default)]
2613    pub can_remove: bool,
2614    #[serde(default)]
2615    pub in_use: bool,
2616    #[serde(default)]
2617    pub management_evidence: Option<String>,
2618    /// Whether this row fits the machine that answered, judged by the same
2619    /// rule `models.recommend` partitions on: the cold-load estimate at the
2620    /// recommendation context against the active resource policy's ceiling.
2621    /// `fits` for remote, delegated and operator-managed rows, whose memory is
2622    /// not this machine's. Computed per request from `HardwareInfo::detect()`
2623    /// and the engine's active policy; never persisted.
2624    ///
2625    /// `#[serde(default)]` (→ `unknown`) so a client built against this
2626    /// version can parse a catalog from a daemon that predates the field, and
2627    /// so an absent verdict never reads as too-big.
2628    #[serde(default)]
2629    pub fit: crate::recommend::ModelFitStatus,
2630    /// The cold-load peak `fit` was judged against, in MB. `null` when the
2631    /// row's memory is not this machine's or it declares no size/RAM figure.
2632    #[serde(default)]
2633    pub estimated_peak_mb: Option<u64>,
2634    /// Whether this machine's accelerator can run the row at all — `false`
2635    /// for a Metal-only row (MLX, managed vLLM-MLX, Apple FoundationModels)
2636    /// on a non-Apple machine, the rows `models.recommend` never offers.
2637    /// Defaults to `true` for the same older-daemon reason as `fit`.
2638    #[serde(default = "default_true")]
2639    pub platform_compatible: bool,
2640    /// Carried through from `ModelSchema::deprecated`: superseded in the
2641    /// catalog. The row stays listed (an installed deprecated model is still
2642    /// usable); clients lead with current models by sorting or hiding on it.
2643    #[serde(default)]
2644    pub deprecated: bool,
2645    /// The credential this row is waiting on, when a missing credential is the
2646    /// ONLY reason it is unavailable.
2647    ///
2648    /// `Some(env_name)` means: this is a remote row, it is not available, and
2649    /// the single thing standing between the user and using it is that named
2650    /// credential. `None` means everything else — the row is available, or it
2651    /// is local, or it is unavailable for a reason a credential will not fix
2652    /// (deprecated, a gateway with no upstream, the managed
2653    /// `parslee/openrouter/*` rows whose availability is a session-plus-upstream
2654    /// question rather than a key).
2655    ///
2656    /// **The daemon answers this so the client never has to guess it.** The
2657    /// obvious client-side shortcut — "a remote row with `available: false`
2658    /// probably needs a key" — is wrong for every case in the previous
2659    /// paragraph, and the protocol already tells clients not to reconstruct
2660    /// state from `available` alone. The alternative, a provider-to-variable
2661    /// map in the UI, duplicates the catalog that owns `api_key_env`.
2662    ///
2663    /// Naming the variable is not a disclosure: `ANTHROPIC_API_KEY` is the
2664    /// public name of an environment variable, carries no value, and is
2665    /// already printed by `car models list` and every provider's own docs.
2666    ///
2667    /// **Only ever populated on an explicit status read**, never on a passive
2668    /// one. A passive refresh resolves credentials from the process
2669    /// environment alone, so a key held in the OS secret store reads as absent
2670    /// — and a row naming a credential the user already has, telling them to
2671    /// add it, is the one piece of advice worse than none. The field's
2672    /// contract is "a missing credential is the SOLE reason this row is
2673    /// unavailable"; a passive refresh cannot establish that, so it does not
2674    /// claim it. `available` is left alone and remains passive-accurate on
2675    /// that path.
2676    ///
2677    /// Serialized as `null` rather than omitted, matching every other optional
2678    /// field on this row (`management_evidence`, `estimated_peak_mb`). The row
2679    /// has a fixed key set, and `tests/models_surface_gate.rs` asserts it
2680    /// exactly — a conditionally-present key would make that gate
2681    /// unmaintainable and hand clients a shape that changes per row.
2682    #[serde(default)]
2683    pub credential_required: Option<String>,
2684    /// Actionable provider setup guidance; absent when no specific remedy is known.
2685    #[serde(default)]
2686    pub unavailable_reason: Option<String>,
2687    /// Model line for grouping (`qwen3`, `gemma-4`), published for local rows
2688    /// only. Remote rows report `null` because this response deliberately
2689    /// carries no upstream identifier for the managed `parslee/openrouter/*`
2690    /// aliases, whose `family` names the upstream model; `models.search`
2691    /// publishes it for every row.
2692    #[serde(default)]
2693    pub family: Option<String>,
2694    /// Version or checkpoint label within the family, local rows only for
2695    /// the same reason as `family` (an alias's `version` is the upstream
2696    /// snapshot).
2697    #[serde(default)]
2698    pub version: Option<String>,
2699}
2700
2701/// Explain why a listed model is unavailable when CAR has one actionable,
2702/// provider-specific answer.
2703pub fn model_unavailable_reason(info: &ModelInfo) -> Option<&'static str> {
2704    (!info.available && info.provider.eq_ignore_ascii_case(OPENAI_CODEX_PROVIDER))
2705        .then_some(crate::schema::OPENAI_CODEX_SIGN_IN_HINT)
2706}
2707
2708fn default_true() -> bool {
2709    true
2710}
2711
2712impl ModelInfo {
2713    /// Stamp a computed fit annotation onto the row. Kept as one call so no
2714    /// caller can publish `fit` without the estimate and platform flag it
2715    /// was judged with.
2716    pub fn with_fit(mut self, fit: crate::recommend::ModelFit) -> Self {
2717        self.fit = fit.fit;
2718        self.estimated_peak_mb = fit.estimated_peak_mb;
2719        self.platform_compatible = fit.platform_compatible;
2720        self
2721    }
2722}
2723
2724impl From<&ModelSchema> for ModelInfo {
2725    fn from(s: &ModelSchema) -> Self {
2726        let mut info = ModelInfo {
2727            id: s.id.clone(),
2728            name: s.name.clone(),
2729            provider: s.provider.clone(),
2730            capabilities: s.capabilities.clone(),
2731            param_count: s.param_count.clone(),
2732            size_mb: s.size_mb(),
2733            context_length: s.context_length,
2734            available: s.available_now(),
2735            is_local: s.is_local(),
2736            operator_managed_external_runtime: matches!(s.source, ModelSource::VllmMlx { .. }),
2737            weights_ready: s.weights_ready,
2738            downloads_weights: s.downloads_weights(),
2739            max_output_tokens: s.max_output_tokens,
2740            public_benchmarks: s.public_benchmarks.clone(),
2741            // Prices only — the managed `parslee/…` aliases carry the same
2742            // `CostModel` as the personal row they front, and a `CostModel`
2743            // holds no identifiers, so publishing it discloses nothing about
2744            // the upstream model id.
2745            cost: s.cost.clone(),
2746            car_enabled: true,
2747            can_remove: false,
2748            in_use: false,
2749            management_evidence: None,
2750            // Not a verdict: the machine and policy are the engine's to
2751            // supply. `list_models_unified` stamps the real annotation with
2752            // `with_fit`; an unannotated projection reads as "unknown", the
2753            // value that hides nothing.
2754            fit: crate::recommend::ModelFitStatus::Unknown,
2755            estimated_peak_mb: None,
2756            platform_compatible: true,
2757            credential_required: s.credential_required(),
2758            unavailable_reason: None,
2759            deprecated: s.deprecated,
2760            // Local rows only — see the field docs for the non-disclosure
2761            // rule on managed aliases.
2762            family: s.is_local().then(|| s.family.clone()),
2763            version: s.is_local().then(|| s.version.clone()),
2764        };
2765        info.unavailable_reason = model_unavailable_reason(&info).map(str::to_owned);
2766        info
2767    }
2768}
2769
2770/// Emit a lightweight `FileStarted` marker for an MLX auxiliary/weight file.
2771/// MLX pulls don't know `total_files` up front (single vs sharded), so these
2772/// carry `total_files = 0` ("unknown") and the UI shows the file name. The
2773/// sharded weight loop, which knows its count, pairs Started/Completed itself.
2774fn emit_file(sink: &ProgressSink, name: &str, index: u32, size_mb: u64) {
2775    sink.emit(DownloadEvent::FileStarted {
2776        filename: name.to_string(),
2777        index,
2778        total_files: 0,
2779        size_mb,
2780    });
2781}
2782
2783/// Download a single file from a HuggingFace repo.
2784/// Mirror every file in a HuggingFace repo into `model_dir`, preserving its
2785/// subdirectory structure. For diffusers-layout models (Flux/LTX) whose weights
2786/// live in component subdirs (`transformer/`, `vae/`, `text_encoder/`, …) with
2787/// no root `config.json` or top-level weight index — the file-by-file MLX flow
2788/// can't enumerate them, so the whole snapshot is fetched. The repo file list
2789/// comes from the HF models API (version-independent, unlike hf-hub's `info()`);
2790/// metadata (`.gitattributes`, dotfiles, markdown) is skipped.
2791async fn download_repo_snapshot(
2792    repo: &str,
2793    model_dir: &Path,
2794    sink: &ProgressSink,
2795) -> Result<(), InferenceError> {
2796    #[derive(serde::Deserialize)]
2797    struct RepoInfo {
2798        siblings: Vec<Sibling>,
2799    }
2800    #[derive(serde::Deserialize)]
2801    struct Sibling {
2802        rfilename: String,
2803    }
2804    let url = format!("https://huggingface.co/api/models/{repo}");
2805    // Not `Client::new()`: that is `build().expect(..)`, which panics when the
2806    // OS trust store loads zero valid certificates — the failure the ladder in
2807    // `tls_client` exists to survive. huggingface.co is a public endpoint, so
2808    // the public-CA rung serves this call fully.
2809    let info: RepoInfo = crate::tls_client::model_download_client()
2810        .get(&url)
2811        .send()
2812        .await
2813        .map_err(|e| InferenceError::DownloadFailed(format!("list {repo}: {e}")))?
2814        .error_for_status()
2815        .map_err(|e| InferenceError::DownloadFailed(format!("list {repo}: {e}")))?
2816        .json()
2817        .await
2818        .map_err(|e| InferenceError::DownloadFailed(format!("parse {repo} file list: {e}")))?;
2819
2820    let files: Vec<String> = info
2821        .siblings
2822        .into_iter()
2823        .map(|s| s.rfilename)
2824        .filter(|f| !f.starts_with('.') && !f.to_ascii_lowercase().ends_with(".md"))
2825        .collect();
2826    if files.is_empty() {
2827        return Err(InferenceError::DownloadFailed(format!(
2828            "{repo}: repo lists no downloadable files"
2829        )));
2830    }
2831
2832    let total = files.len() as u32;
2833    for (i, fname) in files.iter().enumerate() {
2834        let dest = model_dir.join(fname);
2835        if crate::download::cache_file_usable(&dest) {
2836            continue;
2837        }
2838        if let Some(parent) = dest.parent() {
2839            std::fs::create_dir_all(parent)?;
2840        }
2841        sink.emit(DownloadEvent::FileStarted {
2842            filename: fname.clone(),
2843            index: (i + 1) as u32,
2844            total_files: total,
2845            size_mb: 0,
2846        });
2847        download_file(repo, fname, &dest).await?;
2848        sink.emit(DownloadEvent::FileCompleted {
2849            filename: fname.clone(),
2850        });
2851    }
2852    Ok(())
2853}
2854
2855/// Tokenizer filenames seen across the repos CAR pulls, in preference order.
2856///
2857/// There is no single convention, which is the whole point of this list:
2858///
2859/// | Layout | Files | Example |
2860/// |---|---|---|
2861/// | `tokenizers` library | `tokenizer.json` | most modern text MLX repos |
2862/// | GPT-2 BPE pair | `vocab.json`, `merges.txt` | `Qwen3-TTS-12Hz-1.7B-Base-5bit` |
2863/// | SentencePiece | `tokenizer.model`, `tokenizer.vocab`, `vocab.txt` | `parakeet-tdt-0.6b-v3` |
2864/// | none at all | — | `Kokoro-82M-*` (phoneme-based, needs no tokenizer) |
2865const TOKENIZER_FILENAMES: &[&str] = &[
2866    "tokenizer.json",
2867    "vocab.json",
2868    "merges.txt",
2869    "tokenizer.model",
2870    "tokenizer.vocab",
2871    "vocab.txt",
2872];
2873
2874/// Fetch whatever tokenizer assets a repo actually has. **Never fails the pull
2875/// over their absence.**
2876///
2877/// This used to be a single `download_file(repo, "tokenizer.json", …).await?` —
2878/// an unconditional hard requirement that aborted the entire multi-gigabyte
2879/// acquisition when the file 404'd. That made every curated speech model
2880/// impossible to install, in either direction: both Kokoro TTS builds, the
2881/// Qwen3-TTS default, and the Parakeet STT default, so local speech did not work
2882/// at all on a clean machine (Parslee-ai/car#639).
2883///
2884/// Best-effort is the correct posture here, not a workaround. The downloader
2885/// cannot know what a given backend needs — and Kokoro settles it: it ships
2886/// **no** tokenizer files whatsoever, because a phoneme-based TTS has no use for
2887/// one. Any rule of the form "a tokenizer must be present" is therefore wrong
2888/// for some model CAR legitimately supports. A loader that genuinely needs a
2889/// tokenizer will fail at load with a precise, model-specific error, which beats
2890/// a blanket download-time abort that names a file the model was never going to
2891/// have. This also matches the treatment `tokenizer_config.json` already gets
2892/// on the very next line.
2893async fn download_tokenizer_assets(hf_repo: &str, model_dir: &Path, sink: &ProgressSink) {
2894    // Already satisfied by any recognized layout? Then don't re-probe the API.
2895    if TOKENIZER_FILENAMES
2896        .iter()
2897        .any(|f| crate::download::cache_file_usable(&model_dir.join(f)))
2898    {
2899        return;
2900    }
2901    emit_file(sink, "tokenizer", 0, 0);
2902    let mut fetched: Vec<&str> = Vec::new();
2903    for name in TOKENIZER_FILENAMES {
2904        let dest = model_dir.join(name);
2905        if crate::download::cache_file_usable(&dest) {
2906            continue;
2907        }
2908        if download_file(hf_repo, name, &dest).await.is_ok() {
2909            fetched.push(name);
2910        }
2911    }
2912    if fetched.is_empty() {
2913        // Not an error: see the doc comment — Kokoro has none by design.
2914        tracing::debug!(
2915            repo = %hf_repo,
2916            "no tokenizer assets in this repo; continuing (the backend may not need one)"
2917        );
2918    } else {
2919        tracing::debug!(repo = %hf_repo, files = ?fetched, "fetched tokenizer assets");
2920    }
2921}
2922
2923async fn download_file(repo: &str, filename: &str, dest: &Path) -> Result<(), InferenceError> {
2924    // Do not even consult the remote/cache when a prior install is usable.
2925    // Conversely, bare existence is not enough: zero-byte files and dangling
2926    // symlinks must be repaired rather than accepted forever.
2927    if crate::download::cache_file_usable(dest) {
2928        return Ok(());
2929    }
2930
2931    let api = crate::hf_cache::api()
2932        .build()
2933        .map_err(|e| InferenceError::DownloadFailed(e.to_string()))?;
2934
2935    // hf-hub's `get` returns an already-cached file without fetching; only a
2936    // file the cache did not hold is one CAR downloaded.
2937    let already_cached = hf_hub::Cache::new(crate::hf_cache::hub_dir())
2938        .repo(hf_hub::Repo::model(repo.to_string()))
2939        .get(filename)
2940        .is_some();
2941    let repo_id = repo;
2942    let repo = api.model(repo.to_string());
2943    let path = repo
2944        .get(filename)
2945        .await
2946        .map_err(|e| InferenceError::DownloadFailed(format!("{filename}: {e}")))?;
2947    if !already_cached {
2948        crate::retire::note_fetched(repo_id, filename, crate::retire::snapshot_revision(&path));
2949    }
2950
2951    install_fetched_file(&path, dest)
2952}
2953
2954/// Install an already-fetched cache file without consulting the network.
2955fn install_fetched_file(src: &Path, dest: &Path) -> Result<(), InferenceError> {
2956    if crate::download::cache_file_usable(dest) {
2957        return Ok(());
2958    }
2959
2960    match std::fs::symlink_metadata(dest) {
2961        Ok(_) => std::fs::remove_file(dest).map_err(|e| {
2962            InferenceError::DownloadFailed(format!(
2963                "remove unusable destination {}: {e}",
2964                dest.display()
2965            ))
2966        })?,
2967        Err(error) if error.kind() == std::io::ErrorKind::NotFound => {}
2968        Err(error) => {
2969            return Err(InferenceError::DownloadFailed(format!(
2970                "inspect destination {}: {error}",
2971                dest.display()
2972            )));
2973        }
2974    }
2975
2976    // Symlink creation publishes one complete directory entry. If the
2977    // platform/filesystem does not support it, copy beside the destination and
2978    // rename only after the complete source bytes are present.
2979    #[cfg(unix)]
2980    {
2981        if std::os::unix::fs::symlink(src, dest).is_ok() {
2982            return Ok(());
2983        }
2984    }
2985
2986    static INSTALL_SEQUENCE: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0);
2987    let sequence = INSTALL_SEQUENCE.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2988    let file_name = dest
2989        .file_name()
2990        .and_then(|name| name.to_str())
2991        .unwrap_or("download");
2992    let temp = dest.with_file_name(format!(
2993        ".{file_name}.car-install-{}-{sequence}.tmp",
2994        std::process::id()
2995    ));
2996    let _ = std::fs::remove_file(&temp);
2997    std::fs::copy(src, &temp).map_err(|error| {
2998        InferenceError::DownloadFailed(format!(
2999            "copy to temporary destination {}: {error}",
3000            temp.display()
3001        ))
3002    })?;
3003    if let Err(error) = std::fs::rename(&temp, dest) {
3004        let _ = std::fs::remove_file(&temp);
3005        return Err(InferenceError::DownloadFailed(format!(
3006            "publish downloaded file at {}: {error}",
3007            dest.display()
3008        )));
3009    }
3010    Ok(())
3011}
3012
3013/// The Flux checkpoint whose T5 tokenizer the MLX Flux row fetches from its base
3014/// model: `(mlx repo, mlx model name, base repo, file)`. One definition for the
3015/// downloader here and for [`crate::retire`]'s ownership, which must know that
3016/// CAR takes one file from the base repo and owns nothing else in it.
3017pub(crate) const FLUX_AUXILIARY: (&str, &str, &str, &str) = (
3018    "mlx-community/Flux-1.lite-8B-MLX-Q4",
3019    "Flux-1.lite-8B-MLX-Q4",
3020    "Freepik/flux.1-lite-8B",
3021    "tokenizer_2/tokenizer.json",
3022);
3023
3024/// The text encoder every LTX video generation loads. CAR passes it to the
3025/// `ltx-2-mlx` runtime explicitly rather than inheriting the runtime's default,
3026/// so what CAR's video rows fetch is stated here — for the runtime, and for
3027/// [`crate::retire`], which would otherwise see an 8 GB repo nothing references
3028/// and report it as the user's.
3029pub(crate) const LTX_TEXT_ENCODER: &str = "mlx-community/gemma-3-12b-it-4bit";
3030
3031/// Whether `schema` generates video through the `ltx-2-mlx` runtime — every
3032/// MLX video row does.
3033pub(crate) fn uses_ltx_text_encoder(schema: &ModelSchema) -> bool {
3034    matches!(schema.source, ModelSource::Mlx { .. })
3035        && schema.has_capability(ModelCapability::VideoGeneration)
3036}
3037
3038pub(crate) fn uses_flux_auxiliary(model_name: &str, hf_repo: &str) -> bool {
3039    hf_repo == FLUX_AUXILIARY.0 || model_name == FLUX_AUXILIARY.1
3040}
3041
3042fn auxiliary_mlx_files_missing(model_name: &str, hf_repo: &str, model_dir: &Path) -> bool {
3043    uses_flux_auxiliary(model_name, hf_repo)
3044        && !crate::download::cache_file_usable(
3045            &model_dir.join("tokenizer_2").join("tokenizer.json"),
3046        )
3047}
3048
3049async fn ensure_auxiliary_mlx_files(
3050    model_name: &str,
3051    hf_repo: &str,
3052    model_dir: &Path,
3053) -> Result<(), InferenceError> {
3054    if auxiliary_mlx_files_missing(model_name, hf_repo, model_dir) {
3055        let t5_tokenizer_path = model_dir.join("tokenizer_2").join("tokenizer.json");
3056        std::fs::create_dir_all(
3057            t5_tokenizer_path
3058                .parent()
3059                .ok_or_else(|| InferenceError::InferenceFailed("invalid tokenizer path".into()))?,
3060        )?;
3061        info!(
3062            path = %t5_tokenizer_path.display(),
3063            "downloading missing Flux tokenizer_2/tokenizer.json from base model"
3064        );
3065        download_file(FLUX_AUXILIARY.2, FLUX_AUXILIARY.3, &t5_tokenizer_path).await?;
3066    }
3067    Ok(())
3068}
3069
3070fn mlx_auxiliary_ready_without_download(model_name: &str, model_dir: &Path) -> bool {
3071    if model_name == "Flux-1.lite-8B-MLX-Q4" {
3072        return crate::download::cache_file_usable(
3073            &model_dir.join("tokenizer_2").join("tokenizer.json"),
3074        );
3075    }
3076    true
3077}
3078
3079/// The single physical-readiness projection used by registration, refresh,
3080/// recommendation/setup state, and ready-without-download checks.
3081fn physical_weights_ready(schema: &ModelSchema, models_dir: &Path) -> bool {
3082    physical_weights_ready_with_huggingface_hub(schema, models_dir, None)
3083}
3084
3085pub(crate) fn physical_weights_ready_with_huggingface_hub(
3086    schema: &ModelSchema,
3087    models_dir: &Path,
3088    huggingface_hub_root: Option<&Path>,
3089) -> bool {
3090    match &schema.source {
3091        ModelSource::Mlx { hf_repo, .. } | ModelSource::ManagedVllmMlx { hf_repo, .. } => {
3092            let managed_dir = models_dir.join(&schema.name);
3093            if mlx_snapshot_complete(schema, &managed_dir) {
3094                return true;
3095            }
3096            let shared_snapshot = match huggingface_hub_root {
3097                Some(root) => latest_huggingface_repo_snapshot_in(&crate::hf_cache::repo_dir_in(
3098                    root, hf_repo,
3099                )),
3100                None => latest_huggingface_repo_snapshot(hf_repo),
3101            };
3102            shared_snapshot
3103                .as_deref()
3104                .is_some_and(|snapshot| mlx_snapshot_complete(schema, snapshot))
3105        }
3106        ModelSource::WhisperCpp { model } => car_whisper::model_cached(model),
3107        ModelSource::Local { .. } => {
3108            crate::download::cache_file_usable(&models_dir.join(&schema.name).join("model.gguf"))
3109        }
3110        // These sources have no CAR-downloaded artifact. `weights_ready` is a
3111        // readiness sentinel for legacy routing; callers must gate install
3112        // language with `downloads_weights()`.
3113        ModelSource::WindowsSpeech {}
3114        | ModelSource::AppleFoundationModels { .. }
3115        | ModelSource::VllmMlx { .. }
3116        | ModelSource::Ollama { .. }
3117        | ModelSource::RemoteApi { .. }
3118        | ModelSource::CodexCli { .. }
3119        | ModelSource::Proprietary { .. }
3120        | ModelSource::Delegated { .. } => true,
3121    }
3122}
3123
3124#[cfg(test)]
3125fn mlx_weights_ready_at(
3126    schema: &ModelSchema,
3127    managed_dir: &Path,
3128    shared_snapshot: Option<&Path>,
3129) -> bool {
3130    mlx_snapshot_complete(schema, managed_dir)
3131        || shared_snapshot.is_some_and(|snapshot| mlx_snapshot_complete(schema, snapshot))
3132}
3133
3134/// A complete MLX location, whether CAR's managed directory or a shared
3135/// Hugging Face snapshot. Standard text models need config + all indexed
3136/// shards. Diffusers layouts have no root config, so their recursively
3137/// validated component weights are the completeness authority. Tokenizer
3138/// readiness is a stricter invocation concern handled by
3139/// `ready_without_download`; `weights_ready` remains literal physical weights.
3140fn mlx_snapshot_complete(schema: &ModelSchema, dir: &Path) -> bool {
3141    mlx_layout_complete(mlx_schema_is_diffusers(schema), dir)
3142        && mlx_auxiliary_ready_without_download(&schema.name, dir)
3143}
3144
3145/// Whether a row stores its weights in the diffusers component layout (Flux
3146/// image generation, LTX video): component subdirectories with no root
3147/// `config.json` and no top-level weight index.
3148fn mlx_schema_is_diffusers(schema: &ModelSchema) -> bool {
3149    schema.capabilities.iter().any(|capability| {
3150        matches!(
3151            capability,
3152            ModelCapability::ImageGeneration | ModelCapability::VideoGeneration
3153        )
3154    })
3155}
3156
3157/// Root metadata plus weights: the one layout test shared by every reuse
3158/// gate and by the installed-state projection, so the two can never disagree
3159/// about the same directory. A standard MLX directory needs a `config.json`
3160/// that resolves and is non-empty (`cache_file_usable`, the bar each
3161/// per-file download applies) and every weight file present; a zero-byte or
3162/// dangling config is therefore re-fetched rather than reused. Diffusers
3163/// layouts have no root config, so their recursively validated component
3164/// weights are the whole signal.
3165fn mlx_layout_complete(is_diffusers: bool, dir: &Path) -> bool {
3166    (is_diffusers || crate::download::cache_file_usable(&dir.join("config.json")))
3167        && mlx_dir_has_weights(dir)
3168}
3169
3170/// Whether a `ModelSource::Mlx` row is executed by the **in-process** Swift
3171/// loader, as opposed to one of the subprocesses that also read MLX weights.
3172///
3173/// `ModelSource::Mlx` is a statement about the weight format, not about who
3174/// runs it. Five executors share it today: the in-process text loader, the
3175/// managed `mlx-audio` venv (speech), `mflux` (image), `ltx-2-mlx` (video) and
3176/// the `mlx_vlm` Python CLI. Only the first needs the Swift stack linked into
3177/// this binary, and gating the other four on it marked working subprocess
3178/// models unavailable — `car image` and `car video` lose their only local
3179/// backend on a machine whose Swift build merely failed.
3180///
3181/// Both availability paths call this, because they are required to agree: a
3182/// model whose `available` flips depending on whether registration or refresh
3183/// ran last is worse than either answer.
3184// Both call sites sit inside `cfg(macos + aarch64 + not(car_skip_mlx))`; the
3185// else-arm marks every MLX row unavailable without asking, because there is no
3186// MLX there to ask about. So this is genuinely dead off Apple Silicon — as the
3187// `cargo check --cfg=car_skip_mlx` gate in ci-local.sh reported the moment it
3188// was added.
3189#[allow(dead_code)]
3190fn mlx_row_needs_in_process_loader(schema: &ModelSchema) -> bool {
3191    use crate::schema::ModelCapability as C;
3192    let tagged = |t: &str| schema.tags.iter().any(|x| x == t);
3193    if tagged("speech") || tagged("requires-mlx-vlm") {
3194        return false;
3195    }
3196    if schema
3197        .capabilities
3198        .iter()
3199        .any(|c| matches!(c, C::ImageGeneration | C::VideoGeneration))
3200    {
3201        return false;
3202    }
3203    matches!(schema.source, ModelSource::Mlx { .. })
3204}
3205
3206/// Does a managed MLX model dir actually contain resolvable weights?
3207///
3208/// A dir holding only `config.json`/`tokenizer.json` stubs — or a dangling
3209/// `*.safetensors` symlink into a pruned HuggingFace cache — is NOT a complete
3210/// install. Treating it as installed makes `ensure_local`/`car models pull`
3211/// no-op and inference then fails at load with "no safetensors weights found"
3212/// (car-releases#391). We check for at least one `*.safetensors` whose target
3213/// resolves: `Path::exists()` follows symlinks, so a dangling link counts as
3214/// absent. This covers single-file (`model.safetensors`) and sharded
3215/// (`model-00001-of-0000N.safetensors`) layouts; the bare
3216/// `model.safetensors.index.json` (extension `.json`) correctly doesn't count.
3217///
3218/// **Sharded models are checked against their index, not by existence of any
3219/// one shard.** "At least one `*.safetensors` resolves" is the right question
3220/// for a single-file model and the wrong one for a sharded install: an
3221/// interrupted download that landed shard 2 of 2 satisfies it, so
3222/// `ensure_local` returns early without repairing, `car models pull` reports
3223/// success having done nothing, and the failure surfaces only at load —
3224/// `load model-00001-of-00002.safetensors: Path must point to a local file`
3225/// (Parslee-ai/car#808). The index enumerates every required shard, so when it
3226/// is present it is authoritative and cheap to consult.
3227pub(crate) fn mlx_dir_has_weights(dir: &Path) -> bool {
3228    // An index present at the top level means a sharded layout: require EVERY
3229    // shard it names. Absent (single-file, or a diffusers-layout model whose
3230    // components live in subdirs), fall back to the recursive any-weights walk.
3231    let index = dir.join("model.safetensors.index.json");
3232    if index.is_file() {
3233        return sharded_weight_files(&index).is_some_and(|required| {
3234            !required.is_empty()
3235                && required
3236                    .iter()
3237                    .all(|shard| crate::download::cache_file_usable(&dir.join(shard)))
3238        });
3239    }
3240    mlx_dir_has_weights_depth(dir, 0)
3241}
3242
3243/// Weight shards the index requires but that are not present on disk (#808).
3244///
3245/// Empty means "no shard filename could be reported" — including for
3246/// single-file/diffusers layouts and malformed indexes. Callers must pair this
3247/// with [`mlx_dir_has_weights`]; an empty diagnostic is never proof that the
3248/// model is complete.
3249///
3250/// Exists so a failed pull can say *which* shard is absent. The reported case
3251/// (#808) left a 6.7 GB model with one of two shards on disk for weeks, and the
3252/// only symptom was an inference-time `Path must point to a local file` naming
3253/// a shard the caller had no reason to know about.
3254pub(crate) fn missing_weight_shards(dir: &Path) -> Vec<String> {
3255    let index = dir.join("model.safetensors.index.json");
3256    if !index.is_file() {
3257        return Vec::new();
3258    }
3259    let Some(required) = sharded_weight_files(&index) else {
3260        // Unreadable index: there is no trustworthy shard filename to report.
3261        // Readiness still fails closed in `mlx_dir_has_weights`.
3262        return Vec::new();
3263    };
3264    required
3265        .into_iter()
3266        .filter(|shard| !dir.join(shard).exists())
3267        .collect()
3268}
3269
3270/// Shard filenames an MLX/HF `model.safetensors.index.json` says are required.
3271///
3272/// `weight_map` maps every tensor name to the file holding it, so the distinct
3273/// values are exactly the shard set. Returns `None` when the index cannot be
3274/// read or has no `weight_map`; readiness callers must fail closed when an
3275/// index exists but this metadata cannot be validated.
3276fn sharded_weight_files(index: &Path) -> Option<Vec<String>> {
3277    let raw = std::fs::read_to_string(index).ok()?;
3278    let parsed: serde_json::Value = serde_json::from_str(&raw).ok()?;
3279    let map = parsed.get("weight_map")?.as_object()?;
3280    let mut files: Vec<String> = map
3281        .values()
3282        .filter_map(|v| v.as_str().map(str::to_string))
3283        .collect();
3284    files.sort();
3285    files.dedup();
3286    Some(files)
3287}
3288
3289/// Recurse into component subdirs to find weights, but bounded. Diffusers-layout
3290/// models (Flux image, LTX video, …) keep their component weights in subdirs
3291/// (`transformer/`, `vae/`, `text_encoder/`, …), not at the top level, so a flat
3292/// scan misjudges a fully-cached snapshot as weightless (the miss that sent the
3293/// image/video path into a re-download that 404s on the `config.json` these
3294/// repos don't ship). `load_all_tensors` reads those component safetensors
3295/// recursively. The walk is depth-capped and skips symlinked directories so a
3296/// symlink cycle in a cache dir can't stack-overflow us — the HF cache nests
3297/// real component dirs at most ~2 deep, with only leaf blob files symlinked
3298/// (those are files, resolved by `cache_file_usable`, not followed here).
3299fn mlx_dir_has_weights_depth(dir: &Path, depth: usize) -> bool {
3300    if depth > 4 {
3301        return false;
3302    }
3303    let Ok(rd) = std::fs::read_dir(dir) else {
3304        return false;
3305    };
3306    rd.flatten().any(|e| {
3307        let p = e.path();
3308        let is_symlink = std::fs::symlink_metadata(&p)
3309            .map(|m| m.file_type().is_symlink())
3310            .unwrap_or(true);
3311        if p.is_dir() {
3312            !is_symlink && mlx_dir_has_weights_depth(&p, depth + 1)
3313        } else {
3314            // `cache_file_usable` follows the symlink and requires non-empty, so
3315            // a dangling weight symlink (pruned blob) or a zero-length partial
3316            // write does not count as installed.
3317            p.extension().and_then(|x| x.to_str()) == Some("safetensors")
3318                && crate::download::cache_file_usable(&p)
3319        }
3320    })
3321}
3322
3323#[allow(dead_code)] // conditionally compiled — used only on MLX-backend (macOS) snapshot-resolution paths
3324fn huggingface_repo_has_snapshot(repo_id: &str) -> bool {
3325    latest_huggingface_repo_snapshot(repo_id).is_some()
3326}
3327
3328fn resolve_huggingface_ref_snapshot(repo_dir: &Path, name: &str) -> Option<PathBuf> {
3329    let sha = std::fs::read_to_string(repo_dir.join("refs").join(name))
3330        .ok()?
3331        .trim()
3332        .to_string();
3333    if sha.is_empty() {
3334        return None;
3335    }
3336
3337    let snapshot = repo_dir.join("snapshots").join(sha);
3338    if snapshot_looks_ready(&snapshot) {
3339        Some(snapshot)
3340    } else {
3341        None
3342    }
3343}
3344
3345pub(crate) fn latest_huggingface_repo_snapshot(repo_id: &str) -> Option<PathBuf> {
3346    let repo_dir = crate::hf_cache::repo_dir(repo_id);
3347    latest_huggingface_repo_snapshot_in(&repo_dir)
3348}
3349
3350fn latest_huggingface_repo_snapshot_in(repo_dir: &Path) -> Option<PathBuf> {
3351    if let Some(snapshot) = resolve_huggingface_ref_snapshot(repo_dir, "main") {
3352        return Some(snapshot);
3353    }
3354
3355    let snapshots = repo_dir.join("snapshots");
3356    let mut candidates: Vec<(SystemTime, PathBuf)> = std::fs::read_dir(snapshots)
3357        .ok()?
3358        .filter_map(Result::ok)
3359        .map(|e| e.path())
3360        .filter(|p| p.is_dir() && snapshot_looks_ready(p))
3361        .map(|path| {
3362            let modified = path
3363                .metadata()
3364                .and_then(|metadata| metadata.modified())
3365                .unwrap_or(SystemTime::UNIX_EPOCH);
3366            (modified, path)
3367        })
3368        .collect();
3369    candidates.sort();
3370    candidates.pop().map(|(_, path)| path)
3371}
3372
3373fn snapshot_looks_ready(path: &Path) -> bool {
3374    if path.join("config.json").exists() || path.join("model_index.json").exists() {
3375        return true;
3376    }
3377    snapshot_contains_ext(path, "safetensors")
3378}
3379
3380fn snapshot_contains_ext(root: &Path, ext: &str) -> bool {
3381    let Ok(entries) = std::fs::read_dir(root) else {
3382        return false;
3383    };
3384    entries.filter_map(Result::ok).any(|entry| {
3385        let path = entry.path();
3386        if path.is_dir() {
3387            snapshot_contains_ext(&path, ext)
3388        } else {
3389            let ext_matches = path
3390                .extension()
3391                .and_then(|value| value.to_str())
3392                .map(|value| value.eq_ignore_ascii_case(ext))
3393                .unwrap_or(false);
3394            // A matching extension only counts when the file is actually usable
3395            // — a dangling symlink into a pruned blob or a zero-length partial
3396            // must not make a snapshot look ready.
3397            ext_matches && crate::download::cache_file_usable(&path)
3398        }
3399    })
3400}
3401
3402/// Built-in catalog parsed from `builtin_catalog.json`.
3403///
3404/// Adding, removing, or editing a model is a JSON-only change — Rust
3405/// source stays put. The JSON is embedded at compile time via
3406/// `include_str!`, parsed once into a `LazyLock`, and cloned on each
3407/// call. A malformed JSON file fails the integration test
3408/// `builtin_catalog_json_parses` so the binary never ships unable
3409/// to load its own catalog.
3410const BUILTIN_CATALOG_JSON: &str = include_str!("builtin_catalog.json");
3411
3412static BUILTIN_CATALOG: std::sync::LazyLock<Vec<ModelSchema>> = std::sync::LazyLock::new(|| {
3413    serde_json::from_str(BUILTIN_CATALOG_JSON)
3414        .expect("builtin_catalog.json failed to parse — fix the JSON, not this code")
3415});
3416
3417pub(crate) fn builtin_catalog() -> Vec<ModelSchema> {
3418    let mut catalog = BUILTIN_CATALOG.clone();
3419    catalog.extend(crate::openrouter::builtin_schemas());
3420    catalog
3421}
3422
3423/// Credential environment names declared by CAR's built-in model catalog.
3424///
3425/// This is a value-free catalog projection for child-process scrubbing. It
3426/// does not construct a live registry, probe credentials, read caches, or
3427/// perform discovery I/O.
3428pub fn builtin_credential_env_names() -> std::collections::BTreeSet<String> {
3429    let mut names = std::collections::BTreeSet::new();
3430    for model in builtin_catalog() {
3431        match model.source {
3432            ModelSource::RemoteApi {
3433                api_key_env,
3434                api_key_envs,
3435                ..
3436            } => {
3437                names.insert(api_key_env);
3438                names.extend(api_key_envs);
3439            }
3440            ModelSource::Proprietary { auth, .. } => match auth {
3441                ProprietaryAuth::ApiKeyEnv { env_var }
3442                | ProprietaryAuth::BearerTokenEnv { env_var } => {
3443                    names.insert(env_var);
3444                }
3445                ProprietaryAuth::OAuth2Pkce { .. } | ProprietaryAuth::ChatGptSubscription {} => {}
3446            },
3447            ModelSource::Local { .. }
3448            | ModelSource::CodexCli { .. }
3449            | ModelSource::Ollama { .. }
3450            | ModelSource::Mlx { .. }
3451            | ModelSource::WhisperCpp { .. }
3452            | ModelSource::WindowsSpeech {}
3453            | ModelSource::VllmMlx { .. }
3454            | ModelSource::ManagedVllmMlx { .. }
3455            | ModelSource::AppleFoundationModels { .. }
3456            | ModelSource::Delegated { .. } => {}
3457        }
3458    }
3459    names
3460}
3461
3462/// Report whether a built-in model is paid for by a ChatGPT subscription.
3463///
3464/// Some callers hold only the durable model id, not a [`UnifiedRegistry`]. In
3465/// particular, the coder A/B harness reads a completed session after the
3466/// registry that ran it may be gone. Giving those callers a narrow catalog
3467/// lookup lets them render `subscription` instead of inventing a dollar cost,
3468/// without starting a registry merely to recover billing provenance.
3469pub fn is_builtin_subscription_billed(model_id: &str) -> bool {
3470    let latest_id = (!model_id.contains(':')).then(|| format!("{model_id}:latest"));
3471    builtin_catalog().into_iter().any(|model| {
3472        (model.id == model_id || latest_id.as_deref() == Some(model.id.as_str()))
3473            && model.is_subscription_billed()
3474    })
3475}
3476
3477/// Test support for consumers that need real built-in rows without consulting
3478/// the developer machine's shared Hugging Face cache.
3479#[doc(hidden)]
3480pub fn builtin_catalog_with_huggingface_hub_for_testing(
3481    models_dir: &Path,
3482    huggingface_hub_root: &Path,
3483) -> Vec<ModelSchema> {
3484    let mut catalog = builtin_catalog();
3485    for schema in &mut catalog {
3486        schema.weights_ready = physical_weights_ready_with_huggingface_hub(
3487            schema,
3488            models_dir,
3489            Some(huggingface_hub_root),
3490        );
3491    }
3492    catalog
3493}
3494
3495#[cfg(test)]
3496mod tests {
3497    use crate::openrouter::StateRootScope;
3498
3499    /// A sharded install missing one shard is NOT installed.
3500    ///
3501    /// The interrupted-download shape from car#808: shard 2 of 2 landed, shard
3502    /// 1 did not. The old any-weights check said "installed", so `ensure_local`
3503    /// returned early without repairing, `pull_model` reported success having
3504    /// done nothing, and the failure surfaced only at load. Synthetic dir — no
3505    /// real weights needed, which is what makes it runnable in CI.
3506    #[test]
3507    fn a_sharded_model_missing_one_shard_is_not_installed() {
3508        let tmp = tempfile::tempdir().unwrap();
3509        let dir = tmp.path();
3510        std::fs::write(
3511            dir.join("model.safetensors.index.json"),
3512            r#"{"weight_map":{"a":"model-00001-of-00002.safetensors",
3513                              "b":"model-00002-of-00002.safetensors"}}"#,
3514        )
3515        .unwrap();
3516        // Only shard 2 present — exactly the reported state.
3517        std::fs::write(dir.join("model-00002-of-00002.safetensors"), b"x").unwrap();
3518        assert!(
3519            !mlx_dir_has_weights(dir),
3520            "a missing shard must read as not-installed, or pull silently no-ops"
3521        );
3522
3523        // Completing the set flips it.
3524        std::fs::write(dir.join("model-00001-of-00002.safetensors"), b"x").unwrap();
3525        assert!(
3526            mlx_dir_has_weights(dir),
3527            "a complete shard set must read as installed"
3528        );
3529    }
3530
3531    /// A failed pull must be able to NAME the shard it is missing (car#808).
3532    ///
3533    /// The reported symptom was an inference-time
3534    /// `load model-00001-of-00002.safetensors: Path must point to a local file`
3535    /// — a filename the caller had no way to anticipate, surfacing inside a
3536    /// benchmarking run rather than at pull time. `missing_weight_shards` is
3537    /// what lets the pull itself say which shard is absent.
3538    #[test]
3539    fn missing_shards_are_reported_by_name() {
3540        let tmp = tempfile::tempdir().unwrap();
3541        let dir = tmp.path();
3542        std::fs::write(
3543            dir.join("model.safetensors.index.json"),
3544            r#"{"weight_map":{"a":"model-00001-of-00002.safetensors",
3545                              "b":"model-00002-of-00002.safetensors"}}"#,
3546        )
3547        .unwrap();
3548        std::fs::write(dir.join("model-00002-of-00002.safetensors"), b"x").unwrap();
3549
3550        assert_eq!(
3551            missing_weight_shards(dir),
3552            vec!["model-00001-of-00002.safetensors".to_string()],
3553            "the absent shard must be named, not just counted"
3554        );
3555
3556        std::fs::write(dir.join("model-00001-of-00002.safetensors"), b"x").unwrap();
3557        assert!(
3558            missing_weight_shards(dir).is_empty(),
3559            "a complete shard set must report nothing missing"
3560        );
3561    }
3562
3563    /// No index (single-file or diffusers layout) means there is no shard list
3564    /// to check against — report nothing missing rather than inventing a
3565    /// failure. `mlx_dir_has_weights` remains the completeness signal there.
3566    #[test]
3567    fn missing_shards_is_empty_without_an_index() {
3568        let tmp = tempfile::tempdir().unwrap();
3569        std::fs::write(tmp.path().join("model.safetensors"), b"x").unwrap();
3570        assert!(missing_weight_shards(tmp.path()).is_empty());
3571
3572        // An unreadable index is a metadata problem, not proof of a missing
3573        // shard — same call `mlx_dir_has_weights` makes, so they cannot disagree.
3574        let bad = tempfile::tempdir().unwrap();
3575        std::fs::write(bad.path().join("model.safetensors.index.json"), b"not json").unwrap();
3576        assert!(missing_weight_shards(bad.path()).is_empty());
3577    }
3578
3579    /// A single-file model has no index; the any-weights walk still governs.
3580    #[test]
3581    fn a_single_file_model_still_counts_without_an_index() {
3582        let tmp = tempfile::tempdir().unwrap();
3583        std::fs::write(tmp.path().join("model.safetensors"), b"x").unwrap();
3584        assert!(mlx_dir_has_weights(tmp.path()));
3585    }
3586
3587    /// A present but unreadable index is authoritative evidence we cannot
3588    /// validate, so readiness fails closed even if a stray weight file exists.
3589    #[test]
3590    fn an_unparseable_index_fails_closed() {
3591        let tmp = tempfile::tempdir().unwrap();
3592        std::fs::write(
3593            tmp.path().join("model.safetensors.index.json"),
3594            b"{not-json",
3595        )
3596        .unwrap();
3597        std::fs::write(tmp.path().join("model.safetensors"), b"x").unwrap();
3598        assert!(
3599            !mlx_dir_has_weights(tmp.path()),
3600            "an unreadable index must not fall back to a stray weight"
3601        );
3602    }
3603
3604    use super::*;
3605    use tempfile::TempDir;
3606
3607    #[test]
3608    fn mlx_dir_has_weights_detects_completeness() {
3609        let tmp = TempDir::new().unwrap();
3610        let dir = tmp.path();
3611
3612        // A config-only stub is NOT complete (car-releases#391).
3613        std::fs::write(dir.join("config.json"), "{}").unwrap();
3614        std::fs::write(dir.join("tokenizer.json"), "{}").unwrap();
3615        assert!(
3616            !mlx_dir_has_weights(dir),
3617            "config-only stub must not count as installed"
3618        );
3619
3620        // The bare sharded index without shards is still incomplete.
3621        std::fs::write(dir.join("model.safetensors.index.json"), "{}").unwrap();
3622        assert!(!mlx_dir_has_weights(dir), "index.json alone is not weights");
3623
3624        // A real single-file layout has no shard index.
3625        std::fs::remove_file(dir.join("model.safetensors.index.json")).unwrap();
3626        std::fs::write(dir.join("model.safetensors"), b"\x00\x01\x02").unwrap();
3627        assert!(mlx_dir_has_weights(dir));
3628    }
3629
3630    #[test]
3631    fn mlx_dir_has_weights_handles_sharded_and_dangling_symlinks() {
3632        let sharded = TempDir::new().unwrap();
3633        std::fs::write(sharded.path().join("config.json"), "{}").unwrap();
3634        std::fs::write(
3635            sharded.path().join("model-00001-of-00002.safetensors"),
3636            b"\x00",
3637        )
3638        .unwrap();
3639        assert!(mlx_dir_has_weights(sharded.path()), "sharded shard counts");
3640
3641        // A DANGLING *.safetensors symlink (into a pruned HF cache) must NOT
3642        // count — Path::exists() follows the link and returns false.
3643        #[cfg(unix)]
3644        {
3645            let dangling = TempDir::new().unwrap();
3646            std::fs::write(dangling.path().join("config.json"), "{}").unwrap();
3647            std::os::unix::fs::symlink(
3648                dangling.path().join("does-not-exist"),
3649                dangling.path().join("model.safetensors"),
3650            )
3651            .unwrap();
3652            assert!(
3653                !mlx_dir_has_weights(dangling.path()),
3654                "dangling weight symlink must count as absent"
3655            );
3656        }
3657    }
3658
3659    /// A registry over a temp state root, bundled with the two things that
3660    /// must outlive it: the temp dir, and this crate's environment lock.
3661    ///
3662    /// Holding the lock is not optional, and RETURNING it rather than binding
3663    /// it inside the helper is the whole point — a guard dropped at the end of
3664    /// `test_registry()` protects nothing. Every `UnifiedRegistry::new*` ends
3665    /// in `refresh_availability()`, which clears the process-global gateway
3666    /// observation AND `$CAR_HOME/gateway-state.json` whenever no Parslee
3667    /// token is present — exactly a CI runner's state. Two tests in this
3668    /// module (`a_gateway_that_reports_no_upstream_stops_being_advertised`,
3669    /// `a_rejected_credential_stops_the_managed_lane_being_advertised`) assert
3670    /// on that same observation, so ANY unserialized construction in this
3671    /// module can fail them. #989 locked the two direct
3672    /// `refresh_availability()` call sites but not the constructor path; this
3673    /// closes it. `cargo nextest` hides the family by giving every test its
3674    /// own process, which is why it stayed hidden until `shared-process-test`
3675    /// put a threaded `cargo test` on the per-PR path — see
3676    /// docs/solutions/process-global-state-in-tests.md.
3677    ///
3678    /// Modelled on `adaptive_router::tests::TestRegistry`, which already has
3679    /// this shape for the same reason.
3680    struct TestRegistry {
3681        registry: UnifiedRegistry,
3682        // Field drop order is intentional: tear the temp state root down
3683        // before releasing the lock, so the next test never sees a half-gone
3684        // `CAR_HOME`-shaped directory.
3685        _tmp: TempDir,
3686        _environment: tokio::sync::MutexGuard<'static, ()>,
3687    }
3688
3689    impl std::ops::Deref for TestRegistry {
3690        type Target = UnifiedRegistry;
3691
3692        fn deref(&self) -> &Self::Target {
3693            &self.registry
3694        }
3695    }
3696
3697    impl std::ops::DerefMut for TestRegistry {
3698        fn deref_mut(&mut self) -> &mut Self::Target {
3699            &mut self.registry
3700        }
3701    }
3702
3703    fn test_registry() -> TestRegistry {
3704        let _environment = crate::openrouter::test_environment_scope();
3705        let tmp = TempDir::new().unwrap();
3706        let registry = UnifiedRegistry::new_with_state_root(
3707            tmp.path().to_path_buf(),
3708            tmp.path().join("models"),
3709        );
3710        TestRegistry {
3711            registry,
3712            _tmp: tmp,
3713            _environment,
3714        }
3715    }
3716
3717    /// **`credential_required` must only ever name a credential that would
3718    /// actually fix the row.**
3719    ///
3720    /// The field exists so a client never has to infer "this needs a key"
3721    /// from `available: false`. That inference is wrong for several real
3722    /// shapes, and saying "add ANTHROPIC_API_KEY" to someone whose row is
3723    /// deprecated, OAuth-backed, or gated on a gateway upstream is worse than
3724    /// saying nothing — it sends them to do something that cannot help.
3725    /// car#1920. A secret store that ERRORS is not a store that said "absent",
3726    /// and the difference decides whether the model screen tells someone to add
3727    /// a credential they may already hold. The registry records the
3728    /// indeterminate case so the projection can stay silent; the row itself
3729    /// remains unavailable, which is the conservative direction.
3730    #[test]
3731    fn an_unreadable_credential_is_recorded_as_unknown_not_absent() {
3732        // The default registry has run no authoritative refresh, so it claims
3733        // nothing is unknown -- the value that lets `credential_required`
3734        // behave exactly as before wherever this state is not established.
3735        let registry = UnifiedRegistry::new_empty(std::path::PathBuf::from("/nonexistent"));
3736        assert!(
3737            !registry.credential_state_is_unknown("VENDOR_API_KEY"),
3738            "an unprobed registry must not claim a credential state is unknown"
3739        );
3740
3741        // And the three-way probe keeps the distinction the bool could not:
3742        // a name that cannot exist in any store is ABSENT, not unknown, so a
3743        // genuinely missing key still gets its "add this key" advice.
3744        assert_eq!(
3745            credential_presence("CAR_TEST_CREDENTIAL_THAT_DOES_NOT_EXIST"),
3746            CredentialPresence::Absent
3747        );
3748    }
3749
3750    #[test]
3751    fn credential_required_names_a_key_only_when_a_key_is_the_answer() {
3752        let remote = |available: bool, deprecated: bool| {
3753            let mut m = test_generate_schema(
3754                "vendor/m",
3755                "m",
3756                ModelSource::RemoteApi {
3757                    protocol: crate::schema::ApiProtocol::OpenAiCompat,
3758                    endpoint: "https://example.invalid/v1".into(),
3759                    api_key_env: "VENDOR_API_KEY".into(),
3760                    api_key_envs: vec![],
3761                    api_version: None,
3762                },
3763            );
3764            m.provider = "vendor".into();
3765            m.available = available;
3766            m.deprecated = deprecated;
3767            m
3768        };
3769
3770        assert_eq!(
3771            remote(false, false).credential_required().as_deref(),
3772            Some("VENDOR_API_KEY"),
3773            "an unavailable plain remote row is exactly the case this field is for"
3774        );
3775
3776        assert_eq!(
3777            remote(true, false).credential_required(),
3778            None,
3779            "an available row needs nothing"
3780        );
3781        assert_eq!(
3782            remote(false, true).credential_required(),
3783            None,
3784            "a deprecated row does not become usable by adding a key"
3785        );
3786
3787        // Several alternative variables: telling someone to set one of four
3788        // names is not actionable, and choosing for them guesses which
3789        // account they hold.
3790        let mut multi = remote(false, false);
3791        if let ModelSource::RemoteApi {
3792            ref mut api_key_envs,
3793            ..
3794        } = multi.source
3795        {
3796            *api_key_envs = vec!["VENDOR_A".into(), "VENDOR_B".into()];
3797        }
3798        assert_eq!(multi.credential_required(), None);
3799
3800        // OpenRouter resolves through a credential SOURCE (pasted key or
3801        // OAuth), not a variable a person types.
3802        let mut router = remote(false, false);
3803        router.source = ModelSource::RemoteApi {
3804            protocol: crate::schema::ApiProtocol::OpenRouter,
3805            endpoint: "https://openrouter.ai/api/v1".into(),
3806            api_key_env: "OPENROUTER_API_KEY".into(),
3807            api_key_envs: vec![],
3808            api_version: None,
3809        };
3810        assert_eq!(router.credential_required(), None);
3811
3812        // A local row has no credential at all.
3813        let mut local = test_generate_schema(
3814            "mlx/local",
3815            "local",
3816            ModelSource::Mlx {
3817                hf_repo: "org/repo".into(),
3818                hf_weight_file: None,
3819            },
3820        );
3821        local.available = false;
3822        assert_eq!(local.credential_required(), None);
3823    }
3824
3825    fn test_generate_schema(id: &str, name: &str, source: ModelSource) -> ModelSchema {
3826        ModelSchema {
3827            id: id.into(),
3828            name: name.into(),
3829            provider: "local".into(),
3830            family: "qwen3".into(),
3831            version: "test".into(),
3832            capabilities: vec![ModelCapability::Generate],
3833            context_length: 4096,
3834            max_output_tokens: None,
3835            param_count: String::new(),
3836            quantization: None,
3837            performance: PerformanceEnvelope::default(),
3838            cost: CostModel::default(),
3839            source,
3840            tags: vec![],
3841            supported_params: vec![],
3842            public_benchmarks: vec![],
3843            trust_tier: crate::schema::TrustTier::Curated,
3844            deprecated: false,
3845            available: false,
3846            weights_ready: false,
3847        }
3848    }
3849
3850    /// Parslee-ai/car#894: the schema knows whether weights are on disk, but
3851    /// the schema → `ModelInfo` projection dropped that fact, so the CLI could
3852    /// only render `available` and contradicted `car doctor`. The projection
3853    /// must carry `weights_ready` through in both directions.
3854    #[test]
3855    fn model_info_carries_weights_ready_through_the_projection() {
3856        let mut schema = test_generate_schema(
3857            "mlx-community/car894-test-4bit",
3858            "car894-test-4bit",
3859            ModelSource::Mlx {
3860                hf_repo: "mlx-community/car894-test-4bit".into(),
3861                hf_weight_file: None,
3862            },
3863        );
3864
3865        schema.weights_ready = false;
3866        assert!(
3867            !ModelInfo::from(&schema).weights_ready,
3868            "a schema with no weights on disk must project weights_ready = false"
3869        );
3870
3871        schema.weights_ready = true;
3872        assert!(
3873            ModelInfo::from(&schema).weights_ready,
3874            "a schema with weights on disk must project weights_ready = true"
3875        );
3876    }
3877
3878    /// The `weights_ready` flag is only meaningful for entries CAR actually
3879    /// downloads. `downloads_weights` is that gate, and it has to survive the
3880    /// schema → `ModelInfo` projection or the CLI is back to guessing from
3881    /// `is_local` — which is what made `windows/speech-synthesis:os` claim an
3882    /// install it does not have (#894).
3883    #[test]
3884    fn model_info_carries_downloads_weights_through_the_projection() {
3885        let mlx = test_generate_schema(
3886            "mlx-community/car894-test-4bit",
3887            "car894-test-4bit",
3888            ModelSource::Mlx {
3889                hf_repo: "mlx-community/car894-test-4bit".into(),
3890                hf_weight_file: None,
3891            },
3892        );
3893        assert!(
3894            ModelInfo::from(&mlx).downloads_weights,
3895            "an MLX entry downloads weights"
3896        );
3897
3898        // OS-owned sources are local even though CAR never downloads for
3899        // them, so a projection that reused `is_local` would get them wrong.
3900        for (label, source) in [
3901            ("windows speech", ModelSource::WindowsSpeech {}),
3902            (
3903                "apple foundation",
3904                ModelSource::AppleFoundationModels { use_case: None },
3905            ),
3906        ] {
3907            let schema = test_generate_schema("car894/os-model", "os-model", source);
3908            let info = ModelInfo::from(&schema);
3909            assert!(
3910                !info.downloads_weights,
3911                "{label} installs nothing, so the projection must say so"
3912            );
3913            assert!(
3914                info.is_local,
3915                "{label} is still local — which is exactly why is_local cannot stand in"
3916            );
3917        }
3918
3919        // Raw vLLM-MLX is independently owned and therefore remote even when
3920        // its endpoint happens to use a loopback spelling. CAR neither
3921        // downloads its weights nor imposes the client machine's hardware
3922        // requirements on it.
3923        let external = test_generate_schema(
3924            "car894/external-model",
3925            "external-model",
3926            ModelSource::VllmMlx {
3927                endpoint: "http://localhost:8000".into(),
3928                model_name: "mlx-community/car894-test-4bit".into(),
3929            },
3930        );
3931        assert!(!external.is_local());
3932        assert!(external.is_remote());
3933        assert!(!external.requires_apple_silicon());
3934        let info = ModelInfo::from(&external);
3935        assert!(!info.is_local);
3936        assert!(
3937            !info.downloads_weights,
3938            "external vllm-mlx owns its weights, so CAR installs nothing"
3939        );
3940    }
3941
3942    #[test]
3943    fn model_info_classifies_only_raw_vllm_mlx_as_operator_managed_external() {
3944        for endpoint in ["http://localhost:8000", "https://models.example.invalid/v1"] {
3945            let schema = test_generate_schema(
3946                "external/model",
3947                "external-model",
3948                ModelSource::VllmMlx {
3949                    endpoint: endpoint.into(),
3950                    model_name: "mlx-community/external-model".into(),
3951                },
3952            );
3953            let info = ModelInfo::from(&schema);
3954            assert!(info.operator_managed_external_runtime);
3955            assert_eq!(
3956                serde_json::to_value(info).unwrap()["operator_managed_external_runtime"],
3957                true
3958            );
3959        }
3960
3961        for source in [
3962            ModelSource::RemoteApi {
3963                endpoint: "https://cloud.example.invalid/v1".into(),
3964                api_key_env: "CAR_TEST_KEY".into(),
3965                api_key_envs: vec![],
3966                api_version: None,
3967                protocol: crate::schema::ApiProtocol::OpenAiCompat,
3968            },
3969            ModelSource::ManagedVllmMlx {
3970                hf_repo: "mlx-community/car-owned-model".into(),
3971                hf_weight_file: None,
3972            },
3973        ] {
3974            assert!(
3975                !ModelInfo::from(&test_generate_schema(
3976                    "not-external/model",
3977                    "not-external-model",
3978                    source,
3979                ))
3980                .operator_managed_external_runtime
3981            );
3982        }
3983    }
3984
3985    /// The fresh-machine state from Parslee-ai/car#894, end to end: an MLX
3986    /// entry with a declared `hf_repo` and an EMPTY models dir. It is
3987    /// `available` (ensure_local lazy-downloads on first use, #164) and
3988    /// simultaneously NOT `weights_ready` (nothing has been fetched). Those
3989    /// two facts are what `car models list` and `car doctor` were each
3990    /// reporting in isolation, which is why they disagreed.
3991    #[test]
3992    fn fresh_machine_mlx_entry_is_available_but_not_weights_ready() {
3993        let mut reg = test_registry();
3994        let id = "mlx-community/car894-fresh-4bit";
3995        reg.register(test_generate_schema(
3996            id,
3997            "car894-fresh-4bit",
3998            ModelSource::Mlx {
3999                hf_repo: "mlx-community/car894-fresh-4bit".into(),
4000                hf_weight_file: None,
4001            },
4002        ));
4003
4004        let registered = reg
4005            .get(id)
4006            .expect("the model just registered must be in the registry");
4007        let info = ModelInfo::from(registered);
4008
4009        // The models dir is a fresh TempDir, so nothing is downloaded.
4010        assert!(
4011            !registered.weights_ready,
4012            "an empty models dir means no weights on disk"
4013        );
4014        assert!(
4015            !info.weights_ready,
4016            "the CLI-facing projection must report the same: nothing installed"
4017        );
4018
4019        #[cfg(car_mlxlm_swift_built)]
4020        {
4021            // With the Swift MLX backend built, this is the exact contradiction
4022            // #894 reported: runnable here, but not installed.
4023            assert!(
4024                registered.available,
4025                "a declared hf_repo makes an MLX entry runnable before download (#164)"
4026            );
4027            assert!(
4028                info.available,
4029                "the projection must keep reporting it as runnable"
4030            );
4031        }
4032        #[cfg(not(car_mlxlm_swift_built))]
4033        {
4034            // Without the Swift MLX backend it is neither runnable nor
4035            // installed — still not a contradiction.
4036            assert!(
4037                !registered.available,
4038                "the Swift MLX backend is absent, so it must not be runnable"
4039            );
4040            assert!(!info.available);
4041        }
4042    }
4043
4044    fn write_complete_mlx_snapshot(dir: &Path) {
4045        std::fs::create_dir_all(dir).unwrap();
4046        std::fs::write(dir.join("config.json"), b"{}").unwrap();
4047        std::fs::write(dir.join("tokenizer.json"), b"{}").unwrap();
4048        std::fs::write(
4049            dir.join("model.safetensors.index.json"),
4050            r#"{"weight_map":{"a":"model-00001-of-00002.safetensors","b":"model-00002-of-00002.safetensors"}}"#,
4051        )
4052        .unwrap();
4053        std::fs::write(dir.join("model-00001-of-00002.safetensors"), b"one").unwrap();
4054        std::fs::write(dir.join("model-00002-of-00002.safetensors"), b"two").unwrap();
4055    }
4056
4057    #[test]
4058    fn complete_managed_and_shared_mlx_snapshots_are_physically_ready() {
4059        let schema = test_generate_schema(
4060            "mlx/qwen3-4b:4bit",
4061            "Qwen3-4B-MLX",
4062            ModelSource::Mlx {
4063                hf_repo: "mlx-community/Qwen3-4B-4bit".into(),
4064                hf_weight_file: None,
4065            },
4066        );
4067        let root = tempfile::tempdir().unwrap();
4068        let managed = root.path().join("managed");
4069        let shared = root.path().join("shared");
4070
4071        write_complete_mlx_snapshot(&managed);
4072        assert!(mlx_weights_ready_at(&schema, &managed, None));
4073
4074        std::fs::remove_dir_all(&managed).unwrap();
4075        write_complete_mlx_snapshot(&shared);
4076        assert!(mlx_weights_ready_at(&schema, &managed, Some(&shared)));
4077    }
4078
4079    #[test]
4080    fn zero_byte_gguf_is_not_physically_ready() {
4081        let schema = test_generate_schema(
4082            "qwen/qwen3-4b:q4_k_m",
4083            "Qwen3-4B",
4084            ModelSource::Local {
4085                hf_repo: "Qwen/Qwen3-4B-GGUF".into(),
4086                hf_filename: "model.gguf".into(),
4087                tokenizer_repo: "Qwen/Qwen3-4B".into(),
4088            },
4089        );
4090        let root = tempfile::tempdir().unwrap();
4091        let model_dir = root.path().join(&schema.name);
4092        std::fs::create_dir_all(&model_dir).unwrap();
4093        std::fs::write(model_dir.join("model.gguf"), b"").unwrap();
4094
4095        assert!(!physical_weights_ready(&schema, root.path()));
4096        std::fs::write(model_dir.join("model.gguf"), b"gguf").unwrap();
4097        assert!(physical_weights_ready(&schema, root.path()));
4098    }
4099
4100    #[test]
4101    fn shared_mlx_snapshot_missing_an_indexed_shard_is_not_physically_ready() {
4102        let schema = test_generate_schema(
4103            "mlx/qwen3-8b:4bit",
4104            "Qwen3-8B-MLX",
4105            ModelSource::Mlx {
4106                hf_repo: "mlx-community/Qwen3-8B-4bit".into(),
4107                hf_weight_file: None,
4108            },
4109        );
4110        let root = tempfile::tempdir().unwrap();
4111        let managed = root.path().join("managed");
4112        let shared = root.path().join("shared");
4113        write_complete_mlx_snapshot(&shared);
4114        std::fs::remove_file(shared.join("model-00001-of-00002.safetensors")).unwrap();
4115
4116        assert!(!mlx_weights_ready_at(&schema, &managed, Some(&shared)));
4117    }
4118
4119    /// `refresh_availability` must probe each DISTINCT credential once, not
4120    /// once per model (car-releases#75).
4121    ///
4122    /// A credential probe is a keychain query on macOS — ~14 ms. The stock
4123    /// catalog is ~72 models over roughly a dozen providers, so probing
4124    /// per-model cost ~1 s per refresh; a snapshot runs per request, and
4125    /// `estimated_tokens` ran a second one, so a delegated call waited ~2 s
4126    /// before it ever reached the host's runner.
4127    ///
4128    /// Asserted by COUNTING probes rather than by timing: a latency assertion
4129    /// would be flaky on a loaded CI runner, and the invariant that actually
4130    /// matters is "O(providers), not O(models)".
4131    /// Parslee-ai/car#786 — the catalog must stop advertising a managed
4132    /// namespace once the gateway has said it has no upstream for it.
4133    ///
4134    /// This is the reported symptom directly: `car models list` showed all ten
4135    /// `parslee/openrouter/*` aliases as `avail=yes` while every one of them
4136    /// 503'd, because availability for those rows resolves to "is a Parslee
4137    /// session signed in" and nothing more. A harness picked one on the
4138    /// strength of that claim and lost a full benchmark sweep to it.
4139    ///
4140    /// Asserted through the real `curated_schemas()` rows rather than a
4141    /// fixture, because the bug lives in how THOSE rows derive availability.
4142    ///
4143    /// Isolates `CAR_HOME` for the same reason its credential sibling does.
4144    /// `note_gateway_unconfigured` writes `gateway-state.json` under the state
4145    /// root, and cleaning up afterwards is not enough: nextest runs every test
4146    /// in its own process, so while this one holds the flag a DIFFERENT process
4147    /// can load it and fail on an availability assertion it never made
4148    /// (Parslee-ai/car#986). The crate's serial lock cannot serialize across
4149    /// processes; a private state root can.
4150    #[test]
4151    fn a_gateway_that_reports_no_upstream_stops_being_advertised() {
4152        let _guard = crate::openrouter::test_environment_scope();
4153        // See the note above: `StateRootScope` is the RAII form, so a panic in
4154        // an assertion below cannot leave `CAR_HOME` pointing at a deleted dir.
4155        let _home = StateRootScope::new();
4156        crate::openrouter::clear_gateway_unconfigured();
4157
4158        let managed: Vec<ModelSchema> = crate::openrouter::curated_schemas()
4159            .into_iter()
4160            .filter(|s| crate::openrouter::is_curated_managed_gateway_alias(&s.id))
4161            .collect();
4162        assert!(
4163            !managed.is_empty(),
4164            "precondition: the curated catalog must still carry managed aliases"
4165        );
4166
4167        let availability_of = |schema: &ModelSchema| match &schema.source {
4168            ModelSource::Proprietary { provider, auth, .. } => proprietary_auth_available(
4169                &schema.id,
4170                &schema.provider,
4171                provider,
4172                auth,
4173                // Signed in — the state in which the bug reported `yes`.
4174                true,
4175                &std::collections::HashMap::new(),
4176            ),
4177            other => panic!("managed aliases must be Proprietary, got {other:?}"),
4178        };
4179
4180        assert!(
4181            managed.iter().all(availability_of),
4182            "precondition: an authenticated session advertises these today"
4183        );
4184
4185        crate::openrouter::note_gateway_unconfigured();
4186        assert!(
4187            managed.iter().all(|s| !availability_of(s)),
4188            "after the gateway says it has no OpenRouter upstream, every alias in \
4189             the namespace must report unavailable — that claim is what cost the \
4190             benchmark sweep in #786"
4191        );
4192
4193        // Self-correcting: forgetting the observation restores the optimistic
4194        // claim, so an environment that later gains OpenRouter is not
4195        // permanently written off.
4196        crate::openrouter::clear_gateway_unconfigured();
4197        assert!(
4198            managed.iter().all(availability_of),
4199            "the suppression must be recoverable, not a one-way latch"
4200        );
4201    }
4202
4203    /// Parslee-ai/car#986 — building a registry must not clear the gateway
4204    /// observation as a side effect of construction.
4205    ///
4206    /// `new_with_catalog_public_key` ends in a `refresh_availability`, and that
4207    /// refresh forgets the session-scoped evidence whenever no Parslee session
4208    /// is present. #989 took the crate's serial lock in the two tests that call
4209    /// `refresh_availability` directly, but roughly twenty others reach the same
4210    /// clear just by constructing a registry, and none of them took the lock —
4211    /// so the class of bug survived the fix that was supposed to close it.
4212    ///
4213    /// This first case pins the harmless direction: a live session, so nothing
4214    /// is forgotten. `CAR_HOME` is isolated so the durable half of the
4215    /// observation lands in a temp dir rather than the developer's `~/.car`.
4216    #[test]
4217    fn constructing_a_registry_with_a_live_session_leaves_the_gateway_observation_alone() {
4218        let _guard = crate::openrouter::test_environment_scope();
4219        let home = StateRootScope::new();
4220
4221        crate::openrouter::note_gateway_unconfigured();
4222        let _registry = UnifiedRegistry::new_with_session(
4223            home.path().to_path_buf(),
4224            home.path().join("models"),
4225            None,
4226            SessionProbe::Fixed(true),
4227        );
4228
4229        let observed = crate::openrouter::gateway_unconfigured();
4230        let persisted = crate::openrouter::gateway_state_path().exists();
4231
4232        crate::openrouter::clear_gateway_unconfigured();
4233
4234        assert!(
4235            observed,
4236            "a signed-in session has no reason to forget what the gateway said"
4237        );
4238        assert!(
4239            persisted,
4240            "the durable half of the observation must survive construction too"
4241        );
4242    }
4243
4244    /// The other half of Parslee-ai/car#986: the production guarantee from
4245    /// Parslee-ai/car#786 is unchanged. When the session really is gone, the
4246    /// evidence learned about it is discarded — in memory and on disk — because
4247    /// the next sign-in may be to an org that has OpenRouter configured.
4248    ///
4249    /// Pinned rather than bypassed: the fix narrows *who* may forget, not
4250    /// whether forgetting still happens.
4251    #[test]
4252    fn constructing_a_registry_with_no_session_still_forgets_the_gateway_observation() {
4253        let _guard = crate::openrouter::test_environment_scope();
4254        let home = StateRootScope::new();
4255
4256        crate::openrouter::note_gateway_unconfigured();
4257        let recorded = crate::openrouter::gateway_unconfigured();
4258        let _registry = UnifiedRegistry::new_with_session(
4259            home.path().to_path_buf(),
4260            home.path().join("models"),
4261            None,
4262            SessionProbe::Fixed(false),
4263        );
4264
4265        let observed = crate::openrouter::gateway_unconfigured();
4266        let persisted = crate::openrouter::gateway_state_path().exists();
4267
4268        crate::openrouter::clear_gateway_unconfigured();
4269
4270        assert!(
4271            recorded,
4272            "precondition: the observation is on record before construction"
4273        );
4274        assert!(
4275            !observed,
4276            "sign-out must still discard the session-scoped verdict (#786)"
4277        );
4278        assert!(
4279            !persisted,
4280            "and the durable copy with it — otherwise the next sign-in inherits it from disk"
4281        );
4282    }
4283
4284    /// The case that actually failed before Parslee-ai/car#986 was fixed: an
4285    /// ordinary test registry, built the way every other test in this crate
4286    /// builds one, with no serial lock and no opinion about auth.
4287    ///
4288    /// On a machine with no Parslee session — a CI runner — construction alone
4289    /// used to wipe the observation, which is how
4290    /// `a_gateway_that_reports_no_upstream_stops_being_advertised` lost its flag
4291    /// between the `note_` call and the assertion under `cargo test`.
4292    ///
4293    /// Two assertions, deliberately. The observation one reproduces the CI
4294    /// condition and is the regression proper — but it only *fails* on main
4295    /// when the runner is signed out, so on a developer Mac with a live
4296    /// keychain session it would pass either way. The probe one holds
4297    /// regardless of the machine: the default a test registry is built with
4298    /// must be one that cannot forget session evidence, which is the whole
4299    /// content of the fix.
4300    #[test]
4301    fn an_ordinary_test_registry_does_not_disturb_a_separately_set_observation() {
4302        let _guard = crate::openrouter::test_environment_scope();
4303        let home = StateRootScope::new();
4304
4305        crate::openrouter::note_gateway_unconfigured();
4306        let _registry = UnifiedRegistry::new_with_state_root(
4307            home.path().to_path_buf(),
4308            home.path().join("models"),
4309        );
4310
4311        let observed = crate::openrouter::gateway_unconfigured();
4312
4313        crate::openrouter::clear_gateway_unconfigured();
4314
4315        assert!(
4316            observed,
4317            "constructing a registry is not a statement about the session, so it \
4318             must not erase an observation another test just recorded (#986)"
4319        );
4320        assert!(
4321            !SessionProbe::Inert.may_forget_session_evidence(),
4322            "the `cfg(test)` construction default must be a probe that answers \
4323             the session question without acting on it — this is the half of \
4324             the guarantee that does not depend on whether the runner happens \
4325             to be signed in"
4326        );
4327    }
4328
4329    /// Parslee-ai/car#887 — a signed-in session is not a WORKING one, and the
4330    /// catalog must stop advertising `parslee/*` once the server has rejected
4331    /// the credential.
4332    ///
4333    /// The reported symptom: on a machine with a stale sign-in, every routing
4334    /// pass offered the managed aliases as top-quality candidates and every
4335    /// request 401'd, so the router walked the fallback chain silently on every
4336    /// call. `access_token_is_available()` is existence-only, so `true` is
4337    /// exactly the state the bug reported as available.
4338    ///
4339    /// Isolates `CAR_HOME` rather than writing the observation into the
4340    /// developer's real `~/.car`; the sibling gateway test now does the same.
4341    #[test]
4342    fn a_rejected_credential_stops_the_managed_lane_being_advertised() {
4343        let _guard = crate::openrouter::test_environment_scope();
4344        let _home = StateRootScope::new();
4345        crate::parslee_credential::clear_credential_rejected();
4346
4347        let managed: Vec<ModelSchema> = crate::openrouter::curated_schemas()
4348            .into_iter()
4349            .filter(|s| s.provider == "parslee")
4350            .collect();
4351        assert!(
4352            !managed.is_empty(),
4353            "precondition: the curated catalog must still carry parslee rows"
4354        );
4355
4356        let availability_of = |schema: &ModelSchema| match &schema.source {
4357            ModelSource::Proprietary { provider, auth, .. } => proprietary_auth_available(
4358                &schema.id,
4359                &schema.provider,
4360                provider,
4361                auth,
4362                // Signed in — the state in which the bug reported `yes`.
4363                true,
4364                &std::collections::HashMap::new(),
4365            ),
4366            other => panic!("parslee rows must be Proprietary, got {other:?}"),
4367        };
4368
4369        assert!(
4370            managed.iter().all(availability_of),
4371            "precondition: an authenticated session advertises these today"
4372        );
4373
4374        crate::parslee_credential::note_credential_rejected();
4375        assert!(
4376            managed.iter().all(|s| !availability_of(s)),
4377            "after the server rejects the credential, EVERY parslee row must \
4378             report unavailable — unlike the gateway verdict this is not scoped \
4379             to the curated OpenRouter aliases, because a dead credential kills \
4380             the whole namespace"
4381        );
4382
4383        // Recoverable, not a one-way latch: the operator signs in again, the
4384        // next org lookup succeeds and clears the verdict.
4385        crate::parslee_credential::clear_credential_rejected();
4386        assert!(
4387            managed.iter().all(availability_of),
4388            "the suppression must lift once the credential works again"
4389        );
4390    }
4391
4392    /// **An explicit status request resolves env-or-keychain; the passive
4393    /// refresh resolves the environment alone.**
4394    ///
4395    /// The distinction is deliberate — the passive refresh runs on startup and
4396    /// catalog paths, where a secret-store read per provider would be wrong —
4397    /// but it reached a user-facing column. With a key kept in `car secrets`
4398    /// rather than exported, `car models list` reported a usable model as
4399    /// `RUNNABLE no`; measured with `OPENAI_API_KEY` in the keychain and
4400    /// nothing in the environment, `openai/gpt-5.5` read as not runnable while
4401    /// inference through the same binary worked.
4402    ///
4403    /// What this pins is the ENVIRONMENT half, which is all a test can reach:
4404    /// both modes must agree when the credential is exported, so the shared
4405    /// machinery is wired and the explicit mode has not silently become a
4406    /// no-op. The keychain half needs a populated OS store and cannot run in
4407    /// CI; it was verified by hand against the state above.
4408    #[test]
4409    fn an_explicit_status_refresh_sees_the_same_environment_credential() {
4410        let _environment = crate::openrouter::test_environment_scope();
4411        let tmp = TempDir::new().unwrap();
4412        let env_var = "CAR_TEST_EXPLICIT_STATUS_KEY";
4413        let mut registry = UnifiedRegistry::new_empty(tmp.path().join("models"));
4414        let mut schema = test_generate_schema(
4415            "vendor/explicit-status",
4416            "explicit-status",
4417            ModelSource::RemoteApi {
4418                protocol: crate::schema::ApiProtocol::OpenAiCompat,
4419                endpoint: "https://example.invalid/v1".into(),
4420                api_key_env: env_var.into(),
4421                api_key_envs: vec![],
4422                api_version: None,
4423            },
4424        );
4425        schema.provider = "vendor".into();
4426        registry.register(schema);
4427
4428        std::env::remove_var(env_var);
4429        registry.refresh_availability();
4430        assert!(
4431            !registry.get("vendor/explicit-status").unwrap().available,
4432            "no credential anywhere: the passive refresh must report unavailable"
4433        );
4434        registry.refresh_availability_for_explicit_status();
4435        assert!(
4436            !registry.get("vendor/explicit-status").unwrap().available,
4437            "no credential anywhere: the explicit refresh must agree"
4438        );
4439
4440        std::env::set_var(env_var, "sk-present");
4441        registry.refresh_availability();
4442        assert!(
4443            registry.get("vendor/explicit-status").unwrap().available,
4444            "exported credential: the passive refresh must see it"
4445        );
4446        registry.refresh_availability_for_explicit_status();
4447        assert!(
4448            registry.get("vendor/explicit-status").unwrap().available,
4449            "exported credential: the explicit refresh must see it too — if this fails \
4450             the authoritative mode is resolving through a path that cannot see the \
4451             environment, and `car models list` is answering from something else again"
4452        );
4453        std::env::remove_var(env_var);
4454    }
4455
4456    #[test]
4457    fn refresh_availability_probes_each_credential_once_not_per_model() {
4458        // Serialized with the other tests that touch the gateway observation.
4459        // `refresh_availability` CLEARS it when no Parslee token is present
4460        // (see the `!parslee_oauth_available` branch above), which is exactly
4461        // the state a CI runner is in — so without this lock, running under
4462        // `cargo test` (one process per crate target, tests threaded in
4463        // parallel inside it) this test can wipe the flag that
4464        // `a_gateway_that_reports_no_upstream_stops_being_advertised` set,
4465        // between its `note_` call and its assertion. `cargo nextest` hides it
4466        // by giving every test its own process, and every per-PR job runs
4467        // nextest — which is why it surfaced in `check-windows` instead
4468        // (Parslee-ai/car#986). See docs/solutions/process-global-state-in-tests.md
4469        // for the family and the full list of shared-process CI runs.
4470        let _environment = crate::openrouter::test_environment_scope();
4471        let tmp = TempDir::new().unwrap();
4472        let mut registry = UnifiedRegistry::new_empty(tmp.path().join("models"));
4473        for i in 0..25 {
4474            let mut schema = test_generate_schema(
4475                &format!("openrouter/model-{i}"),
4476                &format!("model-{i}"),
4477                ModelSource::RemoteApi {
4478                    protocol: crate::schema::ApiProtocol::OpenRouter,
4479                    endpoint: "https://openrouter.ai/api/v1".into(),
4480                    api_key_env: "OPENROUTER_API_KEY".into(),
4481                    api_key_envs: vec![],
4482                    api_version: None,
4483                },
4484            );
4485            schema.provider = "openrouter".into();
4486            registry.register(schema);
4487        }
4488
4489        crate::openrouter::reset_credential_source_call_count();
4490        registry.refresh_availability();
4491        let calls = crate::openrouter::credential_source_call_count();
4492
4493        assert_eq!(
4494            calls, 1,
4495            "refresh_availability probed the OpenRouter credential {calls} times for 25 models; \
4496             it must resolve each distinct credential once per refresh, not once per model"
4497        );
4498    }
4499
4500    /// `pull` on a `vllm-mlx` model must report the HuggingFace cache directory
4501    /// the external runtime will actually open — CAR's own
4502    /// `~/.car/models/<name>` layout is the wrong shape for a model a Python
4503    /// runtime resolves by repo id, and reporting the wrong path would make
4504    /// "where did my 16 GB go" unanswerable.
4505    #[test]
4506    fn a_vllm_mlx_pull_targets_the_shared_huggingface_cache() {
4507        let repo = "mlx-community/Qwen3.8-27B-4bit";
4508        let dir = crate::hf_cache::repo_dir(repo);
4509        assert!(
4510            dir.ends_with("models--mlx-community--Qwen3.8-27B-4bit"),
4511            "got {}",
4512            dir.display()
4513        );
4514        assert!(
4515            dir.parent().is_some_and(|p| p.ends_with("hub")),
4516            "must live under the HF cache's hub/ root, got {}",
4517            dir.display()
4518        );
4519    }
4520
4521    /// A raw `VllmMlx` row is operator-managed, even when its endpoint is a
4522    /// loopback URL. Only `ManagedVllmMlx` opts into CAR's provisioned runtime.
4523    #[test]
4524    fn external_vllm_mlx_does_not_become_managed_from_a_loopback_endpoint() {
4525        // Serialized with the other tests that touch the gateway observation.
4526        // `refresh_availability` CLEARS it when no Parslee token is present
4527        // (see the `!parslee_oauth_available` branch above), which is exactly
4528        // the state a CI runner is in — so without this lock, running under
4529        // `cargo test` (one process per crate target, tests threaded in
4530        // parallel inside it) this test can wipe the flag that
4531        // `a_gateway_that_reports_no_upstream_stops_being_advertised` set,
4532        // between its `note_` call and its assertion. `cargo nextest` hides it
4533        // by giving every test its own process, and every per-PR job runs
4534        // nextest — which is why it surfaced in `check-windows` instead
4535        // (Parslee-ai/car#986). See docs/solutions/process-global-state-in-tests.md
4536        // for the family and the full list of shared-process CI runs.
4537        let _environment = crate::openrouter::test_environment_scope();
4538        let tmp = TempDir::new().unwrap();
4539        let mut registry = UnifiedRegistry::new_empty(tmp.path().join("models"));
4540        let schema = test_generate_schema(
4541            "vllm-mlx/arch-the-rust-backend-cannot-load",
4542            "external-only-model",
4543            ModelSource::VllmMlx {
4544                endpoint: "http://localhost:8000".into(),
4545                model_name: "mlx-community/Qwen3.8-27B-4bit".into(),
4546            },
4547        );
4548        registry.register(schema);
4549
4550        // Deliberately NOT set: no external endpoint has been declared healthy.
4551        assert!(
4552            std::env::var("VLLM_MLX_ENDPOINT").is_err(),
4553            "test precondition: VLLM_MLX_ENDPOINT must be unset"
4554        );
4555        registry.refresh_availability();
4556
4557        let model = registry
4558            .get("vllm-mlx/arch-the-rust-backend-cannot-load")
4559            .expect("registered model should be present");
4560        assert!(
4561            !model.available,
4562            "an external vllm-mlx row remains external even on loopback; only an \
4563             explicit ManagedVllmMlx source may use CAR's runtime"
4564        );
4565    }
4566
4567    #[test]
4568    fn user_config_load_and_save_force_community_trust() {
4569        let tmp = TempDir::new().unwrap();
4570        let models_dir = tmp.path().join("models");
4571        let config_path = tmp.path().join("models.json");
4572        let schema = test_generate_schema(
4573            "user/test-model",
4574            "user-test-model",
4575            ModelSource::RemoteApi {
4576                endpoint: "https://attacker.invalid/v1/chat/completions".into(),
4577                api_key_env: "CAR_USER_MODEL_TEST_KEY".into(),
4578                api_key_envs: vec![],
4579                api_version: None,
4580                protocol: crate::schema::ApiProtocol::OpenAiCompat,
4581            },
4582        );
4583        let mut omitted_tier = serde_json::to_value(schema.clone()).unwrap();
4584        omitted_tier.as_object_mut().unwrap().remove("trust_tier");
4585        std::fs::write(
4586            &config_path,
4587            serde_json::to_vec_pretty(&vec![omitted_tier]).unwrap(),
4588        )
4589        .unwrap();
4590
4591        let mut loaded = UnifiedRegistry::new_empty(models_dir.clone());
4592        loaded.load_user_config().unwrap();
4593        assert_eq!(
4594            loaded.get("user/test-model").unwrap().trust_tier,
4595            crate::schema::TrustTier::Community
4596        );
4597
4598        let mut persisted = UnifiedRegistry::new_empty(models_dir);
4599        persisted.register_user_model(schema);
4600        persisted.save_user_config().unwrap();
4601        let saved: Vec<ModelSchema> =
4602            serde_json::from_slice(&std::fs::read(config_path).unwrap()).unwrap();
4603        assert_eq!(saved.len(), 1);
4604        assert_eq!(saved[0].trust_tier, crate::schema::TrustTier::Community);
4605    }
4606
4607    #[test]
4608    fn persisted_user_model_cannot_shadow_managed_openrouter_alias() {
4609        // Constructing a `UnifiedRegistry` ends in `refresh_availability()`,
4610        // which WRITES the process-global gateway observation. Full reasoning
4611        // on `TestRegistry` in this crate; family in
4612        // docs/solutions/process-global-state-in-tests.md.
4613        let _environment = crate::openrouter::test_environment_scope();
4614        let tmp = TempDir::new().unwrap();
4615        let models_dir = tmp.path().join("models");
4616        let config_path = tmp.path().join("models.json");
4617        let mut shadow = crate::openrouter::curated_schemas()
4618            .into_iter()
4619            .find(|schema| schema.id == "parslee/openrouter/frontier-general")
4620            .unwrap();
4621        shadow.provider = "attacker".into();
4622        std::fs::write(
4623            &config_path,
4624            serde_json::to_vec_pretty(&vec![shadow]).unwrap(),
4625        )
4626        .unwrap();
4627
4628        let registry = UnifiedRegistry::new_with_state_root(tmp.path().to_path_buf(), models_dir);
4629        let actual = registry
4630            .get("parslee/openrouter/frontier-general")
4631            .expect("compiled managed alias must remain present");
4632        assert_eq!(actual.provider, "parslee");
4633        assert_eq!(
4634            crate::openrouter::canonical_managed_gateway_selector(actual),
4635            Some("parslee/openrouter/frontier-general")
4636        );
4637    }
4638
4639    fn signed_remote(id: &str, name: &str) -> ModelSchema {
4640        test_generate_schema(
4641            id,
4642            name,
4643            ModelSource::RemoteApi {
4644                endpoint: "https://catalog.example/v1".into(),
4645                api_key_env: "SIGNED_CATALOG_TEST_KEY".into(),
4646                api_key_envs: vec![],
4647                api_version: None,
4648                protocol: crate::schema::ApiProtocol::OpenAiCompat,
4649            },
4650        )
4651    }
4652
4653    #[test]
4654    fn a_user_registered_row_wins_a_live_signed_swap_and_stays_in_models_json() {
4655        let _environment = crate::openrouter::test_environment_scope();
4656        let tmp = TempDir::new().unwrap();
4657        let mut registry = UnifiedRegistry::new_with_catalog_public_key(
4658            tmp.path().to_path_buf(),
4659            tmp.path().join("models"),
4660            None,
4661        );
4662        let mut mine = signed_remote("shared/id", "mine");
4663        mine.context_length = 1234;
4664        registry.register_user_model(mine);
4665        let mut publishers = signed_remote("shared/id", "publishers");
4666        publishers.context_length = 9999;
4667        assert_eq!(
4668            registry.replace_signed_catalog(
4669                vec![publishers],
4670                None,
4671                &HashSet::new(),
4672                &HashSet::new()
4673            ),
4674            0
4675        );
4676        assert_eq!(registry.get("shared/id").unwrap().context_length, 1234);
4677        registry.save_user_config().unwrap();
4678        let saved = std::fs::read_to_string(tmp.path().join(USER_MODELS_FILE)).unwrap();
4679        assert!(saved.contains("1234") && !saved.contains("9999"), "{saved}");
4680        assert!(registry.unregister_user_model("shared/id").is_some());
4681    }
4682
4683    #[test]
4684    fn a_withdrawn_row_something_depends_on_stays_deprecated_until_nothing_does() {
4685        let _environment = crate::openrouter::test_environment_scope();
4686        let tmp = TempDir::new().unwrap();
4687        let mut registry = UnifiedRegistry::new_with_catalog_public_key(
4688            tmp.path().to_path_buf(),
4689            tmp.path().join("models"),
4690            None,
4691        );
4692        let used = signed_remote("signed/used", "used");
4693        let idle = signed_remote("signed/idle", "idle");
4694        assert_eq!(
4695            registry.replace_signed_catalog(
4696                vec![used, idle],
4697                None,
4698                &HashSet::new(),
4699                &HashSet::new()
4700            ),
4701            2
4702        );
4703        let keep: HashSet<String> = ["signed/used".to_string()].into();
4704        assert_eq!(
4705            registry.replace_signed_catalog(vec![], None, &keep, &HashSet::new()),
4706            0
4707        );
4708        assert!(
4709            registry.get("signed/idle").is_none(),
4710            "nothing depended on it"
4711        );
4712        let kept = registry.get("signed/used").expect("a dependent keeps it");
4713        assert!(kept.deprecated, "kept, but never suggested");
4714        // Republished: it comes back current.
4715        let used = signed_remote("signed/used", "used");
4716        assert_eq!(
4717            registry.replace_signed_catalog(vec![used], None, &keep, &HashSet::new()),
4718            1
4719        );
4720        assert!(!registry.get("signed/used").unwrap().deprecated);
4721        // Nothing depends on it any more: the next catalog that omits it drops it.
4722        registry.replace_signed_catalog(vec![], None, &HashSet::new(), &HashSet::new());
4723        assert!(registry.get("signed/used").is_none());
4724    }
4725
4726    #[test]
4727    fn a_kept_row_survives_a_restart_only_while_its_signed_catalog_verifies() {
4728        let _environment = crate::openrouter::test_environment_scope();
4729        let tmp = TempDir::new().unwrap();
4730        let open = |key: Option<&str>| {
4731            UnifiedRegistry::new_with_catalog_public_key(
4732                tmp.path().to_path_buf(),
4733                tmp.path().join("models"),
4734                key,
4735            )
4736        };
4737        let cache = crate::catalog::cache_path(tmp.path());
4738        let retained = crate::catalog::retained_path(tmp.path());
4739        let catalog = |models, version| {
4740            crate::catalog::signed_test_catalog(
4741                crate::catalog::CatalogDoc {
4742                    revoked: Vec::new(),
4743                    version,
4744                    models,
4745                },
4746                21,
4747            )
4748        };
4749        let (v1, key) = catalog(
4750            vec![
4751                signed_remote("signed/used", "used"),
4752                signed_remote("signed/idle", "idle"),
4753            ],
4754            1,
4755        );
4756        crate::catalog::save_verified(&cache, &v1).unwrap();
4757        let mut running = open(Some(&key));
4758        assert!(running.get("signed/used").is_some());
4759
4760        // v2 withdraws both; something depends on `used`.
4761        let (v2, _) = catalog(vec![], 2);
4762        let keep: HashSet<String> = ["signed/used".to_string()].into();
4763        running.replace_signed_catalog(
4764            v2.clone().into_models(),
4765            Some(std::sync::Arc::new(v2.envelope())),
4766            &keep,
4767            &HashSet::new(),
4768        );
4769        crate::catalog::save_retained(&retained, &running.retained_rows()).unwrap();
4770        crate::catalog::save_verified(&cache, &v2).unwrap();
4771
4772        let restarted = open(Some(&key));
4773        let kept = restarted.get("signed/used").expect("kept across a restart");
4774        assert!(kept.deprecated);
4775        assert!(restarted.get("signed/idle").is_none());
4776        // Another key: nothing verifies, nothing comes back.
4777        let (_, other_key) = crate::catalog::signed_test_catalog(
4778            crate::catalog::CatalogDoc {
4779                revoked: Vec::new(),
4780                version: 9,
4781                models: vec![],
4782            },
4783            22,
4784        );
4785        assert!(open(Some(&other_key)).get("signed/used").is_none());
4786        // A tampered file: the envelope no longer verifies.
4787        let json = std::fs::read_to_string(&retained).unwrap();
4788        std::fs::write(&retained, json.replace("signed/idle", "signed/idlx")).unwrap();
4789        assert!(open(Some(&key)).get("signed/used").is_none());
4790    }
4791
4792    #[test]
4793    fn a_revoked_row_leaves_even_when_something_depends_on_it() {
4794        let _environment = crate::openrouter::test_environment_scope();
4795        let tmp = TempDir::new().unwrap();
4796        let mut registry = UnifiedRegistry::new_with_catalog_public_key(
4797            tmp.path().to_path_buf(),
4798            tmp.path().join("models"),
4799            None,
4800        );
4801        let none = HashSet::new();
4802        let bad: HashSet<String> = ["signed/bad".to_string()].into();
4803        registry.replace_signed_catalog(
4804            vec![signed_remote("signed/bad", "bad")],
4805            None,
4806            &none,
4807            &none,
4808        );
4809        // Withdrawn, but a lane depends on it: an ordinary withdrawal keeps it…
4810        registry.replace_signed_catalog(vec![], None, &bad, &none);
4811        assert!(registry.get("signed/bad").is_some());
4812        // …a revocation does not, even with nothing else changing.
4813        registry.replace_signed_catalog(vec![], None, &bad, &bad);
4814        assert!(registry.get("signed/bad").is_none());
4815        // And a catalog never registers a row it revokes itself.
4816        registry.replace_signed_catalog(
4817            vec![signed_remote("signed/bad", "bad")],
4818            None,
4819            &none,
4820            &bad,
4821        );
4822        assert!(registry.get("signed/bad").is_none());
4823        assert!(registry.retained_rows().is_empty());
4824
4825        // A revoked builtin is deprecated, and restored once it is not.
4826        let builtin = builtin_catalog()
4827            .into_iter()
4828            .find(|m| !m.deprecated)
4829            .expect("a current builtin");
4830        let revoked: HashSet<String> = [builtin.id.clone()].into();
4831        registry.replace_signed_catalog(vec![], None, &none, &revoked);
4832        assert!(registry.get(&builtin.id).unwrap().deprecated);
4833        registry.replace_signed_catalog(vec![], None, &none, &none);
4834        assert!(!registry.get(&builtin.id).unwrap().deprecated);
4835    }
4836
4837    #[test]
4838    fn a_restart_never_restores_a_row_the_current_catalog_revokes() {
4839        let _environment = crate::openrouter::test_environment_scope();
4840        let tmp = TempDir::new().unwrap();
4841        let catalog = |models, revoked, version| {
4842            crate::catalog::signed_test_catalog(
4843                crate::catalog::CatalogDoc {
4844                    version,
4845                    models,
4846                    revoked,
4847                },
4848                23,
4849            )
4850        };
4851        let (v1, key) = catalog(vec![signed_remote("signed/kept", "kept")], vec![], 1);
4852        let mut running = UnifiedRegistry::new_with_catalog_public_key(
4853            tmp.path().to_path_buf(),
4854            tmp.path().join("models"),
4855            Some(&key),
4856        );
4857        running.replace_signed_catalog(
4858            v1.clone().into_models(),
4859            Some(std::sync::Arc::new(v1.envelope())),
4860            &HashSet::new(),
4861            &HashSet::new(),
4862        );
4863        // v2 withdraws it while a lane depends on it: kept and recorded.
4864        let (v2, _) = catalog(vec![], vec![], 2);
4865        let keep: HashSet<String> = ["signed/kept".to_string()].into();
4866        running.replace_signed_catalog(
4867            vec![],
4868            Some(std::sync::Arc::new(v2.envelope())),
4869            &keep,
4870            &HashSet::new(),
4871        );
4872        let retained = crate::catalog::retained_path(tmp.path());
4873        crate::catalog::save_retained(&retained, &running.retained_rows()).unwrap();
4874        // v3 revokes it; the daemon restarts before anything rewrites the file.
4875        let (v3, _) = catalog(vec![], vec!["signed/kept".to_string()], 3);
4876        crate::catalog::save_verified(&crate::catalog::cache_path(tmp.path()), &v3).unwrap();
4877        let restarted = UnifiedRegistry::new_with_catalog_public_key(
4878            tmp.path().to_path_buf(),
4879            tmp.path().join("models"),
4880            Some(&key),
4881        );
4882        assert!(restarted.get("signed/kept").is_none());
4883    }
4884
4885    #[test]
4886    fn a_signed_overlay_updates_a_builtins_measurements_never_its_weights() {
4887        let _environment = crate::openrouter::test_environment_scope();
4888        let tmp = TempDir::new().unwrap();
4889        let mut registry = UnifiedRegistry::new_with_catalog_public_key(
4890            tmp.path().to_path_buf(),
4891            tmp.path().join("models"),
4892            None,
4893        );
4894        let compiled = builtin_catalog()
4895            .into_iter()
4896            .find(|m| m.public_benchmarks.iter().any(|b| b.name == "car-judged"))
4897            .expect("a scored builtin");
4898        let mut remeasured = compiled.clone();
4899        remeasured.public_benchmarks[0].score = 0.42;
4900        remeasured.public_benchmarks[0].runs = Some(3);
4901        remeasured.context_length = compiled.context_length + 1;
4902        remeasured.name = "renamed by the publisher".into();
4903        assert_eq!(
4904            registry.replace_signed_catalog(
4905                vec![remeasured.clone()],
4906                None,
4907                &HashSet::new(),
4908                &HashSet::new()
4909            ),
4910            1
4911        );
4912        let row = registry.get(&compiled.id).unwrap();
4913        assert_eq!(row.public_benchmarks[0].score, 0.42);
4914        assert_eq!(row.public_benchmarks[0].runs, Some(3));
4915        assert_eq!(
4916            row.context_length, compiled.context_length,
4917            "not an overlay field"
4918        );
4919        assert_eq!(row.name, compiled.name, "not an overlay field");
4920        // A catalog that omits it restores what the build shipped.
4921        registry.replace_signed_catalog(vec![], None, &HashSet::new(), &HashSet::new());
4922        assert_eq!(
4923            registry.get(&compiled.id).unwrap().public_benchmarks[0].score,
4924            compiled.public_benchmarks[0].score
4925        );
4926        // Different weights under a builtin id: refused outright.
4927        let mut repointed = remeasured;
4928        repointed.source = signed_remote("x/y", "y").source;
4929        assert_eq!(
4930            registry.replace_signed_catalog(
4931                vec![repointed],
4932                None,
4933                &HashSet::new(),
4934                &HashSet::new()
4935            ),
4936            0
4937        );
4938        assert_eq!(
4939            registry.get(&compiled.id).unwrap().public_benchmarks[0].score,
4940            compiled.public_benchmarks[0].score
4941        );
4942    }
4943
4944    #[test]
4945    fn user_config_persistence_excludes_signed_rows_and_keeps_builtin_tagged_user_rows() {
4946        // Constructing a `UnifiedRegistry` ends in `refresh_availability()`,
4947        // which WRITES the process-global gateway observation. Full reasoning
4948        // on `TestRegistry` in this crate; family in
4949        // docs/solutions/process-global-state-in-tests.md.
4950        let _environment = crate::openrouter::test_environment_scope();
4951        let tmp = TempDir::new().unwrap();
4952        let models_dir = tmp.path().join("models");
4953        let config_path = tmp.path().join("models.json");
4954
4955        let signed = test_generate_schema(
4956            "signed/catalog-only",
4957            "signed-catalog-only",
4958            ModelSource::RemoteApi {
4959                endpoint: "https://catalog.example/v1".into(),
4960                api_key_env: "SIGNED_CATALOG_TEST_KEY".into(),
4961                api_key_envs: vec![],
4962                api_version: None,
4963                protocol: crate::schema::ApiProtocol::OpenAiCompat,
4964            },
4965        );
4966        assert!(!signed.tags.iter().any(|tag| tag == "builtin"));
4967        let (verified, public_key) = crate::catalog::signed_test_catalog(
4968            crate::catalog::CatalogDoc {
4969                revoked: Vec::new(),
4970                version: 81,
4971                models: vec![signed],
4972            },
4973            81,
4974        );
4975        crate::catalog::save_verified(&crate::catalog::cache_path(tmp.path()), &verified).unwrap();
4976
4977        let mut registry = UnifiedRegistry::new_with_catalog_public_key(
4978            tmp.path().to_path_buf(),
4979            models_dir.clone(),
4980            Some(public_key.as_str()),
4981        );
4982        let mut user = test_generate_schema(
4983            "user/builtin-tagged",
4984            "user-builtin-tagged",
4985            ModelSource::RemoteApi {
4986                endpoint: "https://user.example/v1".into(),
4987                api_key_env: "USER_MODEL_TEST_KEY".into(),
4988                api_key_envs: vec![],
4989                api_version: None,
4990                protocol: crate::schema::ApiProtocol::OpenAiCompat,
4991            },
4992        );
4993        user.tags.push("builtin".into());
4994        registry.register_user_model(user);
4995        registry.save_user_config().unwrap();
4996
4997        let saved: Vec<ModelSchema> =
4998            serde_json::from_slice(&std::fs::read(&config_path).unwrap()).unwrap();
4999        assert_eq!(
5000            saved
5001                .iter()
5002                .map(|model| model.id.as_str())
5003                .collect::<Vec<_>>(),
5004            vec!["user/builtin-tagged"],
5005            "models.json must contain only explicitly user-registered rows"
5006        );
5007        assert_eq!(saved[0].trust_tier, crate::schema::TrustTier::Community);
5008
5009        let mut restarted = UnifiedRegistry::new_with_catalog_public_key(
5010            tmp.path().to_path_buf(),
5011            models_dir,
5012            Some(public_key.as_str()),
5013        );
5014        assert_eq!(
5015            restarted.get("signed/catalog-only").unwrap().trust_tier,
5016            crate::schema::TrustTier::Curated,
5017            "user persistence must not demote an unrelated signed catalog row"
5018        );
5019        assert_eq!(
5020            restarted.get("user/builtin-tagged").unwrap().trust_tier,
5021            crate::schema::TrustTier::Community
5022        );
5023        let mut signed_shadow = restarted.get("signed/catalog-only").unwrap().clone();
5024        signed_shadow.name = "user-shadow-of-signed-row".into();
5025        restarted.register_user_model(signed_shadow);
5026        assert_eq!(
5027            restarted.get("signed/catalog-only").unwrap().name,
5028            "signed-catalog-only",
5029            "a user row must not shadow a signature-verified project exact id"
5030        );
5031    }
5032
5033    #[test]
5034    fn empty_user_config_save_clears_stale_rows() {
5035        let tmp = TempDir::new().unwrap();
5036        let models_dir = tmp.path().join("models");
5037        let config_path = tmp.path().join("models.json");
5038        let stale = test_generate_schema(
5039            "user/stale",
5040            "stale",
5041            ModelSource::RemoteApi {
5042                endpoint: "https://stale.example/v1".into(),
5043                api_key_env: "STALE_USER_MODEL_TEST_KEY".into(),
5044                api_key_envs: vec![],
5045                api_version: None,
5046                protocol: crate::schema::ApiProtocol::OpenAiCompat,
5047            },
5048        );
5049        std::fs::write(
5050            &config_path,
5051            serde_json::to_vec_pretty(&vec![stale]).unwrap(),
5052        )
5053        .unwrap();
5054
5055        UnifiedRegistry::new_empty(models_dir)
5056            .save_user_config()
5057            .unwrap();
5058
5059        let saved: Vec<ModelSchema> =
5060            serde_json::from_slice(&std::fs::read(config_path).unwrap()).unwrap();
5061        assert!(
5062            saved.is_empty(),
5063            "saving an empty user set must overwrite stale models.json rows"
5064        );
5065    }
5066
5067    #[test]
5068    fn unregister_then_save_removes_the_user_row_from_disk() {
5069        let tmp = TempDir::new().unwrap();
5070        let models_dir = tmp.path().join("models");
5071        let config_path = tmp.path().join("models.json");
5072        let mut registry = UnifiedRegistry::new_empty(models_dir);
5073        registry.register_project_model(test_generate_schema(
5074            "signed/not-user-removable",
5075            "not-user-removable",
5076            ModelSource::RemoteApi {
5077                endpoint: "https://catalog.example/v1".into(),
5078                api_key_env: "SIGNED_CATALOG_TEST_KEY".into(),
5079                api_key_envs: vec![],
5080                api_version: None,
5081                protocol: crate::schema::ApiProtocol::OpenAiCompat,
5082            },
5083        ));
5084        assert!(
5085            registry
5086                .unregister_user_model("signed/not-user-removable")
5087                .is_none(),
5088            "the user boundary cannot unregister an untracked catalog row"
5089        );
5090        assert!(registry.get("signed/not-user-removable").is_some());
5091        registry.register_user_model(test_generate_schema(
5092            "user/removable",
5093            "removable",
5094            ModelSource::RemoteApi {
5095                endpoint: "https://user.example/v1".into(),
5096                api_key_env: "REMOVABLE_USER_MODEL_TEST_KEY".into(),
5097                api_key_envs: vec![],
5098                api_version: None,
5099                protocol: crate::schema::ApiProtocol::OpenAiCompat,
5100            },
5101        ));
5102        registry.save_user_config().unwrap();
5103        assert!(registry.unregister_user_model("user/removable").is_some());
5104        registry.save_user_config().unwrap();
5105
5106        let saved: Vec<ModelSchema> =
5107            serde_json::from_slice(&std::fs::read(config_path).unwrap()).unwrap();
5108        assert!(saved.is_empty());
5109    }
5110
5111    /// The daemon's `models.register` handler writes through
5112    /// [`user_config_path`]; the registry reads through the state root it was
5113    /// built with. Those two resolvers must land on one file.
5114    ///
5115    /// They did not. The write side moved to `CAR_HOME` while the read side
5116    /// stayed `<models_dir>/../models.json` — and `models_dir` is deliberately
5117    /// the machine-shared weights cache, which does NOT move. So against a
5118    /// relocated daemon the registration was persisted to
5119    /// `$CAR_HOME/models.json`, the next boot read `~/.car/models.json`, and
5120    /// the model silently never appeared. No error anywhere; the handler even
5121    /// reported the path it had written.
5122    #[test]
5123    fn a_registration_written_under_car_home_is_the_file_the_registry_reads() {
5124        let _environment = crate::openrouter::test_environment_scope();
5125        let prior = std::env::var_os(car_home::ENV_VAR);
5126
5127        let state_root = TempDir::new().unwrap();
5128        // Deliberately NOT inside the state root — this stands in for the
5129        // machine-shared `~/.car/models`, which stays put under `CAR_HOME`.
5130        let weights = TempDir::new().unwrap();
5131        let models_dir = weights.path().join("models");
5132        std::fs::create_dir_all(&models_dir).unwrap();
5133
5134        unsafe { std::env::set_var(car_home::ENV_VAR, state_root.path()) };
5135
5136        // Exactly what `handle_models_register` resolves and writes.
5137        let write_path = user_config_path().expect("CAR_HOME must resolve a models.json path");
5138        assert_eq!(write_path, state_root.path().join(USER_MODELS_FILE));
5139        let registered = test_generate_schema(
5140            "user/relocated-daemon-model",
5141            "relocated-daemon-model",
5142            ModelSource::RemoteApi {
5143                endpoint: "https://relocated.example/v1".into(),
5144                api_key_env: "RELOCATED_DAEMON_MODEL_TEST_KEY".into(),
5145                api_key_envs: vec![],
5146                api_version: None,
5147                protocol: crate::schema::ApiProtocol::OpenAiCompat,
5148            },
5149        );
5150        std::fs::write(
5151            &write_path,
5152            serde_json::to_vec_pretty(&vec![registered]).unwrap(),
5153        )
5154        .unwrap();
5155
5156        // …and exactly what the next daemon boot constructs.
5157        let registry = UnifiedRegistry::new(models_dir.clone());
5158
5159        match prior {
5160            Some(value) => unsafe { std::env::set_var(car_home::ENV_VAR, value) },
5161            None => unsafe { std::env::remove_var(car_home::ENV_VAR) },
5162        }
5163
5164        assert!(
5165            registry.get("user/relocated-daemon-model").is_some(),
5166            "the registry must load the models.json that `models.register` wrote; \
5167             it looked at {} instead",
5168            registry.user_config_path.display(),
5169        );
5170        assert_eq!(registry.user_config_path, write_path);
5171        assert!(
5172            !weights.path().join(USER_MODELS_FILE).exists(),
5173            "nothing may be written beside the shared weights cache",
5174        );
5175    }
5176
5177    #[test]
5178    fn ready_without_download_is_strict_for_local_model_files() {
5179        let tmp = TempDir::new().unwrap();
5180        let models = tmp.path().join("models");
5181        let mut reg = UnifiedRegistry::new_empty(models.clone());
5182        reg.register(test_generate_schema(
5183            "local/test",
5184            "TestLocal",
5185            ModelSource::Local {
5186                hf_repo: "example/repo".into(),
5187                hf_filename: "model.gguf".into(),
5188                tokenizer_repo: "example/repo".into(),
5189            },
5190        ));
5191
5192        assert_eq!(reg.ready_without_download("local/test"), Some(false));
5193
5194        let dir = models.join("TestLocal");
5195        std::fs::create_dir_all(&dir).unwrap();
5196        std::fs::write(dir.join("model.gguf"), b"weights").unwrap();
5197        assert_eq!(
5198            reg.ready_without_download("local/test"),
5199            Some(false),
5200            "tokenizer is required too"
5201        );
5202        std::fs::write(dir.join("tokenizer.json"), "{}").unwrap();
5203        assert_eq!(reg.ready_without_download("local/test"), Some(true));
5204    }
5205
5206    #[test]
5207    fn ready_without_download_rejects_mlx_config_only_stub() {
5208        let tmp = TempDir::new().unwrap();
5209        let models = tmp.path().join("models");
5210        let mut reg = UnifiedRegistry::new_empty(models.clone());
5211        reg.register(test_generate_schema(
5212            "mlx/test",
5213            "TestMlx",
5214            ModelSource::Mlx {
5215                hf_repo: "example/repo".into(),
5216                hf_weight_file: None,
5217            },
5218        ));
5219
5220        let dir = models.join("TestMlx");
5221        std::fs::create_dir_all(&dir).unwrap();
5222        std::fs::write(dir.join("config.json"), "{}").unwrap();
5223        std::fs::write(dir.join("tokenizer.json"), "{}").unwrap();
5224        assert_eq!(
5225            reg.ready_without_download("mlx/test"),
5226            Some(false),
5227            "config/tokenizer stubs must not start assistant inference"
5228        );
5229
5230        std::fs::write(dir.join("model.safetensors"), b"weights").unwrap();
5231        assert_eq!(reg.ready_without_download("mlx/test"), Some(true));
5232    }
5233
5234    fn write_mlx_dir(root: &Path, name: &str, model_type: &str) {
5235        let dir = root.join(name);
5236        std::fs::create_dir_all(&dir).unwrap();
5237        std::fs::write(
5238            dir.join("config.json"),
5239            serde_json::json!({
5240                "model_type": model_type,
5241                "max_position_embeddings": 40_960,
5242                "quantization": { "bits": 8, "group_size": 32, "mode": "mxfp8" },
5243            })
5244            .to_string(),
5245        )
5246        .unwrap();
5247        std::fs::write(dir.join("model.safetensors"), b"weights").unwrap();
5248    }
5249
5250    struct ScopedEnvVar {
5251        name: &'static str,
5252        previous: Option<std::ffi::OsString>,
5253    }
5254
5255    impl ScopedEnvVar {
5256        fn set(name: &'static str, value: &Path) -> Self {
5257            let previous = std::env::var_os(name);
5258            // SAFETY: acquisition tests hold `test_environment_scope_async` for
5259            // the lifetime of this process-global environment mutation.
5260            unsafe { std::env::set_var(name, value) };
5261            Self { name, previous }
5262        }
5263
5264        fn unset(name: &'static str) -> Self {
5265            let previous = std::env::var_os(name);
5266            // SAFETY: as for `set`.
5267            unsafe { std::env::remove_var(name) };
5268            Self { name, previous }
5269        }
5270    }
5271
5272    impl Drop for ScopedEnvVar {
5273        fn drop(&mut self) {
5274            // SAFETY: the environment mutex is still held because this guard is
5275            // declared after it and therefore drops first.
5276            unsafe {
5277                match self.previous.take() {
5278                    Some(value) => std::env::set_var(self.name, value),
5279                    None => std::env::remove_var(self.name),
5280                }
5281            }
5282        }
5283    }
5284
5285    #[derive(Default)]
5286    struct AcquisitionRecorder {
5287        events: std::sync::Mutex<Vec<DownloadEvent>>,
5288        replace_dir_on_started: std::sync::Mutex<Option<PathBuf>>,
5289    }
5290
5291    impl AcquisitionRecorder {
5292        fn replacing_dir_on_started(path: PathBuf) -> Self {
5293            Self {
5294                events: std::sync::Mutex::new(Vec::new()),
5295                replace_dir_on_started: std::sync::Mutex::new(Some(path)),
5296            }
5297        }
5298
5299        fn events(&self) -> Vec<DownloadEvent> {
5300            self.events.lock().unwrap().clone()
5301        }
5302    }
5303
5304    impl crate::download::DownloadProgress for AcquisitionRecorder {
5305        fn on_event(&self, event: &DownloadEvent) {
5306            self.events.lock().unwrap().push(event.clone());
5307            if matches!(event, DownloadEvent::Started { .. }) {
5308                if let Some(path) = self.replace_dir_on_started.lock().unwrap().take() {
5309                    std::fs::remove_dir_all(&path).unwrap();
5310                    std::fs::write(path, b"make create_dir_all fail before any network access")
5311                        .unwrap();
5312                }
5313            }
5314        }
5315    }
5316
5317    fn started_count(events: &[DownloadEvent]) -> usize {
5318        events
5319            .iter()
5320            .filter(|event| matches!(event, DownloadEvent::Started { .. }))
5321            .count()
5322    }
5323
5324    fn mlx_schema(id: &str, name: &str, hf_repo: &str) -> ModelSchema {
5325        test_generate_schema(
5326            id,
5327            name,
5328            ModelSource::Mlx {
5329                hf_repo: hf_repo.into(),
5330                hf_weight_file: None,
5331            },
5332        )
5333    }
5334
5335    #[test]
5336    fn installing_a_fetched_file_replaces_only_unusable_destinations() {
5337        let tmp = TempDir::new().unwrap();
5338        let src = tmp.path().join("fetched");
5339        std::fs::write(&src, b"fetched bytes").unwrap();
5340
5341        let missing = tmp.path().join("missing");
5342        install_fetched_file(&src, &missing).unwrap();
5343        assert_eq!(std::fs::read(&missing).unwrap(), b"fetched bytes");
5344
5345        let usable = tmp.path().join("usable");
5346        std::fs::write(&usable, b"keep these bytes").unwrap();
5347        let usable_before = std::fs::symlink_metadata(&usable).unwrap();
5348        let modified_before = usable_before.modified().unwrap();
5349        #[cfg(unix)]
5350        let inode_before = {
5351            use std::os::unix::fs::MetadataExt;
5352            usable_before.ino()
5353        };
5354        install_fetched_file(&src, &usable).unwrap();
5355        let usable_after = std::fs::symlink_metadata(&usable).unwrap();
5356        assert_eq!(std::fs::read(&usable).unwrap(), b"keep these bytes");
5357        assert_eq!(usable_after.modified().unwrap(), modified_before);
5358        #[cfg(unix)]
5359        {
5360            use std::os::unix::fs::MetadataExt;
5361            assert_eq!(usable_after.ino(), inode_before);
5362        }
5363
5364        let zero_byte = tmp.path().join("zero-byte");
5365        std::fs::write(&zero_byte, b"").unwrap();
5366        install_fetched_file(&src, &zero_byte).unwrap();
5367        assert_eq!(std::fs::read(&zero_byte).unwrap(), b"fetched bytes");
5368
5369        #[cfg(unix)]
5370        {
5371            let dangling = tmp.path().join("dangling");
5372            std::os::unix::fs::symlink(tmp.path().join("absent"), &dangling).unwrap();
5373            assert!(std::fs::symlink_metadata(&dangling).unwrap().is_symlink());
5374            install_fetched_file(&src, &dangling).unwrap();
5375            assert_eq!(std::fs::read(&dangling).unwrap(), b"fetched bytes");
5376        }
5377    }
5378
5379    // The fixture redirects the cache through env vars and unix paths; keep it
5380    // out of nightly Windows lib tests.
5381    #[cfg(unix)]
5382    #[tokio::test]
5383    async fn a_zero_byte_flux_auxiliary_file_is_not_accepted_as_present() {
5384        let _environment = crate::openrouter::test_environment_scope_async().await;
5385        let tmp = TempDir::new().unwrap();
5386        let home = tmp.path().join("home");
5387        let _home = ScopedEnvVar::set("HOME", &home);
5388        let _hf_home = ScopedEnvVar::set("HF_HOME", &tmp.path().join("hf-home"));
5389        let _hub = ScopedEnvVar::unset("HF_HUB_CACHE");
5390        let _legacy_hub = ScopedEnvVar::unset("HUGGINGFACE_HUB_CACHE");
5391
5392        // `download_file` writes the resolved hub — `HF_HOME/hub` here, written
5393        // out literally so the fixture does not borrow the code under test.
5394        // Populate the exact base-model pointer it asks for, fully offline.
5395        let cached_auxiliary = tmp
5396            .path()
5397            .join("hf-home/hub")
5398            .join("models--Freepik--flux.1-lite-8B")
5399            .join("snapshots/fixture/tokenizer_2/tokenizer.json");
5400        std::fs::create_dir_all(cached_auxiliary.parent().unwrap()).unwrap();
5401        std::fs::write(&cached_auxiliary, b"repaired tokenizer").unwrap();
5402        let refs = tmp
5403            .path()
5404            .join("hf-home/hub/models--Freepik--flux.1-lite-8B/refs");
5405        std::fs::create_dir_all(&refs).unwrap();
5406        std::fs::write(refs.join("main"), b"fixture").unwrap();
5407
5408        let models = tmp.path().join("models");
5409        let name = "Flux-1.lite-8B-MLX-Q4";
5410        write_mlx_dir(&models, name, "flux");
5411        let auxiliary = models.join(name).join("tokenizer_2/tokenizer.json");
5412        std::fs::create_dir_all(auxiliary.parent().unwrap()).unwrap();
5413        std::fs::write(&auxiliary, b"").unwrap();
5414        let mut reg = UnifiedRegistry::new_empty(models.clone());
5415        reg.register(mlx_schema(
5416            "mlx/flux-zero-byte-aux",
5417            name,
5418            "mlx-community/Flux-1.lite-8B-MLX-Q4",
5419        ));
5420        let recorder = std::sync::Arc::new(AcquisitionRecorder::default());
5421        let sink = ProgressSink::new(recorder.clone());
5422
5423        let result = reg
5424            .acquire_and_ensure("mlx/flux-zero-byte-aux", &sink, false, None)
5425            .await;
5426        let events = recorder.events();
5427        assert!(matches!(
5428            events.first(),
5429            Some(DownloadEvent::Started { .. })
5430        ));
5431        assert_eq!(started_count(&events), 1);
5432        assert!(
5433            result.is_err() || crate::download::cache_file_usable(&auxiliary),
5434            "acquisition must replace the zero-byte auxiliary file or report failure"
5435        );
5436        if result.is_ok() {
5437            assert_eq!(std::fs::read(auxiliary).unwrap(), b"repaired tokenizer");
5438        }
5439        // The auxiliary came from a file the hub already held: nothing was
5440        // fetched, so nothing is recorded as CAR's.
5441        assert!(
5442            !tmp.path()
5443                .join("model-management")
5444                .join(crate::retire::PROVENANCE_FILE)
5445                .exists(),
5446            "reusing a cached file must not record it as a CAR download"
5447        );
5448    }
5449
5450    // The fixture redirects the cache through env vars and unix paths; keep it
5451    // out of nightly Windows lib tests.
5452    #[cfg(unix)]
5453    #[tokio::test]
5454    async fn managed_flux_missing_auxiliary_does_not_fall_through_to_hf_snapshot() {
5455        let _environment = crate::openrouter::test_environment_scope_async().await;
5456        let tmp = TempDir::new().unwrap();
5457        let home = tmp.path().join("home");
5458        let hf_home = tmp.path().join("hf-home");
5459        let _home = ScopedEnvVar::set("HOME", &home);
5460        let _hf_home = ScopedEnvVar::set("HF_HOME", &hf_home);
5461        let _hub = ScopedEnvVar::unset("HF_HUB_CACHE");
5462        let _legacy_hub = ScopedEnvVar::unset("HUGGINGFACE_HUB_CACHE");
5463
5464        // The managed repair fetches Flux's auxiliary from the base model.
5465        // Populate the resolved hub (`HF_HOME/hub`, written out literally) so
5466        // that repair succeeds offline.
5467        let cached_auxiliary = hf_home
5468            .join("hub")
5469            .join("models--Freepik--flux.1-lite-8B")
5470            .join("snapshots/fixture/tokenizer_2/tokenizer.json");
5471        std::fs::create_dir_all(cached_auxiliary.parent().unwrap()).unwrap();
5472        std::fs::write(&cached_auxiliary, b"managed repair tokenizer").unwrap();
5473        let refs = hf_home.join("hub/models--Freepik--flux.1-lite-8B/refs");
5474        std::fs::create_dir_all(&refs).unwrap();
5475        std::fs::write(refs.join("main"), b"fixture").unwrap();
5476
5477        let models = tmp.path().join("models");
5478        let name = "Flux-1.lite-8B-MLX-Q4";
5479        let managed = models.join(name);
5480        write_mlx_dir(&models, name, "flux");
5481        let managed_auxiliary = managed.join("tokenizer_2/tokenizer.json");
5482
5483        // Also provide a fully reusable shared snapshot. The managed install
5484        // must still be repaired and returned, preserving base precedence.
5485        let snapshot = hf_home
5486            .join("hub/models--mlx-community--Flux-1.lite-8B-MLX-Q4")
5487            .join("snapshots/fixture");
5488        write_complete_mlx_snapshot(&snapshot);
5489        let snapshot_auxiliary = snapshot.join("tokenizer_2/tokenizer.json");
5490        std::fs::create_dir_all(snapshot_auxiliary.parent().unwrap()).unwrap();
5491        std::fs::write(&snapshot_auxiliary, b"snapshot tokenizer").unwrap();
5492
5493        let id = "mlx/flux-managed-precedence";
5494        let mut reg = UnifiedRegistry::new_empty(models);
5495        reg.register(mlx_schema(id, name, "mlx-community/Flux-1.lite-8B-MLX-Q4"));
5496        let recorder = std::sync::Arc::new(AcquisitionRecorder::default());
5497        let sink = ProgressSink::new(recorder.clone());
5498
5499        let path = reg
5500            .acquire_and_ensure(id, &sink, false, None)
5501            .await
5502            .unwrap();
5503
5504        assert_eq!(path, managed);
5505        assert_eq!(
5506            std::fs::read(managed_auxiliary).unwrap(),
5507            b"managed repair tokenizer"
5508        );
5509        let events = recorder.events();
5510        assert!(matches!(
5511            events.first(),
5512            Some(DownloadEvent::Started { .. })
5513        ));
5514        assert_eq!(started_count(&events), 1);
5515    }
5516
5517    #[tokio::test]
5518    async fn reusing_a_complete_managed_mlx_dir_emits_no_acquisition_lifecycle() {
5519        let _environment = crate::openrouter::test_environment_scope_async().await;
5520        let tmp = TempDir::new().unwrap();
5521        let _hf_home = ScopedEnvVar::set("HF_HOME", &tmp.path().join("hf-home"));
5522        let _hub = ScopedEnvVar::unset("HF_HUB_CACHE");
5523        let _legacy_hub = ScopedEnvVar::unset("HUGGINGFACE_HUB_CACHE");
5524        let models = tmp.path().join("models");
5525        let mut reg = UnifiedRegistry::new_empty(models.clone());
5526        reg.register(mlx_schema(
5527            "mlx/reuse-complete",
5528            "Reuse-Complete-MLX",
5529            "example/reuse-complete",
5530        ));
5531        write_mlx_dir(&models, "Reuse-Complete-MLX", "qwen3");
5532        std::fs::write(
5533            models.join("Reuse-Complete-MLX").join("tokenizer.json"),
5534            b"{}",
5535        )
5536        .unwrap();
5537        let recorder = std::sync::Arc::new(AcquisitionRecorder::default());
5538        let sink = ProgressSink::new(recorder.clone());
5539
5540        let path = reg
5541            .ensure_local_with_progress("mlx/reuse-complete", &sink)
5542            .await
5543            .unwrap();
5544
5545        assert_eq!(path, models.join("Reuse-Complete-MLX"));
5546        assert!(recorder.events().is_empty());
5547    }
5548
5549    #[tokio::test]
5550    async fn reusing_a_managed_mlx_dir_without_tokenizer_json_emits_no_lifecycle() {
5551        let _environment = crate::openrouter::test_environment_scope_async().await;
5552        let tmp = TempDir::new().unwrap();
5553        let _hf_home = ScopedEnvVar::set("HF_HOME", &tmp.path().join("hf-home"));
5554        let _hub = ScopedEnvVar::unset("HF_HUB_CACHE");
5555        let _legacy_hub = ScopedEnvVar::unset("HUGGINGFACE_HUB_CACHE");
5556        let models = tmp.path().join("models");
5557        let mut reg = UnifiedRegistry::new_empty(models.clone());
5558        reg.register(mlx_schema(
5559            "mlx/reuse-no-tokenizer",
5560            "Reuse-No-Tokenizer-MLX",
5561            "example/reuse-no-tokenizer",
5562        ));
5563        write_mlx_dir(&models, "Reuse-No-Tokenizer-MLX", "qwen3");
5564        let recorder = std::sync::Arc::new(AcquisitionRecorder::default());
5565        let sink = ProgressSink::new(recorder.clone());
5566
5567        let path = reg
5568            .ensure_local_with_progress("mlx/reuse-no-tokenizer", &sink)
5569            .await
5570            .unwrap();
5571
5572        assert_eq!(path, models.join("Reuse-No-Tokenizer-MLX"));
5573        assert!(recorder.events().is_empty());
5574    }
5575
5576    #[tokio::test]
5577    async fn reusing_a_complete_hf_snapshot_emits_no_acquisition_lifecycle() {
5578        let _environment = crate::openrouter::test_environment_scope_async().await;
5579        let tmp = TempDir::new().unwrap();
5580        let hf_home = tmp.path().join("hf-home");
5581        let _hf_home = ScopedEnvVar::set("HF_HOME", &hf_home);
5582        let _hub = ScopedEnvVar::unset("HF_HUB_CACHE");
5583        let _legacy_hub = ScopedEnvVar::unset("HUGGINGFACE_HUB_CACHE");
5584        let models = tmp.path().join("models");
5585        let repo = "example/reuse-hf-snapshot";
5586        let snapshot = hf_home
5587            .join("hub")
5588            .join("models--example--reuse-hf-snapshot")
5589            .join("snapshots")
5590            .join("fixture");
5591        write_complete_mlx_snapshot(&snapshot);
5592        let mut reg = UnifiedRegistry::new_empty(models);
5593        reg.register(mlx_schema("mlx/reuse-hf", "Reuse-HF-MLX", repo));
5594        let recorder = std::sync::Arc::new(AcquisitionRecorder::default());
5595        let sink = ProgressSink::new(recorder.clone());
5596
5597        let path = reg
5598            .ensure_local_with_progress("mlx/reuse-hf", &sink)
5599            .await
5600            .unwrap();
5601
5602        assert_eq!(path, snapshot);
5603        assert!(recorder.events().is_empty());
5604    }
5605
5606    #[tokio::test]
5607    async fn a_zero_byte_managed_config_json_starts_acquisition() {
5608        let _environment = crate::openrouter::test_environment_scope_async().await;
5609        let tmp = TempDir::new().unwrap();
5610        let _hf_home = ScopedEnvVar::set("HF_HOME", &tmp.path().join("hf-home"));
5611        let _hub = ScopedEnvVar::unset("HF_HUB_CACHE");
5612        let _legacy_hub = ScopedEnvVar::unset("HUGGINGFACE_HUB_CACHE");
5613        let models = tmp.path().join("models");
5614        let name = "Zero-Byte-Config-MLX";
5615        write_mlx_dir(&models, name, "qwen3");
5616        let model_dir = models.join(name);
5617        std::fs::write(model_dir.join("tokenizer.json"), b"{}").unwrap();
5618        std::fs::write(model_dir.join("config.json"), b"").unwrap();
5619        let mut reg = UnifiedRegistry::new_empty(models);
5620        reg.register(mlx_schema(
5621            "mlx/zero-byte-config",
5622            name,
5623            "example/zero-byte-config",
5624        ));
5625        let recorder =
5626            std::sync::Arc::new(AcquisitionRecorder::replacing_dir_on_started(model_dir));
5627        let sink = ProgressSink::new(recorder.clone());
5628
5629        let result = reg
5630            .ensure_local_with_progress("mlx/zero-byte-config", &sink)
5631            .await;
5632
5633        let events = recorder.events();
5634        assert!(
5635            matches!(events.first(), Some(DownloadEvent::Started { .. })),
5636            "weights beside a zero-byte config.json must start an acquisition, not be reused; events: {events:?}"
5637        );
5638        assert_eq!(started_count(&events), 1);
5639        assert!(result.is_err());
5640    }
5641
5642    #[tokio::test]
5643    async fn a_zero_byte_hf_snapshot_config_json_starts_acquisition() {
5644        let _environment = crate::openrouter::test_environment_scope_async().await;
5645        let tmp = TempDir::new().unwrap();
5646        let hf_home = tmp.path().join("hf-home");
5647        let _hf_home = ScopedEnvVar::set("HF_HOME", &hf_home);
5648        let _hub = ScopedEnvVar::unset("HF_HUB_CACHE");
5649        let _legacy_hub = ScopedEnvVar::unset("HUGGINGFACE_HUB_CACHE");
5650        let models = tmp.path().join("models");
5651        let name = "Zero-Byte-Snapshot-Config-MLX";
5652        let repo = "example/zero-byte-snapshot-config";
5653        let snapshot = hf_home
5654            .join("hub")
5655            .join("models--example--zero-byte-snapshot-config")
5656            .join("snapshots")
5657            .join("fixture");
5658        write_complete_mlx_snapshot(&snapshot);
5659        std::fs::write(snapshot.join("config.json"), b"").unwrap();
5660        // An empty managed directory leaves the snapshot as the only reuse
5661        // candidate.
5662        let model_dir = models.join(name);
5663        std::fs::create_dir_all(&model_dir).unwrap();
5664        let mut reg = UnifiedRegistry::new_empty(models);
5665        reg.register(mlx_schema("mlx/zero-byte-snapshot-config", name, repo));
5666        let recorder =
5667            std::sync::Arc::new(AcquisitionRecorder::replacing_dir_on_started(model_dir));
5668        let sink = ProgressSink::new(recorder.clone());
5669
5670        let result = reg
5671            .ensure_local_with_progress("mlx/zero-byte-snapshot-config", &sink)
5672            .await;
5673
5674        let events = recorder.events();
5675        assert!(
5676            matches!(events.first(), Some(DownloadEvent::Started { .. })),
5677            "a snapshot with a zero-byte config.json must start an acquisition, not be reused; events: {events:?}"
5678        );
5679        assert_eq!(started_count(&events), 1);
5680        assert!(result.is_err());
5681    }
5682
5683    #[cfg(unix)]
5684    #[tokio::test]
5685    async fn an_oversized_installed_model_is_reused_below_the_disk_threshold() {
5686        let _environment = crate::openrouter::test_environment_scope_async().await;
5687        let tmp = TempDir::new().unwrap();
5688        let _hf_home = ScopedEnvVar::set("HF_HOME", &tmp.path().join("hf-home"));
5689        let _hub = ScopedEnvVar::unset("HF_HUB_CACHE");
5690        let _legacy_hub = ScopedEnvVar::unset("HUGGINGFACE_HUB_CACHE");
5691        let models = tmp.path().join("models");
5692        let mut schema = mlx_schema(
5693            "mlx/reuse-oversized",
5694            "Reuse-Oversized-MLX",
5695            "example/reuse-oversized",
5696        );
5697        // One petabyte in MB: safely representable by the canonical catalog
5698        // identity while still exceeding any machine this test can run on.
5699        schema.cost.size_mb = Some(1_000_000_000);
5700        let mut reg = UnifiedRegistry::new_empty(models.clone());
5701        reg.register(schema);
5702        write_mlx_dir(&models, "Reuse-Oversized-MLX", "qwen3");
5703        let recorder = std::sync::Arc::new(AcquisitionRecorder::default());
5704        let sink = ProgressSink::new(recorder.clone());
5705
5706        let path = reg
5707            .ensure_local_with_progress("mlx/reuse-oversized", &sink)
5708            .await
5709            .unwrap();
5710
5711        assert_eq!(path, models.join("Reuse-Oversized-MLX"));
5712        assert!(recorder.events().is_empty());
5713    }
5714
5715    #[tokio::test]
5716    async fn a_missing_shard_starts_acquisition_exactly_once_after_the_preflight() {
5717        let _environment = crate::openrouter::test_environment_scope_async().await;
5718        let tmp = TempDir::new().unwrap();
5719        let _hf_home = ScopedEnvVar::set("HF_HOME", &tmp.path().join("hf-home"));
5720        let _hub = ScopedEnvVar::unset("HF_HUB_CACHE");
5721        let _legacy_hub = ScopedEnvVar::unset("HUGGINGFACE_HUB_CACHE");
5722        let models = tmp.path().join("models");
5723        let model_dir = models.join("Missing-Shard-MLX");
5724        std::fs::create_dir_all(&model_dir).unwrap();
5725        std::fs::write(model_dir.join("config.json"), b"{}").unwrap();
5726        std::fs::write(
5727            model_dir.join("model.safetensors.index.json"),
5728            r#"{"weight_map":{"a":"model-00001-of-00002.safetensors","b":"model-00002-of-00002.safetensors"}}"#,
5729        )
5730        .unwrap();
5731        std::fs::write(model_dir.join("model-00002-of-00002.safetensors"), b"two").unwrap();
5732        let mut reg = UnifiedRegistry::new_empty(models);
5733        reg.register(mlx_schema(
5734            "mlx/missing-shard-lifecycle",
5735            "Missing-Shard-MLX",
5736            "example/missing-shard",
5737        ));
5738        let recorder =
5739            std::sync::Arc::new(AcquisitionRecorder::replacing_dir_on_started(model_dir));
5740        let sink = ProgressSink::new(recorder.clone());
5741
5742        assert!(reg
5743            .ensure_local_with_progress("mlx/missing-shard-lifecycle", &sink)
5744            .await
5745            .is_err());
5746        let events = recorder.events();
5747        assert!(matches!(
5748            events.first(),
5749            Some(DownloadEvent::Started { .. })
5750        ));
5751        assert_eq!(started_count(&events), 1);
5752    }
5753
5754    #[tokio::test]
5755    async fn a_flux_dir_missing_its_auxiliary_tokenizer_keeps_the_lifecycle() {
5756        let _environment = crate::openrouter::test_environment_scope_async().await;
5757        let tmp = TempDir::new().unwrap();
5758        let _hf_home = ScopedEnvVar::set("HF_HOME", &tmp.path().join("hf-home"));
5759        let _hub = ScopedEnvVar::unset("HF_HUB_CACHE");
5760        let _legacy_hub = ScopedEnvVar::unset("HUGGINGFACE_HUB_CACHE");
5761        let models = tmp.path().join("models");
5762        let name = "Flux-1.lite-8B-MLX-Q4";
5763        write_mlx_dir(&models, name, "flux");
5764        std::fs::write(models.join(name).join("tokenizer_2"), b"not a directory").unwrap();
5765        let mut reg = UnifiedRegistry::new_empty(models);
5766        reg.register(mlx_schema(
5767            "mlx/flux-missing-aux",
5768            name,
5769            "mlx-community/Flux-1.lite-8B-MLX-Q4",
5770        ));
5771        let recorder = std::sync::Arc::new(AcquisitionRecorder::default());
5772        let sink = ProgressSink::new(recorder.clone());
5773
5774        assert!(reg
5775            .ensure_local_with_progress("mlx/flux-missing-aux", &sink)
5776            .await
5777            .is_err());
5778        let events = recorder.events();
5779        assert!(matches!(
5780            events.first(),
5781            Some(DownloadEvent::Started { .. })
5782        ));
5783        assert_eq!(started_count(&events), 1);
5784    }
5785
5786    #[tokio::test]
5787    async fn force_bypasses_reuse() {
5788        let _environment = crate::openrouter::test_environment_scope_async().await;
5789        let tmp = TempDir::new().unwrap();
5790        let models = tmp.path().join("models");
5791        let name = "Force-Local";
5792        let model_dir = models.join(name);
5793        std::fs::create_dir_all(&model_dir).unwrap();
5794        std::fs::write(model_dir.join("model.gguf"), b"weights").unwrap();
5795        std::fs::write(model_dir.join("tokenizer.json"), b"{}").unwrap();
5796        let mut reg = UnifiedRegistry::new_empty(models);
5797        reg.register(test_generate_schema(
5798            "local/force-lifecycle",
5799            name,
5800            ModelSource::Local {
5801                hf_repo: String::new(),
5802                hf_filename: "model.gguf".into(),
5803                tokenizer_repo: String::new(),
5804            },
5805        ));
5806        let recorder = std::sync::Arc::new(AcquisitionRecorder::default());
5807        let sink = ProgressSink::new(recorder.clone());
5808
5809        assert!(reg
5810            .acquire_and_ensure("local/force-lifecycle", &sink, true, None)
5811            .await
5812            .is_err());
5813        let events = recorder.events();
5814        assert!(matches!(
5815            events.first(),
5816            Some(DownloadEvent::Started { .. })
5817        ));
5818        assert_eq!(started_count(&events), 1);
5819    }
5820
5821    #[tokio::test]
5822    async fn staged_pulls_keep_the_lifecycle() {
5823        let _environment = crate::openrouter::test_environment_scope_async().await;
5824        let tmp = TempDir::new().unwrap();
5825        let models = tmp.path().join("models");
5826        let canonical = models.join("Staged-Local");
5827        std::fs::create_dir_all(&canonical).unwrap();
5828        std::fs::write(canonical.join("model.gguf"), b"canonical weights").unwrap();
5829        std::fs::write(canonical.join("tokenizer.json"), b"{}").unwrap();
5830        let mut reg = UnifiedRegistry::new_empty(models);
5831        reg.register(test_generate_schema(
5832            "local/staged-lifecycle",
5833            "Staged-Local",
5834            ModelSource::Local {
5835                hf_repo: String::new(),
5836                hf_filename: "model.gguf".into(),
5837                tokenizer_repo: String::new(),
5838            },
5839        ));
5840        let staging = tmp.path().join("staging");
5841        std::fs::create_dir_all(&staging).unwrap();
5842        std::fs::write(staging.join("model.gguf"), b"weights").unwrap();
5843        std::fs::write(staging.join("tokenizer.json"), b"{}").unwrap();
5844        let recorder = std::sync::Arc::new(AcquisitionRecorder::default());
5845        let sink = ProgressSink::new(recorder.clone());
5846
5847        let path = reg
5848            .acquire_and_ensure("local/staged-lifecycle", &sink, false, Some(&staging))
5849            .await
5850            .unwrap();
5851        assert_eq!(path, staging);
5852        let events = recorder.events();
5853        assert!(matches!(
5854            events.first(),
5855            Some(DownloadEvent::Started { .. })
5856        ));
5857        assert_eq!(started_count(&events), 1);
5858    }
5859
5860    #[test]
5861    fn synthesize_local_schema_classifies_by_name_and_arch() {
5862        let tmp = TempDir::new().unwrap();
5863        let root = tmp.path();
5864
5865        write_mlx_dir(root, "MyCustom-Qwen3-7B", "qwen3");
5866        let gen = synthesize_local_schema("MyCustom-Qwen3-7B", &root.join("MyCustom-Qwen3-7B"))
5867            .expect("text LLM should be recognized");
5868        assert_eq!(
5869            gen.capabilities,
5870            vec![
5871                ModelCapability::Generate,
5872                ModelCapability::Code,
5873                ModelCapability::Reasoning
5874            ]
5875        );
5876        assert_eq!(gen.context_length, 40_960);
5877        assert_eq!(gen.provider, "local");
5878        assert!(matches!(gen.source, ModelSource::Mlx { .. }));
5879
5880        write_mlx_dir(root, "Some-Embedding-0.6B", "qwen3");
5881        let emb = synthesize_local_schema("Some-Embedding-0.6B", &root.join("Some-Embedding-0.6B"))
5882            .expect("embedding model recognized");
5883        assert_eq!(emb.capabilities, vec![ModelCapability::Embed]);
5884
5885        // Unknown architecture: refuse rather than guess.
5886        write_mlx_dir(root, "Mystery-Net", "some_unknown_arch");
5887        assert!(synthesize_local_schema("Mystery-Net", &root.join("Mystery-Net")).is_none());
5888
5889        // Speech/vision/etc. names are skipped even with a valid LLM config.
5890        write_mlx_dir(root, "silero-vad-v6-mlx", "qwen3");
5891        assert!(
5892            synthesize_local_schema("silero-vad-v6-mlx", &root.join("silero-vad-v6-mlx")).is_none()
5893        );
5894
5895        // A bare directory with no weights is not a model.
5896        std::fs::create_dir_all(root.join("empty")).unwrap();
5897        assert!(synthesize_local_schema("empty", &root.join("empty")).is_none());
5898    }
5899
5900    /// A GGUF directory found by the scan carries no `hf_repo` — its weights
5901    /// are whatever the user put there. When the layout is not what the Candle
5902    /// backend reads, CAR used to attempt a download against the empty repo and
5903    /// report whatever `https://huggingface.co//…` returned, which explains
5904    /// nothing about the actual problem.
5905    #[test]
5906    fn a_scanned_gguf_directory_diagnoses_instead_of_downloading_from_nowhere() {
5907        let _environment = crate::openrouter::test_environment_scope();
5908        let tmp = TempDir::new().unwrap();
5909        let models = tmp.path().join("models");
5910        let dir = models.join("Dropped-In-Llama");
5911        std::fs::create_dir_all(&dir).unwrap();
5912        // A real user drop-in: the file keeps the name it was published under.
5913        std::fs::write(dir.join("Llama-3-8B-Q4_K_M.gguf"), b"weights").unwrap();
5914
5915        // Constructed outside the runtime: `new_with_state_root` ends in
5916        // `refresh_availability`, which blocks.
5917        let reg = UnifiedRegistry::new_with_state_root(tmp.path().to_path_buf(), models);
5918        let err = tokio::runtime::Runtime::new()
5919            .unwrap()
5920            .block_on(reg.ensure_local("Dropped-In-Llama"))
5921            .expect_err("an unloadable layout must not report success");
5922        let err = err.to_string();
5923
5924        assert!(err.contains("model.gguf"), "must name what it reads: {err}");
5925        assert!(
5926            err.contains("tokenizer.json"),
5927            "must name the missing tokenizer too: {err}"
5928        );
5929        assert!(
5930            !err.contains("huggingface.co//"),
5931            "must not have tried to fetch from an empty repo: {err}"
5932        );
5933    }
5934
5935    #[test]
5936    fn discovery_registers_uncatalogued_local_model() {
5937        // Constructing a `UnifiedRegistry` ends in `refresh_availability()`,
5938        // which WRITES the process-global gateway observation. Full reasoning
5939        // on `TestRegistry` in this crate; family in
5940        // docs/solutions/process-global-state-in-tests.md.
5941        let _environment = crate::openrouter::test_environment_scope();
5942        let tmp = TempDir::new().unwrap();
5943        let models = tmp.path().join("models");
5944        std::fs::create_dir_all(&models).unwrap();
5945        write_mlx_dir(&models, "Totally-Custom-Llama-3B", "llama");
5946
5947        let reg = UnifiedRegistry::new_with_state_root(tmp.path().to_path_buf(), models);
5948        let found = reg
5949            .list()
5950            .into_iter()
5951            .find(|m| m.name == "Totally-Custom-Llama-3B");
5952        assert!(
5953            found.is_some(),
5954            "uncatalogued on-disk model should be registered"
5955        );
5956        assert!(found.unwrap().tags.iter().any(|t| t == "auto-discovered"));
5957    }
5958
5959    #[test]
5960    fn explicit_user_row_is_not_pruned_by_auto_discovery_tag() {
5961        let _environment = crate::openrouter::test_environment_scope();
5962        let tmp = TempDir::new().unwrap();
5963        let models = tmp.path().join("models");
5964        let model_dir = models.join("Explicit-Custom-Llama");
5965        write_mlx_dir(&models, "Explicit-Custom-Llama", "llama");
5966
5967        let mut registry = UnifiedRegistry::new_with_state_root(tmp.path().to_path_buf(), models);
5968        let model_id = "local/explicit-custom-llama";
5969        let schema = registry.get(model_id).cloned().unwrap();
5970        assert!(schema.tags.iter().any(|tag| tag == "auto-discovered"));
5971
5972        registry.register_user_model(schema);
5973        std::fs::remove_dir_all(model_dir).unwrap();
5974        registry.prune_missing_on_disk_models();
5975
5976        assert!(
5977            registry.get(model_id).is_some(),
5978            "explicit registration provenance must outrank a user-controlled tag"
5979        );
5980    }
5981
5982    #[test]
5983    fn signed_catalog_cannot_shadow_builtin_exact_id() {
5984        // Constructing a `UnifiedRegistry` ends in `refresh_availability()`,
5985        // which WRITES the process-global gateway observation. Full reasoning
5986        // on `TestRegistry` in this crate; family in
5987        // docs/solutions/process-global-state-in-tests.md.
5988        let _environment = crate::openrouter::test_environment_scope();
5989        // A signature authenticates the catalog publisher, but it does not
5990        // authorize changing a project-owned exact id compiled into this CAR
5991        // build. An additive signed row may not shadow a builtin.
5992        let tmp = TempDir::new().unwrap();
5993        let models_dir = tmp.path().join("models");
5994
5995        let builtin = builtin_catalog();
5996        let mut overriding = builtin.first().expect("a built-in model").clone();
5997        let target_id = overriding.id.clone();
5998        overriding.name = "REPLACED-BY-CATALOG".into();
5999
6000        let (verified, public_key) = crate::catalog::signed_test_catalog(
6001            crate::catalog::CatalogDoc {
6002                revoked: Vec::new(),
6003                version: 1,
6004                models: vec![overriding],
6005            },
6006            51,
6007        );
6008        crate::catalog::save_verified(&crate::catalog::cache_path(tmp.path()), &verified).unwrap();
6009
6010        // new() loads the verified cache additively after the built-ins.
6011        let reg = UnifiedRegistry::new_with_catalog_public_key(
6012            tmp.path().to_path_buf(),
6013            models_dir,
6014            Some(public_key.as_str()),
6015        );
6016        assert_eq!(
6017            reg.get(&target_id).map(|m| m.name.as_str()),
6018            Some(builtin.first().unwrap().name.as_str()),
6019            "a signed cache row must not replace a builtin exact id"
6020        );
6021    }
6022
6023    #[test]
6024    fn user_model_cannot_shadow_project_owned_exact_id() {
6025        let mut reg = test_registry();
6026        let original = builtin_catalog().first().expect("a builtin").clone();
6027        let mut forged = original.clone();
6028        forged.name = "USER-SHADOW".into();
6029
6030        reg.register_user_model(forged);
6031
6032        assert_eq!(
6033            reg.get(&original.id).map(|model| model.name.as_str()),
6034            Some(original.name.as_str()),
6035            "a user row must not replace a project-owned exact id"
6036        );
6037    }
6038
6039    #[test]
6040    fn legacy_unsigned_catalog_cache_cannot_replace_builtin() {
6041        // Constructing a `UnifiedRegistry` ends in `refresh_availability()`,
6042        // which WRITES the process-global gateway observation. Full reasoning
6043        // on `TestRegistry` in this crate; family in
6044        // docs/solutions/process-global-state-in-tests.md.
6045        let _environment = crate::openrouter::test_environment_scope();
6046        let tmp = TempDir::new().unwrap();
6047        let models_dir = tmp.path().join("models");
6048        let builtin = builtin_catalog();
6049        let original = builtin.first().expect("a built-in model");
6050        let mut forged = original.clone();
6051        forged.name = "FORGED-UNSIGNED-CATALOG".into();
6052        let path = crate::catalog::cache_path(tmp.path());
6053        std::fs::write(
6054            &path,
6055            serde_json::to_vec_pretty(&crate::catalog::CatalogDoc {
6056                revoked: Vec::new(),
6057                version: u64::MAX,
6058                models: vec![forged],
6059            })
6060            .unwrap(),
6061        )
6062        .unwrap();
6063
6064        let reg = UnifiedRegistry::new_with_state_root(tmp.path().to_path_buf(), models_dir);
6065        assert_eq!(
6066            reg.get(&original.id).map(|model| model.name.as_str()),
6067            Some(original.name.as_str()),
6068            "legacy unsigned cache JSON must fail closed and preserve the built-in"
6069        );
6070    }
6071
6072    #[test]
6073    fn tampered_signed_managed_row_preserves_builtin() {
6074        // Constructing a `UnifiedRegistry` ends in `refresh_availability()`,
6075        // which WRITES the process-global gateway observation. Full reasoning
6076        // on `TestRegistry` in this crate; family in
6077        // docs/solutions/process-global-state-in-tests.md.
6078        let _environment = crate::openrouter::test_environment_scope();
6079        let tmp = TempDir::new().unwrap();
6080        let models_dir = tmp.path().join("models");
6081        let original = builtin_catalog()
6082            .into_iter()
6083            .find(|model| model.id == "parslee/openrouter/frontier-general")
6084            .expect("managed frontier alias");
6085        let mut forged = original.clone();
6086        forged.name = "SIGNED-THEN-TAMPERED-MANAGED".into();
6087        let (verified, public_key) = crate::catalog::signed_test_catalog(
6088            crate::catalog::CatalogDoc {
6089                revoked: Vec::new(),
6090                version: 9,
6091                models: vec![forged],
6092            },
6093            52,
6094        );
6095        let path = crate::catalog::cache_path(tmp.path());
6096        crate::catalog::save_verified(&path, &verified).unwrap();
6097        let cache = std::fs::read_to_string(&path)
6098            .unwrap()
6099            .replace("SIGNED-THEN-TAMPERED-MANAGED", "ATTACKER-MUTATION");
6100        std::fs::write(&path, cache).unwrap();
6101
6102        let reg = UnifiedRegistry::new_with_catalog_public_key(
6103            tmp.path().to_path_buf(),
6104            models_dir,
6105            Some(public_key.as_str()),
6106        );
6107        assert_eq!(
6108            reg.get(&original.id).map(|model| model.name.as_str()),
6109            Some(original.name.as_str()),
6110            "a tampered same-id managed row must fail verification and preserve the builtin"
6111        );
6112    }
6113
6114    #[test]
6115    fn builtin_catalog_loads() {
6116        let reg = test_registry();
6117        let all = reg.list();
6118        assert_eq!(all.len(), builtin_catalog().len());
6119    }
6120
6121    #[test]
6122    fn subscription_billing_lookup_uses_exact_or_implicit_latest_ids() {
6123        let codex_ids: Vec<_> = builtin_catalog()
6124            .into_iter()
6125            .filter(|model| model.id.starts_with("openai-codex/"))
6126            .map(|model| model.id)
6127            .collect();
6128        assert!(!codex_ids.is_empty());
6129        for id in codex_ids {
6130            assert!(is_builtin_subscription_billed(&id), "{id}");
6131            if let Some(untagged) = id.strip_suffix(":latest") {
6132                assert!(is_builtin_subscription_billed(untagged), "{untagged}");
6133            }
6134        }
6135        assert!(!is_builtin_subscription_billed("parslee/reasoning"));
6136        assert!(!is_builtin_subscription_billed("openai/gpt-5.5-2026-04-23"));
6137        assert!(!is_builtin_subscription_billed("unknown/not-a-model"));
6138    }
6139
6140    #[test]
6141    fn shipped_supervised_vllm_models_use_the_managed_source_contract() {
6142        let managed = builtin_catalog()
6143            .into_iter()
6144            .filter(|model| model.id.starts_with("vllm-mlx/"))
6145            .collect::<Vec<_>>();
6146        assert_eq!(managed.len(), 8);
6147        assert!(managed.iter().all(ModelSchema::is_car_managed_vllm_mlx));
6148        assert!(managed.iter().all(ModelSchema::downloads_weights));
6149    }
6150
6151    /// #137: a model tagged `requires-mlx-vlm` must report
6152    /// `available: true` if and only if `mlx_vlm_cli::is_available()`
6153    /// returns true. Without this, registry consumers (FFI
6154    /// `listModelsUnified`, the tray Models submenu, agent routing)
6155    /// see a model as available, the user picks it, and inference
6156    /// bails with `mlx-vlm CLI not found on PATH`.
6157    ///
6158    /// The probe is environmental — runs the same check on the host
6159    /// the test executes on. CI usually doesn't have `mlx_vlm`
6160    /// installed → expected unavailable; a dev box with it installed
6161    /// → expected available. Either way, the registry tracks the
6162    /// runtime probe.
6163    #[test]
6164    fn mlx_vlm_models_reflect_runtime_availability() {
6165        let reg = test_registry();
6166        let mlx_vlm_models: Vec<&ModelSchema> = reg
6167            .list()
6168            .into_iter()
6169            .filter(|m| m.tags.iter().any(|t| t == "requires-mlx-vlm"))
6170            .collect();
6171        assert!(
6172            !mlx_vlm_models.is_empty(),
6173            "catalog should contain at least one model tagged \
6174             `requires-mlx-vlm` — otherwise this regression has \
6175             nothing to guard"
6176        );
6177
6178        #[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
6179        let expected = crate::backend::mlx_vlm_cli::is_available();
6180        #[cfg(not(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx))))]
6181        let expected = false;
6182
6183        for m in mlx_vlm_models {
6184            assert_eq!(
6185                m.available, expected,
6186                "model {} `available` field should reflect \
6187                 mlx_vlm CLI presence (expected {expected}, got {})",
6188                m.id, m.available
6189            );
6190        }
6191    }
6192
6193    /// F1 (Parslee-ai/car#231 §7.1): MLX models must report
6194    /// `available: false` on any non-Apple-Silicon-with-MLX build target.
6195    /// Without this, the Windows / Linux / Intel-Mac router happily
6196    /// adds MLX entries to fallback chains and inference fails at
6197    /// dispatch with "model not found", producing a 7-deep cascade
6198    /// of useless errors on a fresh install.
6199    ///
6200    /// The test runs on every platform. On macOS arm64 (default-features),
6201    /// at least one MLX model with an `hf_repo` should report available;
6202    /// on Linux / Windows / Intel-Mac / `car_skip_mlx`, every plain MLX
6203    /// model must report unavailable.
6204    #[test]
6205    fn mlx_models_unavailable_on_non_mlx_targets() {
6206        let reg = test_registry();
6207        let mlx_models: Vec<&ModelSchema> = reg
6208            .list()
6209            .into_iter()
6210            .filter(|m| {
6211                m.is_mlx()
6212                    // Exclude `requires-mlx-vlm` and `speech` — those have
6213                    // their own availability logic (mlx_vlm CLI probe,
6214                    // speech_mlx_available). The plain MLX models are
6215                    // what §7.1 surfaced as broken on Windows.
6216                    && !m.tags.iter().any(|t| t == "requires-mlx-vlm")
6217                    && !m.tags.contains(&"speech".to_string())
6218            })
6219            .collect();
6220        assert!(
6221            !mlx_models.is_empty(),
6222            "catalog should contain at least one plain MLX model — \
6223             otherwise this F1 regression guard has nothing to guard"
6224        );
6225
6226        #[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
6227        {
6228            // On a real MLX target, models with an `hf_repo` should be
6229            // marked available (the `ensure_local()` lazy-download path).
6230            // At least one must qualify.
6231            let any_available = mlx_models.iter().any(|m| m.available);
6232            assert!(
6233                any_available,
6234                "on macOS arm64 with MLX enabled, at least one plain MLX \
6235                 model with hf_repo should be available — none were"
6236            );
6237        }
6238        #[cfg(not(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx))))]
6239        {
6240            // On every other target, MLX models cannot execute, so the
6241            // registry must mark them all unavailable.
6242            for m in &mlx_models {
6243                assert!(
6244                    !m.available,
6245                    "MLX model {} is marked available on a non-MLX target — \
6246                     the adaptive router will add it to fallback chains \
6247                     and dispatch will fail (Parslee-ai/car#231 §7.1)",
6248                    m.id
6249                );
6250            }
6251        }
6252    }
6253
6254    /// Embedded JSON must parse cleanly — if it doesn't, the runtime
6255    /// would panic on first registry load. Catch it in CI instead.
6256    /// The `provider/name[:tag]` form `InferenceModelIdentity::resolved_model_id`
6257    /// documents for built-in ids: each part is `[A-Za-z0-9][A-Za-z0-9._-]*`.
6258    fn is_builtin_model_id_form(id: &str) -> bool {
6259        fn part(s: &str) -> bool {
6260            let mut chars = s.chars();
6261            chars.next().is_some_and(|c| c.is_ascii_alphanumeric())
6262                && chars.all(|c| c.is_ascii_alphanumeric() || "._-".contains(c))
6263        }
6264        let (body, tag) = match id.split_once(':') {
6265            Some((body, tag)) => (body, Some(tag)),
6266            None => (id, None),
6267        };
6268        let Some((provider, name)) = body.split_once('/') else {
6269            return false;
6270        };
6271        part(provider) && part(name) && tag.is_none_or(part)
6272    }
6273
6274    #[test]
6275    fn builtin_model_ids_follow_the_resolved_model_id_contract() {
6276        for good in [
6277            "anthropic/claude-opus-4-8",
6278            "openai/gpt-5.5-2026-04-23",
6279            "qwen/qwen3-4b:q4_k_m",
6280        ] {
6281            assert!(is_builtin_model_id_form(good), "{good} should pass");
6282        }
6283        for bad in [
6284            "",
6285            "gpt-5.5",
6286            "/name",
6287            "openai/",
6288            "-openai/x",
6289            "openai/-x",
6290            "open ai/x",
6291            "openai/x y",
6292            "openai/é",
6293            "parslee/openrouter/open-fast",
6294            "openai/x:",
6295            "openai/x:-a",
6296            "openai/x:a b",
6297            "openai/x:a:b",
6298            "openai/x:a/b",
6299        ] {
6300            assert!(!is_builtin_model_id_form(bad), "{bad:?} should fail");
6301        }
6302        let catalog: Vec<ModelSchema> = serde_json::from_str(BUILTIN_CATALOG_JSON).unwrap();
6303        let off: Vec<&str> = catalog
6304            .iter()
6305            .map(|m| m.id.as_str())
6306            .filter(|id| !is_builtin_model_id_form(id))
6307            .collect();
6308        assert!(off.is_empty(), "built-in ids outside the contract: {off:?}");
6309    }
6310
6311    #[test]
6312    fn builtin_catalog_json_parses() {
6313        let catalog: Vec<ModelSchema> = serde_json::from_str(BUILTIN_CATALOG_JSON)
6314            .expect("builtin_catalog.json must be valid ModelSchema array");
6315        assert!(
6316            !catalog.is_empty(),
6317            "embedded catalog has no entries — that's almost certainly wrong"
6318        );
6319
6320        let mut seen = std::collections::HashSet::new();
6321        for entry in &catalog {
6322            assert!(
6323                seen.insert(entry.id.clone()),
6324                "duplicate id in builtin_catalog.json: {}",
6325                entry.id
6326            );
6327        }
6328    }
6329
6330    #[test]
6331    fn codex_subscription_row_has_pinnable_tool_proposal_identity() {
6332        let catalog: Vec<ModelSchema> = serde_json::from_str(BUILTIN_CATALOG_JSON).unwrap();
6333        let row = catalog
6334            .iter()
6335            .find(|model| model.id == "openai/gpt-5.6-sol:high")
6336            .expect("catalog publishes the Codex subscription identity");
6337        assert_eq!(row.name, "gpt-5.6-sol:high");
6338        assert!(matches!(
6339            &row.source,
6340            ModelSource::CodexCli { model } if model == "gpt-5.6-sol:high"
6341        ));
6342        assert!(row.has_capability(ModelCapability::Generate));
6343        assert!(row.has_capability(ModelCapability::Reasoning));
6344        assert!(row.has_capability(ModelCapability::ToolUse));
6345        assert!(row.has_capability(ModelCapability::MultiToolCall));
6346        assert!(!row.has_capability(ModelCapability::Vision));
6347        assert_eq!(row.supported_params, vec![GenerateParam::MaxTokens]);
6348        assert!(!row.downloads_weights());
6349    }
6350
6351    #[test]
6352    fn codex_subscription_rows_publish_the_gpt_5_6_family() {
6353        let catalog: Vec<ModelSchema> = serde_json::from_str(BUILTIN_CATALOG_JSON).unwrap();
6354        assert_eq!(
6355            catalog
6356                .iter()
6357                .filter(|model| model.provider == "openai-codex")
6358                .count(),
6359            6,
6360            "the explicit checks below must cover every openai-codex row"
6361        );
6362        for id in [
6363            "openai-codex/gpt-5.6-sol:latest",
6364            "openai-codex/gpt-5.6-sol:high",
6365            "openai-codex/gpt-5.6-luna:latest",
6366            "openai-codex/gpt-5.6-luna:high",
6367            "openai-codex/gpt-5.6-terra:latest",
6368            "openai-codex/gpt-5.6-terra:high",
6369        ] {
6370            let row = catalog
6371                .iter()
6372                .find(|model| model.id == id)
6373                .unwrap_or_else(|| panic!("missing {id}"));
6374            assert_eq!(row.provider, "openai-codex", "{id}");
6375            assert!(
6376                matches!(&row.source, ModelSource::Proprietary { provider, auth: ProprietaryAuth::ChatGptSubscription {}, .. } if provider == "openai-codex"),
6377                "{id}"
6378            );
6379            assert!(row.is_subscription_billed(), "{id}");
6380            assert!(
6381                row.cost.input_per_mtok.is_none() && row.cost.output_per_mtok.is_none(),
6382                "{id} must carry no dollar price"
6383            );
6384            assert!(row.tags.iter().any(|tag| tag == "subscription"), "{id}");
6385            assert!(
6386                row.cost
6387                    .estimated_usd_bounded(Some(1_000), 1_000, 1_000, 0, 0)
6388                    .is_none(),
6389                "{id} must not produce an approximate dollar cost"
6390            );
6391            let mut accidentally_priced = row.clone();
6392            accidentally_priced.cost.input_per_mtok = Some(1.0);
6393            accidentally_priced.cost.output_per_mtok = Some(1.0);
6394            assert!(
6395                crate::scoreboard_price_model(&accidentally_priced).is_none(),
6396                "{id} must stay out of the dollar scoreboard"
6397            );
6398            assert!(row.has_capability(ModelCapability::ToolUse), "{id}");
6399            assert_eq!(row.name.ends_with(":high"), id.ends_with(":high"), "{id}");
6400        }
6401        let cli = catalog
6402            .iter()
6403            .find(|model| model.id == "openai/gpt-5.6-sol:high")
6404            .unwrap();
6405        assert!(
6406            matches!(&cli.source, ModelSource::CodexCli { .. }),
6407            "the Codex CLI row stays as it was"
6408        );
6409        assert!(!cli.is_subscription_billed());
6410        let api = catalog
6411            .iter()
6412            .find(|model| model.id == "openai/gpt-5.6-sol:latest")
6413            .unwrap();
6414        assert!(
6415            matches!(&api.source, ModelSource::RemoteApi { .. }),
6416            "the API-key row stays as it was"
6417        );
6418        assert!(!api.is_subscription_billed());
6419        let parslee = catalog
6420            .iter()
6421            .find(|model| model.id == "parslee/fast")
6422            .expect("catalog publishes a Parslee row");
6423        assert!(!parslee.is_subscription_billed());
6424    }
6425
6426    #[test]
6427    fn codex_rows_unavailable_without_sign_in_carry_the_sign_in_reason() {
6428        let _environment = crate::openrouter::test_environment_scope();
6429        remember_codex_sign_in(false);
6430
6431        let tmp = TempDir::new().unwrap();
6432        let mut registry = UnifiedRegistry::new_with_state_root(
6433            tmp.path().to_path_buf(),
6434            tmp.path().join("models"),
6435        );
6436        let expected_ids = [
6437            "openai-codex/gpt-5.6-sol:latest",
6438            "openai-codex/gpt-5.6-sol:high",
6439            "openai-codex/gpt-5.6-luna:latest",
6440            "openai-codex/gpt-5.6-luna:high",
6441            "openai-codex/gpt-5.6-terra:latest",
6442            "openai-codex/gpt-5.6-terra:high",
6443        ];
6444
6445        for id in expected_ids {
6446            let info = ModelInfo::from(
6447                registry
6448                    .get(id)
6449                    .unwrap_or_else(|| panic!("missing builtin Codex row {id}")),
6450            );
6451            assert!(!info.available, "{id} must be unavailable while signed out");
6452            let reason = model_unavailable_reason(&info).expect("signed-out Codex reason");
6453            assert_eq!(reason, crate::schema::OPENAI_CODEX_SIGN_IN_HINT);
6454            assert_eq!(
6455                serde_json::to_value(&info).unwrap()["unavailable_reason"],
6456                reason
6457            );
6458            assert!(reason.contains("car auth login --provider openai-codex"));
6459        }
6460
6461        remember_codex_sign_in(true);
6462        registry.refresh_availability();
6463
6464        for id in expected_ids {
6465            let info = ModelInfo::from(registry.get(id).unwrap());
6466            assert!(info.available, "{id} must be available while signed in");
6467            assert_eq!(model_unavailable_reason(&info), None, "{id}");
6468            assert!(serde_json::to_value(&info).unwrap()["unavailable_reason"].is_null());
6469        }
6470
6471        remember_codex_sign_in(false);
6472    }
6473
6474    #[test]
6475    fn work_context_availability_is_independent_of_other_org_rejection() {
6476        let _environment = crate::openrouter::test_environment_scope();
6477        crate::parslee_credential::clear_credential_rejected();
6478        let tmp = TempDir::new().unwrap();
6479        let mut parslee = UnifiedRegistry::new_with_state_root(
6480            tmp.path().to_path_buf(),
6481            tmp.path().join("models"),
6482        );
6483        let mut flyexclusive = parslee.clone();
6484        crate::parslee_credential::note_credential_rejected();
6485        parslee.refresh_work_context_availability("https://authority.example", true);
6486        flyexclusive.refresh_work_context_availability("https://authority.example", false);
6487        let scoped_rows: Vec<_> = parslee.all().filter(|row| matches!(
6488            &row.source, ModelSource::Proprietary { provider, auth: ProprietaryAuth::OAuth2Pkce { .. }, .. }
6489                if provider == "parslee"
6490        )).collect();
6491        assert!(!scoped_rows.is_empty());
6492        for row in scoped_rows {
6493            assert!(row.available);
6494            assert!(!flyexclusive.get(&row.id).unwrap().available);
6495        }
6496        assert!(
6497            crate::parslee_credential::credential_rejected(),
6498            "scoped routing must not clear another context's observation"
6499        );
6500        crate::parslee_credential::clear_credential_rejected();
6501    }
6502
6503    #[test]
6504    fn codex_availability_leaves_openai_and_parslee_rows_alone() {
6505        let _environment = crate::openrouter::test_environment_scope();
6506        remember_codex_sign_in(false);
6507
6508        let tmp = TempDir::new().unwrap();
6509        let mut registry = UnifiedRegistry::new_with_state_root(
6510            tmp.path().to_path_buf(),
6511            tmp.path().join("models"),
6512        );
6513        registry.refresh_availability();
6514
6515        let unaffected_ids: Vec<String> = registry
6516            .list()
6517            .into_iter()
6518            .filter(|row| {
6519                matches!(
6520                    row.id.as_str(),
6521                    "openai/gpt-5.6-sol:latest" | "openai/gpt-5.6-sol:high"
6522                ) || row.id.starts_with("parslee/")
6523            })
6524            .map(|row| row.id.clone())
6525            .collect();
6526        assert!(unaffected_ids
6527            .iter()
6528            .any(|id| id == "openai/gpt-5.6-sol:latest"));
6529        assert!(unaffected_ids
6530            .iter()
6531            .any(|id| id == "openai/gpt-5.6-sol:high"));
6532        assert!(
6533            unaffected_ids.iter().any(|id| id.starts_with("parslee/")),
6534            "builtin catalog must contain Parslee rows"
6535        );
6536        let signed_out: std::collections::HashMap<String, bool> = unaffected_ids
6537            .iter()
6538            .map(|id| (id.clone(), registry.get(id).unwrap().available))
6539            .collect();
6540
6541        remember_codex_sign_in(true);
6542        registry.refresh_availability();
6543
6544        for id in unaffected_ids {
6545            assert_eq!(
6546                registry.get(&id).unwrap().available,
6547                signed_out[&id],
6548                "Codex sign-in must not change {id} availability"
6549            );
6550        }
6551
6552        remember_codex_sign_in(false);
6553    }
6554
6555    #[test]
6556    fn builtin_catalog_names_are_unique_for_codex_rows() {
6557        let catalog: Vec<ModelSchema> = serde_json::from_str(BUILTIN_CATALOG_JSON).unwrap();
6558        for row in catalog.iter().filter(|row| row.provider == "openai-codex") {
6559            let occurrences = catalog
6560                .iter()
6561                .filter(|candidate| candidate.name == row.name)
6562                .count();
6563            assert_eq!(
6564                occurrences, 1,
6565                "openai-codex name must be unique: {}",
6566                row.name
6567            );
6568        }
6569    }
6570
6571    /// The Apple FoundationModels row must advertise `ToolUse` and must NOT
6572    /// advertise `MultiToolCall`, and its context window must match the
6573    /// framework's real limit. All three were wrong or missing, and each
6574    /// failure was silent:
6575    ///
6576    /// - `AdaptiveRouter` pushes `ModelCapability::ToolUse` as a HARD required
6577    ///   capability whenever `has_tools` is set, so a row without it can never
6578    ///   be selected for a tool-calling route. Omitting it made the entire
6579    ///   `backend::foundation_models::generate_with_tools` path — and the Swift
6580    ///   shim's capture-and-return bridge behind it — unreachable through the
6581    ///   router, with no error anywhere: the model was simply never a candidate.
6582    /// - The same filter pushes `MultiToolCall` for multi-step prompts. The
6583    ///   bridge ends the turn on the first captured call, so at most one tool
6584    ///   call per turn is possible and claiming it would be a lie the router
6585    ///   would act on.
6586    /// - `SystemLanguageModel`'s context window is 4096 tokens, not 8192.
6587    ///   `build_context_for_model` sizes the assembly budget from this number,
6588    ///   so an inflated value overcommits the prompt and the framework rejects
6589    ///   the request at generation time.
6590    ///
6591    /// Pins all three so the catalog claim and the implementation stay in
6592    /// lockstep, in the same spirit as the in-process Qwen3 test below.
6593    #[test]
6594    fn apple_foundation_row_claims_single_tool_call_and_the_real_context_window() {
6595        let catalog: Vec<ModelSchema> = serde_json::from_str(BUILTIN_CATALOG_JSON).unwrap();
6596        let row = catalog
6597            .iter()
6598            .find(|model| model.id == "apple/foundation:default")
6599            .expect("catalog publishes the Apple FoundationModels row");
6600        assert!(matches!(
6601            &row.source,
6602            ModelSource::AppleFoundationModels { .. }
6603        ));
6604        assert!(row.has_capability(ModelCapability::Generate));
6605        assert!(
6606            row.has_capability(ModelCapability::ToolUse),
6607            "generate_with_tools exists, so the router must be able to pick this row for tool routes"
6608        );
6609        assert!(
6610            !row.has_capability(ModelCapability::MultiToolCall),
6611            "the capture sentinel ends the turn on the first tool call"
6612        );
6613        assert!(!row.has_capability(ModelCapability::Vision));
6614        assert_eq!(
6615            row.context_length, 4096,
6616            "SystemLanguageModel's context window is 4096 tokens"
6617        );
6618        assert!(!row.downloads_weights());
6619    }
6620
6621    /// The row gains `multi_tool_call` exactly where the host can do it.
6622    ///
6623    /// The catalog literal deliberately omits it, because macOS 26 yields one
6624    /// captured call per turn; macOS 27 yields every proposed call through the
6625    /// same unchanged shim. So the claim is host-dependent, and the registry
6626    /// is where that gets resolved — the same hook that already makes
6627    /// `available` host-dependent for this source.
6628    ///
6629    /// Pins the agreement rather than a platform: whatever
6630    /// `supports_parallel_tool_calls()` reports is what the live row must
6631    /// claim. A regression that upgrades the row on a host that cannot do it
6632    /// (or fails to upgrade one that can) fails here.
6633    #[test]
6634    fn apple_foundation_multi_tool_call_tracks_what_the_host_can_actually_do() {
6635        #[cfg(any(
6636            all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)),
6637            all(target_os = "ios", target_arch = "aarch64")
6638        ))]
6639        {
6640            let catalog: Vec<ModelSchema> = serde_json::from_str(BUILTIN_CATALOG_JSON).unwrap();
6641            let declared = catalog
6642                .iter()
6643                .find(|m| m.id == "apple/foundation:default")
6644                .expect("catalog publishes the Apple FoundationModels row");
6645            assert!(
6646                !declared.has_capability(ModelCapability::MultiToolCall),
6647                "the catalog literal must stay conservative; the host decides"
6648            );
6649
6650            // A temp models dir keeps this off the developer's real ~/.car.
6651            let models_dir = tempfile::tempdir().unwrap();
6652            let registry = UnifiedRegistry::new(models_dir.path().to_path_buf());
6653            // expect(), not a silent early return: the row comes from the
6654            // builtin catalog and is therefore always present here, so a
6655            // missing one is a real regression rather than a reason to skip.
6656            // A test that can quietly pass by doing nothing is the failure
6657            // mode this whole row keeps hitting.
6658            let live = registry
6659                .models
6660                .get("apple/foundation:default")
6661                .expect("the builtin catalog row must survive into the live registry");
6662            let host_can = crate::backend::foundation_models::supports_parallel_tool_calls();
6663            assert_eq!(
6664                live.capabilities.contains(&ModelCapability::MultiToolCall),
6665                host_can,
6666                "live row must claim multi_tool_call iff the host supports it"
6667            );
6668        }
6669    }
6670
6671    /// The row gains `vision` exactly where images can actually be served.
6672    ///
6673    /// Stricter than the `multi_tool_call` rule, and deliberately so. That one
6674    /// only has to be truthful; this one also has to be *serviceable*, because
6675    /// `AdaptiveRouter` treats vision as a hard required capability and will
6676    /// hand image requests to whatever claims it. A row that advertised vision
6677    /// without `generate_with_images` behind it would be picked and then fail
6678    /// at generation — the same shape as the 8192-vs-4096 context bug, where
6679    /// the claim outran what the backend could do.
6680    ///
6681    /// Both sides read the same probe, so they cannot drift apart.
6682    #[test]
6683    fn apple_foundation_claims_vision_only_where_images_can_be_served() {
6684        #[cfg(any(
6685            all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)),
6686            all(target_os = "ios", target_arch = "aarch64")
6687        ))]
6688        {
6689            let catalog: Vec<ModelSchema> = serde_json::from_str(BUILTIN_CATALOG_JSON).unwrap();
6690            let declared = catalog
6691                .iter()
6692                .find(|m| m.id == "apple/foundation:default")
6693                .expect("catalog publishes the Apple FoundationModels row");
6694            assert!(
6695                !declared.has_capability(ModelCapability::Vision),
6696                "the catalog literal must stay conservative; the device decides"
6697            );
6698
6699            let models_dir = tempfile::tempdir().unwrap();
6700            let registry = UnifiedRegistry::new(models_dir.path().to_path_buf());
6701            let live = registry
6702                .models
6703                .get("apple/foundation:default")
6704                .expect("the builtin catalog row must survive into the live registry");
6705
6706            assert_eq!(
6707                live.capabilities.contains(&ModelCapability::Vision),
6708                crate::backend::foundation_models::supports_vision(),
6709                "the live row must claim vision iff this device's model accepts images"
6710            );
6711        }
6712    }
6713
6714    /// On a host WITHOUT the macOS 27 behaviour, the live row must be
6715    /// byte-identical to the catalog row.
6716    ///
6717    /// This is the "does it hurt previous versions?" guard, and it is
6718    /// deliberately written as a property rather than a platform check,
6719    /// because the machine that can run macOS 26 and the machine that runs
6720    /// this suite are not the same machine. Whatever the host reports, the
6721    /// contract is: no parallel support means nothing was added.
6722    ///
6723    /// It therefore proves the macOS 26 no-regression claim by construction
6724    /// instead of by testimony — on such a host
6725    /// `supports_parallel_tool_calls()` is false, this assertion runs, and a
6726    /// capability that leaked in anyway fails the build. On macOS 27 the
6727    /// branch is inert and
6728    /// `apple_foundation_multi_tool_call_tracks_what_the_host_can_actually_do`
6729    /// covers the positive case.
6730    #[test]
6731    fn a_host_without_parallel_support_gets_exactly_the_catalog_capabilities() {
6732        #[cfg(any(
6733            all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)),
6734            all(target_os = "ios", target_arch = "aarch64")
6735        ))]
6736        {
6737            if crate::backend::foundation_models::supports_parallel_tool_calls() {
6738                return;
6739            }
6740            let catalog: Vec<ModelSchema> = serde_json::from_str(BUILTIN_CATALOG_JSON).unwrap();
6741            let declared = catalog
6742                .iter()
6743                .find(|m| m.id == "apple/foundation:default")
6744                .expect("catalog publishes the Apple FoundationModels row");
6745
6746            let models_dir = tempfile::tempdir().unwrap();
6747            let registry = UnifiedRegistry::new(models_dir.path().to_path_buf());
6748            let live = registry
6749                .models
6750                .get("apple/foundation:default")
6751                .expect("the builtin catalog row must survive into the live registry");
6752
6753            assert_eq!(
6754                live.capabilities, declared.capabilities,
6755                "a host without the macOS 27 behaviour must advertise exactly what \
6756                 the catalog declares — nothing added, nothing removed"
6757            );
6758        }
6759    }
6760
6761    /// `context_size()` reports the OS's own answer, and the catalog literal
6762    /// is the fallback when it can't. Both halves matter: the literal above
6763    /// is what macOS 26.0-26.3 and every non-Apple host keep using, because
6764    /// `SystemLanguageModel.contextSize` only exists from 26.4 on.
6765    ///
6766    /// Deliberately not asserting a specific number on a supported host --
6767    /// the whole point of reading it is that Apple owns the value. Measured
6768    /// 4096 on both macOS 26 and macOS 27; a future OS moving it should
6769    /// change behaviour here without failing this test.
6770    #[test]
6771    fn apple_foundation_context_size_is_either_the_os_answer_or_none() {
6772        #[cfg(any(
6773            all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)),
6774            all(target_os = "ios", target_arch = "aarch64")
6775        ))]
6776        {
6777            if let Some(window) = crate::backend::foundation_models::context_size() {
6778                assert!(
6779                    window >= 512,
6780                    "a context window the framework reports must be usable, got {window}"
6781                );
6782            }
6783        }
6784
6785        // Everywhere else the shim isn't built, so the sentinel must be None
6786        // and the declared `context_length` stands untouched.
6787        #[cfg(not(any(
6788            all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)),
6789            all(target_os = "ios", target_arch = "aarch64")
6790        )))]
6791        {
6792            let catalog: Vec<ModelSchema> = serde_json::from_str(BUILTIN_CATALOG_JSON).unwrap();
6793            let row = catalog
6794                .iter()
6795                .find(|model| model.id == "apple/foundation:default")
6796                .expect("catalog publishes the Apple FoundationModels row");
6797            assert_eq!(row.context_length, 4096);
6798        }
6799    }
6800
6801    /// In-process Qwen3 models advertise tool capability because the local
6802    /// generate path now renders Qwen3's `<tools>` chat format and parses
6803    /// `<tool_call>` blocks back out (`tasks::generate::render_chat_prompt` /
6804    /// `parse_tool_calls`). This pins the catalog so the capability claim and
6805    /// the implementation stay in lockstep — if the in-process tool path is
6806    /// ever removed, this test should fail until the claim is dropped too.
6807    #[test]
6808    fn in_process_qwen3_models_declare_tool_use() {
6809        use crate::schema::ModelSource;
6810        let catalog: Vec<ModelSchema> = serde_json::from_str(BUILTIN_CATALOG_JSON).unwrap();
6811        // The mid/large Qwen3 sizes are reliable tool-callers; the 0.6b/1.7b
6812        // tiers legitimately don't advertise tool_use.
6813        let tool_sizes = ["qwen3-4b", "qwen3-8b", "qwen3-30b-a3b"];
6814        let mut checked = 0;
6815        for entry in &catalog {
6816            let in_process = matches!(
6817                entry.source,
6818                ModelSource::Mlx { .. } | ModelSource::Local { .. }
6819            );
6820            if !in_process || !tool_sizes.iter().any(|s| entry.id.contains(s)) {
6821                continue;
6822            }
6823            assert!(
6824                entry.capabilities.contains(&ModelCapability::ToolUse),
6825                "in-process Qwen3 model {} should advertise ToolUse — the local \
6826                 generate path renders/parses tool calls",
6827                entry.id
6828            );
6829            checked += 1;
6830        }
6831        assert_eq!(
6832            checked, 6,
6833            "expected 6 in-process tool-capable Qwen3 entries (3 mlx + 3 gguf)"
6834        );
6835    }
6836
6837    #[test]
6838    fn public_benchmarks_round_trip_through_model_info() {
6839        use crate::schema::BenchmarkScore;
6840        let mut reg = test_registry();
6841        let mut schema = reg
6842            .find_by_name("Qwen3-4B")
6843            .expect("catalog has Qwen3-4B")
6844            .clone();
6845        schema.id = "test/qwen3-4b-with-bench".into();
6846        schema.public_benchmarks = vec![
6847            BenchmarkScore {
6848                name: "MMLU-Pro".into(),
6849                score: 0.482,
6850                harness: Some("5-shot CoT".into()),
6851                source_url: Some("https://example.invalid/qwen3-4b-card".into()),
6852                measured_at: Some("2025-08-12".into()),
6853                runs: None,
6854                spread: None,
6855            },
6856            BenchmarkScore {
6857                name: "HumanEval".into(),
6858                score: 0.713,
6859                harness: Some("pass@1".into()),
6860                source_url: None,
6861                measured_at: None,
6862                runs: None,
6863                spread: None,
6864            },
6865        ];
6866        reg.register(schema);
6867
6868        let stored = reg
6869            .get("test/qwen3-4b-with-bench")
6870            .expect("registered model is retrievable");
6871        let info = ModelInfo::from(stored);
6872        assert_eq!(info.public_benchmarks.len(), 2);
6873
6874        // The serialized JSON shape is what the WS / FFI clients consume.
6875        let json = serde_json::to_string(&info).unwrap();
6876        assert!(json.contains("\"public_benchmarks\""));
6877        assert!(json.contains("\"MMLU-Pro\""));
6878        assert!(json.contains("\"5-shot CoT\""));
6879
6880        // Round-trip back through serde to confirm deserialization works.
6881        let decoded: ModelInfo = serde_json::from_str(&json).unwrap();
6882        assert_eq!(decoded.public_benchmarks.len(), 2);
6883        assert_eq!(decoded.public_benchmarks[0].name, "MMLU-Pro");
6884        assert_eq!(decoded.public_benchmarks[1].name, "HumanEval");
6885    }
6886
6887    #[test]
6888    fn public_benchmarks_default_to_empty_when_absent_in_json() {
6889        // Older user-config JSON written before this field existed must
6890        // still deserialize cleanly into the new ModelSchema shape.
6891        let legacy_json = r#"{
6892            "id": "legacy/test:1",
6893            "name": "Legacy Test",
6894            "provider": "test",
6895            "family": "test",
6896            "version": "",
6897            "capabilities": ["generate"],
6898            "context_length": 4096,
6899            "param_count": "1B",
6900            "quantization": null,
6901            "performance": {},
6902            "cost": {},
6903            "source": { "type": "ollama", "model_tag": "legacy:1" },
6904            "tags": [],
6905            "supported_params": []
6906        }"#;
6907        let schema: ModelSchema = serde_json::from_str(legacy_json).unwrap();
6908        assert!(schema.public_benchmarks.is_empty());
6909    }
6910
6911    #[test]
6912    fn find_by_name() {
6913        let reg = test_registry();
6914        let m = reg.find_by_name("Qwen3-4B").unwrap();
6915        #[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
6916        assert_eq!(m.id, "mlx/qwen3-4b:4bit");
6917        #[cfg(not(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx))))]
6918        assert_eq!(m.id, "qwen/qwen3-4b:q4_k_m");
6919        assert!(m.has_capability(ModelCapability::Code));
6920    }
6921
6922    #[test]
6923    fn query_by_capability() {
6924        let reg = test_registry();
6925        let embed_models = reg.query_by_capability(ModelCapability::Embed);
6926        assert_eq!(embed_models.len(), 2);
6927        assert!(embed_models
6928            .iter()
6929            .any(|model| model.name == "Qwen3-Embedding-0.6B"));
6930        assert!(embed_models
6931            .iter()
6932            .any(|model| model.name == "Qwen3-Embedding-0.6B-MLX"));
6933    }
6934
6935    #[test]
6936    fn query_with_filter() {
6937        let reg = test_registry();
6938        let code_small = reg.query(&ModelFilter {
6939            capabilities: vec![ModelCapability::Code],
6940            max_size_mb: Some(3000),
6941            local_only: true,
6942            ..Default::default()
6943        });
6944        // Qwen3-1.7B, Qwen3-1.7B-MLX, Qwen3-4B, and Qwen3-4B-MLX fit and have Code capability.
6945        assert_eq!(code_small.len(), 4);
6946    }
6947
6948    #[test]
6949    fn register_remote() {
6950        let mut reg = test_registry();
6951        let initial_len = reg.list().len();
6952        let initial_reasoning_len = reg
6953            .query(&ModelFilter {
6954                capabilities: vec![ModelCapability::Reasoning, ModelCapability::ToolUse],
6955                ..Default::default()
6956            })
6957            .len();
6958        let remote = ModelSchema {
6959            id: "anthropic/claude-sonnet-4-6:latest".into(),
6960            name: "Claude Sonnet 4.6".into(),
6961            provider: "anthropic".into(),
6962            family: "claude-4".into(),
6963            version: "latest".into(),
6964            capabilities: vec![
6965                ModelCapability::Generate,
6966                ModelCapability::Code,
6967                ModelCapability::Reasoning,
6968                ModelCapability::ToolUse,
6969            ],
6970            context_length: 200000,
6971            max_output_tokens: None,
6972            param_count: String::new(),
6973            quantization: None,
6974            performance: PerformanceEnvelope {
6975                latency_p50_ms: Some(2000),
6976                ..Default::default()
6977            },
6978            cost: CostModel {
6979                input_per_mtok: Some(3.0),
6980                output_per_mtok: Some(15.0),
6981                ..Default::default()
6982            },
6983            source: ModelSource::RemoteApi {
6984                endpoint: "https://api.anthropic.com/v1/messages".into(),
6985                api_key_env: "ANTHROPIC_API_KEY".into(),
6986                api_key_envs: vec![],
6987                api_version: Some("2023-06-01".into()),
6988                protocol: ApiProtocol::Anthropic,
6989            },
6990            tags: vec![],
6991            supported_params: vec![],
6992            public_benchmarks: vec![],
6993            trust_tier: crate::schema::TrustTier::Curated,
6994            deprecated: false,
6995            available: false,
6996            weights_ready: false,
6997        };
6998
6999        reg.register(remote);
7000        // Same ID as builtin claude-sonnet-4-6 — replaces, count stays same
7001        assert_eq!(reg.list().len(), initial_len);
7002
7003        let reasoning = reg.query(&ModelFilter {
7004            capabilities: vec![ModelCapability::Reasoning, ModelCapability::ToolUse],
7005            ..Default::default()
7006        });
7007        // Replacing an existing remote slot should not change the reasoning/tool-use lineup size.
7008        assert_eq!(reasoning.len(), initial_reasoning_len);
7009    }
7010
7011    #[test]
7012    fn unregister() {
7013        let mut reg = test_registry();
7014        let initial_len = reg.list().len();
7015        let removed = reg.unregister("qwen/qwen3-0.6b:q8_0");
7016        assert!(removed.is_some());
7017        assert_eq!(reg.list().len(), initial_len - 1);
7018    }
7019
7020    #[test]
7021    fn speech_models_are_curated() {
7022        let reg = test_registry();
7023        let stt = reg.query_by_capability(ModelCapability::SpeechToText);
7024        let tts = reg.query_by_capability(ModelCapability::TextToSpeech);
7025        // Parakeet-TDT-MLX + Whisper-large-v3-turbo (cross-platform) + scribe_v1.
7026        assert_eq!(stt.len(), 3);
7027        // Kokoro-6bit/bf16 + Windows-Speech (OS, Windows-only) + Qwen3-TTS + eleven.
7028        assert_eq!(tts.len(), 5);
7029        // whisper.cpp STT is a first-class catalog citizen — the cross-platform
7030        // local STT that runs where MLX (Apple-only) can't.
7031        let whisper = stt
7032            .iter()
7033            .find(|m| m.name == "Whisper-large-v3-turbo-q5_0")
7034            .expect("whisper STT model should be curated");
7035        assert!(whisper.is_local());
7036        assert!(matches!(
7037            whisper.source,
7038            crate::schema::ModelSource::WhisperCpp { .. }
7039        ));
7040    }
7041
7042    #[test]
7043    fn qwen_8b_variants_keep_tool_use_consistent() {
7044        // The GGUF and MLX twins of Qwen3-8B must agree on tool capability, and
7045        // both advertise it: the in-process generate path renders Qwen3's
7046        // <tools> chat format and parses <tool_call> blocks back out (see
7047        // tasks::generate::render_chat_prompt / parse_tool_calls).
7048        let reg = test_registry();
7049        for name in ["Qwen3-8B", "Qwen3-8B-MLX"] {
7050            let model = reg.find_by_name(name).expect("model should exist");
7051            assert!(model.has_capability(ModelCapability::ToolUse));
7052            assert!(model.has_capability(ModelCapability::MultiToolCall));
7053        }
7054    }
7055
7056    #[test]
7057    fn mac_name_resolution_prefers_mlx_siblings() {
7058        // Only used inside the aarch64-macos cfg below; non-mac targets
7059        // keep the test as a smoke compile.
7060        #[allow(unused_variables)]
7061        let reg = test_registry();
7062        #[cfg(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx)))]
7063        {
7064            assert_eq!(
7065                reg.find_by_name("Qwen3-0.6B").unwrap().id,
7066                "mlx/qwen3-0.6b:6bit"
7067            );
7068            assert_eq!(
7069                reg.find_by_name("Qwen3-1.7B").unwrap().id,
7070                "mlx/qwen3-1.7b:3bit"
7071            );
7072            assert_eq!(
7073                reg.find_by_name("Qwen3-Embedding-0.6B").unwrap().id,
7074                "mlx/qwen3-embedding-0.6b:mxfp8"
7075            );
7076        }
7077    }
7078
7079    #[test]
7080    fn remote_multimodal_models_are_curated_as_vision_capable() {
7081        let reg = test_registry();
7082        for name in [
7083            "claude-opus-4-7",
7084            "claude-opus-4-6",
7085            "claude-sonnet-4-6",
7086            "claude-haiku-4-5",
7087            "gpt-5.4",
7088            "gpt-5.4-mini",
7089            "o3",
7090            "o4-mini",
7091            "gpt-4.1-mini",
7092            "gemini-2.5-pro",
7093            "gemini-2.5-flash",
7094        ] {
7095            let model = reg.find_by_name(name).expect("model should exist");
7096            assert!(
7097                model.has_capability(ModelCapability::Vision),
7098                "{name} should be curated as vision-capable"
7099            );
7100        }
7101    }
7102
7103    #[test]
7104    fn qwen25vl_entries_are_replaced_by_qwen3vl_in_builtin_catalog() {
7105        let reg = test_registry();
7106
7107        let stale_ids = [
7108            // Native MLX text tower can't tokenize images — never advertise.
7109            "mlx/qwen2.5-vl-3b:4bit",
7110            "mlx/qwen2.5-vl-7b:4bit",
7111            // Qwen2.5-VL is superseded by Qwen3-VL; drop the mlx-vlm CLI
7112            // catalog entries so callers route to the upgraded family.
7113            "mlx-vlm/qwen2.5-vl-3b:4bit",
7114            "mlx-vlm/qwen2.5-vl-7b:4bit",
7115            // Same supersession applies to the vLLM-MLX route.
7116            "vllm-mlx/qwen2.5-vl-3b:4bit",
7117        ];
7118        for id in stale_ids {
7119            assert!(
7120                reg.get(id).is_none(),
7121                "{id} is superseded by Qwen3-VL; the catalog must not advertise it"
7122            );
7123        }
7124
7125        let vision_ids: Vec<&str> = reg
7126            .query_by_capability(ModelCapability::Vision)
7127            .into_iter()
7128            .map(|model| model.id.as_str())
7129            .collect();
7130        for stale in stale_ids {
7131            assert!(
7132                !vision_ids.contains(&stale),
7133                "{stale} must not be reachable through the Vision capability index"
7134            );
7135        }
7136        assert!(
7137            vision_ids.contains(&"mlx-vlm/qwen3-vl-2b:bf16"),
7138            "Qwen3-VL is the supported local VL family and must route as Vision"
7139        );
7140    }
7141
7142    #[test]
7143    fn gemini_models_are_curated_for_multimodal_tool_use() {
7144        let reg = test_registry();
7145        for name in ["gemini-2.5-pro", "gemini-2.5-flash"] {
7146            let model = reg.find_by_name(name).expect("model should exist");
7147            assert!(model.has_capability(ModelCapability::Vision));
7148            assert!(model.has_capability(ModelCapability::ToolUse));
7149            assert!(model.has_capability(ModelCapability::MultiToolCall));
7150        }
7151    }
7152
7153    #[test]
7154    fn model_info_publishes_declared_prices_and_keeps_unpriced_distinct_from_free() {
7155        let reg = test_registry();
7156
7157        // A priced remote row publishes every declared rate plus its tiers.
7158        let opus = reg
7159            .list()
7160            .into_iter()
7161            .find(|m| m.id == "openrouter/anthropic/claude-opus-4.8")
7162            .map(ModelInfo::from)
7163            .expect("curated opus-4.8 row is present on first boot");
7164        assert_eq!(opus.cost.input_per_mtok, Some(5.0));
7165        assert_eq!(opus.cost.output_per_mtok, Some(25.0));
7166        assert_eq!(opus.cost.cache_read_input_per_mtok, Some(0.5));
7167        assert_eq!(opus.cost.cache_write_input_per_mtok, Some(6.25));
7168
7169        // Tiered pricing survives the projection where a model declares it.
7170        let gpt = reg
7171            .list()
7172            .into_iter()
7173            .find(|m| m.id == "openrouter/openai/gpt-5.4")
7174            .map(ModelInfo::from)
7175            .expect("curated gpt-5.4 row");
7176        assert_eq!(gpt.cost.pricing_tiers.len(), 1);
7177        assert_eq!(gpt.cost.prices_for(272_000).input_per_mtok, Some(5.0));
7178
7179        // A local model that declares no prices reports them ABSENT. `null`
7180        // and `0.0` are different facts and must stay different on the wire.
7181        let local = reg
7182            .list()
7183            .into_iter()
7184            .find(|m| m.is_local() && m.cost.input_per_mtok.is_none())
7185            .map(ModelInfo::from)
7186            .expect("the built-in catalog ships unpriced local models");
7187        let json = serde_json::to_value(&local).unwrap();
7188        assert!(json["cost"]["input_per_mtok"].is_null());
7189        assert!(json["cost"]["output_per_mtok"].is_null());
7190        assert_ne!(json["cost"]["input_per_mtok"], serde_json::json!(0.0));
7191    }
7192
7193    #[test]
7194    fn a_hand_registered_copy_of_a_curated_id_does_not_double_the_row() {
7195        let mut reg = test_registry();
7196        let id = "openrouter/anthropic/claude-opus-4.8";
7197        assert_eq!(reg.list().iter().filter(|m| m.id == id).count(), 1);
7198
7199        // The runtime registry is keyed by id, so a user registration of the
7200        // same id replaces the curated row rather than appending a duplicate.
7201        let mut copy = reg
7202            .list()
7203            .into_iter()
7204            .find(|m| m.id == id)
7205            .cloned()
7206            .expect("curated row");
7207        copy.name = "hand-registered".into();
7208        reg.register_user_model(copy);
7209        assert_eq!(reg.list().iter().filter(|m| m.id == id).count(), 1);
7210    }
7211
7212    /// Scoped to `ModelInfo` — the `models.list_unified` projection — on
7213    /// purpose. `models.search` wraps this same struct but adds `family`,
7214    /// which for a managed alias names the upstream model family and is
7215    /// matched against by queries. That is pre-existing (`frontier-deep`
7216    /// already carries `claude-4.6`) and out of scope here; this test locks
7217    /// the surface the guarantee actually holds on rather than implying a
7218    /// registry-wide one it does not.
7219    #[test]
7220    fn managed_alias_publishes_prices_without_disclosing_the_upstream_id_in_the_catalog_view() {
7221        let reg = test_registry();
7222        let alias = reg
7223            .list()
7224            .into_iter()
7225            .find(|m| m.id == "parslee/openrouter/frontier-deep-next")
7226            .map(ModelInfo::from)
7227            .expect("managed alias for the new curated row");
7228
7229        assert_eq!(alias.cost.input_per_mtok, Some(5.0));
7230        assert_eq!(alias.cost.output_per_mtok, Some(25.0));
7231        assert_eq!(alias.cost.cache_read_input_per_mtok, Some(0.5));
7232        assert_eq!(alias.cost.cache_write_input_per_mtok, Some(6.25));
7233
7234        let wire = serde_json::to_string(&alias).unwrap();
7235        assert!(!wire.contains("claude-opus-4.8"));
7236        assert!(!wire.contains("anthropic/"));
7237    }
7238
7239    #[test]
7240    fn model_info_from_an_older_daemon_without_cost_still_parses() {
7241        // A newly built client must not fail to read a catalog produced before
7242        // `cost` existed; the row deserializes with an all-absent cost.
7243        let legacy = serde_json::json!({
7244            "id": "legacy/model",
7245            "name": "legacy",
7246            "provider": "legacy",
7247            "capabilities": ["generate"],
7248            "param_count": "",
7249            "size_mb": 0,
7250            "context_length": 8192,
7251            "available": true,
7252            "is_local": false
7253        });
7254        let info: ModelInfo = serde_json::from_value(legacy).expect("older catalog row parses");
7255        assert!(info.cost.input_per_mtok.is_none());
7256        assert!(info.cost.output_per_mtok.is_none());
7257        assert!(info.cost.pricing_tiers.is_empty());
7258        assert!(info.max_output_tokens.is_none());
7259        assert!(info.car_enabled, "legacy rows default to enabled");
7260        assert!(!info.can_remove);
7261        assert!(!info.in_use);
7262        assert!(info.management_evidence.is_none());
7263    }
7264
7265    #[test]
7266    fn visual_generation_models_are_curated() {
7267        let reg = test_registry();
7268        assert_eq!(
7269            reg.query_by_capability(ModelCapability::ImageGeneration)
7270                .len(),
7271            1
7272        );
7273        assert_eq!(
7274            reg.query_by_capability(ModelCapability::VideoGeneration)
7275                .len(),
7276            1
7277        );
7278    }
7279}
7280
7281/// Structural validation of the built-in catalog.
7282///
7283/// The catalog is hand-edited JSON compiled into the binary, so a typo in it is
7284/// a shipped bug that no compiler catches. It went seven weeks without a new
7285/// text model while the open-weight world moved through three generations, and
7286/// the entry added to end that drought carried `param_count: "MoE"` — a string
7287/// the scorer cannot parse — which nothing would have caught. These are the
7288/// invariants the rest of the code already assumes.
7289#[cfg(test)]
7290mod builtin_catalog_validation {
7291    use super::*;
7292    use crate::schema::ModelSource;
7293
7294    /// Sources whose weights CAR (or a runtime CAR supervises) actually loads.
7295    fn weight_repo(source: &ModelSource) -> Option<&str> {
7296        match source {
7297            ModelSource::Mlx { hf_repo, .. } => Some(hf_repo),
7298            ModelSource::Local { hf_repo, .. } => Some(hf_repo),
7299            ModelSource::ManagedVllmMlx { hf_repo, .. } => Some(hf_repo),
7300            _ => None,
7301        }
7302    }
7303
7304    #[test]
7305    fn ids_are_unique() {
7306        let catalog = builtin_catalog();
7307        let mut seen: Vec<&str> = Vec::new();
7308        for model in &catalog {
7309            assert!(
7310                !seen.contains(&model.id.as_str()),
7311                "duplicate catalog id `{}` — the later entry silently shadows the earlier",
7312                model.id
7313            );
7314            seen.push(&model.id);
7315        }
7316    }
7317
7318    #[test]
7319    fn exact_frontier_rows_lock_native_selectors_and_digests() {
7320        let catalog = builtin_catalog();
7321        for (id, name, version, expected_digest) in [
7322            (
7323                "openai/gpt-5.5-2026-04-23",
7324                "gpt-5.5-2026-04-23",
7325                "2026-04-23",
7326                "7eeaecc6e83ce409b37cc68e9c44ba84d64c5f9690179a494bdcb25a93f0aac3",
7327            ),
7328            (
7329                "anthropic/claude-opus-4-8",
7330                "claude-opus-4-8",
7331                "4.8",
7332                "10504959e51dc76c3563df91ae2eaba57cf814834ecd657232c50f264f9e735e",
7333            ),
7334            (
7335                "openai/gpt-5.6-sol:high",
7336                "gpt-5.6-sol:high",
7337                "latest",
7338                "0f580e5c2cebd169be21d511ee4c8c7b2e97e83dc62d8462366a3520ae74de5b",
7339            ),
7340        ] {
7341            let row = catalog
7342                .iter()
7343                .find(|model| model.id == id)
7344                .unwrap_or_else(|| panic!("missing exact production row {id}"));
7345            assert_eq!(row.name, name);
7346            assert_eq!(row.version, version);
7347            assert_eq!(
7348                crate::catalog_identity::row_digest(row).unwrap(),
7349                expected_digest,
7350                "CAR row digest drifted for {id}"
7351            );
7352        }
7353    }
7354
7355    #[test]
7356    fn weight_repos_are_well_formed_huggingface_ids() {
7357        for model in builtin_catalog() {
7358            let Some(repo) = weight_repo(&model.source) else {
7359                continue;
7360            };
7361            assert_eq!(
7362                repo.split('/').count(),
7363                2,
7364                "{}: `{repo}` is not an `org/name` HuggingFace id",
7365                model.id
7366            );
7367            assert!(
7368                !repo.split('/').any(str::is_empty),
7369                "{}: `{repo}` has an empty path segment",
7370                model.id
7371            );
7372            assert!(
7373                !repo.contains(char::is_whitespace),
7374                "{}: `{repo}` contains whitespace",
7375                model.id
7376            );
7377        }
7378    }
7379
7380    /// `param_billions_total` parses a leading number; anything else silently
7381    /// falls back to a size estimate, so an unparseable value is a scoring bug
7382    /// that presents as a mysteriously bad recommendation.
7383    #[test]
7384    fn param_counts_are_parseable_or_deliberately_empty() {
7385        for model in builtin_catalog() {
7386            if weight_repo(&model.source).is_none() || model.param_count.is_empty() {
7387                continue;
7388            }
7389            assert!(
7390                model.param_count.starts_with(|c: char| c.is_ascii_digit()),
7391                "{}: param_count `{}` does not start with a number, so the quality \
7392                 prior cannot read it — leave it empty rather than descriptive",
7393                model.id,
7394                model.param_count
7395            );
7396        }
7397    }
7398
7399    /// CAR supervision is an explicit source contract. Endpoint topology must
7400    /// never be used to infer ownership.
7401    #[test]
7402    fn catalog_vllm_mlx_entries_use_explicit_managed_ownership() {
7403        for model in builtin_catalog() {
7404            if !model.is_vllm_mlx() {
7405                continue;
7406            }
7407            assert!(
7408                model.is_car_managed_vllm_mlx(),
7409                "{}: a CAR-supervised catalog row must opt into ManagedVllmMlx; \
7410                 loopback alone cannot confer ownership",
7411                model.id
7412            );
7413        }
7414    }
7415
7416    #[test]
7417    fn generate_capable_models_declare_a_context_window() {
7418        for model in builtin_catalog() {
7419            if !model.has_capability(crate::schema::ModelCapability::Generate) {
7420                continue;
7421            }
7422            assert!(
7423                model.context_length > 0,
7424                "{}: a generate-capable model with no context_length breaks budget sizing",
7425                model.id
7426            );
7427        }
7428    }
7429
7430    #[test]
7431    fn every_entry_declares_at_least_one_capability() {
7432        for model in builtin_catalog() {
7433            assert!(
7434                !model.capabilities.is_empty(),
7435                "{}: an entry with no capabilities can never be routed to",
7436                model.id
7437            );
7438        }
7439    }
7440}
7441
7442#[cfg(test)]
7443mod gguf_quantization_tests {
7444    use crate::schema::{QuantScheme, Quantization};
7445
7446    fn quantization_from_gguf_filename(name: &str) -> Option<Quantization> {
7447        Quantization::from_gguf_filename(name)
7448    }
7449
7450    #[test]
7451    fn reads_the_quantization_a_gguf_file_names() {
7452        let cases = [
7453            ("Qwen3-8B-Q4_K_M.gguf", "Q4_K_M", QuantScheme::KQuantMixed),
7454            (
7455                "Qwen3-Embedding-0.6B-Q8_0.gguf",
7456                "Q8_0",
7457                QuantScheme::RtnBlock,
7458            ),
7459            (
7460                "ggml-large-v3-turbo-q5_0.gguf",
7461                "q5_0",
7462                QuantScheme::RtnBlock,
7463            ),
7464            ("model-IQ4_XS.gguf", "IQ4_XS", QuantScheme::KQuantMixed),
7465        ];
7466        for (filename, label, scheme) in cases {
7467            let q = quantization_from_gguf_filename(filename)
7468                .unwrap_or_else(|| panic!("no quantization found in {filename}"));
7469            assert_eq!(q.label, label, "label for {filename}");
7470            assert_eq!(q.scheme, scheme, "scheme for {filename}");
7471        }
7472    }
7473
7474    /// A name with nothing quant-shaped in it must yield nothing, not a guess.
7475    #[test]
7476    fn returns_none_when_the_name_says_nothing() {
7477        for filename in ["model.gguf", "llama-2-7b-chat.gguf", "ggml-base.gguf"] {
7478            assert!(
7479                quantization_from_gguf_filename(filename).is_none(),
7480                "should not have guessed from {filename}"
7481            );
7482        }
7483    }
7484
7485    /// Scanning from the right means a quant-shaped word earlier in the model
7486    /// name cannot outrank the real suffix.
7487    #[test]
7488    fn the_rightmost_match_wins() {
7489        let q = quantization_from_gguf_filename("q8-experiment-Q4_K_M.gguf").unwrap();
7490        assert_eq!(q.label, "Q4_K_M");
7491    }
7492}
7493
7494#[cfg(test)]
7495mod mlx_executor_discrimination_tests {
7496    use super::*;
7497
7498    /// `ModelSource::Mlx` is a weight-format tag shared by five executors, and
7499    /// only one of them is the in-process Swift loader. Gating the whole source
7500    /// on that loader marked `mflux`, `ltx-2-mlx` and the `mlx_vlm` CLI rows
7501    /// unavailable on any Mac whose Swift build merely failed — subprocesses
7502    /// that never touch the linked library, and the only local image/video
7503    /// backends there are.
7504    ///
7505    /// Asserted against the real builtin catalog rather than a hand-built
7506    /// fixture: a fixture written beside this helper would agree with it by
7507    /// construction, and the thing under test is precisely whether the
7508    /// predicate matches the catalog's actual shape.
7509    #[test]
7510    fn only_in_process_text_rows_need_the_swift_loader() {
7511        let catalog = builtin_catalog();
7512        let row = |id: &str| {
7513            catalog
7514                .iter()
7515                .find(|m| m.id == id)
7516                .unwrap_or_else(|| panic!("{id} missing from the builtin catalog"))
7517        };
7518
7519        for id in [
7520            "mlx/flux-1-lite-8b:q4",    // mflux subprocess
7521            "mlx/ltx-2.3:q4",           // ltx-2-mlx subprocess
7522            "mlx-vlm/qwen3-vl-2b:bf16", // mlx_vlm Python CLI
7523            "mlx/kokoro-82m:bf16",      // managed mlx-audio venv
7524            "mlx/parakeet-tdt-0.6b-v3:default",
7525        ] {
7526            let m = row(id);
7527            assert!(
7528                matches!(m.source, ModelSource::Mlx { .. }),
7529                "{id}: precondition — this test is only meaningful for Mlx-sourced rows"
7530            );
7531            assert!(
7532                !mlx_row_needs_in_process_loader(m),
7533                "{id} runs in a subprocess; its availability must not depend on the \
7534                 in-process Swift loader"
7535            );
7536        }
7537
7538        // Positive control: the predicate must still say yes to something, or
7539        // the assertions above would pass against a function returning false.
7540        for id in ["mlx/qwen3-4b:4bit", "mlx/gemma-4-12b-it:4bit"] {
7541            assert!(
7542                mlx_row_needs_in_process_loader(row(id)),
7543                "{id} is served in-process and does need the loader"
7544            );
7545        }
7546    }
7547}
7548
7549#[cfg(test)]
7550mod local_availability_tests {
7551    use super::*;
7552    use crate::schema::ModelSchema;
7553    use tempfile::TempDir;
7554
7555    fn gguf_row(id: &str, hf_repo: &str) -> ModelSchema {
7556        let mut schema: ModelSchema = serde_json::from_value(serde_json::json!({
7557            "id": id,
7558            "name": id.replace('/', "-"),
7559            "provider": "qwen",
7560            "family": "qwen3",
7561            "capabilities": ["generate"],
7562            "context_length": 32768,
7563            "param_count": "8B",
7564            "source": {
7565                "type": "local",
7566                "hf_repo": hf_repo,
7567                "hf_filename": "model.gguf",
7568                "tokenizer_repo": hf_repo,
7569            },
7570            "cost": { "size_mb": 4900 },
7571        }))
7572        .unwrap();
7573        schema.available = false;
7574        schema
7575    }
7576
7577    /// GGUF is the local path on every machine that is not Apple-Silicon-MLX —
7578    /// i.e. the CUDA machines. `ensure_local` downloads `model.gguf` and
7579    /// `tokenizer.json` from the declared repo on first use, exactly as the MLX
7580    /// path does, so a declared repo is functionally available.
7581    ///
7582    /// This branch was left out of #164, which gave the MLX path that rule, and
7583    /// out of whisper.cpp's unconditional `true`. The result was a fresh CUDA
7584    /// install reporting every local model unavailable, the router dropping
7585    /// them from fallback chains, and the user routed to cloud with an idle GPU
7586    /// one download away.
7587    #[cfg(not(all(target_os = "macos", target_arch = "aarch64", not(car_skip_mlx))))]
7588    #[test]
7589    fn a_declared_repo_is_available_before_it_is_downloaded() {
7590        let _environment = crate::openrouter::test_environment_scope();
7591        let tmp = TempDir::new().unwrap();
7592        let models = tmp.path().join("models");
7593        std::fs::create_dir_all(&models).unwrap();
7594
7595        let mut reg = UnifiedRegistry::new_with_state_root(tmp.path().to_path_buf(), models);
7596        reg.register(gguf_row("qwen/test-8b:q4_k_m", "Qwen/Qwen3-8B-GGUF"));
7597
7598        let model = reg.get("qwen/test-8b:q4_k_m").expect("registered");
7599        assert!(
7600            model.available,
7601            "a GGUF model with a repo to fetch from must not report unavailable \
7602             just because nothing has downloaded it yet"
7603        );
7604    }
7605
7606    /// The exception, and the reason the rule is `!hf_repo.is_empty()` rather
7607    /// than blanket `true`: rows from the local-directory scan have nowhere to
7608    /// fetch from, so they stay gated on the file being physically present.
7609    #[test]
7610    fn a_row_with_nowhere_to_fetch_from_stays_unavailable() {
7611        let _environment = crate::openrouter::test_environment_scope();
7612        let tmp = TempDir::new().unwrap();
7613        let models = tmp.path().join("models");
7614        std::fs::create_dir_all(&models).unwrap();
7615
7616        let mut reg = UnifiedRegistry::new_with_state_root(tmp.path().to_path_buf(), models);
7617        reg.register(gguf_row("local/scanned", ""));
7618
7619        let model = reg.get("local/scanned").expect("registered");
7620        assert!(
7621            !model.available,
7622            "an empty hf_repo has no download to promise"
7623        );
7624    }
7625
7626    #[test]
7627    fn builtin_credential_names_cover_remote_and_proprietary_auth() {
7628        let names = builtin_credential_env_names();
7629        for name in [
7630            "OPENAI_API_KEY",
7631            "ANTHROPIC_API_KEY",
7632            crate::openrouter::API_KEY_ENV,
7633        ] {
7634            assert!(names.contains(name), "built-in catalog omitted {name}");
7635        }
7636    }
7637}