Skip to main content

car_inference/
recommend.rs

1//! Model recommender — hardware + intent → ranked, explained model picks.
2//!
3//! This is the framing layer that turns the typed registry
4//! (`ModelSchema`), hardware facts (`HardwareInfo` / `SupportedAcceleration`),
5//! and the acquisition-intent vocabulary (`UseCase` / `QualityTier` /
6//! `Privacy`) into something a non-expert can act on: "given this machine
7//! and what I want to do, which model should I install, and why?"
8//!
9//! The full contract lives in docs/solutions/first-class-model-ux.md. The
10//! selection pipeline is deterministic — same `(models, hardware, intent)`
11//! always yields the same ranking — and pure (no disk, no network), so the
12//! hardware × use-case matrix is unit-testable.
13//!
14//! This supersedes `hardware::recommend_model` for the user-facing flows
15//! (`car setup`, `models.recommend`). That standalone heuristic survives only
16//! as a registry-less bootstrap inside `HardwareInfo::detect`; everything that
17//! has a registry in hand should call [`recommend`].
18
19use serde::{Deserialize, Serialize};
20
21use crate::hardware::{HardwareInfo, SupportedAcceleration};
22use crate::intent::{Privacy, QualityTier, UseCase, UseCaseRole};
23use crate::resource_policy::{
24    estimate_model_memory, model_parameter_billions_active, model_parameter_billions_total,
25    ResourcePolicy, ResourceProfile, RECOMMENDATION_CONTEXT_TOKENS,
26};
27use crate::schema::{ModelSchema, TrustTier};
28
29/// Whether a model fits in the machine's memory for local execution.
30#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
31#[serde(rename_all = "snake_case")]
32pub enum FitStatus {
33    /// Comfortably fits with headroom for KV cache + overhead.
34    Fits,
35    /// Too large for this machine's memory budget.
36    TooBig,
37    /// Runs on an external server (vLLM-MLX) or remote API — local memory
38    /// does not apply. Never claimed as a local "fits".
39    ServerProvided,
40    /// The model declares no usable memory figure, so fit can't be asserted.
41    Unknown,
42}
43
44/// One ranked recommendation, explained in plain language.
45#[derive(Debug, Clone, Serialize, Deserialize)]
46pub struct Recommendation {
47    /// Registry id (e.g. "qwen/qwen3-4b:q4_k_m"). The caller pulls by this.
48    pub model_id: String,
49    /// Human-readable name shown to the user.
50    pub display_name: String,
51    /// The role lane this pick serves.
52    pub role: UseCaseRole,
53    /// Plain-language reason, generated from the winning factors. Never
54    /// exposes quantization / repo / file jargon.
55    pub rationale: String,
56    /// Download size in MB.
57    pub download_mb: u64,
58    /// True if the model is already installed.
59    pub already_installed: bool,
60    /// Local memory fit.
61    pub fit: FitStatus,
62    /// The acceleration tier this machine would run it on.
63    pub acceleration: SupportedAcceleration,
64    /// True for on-device models, false for remote/cloud.
65    pub is_local: bool,
66    /// True when running this pick sends prompts off the machine — the
67    /// caller must obtain one-time consent before the first cloud inference.
68    pub requires_cloud_consent: bool,
69    /// How much the project vouches for this model.
70    pub trust_tier: TrustTier,
71    /// Internal blended score (higher is better). Exposed for tests/debug.
72    pub score: f32,
73    /// Whether CAR may choose this candidate automatically for the current
74    /// policy lane. False candidates remain visible as explicit heavier or
75    /// unknown-memory alternatives that a person can choose deliberately.
76    ///
77    /// The default preserves deserialization compatibility with responses
78    /// produced before this additive field existed.
79    #[serde(default = "default_true")]
80    pub within_recommendation_target: bool,
81}
82
83const fn default_true() -> bool {
84    true
85}
86
87// --- tuning constants (documented; consistent with hardware.rs scale) ------
88
89/// RAM the OS + other apps need; subtracted from total before the budget.
90const OS_RESERVE_MB: u64 = 3072;
91
92/// The result of a recommendation query — never a bare `Vec`, so the caller
93/// can tell "here are your picks" from "nothing fits, here's why".
94#[derive(Debug, Clone, Serialize, Deserialize)]
95pub struct RecommendationSet {
96    /// Ranked, actionable picks (best first). Empty when nothing is eligible
97    /// or everything is too big — read `note` in that case.
98    pub picks: Vec<Recommendation>,
99    /// Eligible models that don't fit this machine's memory, for an honest
100    /// "needs more RAM" listing. Ranked by score so the closest miss is first.
101    pub not_enough_memory: Vec<Recommendation>,
102    /// Plain-language explanation when `picks` is empty, or a heads-up worth
103    /// surfacing (e.g. the top pick runs in the cloud). `None` otherwise.
104    pub note: Option<String>,
105}
106
107/// Memory fit of one catalog row on one machine, as `models.list_unified`
108/// publishes it. Three-valued on purpose: a catalog consumer wants to know
109/// whether to offer the row, and "the server owns its memory" is, for that
110/// question, a fit. [`FitStatus`] keeps the four-way distinction for the
111/// recommender, which ranks rather than filters.
112#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)]
113#[serde(rename_all = "snake_case")]
114pub enum ModelFitStatus {
115    /// Fits this machine's active local-model budget, or runs somewhere that
116    /// is not this machine's memory at all (remote API, delegated runner,
117    /// operator-managed external endpoint).
118    Fits,
119    /// Too large for this machine's active local-model budget.
120    TooBig,
121    /// The row declares no usable memory figure, or the accelerator's memory
122    /// is unknown, so fit cannot be asserted either way. Also what a client
123    /// sees from a daemon that predates the annotation, which is why it is
124    /// the `Default`: an absent verdict must never read as too-big.
125    #[default]
126    Unknown,
127}
128
129/// A row's fit annotation: the verdict, the estimate it was judged on, and
130/// whether the accelerator can run the row at all.
131#[derive(Debug, Clone, Copy, PartialEq, Eq)]
132pub struct ModelFit {
133    pub fit: ModelFitStatus,
134    /// The cold-load peak the verdict was judged against, in MB. `None` when
135    /// the row's memory is not this machine's (remote, delegated, external)
136    /// or when it declares neither a size nor a RAM figure — the same rows
137    /// the recommender refuses to judge by memory.
138    pub estimated_peak_mb: Option<u64>,
139    /// See [`platform_compatible`].
140    pub platform_compatible: bool,
141}
142
143/// The `models.list_unified` fit rule. It is `build_recommendation`'s own
144/// verdict — the same [`estimate_model_memory`] at
145/// [`RECOMMENDATION_CONTEXT_TOKENS`], the same `fit_status`, the same policy
146/// ceiling — collapsed to the three values a catalog consumer acts on, plus
147/// `passes_base_filter`'s platform check. Pure over its inputs: nothing here
148/// touches disk or the network, and nothing is persisted.
149///
150/// The estimate never reads a checkpoint's own KV geometry: the KV cache is
151/// sized from the parameter count, installed or not. Load-time admission alone
152/// reads an installed checkpoint's `config.json` and live memory, so for the
153/// same row on the same machine it can decide either way from this. Every
154/// reader of this verdict shares that limit — `models.list_unified`'s `fit`,
155/// the portfolio's `never_runs_here`, and `car models fit`.
156pub fn model_fit(m: &ModelSchema, hw: &HardwareInfo, policy: Option<&ResourcePolicy>) -> ModelFit {
157    let memory_limits = RecommendationMemoryLimits::for_policy(hw, policy);
158    let estimate = estimate_model_memory(m, hw, RECOMMENDATION_CONTEXT_TOKENS);
159    let status = fit_status(m, hw, &estimate, &memory_limits);
160    let fit = match status {
161        FitStatus::Fits | FitStatus::ServerProvided => ModelFitStatus::Fits,
162        FitStatus::TooBig => ModelFitStatus::TooBig,
163        FitStatus::Unknown => ModelFitStatus::Unknown,
164    };
165    let estimated_peak_mb = match status {
166        FitStatus::ServerProvided => None,
167        // Mirrors `fit_status`'s own "nothing declared" gate: an estimate
168        // derived from no declared figure is not evidence worth publishing.
169        _ if m.size_mb() == 0 && m.ram_mb() == 0 => None,
170        _ => Some(estimate.estimated_peak_mb),
171    };
172    ModelFit {
173        fit,
174        estimated_peak_mb,
175        platform_compatible: platform_compatible(m, hw),
176    }
177}
178
179/// Rank model picks for a machine + intent, best first. Pure over its inputs.
180///
181/// `models` is typically `UnifiedRegistry::list()`. Returns at most one
182/// role's lane today (every `UseCase` is single-role). Too-big-but-eligible
183/// models are surfaced separately in `not_enough_memory` rather than dropped
184/// silently, and an empty `picks` always comes with an explanatory `note`.
185pub fn recommend(
186    models: &[&ModelSchema],
187    hw: &HardwareInfo,
188    use_case: UseCase,
189    tier: QualityTier,
190    privacy: Privacy,
191) -> RecommendationSet {
192    recommend_inner(models, hw, use_case, tier, privacy, None)
193}
194
195/// Policy-aware recommendation entry point. Every profile supplies the fit
196/// ceiling and automatic-selection target. Only Everyday Assistant/Balanced
197/// changes the legacy ranking and capability floor.
198pub fn recommend_with_policy(
199    models: &[&ModelSchema],
200    hw: &HardwareInfo,
201    policy: &ResourcePolicy,
202    use_case: UseCase,
203    tier: QualityTier,
204    privacy: Privacy,
205) -> RecommendationSet {
206    recommend_inner(models, hw, use_case, tier, privacy, Some(policy))
207}
208
209fn recommend_inner(
210    models: &[&ModelSchema],
211    hw: &HardwareInfo,
212    use_case: UseCase,
213    tier: QualityTier,
214    privacy: Privacy,
215    policy: Option<&ResourcePolicy>,
216) -> RecommendationSet {
217    let accel = hw.supported_acceleration();
218    // A model without `ToolUse` cannot run the assistant: `car do` refuses it
219    // and substitutes a tool-capable one. That is a property of the USE CASE,
220    // not of the resource policy or the quality tier.
221    //
222    // It used to ride on `everyday_assistant_balanced` below, which also
223    // selects a sort order, so the requirement silently applied only on the
224    // default profile at the default tier. `--tier fastest` and
225    // `--tier most-capable` therefore ranked models the assistant will not use
226    // — and told the reader to install the top one with `car setup --for
227    // assistant`. Measured before this split: `--tier fastest` returned
228    // Qwen3-0.6B first, which has no `ToolUse` (car#1822).
229    let assistant_requires_tools = use_case == UseCase::Assistant;
230    let everyday_assistant_balanced = policy.is_some_and(|policy| {
231        policy.profile == ResourceProfile::Everyday
232            && use_case == UseCase::Assistant
233            && tier == QualityTier::Balanced
234    });
235    let memory_limits = RecommendationMemoryLimits::for_policy(hw, policy);
236    let sort = |v: &mut Vec<RankedRecommendation>| {
237        // Everyday Assistant/Balanced is intentionally memory-first for local
238        // candidates. The policy applies only in that lane, so every legacy
239        // caller keeps the established score ordering.
240        v.sort_by(|a, b| {
241            if everyday_assistant_balanced {
242                // This is an item-local key, not a pair-dependent branch.
243                // That matters when local and cloud candidates are mixed:
244                // every comparison observes the same total ordering, so sort
245                // transitivity cannot depend on the input permutation.
246                let policy_class = |recommendation: &Recommendation| {
247                    if recommendation.fit == FitStatus::Unknown {
248                        3
249                    } else if recommendation.is_local && recommendation.within_recommendation_target
250                    {
251                        0
252                    } else if !recommendation.is_local
253                        && recommendation.within_recommendation_target
254                    {
255                        1
256                    } else {
257                        2
258                    }
259                };
260                let a_recommendation = &a.recommendation;
261                let b_recommendation = &b.recommendation;
262                return policy_class(a_recommendation)
263                    .cmp(&policy_class(b_recommendation))
264                    .then(
265                        b_recommendation
266                            .already_installed
267                            .cmp(&a_recommendation.already_installed),
268                    )
269                    .then(a.estimated_peak_mb.cmp(&b.estimated_peak_mb))
270                    .then(a.latency_p50_ms.cmp(&b.latency_p50_ms))
271                    .then_with(|| b_recommendation.score.total_cmp(&a_recommendation.score))
272                    .then(a_recommendation.model_id.cmp(&b_recommendation.model_id));
273            }
274            b.recommendation
275                .score
276                .total_cmp(&a.recommendation.score)
277                .then(
278                    b.recommendation
279                        .already_installed
280                        .cmp(&a.recommendation.already_installed),
281                )
282                .then(
283                    a.recommendation
284                        .download_mb
285                        .cmp(&b.recommendation.download_mb),
286                )
287                .then(a.recommendation.model_id.cmp(&b.recommendation.model_id))
288        });
289    };
290
291    let (mut picks, mut not_enough_memory): (Vec<_>, Vec<_>) = models
292        .iter()
293        .filter(|m| {
294            passes_base_filter(m, hw, use_case, privacy)
295                && (!assistant_requires_tools
296                    || m.has_capability(crate::schema::ModelCapability::ToolUse))
297        })
298        .map(|model| build_recommendation(model, hw, &accel, use_case, tier, &memory_limits))
299        .partition(|ranked| ranked.recommendation.fit != FitStatus::TooBig);
300    sort(&mut picks);
301    sort(&mut not_enough_memory);
302    let picks: Vec<_> = picks
303        .into_iter()
304        .map(|ranked| ranked.recommendation)
305        .collect();
306    let not_enough_memory: Vec<_> = not_enough_memory
307        .into_iter()
308        .map(|ranked| ranked.recommendation)
309        .collect();
310
311    let note = explain_if_needed(&picks, &not_enough_memory, hw, use_case, tier, privacy)
312        .or_else(|| unmeasured_larger_candidates(models, &picks, tier));
313    RecommendationSet {
314        picks,
315        not_enough_memory,
316        note,
317    }
318}
319
320struct RankedRecommendation {
321    recommendation: Recommendation,
322    estimated_peak_mb: u64,
323    latency_p50_ms: u64,
324}
325
326#[derive(Clone, Copy)]
327struct RecommendationMemoryLimits {
328    legacy_budget_mb: u64,
329    policy_host_budget_mb: Option<u64>,
330    recommendation_target_mb: Option<u64>,
331}
332
333impl RecommendationMemoryLimits {
334    /// The budgets every fit verdict is judged against: the legacy
335    /// acceleration-tier budget, plus the policy's host ceiling and automatic
336    /// recommendation target when a policy is in force. One constructor, so
337    /// [`recommend_with_policy`] and [`model_fit`] cannot drift apart.
338    fn for_policy(hw: &HardwareInfo, policy: Option<&ResourcePolicy>) -> Self {
339        Self {
340            legacy_budget_mb: memory_budget_mb(hw),
341            policy_host_budget_mb: policy.map(|policy| {
342                policy
343                    .effective_budget(hw.total_ram_mb)
344                    .configured_model_ceiling_mb
345            }),
346            recommendation_target_mb: policy
347                .map(|policy| policy.recommendation_target_mb(hw.total_ram_mb)),
348        }
349    }
350}
351
352/// Hard eligibility, minus the fit check (fit is handled by partitioning so
353/// too-big models can still be surfaced). A model failing any of these is
354/// never shown at all.
355fn passes_base_filter(
356    m: &ModelSchema,
357    hw: &HardwareInfo,
358    use_case: UseCase,
359    privacy: Privacy,
360) -> bool {
361    if m.deprecated {
362        return false;
363    }
364    // Capability requirement encodes the role lane (Search⇒Embed, etc.).
365    if !use_case
366        .required_capabilities()
367        .iter()
368        .all(|c| m.has_capability(*c))
369    {
370        return false;
371    }
372    // Privacy: on-device excludes anything that leaves the machine.
373    if privacy == Privacy::OnDevice && !m.is_local() {
374        return false;
375    }
376    // A Metal-only model on a non-Apple machine can't run at all.
377    platform_compatible(m, hw)
378}
379
380/// Whether this machine can run the model at all. Metal-only sources (MLX,
381/// CAR-managed vLLM-MLX, Apple FoundationModels) need Apple Silicon, and
382/// GGUF/Candle rows never run there;
383/// WindowsSpeech and rows tagged `windows-only` need Windows; rows tagged
384/// `linux-only` need Linux. Everything else runs anywhere CAR does. Shared by
385/// the recommender's eligibility filter and the `models.list_unified` fit
386/// annotation so the two cannot disagree about which rows a machine can host.
387pub fn platform_compatible(m: &ModelSchema, hw: &HardwareInfo) -> bool {
388    let windows_only = matches!(m.source, crate::schema::ModelSource::WindowsSpeech { .. })
389        || m.tags.iter().any(|tag| tag == "windows-only");
390    let linux_only = m.tags.iter().any(|tag| tag == "linux-only");
391    let apple = matches!(
392        hw.supported_acceleration(),
393        SupportedAcceleration::Apple { .. }
394    );
395    let apple_compatible = !m.requires_apple_silicon() || apple;
396    // On Apple Silicon CAR runs local text only through MLX: a GGUF request
397    // executes as its MLX twin, an exact GGUF pin is refused, and a GGUF row
398    // with no twin cannot run at all. Offering the GGUF row there recommended
399    // a download that would never execute — and listed every Qwen3 twice.
400    let candle_runs_here = !(apple && matches!(m.source, crate::schema::ModelSource::Local { .. }));
401    let windows_compatible = !windows_only || hw.os.eq_ignore_ascii_case("windows");
402    let linux_compatible = !linux_only || hw.os.eq_ignore_ascii_case("linux");
403    apple_compatible && candle_runs_here && windows_compatible && linux_compatible
404}
405
406/// Build the honest `note` for the result set. Empty picks always get one.
407fn explain_if_needed(
408    picks: &[Recommendation],
409    too_big: &[Recommendation],
410    hw: &HardwareInfo,
411    use_case: UseCase,
412    tier: QualityTier,
413    privacy: Privacy,
414) -> Option<String> {
415    let purpose = use_case_purpose(use_case);
416    if picks.is_empty() {
417        let ram_gb = hw.total_ram_mb / 1024;
418        return Some(if !too_big.is_empty() {
419            match privacy {
420                Privacy::OnDevice => format!(
421                    "No on-device model for {purpose} fits your {ram_gb} GB machine. \
422                     Free up memory, pick a smaller tier, or allow cloud models."
423                ),
424                Privacy::CloudOk => format!(
425                    "No local model for {purpose} fits your {ram_gb} GB machine, and no \
426                     cloud model is configured. Add an API key or free up memory."
427                ),
428            }
429        } else {
430            format!("No model available for {purpose} on this machine.")
431        });
432    }
433    // Heads-up when the best pick we can offer is the most-capable tier but
434    // still ran out of bigger options, or when the top pick is cloud.
435    if picks[0].requires_cloud_consent {
436        return Some(format!(
437            "The best {purpose} pick runs in the cloud and needs your OK before first use. \
438             {} fits locally if you prefer on-device.",
439            picks
440                .iter()
441                .find(|p| p.is_local)
442                .map(|p| p.display_name.as_str())
443                .unwrap_or("No local model")
444        ));
445    }
446    let _ = tier;
447    None
448}
449
450/// A heads-up when "most capable" had larger candidates it could not judge.
451///
452/// The quality prior is a published benchmark when one exists and a saturating
453/// size curve otherwise, and those are not the same scale — the curve tops out
454/// below what a real score reaches, so a measured mid-size model outranks an
455/// unmeasured larger one by construction. That is defensible ranking (an
456/// unmeasured model is genuinely unproven) but it is not something to state as
457/// "the most capable model that fits" while saying nothing about the rest.
458///
459/// So: name them. The user can see that the answer is bounded by what has been
460/// measured, and knows there is something to do about it.
461fn unmeasured_larger_candidates(
462    models: &[&ModelSchema],
463    picks: &[Recommendation],
464    tier: QualityTier,
465) -> Option<String> {
466    if tier != QualityTier::MostCapable {
467        return None;
468    }
469    let top = picks.first()?;
470    let top_params = models
471        .iter()
472        .find(|m| m.id == top.model_id)
473        .map(|m| param_billions_total(m))?;
474
475    let mut larger: Vec<&str> = models
476        .iter()
477        .filter(|m| {
478            m.is_local()
479                && m.public_benchmarks.is_empty()
480                && param_billions_total(m) > top_params * 1.5
481        })
482        .map(|m| m.name.as_str())
483        .collect();
484    if larger.is_empty() {
485        return None;
486    }
487    larger.sort_unstable();
488    larger.dedup();
489    let shown = larger
490        .iter()
491        .take(3)
492        .copied()
493        .collect::<Vec<_>>()
494        .join(", ");
495    let rest = larger.len().saturating_sub(3);
496    let and_more = if rest > 0 {
497        format!(" and {rest} more")
498    } else {
499        String::new()
500    };
501    Some(format!(
502        "Ranked among models with a measured quality score. Larger ones this \
503         machine can run are unscored, so they cannot be ranked here yet: \
504         {shown}{and_more}. Run `scripts/bench-contribute.sh` to score them."
505    ))
506}
507
508fn use_case_purpose(use_case: UseCase) -> &'static str {
509    match use_case {
510        UseCase::Assistant => "chat & general help",
511        UseCase::Coding => "coding",
512        UseCase::Summarize => "summarizing",
513        UseCase::Vision => "understanding images",
514        UseCase::Transcription => "transcription",
515        UseCase::Search => "semantic search",
516    }
517}
518
519fn build_recommendation(
520    m: &ModelSchema,
521    hw: &HardwareInfo,
522    accel: &SupportedAcceleration,
523    use_case: UseCase,
524    tier: QualityTier,
525    memory_limits: &RecommendationMemoryLimits,
526) -> RankedRecommendation {
527    let estimate = estimate_model_memory(m, hw, RECOMMENDATION_CONTEXT_TOKENS);
528    let fit = fit_status(m, hw, &estimate, memory_limits);
529    let quality = quality_score(m);
530    let latency = latency_score(m, accel);
531    let pressure = memory_pressure(&estimate, memory_limits.legacy_budget_mb);
532    let w = tier.weights();
533    // Higher score is better, so memory *pressure* is inverted.
534    let mut score =
535        w.quality * quality + w.latency * latency + w.memory_pressure * (1.0 - pressure);
536    // Preferred capabilities are a soft bonus, never an eligibility gate.
537    let pref_hits = use_case
538        .preferred_capabilities()
539        .iter()
540        .filter(|c| m.has_capability(**c))
541        .count();
542    score += 0.05 * pref_hits as f32;
543
544    let is_local = m.is_local();
545    let within_recommendation_target = match memory_limits.recommendation_target_mb {
546        None => true,
547        Some(_) if fit == FitStatus::Unknown => false,
548        Some(target_mb) if is_local => match hw.supported_acceleration() {
549            SupportedAcceleration::Cuda {
550                device_memory_mb: Some(device_memory_mb),
551            } => {
552                let host_required_mb = estimate
553                    .estimated_peak_mb
554                    .saturating_sub(estimate.weights_mb);
555                fit == FitStatus::Fits
556                    && estimate.weights_mb <= device_memory_mb
557                    && host_required_mb <= target_mb
558            }
559            SupportedAcceleration::Cuda {
560                device_memory_mb: None,
561            } => false,
562            _ => fit == FitStatus::Fits && estimate.estimated_peak_mb <= target_mb,
563        },
564        Some(_) => true,
565    };
566    RankedRecommendation {
567        estimated_peak_mb: if is_local {
568            estimate.estimated_peak_mb
569        } else {
570            0
571        },
572        latency_p50_ms: m.performance.latency_p50_ms.unwrap_or(u64::MAX),
573        recommendation: Recommendation {
574            model_id: m.id.clone(),
575            display_name: m.name.clone(),
576            role: use_case.role(),
577            rationale: rationale(m, hw, use_case, tier, fit, quality),
578            download_mb: if m.downloads_weights() {
579                m.size_mb()
580            } else {
581                0
582            },
583            already_installed: m.has_installed_weights(),
584            fit,
585            acceleration: accel.clone(),
586            is_local,
587            requires_cloud_consent: !is_local,
588            trust_tier: m.trust_tier,
589            score,
590            within_recommendation_target,
591        },
592    }
593}
594
595/// 0.0–1.0 quality prior: published benchmarks when present, else a
596/// param-count heuristic (bigger ⇒ generally more capable, with diminishing
597/// returns).
598fn quality_score(m: &ModelSchema) -> f32 {
599    if !m.public_benchmarks.is_empty() {
600        let sum: f64 = m.public_benchmarks.iter().map(|b| b.score).sum();
601        return (sum / m.public_benchmarks.len() as f64).clamp(0.0, 1.0) as f32;
602    }
603    // Quality tracks *total* params (a 30B MoE has 30B-class knowledge even
604    // with 3B active). Saturating: 0.6B≈0.08, 4B≈0.36, 8B≈0.53, 30B≈0.81.
605    let b = param_billions_total(m).max(0.1);
606    (b / (b + 7.0)).clamp(0.0, 1.0)
607}
608
609/// 0.0–1.0 latency prior: smaller + better-accelerated ⇒ faster ⇒ higher.
610/// Uses *active* params — a 30B MoE with 3B active runs at 3B-ish speed.
611fn latency_score(m: &ModelSchema, accel: &SupportedAcceleration) -> f32 {
612    let b = param_billions_active(m).max(0.1);
613    // Smaller models score higher; 0.6B≈0.93, 4B≈0.6, 8B≈0.43, 30B≈0.14.
614    let size_term = 8.0 / (b + 8.0);
615    let accel_bonus = match accel {
616        SupportedAcceleration::Apple { .. } | SupportedAcceleration::Cuda { .. } => 0.1,
617        _ => 0.0,
618    };
619    (size_term + accel_bonus).clamp(0.0, 1.0)
620}
621
622/// Fraction of the memory budget this model consumes (0.0–1.0+). Used
623/// inverted in scoring so leaner picks win under memory-pressure weight.
624fn memory_pressure(estimate: &crate::resource_policy::ModelMemoryEstimate, budget: u64) -> f32 {
625    if budget == 0 {
626        return 1.0;
627    }
628    (estimate.estimated_peak_mb as f32 / budget as f32).clamp(0.0, 1.5)
629}
630
631/// Local memory fit verdict for this model on this machine.
632fn fit_status(
633    m: &ModelSchema,
634    hw: &HardwareInfo,
635    estimate: &crate::resource_policy::ModelMemoryEstimate,
636    memory_limits: &RecommendationMemoryLimits,
637) -> FitStatus {
638    // Explicitly external servers and remote/delegated models own their own
639    // memory. ManagedVllmMlx is CAR-owned and must pass the same local budget
640    // check as an in-process model; endpoint spelling never changes ownership.
641    if m.is_remote() || m.is_delegated() {
642        return FitStatus::ServerProvided;
643    }
644    // OS-provided implementations own their weights and memory, so the lack of
645    // a catalog size is affirmative ownership evidence rather than an unknown
646    // local-model estimate.
647    if m.is_os_provided() {
648        return FitStatus::Fits;
649    }
650    if m.size_mb() == 0 && m.ram_mb() == 0 {
651        return FitStatus::Unknown;
652    }
653    let required_mb = estimate.estimated_peak_mb;
654    let fits = if let Some(host_budget) = memory_limits.policy_host_budget_mb {
655        match hw.supported_acceleration() {
656            SupportedAcceleration::Cuda {
657                device_memory_mb: Some(device_memory_mb),
658            } => {
659                let host_required_mb = estimate
660                    .estimated_peak_mb
661                    .saturating_sub(estimate.weights_mb);
662                estimate.weights_mb <= device_memory_mb && host_required_mb <= host_budget
663            }
664            SupportedAcceleration::Cuda {
665                device_memory_mb: None,
666            } => return FitStatus::Unknown,
667            _ => required_mb <= host_budget,
668        }
669    } else {
670        required_mb <= memory_limits.legacy_budget_mb
671    };
672    if fits {
673        FitStatus::Fits
674    } else {
675        FitStatus::TooBig
676    }
677}
678
679/// Resident memory this model needs: weights + KV cache at a typical context
680/// + backend overhead.
681/// Memory available for a model after OS reserve, by acceleration tier.
682pub(crate) fn memory_budget_mb(hw: &HardwareInfo) -> u64 {
683    match hw.supported_acceleration() {
684        SupportedAcceleration::Apple { unified_memory_mb } => {
685            unified_memory_mb.saturating_sub(OS_RESERVE_MB)
686        }
687        SupportedAcceleration::Cuda { device_memory_mb } => {
688            device_memory_mb.unwrap_or(hw.total_ram_mb)
689        }
690        // CPU and unsupported-discrete both run from system RAM.
691        _ => hw.total_ram_mb.saturating_sub(OS_RESERVE_MB),
692    }
693}
694
695/// Total parameter count in billions from `param_count` ("4B", "30B (3B
696/// active)" → 30). Drives the *quality* prior. When `param_count` is blank
697/// (common for under-curated entries), estimate from on-disk size rather
698/// than treating the model as 0B — a 4-bit GGUF is ~0.6 GB per B params.
699fn param_billions_total(m: &ModelSchema) -> f32 {
700    model_parameter_billions_total(m)
701}
702
703/// Active parameter count in billions — the "(N active)" hint for MoE models,
704/// falling back to total for dense models. Drives *latency* and KV sizing.
705fn param_billions_active(m: &ModelSchema) -> f32 {
706    model_parameter_billions_active(m)
707}
708
709/// Plain-language rationale from the winning factors. Fixed templates, no
710/// free-form prose, so copy stays consistent and translatable.
711fn rationale(
712    m: &ModelSchema,
713    hw: &HardwareInfo,
714    use_case: UseCase,
715    tier: QualityTier,
716    fit: FitStatus,
717    quality: f32,
718) -> String {
719    let purpose = use_case_purpose(use_case);
720    let machine = match hw.supported_acceleration() {
721        SupportedAcceleration::Apple { unified_memory_mb } => {
722            format!(
723                "your {} GB Apple Silicon Mac (Metal)",
724                unified_memory_mb / 1024
725            )
726        }
727        SupportedAcceleration::Cuda { device_memory_mb } => match device_memory_mb {
728            Some(mb) => format!("your {} GB NVIDIA GPU (CUDA)", mb / 1024),
729            None => "your NVIDIA GPU (CUDA)".to_string(),
730        },
731        SupportedAcceleration::UnsupportedDiscreteGpu { .. } | SupportedAcceleration::Cpu => {
732            format!("your {} GB machine (CPU)", hw.total_ram_mb / 1024)
733        }
734    };
735
736    match fit {
737        FitStatus::ServerProvided
738            if matches!(&m.source, crate::schema::ModelSource::VllmMlx { .. }) =>
739        {
740            format!(
741                "{}: external server for {} — its operator runs the model, nothing to download",
742                m.name, purpose
743            )
744        }
745        FitStatus::ServerProvided if m.is_remote() => format!(
746            "{}: cloud model for {} — runs on Parslee's servers, nothing to download",
747            m.name, purpose
748        ),
749        FitStatus::ServerProvided => format!(
750            "{}: served externally for {} — no local memory needed",
751            m.name, purpose
752        ),
753        _ => {
754            let tier_word = match tier {
755                QualityTier::Fastest => "fastest",
756                QualityTier::Balanced => "best-balanced",
757                QualityTier::MostCapable => "most capable",
758            };
759            let quality_note = if quality >= 0.7 { "high-quality " } else { "" };
760            let size = if m.size_mb() >= 1024 {
761                format!("{:.1} GB download", m.size_mb() as f64 / 1024.0)
762            } else {
763                format!("{} MB download", m.size_mb())
764            };
765            format!(
766                "{}: the {} {}{} model that fits {} ({})",
767                m.name, tier_word, quality_note, purpose, machine, size
768            )
769        }
770    }
771}
772
773#[cfg(test)]
774mod tests {
775    use super::*;
776    use crate::hardware::{GpuBackend, GpuDevice, GpuVendor};
777    use crate::schema::{CostModel, ModelCapability, ModelSource, PerformanceEnvelope};
778
779    pub(super) fn hw(accel_backend: GpuBackend, ram_mb: u64, gpu_mb: Option<u64>) -> HardwareInfo {
780        HardwareInfo {
781            os: "test".into(),
782            arch: "test".into(),
783            cpu_cores: 8,
784            total_ram_mb: ram_mb,
785            gpu_backend: accel_backend,
786            gpu_memory_mb: gpu_mb,
787            gpu_devices: vec![],
788            recommended_model: String::new(),
789            recommended_context: 4096,
790            max_model_mb: 0,
791        }
792    }
793
794    pub(super) fn mac(ram_gb: u64) -> HardwareInfo {
795        // Metal backend ⇒ SupportedAcceleration::Apple with unified memory.
796        hw(GpuBackend::Metal, ram_gb * 1024, None)
797    }
798
799    pub(super) fn local_model(id: &str, name: &str, params: &str, size_mb: u64) -> ModelSchema {
800        ModelSchema {
801            id: id.into(),
802            name: name.into(),
803            provider: "qwen".into(),
804            family: "qwen3".into(),
805            version: String::new(),
806            capabilities: vec![ModelCapability::Generate, ModelCapability::Code],
807            context_length: 32768,
808            max_output_tokens: None,
809            param_count: params.into(),
810            quantization: Some(crate::schema::Quantization::parse("Q4_K_M")),
811            performance: PerformanceEnvelope::default(),
812            cost: CostModel {
813                size_mb: Some(size_mb),
814                ram_mb: Some(size_mb),
815                ..Default::default()
816            },
817            source: ModelSource::Local {
818                hf_repo: "x/y".into(),
819                hf_filename: "m.gguf".into(),
820                tokenizer_repo: "x/y".into(),
821            },
822            tags: vec![],
823            supported_params: vec![],
824            public_benchmarks: vec![],
825            trust_tier: TrustTier::Curated,
826            deprecated: false,
827            available: false,
828            weights_ready: false,
829        }
830    }
831
832    fn catalog() -> Vec<ModelSchema> {
833        vec![
834            local_model("qwen/qwen3-0.6b", "Qwen3-0.6B", "0.6B", 650),
835            local_model("qwen/qwen3-4b", "Qwen3-4B", "4B", 2500),
836            local_model("qwen/qwen3-8b", "Qwen3-8B", "8B", 4900),
837            local_model("qwen/qwen3-30b", "Qwen3-30B-A3B", "30B (3B active)", 17000),
838        ]
839    }
840
841    /// The same catalog as a Mac sees it: MLX rows. GGUF rows never run on
842    /// Apple Silicon — see [`platform_compatible`].
843    fn mac_catalog() -> Vec<ModelSchema> {
844        catalog().into_iter().map(as_mlx).collect()
845    }
846
847    fn as_mlx(mut m: ModelSchema) -> ModelSchema {
848        let name =
849            m.id.rsplit('/')
850                .next()
851                .expect("rsplit yields one")
852                .to_string();
853        m.id = format!("mlx/{name}");
854        m.source = ModelSource::Mlx {
855            hf_repo: format!("mlx-community/{}", m.name),
856            hf_weight_file: None,
857        };
858        m
859    }
860
861    fn qwen_mlx_policy_catalog() -> Vec<ModelSchema> {
862        let mut four = local_model("mlx/qwen3-4b:4bit", "Qwen3-4B-MLX", "4B", 2400);
863        four.source = ModelSource::Mlx {
864            hf_repo: "mlx-community/Qwen3-4B-4bit".into(),
865            hf_weight_file: None,
866        };
867        four.capabilities = vec![
868            ModelCapability::Generate,
869            ModelCapability::Code,
870            ModelCapability::ToolUse,
871        ];
872        four.performance.latency_p50_ms = Some(294);
873
874        let mut eight = local_model("mlx/qwen3-8b:4bit", "Qwen3-8B-MLX", "8B", 4800);
875        eight.source = ModelSource::Mlx {
876            hf_repo: "mlx-community/Qwen3-8B-4bit".into(),
877            hf_weight_file: None,
878        };
879        eight.capabilities = vec![
880            ModelCapability::Generate,
881            ModelCapability::Code,
882            ModelCapability::ToolUse,
883        ];
884        eight.performance.latency_p50_ms = Some(451);
885        vec![four, eight]
886    }
887
888    fn refs(v: &[ModelSchema]) -> Vec<&ModelSchema> {
889        v.iter().collect()
890    }
891
892    #[test]
893    fn fastest_prefers_the_small_model() {
894        let cat = mac_catalog();
895        let recs = recommend(
896            &refs(&cat),
897            &mac(36),
898            UseCase::Coding,
899            QualityTier::Fastest,
900            Privacy::OnDevice,
901        )
902        .picks;
903        assert_eq!(recs[0].display_name, "Qwen3-0.6B");
904    }
905
906    #[test]
907    fn downloadable_catalog_entry_is_not_installed_until_weights_are_ready() {
908        let mut model = qwen_mlx_policy_catalog().remove(0);
909        model.available = true;
910        model.weights_ready = false;
911
912        let set = recommend(
913            &[&model],
914            &mac(32),
915            UseCase::Assistant,
916            QualityTier::Balanced,
917            Privacy::OnDevice,
918        );
919
920        assert_eq!(set.picks.len(), 1);
921        assert!(!set.picks[0].already_installed);
922    }
923
924    #[test]
925    fn everyday_32gb_apple_assistant_balanced_prefers_four_b_and_keeps_eight_b() {
926        let mut catalog = qwen_mlx_policy_catalog();
927        let mut no_tools = local_model("mlx/qwen3-1.7b:3bit", "Qwen3-1.7B-MLX", "1.7B", 900);
928        no_tools.source = ModelSource::Mlx {
929            hf_repo: "mlx-community/Qwen3-1.7B-3bit".into(),
930            hf_weight_file: None,
931        };
932        no_tools.capabilities = vec![ModelCapability::Generate];
933        catalog.push(no_tools);
934        let set = recommend_with_policy(
935            &refs(&catalog),
936            &mac(32),
937            &crate::resource_policy::ResourcePolicy::everyday(),
938            UseCase::Assistant,
939            QualityTier::Balanced,
940            Privacy::OnDevice,
941        );
942
943        let ids: Vec<&str> = set
944            .picks
945            .iter()
946            .map(|pick| pick.model_id.as_str())
947            .collect();
948        assert_eq!(ids, ["mlx/qwen3-4b:4bit", "mlx/qwen3-8b:4bit"]);
949        let budget = crate::resource_policy::ResourcePolicy::everyday().effective_budget(32 * 1024);
950        for model in &catalog[..2] {
951            assert!(
952                estimate_model_memory(model, &mac(32), RECOMMENDATION_CONTEXT_TOKENS)
953                    .estimated_peak_mb
954                    < budget.configured_model_ceiling_mb
955            );
956        }
957    }
958
959    #[test]
960    fn everyday_target_prefers_under_half_ceiling_but_keeps_heavier_fit_visible() {
961        let mut four = qwen_mlx_policy_catalog().remove(0);
962        four.cost.ram_mb = Some(3_500);
963        four.cost.size_mb = Some(3_500);
964        let mut nine = four.clone();
965        nine.id = "mlx/qwen3-9b:4bit".into();
966        nine.name = "Qwen3-9B-MLX".into();
967        nine.param_count = "9B".into();
968        nine.cost.ram_mb = Some(9_000);
969        nine.cost.size_mb = Some(9_000);
970        nine.weights_ready = true;
971        nine.public_benchmarks = vec![crate::schema::BenchmarkScore {
972            name: "quality".into(),
973            score: 0.99,
974            harness: None,
975            source_url: None,
976            measured_at: None,
977            runs: None,
978            spread: None,
979        }];
980        let catalog = vec![nine, four];
981
982        let set = recommend_with_policy(
983            &refs(&catalog),
984            &mac(32),
985            &ResourcePolicy::everyday(),
986            UseCase::Assistant,
987            QualityTier::Balanced,
988            Privacy::OnDevice,
989        );
990
991        assert_eq!(set.picks[0].model_id, "mlx/qwen3-4b:4bit");
992        assert_eq!(set.picks[1].model_id, "mlx/qwen3-9b:4bit");
993        assert!(set.picks[0].within_recommendation_target);
994        assert!(!set.picks[1].within_recommendation_target);
995
996        let only_heavy = recommend_with_policy(
997            &[&catalog[0]],
998            &mac(32),
999            &ResourcePolicy::everyday(),
1000            UseCase::Assistant,
1001            QualityTier::Balanced,
1002            Privacy::OnDevice,
1003        );
1004        assert_eq!(only_heavy.picks[0].fit, FitStatus::Fits);
1005        assert!(!only_heavy.picks[0].within_recommendation_target);
1006    }
1007
1008    #[test]
1009    fn everyday_ranking_is_permutation_stable_across_local_and_cloud_candidates() {
1010        let mut four = qwen_mlx_policy_catalog().remove(0);
1011        four.cost.ram_mb = Some(3_500);
1012        four.cost.size_mb = Some(3_500);
1013
1014        let mut nine = four.clone();
1015        nine.id = "mlx/qwen3-9b:4bit".into();
1016        nine.name = "Qwen3-9B-MLX".into();
1017        nine.param_count = "9B".into();
1018        nine.cost.ram_mb = Some(9_000);
1019        nine.cost.size_mb = Some(9_000);
1020        nine.public_benchmarks = vec![crate::schema::BenchmarkScore {
1021            name: "quality".into(),
1022            score: 0.99,
1023            harness: None,
1024            source_url: None,
1025            measured_at: None,
1026            runs: None,
1027            spread: None,
1028        }];
1029
1030        let mut cloud = four.clone();
1031        cloud.id = "remote/tool-use".into();
1032        cloud.name = "ToolUse Cloud".into();
1033        cloud.source = ModelSource::RemoteApi {
1034            endpoint: "https://example.invalid".into(),
1035            api_key_env: "TEST_KEY".into(),
1036            api_key_envs: vec![],
1037            api_version: None,
1038            protocol: crate::schema::ApiProtocol::OpenAiCompat,
1039        };
1040        cloud.cost.ram_mb = None;
1041        cloud.cost.size_mb = None;
1042        cloud.public_benchmarks = vec![crate::schema::BenchmarkScore {
1043            name: "quality".into(),
1044            score: 1.0,
1045            harness: None,
1046            source_url: None,
1047            measured_at: None,
1048            runs: None,
1049            spread: None,
1050        }];
1051
1052        let candidates = [four, nine, cloud];
1053        let permutations = [
1054            [0, 1, 2],
1055            [0, 2, 1],
1056            [1, 0, 2],
1057            [1, 2, 0],
1058            [2, 0, 1],
1059            [2, 1, 0],
1060        ];
1061        let expected = ["mlx/qwen3-4b:4bit", "remote/tool-use", "mlx/qwen3-9b:4bit"];
1062        for permutation in permutations {
1063            let catalog: Vec<ModelSchema> = permutation
1064                .into_iter()
1065                .map(|index| candidates[index].clone())
1066                .collect();
1067            let set = recommend_with_policy(
1068                &refs(&catalog),
1069                &mac(32),
1070                &ResourcePolicy::everyday(),
1071                UseCase::Assistant,
1072                QualityTier::Balanced,
1073                Privacy::CloudOk,
1074            );
1075            let actual: Vec<&str> = set
1076                .picks
1077                .iter()
1078                .map(|pick| pick.model_id.as_str())
1079                .collect();
1080            assert_eq!(actual, expected, "permutation {permutation:?}");
1081        }
1082    }
1083
1084    #[test]
1085    fn cuda_policy_checks_gpu_weights_and_host_overhead_as_separate_pools() {
1086        let mut model = qwen_mlx_policy_catalog().remove(0);
1087        model.source = ModelSource::Local {
1088            hf_repo: "x/y".into(),
1089            hf_filename: "m.gguf".into(),
1090            tokenizer_repo: "x/y".into(),
1091        };
1092
1093        model.cost.ram_mb = Some(1_000);
1094        model.cost.size_mb = Some(1_000);
1095        let vram_too_small = recommend_with_policy(
1096            &[&model],
1097            &hw(GpuBackend::Cuda, 64 * 1024, Some(900)),
1098            &ResourcePolicy::everyday(),
1099            UseCase::Assistant,
1100            QualityTier::Balanced,
1101            Privacy::OnDevice,
1102        );
1103        assert_eq!(vram_too_small.not_enough_memory[0].fit, FitStatus::TooBig);
1104
1105        model.cost.ram_mb = Some(5_000);
1106        model.cost.size_mb = Some(5_000);
1107        // The CUDA weights use the 16 GB VRAM pool. Unknown geometry keeps
1108        // 2,688 MB of host overhead (512 runtime + 1,152 context + 1,024
1109        // transient), below a 16 GB Everyday target of 3,276 MB.
1110        let separate_pools_fit = recommend_with_policy(
1111            &[&model],
1112            &hw(GpuBackend::Cuda, 16 * 1024, Some(16 * 1024)),
1113            &ResourcePolicy::everyday(),
1114            UseCase::Assistant,
1115            QualityTier::Balanced,
1116            Privacy::OnDevice,
1117        );
1118        assert_eq!(separate_pools_fit.picks[0].fit, FitStatus::Fits);
1119        assert!(separate_pools_fit.picks[0].within_recommendation_target);
1120    }
1121
1122    #[test]
1123    fn custom_zero_blocks_automatic_and_explicit_local_fit() {
1124        let model = qwen_mlx_policy_catalog().remove(0);
1125        let set = recommend_with_policy(
1126            &[&model],
1127            &mac(32),
1128            &ResourcePolicy::custom_gb(0.0).unwrap(),
1129            UseCase::Assistant,
1130            QualityTier::Balanced,
1131            Privacy::OnDevice,
1132        );
1133
1134        assert!(set.picks.is_empty());
1135        assert_eq!(set.not_enough_memory[0].fit, FitStatus::TooBig);
1136        assert!(!set.not_enough_memory[0].within_recommendation_target);
1137    }
1138
1139    #[test]
1140    fn local_focused_uses_its_full_configured_ceiling() {
1141        let mut model = qwen_mlx_policy_catalog().remove(0);
1142        // 10,400 weights + 512 runtime + 1,152 context + 1,024 transient =
1143        // 13,088 MB: below Local Focused's full 13,107 MB ceiling, but well
1144        // above Everyday's 6,553 MB ceiling on this 16 GB fixture.
1145        model.cost.ram_mb = Some(10_400);
1146        model.cost.size_mb = Some(10_400);
1147        let set = recommend_with_policy(
1148            &[&model],
1149            &mac(16),
1150            &ResourcePolicy::local_focused(),
1151            UseCase::Assistant,
1152            QualityTier::Balanced,
1153            Privacy::OnDevice,
1154        );
1155
1156        assert_eq!(set.picks[0].fit, FitStatus::Fits);
1157        assert!(set.picks[0].within_recommendation_target);
1158    }
1159
1160    #[test]
1161    fn unknown_memory_never_outranks_a_known_fit() {
1162        let mut known = qwen_mlx_policy_catalog().remove(0);
1163        known.public_benchmarks.clear();
1164        let mut unknown = known.clone();
1165        unknown.id = "local/unknown-memory".into();
1166        unknown.name = "Unknown Memory".into();
1167        unknown.param_count.clear();
1168        unknown.cost.ram_mb = None;
1169        unknown.cost.size_mb = None;
1170        unknown.public_benchmarks = vec![crate::schema::BenchmarkScore {
1171            name: "quality".into(),
1172            score: 1.0,
1173            harness: None,
1174            source_url: None,
1175            measured_at: None,
1176            runs: None,
1177            spread: None,
1178        }];
1179
1180        let set = recommend_with_policy(
1181            &refs(&[unknown, known]),
1182            &mac(32),
1183            &ResourcePolicy::everyday(),
1184            UseCase::Assistant,
1185            QualityTier::Balanced,
1186            Privacy::OnDevice,
1187        );
1188        assert_eq!(set.picks[0].model_id, "mlx/qwen3-4b:4bit");
1189        assert_eq!(set.picks[1].fit, FitStatus::Unknown);
1190        assert!(!set.picks[1].within_recommendation_target);
1191    }
1192
1193    #[test]
1194    fn everyday_assistant_tool_floor_excludes_generate_only_cloud_models() {
1195        let mut local = qwen_mlx_policy_catalog().remove(0);
1196        local.public_benchmarks.clear();
1197        let mut cloud = local.clone();
1198        cloud.id = "remote/high-score-generate-only".into();
1199        cloud.name = "Remote Generate Only".into();
1200        cloud.source = ModelSource::RemoteApi {
1201            endpoint: "https://api".into(),
1202            api_key_env: "K".into(),
1203            api_key_envs: vec![],
1204            api_version: None,
1205            protocol: crate::schema::ApiProtocol::OpenAiCompat,
1206        };
1207        cloud.capabilities = vec![ModelCapability::Generate];
1208        cloud.public_benchmarks = vec![crate::schema::BenchmarkScore {
1209            name: "quality".into(),
1210            score: 1.0,
1211            harness: None,
1212            source_url: None,
1213            measured_at: None,
1214            runs: None,
1215            spread: None,
1216        }];
1217
1218        let set = recommend_with_policy(
1219            &refs(&[cloud, local]),
1220            &mac(32),
1221            &ResourcePolicy::everyday(),
1222            UseCase::Assistant,
1223            QualityTier::Balanced,
1224            Privacy::CloudOk,
1225        );
1226        assert_eq!(set.picks.len(), 1);
1227        assert_eq!(set.picks[0].model_id, "mlx/qwen3-4b:4bit");
1228    }
1229
1230    /// The assistant tool floor is a property of the use case, not of the tier
1231    /// or the entry point.
1232    ///
1233    /// Positive control: with the requirement welded to
1234    /// `everyday_assistant_balanced`, every assertion below except the
1235    /// `Balanced` one fails — `Fastest` and `MostCapable` returned the
1236    /// generate-only model, and `recommend` (no policy) returned it at every
1237    /// tier. That is car#1822: `--tier fastest` ranked Qwen3-0.6B first and
1238    /// told the reader to install it with `car setup --for assistant`, while
1239    /// `car do` refuses a model without `ToolUse` and substitutes one.
1240    #[test]
1241    fn the_assistant_tool_floor_holds_at_every_tier_and_entry_point() {
1242        let mut tool_capable = as_mlx(local_model("mlx/tools-4b", "Tools-4B", "4B", 2500));
1243        tool_capable.capabilities.push(ModelCapability::ToolUse);
1244        // `local_model` is Generate+Code, so this one is the excluded shape.
1245        let generate_only = as_mlx(local_model("mlx/chat-0.6b", "Chat-0.6B", "0.6B", 650));
1246        let models = [tool_capable, generate_only];
1247
1248        for tier in [
1249            QualityTier::Fastest,
1250            QualityTier::Balanced,
1251            QualityTier::MostCapable,
1252        ] {
1253            for with_policy in [false, true] {
1254                let set = if with_policy {
1255                    recommend_with_policy(
1256                        &refs(&models),
1257                        &mac(32),
1258                        &ResourcePolicy::everyday(),
1259                        UseCase::Assistant,
1260                        tier,
1261                        Privacy::OnDevice,
1262                    )
1263                } else {
1264                    recommend(
1265                        &refs(&models),
1266                        &mac(32),
1267                        UseCase::Assistant,
1268                        tier,
1269                        Privacy::OnDevice,
1270                    )
1271                };
1272                let ids: Vec<&str> = set.picks.iter().map(|p| p.model_id.as_str()).collect();
1273                assert_eq!(
1274                    ids,
1275                    vec!["mlx/tools-4b"],
1276                    "tier {tier:?}, with_policy {with_policy}: a model without ToolUse \
1277                     cannot serve the assistant"
1278                );
1279            }
1280        }
1281
1282        // Control: the floor is scoped to Assistant. Coding ranks both, so the
1283        // assertion above is about the use case and not about the catalog.
1284        let coding = recommend(
1285            &refs(&models),
1286            &mac(32),
1287            UseCase::Coding,
1288            QualityTier::Fastest,
1289            Privacy::OnDevice,
1290        );
1291        assert_eq!(
1292            coding.picks.len(),
1293            2,
1294            "coding has no tool floor; both candidates should rank"
1295        );
1296    }
1297
1298    #[test]
1299    fn policy_entry_point_preserves_legacy_order_among_policy_eligible_candidates() {
1300        let catalog = mac_catalog();
1301        for (policy, use_case, tier) in [
1302            (
1303                ResourcePolicy::everyday(),
1304                UseCase::Assistant,
1305                QualityTier::Fastest,
1306            ),
1307            (
1308                ResourcePolicy::everyday(),
1309                UseCase::Assistant,
1310                QualityTier::MostCapable,
1311            ),
1312            (
1313                ResourcePolicy::everyday(),
1314                UseCase::Coding,
1315                QualityTier::Balanced,
1316            ),
1317            (
1318                ResourcePolicy::local_focused(),
1319                UseCase::Assistant,
1320                QualityTier::Balanced,
1321            ),
1322        ] {
1323            let legacy = recommend(&refs(&catalog), &mac(36), use_case, tier, Privacy::OnDevice);
1324            let policy_aware = recommend_with_policy(
1325                &refs(&catalog),
1326                &mac(36),
1327                &policy,
1328                use_case,
1329                tier,
1330                Privacy::OnDevice,
1331            );
1332            let legacy_common: Vec<&str> = legacy
1333                .picks
1334                .iter()
1335                .filter(|pick| {
1336                    policy_aware
1337                        .picks
1338                        .iter()
1339                        .any(|candidate| candidate.model_id == pick.model_id)
1340                })
1341                .map(|pick| pick.model_id.as_str())
1342                .collect();
1343            let policy_common: Vec<&str> = policy_aware
1344                .picks
1345                .iter()
1346                .filter(|pick| {
1347                    legacy
1348                        .picks
1349                        .iter()
1350                        .any(|candidate| candidate.model_id == pick.model_id)
1351                })
1352                .map(|pick| pick.model_id.as_str())
1353                .collect();
1354            assert_eq!(
1355                policy_common, legacy_common,
1356                "{policy:?} {use_case:?} {tier:?}"
1357            );
1358        }
1359    }
1360
1361    #[test]
1362    fn most_capable_prefers_the_big_model_when_it_fits() {
1363        let cat = mac_catalog();
1364        let recs = recommend(
1365            &refs(&cat),
1366            &mac(36), // 36 GB Mac fits the 17 GB model
1367            UseCase::Coding,
1368            QualityTier::MostCapable,
1369            Privacy::OnDevice,
1370        )
1371        .picks;
1372        assert_eq!(recs[0].display_name, "Qwen3-30B-A3B");
1373        assert_eq!(recs[0].fit, FitStatus::Fits);
1374    }
1375
1376    #[test]
1377    fn too_big_models_are_excluded_on_small_machines() {
1378        let cat = mac_catalog();
1379        let recs = recommend(
1380            &refs(&cat),
1381            &mac(8), // 8 GB: 17 GB and 8 GB models shouldn't be offered
1382            UseCase::Coding,
1383            QualityTier::MostCapable,
1384            Privacy::OnDevice,
1385        )
1386        .picks;
1387        let names: Vec<&str> = recs.iter().map(|r| r.display_name.as_str()).collect();
1388        assert!(!names.contains(&"Qwen3-30B-A3B"), "30B must not fit 8GB");
1389        assert!(recs.iter().all(|r| r.fit == FitStatus::Fits));
1390        assert!(!recs.is_empty(), "the 0.6B model should still be offered");
1391    }
1392
1393    #[test]
1394    fn balanced_picks_a_capable_model_that_fits() {
1395        let cat = mac_catalog();
1396        let recs = recommend(
1397            &refs(&cat),
1398            &mac(16),
1399            UseCase::Coding,
1400            QualityTier::Balanced,
1401            Privacy::OnDevice,
1402        )
1403        .picks;
1404        // On a 16 GB Mac, Balanced should land on a mid model, not the 0.6B
1405        // and not necessarily the largest.
1406        assert!(matches!(
1407            recs[0].display_name.as_str(),
1408            "Qwen3-4B" | "Qwen3-8B"
1409        ));
1410    }
1411
1412    /// On Apple Silicon a GGUF row never runs — a request for it executes as
1413    /// its MLX twin — so it is not offered there, and each Qwen3 appears once.
1414    /// Everywhere else the GGUF row is the local model.
1415    #[test]
1416    fn gguf_rows_are_offered_everywhere_but_apple_silicon() {
1417        let gguf = catalog();
1418        let both: Vec<ModelSchema> = gguf.iter().cloned().chain(mac_catalog()).collect();
1419        let on_mac = recommend(
1420            &refs(&both),
1421            &mac(64),
1422            UseCase::Coding,
1423            QualityTier::Balanced,
1424            Privacy::OnDevice,
1425        );
1426        assert!(!on_mac.picks.is_empty());
1427        assert!(on_mac
1428            .picks
1429            .iter()
1430            .chain(&on_mac.not_enough_memory)
1431            .all(|p| !gguf.iter().any(|g| g.id == p.model_id)));
1432        let cuda = hw(GpuBackend::Cuda, 64 * 1024, Some(24 * 1024));
1433        let on_cuda = recommend(
1434            &refs(&both),
1435            &cuda,
1436            UseCase::Coding,
1437            QualityTier::Balanced,
1438            Privacy::OnDevice,
1439        );
1440        assert!(!on_cuda.picks.is_empty());
1441        assert!(on_cuda
1442            .picks
1443            .iter()
1444            .all(|p| gguf.iter().any(|g| g.id == p.model_id)));
1445        assert!(!platform_compatible(&gguf[0], &mac(64)));
1446        assert!(platform_compatible(&gguf[0], &cuda));
1447    }
1448
1449    #[test]
1450    fn search_only_returns_embedding_models() {
1451        let mut cat = mac_catalog();
1452        let mut embed = as_mlx(local_model("qwen/embed", "Qwen3-Embedding", "0.6B", 640));
1453        embed.capabilities = vec![ModelCapability::Embed];
1454        cat.push(embed);
1455        let recs = recommend(
1456            &refs(&cat),
1457            &mac(16),
1458            UseCase::Search,
1459            QualityTier::Balanced,
1460            Privacy::OnDevice,
1461        )
1462        .picks;
1463        assert_eq!(recs.len(), 1, "only the embed model is in the Search lane");
1464        assert_eq!(recs[0].display_name, "Qwen3-Embedding");
1465        assert_eq!(recs[0].role, UseCaseRole::Retrieval);
1466    }
1467
1468    #[test]
1469    fn deprecated_models_are_never_recommended() {
1470        let mut cat = mac_catalog();
1471        cat[1].deprecated = true; // deprecate the 4B
1472        let recs = recommend(
1473            &refs(&cat),
1474            &mac(16),
1475            UseCase::Coding,
1476            QualityTier::Balanced,
1477            Privacy::OnDevice,
1478        )
1479        .picks;
1480        assert!(recs.iter().all(|r| r.display_name != "Qwen3-4B"));
1481    }
1482
1483    #[test]
1484    fn on_device_excludes_cloud_but_cloud_ok_includes_it_with_consent() {
1485        let mut cat = mac_catalog();
1486        let mut cloud = local_model("anthropic/sonnet", "Claude Sonnet", "", 0);
1487        cloud.capabilities = vec![ModelCapability::Generate, ModelCapability::Code];
1488        cloud.source = ModelSource::RemoteApi {
1489            endpoint: "https://api".into(),
1490            api_key_env: "K".into(),
1491            api_key_envs: vec![],
1492            api_version: None,
1493            protocol: crate::schema::ApiProtocol::Anthropic,
1494        };
1495        cloud.public_benchmarks = vec![crate::schema::BenchmarkScore {
1496            name: "SWE-bench".into(),
1497            score: 0.7,
1498            harness: None,
1499            source_url: None,
1500            measured_at: None,
1501            runs: None,
1502            spread: None,
1503        }];
1504        cat.push(cloud);
1505
1506        let on_device = recommend(
1507            &refs(&cat),
1508            &mac(16),
1509            UseCase::Coding,
1510            QualityTier::MostCapable,
1511            Privacy::OnDevice,
1512        )
1513        .picks;
1514        assert!(on_device.iter().all(|r| r.is_local));
1515
1516        let cloud_ok = recommend(
1517            &refs(&cat),
1518            &mac(16),
1519            UseCase::Coding,
1520            QualityTier::MostCapable,
1521            Privacy::CloudOk,
1522        )
1523        .picks;
1524        let claude = cloud_ok
1525            .iter()
1526            .find(|r| r.display_name == "Claude Sonnet")
1527            .expect("cloud model eligible under CloudOk");
1528        assert!(claude.requires_cloud_consent);
1529        assert_eq!(claude.fit, FitStatus::ServerProvided);
1530    }
1531
1532    #[test]
1533    fn metal_only_model_excluded_on_cpu_host() {
1534        let mut cat = catalog();
1535        let mut mlx = local_model("mlx/qwen3-4b", "Qwen3-4B-MLX", "4B", 2400);
1536        mlx.source = ModelSource::Mlx {
1537            hf_repo: "mlx-community/x".into(),
1538            hf_weight_file: None,
1539        };
1540        cat.push(mlx);
1541        // CPU-only Linux box.
1542        let recs = recommend(
1543            &refs(&cat),
1544            &hw(GpuBackend::Cpu, 32 * 1024, None),
1545            UseCase::Coding,
1546            QualityTier::Balanced,
1547            Privacy::OnDevice,
1548        )
1549        .picks;
1550        assert!(recs.iter().all(|r| r.display_name != "Qwen3-4B-MLX"));
1551    }
1552
1553    #[test]
1554    fn ranking_is_deterministic() {
1555        let cat = mac_catalog();
1556        let a = recommend(
1557            &refs(&cat),
1558            &mac(16),
1559            UseCase::Assistant,
1560            QualityTier::Balanced,
1561            Privacy::OnDevice,
1562        );
1563        let b = recommend(
1564            &refs(&cat),
1565            &mac(16),
1566            UseCase::Assistant,
1567            QualityTier::Balanced,
1568            Privacy::OnDevice,
1569        );
1570        let ids_a: Vec<&str> = a.picks.iter().map(|r| r.model_id.as_str()).collect();
1571        let ids_b: Vec<&str> = b.picks.iter().map(|r| r.model_id.as_str()).collect();
1572        assert_eq!(ids_a, ids_b);
1573    }
1574
1575    #[test]
1576    fn rationale_is_plain_language_no_jargon() {
1577        let cat = mac_catalog();
1578        let recs = recommend(
1579            &refs(&cat),
1580            &mac(36),
1581            UseCase::Coding,
1582            QualityTier::Balanced,
1583            Privacy::OnDevice,
1584        )
1585        .picks;
1586        let r = &recs[0].rationale;
1587        assert!(!r.contains("Q4_K_M"), "no quantization jargon");
1588        assert!(!r.contains("gguf") && !r.contains("hf_repo"));
1589        assert!(r.contains("coding"), "states the purpose");
1590    }
1591
1592    #[test]
1593    fn all_too_big_surfaces_needs_more_ram_with_a_note() {
1594        // A 2 GB machine fits nothing in the catalog.
1595        let cat = catalog();
1596        let set = recommend(
1597            &refs(&cat),
1598            &hw(GpuBackend::Cpu, 2 * 1024, None),
1599            UseCase::Coding,
1600            QualityTier::Balanced,
1601            Privacy::OnDevice,
1602        );
1603        assert!(set.picks.is_empty(), "nothing should fit 2 GB");
1604        assert!(
1605            !set.not_enough_memory.is_empty(),
1606            "too-big models surfaced, not dropped"
1607        );
1608        let note = set.note.expect("empty picks must carry a note");
1609        assert!(note.contains("fits"), "note explains the no-fit: {note}");
1610        // Closest miss ranked first.
1611        assert_eq!(set.not_enough_memory[0].fit, FitStatus::TooBig);
1612    }
1613
1614    #[test]
1615    fn all_deprecated_gives_generic_note_not_a_memory_note() {
1616        // Deprecated models are filtered before the fit partition, so they
1617        // land in neither picks nor not_enough_memory — the note must be the
1618        // generic "no model available", not a misleading "needs more RAM".
1619        let mut cat = mac_catalog();
1620        for m in &mut cat {
1621            m.deprecated = true;
1622        }
1623        let set = recommend(
1624            &refs(&cat),
1625            &mac(36), // plenty of RAM — so a memory note would be wrong
1626            UseCase::Coding,
1627            QualityTier::Balanced,
1628            Privacy::OnDevice,
1629        );
1630        assert!(set.picks.is_empty());
1631        assert!(set.not_enough_memory.is_empty());
1632        let note = set.note.expect("must explain");
1633        assert!(
1634            !note.contains("fits") && !note.contains("memory"),
1635            "deprecated-only must not claim a memory problem: {note}"
1636        );
1637    }
1638
1639    #[test]
1640    fn not_enough_memory_is_ordered_deterministically() {
1641        let cat = catalog();
1642        let mk = || {
1643            recommend(
1644                &refs(&cat),
1645                &hw(GpuBackend::Cpu, 3 * 1024, None), // only the 0.6B fits
1646                UseCase::Coding,
1647                QualityTier::Balanced,
1648                Privacy::OnDevice,
1649            )
1650            .not_enough_memory
1651            .into_iter()
1652            .map(|r| r.model_id)
1653            .collect::<Vec<_>>()
1654        };
1655        assert!(mk().len() >= 2, "several models should be too big for 3 GB");
1656        assert_eq!(mk(), mk(), "too-big ordering must be deterministic");
1657    }
1658
1659    #[test]
1660    fn empty_registry_returns_empty_with_a_note() {
1661        let set = recommend(
1662            &[],
1663            &mac(16),
1664            UseCase::Assistant,
1665            QualityTier::Balanced,
1666            Privacy::OnDevice,
1667        );
1668        assert!(set.picks.is_empty());
1669        assert!(set.not_enough_memory.is_empty());
1670        assert!(set.note.is_some(), "no-model case must explain itself");
1671    }
1672
1673    #[test]
1674    fn cuda_box_sizes_against_vram() {
1675        // A 24 GB CUDA GPU fits the 17 GB model under MostCapable.
1676        let cat = catalog();
1677        let h = hw(GpuBackend::Cuda, 64 * 1024, Some(24 * 1024));
1678        let recs = recommend(
1679            &refs(&cat),
1680            &h,
1681            UseCase::Coding,
1682            QualityTier::MostCapable,
1683            Privacy::OnDevice,
1684        )
1685        .picks;
1686        assert_eq!(recs[0].display_name, "Qwen3-30B-A3B");
1687    }
1688
1689    #[test]
1690    fn unsupported_discrete_gpu_uses_system_ram_not_vram() {
1691        // A 24 GB discrete GPU CAR can't drive must NOT be used as the budget;
1692        // a 16 GB-RAM CPU host can't fit the 17 GB model despite the big card.
1693        let cat = catalog();
1694        let mut h = hw(GpuBackend::Cpu, 16 * 1024, None);
1695        h.gpu_devices = vec![GpuDevice {
1696            vendor: GpuVendor::Nvidia,
1697            name: "GeForce RTX 4090".into(),
1698            memory_mb: Some(24_000),
1699        }];
1700        // Sanity: this is the UnsupportedDiscreteGpu tier.
1701        assert!(matches!(
1702            h.supported_acceleration(),
1703            crate::hardware::SupportedAcceleration::UnsupportedDiscreteGpu { .. }
1704        ));
1705        let recs = recommend(
1706            &refs(&cat),
1707            &h,
1708            UseCase::Coding,
1709            QualityTier::MostCapable,
1710            Privacy::OnDevice,
1711        )
1712        .picks;
1713        assert!(
1714            recs.iter().all(|r| r.display_name != "Qwen3-30B-A3B"),
1715            "17 GB model must not fit a 16 GB-RAM CPU host"
1716        );
1717        assert!(!recs.is_empty(), "smaller models still fit");
1718    }
1719
1720    #[test]
1721    fn recommendation_set_wire_shape_is_snake_case_and_stable() {
1722        // Guards the JSON contract that FFI / JSON-RPC clients decode.
1723        let cat = mac_catalog();
1724        let set = recommend(
1725            &refs(&cat),
1726            &mac(36),
1727            UseCase::Coding,
1728            QualityTier::Balanced,
1729            Privacy::OnDevice,
1730        );
1731        let json = serde_json::to_string(&set).unwrap();
1732        assert!(json.contains("\"picks\""));
1733        assert!(json.contains("\"not_enough_memory\""));
1734        assert!(json.contains("\"model_id\""));
1735        assert!(json.contains("\"already_installed\""));
1736        assert!(json.contains("\"requires_cloud_consent\""));
1737        assert!(json.contains("\"within_recommendation_target\""));
1738        assert!(json.contains("\"fit\""));
1739
1740        let mut legacy = serde_json::to_value(&set.picks[0]).unwrap();
1741        legacy
1742            .as_object_mut()
1743            .unwrap()
1744            .remove("within_recommendation_target");
1745        let decoded: Recommendation = serde_json::from_value(legacy).unwrap();
1746        assert!(decoded.within_recommendation_target);
1747    }
1748
1749    #[test]
1750    fn blank_param_count_estimates_from_size_not_zero() {
1751        // An under-curated entry with no param_count must not be treated as a
1752        // 0B model (which would falsely look tiny + low quality).
1753        let mut m = local_model("x/unknown", "Unknown-Model", "", 4900);
1754        m.param_count = String::new();
1755        assert!(
1756            param_billions_total(&m) > 5.0,
1757            "4.9 GB ⇒ roughly an 8B model, not 0B"
1758        );
1759    }
1760
1761    // --- models.list_unified fit annotation (car#1399) --------------------
1762
1763    fn cloud_row(id: &str) -> ModelSchema {
1764        let mut cloud = local_model(id, "Cloud", "", 0);
1765        cloud.source = ModelSource::RemoteApi {
1766            endpoint: "https://example.invalid".into(),
1767            api_key_env: "TEST_KEY".into(),
1768            api_key_envs: vec![],
1769            api_version: None,
1770            protocol: crate::schema::ApiProtocol::OpenAiCompat,
1771        };
1772        cloud.cost.ram_mb = None;
1773        cloud.cost.size_mb = None;
1774        cloud
1775    }
1776
1777    /// Everyday keeps 40% of unified memory for local models: 3,276 MB on an
1778    /// 8 GB Mac, 6,553 MB on 16 GB, 13,107 MB on 32 GB. Unknown fixture
1779    /// geometry conservatively reserves 1,152 MB of context at 8,192 tokens,
1780    /// yielding Metal peaks of 3,188 MB (0.6B), 5,088 MB (4B), 7,488 MB (8B),
1781    /// and 19,188 MB (30B-A3B).
1782    #[test]
1783    fn unified_fit_is_the_recommenders_verdict_on_every_machine_size() {
1784        let policy = ResourcePolicy::everyday();
1785        let cat = qwen_mlx_policy_catalog();
1786        let four = &cat[0];
1787        let eight = &cat[1];
1788        let small = local_model("mlx/qwen3-0.6b:6bit", "Qwen3-0.6B", "0.6B", 500);
1789        let thirty = local_model(
1790            "mlx/qwen3-30b-a3b:4bit",
1791            "Qwen3-30B-A3B",
1792            "30B (3B active)",
1793            16_500,
1794        );
1795        let at = |m: &ModelSchema, gb: u64| model_fit(m, &mac(gb), Some(&policy));
1796
1797        assert_eq!(at(eight, 8).fit, ModelFitStatus::TooBig);
1798        assert_eq!(at(eight, 16).fit, ModelFitStatus::TooBig);
1799        assert_eq!(at(eight, 32).fit, ModelFitStatus::Fits);
1800        assert_eq!(at(four, 8).fit, ModelFitStatus::TooBig);
1801        assert_eq!(at(four, 16).fit, ModelFitStatus::Fits);
1802        assert_eq!(at(&small, 8).fit, ModelFitStatus::Fits);
1803        assert_eq!(at(&thirty, 16).fit, ModelFitStatus::TooBig);
1804        assert_eq!(at(&thirty, 32).fit, ModelFitStatus::TooBig);
1805
1806        // The published estimate is the one the verdict was judged on, and
1807        // the accelerator can run every one of these on a Mac.
1808        let eight_at_32 = at(eight, 32);
1809        assert_eq!(
1810            eight_at_32.estimated_peak_mb,
1811            Some(
1812                estimate_model_memory(eight, &mac(32), RECOMMENDATION_CONTEXT_TOKENS)
1813                    .estimated_peak_mb
1814            )
1815        );
1816        assert!(eight_at_32.platform_compatible);
1817
1818        // Same rule as the recommender: whatever it partitions as a pick
1819        // reads `fits`, whatever it lists under not_enough_memory reads
1820        // `too_big`, on each machine.
1821        let all = vec![four.clone(), eight.clone(), small, thirty];
1822        let by_id = |id: &str| all.iter().find(|m| m.id == id).unwrap();
1823        for gb in [8u64, 16, 32] {
1824            let set = recommend_with_policy(
1825                &refs(&all),
1826                &mac(gb),
1827                &policy,
1828                UseCase::Assistant,
1829                QualityTier::Balanced,
1830                Privacy::OnDevice,
1831            );
1832            for pick in &set.picks {
1833                assert_eq!(
1834                    model_fit(by_id(&pick.model_id), &mac(gb), Some(&policy)).fit,
1835                    ModelFitStatus::Fits,
1836                    "{gb} GB pick {}",
1837                    pick.model_id
1838                );
1839            }
1840            for miss in &set.not_enough_memory {
1841                assert_eq!(
1842                    model_fit(by_id(&miss.model_id), &mac(gb), Some(&policy)).fit,
1843                    ModelFitStatus::TooBig,
1844                    "{gb} GB miss {}",
1845                    miss.model_id
1846                );
1847            }
1848        }
1849    }
1850
1851    /// A model no machine could ever hold reads `too_big` on every machine
1852    /// CAR runs on — including the CUDA-compiled Linux/Windows build with no
1853    /// NVIDIA card, which used to answer `unknown` and so showed the row.
1854    ///
1855    /// This is the fit half of `car-cli`'s
1856    /// `models_list_hides_too_big_and_deprecated_rows_unless_all`. That test
1857    /// spawns the real binary, so its verdict came from whatever hardware the
1858    /// runner reported: it passed under `ci.yml` (`--cfg=car_skip_cuda`, CPU
1859    /// tier) and failed under `build.yml` (CUDA toolkit installed, no GPU) on
1860    /// the very same commit. Judging the fixture here pins the rule to the
1861    /// machine profile instead of to the host.
1862    #[test]
1863    fn a_model_larger_than_any_machine_is_too_big_on_every_machine() {
1864        let policy = ResourcePolicy::everyday();
1865        // The `car models list` fixture: 900 TB of weights.
1866        let enormous = local_model("test/enormous-model:q4", "Enormous", "9000B", 900_000_000);
1867        let machines = [
1868            ("apple 8 GB", mac(8)),
1869            ("apple 128 GB", mac(128)),
1870            ("cpu 32 GB", hw(GpuBackend::Cpu, 32 * 1024, None)),
1871            // The shipped x86_64 Linux/Windows build on a GPU-less box:
1872            // `detect_gpu_backend` says Cuda from a `#[cfg]`, nvidia-smi
1873            // reports nothing.
1874            ("cuda build, no card", hw(GpuBackend::Cuda, 64 * 1024, None)),
1875            (
1876                "cuda 24 GB card",
1877                hw(GpuBackend::Cuda, 64 * 1024, Some(24 * 1024)),
1878            ),
1879        ];
1880        for (label, machine) in machines {
1881            assert_eq!(
1882                model_fit(&enormous, &machine, Some(&policy)).fit,
1883                ModelFitStatus::TooBig,
1884                "{label} must not claim to hold a 900 TB model"
1885            );
1886            assert_eq!(
1887                model_fit(&enormous, &machine, None).fit,
1888                ModelFitStatus::TooBig,
1889                "{label} without a policy must not claim to hold a 900 TB model"
1890            );
1891        }
1892    }
1893
1894    /// The same machine still judges a model it CAN hold as fitting, so the
1895    /// rule above is a memory verdict and not a blanket refusal.
1896    #[test]
1897    fn cuda_build_without_a_card_still_fits_models_that_fit_system_ram() {
1898        let policy = ResourcePolicy::everyday();
1899        let small = local_model("qwen/qwen3-0.6b:q4_k_m", "Qwen3-0.6B", "0.6B", 500);
1900        let no_card = hw(GpuBackend::Cuda, 64 * 1024, None);
1901        assert_eq!(
1902            model_fit(&small, &no_card, Some(&policy)).fit,
1903            ModelFitStatus::Fits
1904        );
1905    }
1906
1907    #[test]
1908    fn unified_fit_platform_check_is_the_base_filters() {
1909        let cat = qwen_mlx_policy_catalog();
1910        let mlx = &cat[0];
1911        let cpu_box = hw(GpuBackend::Cpu, 32 * 1024, None);
1912        let fit = model_fit(mlx, &cpu_box, Some(&ResourcePolicy::everyday()));
1913        assert!(!fit.platform_compatible, "MLX needs Apple Silicon");
1914        assert!(!passes_base_filter(
1915            mlx,
1916            &cpu_box,
1917            UseCase::Coding,
1918            Privacy::OnDevice
1919        ));
1920        assert!(platform_compatible(mlx, &mac(32)));
1921
1922        let gguf = local_model("qwen/qwen3-4b:q4_k_m", "Qwen3-4B", "4B", 2_500);
1923        assert!(model_fit(&gguf, &cpu_box, None).platform_compatible);
1924        assert!(passes_base_filter(
1925            &gguf,
1926            &cpu_box,
1927            UseCase::Coding,
1928            Privacy::OnDevice
1929        ));
1930    }
1931
1932    #[test]
1933    fn unified_fit_for_rows_whose_memory_is_not_this_machines() {
1934        // Remote: the server owns its memory → fits, no estimate, any platform.
1935        let cloud = cloud_row("remote/cloud");
1936        let fit = model_fit(&cloud, &mac(8), Some(&ResourcePolicy::everyday()));
1937        assert_eq!(fit.fit, ModelFitStatus::Fits);
1938        assert_eq!(fit.estimated_peak_mb, None);
1939        assert!(fit.platform_compatible);
1940        assert!(model_fit(&cloud, &hw(GpuBackend::Cpu, 8 * 1024, None), None).platform_compatible);
1941
1942        // A local row that declares neither size nor RAM: nothing to judge.
1943        let mut undeclared = local_model("local/undeclared", "Undeclared", "4B", 0);
1944        undeclared.cost.ram_mb = None;
1945        undeclared.cost.size_mb = None;
1946        let fit = model_fit(&undeclared, &mac(32), Some(&ResourcePolicy::everyday()));
1947        assert_eq!(fit.fit, ModelFitStatus::Unknown);
1948        assert_eq!(fit.estimated_peak_mb, None);
1949
1950        // OS-owned rows fit by definition because their memory is outside the
1951        // CAR model budget. Their own platform requirement remains separate.
1952        let mut foundation = local_model("apple/foundation:default", "Apple", "", 0);
1953        foundation.source = ModelSource::AppleFoundationModels { use_case: None };
1954        foundation.cost.ram_mb = None;
1955        foundation.cost.size_mb = None;
1956        let on_mac = model_fit(&foundation, &mac(8), Some(&ResourcePolicy::everyday()));
1957        assert_eq!(on_mac.fit, ModelFitStatus::Fits);
1958        assert_eq!(on_mac.estimated_peak_mb, None);
1959        assert!(on_mac.platform_compatible);
1960        assert!(
1961            !model_fit(&foundation, &hw(GpuBackend::Cpu, 64 * 1024, None), None)
1962                .platform_compatible
1963        );
1964
1965        let mut windows = local_model("windows/speech-synthesis:os", "Windows", "", 0);
1966        windows.source = ModelSource::WindowsSpeech {};
1967        windows.cost.ram_mb = None;
1968        windows.cost.size_mb = None;
1969        let on_mac = model_fit(&windows, &mac(8), Some(&ResourcePolicy::everyday()));
1970        assert_eq!(on_mac.fit, ModelFitStatus::Fits);
1971        assert_eq!(on_mac.estimated_peak_mb, None);
1972        assert!(!on_mac.platform_compatible);
1973        let mut windows_host = hw(GpuBackend::Cpu, 8 * 1024, None);
1974        windows_host.os = "windows".into();
1975        assert!(model_fit(&windows, &windows_host, None).platform_compatible);
1976
1977        // The explicit platform tags cover future sources without teaching
1978        // this rule another backend variant.
1979        let mut linux = undeclared.clone();
1980        linux.tags.push("linux-only".into());
1981        let mut linux_host = hw(GpuBackend::Cpu, 8 * 1024, None);
1982        linux_host.os = "linux".into();
1983        assert!(platform_compatible(&linux, &linux_host));
1984        assert!(!platform_compatible(&linux, &windows_host));
1985        let mut tagged_windows = undeclared.clone();
1986        tagged_windows.tags.push("windows-only".into());
1987        assert!(platform_compatible(&tagged_windows, &windows_host));
1988        assert!(!platform_compatible(&tagged_windows, &linux_host));
1989
1990        // Without a policy the legacy tier budget applies — unified memory
1991        // minus the 3 GB OS reserve, 5,120 MB on an 8 GB Mac — exactly as
1992        // `recommend` does: the 4B (≈3,940 MB) fits, the 8B (≈6,344 MB) does
1993        // not, and that is the partition the recommender publishes.
1994        let cat = qwen_mlx_policy_catalog();
1995        assert_eq!(model_fit(&cat[0], &mac(8), None).fit, ModelFitStatus::Fits);
1996        assert_eq!(
1997            model_fit(&cat[1], &mac(8), None).fit,
1998            ModelFitStatus::TooBig
1999        );
2000        let legacy = recommend(
2001            &refs(&cat),
2002            &mac(8),
2003            UseCase::Coding,
2004            QualityTier::Balanced,
2005            Privacy::OnDevice,
2006        );
2007        fn ids(set: &[Recommendation]) -> Vec<&str> {
2008            set.iter().map(|pick| pick.model_id.as_str()).collect()
2009        }
2010        assert_eq!(ids(&legacy.picks), vec!["mlx/qwen3-4b:4bit"]);
2011        assert_eq!(ids(&legacy.not_enough_memory), vec!["mlx/qwen3-8b:4bit"]);
2012    }
2013}
2014
2015#[cfg(test)]
2016mod local_server_fit_tests {
2017    use super::*;
2018    use crate::schema::{ModelCapability, ModelSource};
2019
2020    fn managed_vllm_model(id: &str, size_mb: u64) -> ModelSchema {
2021        let mut m = super::tests::local_model(id, id, "12B", size_mb);
2022        m.capabilities.push(ModelCapability::ToolUse);
2023        m.cost.ram_mb = Some(size_mb + size_mb / 4);
2024        m.source = ModelSource::ManagedVllmMlx {
2025            hf_repo: "mlx-community/whatever-4bit".into(),
2026            hf_weight_file: None,
2027        };
2028        m
2029    }
2030
2031    fn external_vllm_model(id: &str, endpoint: &str, size_mb: u64) -> ModelSchema {
2032        let mut m = super::tests::local_model(id, id, "12B", size_mb);
2033        m.capabilities.push(ModelCapability::ToolUse);
2034        m.cost.ram_mb = Some(size_mb + size_mb / 4);
2035        m.source = ModelSource::VllmMlx {
2036            endpoint: endpoint.to_string(),
2037            model_name: "externally-managed-model".into(),
2038        };
2039        m
2040    }
2041
2042    fn small_mac() -> HardwareInfo {
2043        super::tests::hw(crate::hardware::GpuBackend::Metal, 16384, Some(12288))
2044    }
2045
2046    /// The production break: an explicitly CAR-managed vLLM-MLX model was
2047    /// classified as server-provided, so a 20 GB allocation could be
2048    /// recommended to a 16 GB Mac without consuming the configured budget.
2049    #[test]
2050    fn managed_vllm_mlx_is_memory_checked_and_rejected_when_over_budget() {
2051        let big = managed_vllm_model("vllm-mlx/huge:4bit", 20_000);
2052        let set = recommend_with_policy(
2053            &[&big],
2054            &small_mac(),
2055            &ResourcePolicy::everyday(),
2056            UseCase::Assistant,
2057            QualityTier::Balanced,
2058            Privacy::OnDevice,
2059        );
2060
2061        assert!(set.picks.is_empty());
2062        assert_eq!(set.not_enough_memory.len(), 1);
2063        assert_eq!(set.not_enough_memory[0].fit, FitStatus::TooBig);
2064        assert_eq!(
2065            set.not_enough_memory[0].download_mb, 20_000,
2066            "CAR-managed vllm weights must retain their declared download size"
2067        );
2068    }
2069
2070    /// External vLLM is a cloud-consent/server-provided route regardless of the
2071    /// endpoint spelling or the hardware CAR itself is running on.
2072    #[test]
2073    fn external_vllm_mlx_requires_cloud_consent_and_is_cross_platform() {
2074        let machines = [
2075            small_mac(),
2076            super::tests::hw(crate::hardware::GpuBackend::Cpu, 16_384, None),
2077            super::tests::hw(crate::hardware::GpuBackend::Cuda, 16_384, Some(12_288)),
2078        ];
2079        for endpoint in [
2080            "http://localhost:8000",
2081            "http://127.0.0.1:8000",
2082            "https://gpu-owner.example/v1",
2083        ] {
2084            let external = external_vllm_model("external/vllm", endpoint, 20_000);
2085            for machine in &machines {
2086                let on_device = recommend_with_policy(
2087                    &[&external],
2088                    machine,
2089                    &ResourcePolicy::everyday(),
2090                    UseCase::Assistant,
2091                    QualityTier::Balanced,
2092                    Privacy::OnDevice,
2093                );
2094                assert!(
2095                    on_device.picks.is_empty(),
2096                    "external endpoint {endpoint} must require cloud consent on {:?}",
2097                    machine.gpu_backend
2098                );
2099
2100                let cloud_ok = recommend_with_policy(
2101                    &[&external],
2102                    machine,
2103                    &ResourcePolicy::everyday(),
2104                    UseCase::Assistant,
2105                    QualityTier::Balanced,
2106                    Privacy::CloudOk,
2107                );
2108                assert_eq!(
2109                    cloud_ok.picks.len(),
2110                    1,
2111                    "external endpoint {endpoint} on {:?}",
2112                    machine.gpu_backend
2113                );
2114                assert_eq!(cloud_ok.picks[0].fit, FitStatus::ServerProvided);
2115                assert_eq!(
2116                    cloud_ok.picks[0].download_mb, 0,
2117                    "external vllm owns its weights, so CAR has no download to report"
2118                );
2119                assert!(
2120                    cloud_ok.picks[0].rationale.contains("external server"),
2121                    "external vllm rationale must describe its actual owner: {}",
2122                    cloud_ok.picks[0].rationale
2123                );
2124                assert!(
2125                    !cloud_ok.picks[0].rationale.contains("Parslee's servers"),
2126                    "external vllm must not be attributed to Parslee: {}",
2127                    cloud_ok.picks[0].rationale
2128                );
2129            }
2130        }
2131    }
2132}
2133
2134#[cfg(test)]
2135mod catalog_capability_gap_tests {
2136    use super::*;
2137    use crate::schema::{ModelCapability, ModelSchema};
2138
2139    fn builtin() -> Vec<ModelSchema> {
2140        serde_json::from_str(include_str!("builtin_catalog.json")).unwrap()
2141    }
2142
2143    fn cuda_box(vram_gb: u64, ram_gb: u64) -> crate::hardware::HardwareInfo {
2144        super::tests::hw(
2145            crate::hardware::GpuBackend::Cuda,
2146            ram_gb * 1024,
2147            Some(vram_gb * 1024),
2148        )
2149    }
2150
2151    /// The real catalog on a Mac offers no GGUF row in any lane: each Qwen3
2152    /// used to appear twice, once as a download that would never run.
2153    #[test]
2154    fn a_mac_is_never_offered_a_gguf_row_from_the_builtin_catalog() {
2155        let catalog = builtin();
2156        let refs: Vec<&ModelSchema> = catalog.iter().collect();
2157        for use_case in [
2158            UseCase::Assistant,
2159            UseCase::Coding,
2160            UseCase::Summarize,
2161            UseCase::Vision,
2162            UseCase::Transcription,
2163            UseCase::Search,
2164        ] {
2165            for tier in [
2166                QualityTier::Fastest,
2167                QualityTier::Balanced,
2168                QualityTier::MostCapable,
2169            ] {
2170                let set = recommend(
2171                    &refs,
2172                    &super::tests::mac(64),
2173                    use_case,
2174                    tier,
2175                    Privacy::OnDevice,
2176                );
2177                for pick in set.picks.iter().chain(&set.not_enough_memory) {
2178                    let row = catalog.iter().find(|m| m.id == pick.model_id).unwrap();
2179                    assert!(
2180                        !matches!(row.source, crate::schema::ModelSource::Local { .. }),
2181                        "{use_case:?}/{tier:?} offered GGUF row {}",
2182                        row.id
2183                    );
2184                }
2185            }
2186        }
2187    }
2188
2189    fn most_capable_on(ram_gb: u64) -> RecommendationSet {
2190        let catalog: &'static Vec<ModelSchema> = Box::leak(Box::new(builtin()));
2191        let refs: Vec<&ModelSchema> = catalog.iter().collect();
2192        recommend(
2193            &refs,
2194            &super::tests::mac(ram_gb),
2195            UseCase::Assistant,
2196            QualityTier::MostCapable,
2197            Privacy::OnDevice,
2198        )
2199    }
2200
2201    /// The outcome this whole area exists for: on a machine with room for it,
2202    /// "most capable" returns the highest-scoring model that machine can run —
2203    /// not the highest-scoring small one.
2204    ///
2205    /// It used to answer gemma-4-12B (6.6 GB) on a 64 GB Mac while `--tier
2206    /// fastest` answered a 35B, because every model carrying a `car-judged`
2207    /// score was an older, smaller one and an unscored model fell back to a
2208    /// size prior that saturates below what a real score reaches.
2209    #[test]
2210    fn most_capable_returns_the_best_model_the_machine_can_run() {
2211        let catalog = builtin();
2212        let set = most_capable_on(64);
2213        let top = set.picks.first().expect("a 64 GB machine has picks");
2214
2215        let top_score = catalog
2216            .iter()
2217            .find(|m| m.id == top.model_id)
2218            .and_then(|m| m.public_benchmarks.first())
2219            .map(|b| b.score)
2220            .unwrap_or(0.0);
2221
2222        for m in catalog
2223            .iter()
2224            .filter(|m| m.is_local() && m.size_mb() < 24_000)
2225        {
2226            if let Some(s) = m.public_benchmarks.first().map(|b| b.score) {
2227                assert!(
2228                    s <= top_score,
2229                    "{} scores {s} but {} ({top_score}) was recommended as most capable",
2230                    m.id,
2231                    top.model_id
2232                );
2233            }
2234        }
2235        assert!(
2236            top.download_mb > 10_000,
2237            "a 64 GB machine should be offered a large model, got {} at {} MB",
2238            top.model_id,
2239            top.download_mb
2240        );
2241    }
2242
2243    /// A small machine still gets something that fits — the capability tier
2244    /// must not simply always return the biggest model.
2245    #[test]
2246    fn a_small_machine_is_not_offered_a_model_it_cannot_hold() {
2247        let set = most_capable_on(8);
2248        if let Some(top) = set.picks.first() {
2249            assert!(
2250                top.fit != FitStatus::TooBig,
2251                "{} does not fit an 8 GB machine",
2252                top.model_id
2253            );
2254        }
2255    }
2256
2257    /// The disclosure fires only while larger models remain unscored, and never
2258    /// claims they are downloaded — available means CAR can fetch and run them.
2259    #[test]
2260    fn the_disclosure_is_accurate_when_it_appears() {
2261        let set = most_capable_on(64);
2262        if let Some(note) = set.note.as_deref() {
2263            if note.contains("unscored") {
2264                assert!(note.contains("bench-contribute"), "must say how: {note}");
2265                assert!(
2266                    !note.contains("installed"),
2267                    "availability is not installation: {note}"
2268                );
2269            }
2270        }
2271    }
2272
2273    /// It is scoped to the tier that makes the capability claim.
2274    #[test]
2275    fn other_tiers_do_not_carry_the_disclosure() {
2276        let catalog = builtin();
2277        let refs: Vec<&ModelSchema> = catalog.iter().collect();
2278        for tier in [QualityTier::Fastest, QualityTier::Balanced] {
2279            let set = recommend(
2280                &refs,
2281                &super::tests::mac(64),
2282                UseCase::Assistant,
2283                tier,
2284                Privacy::OnDevice,
2285            );
2286            let carries = set.note.as_deref().is_some_and(|n| n.contains("unscored"));
2287            assert!(!carries, "{tier:?} should not carry the disclosure");
2288        }
2289    }
2290
2291    /// A CUDA machine must be offered a local model at all. GGUF is its local
2292    /// path — `backend::candle` is compiled on exactly the targets MLX is not —
2293    /// so the catalog's `local` rows are its whole local story.
2294    ///
2295    /// Catalog coverage, not a regression test for the availability fix:
2296    /// `recommend` does not read `ModelSchema::available`. That gates the
2297    /// router's fallback chains and is covered in
2298    /// `registry::local_availability_tests`.
2299    #[test]
2300    fn a_cuda_machine_is_offered_a_local_model() {
2301        let catalog = builtin();
2302        let refs: Vec<&ModelSchema> = catalog.iter().collect();
2303        let set = recommend(
2304            &refs,
2305            &cuda_box(24, 64),
2306            UseCase::Assistant,
2307            QualityTier::MostCapable,
2308            Privacy::OnDevice,
2309        );
2310        assert!(
2311            set.picks.iter().any(|p| p.is_local),
2312            "a 24 GB CUDA GPU must be offered something local, got {:?}",
2313            set.picks.iter().map(|p| &p.model_id).collect::<Vec<_>>()
2314        );
2315    }
2316
2317    /// A blank `param_count` is not neutral: the total falls back to
2318    /// `size_mb / 600` and the *active* count falls back to that total, so a
2319    /// 4-bit MoE reads as a dense model of its whole on-disk size.
2320    #[test]
2321    fn local_generate_models_declare_their_parameter_count() {
2322        let blank: Vec<String> = builtin()
2323            .iter()
2324            .filter(|m| {
2325                m.capabilities.contains(&ModelCapability::Generate)
2326                    && m.is_local()
2327                    && m.param_count.trim().is_empty()
2328            })
2329            .map(|m| m.id.clone())
2330            .collect();
2331        assert!(
2332            blank.is_empty(),
2333            "local generate models with no param_count: {blank:?}"
2334        );
2335    }
2336
2337    /// An MoE's `(N active)` hint is what separates its latency from its size.
2338    #[test]
2339    fn an_moe_is_scored_on_its_active_parameters() {
2340        let catalog = builtin();
2341        let glm = catalog
2342            .iter()
2343            .find(|m| m.id == "vllm-mlx/glm-4.7-flash:4bit")
2344            .expect("catalog entry");
2345        let active = crate::resource_policy::model_parameter_billions_active(glm);
2346        let total = crate::resource_policy::model_parameter_billions_total(glm);
2347        assert!(
2348            active < 6.0,
2349            "top-4-of-64 MoE runs at a few B active, got {active}"
2350        );
2351        assert!(total > 20.0, "and carries 30B-class knowledge, got {total}");
2352    }
2353}