Skip to main content

cortiq_engine/
loader.rs

1//! Weight loader: CMF tensor directory → Pipeline.
2//!
3//! Storage rule: models WITH task masks are dequantized to f32 (masked
4//! execution needs f32 row access; skill files are small by design).
5//! Models without masks keep quantized matrices zero-copy from the mmap
6//! (`QTensor::Mapped`) — this is what lets a 15B file run in a few GB
7//! of RSS instead of 60 GB of f32.
8//!
9//! Layer kinds come from `arch.layer_types`: FullAttention loads
10//! `self_attn.*` (with auto-detected Qwen3.5 extras: per-head qk-norm by
11//! tensor presence, output gate by q_proj row count); LinearAttention
12//! loads the canonical core `vmf_attn.*` (folded at convert time).
13
14use crate::kv_cache::LayerKvCache;
15use crate::linear_core::{
16    GdnCfg, GdnWeights, ShortConvCfg, ShortConvWeights, VmfPhaseCfg, VmfPhaseWeights,
17};
18use crate::pipeline::{
19    AttnKind, DenseFfn, FfnKind, LayerWeights, MoeFfn, MtpModule, Pipeline, PipelineWeights,
20};
21use crate::qtensor::QTensor;
22use crate::sampler::SamplerConfig;
23use crate::tokenizer::Tokenizer;
24use cortiq_core::quant::dequant_tensor;
25use cortiq_core::{CmfError, CmfModel, LayerType, ModelArch};
26use std::sync::Arc;
27
28/// Tensor source selector (spec §9): backbone, one skill's overlay, or
29/// a soft superposition of top-m skills (claim 14 working tensors).
30pub enum Overlay<'a> {
31    None,
32    One(&'a str),
33    /// (skill_id, weight); weights sum to 1 (softmax(−E/T) upstream).
34    Blend(&'a [(String, f32)]),
35}
36
37impl Overlay<'_> {
38    fn blend_touches(&self, model: &CmfModel, name: &str) -> bool {
39        match self {
40            Overlay::Blend(list) => list
41                .iter()
42                .any(|(sid, _)| model.tensor(&format!("skill.{sid}.{name}")).is_some()),
43            _ => false,
44        }
45    }
46}
47
48fn dequant_by_name(model: &CmfModel, name: &str) -> Result<Vec<f32>, String> {
49    let entry = model
50        .tensor(name)
51        .ok_or_else(|| format!("tensor '{name}' not found in CMF directory"))?;
52    let mut out = vec![0.0f32; entry.n_elems()];
53    dequant_tensor(entry, model.entry_bytes(entry), &mut out)?;
54    Ok(out)
55}
56
57/// Weighted working tensor (claim 14): Σ wᵢ·Tᵢ, where Tᵢ is the
58/// skill's replacement when present, else the backbone tensor.
59fn blend_f32(model: &CmfModel, name: &str, list: &[(String, f32)]) -> Result<Vec<f32>, String> {
60    let mut acc: Option<Vec<f32>> = None;
61    for (sid, w) in list {
62        let sname = format!("skill.{sid}.{name}");
63        let src = if model.tensor(&sname).is_some() {
64            &sname
65        } else {
66            name
67        };
68        let t = dequant_by_name(model, src)?;
69        match &mut acc {
70            None => {
71                let mut t = t;
72                for v in t.iter_mut() {
73                    *v *= w;
74                }
75                acc = Some(t);
76            }
77            Some(a) => {
78                for (av, tv) in a.iter_mut().zip(&t) {
79                    *av += w * tv;
80                }
81            }
82        }
83    }
84    acc.ok_or_else(|| "empty blend".into())
85}
86
87/// Dequantize a tensor fully into f32 (norms, masked models).
88pub(crate) fn load_f32(model: &CmfModel, name: &str, ov: &Overlay) -> Result<Vec<f32>, String> {
89    if ov.blend_touches(model, name) {
90        if let Overlay::Blend(list) = ov {
91            return blend_f32(model, name, list);
92        }
93    }
94    let skill = match ov {
95        Overlay::One(s) => Some(*s),
96        _ => None,
97    };
98    let entry = model
99        .resolve_tensor(name, skill)
100        .ok_or_else(|| format!("tensor '{name}' not found in CMF directory"))?;
101    let bytes = model.entry_bytes(entry);
102    let mut out = vec![0.0f32; entry.n_elems()];
103    dequant_tensor(entry, bytes, &mut out)?;
104    Ok(out)
105}
106
107/// Build one layer's FFN (dense or MoE) under a given overlay. Shared
108/// by the static loader AND dynamic per-token skill switching
109/// (`Pipeline::set_active_skill`): switching skills = rebuilding the
110/// FFN of the touched layers, cheap because Mapped tensors are just
111/// re-resolved mmap pointers (no dequant, no copy).
112pub(crate) fn build_layer_ffn(
113    model: &Arc<CmfModel>,
114    arch: &ModelArch,
115    li: usize,
116    force_f32: bool,
117    ov: &Overlay,
118) -> Result<FfnKind, CmfError> {
119    build_ffn_at(model, arch, &format!("model.layers.{li}."), force_f32, ov)
120}
121
122/// The FFN under an arbitrary prefix. Split out of `build_layer_ffn` so the
123/// MTP block can reuse it: Qwen3.6's MTP layer carries a full MoE mlp
124/// (router + 256 experts + shared expert), not the dense one the head was
125/// first written against.
126pub(crate) fn build_ffn_at(
127    model: &Arc<CmfModel>,
128    arch: &ModelArch,
129    prefix: &str,
130    force_f32: bool,
131    ov: &Overlay,
132) -> Result<FfnKind, CmfError> {
133    let prefix = prefix.to_string();
134    let load_dense = |p: &str| -> Result<DenseFfn, CmfError> {
135        let gate_proj = load_matrix(model, &format!("{p}gate_proj.weight"), force_f32, ov)?;
136        let up_proj = load_matrix(model, &format!("{p}up_proj.weight"), force_f32, ov)?;
137        let down_proj = load_matrix(model, &format!("{p}down_proj.weight"), force_f32, ov)?;
138        // FFN triple invariant (holds for dense and each MoE expert;
139        // enforced loudly so a malformed defrag/repack — spec §11 — fails
140        // at load instead of silently mis-computing). inter' is per-layer.
141        let inter = gate_proj.rows();
142        if up_proj.rows() != inter || down_proj.cols() != inter {
143            return Err(CmfError::Parse(format!(
144                "{p}: FFN dims disagree (gate.rows={inter}, up.rows={}, \
145                 down.cols={}); all three must equal inter'",
146                up_proj.rows(),
147                down_proj.cols()
148            )));
149        }
150        if down_proj.rows() != arch.hidden_size {
151            return Err(CmfError::Parse(format!(
152                "{p}: down_proj.rows={} != hidden_size={}",
153                down_proj.rows(),
154                arch.hidden_size
155            )));
156        }
157        Ok(DenseFfn {
158            gate_proj,
159            up_proj,
160            down_proj,
161            act: crate::pipeline::Act::from_arch_full(arch),
162        })
163    };
164    let router_name = format!("{prefix}mlp.gate.weight");
165    if model.tensor(&router_name).is_none() {
166        return Ok(FfnKind::Dense(load_dense(&format!("{prefix}mlp."))?));
167    }
168    let cfg = arch.moe.as_ref().ok_or_else(|| {
169        CmfError::Parse(format!(
170            "{router_name} present but header has no arch.moe block"
171        ))
172    })?;
173    // Experts enumerate by TENSOR PRESENCE up to the header count — a
174    // moe-defrag'd specialist keeps a per-layer contiguous prefix of
175    // renumbered experts (fewer than arch.moe.num_experts), with the
176    // router rows sliced to match.
177    let mut experts = Vec::new();
178    for e in 0..cfg.num_experts {
179        if model
180            .tensor(&format!("{prefix}mlp.experts.{e}.gate_proj.weight"))
181            .is_none()
182        {
183            break;
184        }
185        experts.push(load_dense(&format!("{prefix}mlp.experts.{e}."))?);
186    }
187    if experts.is_empty() {
188        return Err(CmfError::Parse(format!(
189            "{prefix}: router present but no expert tensors"
190        )));
191    }
192    let shared = if model
193        .tensor(&format!("{prefix}mlp.shared_expert.gate_proj.weight"))
194        .is_some()
195    {
196        let gate_name = format!("{prefix}mlp.shared_expert_gate.weight");
197        Some((
198            load_dense(&format!("{prefix}mlp.shared_expert."))?,
199            if model.tensor(&gate_name).is_some() {
200                Some(load_matrix(model, &gate_name, force_f32, ov)?)
201            } else {
202                None
203            },
204        ))
205    } else {
206        None
207    };
208    // LFM2-MoE selection bias (`mlp.expert_bias`): present iff the model
209    // routes with a bias; loaded by tensor presence.
210    let bias_name = format!("{prefix}mlp.expert_bias");
211    let expert_bias = if model.tensor(&bias_name).is_some() {
212        Some(load_f32(model, &bias_name, ov).map_err(CmfError::Parse)?)
213    } else {
214        None
215    };
216    // CMF_MOE_TOPK=N (opt-in): route to fewer experts than the header
217    // asks. MoE decode is memory-bound — every selected expert streams
218    // its three matrices per token — so halving k halves that traffic;
219    // the renormalized top-k keeps the mixture a proper average.
220    // Quality is the experiment — measure ppl before trusting.
221    let top_k = std::env::var("CMF_MOE_TOPK")
222        .ok()
223        .and_then(|v| v.parse::<usize>().ok())
224        .filter(|&k| k >= 1 && k <= cfg.top_k)
225        .inspect(|k| tracing::info!("MoE top_k override: {} (header {})", k, cfg.top_k))
226        .unwrap_or(cfg.top_k);
227    // CMF_MOE_TAU=0.x (opt-in): adaptive routing — see MoeFfn::route_tau.
228    let route_tau = std::env::var("CMF_MOE_TAU")
229        .ok()
230        .and_then(|v| v.parse::<f32>().ok())
231        .filter(|&t| t > 0.0 && t < 1.0)
232        .inspect(|t| tracing::info!("MoE adaptive routing: tau {t}"));
233    let mask = moe_task_mask(&prefix, experts.len());
234    let router = load_matrix(model, &router_name, force_f32, ov)?;
235    if router.rows() != experts.len() {
236        return Err(CmfError::Parse(format!(
237            "{router_name}: {} rows != {} experts",
238            router.rows(),
239            experts.len()
240        )));
241    }
242    let top_k = top_k.min(experts.len());
243    // Gemma-4: per-expert weight scale after the top-k renorm; its
244    // presence also marks the scale-less-rms router input (the folded
245    // router gain — see the converter).
246    let pes_name = format!("{prefix}mlp.per_expert_scale");
247    let per_expert_scale = if model.tensor(&pes_name).is_some() {
248        Some(load_f32(model, &pes_name, ov).map_err(CmfError::Parse)?)
249    } else {
250        None
251    };
252    let router_input_norm = per_expert_scale.is_some();
253    let moe = MoeFfn {
254        router,
255        experts,
256        top_k,
257        route_tau,
258        norm_topk_prob: cfg.norm_topk_prob,
259        router_sigmoid: cfg.router_sigmoid,
260        expert_bias,
261        routed_scaling: cfg.routed_scaling_factor.unwrap_or(1.0),
262        shared,
263        stats: std::cell::RefCell::new(Vec::new()),
264        act_sq: std::cell::RefCell::new(Vec::new()),
265        act_rows: std::cell::RefCell::new(Vec::new()),
266        mask,
267        per_expert_scale,
268        router_input_norm,
269    };
270    // Gemma-4 dual-branch layer: a dense MLP coexists with the routed
271    // experts, each branch inside its own norm sandwich.
272    if model
273        .tensor(&format!("{prefix}mlp.gate_proj.weight"))
274        .is_some()
275    {
276        let norm = |suffix: &str| -> Result<Vec<f32>, CmfError> {
277            load_f32(model, &format!("{prefix}{suffix}.weight"), ov).map_err(CmfError::Parse)
278        };
279        return Ok(FfnKind::DenseMoe(Box::new(crate::pipeline::DenseMoeFfn {
280            dense: load_dense(&format!("{prefix}mlp."))?,
281            moe,
282            post_norm_1: norm("post_feedforward_layernorm_1")?,
283            pre_norm_2: norm("pre_feedforward_layernorm_2")?,
284            post_norm_2: norm("post_feedforward_layernorm_2")?,
285        })));
286    }
287    Ok(FfnKind::Moe(moe))
288}
289
290/// Task mask over routed experts (opt-in, experimental): DTG-MA applied
291/// to MoE. `CMF_MOE_MASK=<stats.json>` points at a claim-12 B-field dump
292/// (`CMF_MOE_STATS` output — per-layer expert-selection counts from a
293/// task-representative run); `CMF_MOE_MASK_COVER` (default 0.9) keeps,
294/// per layer, the smallest top set of experts reaching that fraction of
295/// the recorded routing mass. Selection then happens over the allowed
296/// set only (softmax renormalizes). Gate any real use on a ppl A/B.
297pub(crate) fn moe_task_mask(prefix: &str, ne: usize) -> Option<Vec<bool>> {
298    use std::sync::OnceLock;
299    static CFG: OnceLock<Option<(std::collections::HashMap<usize, Vec<u64>>, f64)>> =
300        OnceLock::new();
301    let cfg = CFG.get_or_init(|| {
302        let path = std::env::var("CMF_MOE_MASK").ok()?;
303        let cover = std::env::var("CMF_MOE_MASK_COVER")
304            .ok()
305            .and_then(|v| v.parse::<f64>().ok())
306            .filter(|&c| c > 0.0 && c <= 1.0)
307            .unwrap_or(0.9);
308        let text = std::fs::read_to_string(&path)
309            .map_err(|e| tracing::warn!("CMF_MOE_MASK: cannot read {path}: {e}"))
310            .ok()?;
311        let map: std::collections::HashMap<String, Vec<u64>> = serde_json::from_str(&text)
312            .map_err(|e| tracing::warn!("CMF_MOE_MASK: bad JSON in {path}: {e}"))
313            .ok()?;
314        tracing::info!("MoE task mask: {path}, cover {cover}");
315        Some((
316            map.into_iter()
317                .filter_map(|(k, v)| Some((k.parse::<usize>().ok()?, v)))
318                .collect(),
319            cover,
320        ))
321    });
322    let (stats, cover) = cfg.as_ref()?;
323    // The layer index rides in the tensor prefix ("model.layers.N.").
324    let li: usize = prefix
325        .split("layers.")
326        .nth(1)?
327        .split('.')
328        .next()?
329        .parse()
330        .ok()?;
331    let counts = stats.get(&li)?;
332    if counts.len() != ne {
333        tracing::warn!(
334            "CMF_MOE_MASK: layer {li} has {} counts, model has {ne} experts — skipped",
335            counts.len()
336        );
337        return None;
338    }
339    let total: u64 = counts.iter().sum();
340    if total == 0 {
341        return None;
342    }
343    let mut order: Vec<usize> = (0..ne).collect();
344    order.sort_unstable_by_key(|&e| std::cmp::Reverse(counts[e]));
345    let mut mask = vec![false; ne];
346    let mut acc = 0u64;
347    let mut kept = 0usize;
348    for &e in &order {
349        mask[e] = true;
350        acc += counts[e];
351        kept += 1;
352        if (acc as f64) >= cover * (total as f64) {
353            break;
354        }
355    }
356    tracing::info!(
357        "MoE task mask L{li}: {kept}/{ne} experts for {:.0}% mass",
358        cover * 100.0
359    );
360    Some(mask)
361}
362
363fn load_matrix(
364    model: &Arc<CmfModel>,
365    name: &str,
366    force_f32: bool,
367    ov: &Overlay,
368) -> Result<QTensor, CmfError> {
369    // Claim 14: a blended working tensor is materialized in f32 and
370    // held resident (the overlay-cache slot); single skills stay
371    // zero-copy pointers into the mmap.
372    if ov.blend_touches(model, name) {
373        if let Overlay::Blend(list) = ov {
374            let entry = model
375                .tensor(name)
376                .ok_or_else(|| CmfError::MissingTensor(name.to_string()))?;
377            let data =
378                blend_f32(model, name, list).map_err(|e| CmfError::Parse(format!("blend: {e}")))?;
379            return Ok(QTensor::from_f32(data, entry.shape[0], entry.shape[1]));
380        }
381    }
382    let skill = match ov {
383        Overlay::One(s) => Some(*s),
384        _ => None,
385    };
386    // Tensor-source indirection (spec §9): the skill's replacement is
387    // read in place of the backbone tensor — either/or, never a sum.
388    let name: &str = &match skill {
389        Some(sid) if model.tensor(&format!("skill.{sid}.{name}")).is_some() => {
390            format!("skill.{sid}.{name}")
391        }
392        _ => name.to_string(),
393    };
394    let err = |e: String| CmfError::Parse(format!("weight loading: {e}"));
395    if force_f32 {
396        let entry = model
397            .tensor(name)
398            .ok_or_else(|| CmfError::MissingTensor(name.to_string()))?;
399        if entry.shape.len() != 2 {
400            return Err(err(format!("'{name}' is not 2-D")));
401        }
402        let data = load_f32(model, name, &Overlay::None).map_err(err)?;
403        Ok(QTensor::from_f32(data, entry.shape[0], entry.shape[1]))
404    } else {
405        QTensor::from_model(model, name).map_err(err)
406    }
407}
408
409impl Pipeline {
410    /// Build a runnable pipeline from an opened CMF model.
411    pub fn from_model(
412        model: &Arc<CmfModel>,
413        sampler_config: SamplerConfig,
414    ) -> Result<Self, CmfError> {
415        Self::from_model_with_skill(model, sampler_config, None)
416    }
417
418    /// Same, with a skill overlaid (spec §9): every layer tensor is
419    /// resolved through tensor-source indirection — the skill's
420    /// full-shape replacement is read in place of the backbone tensor.
421    /// No per-skill model is ever assembled: Mapped tensors are
422    /// pointers into the one shared mmap.
423    pub fn from_model_with_skill(
424        model: &Arc<CmfModel>,
425        sampler_config: SamplerConfig,
426        skill: Option<&str>,
427    ) -> Result<Self, CmfError> {
428        match skill {
429            Some(s) => Self::from_model_with_overlay(model, sampler_config, &Overlay::One(s)),
430            None => Self::from_model_with_overlay(model, sampler_config, &Overlay::None),
431        }
432    }
433
434    /// Soft superposition (claim 14): working tensors accumulated from
435    /// the given (skill, weight) list — softmax(−E/T) upstream.
436    pub fn from_model_with_blend(
437        model: &Arc<CmfModel>,
438        sampler_config: SamplerConfig,
439        blend: &[(String, f32)],
440    ) -> Result<Self, CmfError> {
441        Self::from_model_with_overlay(model, sampler_config, &Overlay::Blend(blend))
442    }
443
444    fn from_model_with_overlay(
445        model: &Arc<CmfModel>,
446        sampler_config: SamplerConfig,
447        ov: &Overlay,
448    ) -> Result<Self, CmfError> {
449        let skill = match ov {
450            Overlay::One(s) => Some(*s),
451            _ => None,
452        };
453        if let Some(sid) = skill {
454            let known = model.header.skills.iter().any(|s| s.id == sid)
455                || model.skill_tensors(sid).next().is_some();
456            if !known {
457                return Err(CmfError::Parse(format!(
458                    "skill '{sid}' not in this container (header.skills: {:?})",
459                    model
460                        .header
461                        .skills
462                        .iter()
463                        .map(|s| &s.id)
464                        .collect::<Vec<_>>()
465                )));
466            }
467            tracing::info!(
468                "skill '{sid}': {} replacement tensors overlaid",
469                model.skill_tensors(sid).count()
470            );
471        }
472        let arch = model.arch().clone();
473        let err = |e: String| CmfError::Parse(format!("weight loading: {e}"));
474        if let Some(heads) = &arch.attention_heads_per_layer {
475            if heads.len() != arch.num_layers {
476                return Err(CmfError::Parse(format!(
477                    "arch.attention_heads_per_layer has {} entries, expected {}",
478                    heads.len(),
479                    arch.num_layers
480                )));
481            }
482            if let Some((li, &nh)) = heads
483                .iter()
484                .enumerate()
485                .find(|(_, nh)| **nh == 0 || **nh % arch.num_kv_heads != 0)
486            {
487                return Err(CmfError::Parse(format!(
488                    "layer {li} has {nh} Q heads, which must be nonzero and divisible by {} KV heads",
489                    arch.num_kv_heads
490                )));
491            }
492        }
493        if arch
494            .layer_types
495            .iter()
496            .any(|t| matches!(t, LayerType::SlidingAttention))
497            && arch.sliding_window.is_none()
498        {
499            return Err(CmfError::Parse(
500                "model has SlidingAttention layers but no arch.sliding_window".into(),
501            ));
502        }
503
504        // Masks × quantized mmap: only ATTENTION keeps f32 (the head-mask
505        // path needs f32 slices). FFN masks now run sparse directly on the
506        // quant bytes (sparse_ffn_quant), and embed/lm_head are never
507        // masked — so a masked model runs at quantized RSS, not the old
508        // whole-model-f32 blowup.
509        let masks_present = !model.masks.masks.is_empty();
510        let force_f32 = masks_present; // attention only (head masks)
511
512        // ── Tokenizer: embedded → sidecar → byte-level fallback ──
513        let mut tokenizer = if let Some(vocab_bytes) = &model.vocab {
514            Tokenizer::from_bytes(vocab_bytes)
515                .map_err(|e| CmfError::Parse(format!("embedded tokenizer: {e}")))?
516        } else {
517            let sidecar = model.path.with_file_name("tokenizer.json");
518            if sidecar.exists() {
519                Tokenizer::from_file(&sidecar)
520                    .map_err(|e| CmfError::Parse(format!("sidecar tokenizer: {e}")))?
521            } else {
522                tracing::warn!("no tokenizer in file or sidecar — using byte-level fallback");
523                Tokenizer::byte_level()
524            }
525        };
526        // Chat/eos bundle (spec §6.1): the FILE defines chat behavior.
527        if let Some(tc) = &model.header.tokenizer_config {
528            tokenizer.chat_template = tc.chat_template.clone();
529            tokenizer.extra_eos.extend(tc.eos_token_ids.iter().copied());
530            if tokenizer.bos_token_id.is_none() {
531                tokenizer.bos_token_id = tc.bos_token_id;
532            }
533            tracing::info!(
534                "chat bundle: template {} chars, {} stop ids",
535                tc.chat_template.as_deref().map(str::len).unwrap_or(0),
536                tc.eos_token_ids.len()
537            );
538        }
539        // Gemma's contract requires <bos> at sequence start, but its
540        // tokenizer.json post-processor does not add it (the chat
541        // template does). Raw prompts need it too — word salad without.
542        if arch.arch_name.to_lowercase().contains("gemma") && tokenizer.bos_token_id.is_some() {
543            tokenizer.add_bos = true;
544        }
545
546        // ── Top-level weights (never masked → always quantized) ──
547        let embed_tokens = load_matrix(model, "model.embed_tokens.weight", false, ov)?;
548        let final_norm = load_f32(model, "model.norm.weight", ov).map_err(err)?;
549        let lm_head = if model.tensor("lm_head.weight").is_some() {
550            load_matrix(model, "lm_head.weight", false, ov)?
551        } else if arch.tie_word_embeddings {
552            // Tied: reuse the embedding matrix (re-open, cheap for Mapped).
553            load_matrix(model, "model.embed_tokens.weight", false, ov)?
554        } else {
555            return Err(CmfError::MissingTensor(
556                "lm_head.weight (and tie_word_embeddings is false)".into(),
557            ));
558        };
559
560        // ── Linear-core geometry (required if any linear layer exists) ──
561        let has_linear = arch
562            .layer_types
563            .iter()
564            .any(|t| matches!(t, LayerType::LinearAttention));
565        let mut vmf_cfg = None;
566        let mut gdn_cfg = None;
567        if has_linear {
568            let lc = arch.linear_core.as_ref().ok_or_else(|| {
569                CmfError::Parse(
570                    "model has LinearAttention layers but no arch.linear_core — \
571                     reconvert with the current converter"
572                        .into(),
573                )
574            })?;
575            let need = |v: Option<usize>, name: &str| {
576                v.ok_or_else(|| CmfError::Parse(format!("linear core needs arch.{name}")))
577            };
578            match lc.kind.as_str() {
579                "vmf_phase" => {
580                    vmf_cfg = Some(VmfPhaseCfg {
581                        num_heads: lc.num_heads,
582                        nphase: need(lc.nphase, "linear_core.nphase")?,
583                        value_head_dim: lc.value_head_dim,
584                        hidden_size: arch.hidden_size,
585                        // θ-mass (η′): default 0 (massless); CMF_PHASE_MASS
586                        // widens the phase kernel for folded-unhealed models.
587                        phase_mass: std::env::var("CMF_PHASE_MASS")
588                            .ok()
589                            .and_then(|v| v.parse().ok())
590                            .unwrap_or(0.0),
591                    });
592                }
593                "gated_delta_net" => {
594                    gdn_cfg = Some(GdnCfg {
595                        num_v_heads: lc.num_heads,
596                        num_k_heads: need(arch.linear_num_key_heads, "linear_num_key_heads")?,
597                        key_head_dim: need(arch.linear_key_head_dim, "linear_key_head_dim")?,
598                        value_head_dim: lc.value_head_dim,
599                        conv_kernel: need(arch.linear_conv_kernel_dim, "linear_conv_kernel_dim")?,
600                        hidden_size: arch.hidden_size,
601                        rms_eps: arch.rms_norm_eps,
602                    });
603                }
604                other => {
605                    return Err(CmfError::Parse(format!(
606                        "unknown linear core '{other}' (this runtime executes: \
607                         gated_delta_net, vmf_phase)"
608                    )));
609                }
610            }
611        }
612
613        // ── KDA geometry (Kimi Linear / Kimi-K3 delta-attention layers) ──
614        let has_kda = arch.layer_types.iter().any(|t| matches!(t, LayerType::Kda));
615        let kda_cfg = if has_kda {
616            let need = |v: Option<usize>, name: &str| {
617                v.ok_or_else(|| CmfError::Parse(format!("KDA core needs arch.{name}")))
618            };
619            Some(crate::linear_core::KdaCfg {
620                num_heads: need(arch.linear_num_key_heads, "linear_num_key_heads")?,
621                head_k_dim: need(arch.linear_key_head_dim, "linear_key_head_dim")?,
622                head_v_dim: need(arch.linear_value_head_dim, "linear_value_head_dim")?,
623                conv_kernel: need(arch.linear_conv_kernel_dim, "linear_conv_kernel_dim")?,
624                hidden_size: arch.hidden_size,
625                rms_eps: arch.rms_norm_eps,
626            })
627        } else {
628            None
629        };
630
631        // ── Short-convolution geometry (LFM2 conv mixer layers) ──
632        let has_short_conv = arch
633            .layer_types
634            .iter()
635            .any(|t| matches!(t, LayerType::ShortConv));
636        let short_conv_cfg = if has_short_conv {
637            Some(ShortConvCfg {
638                hidden_size: arch.hidden_size,
639                kernel: arch.linear_conv_kernel_dim.ok_or_else(|| {
640                    CmfError::Parse(
641                        "model has ShortConv layers but no arch.linear_conv_kernel_dim — \
642                         reconvert with the current converter"
643                            .into(),
644                    )
645                })?,
646            })
647        } else {
648            None
649        };
650
651        // ── Layers ──
652        let load_full_attn = |prefix: &str, layer: Option<usize>| -> Result<AttnKind, CmfError> {
653            let t = |suffix: &str| load_matrix(model, &format!("{prefix}{suffix}"), force_f32, ov);
654            let n = |suffix: &str| -> Option<Vec<f32>> {
655                model
656                    .tensor(&format!("{prefix}{suffix}"))
657                    .and_then(|_| load_f32(model, &format!("{prefix}{suffix}"), ov).ok())
658            };
659            // DeepSeek-V2 MLA: the latent projections replace the k/v pair.
660            if let Some(mla) = arch.mla.as_ref() {
661                // Compressed q (K3/V3): q_a → rms → q_b; direct otherwise.
662                let (q_proj, q_a, q_a_norm) = if mla.q_lora_rank.is_some() {
663                    (
664                        t("self_attn.q_b_proj.weight")?,
665                        Some(t("self_attn.q_a_proj.weight")?),
666                        Some(n("self_attn.q_a_layernorm.weight").ok_or_else(|| {
667                            CmfError::Parse(format!("{prefix}: MLA needs q_a_layernorm"))
668                        })?),
669                    )
670                } else {
671                    (t("self_attn.q_proj.weight")?, None, None)
672                };
673                let hd = mla.qk_rope_head_dim + mla.qk_nope_head_dim;
674                let nh = q_proj.rows() / hd;
675                // YaRN mscale²: DeepSeek corrects the softmax scale by
676                // (0.1·mscale_all_dim·ln(factor)+1)².
677                let mut scale = 1.0 / (hd as f32).sqrt();
678                if let Some(y) = arch.yarn.as_ref() {
679                    if let Some(m) = y.mscale_all_dim.filter(|&m| m > 0.0) {
680                        let ms = 0.1 * m * y.factor.ln() + 1.0;
681                        scale *= ms * ms;
682                    }
683                }
684                return Ok(AttnKind::Mla(Box::new(crate::pipeline::MlaWeights {
685                    q_proj,
686                    q_a,
687                    q_a_norm,
688                    kv_a: t("self_attn.kv_a_proj_with_mqa.weight")?,
689                    kv_a_norm: n("self_attn.kv_a_layernorm.weight").ok_or_else(|| {
690                        CmfError::Parse(format!("{prefix}: MLA needs kv_a_layernorm"))
691                    })?,
692                    kv_b: t("self_attn.kv_b_proj.weight")?,
693                    o_proj: t("self_attn.o_proj.weight")?,
694                    nh,
695                    qk_rope: mla.qk_rope_head_dim,
696                    qk_nope: mla.qk_nope_head_dim,
697                    v_dim: mla.v_head_dim,
698                    lora: mla.kv_lora_rank,
699                    scale,
700                    nope: mla.nope,
701                })));
702            }
703            let wq = t("self_attn.q_proj.weight")?;
704            let nh = layer
705                .and_then(|li| {
706                    arch.attention_heads_per_layer
707                        .as_ref()
708                        .and_then(|v| v.get(li).copied())
709                })
710                .unwrap_or(arch.num_attention_heads);
711            // Qwen3.5 output gate: q_proj rows = 2·nh·hd (per-head [q; gate]).
712            // Gemma-4 global layers legitimately have nh·global_head_dim
713            // rows (which can equal 2·nh·hd) — never gated.
714            let output_gate = arch.global_head_dim.is_none() && wq.rows() == 2 * nh * arch.head_dim;
715            // Gemma-4 global layers run MQA at global_head_dim — their
716            // q_proj legitimately carries nh·ghd rows.
717            let is_global_layer = arch.global_head_dim.is_some()
718                && layer.is_some_and(|li| {
719                    arch.sliding_window_pattern
720                        .is_some_and(|p| p > 0 && (li + 1) % p == 0)
721                });
722            let expect = if is_global_layer {
723                nh * arch.global_head_dim.unwrap_or(arch.head_dim)
724            } else {
725                nh * arch.head_dim
726            };
727            if !output_gate && wq.rows() != expect {
728                return Err(CmfError::Parse(format!(
729                    "{prefix}self_attn.q_proj.weight rows={} != heads({nh}) * head_dim({})",
730                    wq.rows(),
731                    expect / nh.max(1)
732                )));
733            }
734            let gate_name = format!("{prefix}self_attn.g_proj.weight");
735            let softplus_gate = if model.tensor(&gate_name).is_some() {
736                let gate = load_matrix(model, &gate_name, force_f32, ov)?;
737                if gate.cols() != arch.hidden_size {
738                    return Err(CmfError::Parse(format!(
739                        "{gate_name} cols={} != hidden_size ({})",
740                        gate.cols(),
741                        arch.hidden_size
742                    )));
743                }
744                let per_head = if gate.rows() == nh {
745                    true
746                } else if gate.rows() == nh * arch.head_dim {
747                    false
748                } else {
749                    return Err(CmfError::Parse(format!(
750                        "{gate_name} rows={} must equal heads ({nh}) or heads*head_dim ({})",
751                        gate.rows(),
752                        nh * arch.head_dim
753                    )));
754                };
755                Some((gate, per_head))
756            } else {
757                None
758            };
759            // Qwen2-family projection biases (by tensor presence).
760            let bias = match (
761                n("self_attn.q_proj.bias"),
762                n("self_attn.k_proj.bias"),
763                n("self_attn.v_proj.bias"),
764            ) {
765                (Some(a), Some(b), Some(c)) => Some((a, b, c)),
766                _ => None,
767            };
768            Ok(AttnKind::Full {
769                wq,
770                wk: t("self_attn.k_proj.weight")?,
771                wv: t("self_attn.v_proj.weight")?,
772                wo: t("self_attn.o_proj.weight")?,
773                q_norm: n("self_attn.q_norm.weight"),
774                k_norm: n("self_attn.k_norm.weight"),
775                output_gate,
776                softplus_gate,
777                bias,
778            })
779        };
780
781        let load_linear_attn = |prefix: &str| -> Result<AttnKind, CmfError> {
782            if gdn_cfg.is_some() {
783                // Faithful vendor operator: tensor names 1:1 with the source.
784                let t = |suffix: &str| {
785                    load_matrix(
786                        model,
787                        &format!("{prefix}linear_attn.{suffix}"),
788                        force_f32,
789                        ov,
790                    )
791                };
792                let f = |suffix: &str| {
793                    load_f32(model, &format!("{prefix}linear_attn.{suffix}"), ov).map_err(err)
794                };
795                return Ok(AttnKind::LinearGdn(GdnWeights {
796                    in_proj_qkv: t("in_proj_qkv.weight")?,
797                    in_proj_z: t("in_proj_z.weight")?,
798                    in_proj_a: t("in_proj_a.weight")?,
799                    in_proj_b: t("in_proj_b.weight")?,
800                    conv1d: f("conv1d.weight")?,
801                    a_log: f("A_log")?,
802                    dt_bias: f("dt_bias")?,
803                    norm: f("norm.weight")?,
804                    out_proj: t("out_proj.weight")?,
805                }));
806            }
807            let t = |suffix: &str| {
808                load_matrix(model, &format!("{prefix}vmf_attn.{suffix}"), force_f32, ov)
809            };
810            let a_log = load_f32(model, &format!("{prefix}vmf_attn.A_log"), ov).map_err(err)?;
811            // Selective-write gate κ (hybrid_k core): optional by tensor
812            // presence — files without it run the classic phase kernel
813            // bit-identically.
814            let k_gate = if model
815                .tensor(&format!("{prefix}vmf_attn.k_gate.weight"))
816                .is_some()
817            {
818                Some((
819                    t("k_gate.weight")?,
820                    load_f32(model, &format!("{prefix}vmf_attn.k_gate.bias"), ov).map_err(err)?,
821                ))
822            } else {
823                None
824            };
825            Ok(AttnKind::Linear(VmfPhaseWeights {
826                thq: t("thq.weight")?,
827                thk: t("thk.weight")?,
828                v_proj: t("v_proj.weight")?,
829                out_proj: t("out_proj.weight")?,
830                decay: a_log.iter().map(|&a| (-(a as f64).exp()).exp()).collect(),
831                k_gate,
832            }))
833        };
834
835        // LFM2 short-conv mixer: in_proj [3·hidden, hidden], a depthwise
836        // conv (stored f16 as `[hidden, 1, kernel]` → flattened taps), and
837        // out_proj [hidden, hidden]. Names canonicalized at convert time.
838        let load_short_conv = |prefix: &str| -> Result<AttnKind, CmfError> {
839            let t = |suffix: &str| {
840                load_matrix(
841                    model,
842                    &format!("{prefix}short_conv.{suffix}"),
843                    force_f32,
844                    ov,
845                )
846            };
847            Ok(AttnKind::ShortConv(ShortConvWeights {
848                in_proj: t("in_proj.weight")?,
849                conv: load_f32(model, &format!("{prefix}short_conv.conv.weight"), ov)
850                    .map_err(err)?,
851                out_proj: t("out_proj.weight")?,
852            }))
853        };
854
855        // KDA layer (Kimi Linear / Kimi-K3): faithful vendor tensors under
856        // the `kda_attn.` canonical prefix. The output gate is full-rank
857        // (g_proj, K3) or low-rank (g_a/g_b, Kimi-Linear-48B) by presence.
858        let load_kda = |prefix: &str| -> Result<AttnKind, CmfError> {
859            let t = |suffix: &str| {
860                load_matrix(model, &format!("{prefix}kda_attn.{suffix}"), force_f32, ov)
861            };
862            let f = |suffix: &str| {
863                load_f32(model, &format!("{prefix}kda_attn.{suffix}"), ov).map_err(err)
864            };
865            let gate = if model
866                .tensor(&format!("{prefix}kda_attn.g_proj.weight"))
867                .is_some()
868            {
869                crate::linear_core::KdaOutGate::Full(t("g_proj.weight")?)
870            } else {
871                crate::linear_core::KdaOutGate::LowRank(
872                    t("g_a_proj.weight")?,
873                    t("g_b_proj.weight")?,
874                )
875            };
876            Ok(AttnKind::Kda(Box::new(crate::linear_core::KdaWeights {
877                q_proj: t("q_proj.weight")?,
878                k_proj: t("k_proj.weight")?,
879                v_proj: t("v_proj.weight")?,
880                conv_q: f("q_conv1d.weight")?,
881                conv_k: f("k_conv1d.weight")?,
882                conv_v: f("v_conv1d.weight")?,
883                f_a: t("f_a_proj.weight")?,
884                f_b: t("f_b_proj.weight")?,
885                dt_bias: f("dt_bias")?,
886                a_log: f("A_log")?,
887                b_proj: t("b_proj.weight")?,
888                gate,
889                o_norm: f("o_norm.weight")?,
890                o_proj: t("o_proj.weight")?,
891                gate_lower_bound: arch.kda_gate_lower_bound.map(|v| v as f32),
892            })))
893        };
894
895        fn anyhow_like(ok: bool) -> Result<(), ()> {
896            if ok { Ok(()) } else { Err(()) }
897        }
898        let mut layers = Vec::with_capacity(arch.num_layers);
899        let is_g3n = arch.g3n.is_some();
900        // Architectures that load their own layer stack below. DeepSeek-V4
901        // has none of the canonical projections — no q/k/v/o_proj, no
902        // per-layer gate_proj — so the generic loop would demand
903        // `self_attn.q_proj.weight` and fail before its own loader ever ran.
904        let owns_its_layers = is_g3n || arch.arch_name == "deepseek_v4";
905        for li in 0..(if owns_its_layers { 0 } else { arch.num_layers }) {
906            let prefix = format!("model.layers.{li}.");
907            let attn = match arch.layer_types.get(li) {
908                Some(LayerType::LinearAttention) => load_linear_attn(&prefix)?,
909                Some(LayerType::Kda) => load_kda(&prefix)?,
910                Some(LayerType::ShortConv) => load_short_conv(&prefix)?,
911                _ => load_full_attn(&prefix, Some(li))?,
912            };
913            // Gemma-2/3 sandwich: `pre_feedforward_layernorm` present →
914            // it is the pre-FFN norm, and post_attention/post_feedforward
915            // норms apply to the branch OUTPUTS before their residuals.
916            let pre_ffn = format!("{prefix}pre_feedforward_layernorm.weight");
917            let sandwich = model.tensor(&pre_ffn).is_some();
918            layers.push(LayerWeights {
919                input_norm: load_f32(model, &format!("{prefix}input_layernorm.weight"), ov)
920                    .map_err(err)?,
921                post_norm: if sandwich {
922                    load_f32(model, &pre_ffn, ov).map_err(err)?
923                } else {
924                    load_f32(
925                        model,
926                        &format!("{prefix}post_attention_layernorm.weight"),
927                        ov,
928                    )
929                    .map_err(err)?
930                },
931                attn_out_norm: if sandwich {
932                    Some(
933                        load_f32(
934                            model,
935                            &format!("{prefix}post_attention_layernorm.weight"),
936                            ov,
937                        )
938                        .map_err(err)?,
939                    )
940                } else {
941                    None
942                },
943                ffn_out_norm: if sandwich {
944                    Some(
945                        load_f32(
946                            model,
947                            &format!("{prefix}post_feedforward_layernorm.weight"),
948                            ov,
949                        )
950                        .map_err(err)?,
951                    )
952                } else {
953                    None
954                },
955                // Gemma-4: learned scalar multiplying the layer output.
956                layer_scale: model
957                    .tensor(&format!("{prefix}layer_scalar"))
958                    .and_then(|_| {
959                        load_f32(model, &format!("{prefix}layer_scalar"), ov)
960                            .ok()
961                            .and_then(|v| v.first().copied())
962                    }),
963                // FFN always quantized — masks run sparse on quant bytes.
964                ffn: build_layer_ffn(model, &arch, li, false, ov)?,
965                attn,
966            });
967        }
968
969        // ── MTP head (optional, spec §2.1) ──
970        //
971        // The header declaring an MTP head is not the same as the file
972        // carrying one. DeepSeek-V4's config announces a next-token predictor
973        // whose weights the converter does not map (they are spelled `mtp.N.*`
974        // and have none of the canonical projections), so demanding
975        // `model.mtp.layers.0.self_attn.q_proj.weight` failed a model that is
976        // otherwise complete. Presence in the directory decides.
977        let mtp_present = model
978            .tensor("model.mtp.layers.0.self_attn.q_proj.weight")
979            .is_some()
980            || model.tensor("model.mtp.eh_proj.weight").is_some();
981        if arch.mtp.is_some() && !mtp_present {
982            tracing::info!(
983                "header declares an MTP head but the file carries none — \
984                 loading without it"
985            );
986        }
987        let mtp = if let Some(cfg) = arch.mtp.as_ref().filter(|_| mtp_present) {
988            if cfg.num_layers != 1 {
989                return Err(CmfError::Parse(format!(
990                    "MTP with {} blocks not supported yet (only 1)",
991                    cfg.num_layers
992                )));
993            }
994            let p = "model.mtp.";
995            let attn = load_full_attn("model.mtp.layers.0.", None)?;
996            Some(MtpModule {
997                enorm: load_f32(model, &format!("{p}enorm.weight"), ov).map_err(err)?,
998                hnorm: load_f32(model, &format!("{p}hnorm.weight"), ov).map_err(err)?,
999                eh_proj: load_matrix(model, &format!("{p}eh_proj.weight"), false, ov)?,
1000                layer: LayerWeights {
1001                    attn_out_norm: None,
1002                    ffn_out_norm: None,
1003                    layer_scale: None,
1004                    input_norm: load_f32(model, &format!("{p}layers.0.input_layernorm.weight"), ov)
1005                        .map_err(err)?,
1006                    post_norm: load_f32(
1007                        model,
1008                        &format!("{p}layers.0.post_attention_layernorm.weight"),
1009                        ov,
1010                    )
1011                    .map_err(err)?,
1012                    // Whatever the block actually carries: DeepSeek's MTP
1013                    // layer is dense, Qwen3.6's is a full MoE (router + 256
1014                    // experts + shared). Same builder as a backbone layer.
1015                    ffn: build_ffn_at(model, &arch, &format!("{p}layers.0."), false, ov)?,
1016                    attn,
1017                },
1018                final_norm: load_f32(model, &format!("{p}norm.weight"), ov).map_err(err)?,
1019                kv: LayerKvCache::new(arch.num_kv_heads, arch.head_dim),
1020            })
1021        } else {
1022            None
1023        };
1024
1025        tracing::info!(
1026            "Pipeline loaded: {} | {}L ({} linear) | {:.2}B params | storage: {} | MTP: {}",
1027            arch.arch_name,
1028            arch.num_layers,
1029            arch.layer_types
1030                .iter()
1031                .filter(|t| matches!(t, LayerType::LinearAttention))
1032                .count(),
1033            model.total_param_count() as f64 / 1e9,
1034            if force_f32 {
1035                "f32 (masked)"
1036            } else {
1037                "quantized mmap"
1038            },
1039            if mtp.is_some() { "yes" } else { "no" }
1040        );
1041
1042        // KV window: the descriptor's max, capped for dev-box safety;
1043        // CMF_MAX_SEQ overrides the cap (long-context runs).
1044        let cap = std::env::var("CMF_MAX_SEQ")
1045            .ok()
1046            .and_then(|v| v.parse::<usize>().ok())
1047            .unwrap_or(8192);
1048        let max_seq_len = arch.max_position_embeddings.min(cap);
1049
1050        // Looped Transformer: total virtual layers = physical × num_loops.
1051        let total_layers = arch.num_layers * arch.num_loops;
1052
1053        let mut pipeline = Pipeline::new(
1054            tokenizer,
1055            PipelineWeights {
1056                embed_tokens,
1057                layers,
1058                lm_head,
1059                final_norm,
1060            },
1061            arch.hidden_size,
1062            arch.intermediate_size,
1063            arch.num_attention_heads,
1064            arch.num_kv_heads,
1065            arch.head_dim,
1066            total_layers,
1067            arch.num_layers, // physical layers in weights
1068            arch.loop_final_norm,
1069            arch.vocab_size,
1070            arch.rms_norm_eps,
1071            arch.rope_theta as f32,
1072            arch.norm_style,
1073            max_seq_len,
1074            sampler_config,
1075        );
1076        let rotary = ((arch.head_dim as f32 * arch.partial_rotary_factor) as usize).max(2);
1077        pipeline.set_rotary(rotary, arch.rope_theta as f32);
1078        pipeline.attention_heads_per_layer = arch.attention_heads_per_layer.clone();
1079        if let Some(yarn) = &arch.yarn {
1080            pipeline.inv_freq = std::sync::Arc::new(crate::attention::yarn_inv_freq(
1081                rotary,
1082                arch.rope_theta as f32,
1083                yarn.factor,
1084                yarn.original_max_position_embeddings,
1085                yarn.beta_fast,
1086                yarn.beta_slow,
1087            ));
1088            pipeline.rope_scale = yarn.attention_factor;
1089        }
1090        // Gemma-family extras: embedding scale, attention-scale
1091        // override, and (Gemma-3) sliding-window layers with their own
1092        // local RoPE base.
1093        pipeline.embed_multiplier = arch.embed_multiplier;
1094        pipeline.logit_multiplier = arch.logit_multiplier;
1095        if let Some(qpas) = arch.query_pre_attn_scalar {
1096            pipeline.attn_scale = 1.0 / (qpas as f32).sqrt();
1097        }
1098        if let (Some(w), Some(p)) = (arch.sliding_window, arch.sliding_window_pattern) {
1099            pipeline.swa = Some((w, p));
1100            if let Some(base) = arch.rope_local_base_freq {
1101                pipeline.inv_freq_local = Some(std::sync::Arc::new(
1102                    crate::attention::rope_inv_freq(rotary, base as f32),
1103                ));
1104            }
1105        }
1106        let explicit_sliding: Vec<bool> = arch
1107            .layer_types
1108            .iter()
1109            .map(|t| matches!(t, cortiq_core::LayerType::SlidingAttention))
1110            .collect();
1111        if explicit_sliding.iter().any(|&v| v) {
1112            pipeline.sliding_layers = Some(explicit_sliding);
1113            if let Some(w) = arch.sliding_window {
1114                pipeline.swa = Some((w, usize::MAX));
1115            }
1116            let local_rotary = ((arch.head_dim as f32
1117                * arch
1118                    .local_partial_rotary_factor
1119                    .unwrap_or(arch.partial_rotary_factor))
1120                as usize)
1121                .max(2);
1122            pipeline.rotary_dim_local = Some(local_rotary);
1123            if let Some(base) = arch.rope_local_base_freq {
1124                pipeline.inv_freq_local = Some(std::sync::Arc::new(
1125                    crate::attention::rope_inv_freq(local_rotary, base as f32),
1126                ));
1127            }
1128        }
1129        // Gemma-4: global layers run their own geometry (MQA at
1130        // global_head_dim) with a proportional RoPE — the first
1131        // factor·head_dim dims rotate, the zero-padded tail is identity.
1132        if let (Some(ghd), Some(gkv)) = (arch.global_head_dim, arch.num_global_kv_heads) {
1133            pipeline.global_attn = Some((ghd, gkv));
1134            let prf = arch.global_partial_rotary_factor.unwrap_or(1.0);
1135            let half = ghd / 2;
1136            let ra = (((prf * ghd as f32) as usize) / 2).min(half);
1137            let mut f = vec![0.0f32; half];
1138            for (i, slot) in f.iter_mut().enumerate().take(ra) {
1139                *slot = 1.0 / (arch.rope_theta as f32).powf(2.0 * i as f32 / ghd as f32);
1140            }
1141            pipeline.inv_freq_global = Some(std::sync::Arc::new(f));
1142            // Re-shape the global layers' KV storage to their geometry.
1143            // An explicit layer_types map wins over the numeric pattern
1144            // (explicit tags set swa's pattern to usize::MAX, which
1145            // would otherwise leave every global cache mis-shaped).
1146            let global_at = |li: usize| -> bool {
1147                match &pipeline.sliding_layers {
1148                    Some(map) => !map.get(li).copied().unwrap_or(false),
1149                    None => pipeline
1150                        .swa
1151                        .map(|(_, p)| p > 0 && p != usize::MAX && (li + 1) % p == 0)
1152                        .unwrap_or(false),
1153                }
1154            };
1155            for li in 0..arch.num_layers {
1156                if global_at(li) {
1157                    pipeline.kv_cache.layers[li] = crate::kv_cache::LayerKvCache::new(gkv, ghd);
1158                }
1159            }
1160        }
1161        // MLA (DeepSeek-V2): the expand-to-MHA cache holds nh heads of
1162        // rope+nope dims; rotary covers the rope prefix.
1163        if let Some(mla) = arch.mla.as_ref() {
1164            let hd = mla.qk_rope_head_dim + mla.qk_nope_head_dim;
1165            pipeline.head_dim = hd;
1166            pipeline.num_kv_heads = arch.num_attention_heads;
1167            pipeline.rotary_dim = mla.qk_rope_head_dim;
1168            let half = mla.qk_rope_head_dim / 2;
1169            let mut f = vec![0.0f32; half];
1170            for (i, slot) in f.iter_mut().enumerate() {
1171                *slot = 1.0
1172                    / (arch.rope_theta as f32).powf(2.0 * i as f32 / mla.qk_rope_head_dim as f32);
1173            }
1174            pipeline.inv_freq = std::sync::Arc::new(f);
1175            for li in 0..arch.num_layers {
1176                pipeline.kv_cache.layers[li] =
1177                    crate::kv_cache::LayerKvCache::new(arch.num_attention_heads, hd);
1178            }
1179        }
1180        // Per-frequency rope divisors (MiniCPM3 longrope short_factor):
1181        // served at the native window with the trained per-dim factors.
1182        // Applied after every inv_freq build (plain, YaRN, MLA).
1183        if let Some(fac) = &arch.rope_freq_factors {
1184            let mut f = pipeline.inv_freq.as_ref().clone();
1185            for (i, v) in f.iter_mut().enumerate() {
1186                if let Some(&d) = fac.get(i) {
1187                    *v /= d as f32;
1188                }
1189            }
1190            pipeline.inv_freq = std::sync::Arc::new(f);
1191        }
1192        pipeline.attn_v_norm = arch.attn_v_norm;
1193        pipeline.final_softcap = arch.final_logit_softcapping.map(|c| c as f32);
1194        pipeline.attn_softcap = arch.attn_logit_softcapping.unwrap_or(0.0) as f32;
1195        pipeline.vmf_cfg = vmf_cfg;
1196        pipeline.gdn_cfg = gdn_cfg;
1197        pipeline.kda_cfg = kda_cfg;
1198        if let Some(gc) = arch.g3n.as_ref() {
1199            use crate::g3n::{G3nAltUp, G3nGlobals, G3nLaurel, G3nLayer};
1200            anyhow_like(gc.altup_num_inputs == crate::g3n::ALTUP_N).map_err(|_| {
1201                CmfError::Parse(format!(
1202                    "g3n: altup_num_inputs {} != supported {}",
1203                    gc.altup_num_inputs,
1204                    crate::g3n::ALTUP_N
1205                ))
1206            })?;
1207            let t = |name: &str| load_matrix(model, name, force_f32, ov);
1208            let f = |name: &str| load_f32(model, name, ov).map_err(err);
1209            let mut altup_proj = Vec::new();
1210            let mut altup_unembed = Vec::new();
1211            for i in 0..crate::g3n::ALTUP_N - 1 {
1212                altup_proj.push(t(&format!("model.altup_projections.{i}.weight"))?);
1213                altup_unembed.push(t(&format!("model.altup_unembed_projections.{i}.weight"))?);
1214            }
1215            let first_shared = arch.num_layers.saturating_sub(gc.num_kv_shared_layers);
1216            let sliding_of = |li: usize| {
1217                matches!(
1218                    arch.layer_types.get(li),
1219                    Some(cortiq_core::LayerType::SlidingAttention)
1220                )
1221            };
1222            let mut g3n_layers = Vec::with_capacity(arch.num_layers);
1223            for li in 0..arch.num_layers {
1224                let pfx = format!("model.layers.{li}.");
1225                let shared = li >= first_shared && first_shared > 0;
1226                let share_src = if shared {
1227                    let want = sliding_of(li);
1228                    (0..first_shared).rev().find(|&j| sliding_of(j) == want)
1229                } else {
1230                    None
1231                };
1232                g3n_layers.push(G3nLayer {
1233                    altup: G3nAltUp {
1234                        router_norm: f(&format!("{pfx}altup.router_norm.weight"))?,
1235                        modality_router: t(&format!("{pfx}altup.modality_router.weight"))?,
1236                        prediction_coefs: t(&format!("{pfx}altup.prediction_coefs.weight"))?,
1237                        correction_coefs: t(&format!("{pfx}altup.correction_coefs.weight"))?,
1238                        correct_output_scale: f(&format!("{pfx}altup.correct_output_scale"))?,
1239                    },
1240                    laurel: G3nLaurel {
1241                        left: t(&format!("{pfx}laurel.linear_left.weight"))?,
1242                        right: t(&format!("{pfx}laurel.linear_right.weight"))?,
1243                        post_norm: f(&format!("{pfx}laurel.post_laurel_norm.weight"))?,
1244                    },
1245                    input_norm: f(&format!("{pfx}input_layernorm.weight"))?,
1246                    post_attn_norm: f(&format!("{pfx}post_attention_layernorm.weight"))?,
1247                    pre_ffw_norm: f(&format!("{pfx}pre_feedforward_layernorm.weight"))?,
1248                    post_ffw_norm: f(&format!("{pfx}post_feedforward_layernorm.weight"))?,
1249                    wq: t(&format!("{pfx}self_attn.q_proj.weight"))?,
1250                    wk: if shared {
1251                        None
1252                    } else {
1253                        Some(t(&format!("{pfx}self_attn.k_proj.weight"))?)
1254                    },
1255                    wv: if shared {
1256                        None
1257                    } else {
1258                        Some(t(&format!("{pfx}self_attn.v_proj.weight"))?)
1259                    },
1260                    wo: t(&format!("{pfx}self_attn.o_proj.weight"))?,
1261                    q_norm: f(&format!("{pfx}self_attn.q_norm.weight"))?,
1262                    k_norm: if shared {
1263                        None
1264                    } else {
1265                        Some(f(&format!("{pfx}self_attn.k_norm.weight"))?)
1266                    },
1267                    kv_share_src: share_src,
1268                    sliding: sliding_of(li),
1269                    gate: t(&format!("{pfx}mlp.gate_proj.weight"))?,
1270                    up: t(&format!("{pfx}mlp.up_proj.weight"))?,
1271                    down: t(&format!("{pfx}mlp.down_proj.weight"))?,
1272                    sparsity: gc.activation_sparsity.get(li).copied().unwrap_or(0.0),
1273                    ple_gate: t(&format!("{pfx}per_layer_input_gate.weight"))?,
1274                    ple_proj: t(&format!("{pfx}per_layer_projection.weight"))?,
1275                    post_ple_norm: f(&format!("{pfx}post_per_layer_input_norm.weight"))?,
1276                });
1277            }
1278            let hd = arch.head_dim;
1279            let globals = G3nGlobals {
1280                altup_proj,
1281                altup_unembed,
1282                ple_embed: t("model.embed_tokens_per_layer.weight")?,
1283                ple_model_proj: t("model.per_layer_model_projection.weight")?,
1284                ple_norm: f("model.per_layer_projection_norm.weight")?,
1285                ple_vocab: gc.ple_vocab,
1286                ple_dim: gc.ple_dim,
1287                num_layers: arch.num_layers,
1288                hidden: arch.hidden_size,
1289                rms_eps: arch.rms_norm_eps,
1290                inv_freq_local: crate::attention::rope_inv_freq(
1291                    hd,
1292                    arch.rope_local_base_freq.unwrap_or(10_000.0) as f32,
1293                ),
1294                inv_freq_global: crate::attention::rope_inv_freq(hd, arch.rope_theta as f32),
1295                window: arch.sliding_window.unwrap_or(512),
1296            };
1297            pipeline.g3n = Some(Box::new((globals, g3n_layers)));
1298        }
1299        // DeepSeek-V4: its own stack, selected by the arch name the
1300        // converter wrote. Loading failure is fatal rather than a silent
1301        // fallback — the generic loop cannot represent this model at all,
1302        // so a fallback would decode noise.
1303        if arch.arch_name == "deepseek_v4" {
1304            let moe = arch
1305                .moe
1306                .as_ref()
1307                .ok_or_else(|| CmfError::Parse("deepseek_v4: no moe config".into()))?;
1308            let cfg = crate::dsv4::Dsv4Cfg {
1309                dim: arch.hidden_size,
1310                n_heads: arch.num_attention_heads,
1311                head_dim: arch.head_dim,
1312                // The rope tail: `partial_rotary_factor` carries it when the
1313                // conversion recorded it (rd/head_dim), which the tensors
1314                // cannot reveal. Files converted before that carry 1.0,
1315                // meaning "unset" here rather than "rotate everything" —
1316                // for those the release's 64 stands in, which is what they
1317                // were converted from.
1318                rope_head_dim: if arch.partial_rotary_factor < 1.0 {
1319                    (((arch.head_dim as f32 * arch.partial_rotary_factor) as usize) & !1)
1320                        .clamp(2, arch.head_dim)
1321                } else {
1322                    64.min(arch.head_dim)
1323                },
1324                // The LoRA ranks and the group count ARE visible in the
1325                // weights, and reading them there means a re-tuned
1326                // checkpoint loads without touching this code.
1327                q_lora_rank: 0,
1328                o_lora_rank: 0,
1329                // Derived below from wo_a's shape — the attention output is
1330                // n_heads*head_dim wide and wo_a takes one group of it per
1331                // row block, so groups = width / wo_a.cols(). A pinned 8 is
1332                // right for the release and wrong for anything else, which
1333                // is exactly what made a toy checkpoint impossible to
1334                // compare against the reference.
1335                o_groups: 8,
1336                hc_mult: 4,
1337                hc_sinkhorn_iters: 20,
1338                hc_eps: 1e-6,
1339                norm_eps: arch.rms_norm_eps as f32,
1340                n_routed_experts: moe.num_experts,
1341                top_k: moe.top_k,
1342                moe_inter: moe.moe_intermediate_size,
1343                route_scale: moe.routed_scaling_factor.unwrap_or(1.0) as f32,
1344                // config.json's `swiglu_limit`, which the header has no
1345                // field for. The release ships 10.0; a checkpoint that
1346                // retunes it would need this read from the config, so it
1347                // sits next to the other pinned constants rather than
1348                // hiding inside the expert.
1349                swiglu_limit: 10.0,
1350                window: arch.sliding_window.unwrap_or(128),
1351                index_topk: 512,
1352                vocab: arch.vocab_size,
1353            };
1354            let (g, dl) = crate::dsv4::load(model, &cfg, arch.num_layers)
1355                .map_err(|e| CmfError::Parse(format!("deepseek_v4: {e}")))?;
1356            // Read the ranks off the weights that define them: wq_a's
1357            // rows ARE q_lora_rank, and wo_b's columns are groups x
1358            // o_lora_rank. A header field could disagree with the file;
1359            // these cannot.
1360            let mut cfg = cfg;
1361            if let Some(l0) = dl.first() {
1362                cfg.q_lora_rank = l0.wq_a.rows();
1363                let attn_width = arch.num_attention_heads * arch.head_dim;
1364                if l0.wo_a.cols() > 0 && attn_width % l0.wo_a.cols() == 0 {
1365                    cfg.o_groups = (attn_width / l0.wo_a.cols()).max(1);
1366                }
1367                cfg.o_lora_rank = l0.wo_b.cols() / cfg.o_groups.max(1);
1368                cfg.hc_mult = (l0.hc_attn_fn.len() / l0.hc_attn_base.len().max(1)) / cfg.dim.max(1);
1369                if cfg.hc_mult == 0 {
1370                    cfg.hc_mult = 4;
1371                }
1372            }
1373            // RoPE rides only the last `rope_head_dim` of each head, and the
1374            // reference builds its frequencies over THAT width — not over
1375            // head_dim, which is 512 here. The generic path above sized them
1376            // by head_dim, giving 1/base^(2i/512) where 1/base^(2i/64) is
1377            // wanted: every position rotated by the wrong angle.
1378            //
1379            // YaRN is applied unconditionally by the reference (its guard is
1380            // `original_seq_len > 0`, not the sequence length), so it belongs
1381            // in these frequencies too. Older configs spell the key `type`
1382            // rather than `rope_type`; when the header carries no profile the
1383            // release's own numbers stand in, which is better than silently
1384            // decoding with unscaled frequencies.
1385            let (yf, yo, ybf, ybs) = match &arch.yarn {
1386                Some(y) => (
1387                    y.factor,
1388                    y.original_max_position_embeddings,
1389                    y.beta_fast,
1390                    y.beta_slow,
1391                ),
1392                None => {
1393                    tracing::warn!(
1394                        "deepseek_v4: the header carries no YaRN profile — \
1395                         falling back to the release's (factor 16, original \
1396                         65536, beta 32/1). Re-converting with a build that \
1397                         reads rope_scaling.type would make this exact."
1398                    );
1399                    (16.0, 65536, 32.0, 1.0)
1400                }
1401            };
1402            pipeline.inv_freq = std::sync::Arc::new(crate::attention::yarn_inv_freq(
1403                cfg.rope_head_dim,
1404                arch.rope_theta as f32,
1405                yf,
1406                yo,
1407                ybf,
1408                ybs,
1409            ));
1410            // Keep the working set resident. Everything but the routed
1411            // experts is touched by every token, and of the experts only the
1412            // ones the task actually routes to — the page cache cannot know
1413            // that and evicts by age instead.
1414            if let Ok(stats) = std::env::var("CMF_MOE_PIN") {
1415                let cover = std::env::var("CMF_MOE_PIN_COVER")
1416                    .ok()
1417                    .and_then(|v| v.parse::<f64>().ok())
1418                    .filter(|&c| c > 0.0 && c <= 1.0)
1419                    .unwrap_or(0.95);
1420                let hot = crate::pin::hot_experts(&stats, cover);
1421                let mut names: Vec<String> = Vec::new();
1422                for e in &model.tensors {
1423                    let is_expert = e.name.contains(".mlp.experts.");
1424                    if !is_expert {
1425                        names.push(e.name.clone()); // skeleton: always hot
1426                    }
1427                }
1428                let mut kept_experts = 0usize;
1429                if let Some(hot) = &hot {
1430                    for (li, experts) in hot {
1431                        for e in experts {
1432                            for w in ["gate_proj", "up_proj", "down_proj"] {
1433                                names.push(format!("model.layers.{li}.mlp.experts.{e}.{w}.weight"));
1434                            }
1435                            kept_experts += 1;
1436                        }
1437                    }
1438                }
1439                let r = crate::pin::pin_tensors(model, &names);
1440                tracing::info!(
1441                    "закреплено {:.1} ГБ ({} тензоров, горячих экспертов {kept_experts},                      покрытие {cover}); лимит {}",
1442                    r.bytes as f64 / 1e9,
1443                    r.tensors,
1444                    r.limit
1445                        .map(|l| format!("{:.1} ГБ", l as f64 / 1e9))
1446                        .unwrap_or_else(|| "неизвестен".into())
1447                );
1448                if r.skipped > 0 {
1449                    tracing::warn!("не закреплено тензоров: {}", r.skipped);
1450                }
1451            }
1452            let st = crate::dsv4::Dsv4State::new(arch.num_layers);
1453            pipeline.dsv4 = Some(Box::new((g, dl, cfg, st)));
1454        }
1455        pipeline.short_conv_cfg = short_conv_cfg;
1456        pipeline.mtp = mtp;
1457        pipeline.install_dynamic_routing(model, false);
1458        // Record the load-time overlay so a later set_active_skill(None)
1459        // correctly reverts it (the union-diff assumes dyn_active mirrors
1460        // the live overlay). Blend loads have no single index to revert.
1461        match ov {
1462            Overlay::One(sid) => {
1463                pipeline.dyn_active = model.header.skills.iter().position(|s| &s.id == sid);
1464            }
1465            Overlay::Blend(_) => pipeline.dyn_blend_loaded = true,
1466            Overlay::None => {}
1467        }
1468        // B1: apply the measured confidence-calibration temperature, if the
1469        // file carries one (softmax(logits / T) for reported Born mass).
1470        if let Some(c) = &model.header.calibration {
1471            pipeline.set_calib_temp(c.temperature);
1472        }
1473        // O(1) Nyström attention (runtime-level, no format change):
1474        // env CMF_O1 decides; unset falls through to the converter hint
1475        // in header.provenance.o1_attn (`cortiq convert --o1`), and
1476        // CMF_O1=off force-disables even the hint. CLI flags override
1477        // later via set_o1().
1478        let o1 = match crate::nystrom::o1_from_env() {
1479            crate::nystrom::O1Env::Off => None,
1480            crate::nystrom::O1Env::On(cfg) => Some(cfg),
1481            crate::nystrom::O1Env::Unset => model
1482                .header
1483                .provenance
1484                .as_ref()
1485                .and_then(|p| p.get("o1_attn"))
1486                .and_then(crate::nystrom::O1Cfg::from_json),
1487        };
1488        if o1.is_some() {
1489            if pipeline.attn_softcap > 0.0 {
1490                return Err(CmfError::Parse(
1491                    "--o1 with attention-logit soft-capping (Gemma-2) is not supported: \
1492                     the streaming operator has no capped-score form"
1493                        .into(),
1494                ));
1495            }
1496            pipeline.set_o1(o1);
1497        }
1498        Ok(pipeline)
1499    }
1500
1501    /// Record per-skill dynamic-routing metadata: which FFN layers each
1502    /// skill actually replaces (derived from the tensors present, not
1503    /// the meta `layers` field), and whether the skill is eligible for
1504    /// cheap dynamic switching (FFN-only). Called once at load.
1505    pub(crate) fn install_dynamic_routing(&mut self, model: &Arc<CmfModel>, force_f32: bool) {
1506        self.model = Some(model.clone());
1507        self.dyn_force_f32 = force_f32;
1508        let mut per_skill = Vec::with_capacity(model.header.skills.len());
1509        for sk in &model.header.skills {
1510            let mut ffn_layers = std::collections::BTreeSet::new();
1511            let mut non_ffn = false;
1512            let prefix = format!("skill.{}.", sk.id);
1513            for t in model.skill_tensors(&sk.id) {
1514                let rel = &t.name[prefix.len()..]; // e.g. model.layers.20.mlp.down_proj.weight
1515                let toks: Vec<&str> = rel.split('.').collect();
1516                if toks.len() >= 5 && toks[0] == "model" && toks[1] == "layers" && toks[3] == "mlp"
1517                {
1518                    if let Ok(li) = toks[2].parse::<usize>() {
1519                        ffn_layers.insert(li);
1520                        continue;
1521                    }
1522                }
1523                non_ffn = true; // replaces attention / embed / lm_head
1524            }
1525            if non_ffn {
1526                tracing::warn!(
1527                    "skill '{}' replaces non-FFN tensors — excluded from dynamic \
1528                     routing (static overlay still works)",
1529                    sk.id
1530                );
1531                per_skill.push(None);
1532            } else {
1533                per_skill.push(Some(ffn_layers.into_iter().collect::<Vec<_>>()));
1534            }
1535        }
1536        self.dyn_skill_layers = per_skill;
1537    }
1538
1539    /// Switch the overlaid skill for subsequent forwards (dynamic
1540    /// routing). `idx` = index into model.header.skills; None = backbone.
1541    /// Rebuilds the FFN of the union of the old and new skill's touched
1542    /// layers with the new overlay — tensor-source indirection made
1543    /// dynamic. Cheap: Mapped tensors are re-resolved mmap pointers.
1544    /// Result is bit-identical to loading the pipeline with that skill.
1545    pub fn set_active_skill(&mut self, idx: Option<usize>) -> Result<(), CmfError> {
1546        // Overlay swap changes weights → every cached K/V is stale.
1547        self.kv_cache.clear();
1548        self.kv_history.clear();
1549        if self.dyn_active == idx {
1550            return Ok(());
1551        }
1552        let model = self.model.clone().ok_or_else(|| {
1553            CmfError::Parse("dynamic routing needs a model-backed pipeline".into())
1554        })?;
1555        let mut union: std::collections::BTreeSet<usize> = std::collections::BTreeSet::new();
1556        if let Some(old) = self.dyn_active {
1557            if let Some(Some(ls)) = self.dyn_skill_layers.get(old) {
1558                union.extend(ls.iter().copied());
1559            }
1560        }
1561        let new_id: Option<String> = match idx {
1562            Some(n) => match self.dyn_skill_layers.get(n) {
1563                Some(Some(ls)) => {
1564                    union.extend(ls.iter().copied());
1565                    Some(model.header.skills[n].id.clone())
1566                }
1567                _ => {
1568                    return Err(CmfError::Parse(format!(
1569                        "skill index {n} not dynamic-eligible"
1570                    )));
1571                }
1572            },
1573            None => None,
1574        };
1575        let ov = match &new_id {
1576            Some(s) => Overlay::One(s),
1577            None => Overlay::None,
1578        };
1579        let arch = model.arch();
1580        for li in union {
1581            self.weights.layers[li].ffn =
1582                build_layer_ffn(&model, arch, li, self.dyn_force_f32, &ov)?;
1583        }
1584        self.dyn_active = idx;
1585        Ok(())
1586    }
1587}