Skip to main content

cortiq_engine/
loader.rs

1//! Weight loader: CMF tensor directory → Pipeline.
2//!
3//! Storage rule: models WITH task masks are dequantized to f32 (masked
4//! execution needs f32 row access; skill files are small by design).
5//! Models without masks keep quantized matrices zero-copy from the mmap
6//! (`QTensor::Mapped`) — this is what lets a 15B file run in a few GB
7//! of RSS instead of 60 GB of f32.
8//!
9//! Layer kinds come from `arch.layer_types`: FullAttention loads
10//! `self_attn.*` (with auto-detected Qwen3.5 extras: per-head qk-norm by
11//! tensor presence, output gate by q_proj row count); LinearAttention
12//! loads the canonical core `vmf_attn.*` (folded at convert time).
13
14use crate::kv_cache::LayerKvCache;
15use crate::linear_core::{
16    GdnCfg, GdnWeights, ShortConvCfg, ShortConvWeights, VmfPhaseCfg, VmfPhaseWeights,
17};
18use crate::pipeline::{
19    AttnKind, DenseFfn, FfnKind, LayerWeights, MoeFfn, MtpModule, Pipeline, PipelineWeights,
20};
21use crate::qtensor::QTensor;
22use crate::sampler::SamplerConfig;
23use crate::tokenizer::Tokenizer;
24use cortiq_core::quant::dequant_tensor;
25use cortiq_core::{CmfError, CmfModel, LayerType, ModelArch};
26use std::sync::Arc;
27
28/// The refusal every generative entry point (`Pipeline::from_model*`, and
29/// through it run/chat/route/bench) gives a file carrying the DECISION
30/// feature bit. Such a file is served by the decision runtime instead.
31pub const DECISION_MODEL_REFUSAL: &str =
32    "this is a DECISION model; use `cortiq decide` or `cortiq serve`";
33
34/// Tensor source selector (spec §9): backbone, one skill's overlay, or
35/// a soft superposition of top-m skills (claim 14 working tensors).
36pub enum Overlay<'a> {
37    None,
38    One(&'a str),
39    /// (skill_id, weight); weights sum to 1 (softmax(−E/T) upstream).
40    Blend(&'a [(String, f32)]),
41}
42
43impl Overlay<'_> {
44    fn blend_touches(&self, model: &CmfModel, name: &str) -> bool {
45        match self {
46            Overlay::Blend(list) => list
47                .iter()
48                .any(|(sid, _)| model.tensor(&format!("skill.{sid}.{name}")).is_some()),
49            _ => false,
50        }
51    }
52}
53
54fn dequant_by_name(model: &CmfModel, name: &str) -> Result<Vec<f32>, String> {
55    let entry = model
56        .tensor(name)
57        .ok_or_else(|| format!("tensor '{name}' not found in CMF directory"))?;
58    let mut out = vec![0.0f32; entry.n_elems()];
59    dequant_tensor(entry, model.entry_bytes(entry), &mut out)?;
60    Ok(out)
61}
62
63/// Weighted working tensor (claim 14): Σ wᵢ·Tᵢ, where Tᵢ is the
64/// skill's replacement when present, else the backbone tensor.
65fn blend_f32(model: &CmfModel, name: &str, list: &[(String, f32)]) -> Result<Vec<f32>, String> {
66    let mut acc: Option<Vec<f32>> = None;
67    for (sid, w) in list {
68        let sname = format!("skill.{sid}.{name}");
69        let src = if model.tensor(&sname).is_some() {
70            &sname
71        } else {
72            name
73        };
74        let t = dequant_by_name(model, src)?;
75        match &mut acc {
76            None => {
77                let mut t = t;
78                for v in t.iter_mut() {
79                    *v *= w;
80                }
81                acc = Some(t);
82            }
83            Some(a) => {
84                for (av, tv) in a.iter_mut().zip(&t) {
85                    *av += w * tv;
86                }
87            }
88        }
89    }
90    acc.ok_or_else(|| "empty blend".into())
91}
92
93/// Dequantize a tensor fully into f32 (norms, masked models).
94pub(crate) fn load_f32(model: &CmfModel, name: &str, ov: &Overlay) -> Result<Vec<f32>, String> {
95    if ov.blend_touches(model, name) {
96        if let Overlay::Blend(list) = ov {
97            return blend_f32(model, name, list);
98        }
99    }
100    let skill = match ov {
101        Overlay::One(s) => Some(*s),
102        _ => None,
103    };
104    let entry = model
105        .resolve_tensor(name, skill)
106        .ok_or_else(|| format!("tensor '{name}' not found in CMF directory"))?;
107    let bytes = model.entry_bytes(entry);
108    let mut out = vec![0.0f32; entry.n_elems()];
109    dequant_tensor(entry, bytes, &mut out)?;
110    Ok(out)
111}
112
113/// Build one layer's FFN (dense or MoE) under a given overlay. Shared
114/// by the static loader AND dynamic per-token skill switching
115/// (`Pipeline::set_active_skill`): switching skills = rebuilding the
116/// FFN of the touched layers, cheap because Mapped tensors are just
117/// re-resolved mmap pointers (no dequant, no copy).
118pub(crate) fn build_layer_ffn(
119    model: &Arc<CmfModel>,
120    arch: &ModelArch,
121    li: usize,
122    force_f32: bool,
123    ov: &Overlay,
124) -> Result<FfnKind, CmfError> {
125    build_ffn_at(model, arch, &format!("model.layers.{li}."), force_f32, ov)
126}
127
128/// The FFN under an arbitrary prefix. Split out of `build_layer_ffn` so the
129/// MTP block can reuse it: Qwen3.6's MTP layer carries a full MoE mlp
130/// (router + 256 experts + shared expert), not the dense one the head was
131/// first written against.
132pub(crate) fn build_ffn_at(
133    model: &Arc<CmfModel>,
134    arch: &ModelArch,
135    prefix: &str,
136    force_f32: bool,
137    ov: &Overlay,
138) -> Result<FfnKind, CmfError> {
139    let prefix = prefix.to_string();
140    let load_dense = |p: &str| -> Result<DenseFfn, CmfError> {
141        let gate_proj = load_matrix(model, &format!("{p}gate_proj.weight"), force_f32, ov)?;
142        let up_proj = load_matrix(model, &format!("{p}up_proj.weight"), force_f32, ov)?;
143        let down_proj = load_matrix(model, &format!("{p}down_proj.weight"), force_f32, ov)?;
144        // FFN triple invariant (holds for dense and each MoE expert;
145        // enforced loudly so a malformed defrag/repack — spec §11 — fails
146        // at load instead of silently mis-computing). inter' is per-layer.
147        let inter = gate_proj.rows();
148        if up_proj.rows() != inter || down_proj.cols() != inter {
149            return Err(CmfError::Parse(format!(
150                "{p}: FFN dims disagree (gate.rows={inter}, up.rows={}, \
151                 down.cols={}); all three must equal inter'",
152                up_proj.rows(),
153                down_proj.cols()
154            )));
155        }
156        if down_proj.rows() != arch.hidden_size {
157            return Err(CmfError::Parse(format!(
158                "{p}: down_proj.rows={} != hidden_size={}",
159                down_proj.rows(),
160                arch.hidden_size
161            )));
162        }
163        // The transposed down, when the file carries one: the per-token
164        // sparse path needs a neuron's down weights contiguous.
165        let dt_name = format!("{p}down_proj.t.weight");
166        let down_t = match model.tensor(&dt_name) {
167            Some(_) => Some(load_matrix(model, &dt_name, force_f32, ov)?),
168            None => None,
169        };
170        if let Some(t) = &down_t
171            && (t.rows() != inter || t.cols() != arch.hidden_size)
172        {
173            return Err(CmfError::Parse(format!(
174                "{p}down_proj.t: [{}, {}] != [{inter}, {}]",
175                t.rows(),
176                t.cols(),
177                arch.hidden_size
178            )));
179        }
180        // Task tubes (`…gate_proj.tube1.weight`, …): the neurons only
181        // some tasks compute, each stored as its own triple so every
182        // kernel runs it unchanged and an inactive tube's bytes are
183        // never touched. Numbering is dense from 1; the first gap ends
184        // the tube list.
185        let mut segs = Vec::new();
186        let mut start = inter;
187        for k in 1.. {
188            let gn = format!("{p}gate_proj.tube{k}.weight");
189            if model.tensor(&gn).is_none() {
190                break;
191            }
192            let gate = load_matrix(model, &gn, force_f32, ov)?;
193            let up = load_matrix(model, &format!("{p}up_proj.tube{k}.weight"), force_f32, ov)?;
194            let down = load_matrix(
195                model,
196                &format!("{p}down_proj.tube{k}.weight"),
197                force_f32,
198                ov,
199            )?;
200            let width = gate.rows();
201            if up.rows() != width || down.cols() != width || down.rows() != arch.hidden_size {
202                return Err(CmfError::Parse(format!(
203                    "{p}tube{k}: dims disagree (gate.rows={width}, up.rows={}, \
204                     down=[{}, {}], hidden={})",
205                    up.rows(),
206                    down.rows(),
207                    down.cols(),
208                    arch.hidden_size
209                )));
210            }
211            segs.push(crate::pipeline::FfnSeg {
212                gate,
213                up,
214                down,
215                start,
216                width,
217            });
218            start += width;
219        }
220        Ok(DenseFfn {
221            gate_proj,
222            up_proj,
223            down_proj,
224            act: crate::pipeline::Act::from_arch_full(arch),
225            down_t,
226            segs,
227        })
228    };
229    let router_name = format!("{prefix}mlp.gate.weight");
230    // Cortiq Embryo: resonance-routed experts carry NO gate tensor — the
231    // MoE is keyed on the first expert + the arch flag (a grown genome
232    // appends `mlp.experts.{E}.*` records without rewriting anything).
233    let resonance_moe = model.tensor(&router_name).is_none()
234        && arch.moe.as_ref().is_some_and(|m| m.router_resonance)
235        && model
236            .tensor(&format!("{prefix}mlp.experts.0.gate_proj.weight"))
237            .is_some();
238    if model.tensor(&router_name).is_none() && !resonance_moe {
239        return Ok(FfnKind::Dense(load_dense(&format!("{prefix}mlp."))?));
240    }
241    let cfg = arch.moe.as_ref().ok_or_else(|| {
242        CmfError::Parse(format!(
243            "{router_name} present but header has no arch.moe block"
244        ))
245    })?;
246    // Experts enumerate by TENSOR PRESENCE up to the header count — a
247    // moe-defrag'd specialist keeps a per-layer contiguous prefix of
248    // renumbered experts (fewer than arch.moe.num_experts), with the
249    // router rows sliced to match.
250    let mut experts = Vec::new();
251    for e in 0..cfg.num_experts {
252        if model
253            .tensor(&format!("{prefix}mlp.experts.{e}.gate_proj.weight"))
254            .is_none()
255        {
256            break;
257        }
258        experts.push(load_dense(&format!("{prefix}mlp.experts.{e}."))?);
259    }
260    if experts.is_empty() {
261        return Err(CmfError::Parse(format!(
262            "{prefix}: router present but no expert tensors"
263        )));
264    }
265    let shared = if model
266        .tensor(&format!("{prefix}mlp.shared_expert.gate_proj.weight"))
267        .is_some()
268    {
269        let gate_name = format!("{prefix}mlp.shared_expert_gate.weight");
270        Some((
271            load_dense(&format!("{prefix}mlp.shared_expert."))?,
272            if model.tensor(&gate_name).is_some() {
273                Some(load_matrix(model, &gate_name, force_f32, ov)?)
274            } else {
275                None
276            },
277        ))
278    } else {
279        None
280    };
281    // LFM2-MoE selection bias (`mlp.expert_bias`): present iff the model
282    // routes with a bias; loaded by tensor presence.
283    let bias_name = format!("{prefix}mlp.expert_bias");
284    let expert_bias = if model.tensor(&bias_name).is_some() {
285        Some(load_f32(model, &bias_name, ov).map_err(CmfError::Parse)?)
286    } else {
287        None
288    };
289    // CMF_MOE_TOPK=N (opt-in): route to fewer experts than the header
290    // asks. MoE decode is memory-bound — every selected expert streams
291    // its three matrices per token — so halving k halves that traffic;
292    // the renormalized top-k keeps the mixture a proper average.
293    // Quality is the experiment — measure ppl before trusting.
294    let top_k = std::env::var("CMF_MOE_TOPK")
295        .ok()
296        .and_then(|v| v.parse::<usize>().ok())
297        .filter(|&k| k >= 1 && k <= cfg.top_k)
298        .inspect(|k| tracing::info!("MoE top_k override: {} (header {})", k, cfg.top_k))
299        .unwrap_or(cfg.top_k);
300    // CMF_MOE_TAU=0.x (opt-in): adaptive routing — see MoeFfn::route_tau.
301    let route_tau = std::env::var("CMF_MOE_TAU")
302        .ok()
303        .and_then(|v| v.parse::<f32>().ok())
304        .filter(|&t| t > 0.0 && t < 1.0)
305        .inspect(|t| tracing::info!("MoE adaptive routing: tau {t}"));
306    let mask = moe_task_mask(model, &prefix, experts.len());
307    let router = if resonance_moe {
308        // never read (selection is by descriptors); zero placeholder in RAM
309        QTensor::from_f32(
310            vec![0.0; experts.len() * arch.hidden_size],
311            experts.len(),
312            arch.hidden_size,
313        )
314    } else {
315        load_matrix(model, &router_name, force_f32, ov)?
316    };
317    if router.rows() != experts.len() {
318        return Err(CmfError::Parse(format!(
319            "{router_name}: {} rows != {} experts",
320            router.rows(),
321            experts.len()
322        )));
323    }
324    let top_k = top_k.min(experts.len());
325    // Gemma-4: per-expert weight scale after the top-k renorm; its
326    // presence also marks the scale-less-rms router input (the folded
327    // router gain — see the converter).
328    let pes_name = format!("{prefix}mlp.per_expert_scale");
329    let per_expert_scale = if model.tensor(&pes_name).is_some() {
330        Some(load_f32(model, &pes_name, ov).map_err(CmfError::Parse)?)
331    } else {
332        None
333    };
334    let router_input_norm = per_expert_scale.is_some();
335    // Cortiq Embryo resonance descriptors: per-expert records
336    // `mlp.experts.{e}.desc.{mu,u,bias}` (append-only growth) or the legacy
337    // per-layer `mlp.desc.{mu,u,bias}` [E, ...]. Present → the router is a
338    // placeholder and selection is by reconstruction error.
339    let per_expert = model
340        .tensor(&format!("{prefix}mlp.experts.0.desc.mu"))
341        .is_some();
342    let resonance = if per_expert {
343        let ne_d = experts.len();
344        let hidden = arch.hidden_size;
345        let mut mu = Vec::with_capacity(ne_d * hidden);
346        let mut u = Vec::new();
347        let mut bias = Vec::with_capacity(ne_d);
348        let mut k = 0usize;
349        for e in 0..ne_d {
350            let m = load_f32(model, &format!("{prefix}mlp.experts.{e}.desc.mu"), ov)
351                .map_err(CmfError::Parse)?;
352            if m.len() != hidden {
353                return Err(CmfError::Parse(format!(
354                    "{prefix}mlp.experts.{e}.desc.mu: {} != {hidden}",
355                    m.len()
356                )));
357            }
358            mu.extend_from_slice(&m);
359            let un = format!("{prefix}mlp.experts.{e}.desc.u");
360            if model.tensor(&un).is_some() {
361                let ue = load_f32(model, &un, ov).map_err(CmfError::Parse)?;
362                let ke = ue.len() / hidden.max(1);
363                if e == 0 {
364                    k = ke;
365                }
366                if ke != k {
367                    return Err(CmfError::Parse(format!("{un}: rank {ke} != {k}")));
368                }
369                u.extend_from_slice(&ue);
370            }
371            let bn = format!("{prefix}mlp.experts.{e}.desc.bias");
372            bias.push(if model.tensor(&bn).is_some() {
373                load_f32(model, &bn, ov)
374                    .map_err(CmfError::Parse)?
375                    .first()
376                    .copied()
377                    .unwrap_or(0.0)
378            } else {
379                0.0
380            });
381        }
382        Some(crate::pipeline::Resonance { mu, u, k, bias })
383    } else if model.tensor(&format!("{prefix}mlp.desc.mu")).is_some() {
384        let mu = load_f32(model, &format!("{prefix}mlp.desc.mu"), ov).map_err(CmfError::Parse)?;
385        let ne_d = experts.len();
386        let hidden = arch.hidden_size;
387        if mu.len() != ne_d * hidden {
388            return Err(CmfError::Parse(format!(
389                "{prefix}mlp.desc.mu: {} != {ne_d}×{hidden}",
390                mu.len()
391            )));
392        }
393        let u_name = format!("{prefix}mlp.desc.u");
394        let (u, k) = if model.tensor(&u_name).is_some() {
395            let u = load_f32(model, &u_name, ov).map_err(CmfError::Parse)?;
396            let k = u.len() / (ne_d * hidden).max(1);
397            (u, k)
398        } else {
399            (Vec::new(), 0)
400        };
401        let b_name = format!("{prefix}mlp.desc.bias");
402        let bias = if model.tensor(&b_name).is_some() {
403            load_f32(model, &b_name, ov).map_err(CmfError::Parse)?
404        } else {
405            vec![0.0; ne_d]
406        };
407        Some(crate::pipeline::Resonance { mu, u, k, bias })
408    } else {
409        None
410    };
411    let moe = MoeFfn {
412        router,
413        experts,
414        top_k,
415        route_tau,
416        norm_topk_prob: cfg.norm_topk_prob,
417        router_sigmoid: cfg.router_sigmoid,
418        expert_bias,
419        routed_scaling: cfg.routed_scaling_factor.unwrap_or(1.0),
420        shared,
421        stats: std::cell::RefCell::new(Vec::new()),
422        act_sq: std::cell::RefCell::new(Vec::new()),
423        act_rows: std::cell::RefCell::new(Vec::new()),
424        mask,
425        per_expert_scale,
426        router_input_norm,
427        resonance,
428    };
429    // Gemma-4 dual-branch layer: a dense MLP coexists with the routed
430    // experts, each branch inside its own norm sandwich.
431    if model
432        .tensor(&format!("{prefix}mlp.gate_proj.weight"))
433        .is_some()
434    {
435        let norm = |suffix: &str| -> Result<Vec<f32>, CmfError> {
436            load_f32(model, &format!("{prefix}{suffix}.weight"), ov).map_err(CmfError::Parse)
437        };
438        return Ok(FfnKind::DenseMoe(Box::new(crate::pipeline::DenseMoeFfn {
439            dense: load_dense(&format!("{prefix}mlp."))?,
440            moe,
441            post_norm_1: norm("post_feedforward_layernorm_1")?,
442            pre_norm_2: norm("pre_feedforward_layernorm_2")?,
443            post_norm_2: norm("post_feedforward_layernorm_2")?,
444        })));
445    }
446    Ok(FfnKind::Moe(moe))
447}
448
449/// Task mask over routed experts: DTG-MA applied to MoE.
450///
451/// This is deliberately diagnostic-only. A route-mass sidecar can look good
452/// on a short perplexity window and still collapse during a long
453/// autoregressive generation: excluding a low-mass expert changes the model,
454/// and the changed prefix changes every later route. Consequently a nearby
455/// `<model>.moe-mass.json` is NEVER adopted implicitly. `CMF_MOE_MASK=...`
456/// remains an explicit experiment and accepts both weighted and legacy count
457/// files; production acceleration must preserve the full router and treat a
458/// non-resident winner as an exact cold completion.
459pub(crate) fn moe_task_mask(
460    _model: &std::sync::Arc<CmfModel>,
461    prefix: &str,
462    ne: usize,
463) -> Option<Vec<bool>> {
464    use std::sync::OnceLock;
465    static CFG: OnceLock<Option<(std::collections::HashMap<usize, Vec<u64>>, f64)>> =
466        OnceLock::new();
467    let cfg = CFG.get_or_init(|| {
468        let path = std::path::PathBuf::from(std::env::var("CMF_MOE_MASK").ok()?);
469        let shown = path.display();
470        let text = std::fs::read_to_string(&path)
471            .map_err(|e| tracing::warn!("CMF_MOE_MASK: cannot read {shown}: {e}"))
472            .ok()?;
473        let map: std::collections::HashMap<String, Vec<u64>> = serde_json::from_str(&text)
474            .map_err(|e| tracing::warn!("CMF_MOE_MASK: bad JSON in {shown}: {e}"))
475            .ok()?;
476        // Weighted dumps store normalized route mass in fixed-point units
477        // and therefore have per-layer totals many orders above the old
478        // top-k vote counts. The magnitude distinguishes the two explicit
479        // diagnostic formats without relying on a special filename.
480        let weighted = map
481            .values()
482            .any(|row| row.iter().copied().sum::<u64>() >= 1_000_000);
483        let cover = std::env::var("CMF_MOE_MASK_COVER")
484            .ok()
485            .and_then(|v| v.parse::<f64>().ok())
486            .filter(|&c| c > 0.0 && c <= 1.0)
487            .unwrap_or(if weighted { 0.925 } else { 0.9 });
488        tracing::info!("MoE task mask: {shown}, cover {cover}");
489        Some((
490            map.into_iter()
491                .filter_map(|(k, v)| Some((k.parse::<usize>().ok()?, v)))
492                .collect(),
493            cover,
494        ))
495    });
496    let (stats, cover) = cfg.as_ref()?;
497    // The layer index rides in the tensor prefix ("model.layers.N.").
498    let li: usize = prefix
499        .split("layers.")
500        .nth(1)?
501        .split('.')
502        .next()?
503        .parse()
504        .ok()?;
505    let counts = stats.get(&li)?;
506    if counts.len() != ne {
507        tracing::warn!(
508            "CMF_MOE_MASK: layer {li} has {} counts, model has {ne} experts — skipped",
509            counts.len()
510        );
511        return None;
512    }
513    let total: u64 = counts.iter().sum();
514    if total == 0 {
515        return None;
516    }
517    let mut order: Vec<usize> = (0..ne).collect();
518    order.sort_unstable_by_key(|&e| std::cmp::Reverse(counts[e]));
519    let mut mask = vec![false; ne];
520    let mut acc = 0u64;
521    let mut kept = 0usize;
522    for &e in &order {
523        mask[e] = true;
524        acc += counts[e];
525        kept += 1;
526        if (acc as f64) >= cover * (total as f64) {
527            break;
528        }
529    }
530    tracing::info!(
531        "MoE task mask L{li}: {kept}/{ne} experts for {:.0}% mass",
532        cover * 100.0
533    );
534    Some(mask)
535}
536
537pub(crate) fn load_matrix(
538    model: &Arc<CmfModel>,
539    name: &str,
540    force_f32: bool,
541    ov: &Overlay,
542) -> Result<QTensor, CmfError> {
543    // Claim 14: a blended working tensor is materialized in f32 and
544    // held resident (the overlay-cache slot); single skills stay
545    // zero-copy pointers into the mmap.
546    if ov.blend_touches(model, name) {
547        if let Overlay::Blend(list) = ov {
548            let entry = model
549                .tensor(name)
550                .ok_or_else(|| CmfError::MissingTensor(name.to_string()))?;
551            let data =
552                blend_f32(model, name, list).map_err(|e| CmfError::Parse(format!("blend: {e}")))?;
553            return Ok(QTensor::from_f32(data, entry.shape[0], entry.shape[1]));
554        }
555    }
556    let skill = match ov {
557        Overlay::One(s) => Some(*s),
558        _ => None,
559    };
560    // Tensor-source indirection (spec §9): the skill's replacement is
561    // read in place of the backbone tensor — either/or, never a sum.
562    let name: &str = &match skill {
563        Some(sid) if model.tensor(&format!("skill.{sid}.{name}")).is_some() => {
564            format!("skill.{sid}.{name}")
565        }
566        _ => name.to_string(),
567    };
568    let err = |e: String| CmfError::Parse(format!("weight loading: {e}"));
569    if force_f32 {
570        let entry = model
571            .tensor(name)
572            .ok_or_else(|| CmfError::MissingTensor(name.to_string()))?;
573        if entry.shape.len() != 2 {
574            return Err(err(format!("'{name}' is not 2-D")));
575        }
576        let data = load_f32(model, name, &Overlay::None).map_err(err)?;
577        Ok(QTensor::from_f32(data, entry.shape[0], entry.shape[1]))
578    } else {
579        QTensor::from_model(model, name).map_err(err)
580    }
581}
582
583impl Pipeline {
584    /// Build a runnable pipeline from an opened CMF model.
585    pub fn from_model(
586        model: &Arc<CmfModel>,
587        sampler_config: SamplerConfig,
588    ) -> Result<Self, CmfError> {
589        Self::from_model_with_skill(model, sampler_config, None)
590    }
591
592    /// Same, with a skill overlaid (spec §9): every layer tensor is
593    /// resolved through tensor-source indirection — the skill's
594    /// full-shape replacement is read in place of the backbone tensor.
595    /// No per-skill model is ever assembled: Mapped tensors are
596    /// pointers into the one shared mmap.
597    pub fn from_model_with_skill(
598        model: &Arc<CmfModel>,
599        sampler_config: SamplerConfig,
600        skill: Option<&str>,
601    ) -> Result<Self, CmfError> {
602        match skill {
603            Some(s) => Self::from_model_with_overlay(model, sampler_config, &Overlay::One(s)),
604            None => Self::from_model_with_overlay(model, sampler_config, &Overlay::None),
605        }
606    }
607
608    /// Soft superposition (claim 14): working tensors accumulated from
609    /// the given (skill, weight) list — softmax(−E/T) upstream.
610    pub fn from_model_with_blend(
611        model: &Arc<CmfModel>,
612        sampler_config: SamplerConfig,
613        blend: &[(String, f32)],
614    ) -> Result<Self, CmfError> {
615        Self::from_model_with_overlay(model, sampler_config, &Overlay::Blend(blend))
616    }
617
618    fn skill_file_guard(model: &CmfModel) -> Result<(), CmfError> {
619        // A decision profile (encoder + resonance skills) has no language
620        // model in it at all; the generative pipeline must not try to
621        // assemble one from its tensors.
622        if model.required_features & cortiq_core::format::features::DECISION != 0 {
623            return Err(CmfError::Parse(DECISION_MODEL_REFUSAL.into()));
624        }
625        // A standalone skill file carries a PARTIAL tensor set cut against
626        // a base; running it would be half a network answering questions.
627        if model.required_features & cortiq_core::format::features::SKILL_FILE != 0 {
628            return Err(CmfError::Parse(
629                "this file is a standalone SKILL, not a runnable model — attach it: \
630                 cortiq skill apply <base.cmf> <this file> -o specialist.cmf"
631                    .into(),
632            ));
633        }
634        Ok(())
635    }
636
637    fn from_model_with_overlay(
638        model: &Arc<CmfModel>,
639        sampler_config: SamplerConfig,
640        ov: &Overlay,
641    ) -> Result<Self, CmfError> {
642        // Refuse non-runnable files first, before any process-wide state
643        // (the device cache directory, the graph verdict) is touched on
644        // their behalf.
645        Self::skill_file_guard(model)?;
646        // Small device caches (probe verdicts, compiled pipelines) go
647        // beside the model: it is a directory the caller demonstrably
648        // writes to, which `std::env::temp_dir()` is not inside an
649        // Android app sandbox.
650        if let Some(dir) = model.path.parent() {
651            crate::gpu::set_cache_dir(dir.to_path_buf());
652        }
653        // A new model gets a fresh verdict on whether the token graph can
654        // be built: the refusal is remembered per model, not per process.
655        crate::gpu::graph_unsupported_reset();
656        let skill = match ov {
657            Overlay::One(s) => Some(*s),
658            _ => None,
659        };
660        if let Some(sid) = skill {
661            let known = model.header.skills.iter().any(|s| s.id == sid)
662                || model.skill_tensors(sid).next().is_some();
663            if !known {
664                return Err(CmfError::Parse(format!(
665                    "skill '{sid}' not in this container (header.skills: {:?})",
666                    model
667                        .header
668                        .skills
669                        .iter()
670                        .map(|s| &s.id)
671                        .collect::<Vec<_>>()
672                )));
673            }
674            tracing::info!(
675                "skill '{sid}': {} replacement tensors overlaid",
676                model.skill_tensors(sid).count()
677            );
678        }
679        let arch = model.arch().clone();
680        let err = |e: String| CmfError::Parse(format!("weight loading: {e}"));
681        if let Some(heads) = &arch.attention_heads_per_layer {
682            if heads.len() != arch.num_layers {
683                return Err(CmfError::Parse(format!(
684                    "arch.attention_heads_per_layer has {} entries, expected {}",
685                    heads.len(),
686                    arch.num_layers
687                )));
688            }
689            if let Some((li, &nh)) = heads
690                .iter()
691                .enumerate()
692                .find(|(_, nh)| **nh == 0 || **nh % arch.num_kv_heads != 0)
693            {
694                return Err(CmfError::Parse(format!(
695                    "layer {li} has {nh} Q heads, which must be nonzero and divisible by {} KV heads",
696                    arch.num_kv_heads
697                )));
698            }
699        }
700        if arch
701            .layer_types
702            .iter()
703            .any(|t| matches!(t, LayerType::SlidingAttention))
704            && arch.sliding_window.is_none()
705        {
706            return Err(CmfError::Parse(
707                "model has SlidingAttention layers but no arch.sliding_window".into(),
708            ));
709        }
710
711        // Masks × quantized mmap: only the HEAD-mask path needs f32
712        // slices, so f32 is forced only when some mask actually restricts
713        // attention heads. FFN masks run sparse directly on the quant
714        // bytes (sparse_ffn_quant), and embed/lm_head are never masked.
715        // The old condition forced f32 for ANY mask: a 4.17 B file whose
716        // mask touched only the FFN dequantized to 16.7 GB at load and
717        // ran the bare-f32 GEMMs — 0.0 tok/s on a laptop that swapped,
718        // and a 30× crawl on a 48-core server. A mask with no head rows
719        // costs nothing now.
720        let heads_masked = model.masks.masks.iter().any(|m| {
721            m.head_masks.iter().any(|row| {
722                let mut bits = 0usize;
723                for &b in row.iter() {
724                    bits += b.count_ones() as usize;
725                }
726                !row.is_empty() && bits < arch.num_attention_heads
727            })
728        });
729        let force_f32 = heads_masked; // attention only (head masks)
730
731        // ── Tokenizer: embedded → sidecar → byte-level fallback ──
732        let mut tokenizer = if let Some(vocab_bytes) = &model.vocab {
733            Tokenizer::from_bytes(vocab_bytes)
734                .map_err(|e| CmfError::Parse(format!("embedded tokenizer: {e}")))?
735        } else {
736            let sidecar = model.path.with_file_name("tokenizer.json");
737            if sidecar.exists() {
738                Tokenizer::from_file(&sidecar)
739                    .map_err(|e| CmfError::Parse(format!("sidecar tokenizer: {e}")))?
740            } else {
741                tracing::warn!("no tokenizer in file or sidecar — using byte-level fallback");
742                Tokenizer::byte_level()
743            }
744        };
745        // Chat/eos bundle (spec §6.1): the FILE defines chat behavior.
746        if let Some(tc) = &model.header.tokenizer_config {
747            tokenizer.chat_template = tc.chat_template.clone();
748            tokenizer.extra_eos.extend(tc.eos_token_ids.iter().copied());
749            if tokenizer.bos_token_id.is_none() {
750                tokenizer.bos_token_id = tc.bos_token_id;
751            }
752            tracing::info!(
753                "chat bundle: template {} chars, {} stop ids",
754                tc.chat_template.as_deref().map(str::len).unwrap_or(0),
755                tc.eos_token_ids.len()
756            );
757        }
758        // Gemma's contract requires <bos> at sequence start, but its
759        // tokenizer.json post-processor does not add it (the chat
760        // template does). Raw prompts need it too — word salad without.
761        if arch.arch_name.to_lowercase().contains("gemma") && tokenizer.bos_token_id.is_some() {
762            tokenizer.add_bos = true;
763        }
764
765        // ── Top-level weights (never masked → always quantized) ──
766        let embed_tokens = load_matrix(model, "model.embed_tokens.weight", false, ov)?;
767        let final_norm = if arch.qwen4_exp.is_some() {
768            // qwen4_exp folds its four streams through a learned global
769            // hyper mixer and has no conventional final RMSNorm.
770            vec![0.0; arch.hidden_size]
771        } else {
772            load_f32(model, "model.norm.weight", ov).map_err(err)?
773        };
774        let lm_head = if model.tensor("lm_head.weight").is_some() {
775            load_matrix(model, "lm_head.weight", false, ov)?
776        } else if arch.tie_word_embeddings {
777            // Tied: reuse the embedding matrix (re-open, cheap for Mapped).
778            load_matrix(model, "model.embed_tokens.weight", false, ov)?
779        } else {
780            return Err(CmfError::MissingTensor(
781                "lm_head.weight (and tie_word_embeddings is false)".into(),
782            ));
783        };
784
785        // ── Linear-core geometry (required if any linear layer exists) ──
786        let has_linear = arch
787            .layer_types
788            .iter()
789            .any(|t| matches!(t, LayerType::LinearAttention));
790        let mut vmf_cfg = None;
791        let mut gdn_cfg = None;
792        if has_linear {
793            let lc = arch.linear_core.as_ref().ok_or_else(|| {
794                CmfError::Parse(
795                    "model has LinearAttention layers but no arch.linear_core — \
796                     reconvert with the current converter"
797                        .into(),
798                )
799            })?;
800            let need = |v: Option<usize>, name: &str| {
801                v.ok_or_else(|| CmfError::Parse(format!("linear core needs arch.{name}")))
802            };
803            match lc.kind.as_str() {
804                "vmf_phase" => {
805                    vmf_cfg = Some(VmfPhaseCfg {
806                        num_heads: lc.num_heads,
807                        nphase: need(lc.nphase, "linear_core.nphase")?,
808                        value_head_dim: lc.value_head_dim,
809                        hidden_size: arch.hidden_size,
810                        // Phase-mass correction: default 0 (disabled); CMF_PHASE_MASS
811                        // widens the phase kernel for folded-unhealed models.
812                        phase_mass: std::env::var("CMF_PHASE_MASS")
813                            .ok()
814                            .and_then(|v| v.parse().ok())
815                            .unwrap_or(0.0),
816                    });
817                }
818                "gated_delta_net" => {
819                    gdn_cfg = Some(GdnCfg {
820                        num_v_heads: lc.num_heads,
821                        num_k_heads: need(arch.linear_num_key_heads, "linear_num_key_heads")?,
822                        key_head_dim: need(arch.linear_key_head_dim, "linear_key_head_dim")?,
823                        value_head_dim: lc.value_head_dim,
824                        conv_kernel: need(arch.linear_conv_kernel_dim, "linear_conv_kernel_dim")?,
825                        hidden_size: arch.hidden_size,
826                        rms_eps: arch.rms_norm_eps,
827                        output_gate_sigmoid: false,
828                    });
829                }
830                other => {
831                    return Err(CmfError::Parse(format!(
832                        "unknown linear core '{other}' (this runtime executes: \
833                         gated_delta_net, vmf_phase)"
834                    )));
835                }
836            }
837        }
838
839        // ── KDA geometry (Kimi Linear / Kimi-K3 delta-attention layers) ──
840        let has_kda = arch.layer_types.iter().any(|t| matches!(t, LayerType::Kda));
841        let kda_cfg = if has_kda {
842            let need = |v: Option<usize>, name: &str| {
843                v.ok_or_else(|| CmfError::Parse(format!("KDA core needs arch.{name}")))
844            };
845            Some(crate::linear_core::KdaCfg {
846                num_heads: need(arch.linear_num_key_heads, "linear_num_key_heads")?,
847                head_k_dim: need(arch.linear_key_head_dim, "linear_key_head_dim")?,
848                head_v_dim: need(arch.linear_value_head_dim, "linear_value_head_dim")?,
849                conv_kernel: need(arch.linear_conv_kernel_dim, "linear_conv_kernel_dim")?,
850                hidden_size: arch.hidden_size,
851                rms_eps: arch.rms_norm_eps,
852            })
853        } else {
854            None
855        };
856
857        // ── Short-convolution geometry (LFM2 conv mixer layers) ──
858        let has_short_conv = arch
859            .layer_types
860            .iter()
861            .any(|t| matches!(t, LayerType::ShortConv));
862        let short_conv_cfg = if has_short_conv {
863            Some(ShortConvCfg {
864                hidden_size: arch.hidden_size,
865                kernel: arch.linear_conv_kernel_dim.ok_or_else(|| {
866                    CmfError::Parse(
867                        "model has ShortConv layers but no arch.linear_conv_kernel_dim — \
868                         reconvert with the current converter"
869                            .into(),
870                    )
871                })?,
872            })
873        } else {
874            None
875        };
876
877        // ── Layers ──
878        let load_full_attn = |prefix: &str, layer: Option<usize>| -> Result<AttnKind, CmfError> {
879            let t = |suffix: &str| load_matrix(model, &format!("{prefix}{suffix}"), force_f32, ov);
880            let n = |suffix: &str| -> Option<Vec<f32>> {
881                model
882                    .tensor(&format!("{prefix}{suffix}"))
883                    .and_then(|_| load_f32(model, &format!("{prefix}{suffix}"), ov).ok())
884            };
885            // DeepSeek-V2 MLA: the latent projections replace the k/v pair.
886            if let Some(mla) = arch.mla.as_ref() {
887                // Compressed q (K3/V3): q_a → rms → q_b; direct otherwise.
888                let (q_proj, q_a, q_a_norm) = if mla.q_lora_rank.is_some() {
889                    (
890                        t("self_attn.q_b_proj.weight")?,
891                        Some(t("self_attn.q_a_proj.weight")?),
892                        Some(n("self_attn.q_a_layernorm.weight").ok_or_else(|| {
893                            CmfError::Parse(format!("{prefix}: MLA needs q_a_layernorm"))
894                        })?),
895                    )
896                } else {
897                    (t("self_attn.q_proj.weight")?, None, None)
898                };
899                let hd = mla.qk_rope_head_dim + mla.qk_nope_head_dim;
900                let nh = q_proj.rows() / hd;
901                // YaRN mscale²: DeepSeek corrects the softmax scale by
902                // (0.1·mscale_all_dim·ln(factor)+1)².
903                let mut scale = 1.0 / (hd as f32).sqrt();
904                if let Some(y) = arch.yarn.as_ref() {
905                    if let Some(m) = y.mscale_all_dim.filter(|&m| m > 0.0) {
906                        let ms = 0.1 * m * y.factor.ln() + 1.0;
907                        scale *= ms * ms;
908                    }
909                }
910                return Ok(AttnKind::Mla(Box::new(crate::pipeline::MlaWeights {
911                    q_proj,
912                    q_a,
913                    q_a_norm,
914                    kv_a: t("self_attn.kv_a_proj_with_mqa.weight")?,
915                    kv_a_norm: n("self_attn.kv_a_layernorm.weight").ok_or_else(|| {
916                        CmfError::Parse(format!("{prefix}: MLA needs kv_a_layernorm"))
917                    })?,
918                    kv_b: t("self_attn.kv_b_proj.weight")?,
919                    o_proj: t("self_attn.o_proj.weight")?,
920                    nh,
921                    qk_rope: mla.qk_rope_head_dim,
922                    qk_nope: mla.qk_nope_head_dim,
923                    v_dim: mla.v_head_dim,
924                    lora: mla.kv_lora_rank,
925                    scale,
926                    nope: mla.nope,
927                })));
928            }
929            let wq = t("self_attn.q_proj.weight")?;
930            let nh = layer
931                .and_then(|li| {
932                    arch.attention_heads_per_layer
933                        .as_ref()
934                        .and_then(|v| v.get(li).copied())
935                })
936                .unwrap_or(arch.num_attention_heads);
937            // Qwen3.5 output gate: q_proj rows = 2·nh·hd (per-head [q; gate]).
938            // Gemma-4 global layers legitimately have nh·global_head_dim
939            // rows (which can equal 2·nh·hd) — never gated.
940            let output_gate = arch.global_head_dim.is_none() && wq.rows() == 2 * nh * arch.head_dim;
941            // Gemma-4 global layers run MQA at global_head_dim — their
942            // q_proj legitimately carries nh·ghd rows.
943            let is_global_layer = arch.global_head_dim.is_some()
944                && layer.is_some_and(|li| {
945                    arch.sliding_window_pattern
946                        .is_some_and(|p| p > 0 && (li + 1) % p == 0)
947                });
948            let expect = if is_global_layer {
949                nh * arch.global_head_dim.unwrap_or(arch.head_dim)
950            } else {
951                nh * arch.head_dim
952            };
953            if !output_gate && wq.rows() != expect {
954                return Err(CmfError::Parse(format!(
955                    "{prefix}self_attn.q_proj.weight rows={} != heads({nh}) * head_dim({})",
956                    wq.rows(),
957                    expect / nh.max(1)
958                )));
959            }
960            let gate_name = format!("{prefix}self_attn.g_proj.weight");
961            let softplus_gate = if model.tensor(&gate_name).is_some() {
962                let gate = load_matrix(model, &gate_name, force_f32, ov)?;
963                if gate.cols() != arch.hidden_size {
964                    return Err(CmfError::Parse(format!(
965                        "{gate_name} cols={} != hidden_size ({})",
966                        gate.cols(),
967                        arch.hidden_size
968                    )));
969                }
970                let per_head = if gate.rows() == nh {
971                    true
972                } else if gate.rows() == nh * arch.head_dim {
973                    false
974                } else {
975                    return Err(CmfError::Parse(format!(
976                        "{gate_name} rows={} must equal heads ({nh}) or heads*head_dim ({})",
977                        gate.rows(),
978                        nh * arch.head_dim
979                    )));
980                };
981                Some((gate, per_head))
982            } else {
983                None
984            };
985            // Qwen2-family projection biases (by tensor presence).
986            let bias = match (
987                n("self_attn.q_proj.bias"),
988                n("self_attn.k_proj.bias"),
989                n("self_attn.v_proj.bias"),
990            ) {
991                (Some(a), Some(b), Some(c)) => Some((a, b, c)),
992                _ => None,
993            };
994            let wk = t("self_attn.k_proj.weight")?;
995            let wv = t("self_attn.v_proj.weight")?;
996            let wo = t("self_attn.o_proj.weight")?;
997            // K/V/O geometry, checked here and never at run time: a
998            // mapped matvec accepts any `out`/`x` at least as long as its
999            // rows/cols, so a V narrower than the cache's head or an
1000            // o_proj wider than nh·v would otherwise run silently wrong.
1001            // KV heads come from the header — per layer when the model
1002            // varies them (MiMo-V2: 4 full / 8 sliding), Gemma-4 global
1003            // layers carry their own (heads, width).
1004            let (nkv_l, hd_l) = if is_global_layer {
1005                (
1006                    arch.num_global_kv_heads.unwrap_or(arch.num_kv_heads),
1007                    arch.global_head_dim.unwrap_or(arch.head_dim),
1008                )
1009            } else {
1010                (
1011                    layer
1012                        .and_then(|li| {
1013                            arch.kv_heads_per_layer
1014                                .as_ref()
1015                                .and_then(|v| v.get(li).copied())
1016                        })
1017                        .unwrap_or(arch.num_kv_heads),
1018                    arch.head_dim,
1019                )
1020            };
1021            let vd = if is_global_layer {
1022                hd_l
1023            } else {
1024                arch.v_head_dim.unwrap_or(hd_l)
1025            };
1026            let shape_err = |what: String| {
1027                CmfError::Parse(format!(
1028                    "{prefix}self_attn: {what} (heads {nh}, kv heads {nkv_l}, head_dim {hd_l}, \
1029                     v_head_dim {vd}) — file and header disagree"
1030                ))
1031            };
1032            if nkv_l == 0 || nh % nkv_l != 0 {
1033                return Err(shape_err(format!(
1034                    "{nkv_l} KV heads must be nonzero and divide {nh} Q heads"
1035                )));
1036            }
1037            if vd == 0 || vd > hd_l {
1038                return Err(shape_err(format!("v_head_dim {vd} must be in 1..={hd_l}")));
1039            }
1040            if wk.rows() != nkv_l * hd_l {
1041                return Err(shape_err(format!(
1042                    "k_proj rows={} != kv_heads*head_dim={}",
1043                    wk.rows(),
1044                    nkv_l * hd_l
1045                )));
1046            }
1047            if wv.rows() != nkv_l * vd {
1048                return Err(shape_err(format!(
1049                    "v_proj rows={} != kv_heads*v_head_dim={}",
1050                    wv.rows(),
1051                    nkv_l * vd
1052                )));
1053            }
1054            if wo.cols() != nh * vd {
1055                return Err(shape_err(format!(
1056                    "o_proj cols={} != heads*v_head_dim={}",
1057                    wo.cols(),
1058                    nh * vd
1059                )));
1060            }
1061            if vd < hd_l && (output_gate || softplus_gate.is_some()) {
1062                return Err(shape_err(
1063                    "an attention output gate with V heads narrower than Q/K is not supported"
1064                        .into(),
1065                ));
1066            }
1067            Ok(AttnKind::Full {
1068                wq,
1069                wk,
1070                wv,
1071                wo,
1072                q_norm: n("self_attn.q_norm.weight"),
1073                k_norm: n("self_attn.k_norm.weight"),
1074                output_gate,
1075                softplus_gate,
1076                bias,
1077            })
1078        };
1079
1080        let load_linear_attn = |prefix: &str| -> Result<AttnKind, CmfError> {
1081            if gdn_cfg.is_some() {
1082                // Faithful vendor operator: tensor names 1:1 with the source.
1083                let t = |suffix: &str| {
1084                    load_matrix(
1085                        model,
1086                        &format!("{prefix}linear_attn.{suffix}"),
1087                        force_f32,
1088                        ov,
1089                    )
1090                };
1091                let f = |suffix: &str| {
1092                    load_f32(model, &format!("{prefix}linear_attn.{suffix}"), ov).map_err(err)
1093                };
1094                return Ok(AttnKind::LinearGdn(GdnWeights {
1095                    in_proj_qkv: t("in_proj_qkv.weight")?,
1096                    in_proj_z: t("in_proj_z.weight")?,
1097                    in_proj_a: t("in_proj_a.weight")?,
1098                    in_proj_b: t("in_proj_b.weight")?,
1099                    conv1d: f("conv1d.weight")?,
1100                    a_log: f("A_log")?,
1101                    dt_bias: f("dt_bias")?,
1102                    norm: f("norm.weight")?,
1103                    out_proj: t("out_proj.weight")?,
1104                }));
1105            }
1106            let t = |suffix: &str| {
1107                load_matrix(model, &format!("{prefix}vmf_attn.{suffix}"), force_f32, ov)
1108            };
1109            let a_log = load_f32(model, &format!("{prefix}vmf_attn.A_log"), ov).map_err(err)?;
1110            // Selective-write gate κ (hybrid_k core): optional by tensor
1111            // presence — files without it run the classic phase kernel
1112            // bit-identically.
1113            let k_gate = if model
1114                .tensor(&format!("{prefix}vmf_attn.k_gate.weight"))
1115                .is_some()
1116            {
1117                Some((
1118                    t("k_gate.weight")?,
1119                    load_f32(model, &format!("{prefix}vmf_attn.k_gate.bias"), ov).map_err(err)?,
1120                ))
1121            } else {
1122                None
1123            };
1124            // Optional short causal conv (embryo genomes born with
1125            // --conv-k): `[hidden, 1, k]` taps, flattened.
1126            let conv = if model
1127                .tensor(&format!("{prefix}vmf_attn.conv1d.weight"))
1128                .is_some()
1129            {
1130                Some(load_f32(model, &format!("{prefix}vmf_attn.conv1d.weight"), ov).map_err(err)?)
1131            } else {
1132                None
1133            };
1134            Ok(AttnKind::Linear(VmfPhaseWeights {
1135                thq: t("thq.weight")?,
1136                conv,
1137                thk: t("thk.weight")?,
1138                v_proj: t("v_proj.weight")?,
1139                out_proj: t("out_proj.weight")?,
1140                decay: a_log.iter().map(|&a| (-(a as f64).exp()).exp()).collect(),
1141                k_gate,
1142            }))
1143        };
1144
1145        // LFM2 short-conv mixer: in_proj [3·hidden, hidden], a depthwise
1146        // conv (stored f16 as `[hidden, 1, kernel]` → flattened taps), and
1147        // out_proj [hidden, hidden]. Names canonicalized at convert time.
1148        let load_short_conv = |prefix: &str| -> Result<AttnKind, CmfError> {
1149            let t = |suffix: &str| {
1150                load_matrix(
1151                    model,
1152                    &format!("{prefix}short_conv.{suffix}"),
1153                    force_f32,
1154                    ov,
1155                )
1156            };
1157            Ok(AttnKind::ShortConv(ShortConvWeights {
1158                in_proj: t("in_proj.weight")?,
1159                conv: load_f32(model, &format!("{prefix}short_conv.conv.weight"), ov)
1160                    .map_err(err)?,
1161                out_proj: t("out_proj.weight")?,
1162            }))
1163        };
1164
1165        // KDA layer (Kimi Linear / Kimi-K3): faithful vendor tensors under
1166        // the `kda_attn.` canonical prefix. The output gate is full-rank
1167        // (g_proj, K3) or low-rank (g_a/g_b, Kimi-Linear-48B) by presence.
1168        let load_kda = |prefix: &str| -> Result<AttnKind, CmfError> {
1169            let t = |suffix: &str| {
1170                load_matrix(model, &format!("{prefix}kda_attn.{suffix}"), force_f32, ov)
1171            };
1172            let f = |suffix: &str| {
1173                load_f32(model, &format!("{prefix}kda_attn.{suffix}"), ov).map_err(err)
1174            };
1175            let gate = if model
1176                .tensor(&format!("{prefix}kda_attn.g_proj.weight"))
1177                .is_some()
1178            {
1179                crate::linear_core::KdaOutGate::Full(t("g_proj.weight")?)
1180            } else {
1181                crate::linear_core::KdaOutGate::LowRank(
1182                    t("g_a_proj.weight")?,
1183                    t("g_b_proj.weight")?,
1184                )
1185            };
1186            Ok(AttnKind::Kda(Box::new(crate::linear_core::KdaWeights {
1187                q_proj: t("q_proj.weight")?,
1188                k_proj: t("k_proj.weight")?,
1189                v_proj: t("v_proj.weight")?,
1190                conv_q: f("q_conv1d.weight")?,
1191                conv_k: f("k_conv1d.weight")?,
1192                conv_v: f("v_conv1d.weight")?,
1193                f_a: t("f_a_proj.weight")?,
1194                f_b: t("f_b_proj.weight")?,
1195                dt_bias: f("dt_bias")?,
1196                a_log: f("A_log")?,
1197                b_proj: t("b_proj.weight")?,
1198                gate,
1199                o_norm: f("o_norm.weight")?,
1200                o_proj: t("o_proj.weight")?,
1201                gate_lower_bound: arch.kda_gate_lower_bound.map(|v| v as f32),
1202            })))
1203        };
1204
1205        fn anyhow_like(ok: bool) -> Result<(), ()> {
1206            if ok { Ok(()) } else { Err(()) }
1207        }
1208        let mut layers = Vec::with_capacity(arch.num_layers);
1209        let is_g3n = arch.g3n.is_some();
1210        // Architectures that load their own layer stack below. DeepSeek-V4
1211        // has none of the canonical projections — no q/k/v/o_proj, no
1212        // per-layer gate_proj — so the generic loop would demand
1213        // `self_attn.q_proj.weight` and fail before its own loader ever ran.
1214        let owns_its_layers = is_g3n
1215            || arch.arch_name == "deepseek_v4"
1216            || arch.arch_name == "deepseek_v41"
1217            || arch.qwen4_exp.is_some();
1218        for li in 0..(if owns_its_layers { 0 } else { arch.num_layers }) {
1219            let prefix = format!("model.layers.{li}.");
1220            let attn = match arch.layer_types.get(li) {
1221                Some(LayerType::LinearAttention) => load_linear_attn(&prefix)?,
1222                Some(LayerType::Kda) => load_kda(&prefix)?,
1223                Some(LayerType::ShortConv) => load_short_conv(&prefix)?,
1224                _ => load_full_attn(&prefix, Some(li))?,
1225            };
1226            // Gemma-2/3 sandwich: `pre_feedforward_layernorm` present →
1227            // it is the pre-FFN norm, and post_attention/post_feedforward
1228            // норms apply to the branch OUTPUTS before their residuals.
1229            let pre_ffn = format!("{prefix}pre_feedforward_layernorm.weight");
1230            let sandwich = model.tensor(&pre_ffn).is_some();
1231            layers.push(LayerWeights {
1232                input_norm: load_f32(model, &format!("{prefix}input_layernorm.weight"), ov)
1233                    .map_err(err)?,
1234                post_norm: if sandwich {
1235                    load_f32(model, &pre_ffn, ov).map_err(err)?
1236                } else {
1237                    load_f32(
1238                        model,
1239                        &format!("{prefix}post_attention_layernorm.weight"),
1240                        ov,
1241                    )
1242                    .map_err(err)?
1243                },
1244                attn_out_norm: if sandwich {
1245                    Some(
1246                        load_f32(
1247                            model,
1248                            &format!("{prefix}post_attention_layernorm.weight"),
1249                            ov,
1250                        )
1251                        .map_err(err)?,
1252                    )
1253                } else {
1254                    None
1255                },
1256                ffn_out_norm: if sandwich {
1257                    Some(
1258                        load_f32(
1259                            model,
1260                            &format!("{prefix}post_feedforward_layernorm.weight"),
1261                            ov,
1262                        )
1263                        .map_err(err)?,
1264                    )
1265                } else {
1266                    None
1267                },
1268                // Gemma-4: learned scalar multiplying the layer output.
1269                layer_scale: model
1270                    .tensor(&format!("{prefix}layer_scalar"))
1271                    .and_then(|_| {
1272                        load_f32(model, &format!("{prefix}layer_scalar"), ov)
1273                            .ok()
1274                            .and_then(|v| v.first().copied())
1275                    }),
1276                // FFN always quantized — masks run sparse on quant bytes.
1277                ffn: build_layer_ffn(model, &arch, li, false, ov)?,
1278                attn,
1279            });
1280        }
1281
1282        // ── MTP head (optional, spec §2.1) ──
1283        //
1284        // The header declaring an MTP head is not the same as the file
1285        // carrying one. DeepSeek-V4's config announces a next-token predictor
1286        // whose weights the converter does not map (they are spelled `mtp.N.*`
1287        // and have none of the canonical projections), so demanding
1288        // `model.mtp.layers.0.self_attn.q_proj.weight` failed a model that is
1289        // otherwise complete. Presence in the directory decides.
1290        let mtp_present = model
1291            .tensor("model.mtp.layers.0.self_attn.q_proj.weight")
1292            .is_some()
1293            || model.tensor("model.mtp.eh_proj.weight").is_some();
1294        // DeepSeek-V4 writes its own stack under `model.mtp.N.*` — three full
1295        // layers, not a V3-style single block — so it cannot go through the
1296        // path below and is loaded by the dsv4 arm instead. Saying the file
1297        // "carries none" was a false negative worth six gigabytes.
1298        let dsv4_mtp = model.tensor("model.mtp.0.main_proj.weight").is_some();
1299        if arch.mtp.is_some() && !mtp_present && !dsv4_mtp {
1300            tracing::info!(
1301                "header declares an MTP head but the file carries none — \
1302                 loading without it"
1303            );
1304        }
1305        // MiMo-V2 carries its own three-layer stack (sidecar or main file),
1306        // loaded by `mimo_mtp::load_for` below — never this single block.
1307        let mtp = if let Some(cfg) = arch
1308            .mtp
1309            .as_ref()
1310            .filter(|_| mtp_present && arch.arch_name != "mimo_v2")
1311        {
1312            if cfg.num_layers != 1 {
1313                return Err(CmfError::Parse(format!(
1314                    "MTP with {} blocks not supported yet (only 1)",
1315                    cfg.num_layers
1316                )));
1317            }
1318            let p = "model.mtp.";
1319            let attn = load_full_attn("model.mtp.layers.0.", None)?;
1320            Some(MtpModule {
1321                enorm: load_f32(model, &format!("{p}enorm.weight"), ov).map_err(err)?,
1322                hnorm: load_f32(model, &format!("{p}hnorm.weight"), ov).map_err(err)?,
1323                eh_proj: load_matrix(model, &format!("{p}eh_proj.weight"), false, ov)?,
1324                layer: LayerWeights {
1325                    attn_out_norm: None,
1326                    ffn_out_norm: None,
1327                    layer_scale: None,
1328                    input_norm: load_f32(model, &format!("{p}layers.0.input_layernorm.weight"), ov)
1329                        .map_err(err)?,
1330                    post_norm: load_f32(
1331                        model,
1332                        &format!("{p}layers.0.post_attention_layernorm.weight"),
1333                        ov,
1334                    )
1335                    .map_err(err)?,
1336                    // Whatever the block actually carries: DeepSeek's MTP
1337                    // layer is dense, Qwen3.6's is a full MoE (router + 256
1338                    // experts + shared). Same builder as a backbone layer.
1339                    ffn: build_ffn_at(model, &arch, &format!("{p}layers.0."), false, ov)?,
1340                    attn,
1341                },
1342                final_norm: load_f32(model, &format!("{p}norm.weight"), ov).map_err(err)?,
1343                kv: LayerKvCache::new(arch.num_kv_heads, arch.head_dim),
1344            })
1345        } else {
1346            None
1347        };
1348
1349        tracing::info!(
1350            "Pipeline loaded: {} | {}L ({} linear) | {:.2}B params | storage: {} | MTP: {}",
1351            arch.arch_name,
1352            arch.num_layers,
1353            arch.layer_types
1354                .iter()
1355                .filter(|t| matches!(t, LayerType::LinearAttention))
1356                .count(),
1357            model.total_param_count() as f64 / 1e9,
1358            if force_f32 {
1359                "f32 (masked)"
1360            } else {
1361                "quantized mmap"
1362            },
1363            if mtp.is_some() { "yes" } else { "no" }
1364        );
1365
1366        // KV window: the descriptor's max, capped for dev-box safety;
1367        // CMF_MAX_SEQ overrides the cap (long-context runs).
1368        // 8192 was the silent quality cliff of the Qwen3.8 bring-up: at
1369        // the cap the wgpu token graph declines, the host evicts half the
1370        // KV, and a GDN hybrid's recurrent state goes stale — the model
1371        // stays fluent and loses its mind (Django internals, a Turkish
1372        // essay, an em-dash loop; one failure, three costumes). 32768
1373        // covers every long-form run we actually ship while keeping the
1374        // graph's device KV mirror affordable beside the weights;
1375        // CMF_MAX_SEQ still overrides in either direction.
1376        let cap = std::env::var("CMF_MAX_SEQ")
1377            .ok()
1378            .and_then(|v| v.parse::<usize>().ok())
1379            .unwrap_or(32_768);
1380        let max_seq_len = arch.max_position_embeddings.min(cap);
1381
1382        // Looped Transformer: total virtual layers = physical × num_loops.
1383        let total_layers = arch.num_layers * arch.num_loops;
1384
1385        let mut pipeline = Pipeline::new(
1386            tokenizer,
1387            PipelineWeights {
1388                embed_tokens,
1389                layers,
1390                lm_head,
1391                final_norm,
1392            },
1393            arch.hidden_size,
1394            arch.intermediate_size,
1395            arch.num_attention_heads,
1396            arch.num_kv_heads,
1397            arch.head_dim,
1398            total_layers,
1399            arch.num_layers, // physical layers in weights
1400            arch.loop_final_norm,
1401            arch.vocab_size,
1402            arch.rms_norm_eps,
1403            arch.rope_theta as f32,
1404            arch.norm_style,
1405            max_seq_len,
1406            sampler_config,
1407        );
1408        let rotary = ((arch.head_dim as f32 * arch.partial_rotary_factor) as usize).max(2);
1409        pipeline.set_rotary(rotary, arch.rope_theta as f32);
1410        pipeline.attention_heads_per_layer = arch.attention_heads_per_layer.clone();
1411        if let Some(yarn) = &arch.yarn {
1412            pipeline.inv_freq = std::sync::Arc::new(crate::attention::yarn_inv_freq(
1413                rotary,
1414                arch.rope_theta as f32,
1415                yarn.factor,
1416                yarn.original_max_position_embeddings,
1417                yarn.beta_fast,
1418                yarn.beta_slow,
1419            ));
1420            pipeline.rope_scale = yarn.attention_factor;
1421        }
1422        // Gemma-family extras: embedding scale, attention-scale
1423        // override, and (Gemma-3) sliding-window layers with their own
1424        // local RoPE base.
1425        pipeline.embed_multiplier = arch.embed_multiplier;
1426        pipeline.logit_multiplier = arch.logit_multiplier;
1427        if let Some(qpas) = arch.query_pre_attn_scalar {
1428            pipeline.attn_scale = 1.0 / (qpas as f32).sqrt();
1429        }
1430        if let (Some(w), Some(p)) = (arch.sliding_window, arch.sliding_window_pattern) {
1431            pipeline.swa = Some((w, p));
1432            if let Some(base) = arch.rope_local_base_freq {
1433                pipeline.inv_freq_local = Some(std::sync::Arc::new(
1434                    crate::attention::rope_inv_freq(rotary, base as f32),
1435                ));
1436            }
1437        }
1438        let explicit_sliding: Vec<bool> = arch
1439            .layer_types
1440            .iter()
1441            .map(|t| matches!(t, cortiq_core::LayerType::SlidingAttention))
1442            .collect();
1443        if explicit_sliding.iter().any(|&v| v) {
1444            pipeline.sliding_layers = Some(explicit_sliding);
1445            if let Some(w) = arch.sliding_window {
1446                pipeline.swa = Some((w, usize::MAX));
1447            }
1448            let local_rotary = ((arch.head_dim as f32
1449                * arch
1450                    .local_partial_rotary_factor
1451                    .unwrap_or(arch.partial_rotary_factor))
1452                as usize)
1453                .max(2);
1454            pipeline.rotary_dim_local = Some(local_rotary);
1455            if let Some(base) = arch.rope_local_base_freq {
1456                pipeline.inv_freq_local = Some(std::sync::Arc::new(
1457                    crate::attention::rope_inv_freq(local_rotary, base as f32),
1458                ));
1459            }
1460        }
1461        // Gemma-4: global layers run their own geometry (MQA at
1462        // global_head_dim) with a proportional RoPE — the first
1463        // factor·head_dim dims rotate, the zero-padded tail is identity.
1464        if let (Some(ghd), Some(gkv)) = (arch.global_head_dim, arch.num_global_kv_heads) {
1465            pipeline.global_attn = Some((ghd, gkv));
1466            let prf = arch.global_partial_rotary_factor.unwrap_or(1.0);
1467            let half = ghd / 2;
1468            let ra = (((prf * ghd as f32) as usize) / 2).min(half);
1469            let mut f = vec![0.0f32; half];
1470            for (i, slot) in f.iter_mut().enumerate().take(ra) {
1471                *slot = 1.0 / (arch.rope_theta as f32).powf(2.0 * i as f32 / ghd as f32);
1472            }
1473            pipeline.inv_freq_global = Some(std::sync::Arc::new(f));
1474            // Re-shape the global layers' KV storage to their geometry.
1475            // An explicit layer_types map wins over the numeric pattern
1476            // (explicit tags set swa's pattern to usize::MAX, which
1477            // would otherwise leave every global cache mis-shaped).
1478            let global_at = |li: usize| -> bool {
1479                match &pipeline.sliding_layers {
1480                    Some(map) => !map.get(li).copied().unwrap_or(false),
1481                    None => pipeline
1482                        .swa
1483                        .map(|(_, p)| p > 0 && p != usize::MAX && (li + 1) % p == 0)
1484                        .unwrap_or(false),
1485                }
1486            };
1487            for li in 0..arch.num_layers {
1488                if global_at(li) {
1489                    pipeline.kv_cache.layers[li] = crate::kv_cache::LayerKvCache::new(gkv, ghd);
1490                }
1491            }
1492        }
1493        // MLA (DeepSeek-V2): the expand-to-MHA cache holds nh heads of
1494        // rope+nope dims; rotary covers the rope prefix.
1495        if let Some(mla) = arch.mla.as_ref() {
1496            let hd = mla.qk_rope_head_dim + mla.qk_nope_head_dim;
1497            pipeline.head_dim = hd;
1498            pipeline.num_kv_heads = arch.num_attention_heads;
1499            pipeline.rotary_dim = mla.qk_rope_head_dim;
1500            let half = mla.qk_rope_head_dim / 2;
1501            let mut f = vec![0.0f32; half];
1502            for (i, slot) in f.iter_mut().enumerate() {
1503                *slot = 1.0
1504                    / (arch.rope_theta as f32).powf(2.0 * i as f32 / mla.qk_rope_head_dim as f32);
1505            }
1506            pipeline.inv_freq = std::sync::Arc::new(f);
1507            for li in 0..arch.num_layers {
1508                pipeline.kv_cache.layers[li] =
1509                    crate::kv_cache::LayerKvCache::new(arch.num_attention_heads, hd);
1510            }
1511        }
1512        // Per-layer KV geometry (MiMo-V2): KV heads per layer and V heads
1513        // narrower than Q/K. The per-layer shapes were checked against the
1514        // header in load_full_attn; this installs the geometry and gives
1515        // every layer whose KV head count differs its own cache shape.
1516        if !owns_its_layers {
1517            pipeline
1518                .set_attn_geometry(arch.kv_heads_per_layer.clone(), arch.v_head_dim)
1519                .map_err(|e| CmfError::Parse(format!("attention geometry: {e}")))?;
1520        } else if arch.kv_heads_per_layer.is_some() || arch.v_head_dim.is_some() {
1521            return Err(CmfError::Parse(format!(
1522                "{}: kv_heads_per_layer / v_head_dim are not supported by its own layer stack",
1523                arch.arch_name
1524            )));
1525        }
1526        // Learned attention sinks (gpt-oss / MiMo-V2 SWA layers), by tensor
1527        // presence: `model.layers.N.self_attn.sinks`, one f32 logit per Q
1528        // head (`attention_sink_bias` is the HF spelling, accepted too).
1529        // They fold into the grouped softmax; every attention path that
1530        // cannot carry them (chunk GEMM attend, O(1), the GPU graphs)
1531        // refuses the layer.
1532        if !owns_its_layers {
1533            for li in 0..arch.num_layers {
1534                let name = [
1535                    format!("model.layers.{li}.self_attn.sinks"),
1536                    format!("model.layers.{li}.self_attn.attention_sink_bias"),
1537                ]
1538                .into_iter()
1539                .find(|n| model.tensor(n).is_some());
1540                if let Some(name) = name {
1541                    let sinks = load_f32(model, &name, ov).map_err(err)?;
1542                    pipeline
1543                        .set_layer_sinks(li, sinks)
1544                        .map_err(|e| CmfError::Parse(format!("{name}: {e}")))?;
1545                }
1546            }
1547        }
1548        if pipeline.kv_heads_per_layer.is_some()
1549            || pipeline.v_head_dim.is_some()
1550            || pipeline.kv_cache.layers.iter().any(|l| l.sinks.is_some())
1551        {
1552            let per_token: usize = pipeline
1553                .kv_cache
1554                .layers
1555                .iter()
1556                .map(|l| 2 * l.num_kv_heads * l.head_dim * std::mem::size_of::<f32>())
1557                .sum();
1558            tracing::info!(
1559                "attention geometry: kv heads per layer {:?}, v_head_dim {:?}, {} sink layer(s); \
1560                 f32 KV {} B/token (V padded to head_dim), cap {} tokens",
1561                pipeline.kv_heads_per_layer,
1562                pipeline.v_head_dim,
1563                pipeline
1564                    .kv_cache
1565                    .layers
1566                    .iter()
1567                    .filter(|l| l.sinks.is_some())
1568                    .count(),
1569                per_token,
1570                pipeline.kv_cache.max_seq_len
1571            );
1572        }
1573        // Per-frequency rope divisors (MiniCPM3 longrope short_factor):
1574        // served at the native window with the trained per-dim factors.
1575        // Applied after every inv_freq build (plain, YaRN, MLA).
1576        if let Some(fac) = &arch.rope_freq_factors {
1577            let mut f = pipeline.inv_freq.as_ref().clone();
1578            for (i, v) in f.iter_mut().enumerate() {
1579                if let Some(&d) = fac.get(i) {
1580                    *v /= d as f32;
1581                }
1582            }
1583            pipeline.inv_freq = std::sync::Arc::new(f);
1584        }
1585        pipeline.attn_v_norm = arch.attn_v_norm;
1586        pipeline.qk_norm_after_rope = arch.qk_norm_after_rope;
1587        pipeline.final_softcap = arch.final_logit_softcapping.map(|c| c as f32);
1588        // Cortiq Embryo hierarchical head: cluster matrix → two-level log-probs.
1589        if let Some(ncl) = arch.head_clusters {
1590            let cm = load_f32(model, "lm_head.clusters.weight", ov).map_err(err)?;
1591            if cm.len() != ncl * arch.hidden_size {
1592                return Err(CmfError::Parse(format!(
1593                    "lm_head.clusters.weight: {} != {ncl}×{}",
1594                    cm.len(),
1595                    arch.hidden_size
1596                )));
1597            }
1598            pipeline.head_clusters = Some(std::sync::Arc::new(cm));
1599        }
1600        pipeline.attn_softcap = arch.attn_logit_softcapping.unwrap_or(0.0) as f32;
1601        pipeline.vmf_cfg = vmf_cfg;
1602        pipeline.gdn_cfg = gdn_cfg;
1603        pipeline.kda_cfg = kda_cfg;
1604        if arch.qwen4_exp.is_some() {
1605            let (globals, layers, cfg, state) = crate::qwen4_exp::load(model, &arch)?;
1606            pipeline.qwen4_exp = Some(Box::new((globals, layers, cfg, state)));
1607        }
1608        if let Some(gc) = arch.g3n.as_ref() {
1609            use crate::g3n::{G3nAltUp, G3nGlobals, G3nLaurel, G3nLayer};
1610            anyhow_like(gc.altup_num_inputs == crate::g3n::ALTUP_N).map_err(|_| {
1611                CmfError::Parse(format!(
1612                    "g3n: altup_num_inputs {} != supported {}",
1613                    gc.altup_num_inputs,
1614                    crate::g3n::ALTUP_N
1615                ))
1616            })?;
1617            let t = |name: &str| load_matrix(model, name, force_f32, ov);
1618            let f = |name: &str| load_f32(model, name, ov).map_err(err);
1619            let mut altup_proj = Vec::new();
1620            let mut altup_unembed = Vec::new();
1621            for i in 0..crate::g3n::ALTUP_N - 1 {
1622                altup_proj.push(t(&format!("model.altup_projections.{i}.weight"))?);
1623                altup_unembed.push(t(&format!("model.altup_unembed_projections.{i}.weight"))?);
1624            }
1625            let first_shared = arch.num_layers.saturating_sub(gc.num_kv_shared_layers);
1626            let sliding_of = |li: usize| {
1627                matches!(
1628                    arch.layer_types.get(li),
1629                    Some(cortiq_core::LayerType::SlidingAttention)
1630                )
1631            };
1632            let mut g3n_layers = Vec::with_capacity(arch.num_layers);
1633            for li in 0..arch.num_layers {
1634                let pfx = format!("model.layers.{li}.");
1635                let shared = li >= first_shared && first_shared > 0;
1636                let share_src = if shared {
1637                    let want = sliding_of(li);
1638                    (0..first_shared).rev().find(|&j| sliding_of(j) == want)
1639                } else {
1640                    None
1641                };
1642                g3n_layers.push(G3nLayer {
1643                    altup: G3nAltUp {
1644                        router_norm: f(&format!("{pfx}altup.router_norm.weight"))?,
1645                        modality_router: t(&format!("{pfx}altup.modality_router.weight"))?,
1646                        prediction_coefs: t(&format!("{pfx}altup.prediction_coefs.weight"))?,
1647                        correction_coefs: t(&format!("{pfx}altup.correction_coefs.weight"))?,
1648                        correct_output_scale: f(&format!("{pfx}altup.correct_output_scale"))?,
1649                    },
1650                    laurel: G3nLaurel {
1651                        left: t(&format!("{pfx}laurel.linear_left.weight"))?,
1652                        right: t(&format!("{pfx}laurel.linear_right.weight"))?,
1653                        post_norm: f(&format!("{pfx}laurel.post_laurel_norm.weight"))?,
1654                    },
1655                    input_norm: f(&format!("{pfx}input_layernorm.weight"))?,
1656                    post_attn_norm: f(&format!("{pfx}post_attention_layernorm.weight"))?,
1657                    pre_ffw_norm: f(&format!("{pfx}pre_feedforward_layernorm.weight"))?,
1658                    post_ffw_norm: f(&format!("{pfx}post_feedforward_layernorm.weight"))?,
1659                    wq: t(&format!("{pfx}self_attn.q_proj.weight"))?,
1660                    wk: if shared {
1661                        None
1662                    } else {
1663                        Some(t(&format!("{pfx}self_attn.k_proj.weight"))?)
1664                    },
1665                    wv: if shared {
1666                        None
1667                    } else {
1668                        Some(t(&format!("{pfx}self_attn.v_proj.weight"))?)
1669                    },
1670                    wo: t(&format!("{pfx}self_attn.o_proj.weight"))?,
1671                    q_norm: f(&format!("{pfx}self_attn.q_norm.weight"))?,
1672                    k_norm: if shared {
1673                        None
1674                    } else {
1675                        Some(f(&format!("{pfx}self_attn.k_norm.weight"))?)
1676                    },
1677                    kv_share_src: share_src,
1678                    sliding: sliding_of(li),
1679                    gate: t(&format!("{pfx}mlp.gate_proj.weight"))?,
1680                    up: t(&format!("{pfx}mlp.up_proj.weight"))?,
1681                    down: t(&format!("{pfx}mlp.down_proj.weight"))?,
1682                    sparsity: gc.activation_sparsity.get(li).copied().unwrap_or(0.0),
1683                    ple_gate: t(&format!("{pfx}per_layer_input_gate.weight"))?,
1684                    ple_proj: t(&format!("{pfx}per_layer_projection.weight"))?,
1685                    post_ple_norm: f(&format!("{pfx}post_per_layer_input_norm.weight"))?,
1686                });
1687            }
1688            let hd = arch.head_dim;
1689            let globals = G3nGlobals {
1690                altup_proj,
1691                altup_unembed,
1692                ple_embed: t("model.embed_tokens_per_layer.weight")?,
1693                ple_model_proj: t("model.per_layer_model_projection.weight")?,
1694                ple_norm: f("model.per_layer_projection_norm.weight")?,
1695                ple_vocab: gc.ple_vocab,
1696                ple_dim: gc.ple_dim,
1697                num_layers: arch.num_layers,
1698                hidden: arch.hidden_size,
1699                rms_eps: arch.rms_norm_eps,
1700                inv_freq_local: crate::attention::rope_inv_freq(
1701                    hd,
1702                    arch.rope_local_base_freq.unwrap_or(10_000.0) as f32,
1703                ),
1704                inv_freq_global: crate::attention::rope_inv_freq(hd, arch.rope_theta as f32),
1705                window: arch.sliding_window.unwrap_or(512),
1706            };
1707            pipeline.g3n = Some(Box::new((globals, g3n_layers)));
1708        }
1709        // DeepSeek-V4.1: its own stack, selected by the preserved source
1710        // configuration. The generic layer loop cannot represent shared
1711        // CED/CSA2 state or native Engram tables, so loading failure is
1712        // fatal instead of falling back to an unrelated attention layout.
1713        if arch.arch_name == "deepseek_v41" {
1714            let source = arch.deepseek_v41.as_ref().ok_or_else(|| {
1715                CmfError::Parse("deepseek_v41: missing preserved source config".into())
1716            })?;
1717            let tc = source.get("text_config").unwrap_or(source);
1718            let usize_of = |key: &str, fallback: usize| {
1719                tc.get(key)
1720                    .and_then(|v| v.as_u64())
1721                    .map(|v| v as usize)
1722                    .unwrap_or(fallback)
1723            };
1724            let f32_of = |key: &str, fallback: f32| {
1725                tc.get(key)
1726                    .and_then(|v| v.as_f64())
1727                    .map(|v| v as f32)
1728                    .unwrap_or(fallback)
1729            };
1730            let usize_any = |keys: &[&str], fallback: usize| {
1731                keys.iter()
1732                    .find_map(|key| tc.get(key).and_then(|v| v.as_u64()))
1733                    .map(|v| v as usize)
1734                    .unwrap_or(fallback)
1735            };
1736            let f32_any = |keys: &[&str], fallback: f32| {
1737                keys.iter()
1738                    .find_map(|key| tc.get(key).and_then(|v| v.as_f64()))
1739                    .map(|v| v as f32)
1740                    .unwrap_or(fallback)
1741            };
1742            let bool_of = |key: &str, fallback: bool| {
1743                tc.get(key).and_then(|v| v.as_bool()).unwrap_or(fallback)
1744            };
1745            let array_of = |key: &str| -> Vec<usize> {
1746                tc.get(key)
1747                    .and_then(|v| v.as_array())
1748                    .map(|a| {
1749                        a.iter()
1750                            .filter_map(|v| v.as_u64().map(|x| x as usize))
1751                            .collect()
1752                    })
1753                    .unwrap_or_default()
1754            };
1755            let array_alias = |keys: &[&str]| -> Vec<usize> {
1756                keys.iter()
1757                    .find_map(|key| {
1758                        let values = array_of(key);
1759                        (!values.is_empty()).then_some(values)
1760                    })
1761                    .unwrap_or_default()
1762            };
1763            let dim = usize_any(&["hidden_size", "dim"], arch.hidden_size);
1764            let n_layers = usize_any(&["num_hidden_layers", "n_layers"], arch.num_layers);
1765            let n_heads = usize_any(
1766                &["num_attention_heads", "n_heads"],
1767                arch.num_attention_heads,
1768            );
1769            let head_dim = usize_any(&["head_dim"], arch.head_dim);
1770            let rope_head_dim = usize_any(&["rope_head_dim", "qk_rope_head_dim"], 64.min(head_dim));
1771            let moe_inter = usize_any(
1772                &[
1773                    "moe_intermediate_size",
1774                    "moe_inter_dim",
1775                    "intermediate_size",
1776                ],
1777                arch.intermediate_size,
1778            );
1779            let n_experts = usize_any(
1780                &["n_routed_experts"],
1781                arch.moe.as_ref().map(|m| m.num_experts).unwrap_or(384),
1782            );
1783            let top_k = usize_any(
1784                &["num_experts_per_tok", "n_activated_experts"],
1785                arch.moe.as_ref().map(|m| m.top_k).unwrap_or(6),
1786            );
1787            let mut ratios = array_of("compress_ratios");
1788            if ratios.len() >= n_layers {
1789                ratios.truncate(n_layers);
1790            } else {
1791                ratios = (0..n_layers)
1792                    .map(|li| {
1793                        if (2..20).contains(&li) {
1794                            2
1795                        } else if (20..40).contains(&li) {
1796                            1
1797                        } else {
1798                            0
1799                        }
1800                    })
1801                    .collect();
1802            }
1803            let kv_sources = {
1804                let a = array_alias(&["kv_source_layers", "kv_source_layer_ids"]);
1805                if a.is_empty() { vec![2, 8, 14, 20] } else { a }
1806            };
1807            let index_sources = {
1808                let a = array_alias(&["index_source_layers", "index_source_layer_ids"]);
1809                if a.is_empty() {
1810                    vec![2, 8, 14, 20, 24, 28, 32, 36]
1811                } else {
1812                    a
1813                }
1814            };
1815            let engram_layers = array_of("engram_layer_ids");
1816            let engram_embeddings = array_of("engram_num_embeddings");
1817            let cfg = crate::dsv41::Dsv41Cfg {
1818                dim,
1819                n_heads,
1820                head_dim,
1821                rope_head_dim: rope_head_dim.min(head_dim) & !1,
1822                q_lora_rank: usize_of("q_lora_rank", 1280),
1823                o_lora_rank: usize_of("o_lora_rank", 1024),
1824                o_groups: usize_of("o_groups", 8),
1825                hc_mult: usize_of("hc_mult", 4),
1826                hc_sinkhorn_iters: usize_of("hc_sinkhorn_iters", 20),
1827                hc_eps: f32_of("hc_eps", 1e-6),
1828                norm_eps: f32_any(&["norm_eps", "rms_norm_eps"], arch.rms_norm_eps as f32),
1829                n_routed_experts: n_experts,
1830                top_k,
1831                moe_inter,
1832                gate_temp: f32_of("gate_temp", 1.0),
1833                norm_topk_prob: bool_of("norm_topk_prob", true),
1834                route_scale: f32_any(
1835                    &["routed_scaling_factor", "route_scale"],
1836                    arch.moe
1837                        .as_ref()
1838                        .and_then(|m| m.routed_scaling_factor)
1839                        .unwrap_or(1.5),
1840                ),
1841                swiglu_limit: f32_of("swiglu_limit", 10.0),
1842                window: usize_any(
1843                    &["window_size", "sliding_window"],
1844                    arch.sliding_window.unwrap_or(128),
1845                ),
1846                rope_theta: f32_any(&["rope_theta"], arch.rope_theta as f32),
1847                compress_rope_theta: f32_any(&["compress_rope_theta"], 160_000.0),
1848                rope_factor: f32_any(
1849                    &["rope_factor"],
1850                    tc.get("rope_scaling")
1851                        .and_then(|v| v.get("factor"))
1852                        .and_then(|v| v.as_f64())
1853                        .map(|v| v as f32)
1854                        .unwrap_or(16.0),
1855                ),
1856                original_seq_len: usize_any(
1857                    &["original_seq_len", "original_max_position_embeddings"],
1858                    tc.get("rope_scaling")
1859                        .and_then(|v| v.get("original_max_position_embeddings"))
1860                        .and_then(|v| v.as_u64())
1861                        .map(|v| v as usize)
1862                        .unwrap_or(65_536),
1863                ),
1864                beta_fast: f32_any(
1865                    &["beta_fast"],
1866                    tc.get("rope_scaling")
1867                        .and_then(|v| v.get("beta_fast"))
1868                        .and_then(|v| v.as_f64())
1869                        .map(|v| v as f32)
1870                        .unwrap_or(32.0),
1871                ),
1872                beta_slow: f32_any(
1873                    &["beta_slow"],
1874                    tc.get("rope_scaling")
1875                        .and_then(|v| v.get("beta_slow"))
1876                        .and_then(|v| v.as_f64())
1877                        .map(|v| v as f32)
1878                        .unwrap_or(1.0),
1879                ),
1880                index_heads: usize_any(&["index_n_heads", "indexer_n_heads"], 32),
1881                index_head_dim: usize_any(&["index_head_dim", "indexer_head_dim"], 128),
1882                index_topk: usize_any(&["index_topk", "indexer_topk"], 512),
1883                candidate_source: usize_any(
1884                    &["candidate_source_layer", "candidate_source_layer_id"],
1885                    20,
1886                ),
1887                candidate_topk_blocks: usize_any(&["candidate_topk_blocks"], 2048),
1888                candidate_block_size: usize_any(&["candidate_block_size"], 8),
1889                kv_sources,
1890                index_sources,
1891                compress_ratios: ratios,
1892                engram_layers,
1893                engram_vocab: usize_any(&["engram_vocab_size"], 16_000_000),
1894                engram_embeddings,
1895                engram_max_ngram: usize_any(&["engram_max_ngram_size"], 4),
1896                engram_heads: usize_any(&["engram_n_heads"], 8),
1897                engram_head_dim: usize_any(&["engram_head_dim"], 256),
1898                engram_compressed_vocab: usize_any(&["engram_compressed_vocab_size"], 99_092),
1899                engram_pad_id: usize_any(&["engram_pad_id", "engram_pad_token_id"], 2),
1900                vocab: usize_any(&["vocab_size"], arch.vocab_size),
1901            };
1902            let token_map = crate::dsv41::token_map_from_model(model, cfg.vocab);
1903            let (g, dl, hash) = crate::dsv41::load(model, &cfg, n_layers, token_map)
1904                .map_err(|e| CmfError::Parse(format!("deepseek_v41: {e}")))?;
1905            let st = crate::dsv41::Dsv41State::new(&cfg, hash);
1906            // Vision tensors are optional in text-only exports, but when
1907            // present they stay mmap-backed through the dedicated tower.
1908            // Do not make a text-only CMF fail merely because its source
1909            // config still carries the multimodal section.
1910            if let Ok(vision_cfg) = crate::dsv41_vision::VisionConfig::from_source(source) {
1911                if vision_cfg.vision_enabled()
1912                    && model.tensor("vision.patch_embed.proj.weight").is_some()
1913                {
1914                    pipeline.dsv41_vision = Some(
1915                        crate::dsv41_vision::VisionModel::from_model(model, vision_cfg)
1916                            .map_err(|e| CmfError::Parse(format!("deepseek_v41 vision: {e}")))?,
1917                    );
1918                }
1919            }
1920            tracing::info!(
1921                "deepseek_v41: loaded {} layers, {} KV sources, {} index sources, {} Engram layers; experts remain mmap-backed",
1922                dl.len(),
1923                cfg.kv_sources.len(),
1924                cfg.index_sources.len(),
1925                cfg.engram_layers.len()
1926            );
1927            pipeline.dsv41 = Some(Box::new((g, dl, cfg, st)));
1928        }
1929        // DeepSeek-V4: its own stack, selected by the arch name the
1930        // converter wrote. Loading failure is fatal rather than a silent
1931        // fallback — the generic loop cannot represent this model at all,
1932        // so a fallback would decode noise.
1933        if arch.arch_name == "deepseek_v4" {
1934            let moe = arch
1935                .moe
1936                .as_ref()
1937                .ok_or_else(|| CmfError::Parse("deepseek_v4: no moe config".into()))?;
1938            let cfg = crate::dsv4::Dsv4Cfg {
1939                dim: arch.hidden_size,
1940                n_heads: arch.num_attention_heads,
1941                head_dim: arch.head_dim,
1942                // The rope tail: `partial_rotary_factor` carries it when the
1943                // conversion recorded it (rd/head_dim), which the tensors
1944                // cannot reveal. Files converted before that carry 1.0,
1945                // meaning "unset" here rather than "rotate everything" —
1946                // for those the release's 64 stands in, which is what they
1947                // were converted from.
1948                rope_head_dim: if arch.partial_rotary_factor < 1.0 {
1949                    (((arch.head_dim as f32 * arch.partial_rotary_factor) as usize) & !1)
1950                        .clamp(2, arch.head_dim)
1951                } else {
1952                    64.min(arch.head_dim)
1953                },
1954                // The LoRA ranks and the group count ARE visible in the
1955                // weights, and reading them there means a re-tuned
1956                // checkpoint loads without touching this code.
1957                q_lora_rank: 0,
1958                o_lora_rank: 0,
1959                // Derived below from wo_a's shape — the attention output is
1960                // n_heads*head_dim wide and wo_a takes one group of it per
1961                // row block, so groups = width / wo_a.cols(). A pinned 8 is
1962                // right for the release and wrong for anything else, which
1963                // is exactly what made a toy checkpoint impossible to
1964                // compare against the reference.
1965                o_groups: 8,
1966                hc_mult: 4,
1967                hc_sinkhorn_iters: 20,
1968                hc_eps: 1e-6,
1969                norm_eps: arch.rms_norm_eps as f32,
1970                n_routed_experts: moe.num_experts,
1971                top_k: moe.top_k,
1972                moe_inter: moe.moe_intermediate_size,
1973                route_scale: moe.routed_scaling_factor.unwrap_or(1.0),
1974                // config.json's `swiglu_limit`, which the header has no
1975                // field for. The release ships 10.0; a checkpoint that
1976                // retunes it would need this read from the config, so it
1977                // sits next to the other pinned constants rather than
1978                // hiding inside the expert.
1979                swiglu_limit: 10.0,
1980                window: arch.sliding_window.unwrap_or(128),
1981                index_topk: 512,
1982                vocab: arch.vocab_size,
1983            };
1984            let (g, dl) = crate::dsv4::load(model, &cfg, arch.num_layers)
1985                .map_err(|e| CmfError::Parse(format!("deepseek_v4: {e}")))?;
1986            // Read the ranks off the weights that define them: wq_a's
1987            // rows ARE q_lora_rank, and wo_b's columns are groups x
1988            // o_lora_rank. A header field could disagree with the file;
1989            // these cannot.
1990            let mut cfg = cfg;
1991            if let Some(l0) = dl.first() {
1992                cfg.q_lora_rank = l0.wq_a.rows();
1993                let attn_width = arch.num_attention_heads * arch.head_dim;
1994                if l0.wo_a.cols() > 0 && attn_width % l0.wo_a.cols() == 0 {
1995                    cfg.o_groups = (attn_width / l0.wo_a.cols()).max(1);
1996                }
1997                cfg.o_lora_rank = l0.wo_b.cols() / cfg.o_groups.max(1);
1998                cfg.hc_mult = (l0.hc_attn_fn.len() / l0.hc_attn_base.len().max(1)) / cfg.dim.max(1);
1999                if cfg.hc_mult == 0 {
2000                    cfg.hc_mult = 4;
2001                }
2002            }
2003            // RoPE rides only the last `rope_head_dim` of each head, and the
2004            // reference builds its frequencies over THAT width — not over
2005            // head_dim, which is 512 here. The generic path above sized them
2006            // by head_dim, giving 1/base^(2i/512) where 1/base^(2i/64) is
2007            // wanted: every position rotated by the wrong angle.
2008            //
2009            // YaRN is applied unconditionally by the reference (its guard is
2010            // `original_seq_len > 0`, not the sequence length), so it belongs
2011            // in these frequencies too. Older configs spell the key `type`
2012            // rather than `rope_type`; when the header carries no profile the
2013            // release's own numbers stand in, which is better than silently
2014            // decoding with unscaled frequencies.
2015            let (yf, yo, ybf, ybs) = match &arch.yarn {
2016                Some(y) => (
2017                    y.factor,
2018                    y.original_max_position_embeddings,
2019                    y.beta_fast,
2020                    y.beta_slow,
2021                ),
2022                None => {
2023                    tracing::warn!(
2024                        "deepseek_v4: the header carries no YaRN profile — \
2025                         falling back to the release's (factor 16, original \
2026                         65536, beta 32/1). Re-converting with a build that \
2027                         reads rope_scaling.type would make this exact."
2028                    );
2029                    (16.0, 65536, 32.0, 1.0)
2030                }
2031            };
2032            pipeline.inv_freq = std::sync::Arc::new(crate::attention::yarn_inv_freq(
2033                cfg.rope_head_dim,
2034                arch.rope_theta as f32,
2035                yf,
2036                yo,
2037                ybf,
2038                ybs,
2039            ));
2040            // Keep the working set resident. Everything but the routed
2041            // experts is touched by every token, and of the experts only the
2042            // ones the task actually routes to — the page cache cannot know
2043            // that and evicts by age instead.
2044            if let Ok(stats) = std::env::var("CMF_MOE_PIN") {
2045                let cover = std::env::var("CMF_MOE_PIN_COVER")
2046                    .ok()
2047                    .and_then(|v| v.parse::<f64>().ok())
2048                    .filter(|&c| c > 0.0 && c <= 1.0)
2049                    .unwrap_or(0.95);
2050                let hot = crate::pin::hot_experts(&stats, cover);
2051                let mut names: Vec<String> = Vec::new();
2052                for e in &model.tensors {
2053                    let is_expert = e.name.contains(".mlp.experts.");
2054                    if !is_expert {
2055                        names.push(e.name.clone()); // skeleton: always hot
2056                    }
2057                }
2058                let mut kept_experts = 0usize;
2059                if let Some(hot) = &hot {
2060                    for (li, experts) in hot {
2061                        for e in experts {
2062                            for w in ["gate_proj", "up_proj", "down_proj"] {
2063                                names.push(format!("model.layers.{li}.mlp.experts.{e}.{w}.weight"));
2064                            }
2065                            kept_experts += 1;
2066                        }
2067                    }
2068                }
2069                let r = crate::pin::pin_tensors(model, &names);
2070                tracing::info!(
2071                    "закреплено {:.1} ГБ ({} тензоров, горячих экспертов {kept_experts},                      покрытие {cover}); лимит {}",
2072                    r.bytes as f64 / 1e9,
2073                    r.tensors,
2074                    r.limit
2075                        .map(|l| format!("{:.1} ГБ", l as f64 / 1e9))
2076                        .unwrap_or_else(|| "неизвестен".into())
2077                );
2078                if r.skipped > 0 {
2079                    tracing::warn!("не закреплено тензоров: {}", r.skipped);
2080                }
2081            }
2082            let st = crate::dsv4::Dsv4State::new(arch.num_layers);
2083            // The speculation stack, if the file carries one. Reading it is
2084            // metadata only — the expert weights stay in the mapping until a
2085            // draft actually runs — so it costs nothing to know it is there.
2086            let depth = std::env::var("CMF_DSV4_MTP_DEPTH")
2087                .ok()
2088                .and_then(|v| v.parse::<usize>().ok())
2089                .unwrap_or(3);
2090            pipeline.dsv4_mtp = crate::dsv4::load_mtp(model, &cfg, depth);
2091            // Before any trunk pack is built: leave the draft its VRAM.
2092            crate::dsv4::dspark_reserve_note(&pipeline.dsv4_mtp, &cfg, &dl);
2093            pipeline.dsv4 = Some(Box::new((g, dl, cfg, st)));
2094        }
2095        pipeline.short_conv_cfg = short_conv_cfg;
2096        pipeline.mtp = mtp;
2097        if arch.arch_name == "mimo_v2" {
2098            pipeline.mimo_mtp = crate::pipeline::mimo_mtp::load_for(model, &arch)?;
2099        }
2100        pipeline.install_dynamic_routing(model, false);
2101        // Record the load-time overlay so a later set_active_skill(None)
2102        // correctly reverts it (the union-diff assumes dyn_active mirrors
2103        // the live overlay). Blend loads have no single index to revert.
2104        match ov {
2105            Overlay::One(sid) => {
2106                pipeline.dyn_active = model.header.skills.iter().position(|s| &s.id == sid);
2107            }
2108            Overlay::Blend(_) => pipeline.dyn_blend_loaded = true,
2109            Overlay::None => {}
2110        }
2111        // B1: apply the measured confidence-calibration temperature, if the
2112        // file carries one (softmax(logits / T) for reported confidence).
2113        if let Some(c) = &model.header.calibration {
2114            pipeline.set_calib_temp(c.temperature);
2115        }
2116        // O(1) Nyström attention (runtime-level, no format change):
2117        // env CMF_O1 decides; unset falls through to the converter hint
2118        // in header.provenance.o1_attn (`cortiq convert --o1`), and
2119        // CMF_O1=off force-disables even the hint. CLI flags override
2120        // later via set_o1().
2121        let o1 = match crate::nystrom::o1_from_env() {
2122            crate::nystrom::O1Env::Off => None,
2123            crate::nystrom::O1Env::On(cfg) => Some(cfg),
2124            crate::nystrom::O1Env::Unset => model
2125                .header
2126                .provenance
2127                .as_ref()
2128                .and_then(|p| p.get("o1_attn"))
2129                .and_then(crate::nystrom::O1Cfg::from_json),
2130        };
2131        if o1.is_some() {
2132            if pipeline.attn_softcap > 0.0 {
2133                return Err(CmfError::Parse(
2134                    "--o1 with attention-logit soft-capping (Gemma-2) is not supported: \
2135                     the streaming operator has no capped-score form"
2136                        .into(),
2137                ));
2138            }
2139            pipeline.set_o1(o1);
2140        }
2141        Ok(pipeline)
2142    }
2143
2144    /// Record per-skill dynamic-routing metadata: which FFN layers each
2145    /// skill actually replaces (derived from the tensors present, not
2146    /// the meta `layers` field), and whether the skill is eligible for
2147    /// cheap dynamic switching (FFN-only). Called once at load.
2148    pub(crate) fn install_dynamic_routing(&mut self, model: &Arc<CmfModel>, force_f32: bool) {
2149        self.model = Some(model.clone());
2150        self.dyn_force_f32 = force_f32;
2151        let mut per_skill = Vec::with_capacity(model.header.skills.len());
2152        for sk in &model.header.skills {
2153            let mut ffn_layers = std::collections::BTreeSet::new();
2154            let mut non_ffn = false;
2155            let prefix = format!("skill.{}.", sk.id);
2156            for t in model.skill_tensors(&sk.id) {
2157                let rel = &t.name[prefix.len()..]; // e.g. model.layers.20.mlp.down_proj.weight
2158                let toks: Vec<&str> = rel.split('.').collect();
2159                if toks.len() >= 5 && toks[0] == "model" && toks[1] == "layers" && toks[3] == "mlp"
2160                {
2161                    if let Ok(li) = toks[2].parse::<usize>() {
2162                        ffn_layers.insert(li);
2163                        continue;
2164                    }
2165                }
2166                non_ffn = true; // replaces attention / embed / lm_head
2167            }
2168            if non_ffn {
2169                tracing::warn!(
2170                    "skill '{}' replaces non-FFN tensors — excluded from dynamic \
2171                     routing (static overlay still works)",
2172                    sk.id
2173                );
2174                per_skill.push(None);
2175            } else {
2176                per_skill.push(Some(ffn_layers.into_iter().collect::<Vec<_>>()));
2177            }
2178        }
2179        self.dyn_skill_layers = per_skill;
2180    }
2181
2182    /// Switch the overlaid skill for subsequent forwards (dynamic
2183    /// routing). `idx` = index into model.header.skills; None = backbone.
2184    /// Rebuilds the FFN of the union of the old and new skill's touched
2185    /// layers with the new overlay — tensor-source indirection made
2186    /// dynamic. Cheap: Mapped tensors are re-resolved mmap pointers.
2187    /// Result is bit-identical to loading the pipeline with that skill.
2188    pub fn set_active_skill(&mut self, idx: Option<usize>) -> Result<(), CmfError> {
2189        // Overlay swap changes weights → every cached K/V is stale.
2190        // The wgpu graph owns a parallel recurrent/KV mirror keyed by the
2191        // pipeline id.  Overlay swaps are sequence boundaries too; the shared
2192        // reset clears it before the next token so dynamic routing cannot
2193        // read state produced with the prior skill.
2194        self.reset_session();
2195        if self.dyn_active == idx {
2196            return Ok(());
2197        }
2198        let model = self.model.clone().ok_or_else(|| {
2199            CmfError::Parse("dynamic routing needs a model-backed pipeline".into())
2200        })?;
2201        let mut union: std::collections::BTreeSet<usize> = std::collections::BTreeSet::new();
2202        if let Some(old) = self.dyn_active {
2203            if let Some(Some(ls)) = self.dyn_skill_layers.get(old) {
2204                union.extend(ls.iter().copied());
2205            }
2206        }
2207        let new_id: Option<String> = match idx {
2208            Some(n) => match self.dyn_skill_layers.get(n) {
2209                Some(Some(ls)) => {
2210                    union.extend(ls.iter().copied());
2211                    Some(model.header.skills[n].id.clone())
2212                }
2213                _ => {
2214                    return Err(CmfError::Parse(format!(
2215                        "skill index {n} not dynamic-eligible"
2216                    )));
2217                }
2218            },
2219            None => None,
2220        };
2221        let ov = match &new_id {
2222            Some(s) => Overlay::One(s),
2223            None => Overlay::None,
2224        };
2225        let arch = model.arch();
2226        for li in union {
2227            self.weights.layers[li].ffn =
2228                build_layer_ffn(&model, arch, li, self.dyn_force_f32, &ov)?;
2229        }
2230        self.dyn_active = idx;
2231        Ok(())
2232    }
2233}