Skip to main content

cortiq_engine/
loader.rs

1//! Weight loader: CMF tensor directory → Pipeline.
2//!
3//! Storage rule: models WITH task masks are dequantized to f32 (masked
4//! execution needs f32 row access; skill files are small by design).
5//! Models without masks keep quantized matrices zero-copy from the mmap
6//! (`QTensor::Mapped`) — this is what lets a 15B file run in a few GB
7//! of RSS instead of 60 GB of f32.
8//!
9//! Layer kinds come from `arch.layer_types`: FullAttention loads
10//! `self_attn.*` (with auto-detected Qwen3.5 extras: per-head qk-norm by
11//! tensor presence, output gate by q_proj row count); LinearAttention
12//! loads the canonical core `vmf_attn.*` (folded at convert time).
13
14use crate::kv_cache::LayerKvCache;
15use crate::linear_core::{
16    GdnCfg, GdnWeights, ShortConvCfg, ShortConvWeights, VmfPhaseCfg, VmfPhaseWeights,
17};
18use crate::pipeline::{
19    AttnKind, DenseFfn, FfnKind, LayerWeights, MoeFfn, MtpModule, Pipeline, PipelineWeights,
20};
21use crate::qtensor::QTensor;
22use crate::sampler::SamplerConfig;
23use crate::tokenizer::Tokenizer;
24use cortiq_core::quant::dequant_tensor;
25use cortiq_core::{CmfError, CmfModel, LayerType, ModelArch};
26use std::sync::Arc;
27
28/// The refusal every generative entry point (`Pipeline::from_model*`, and
29/// through it run/chat/route/bench) gives a file carrying the DECISION
30/// feature bit. Such a file is served by the decision runtime instead.
31pub const DECISION_MODEL_REFUSAL: &str =
32    "this is a DECISION model; use `cortiq decide` or `cortiq serve`";
33
34/// Tensor source selector (spec §9): backbone, one skill's overlay, or
35/// a soft superposition of top-m skills (claim 14 working tensors).
36pub enum Overlay<'a> {
37    None,
38    One(&'a str),
39    /// (skill_id, weight); weights sum to 1 (softmax(−E/T) upstream).
40    Blend(&'a [(String, f32)]),
41}
42
43impl Overlay<'_> {
44    fn blend_touches(&self, model: &CmfModel, name: &str) -> bool {
45        match self {
46            Overlay::Blend(list) => list
47                .iter()
48                .any(|(sid, _)| model.tensor(&format!("skill.{sid}.{name}")).is_some()),
49            _ => false,
50        }
51    }
52}
53
54fn dequant_by_name(model: &CmfModel, name: &str) -> Result<Vec<f32>, String> {
55    let entry = model
56        .tensor(name)
57        .ok_or_else(|| format!("tensor '{name}' not found in CMF directory"))?;
58    let mut out = vec![0.0f32; entry.n_elems()];
59    dequant_tensor(entry, model.entry_bytes(entry), &mut out)?;
60    Ok(out)
61}
62
63/// Weighted working tensor (claim 14): Σ wᵢ·Tᵢ, where Tᵢ is the
64/// skill's replacement when present, else the backbone tensor.
65fn blend_f32(model: &CmfModel, name: &str, list: &[(String, f32)]) -> Result<Vec<f32>, String> {
66    let mut acc: Option<Vec<f32>> = None;
67    for (sid, w) in list {
68        let sname = format!("skill.{sid}.{name}");
69        let src = if model.tensor(&sname).is_some() {
70            &sname
71        } else {
72            name
73        };
74        let t = dequant_by_name(model, src)?;
75        match &mut acc {
76            None => {
77                let mut t = t;
78                for v in t.iter_mut() {
79                    *v *= w;
80                }
81                acc = Some(t);
82            }
83            Some(a) => {
84                for (av, tv) in a.iter_mut().zip(&t) {
85                    *av += w * tv;
86                }
87            }
88        }
89    }
90    acc.ok_or_else(|| "empty blend".into())
91}
92
93/// Dequantize a tensor fully into f32 (norms, masked models).
94pub(crate) fn load_f32(model: &CmfModel, name: &str, ov: &Overlay) -> Result<Vec<f32>, String> {
95    if ov.blend_touches(model, name) {
96        if let Overlay::Blend(list) = ov {
97            return blend_f32(model, name, list);
98        }
99    }
100    let skill = match ov {
101        Overlay::One(s) => Some(*s),
102        _ => None,
103    };
104    let entry = model
105        .resolve_tensor(name, skill)
106        .ok_or_else(|| format!("tensor '{name}' not found in CMF directory"))?;
107    let bytes = model.entry_bytes(entry);
108    let mut out = vec![0.0f32; entry.n_elems()];
109    dequant_tensor(entry, bytes, &mut out)?;
110    Ok(out)
111}
112
113/// Build one layer's FFN (dense or MoE) under a given overlay. Shared
114/// by the static loader AND dynamic per-token skill switching
115/// (`Pipeline::set_active_skill`): switching skills = rebuilding the
116/// FFN of the touched layers, cheap because Mapped tensors are just
117/// re-resolved mmap pointers (no dequant, no copy).
118pub(crate) fn build_layer_ffn(
119    model: &Arc<CmfModel>,
120    arch: &ModelArch,
121    li: usize,
122    force_f32: bool,
123    ov: &Overlay,
124) -> Result<FfnKind, CmfError> {
125    build_ffn_at(model, arch, &format!("model.layers.{li}."), force_f32, ov)
126}
127
128/// The FFN under an arbitrary prefix. Split out of `build_layer_ffn` so the
129/// MTP block can reuse it: Qwen3.6's MTP layer carries a full MoE mlp
130/// (router + 256 experts + shared expert), not the dense one the head was
131/// first written against.
132pub(crate) fn build_ffn_at(
133    model: &Arc<CmfModel>,
134    arch: &ModelArch,
135    prefix: &str,
136    force_f32: bool,
137    ov: &Overlay,
138) -> Result<FfnKind, CmfError> {
139    let prefix = prefix.to_string();
140    let load_dense = |p: &str| -> Result<DenseFfn, CmfError> {
141        let gate_proj = load_matrix(model, &format!("{p}gate_proj.weight"), force_f32, ov)?;
142        let up_proj = load_matrix(model, &format!("{p}up_proj.weight"), force_f32, ov)?;
143        let down_proj = load_matrix(model, &format!("{p}down_proj.weight"), force_f32, ov)?;
144        // FFN triple invariant (holds for dense and each MoE expert;
145        // enforced loudly so a malformed defrag/repack — spec §11 — fails
146        // at load instead of silently mis-computing). inter' is per-layer.
147        let inter = gate_proj.rows();
148        if up_proj.rows() != inter || down_proj.cols() != inter {
149            return Err(CmfError::Parse(format!(
150                "{p}: FFN dims disagree (gate.rows={inter}, up.rows={}, \
151                 down.cols={}); all three must equal inter'",
152                up_proj.rows(),
153                down_proj.cols()
154            )));
155        }
156        if down_proj.rows() != arch.hidden_size {
157            return Err(CmfError::Parse(format!(
158                "{p}: down_proj.rows={} != hidden_size={}",
159                down_proj.rows(),
160                arch.hidden_size
161            )));
162        }
163        // The transposed down, when the file carries one: the per-token
164        // sparse path needs a neuron's down weights contiguous.
165        let dt_name = format!("{p}down_proj.t.weight");
166        let down_t = match model.tensor(&dt_name) {
167            Some(_) => Some(load_matrix(model, &dt_name, force_f32, ov)?),
168            None => None,
169        };
170        if let Some(t) = &down_t
171            && (t.rows() != inter || t.cols() != arch.hidden_size)
172        {
173            return Err(CmfError::Parse(format!(
174                "{p}down_proj.t: [{}, {}] != [{inter}, {}]",
175                t.rows(),
176                t.cols(),
177                arch.hidden_size
178            )));
179        }
180        // Task tubes (`…gate_proj.tube1.weight`, …): the neurons only
181        // some tasks compute, each stored as its own triple so every
182        // kernel runs it unchanged and an inactive tube's bytes are
183        // never touched. Numbering is dense from 1; the first gap ends
184        // the tube list.
185        let mut segs = Vec::new();
186        let mut start = inter;
187        for k in 1.. {
188            let gn = format!("{p}gate_proj.tube{k}.weight");
189            if model.tensor(&gn).is_none() {
190                break;
191            }
192            let gate = load_matrix(model, &gn, force_f32, ov)?;
193            let up = load_matrix(model, &format!("{p}up_proj.tube{k}.weight"), force_f32, ov)?;
194            let down = load_matrix(
195                model,
196                &format!("{p}down_proj.tube{k}.weight"),
197                force_f32,
198                ov,
199            )?;
200            let width = gate.rows();
201            if up.rows() != width || down.cols() != width || down.rows() != arch.hidden_size {
202                return Err(CmfError::Parse(format!(
203                    "{p}tube{k}: dims disagree (gate.rows={width}, up.rows={}, \
204                     down=[{}, {}], hidden={})",
205                    up.rows(),
206                    down.rows(),
207                    down.cols(),
208                    arch.hidden_size
209                )));
210            }
211            segs.push(crate::pipeline::FfnSeg {
212                gate,
213                up,
214                down,
215                start,
216                width,
217            });
218            start += width;
219        }
220        Ok(DenseFfn {
221            gate_proj,
222            up_proj,
223            down_proj,
224            act: crate::pipeline::Act::from_arch_full(arch),
225            down_t,
226            segs,
227        })
228    };
229    let router_name = format!("{prefix}mlp.gate.weight");
230    // Cortiq Embryo: resonance-routed experts carry NO gate tensor — the
231    // MoE is keyed on the first expert + the arch flag (a grown genome
232    // appends `mlp.experts.{E}.*` records without rewriting anything).
233    let resonance_moe = model.tensor(&router_name).is_none()
234        && arch.moe.as_ref().is_some_and(|m| m.router_resonance)
235        && model
236            .tensor(&format!("{prefix}mlp.experts.0.gate_proj.weight"))
237            .is_some();
238    if model.tensor(&router_name).is_none() && !resonance_moe {
239        return Ok(FfnKind::Dense(load_dense(&format!("{prefix}mlp."))?));
240    }
241    let cfg = arch.moe.as_ref().ok_or_else(|| {
242        CmfError::Parse(format!(
243            "{router_name} present but header has no arch.moe block"
244        ))
245    })?;
246    // Experts enumerate by TENSOR PRESENCE up to the header count — a
247    // moe-defrag'd specialist keeps a per-layer contiguous prefix of
248    // renumbered experts (fewer than arch.moe.num_experts), with the
249    // router rows sliced to match.
250    let mut experts = Vec::new();
251    for e in 0..cfg.num_experts {
252        if model
253            .tensor(&format!("{prefix}mlp.experts.{e}.gate_proj.weight"))
254            .is_none()
255        {
256            break;
257        }
258        experts.push(load_dense(&format!("{prefix}mlp.experts.{e}."))?);
259    }
260    if experts.is_empty() {
261        return Err(CmfError::Parse(format!(
262            "{prefix}: router present but no expert tensors"
263        )));
264    }
265    // Growth records (`kind = "expert_append"`, spec §9.5.1): the experts
266    // the records append to THIS layer, mounted behind the trunk experts
267    // in header order. `arch.moe.num_experts` stays E0; the trunk count
268    // is `e0`, the tail `experts[e0..]` is described by `grown`.
269    let e0 = experts.len();
270    let grown_loads = if resonance_moe {
271        mount_growth(model, &prefix, ov, &load_dense)?
272    } else {
273        Vec::new()
274    };
275    let mut grown_desc: Vec<(Vec<f32>, Vec<f32>, f32, f32)> = Vec::new();
276    let mut grown: Vec<crate::pipeline::GrownExpert> = Vec::new();
277    for g in grown_loads {
278        experts.push(g.ffn);
279        grown_desc.push((g.mu, g.u, g.bias, g.shell));
280        grown.push(g.meta);
281    }
282    let shared = if model
283        .tensor(&format!("{prefix}mlp.shared_expert.gate_proj.weight"))
284        .is_some()
285    {
286        let gate_name = format!("{prefix}mlp.shared_expert_gate.weight");
287        Some((
288            load_dense(&format!("{prefix}mlp.shared_expert."))?,
289            if model.tensor(&gate_name).is_some() {
290                Some(load_matrix(model, &gate_name, force_f32, ov)?)
291            } else {
292                None
293            },
294        ))
295    } else {
296        None
297    };
298    // LFM2-MoE selection bias (`mlp.expert_bias`): present iff the model
299    // routes with a bias; loaded by tensor presence.
300    let bias_name = format!("{prefix}mlp.expert_bias");
301    let expert_bias = if model.tensor(&bias_name).is_some() {
302        Some(load_f32(model, &bias_name, ov).map_err(CmfError::Parse)?)
303    } else {
304        None
305    };
306    // CMF_MOE_TOPK=N (opt-in): route to fewer experts than the header
307    // asks. MoE decode is memory-bound — every selected expert streams
308    // its three matrices per token — so halving k halves that traffic;
309    // the renormalized top-k keeps the mixture a proper average.
310    // Quality is the experiment — measure ppl before trusting.
311    let top_k = std::env::var("CMF_MOE_TOPK")
312        .ok()
313        .and_then(|v| v.parse::<usize>().ok())
314        .filter(|&k| k >= 1 && k <= cfg.top_k)
315        .inspect(|k| tracing::info!("MoE top_k override: {} (header {})", k, cfg.top_k))
316        .unwrap_or(cfg.top_k);
317    // CMF_MOE_TAU=0.x (opt-in): adaptive routing — see MoeFfn::route_tau.
318    let route_tau = std::env::var("CMF_MOE_TAU")
319        .ok()
320        .and_then(|v| v.parse::<f32>().ok())
321        .filter(|&t| t > 0.0 && t < 1.0)
322        .inspect(|t| tracing::info!("MoE adaptive routing: tau {t}"));
323    let mask = moe_task_mask(model, &prefix, experts.len());
324    let router = if resonance_moe {
325        // never read (selection is by descriptors); zero placeholder in RAM
326        QTensor::from_f32(
327            vec![0.0; experts.len() * arch.hidden_size],
328            experts.len(),
329            arch.hidden_size,
330        )
331    } else {
332        load_matrix(model, &router_name, force_f32, ov)?
333    };
334    if router.rows() != experts.len() {
335        return Err(CmfError::Parse(format!(
336            "{router_name}: {} rows != {} experts",
337            router.rows(),
338            experts.len()
339        )));
340    }
341    let top_k = top_k.min(experts.len());
342    // Gemma-4: per-expert weight scale after the top-k renorm; its
343    // presence also marks the scale-less-rms router input (the folded
344    // router gain — see the converter).
345    let pes_name = format!("{prefix}mlp.per_expert_scale");
346    let per_expert_scale = if model.tensor(&pes_name).is_some() {
347        Some(load_f32(model, &pes_name, ov).map_err(CmfError::Parse)?)
348    } else {
349        None
350    };
351    let router_input_norm = per_expert_scale.is_some();
352    // Cortiq Embryo resonance descriptors: per-expert records
353    // `mlp.experts.{e}.desc.{mu,u,bias}` (append-only growth) or the legacy
354    // per-layer `mlp.desc.{mu,u,bias}` [E, ...]. Present → the router is a
355    // placeholder and selection is by reconstruction error.
356    let per_expert = model
357        .tensor(&format!("{prefix}mlp.experts.0.desc.mu"))
358        .is_some();
359    if !per_expert && !grown.is_empty() {
360        return Err(CmfError::Parse(format!(
361            "{prefix}: expert_append records need per-expert descriptors \
362             (mlp.experts.0.desc.mu); the stacked mlp.desc.* form cannot be grown"
363        )));
364    }
365    let resonance = if per_expert {
366        // Trunk descriptors first (E0 rows), then the grown rows behind
367        // them: `shell` is +inf on the trunk, the record's finite
368        // `desc.shell` on a grown expert.
369        let ne_d = e0;
370        let hidden = arch.hidden_size;
371        let mut mu = Vec::with_capacity(experts.len() * hidden);
372        let mut u = Vec::new();
373        let mut bias = Vec::with_capacity(experts.len());
374        let mut k = 0usize;
375        for e in 0..ne_d {
376            let m = load_f32(model, &format!("{prefix}mlp.experts.{e}.desc.mu"), ov)
377                .map_err(CmfError::Parse)?;
378            if m.len() != hidden {
379                return Err(CmfError::Parse(format!(
380                    "{prefix}mlp.experts.{e}.desc.mu: {} != {hidden}",
381                    m.len()
382                )));
383            }
384            mu.extend_from_slice(&m);
385            let un = format!("{prefix}mlp.experts.{e}.desc.u");
386            if model.tensor(&un).is_some() {
387                let ue = load_f32(model, &un, ov).map_err(CmfError::Parse)?;
388                let ke = ue.len() / hidden.max(1);
389                if e == 0 {
390                    k = ke;
391                }
392                if ke != k {
393                    return Err(CmfError::Parse(format!("{un}: rank {ke} != {k}")));
394                }
395                u.extend_from_slice(&ue);
396            }
397            let bn = format!("{prefix}mlp.experts.{e}.desc.bias");
398            bias.push(if model.tensor(&bn).is_some() {
399                load_f32(model, &bn, ov)
400                    .map_err(CmfError::Parse)?
401                    .first()
402                    .copied()
403                    .unwrap_or(0.0)
404            } else {
405                0.0
406            });
407        }
408        let mut shell = vec![f32::INFINITY; ne_d];
409        for (gm, gu, gb, gs) in &grown_desc {
410            if gm.len() != hidden || gu.len() != k * hidden {
411                return Err(CmfError::Parse(format!(
412                    "{prefix}: grown expert descriptor {}×{} != trunk {hidden}×{k}",
413                    gm.len(),
414                    gu.len() / hidden.max(1)
415                )));
416            }
417            mu.extend_from_slice(gm);
418            u.extend_from_slice(gu);
419            bias.push(*gb);
420            shell.push(*gs);
421        }
422        Some(crate::pipeline::Resonance {
423            mu,
424            u,
425            k,
426            bias,
427            shell,
428        })
429    } else if model.tensor(&format!("{prefix}mlp.desc.mu")).is_some() {
430        let mu = load_f32(model, &format!("{prefix}mlp.desc.mu"), ov).map_err(CmfError::Parse)?;
431        let ne_d = experts.len();
432        let hidden = arch.hidden_size;
433        if mu.len() != ne_d * hidden {
434            return Err(CmfError::Parse(format!(
435                "{prefix}mlp.desc.mu: {} != {ne_d}×{hidden}",
436                mu.len()
437            )));
438        }
439        let u_name = format!("{prefix}mlp.desc.u");
440        let (u, k) = if model.tensor(&u_name).is_some() {
441            let u = load_f32(model, &u_name, ov).map_err(CmfError::Parse)?;
442            let k = u.len() / (ne_d * hidden).max(1);
443            (u, k)
444        } else {
445            (Vec::new(), 0)
446        };
447        let b_name = format!("{prefix}mlp.desc.bias");
448        let bias = if model.tensor(&b_name).is_some() {
449            load_f32(model, &b_name, ov).map_err(CmfError::Parse)?
450        } else {
451            vec![0.0; ne_d]
452        };
453        Some(crate::pipeline::Resonance {
454            mu,
455            u,
456            k,
457            bias,
458            shell: Vec::new(),
459        })
460    } else {
461        None
462    };
463    let moe = MoeFfn {
464        grown,
465        router,
466        experts,
467        top_k,
468        route_tau,
469        norm_topk_prob: cfg.norm_topk_prob,
470        router_sigmoid: cfg.router_sigmoid,
471        expert_bias,
472        routed_scaling: cfg.routed_scaling_factor.unwrap_or(1.0),
473        shared,
474        stats: std::cell::RefCell::new(Vec::new()),
475        act_sq: std::cell::RefCell::new(Vec::new()),
476        act_rows: std::cell::RefCell::new(Vec::new()),
477        mask,
478        per_expert_scale,
479        router_input_norm,
480        resonance,
481    };
482    // Gemma-4 dual-branch layer: a dense MLP coexists with the routed
483    // experts, each branch inside its own norm sandwich.
484    if model
485        .tensor(&format!("{prefix}mlp.gate_proj.weight"))
486        .is_some()
487    {
488        let norm = |suffix: &str| -> Result<Vec<f32>, CmfError> {
489            load_f32(model, &format!("{prefix}{suffix}.weight"), ov).map_err(CmfError::Parse)
490        };
491        return Ok(FfnKind::DenseMoe(Box::new(crate::pipeline::DenseMoeFfn {
492            dense: load_dense(&format!("{prefix}mlp."))?,
493            moe,
494            post_norm_1: norm("post_feedforward_layernorm_1")?,
495            pre_norm_2: norm("pre_feedforward_layernorm_2")?,
496            post_norm_2: norm("post_feedforward_layernorm_2")?,
497        })));
498    }
499    Ok(FfnKind::Moe(moe))
500}
501
502// ───────────────────────── growth records ─────────────────────────
503
504/// Which `expert_append` records the loader mounts (`CMF_GROWTH`).
505#[derive(Clone, Copy, PartialEq, Eq, Debug)]
506pub enum GrowthMode {
507    /// Records with `status == "active"` (the default).
508    Active,
509    /// Every record that is not retired (`CMF_GROWTH=all`: quarantine
510    /// and stale_regate too — the gate measurement of a fresh record).
511    All,
512    /// None (`CMF_GROWTH=off`: the plain genome, F0's forward).
513    Off,
514}
515
516impl GrowthMode {
517    pub fn label(self) -> &'static str {
518        match self {
519            Self::Active => "active",
520            Self::All => "all",
521            Self::Off => "off",
522        }
523    }
524
525    /// Does this mode mount a record of `status`?
526    pub fn admits(self, status: Option<&str>) -> bool {
527        match self {
528            Self::Off => false,
529            Self::Active => status == Some("active"),
530            Self::All => status != Some("retired"),
531        }
532    }
533}
534
535/// `CMF_GROWTH` = `active` (unset) | `all` | `off`.
536pub fn growth_mode() -> GrowthMode {
537    match std::env::var("CMF_GROWTH")
538        .ok()
539        .as_deref()
540        .map(|v| v.trim().to_ascii_lowercase())
541        .as_deref()
542    {
543        Some("off") | Some("0") | Some("none") => GrowthMode::Off,
544        Some("all") => GrowthMode::All,
545        _ => GrowthMode::Active,
546    }
547}
548
549/// The `expert_append` records the current [`growth_mode`] mounts, with
550/// their positions in `header.skills` (the chain rule of the format is
551/// evaluated at that position).
552pub fn mounted_growth_records(
553    header: &cortiq_core::CmfHeader,
554) -> Vec<(usize, &cortiq_core::SkillRecord)> {
555    let mode = growth_mode();
556    header
557        .skills
558        .iter()
559        .enumerate()
560        .filter(|(_, s)| {
561            s.kind.as_deref() == Some(cortiq_core::knowledge::skill_kind::EXPERT_APPEND)
562                && mode.admits(s.status.as_deref())
563        })
564        .collect()
565}
566
567/// One grown expert as the loader read it from a record.
568struct GrownLoad {
569    ffn: DenseFfn,
570    mu: Vec<f32>,
571    u: Vec<f32>,
572    bias: f32,
573    shell: f32,
574    meta: crate::pipeline::GrownExpert,
575}
576
577/// The layer index of a `model.layers.{l}.` prefix (None for the MTP
578/// block and other prefixes — no growth record addresses those).
579fn layer_of_prefix(prefix: &str) -> Option<usize> {
580    prefix
581        .strip_prefix("model.layers.")?
582        .strip_suffix('.')?
583        .parse()
584        .ok()
585}
586
587/// The grown experts of the mounted `expert_append` records that grow the
588/// layer `prefix` addresses, in header order, each read from the record's
589/// own tensors (`skill.{id}.model.layers.{l}.mlp.experts.{e}.*`, the
590/// names the format's layout plan gives for the record's position). One
591/// log line per mounted record and layer.
592fn mount_growth(
593    model: &Arc<CmfModel>,
594    prefix: &str,
595    ov: &Overlay,
596    load_dense: &dyn Fn(&str) -> Result<DenseFfn, CmfError>,
597) -> Result<Vec<GrownLoad>, CmfError> {
598    use cortiq_core::knowledge::expert_leaf;
599    let Some(layer) = layer_of_prefix(prefix) else {
600        return Ok(Vec::new());
601    };
602    let records = mounted_growth_records(&model.header);
603    let mut out = Vec::new();
604    for (at, rec) in records {
605        if !rec.layers.contains(&layer) {
606            continue;
607        }
608        let plan = cortiq_core::expert_append_layout(&model.header, &model.tensors, at, rec)
609            .map_err(|e| CmfError::Parse(format!("{prefix}: growth record '{}': {e}", rec.id)))?;
610        let mut by_expert: std::collections::BTreeMap<usize, Vec<&cortiq_core::ExpertTensorSpec>> =
611            Default::default();
612        for p in plan.iter().filter(|p| p.layer == layer) {
613            by_expert.entry(p.expert).or_default().push(p);
614        }
615        let first = out.len();
616        for (e, leaves) in &by_expert {
617            let name = |leaf: &str| -> Result<String, CmfError> {
618                leaves
619                    .iter()
620                    .find(|p| p.leaf == leaf)
621                    .map(|p| p.name.clone())
622                    .ok_or_else(|| {
623                        CmfError::Parse(format!(
624                            "{prefix}: growth record '{}' expert {e}: layout has no '{leaf}'",
625                            rec.id
626                        ))
627                    })
628            };
629            let ffn = load_dense(&format!("skill.{}.{prefix}mlp.experts.{e}.", rec.id))?;
630            let mu = load_f32(model, &name(expert_leaf::MU)?, ov).map_err(CmfError::Parse)?;
631            let u = if leaves.iter().any(|p| p.leaf == expert_leaf::U) {
632                load_f32(model, &name(expert_leaf::U)?, ov).map_err(CmfError::Parse)?
633            } else {
634                Vec::new()
635            };
636            let scalar = |leaf: &str| -> Result<f32, CmfError> {
637                load_f32(model, &name(leaf)?, ov)
638                    .map_err(CmfError::Parse)?
639                    .first()
640                    .copied()
641                    .ok_or_else(|| {
642                        CmfError::Parse(format!(
643                            "{prefix}: growth record '{}' expert {e}: empty '{leaf}'",
644                            rec.id
645                        ))
646                    })
647            };
648            let bias = scalar(expert_leaf::BIAS)?;
649            let shell = scalar(expert_leaf::SHELL)?;
650            if !shell.is_finite() {
651                return Err(CmfError::Parse(format!(
652                    "{prefix}: growth record '{}' expert {e}: desc.shell {shell} is not finite",
653                    rec.id
654                )));
655            }
656            out.push(GrownLoad {
657                ffn,
658                mu,
659                u,
660                bias,
661                shell,
662                meta: crate::pipeline::GrownExpert {
663                    record: rec.id.clone(),
664                    record_index: at,
665                    layer,
666                    expert: *e,
667                },
668            });
669        }
670        let declared: Vec<usize> = by_expert.keys().copied().collect();
671        tracing::info!(
672            "growth: layer {layer}: mounted expert_append '{}' (status {}) — {} experts, declared {:?}, shells {:?}",
673            rec.id,
674            rec.status.as_deref().unwrap_or("?"),
675            by_expert.len(),
676            declared,
677            out[first..].iter().map(|g| g.shell).collect::<Vec<_>>()
678        );
679    }
680    Ok(out)
681}
682
683/// Task mask over routed experts: DTG-MA applied to MoE.
684///
685/// This is deliberately diagnostic-only. A route-mass sidecar can look good
686/// on a short perplexity window and still collapse during a long
687/// autoregressive generation: excluding a low-mass expert changes the model,
688/// and the changed prefix changes every later route. Consequently a nearby
689/// `<model>.moe-mass.json` is NEVER adopted implicitly. `CMF_MOE_MASK=...`
690/// remains an explicit experiment and accepts both weighted and legacy count
691/// files; production acceleration must preserve the full router and treat a
692/// non-resident winner as an exact cold completion.
693pub(crate) fn moe_task_mask(
694    _model: &std::sync::Arc<CmfModel>,
695    prefix: &str,
696    ne: usize,
697) -> Option<Vec<bool>> {
698    use std::sync::OnceLock;
699    static CFG: OnceLock<Option<(std::collections::HashMap<usize, Vec<u64>>, f64)>> =
700        OnceLock::new();
701    let cfg = CFG.get_or_init(|| {
702        let path = std::path::PathBuf::from(std::env::var("CMF_MOE_MASK").ok()?);
703        let shown = path.display();
704        let text = std::fs::read_to_string(&path)
705            .map_err(|e| tracing::warn!("CMF_MOE_MASK: cannot read {shown}: {e}"))
706            .ok()?;
707        let map: std::collections::HashMap<String, Vec<u64>> = serde_json::from_str(&text)
708            .map_err(|e| tracing::warn!("CMF_MOE_MASK: bad JSON in {shown}: {e}"))
709            .ok()?;
710        // Weighted dumps store normalized route mass in fixed-point units
711        // and therefore have per-layer totals many orders above the old
712        // top-k vote counts. The magnitude distinguishes the two explicit
713        // diagnostic formats without relying on a special filename.
714        let weighted = map
715            .values()
716            .any(|row| row.iter().copied().sum::<u64>() >= 1_000_000);
717        let cover = std::env::var("CMF_MOE_MASK_COVER")
718            .ok()
719            .and_then(|v| v.parse::<f64>().ok())
720            .filter(|&c| c > 0.0 && c <= 1.0)
721            .unwrap_or(if weighted { 0.925 } else { 0.9 });
722        tracing::info!("MoE task mask: {shown}, cover {cover}");
723        Some((
724            map.into_iter()
725                .filter_map(|(k, v)| Some((k.parse::<usize>().ok()?, v)))
726                .collect(),
727            cover,
728        ))
729    });
730    let (stats, cover) = cfg.as_ref()?;
731    // The layer index rides in the tensor prefix ("model.layers.N.").
732    let li: usize = prefix
733        .split("layers.")
734        .nth(1)?
735        .split('.')
736        .next()?
737        .parse()
738        .ok()?;
739    let counts = stats.get(&li)?;
740    if counts.len() != ne {
741        tracing::warn!(
742            "CMF_MOE_MASK: layer {li} has {} counts, model has {ne} experts — skipped",
743            counts.len()
744        );
745        return None;
746    }
747    let total: u64 = counts.iter().sum();
748    if total == 0 {
749        return None;
750    }
751    let mut order: Vec<usize> = (0..ne).collect();
752    order.sort_unstable_by_key(|&e| std::cmp::Reverse(counts[e]));
753    let mut mask = vec![false; ne];
754    let mut acc = 0u64;
755    let mut kept = 0usize;
756    for &e in &order {
757        mask[e] = true;
758        acc += counts[e];
759        kept += 1;
760        if (acc as f64) >= cover * (total as f64) {
761            break;
762        }
763    }
764    tracing::info!(
765        "MoE task mask L{li}: {kept}/{ne} experts for {:.0}% mass",
766        cover * 100.0
767    );
768    Some(mask)
769}
770
771pub(crate) fn load_matrix(
772    model: &Arc<CmfModel>,
773    name: &str,
774    force_f32: bool,
775    ov: &Overlay,
776) -> Result<QTensor, CmfError> {
777    // Claim 14: a blended working tensor is materialized in f32 and
778    // held resident (the overlay-cache slot); single skills stay
779    // zero-copy pointers into the mmap.
780    if ov.blend_touches(model, name) {
781        if let Overlay::Blend(list) = ov {
782            let entry = model
783                .tensor(name)
784                .ok_or_else(|| CmfError::MissingTensor(name.to_string()))?;
785            let data =
786                blend_f32(model, name, list).map_err(|e| CmfError::Parse(format!("blend: {e}")))?;
787            return Ok(QTensor::from_f32(data, entry.shape[0], entry.shape[1]));
788        }
789    }
790    let skill = match ov {
791        Overlay::One(s) => Some(*s),
792        _ => None,
793    };
794    // Tensor-source indirection (spec §9): the skill's replacement is
795    // read in place of the backbone tensor — either/or, never a sum.
796    let name: &str = &match skill {
797        Some(sid) if model.tensor(&format!("skill.{sid}.{name}")).is_some() => {
798            format!("skill.{sid}.{name}")
799        }
800        _ => name.to_string(),
801    };
802    let err = |e: String| CmfError::Parse(format!("weight loading: {e}"));
803    if force_f32 {
804        let entry = model
805            .tensor(name)
806            .ok_or_else(|| CmfError::MissingTensor(name.to_string()))?;
807        if entry.shape.len() != 2 {
808            return Err(err(format!("'{name}' is not 2-D")));
809        }
810        let data = load_f32(model, name, &Overlay::None).map_err(err)?;
811        Ok(QTensor::from_f32(data, entry.shape[0], entry.shape[1]))
812    } else {
813        QTensor::from_model(model, name).map_err(err)
814    }
815}
816
817impl Pipeline {
818    /// Build a runnable pipeline from an opened CMF model.
819    pub fn from_model(
820        model: &Arc<CmfModel>,
821        sampler_config: SamplerConfig,
822    ) -> Result<Self, CmfError> {
823        Self::from_model_with_skill(model, sampler_config, None)
824    }
825
826    /// Same, with a skill overlaid (spec §9): every layer tensor is
827    /// resolved through tensor-source indirection — the skill's
828    /// full-shape replacement is read in place of the backbone tensor.
829    /// No per-skill model is ever assembled: Mapped tensors are
830    /// pointers into the one shared mmap.
831    pub fn from_model_with_skill(
832        model: &Arc<CmfModel>,
833        sampler_config: SamplerConfig,
834        skill: Option<&str>,
835    ) -> Result<Self, CmfError> {
836        match skill {
837            Some(s) => Self::from_model_with_overlay(model, sampler_config, &Overlay::One(s)),
838            None => Self::from_model_with_overlay(model, sampler_config, &Overlay::None),
839        }
840    }
841
842    /// Soft superposition (claim 14): working tensors accumulated from
843    /// the given (skill, weight) list — softmax(−E/T) upstream.
844    pub fn from_model_with_blend(
845        model: &Arc<CmfModel>,
846        sampler_config: SamplerConfig,
847        blend: &[(String, f32)],
848    ) -> Result<Self, CmfError> {
849        Self::from_model_with_overlay(model, sampler_config, &Overlay::Blend(blend))
850    }
851
852    fn skill_file_guard(model: &CmfModel) -> Result<(), CmfError> {
853        // A decision profile (encoder + resonance skills) has no language
854        // model in it at all; the generative pipeline must not try to
855        // assemble one from its tensors.
856        if model.required_features & cortiq_core::format::features::DECISION != 0 {
857            return Err(CmfError::Parse(DECISION_MODEL_REFUSAL.into()));
858        }
859        // A standalone skill file carries a PARTIAL tensor set cut against
860        // a base; running it would be half a network answering questions.
861        if model.required_features & cortiq_core::format::features::SKILL_FILE != 0 {
862            return Err(CmfError::Parse(
863                "this file is a standalone SKILL, not a runnable model — attach it: \
864                 cortiq skill apply <base.cmf> <this file> -o specialist.cmf"
865                    .into(),
866            ));
867        }
868        Ok(())
869    }
870
871    fn from_model_with_overlay(
872        model: &Arc<CmfModel>,
873        sampler_config: SamplerConfig,
874        ov: &Overlay,
875    ) -> Result<Self, CmfError> {
876        // Refuse non-runnable files first, before any process-wide state
877        // (the device cache directory, the graph verdict) is touched on
878        // their behalf.
879        Self::skill_file_guard(model)?;
880        // Small device caches (probe verdicts, compiled pipelines) go
881        // beside the model: it is a directory the caller demonstrably
882        // writes to, which `std::env::temp_dir()` is not inside an
883        // Android app sandbox.
884        if let Some(dir) = model.path.parent() {
885            crate::gpu::set_cache_dir(dir.to_path_buf());
886        }
887        // The token-graph refusal verdict lives in each pipeline
888        // (`Pipeline::graph_refused`): a new pipeline starts clean and
889        // loading one — a skill lane mid-traffic — never touches the
890        // verdict of the others (R4/NF-2; this used to reset a
891        // process-wide flag under running slots).
892        let skill = match ov {
893            Overlay::One(s) => Some(*s),
894            _ => None,
895        };
896        if let Some(sid) = skill {
897            let known = model.header.skills.iter().any(|s| s.id == sid)
898                || model.skill_tensors(sid).next().is_some();
899            if !known {
900                return Err(CmfError::Parse(format!(
901                    "skill '{sid}' not in this container (header.skills: {:?})",
902                    model
903                        .header
904                        .skills
905                        .iter()
906                        .map(|s| &s.id)
907                        .collect::<Vec<_>>()
908                )));
909            }
910            tracing::info!(
911                "skill '{sid}': {} replacement tensors overlaid",
912                model.skill_tensors(sid).count()
913            );
914        }
915        let arch = model.arch().clone();
916        let err = |e: String| CmfError::Parse(format!("weight loading: {e}"));
917        if let Some(heads) = &arch.attention_heads_per_layer {
918            if heads.len() != arch.num_layers {
919                return Err(CmfError::Parse(format!(
920                    "arch.attention_heads_per_layer has {} entries, expected {}",
921                    heads.len(),
922                    arch.num_layers
923                )));
924            }
925            if let Some((li, &nh)) = heads
926                .iter()
927                .enumerate()
928                .find(|(_, nh)| **nh == 0 || **nh % arch.num_kv_heads != 0)
929            {
930                return Err(CmfError::Parse(format!(
931                    "layer {li} has {nh} Q heads, which must be nonzero and divisible by {} KV heads",
932                    arch.num_kv_heads
933                )));
934            }
935        }
936        if arch
937            .layer_types
938            .iter()
939            .any(|t| matches!(t, LayerType::SlidingAttention))
940            && arch.sliding_window.is_none()
941        {
942            return Err(CmfError::Parse(
943                "model has SlidingAttention layers but no arch.sliding_window".into(),
944            ));
945        }
946
947        // Masks × quantized mmap: only the HEAD-mask path needs f32
948        // slices, so f32 is forced only when some mask actually restricts
949        // attention heads. FFN masks run sparse directly on the quant
950        // bytes (sparse_ffn_quant), and embed/lm_head are never masked.
951        // The old condition forced f32 for ANY mask: a 4.17 B file whose
952        // mask touched only the FFN dequantized to 16.7 GB at load and
953        // ran the bare-f32 GEMMs — 0.0 tok/s on a laptop that swapped,
954        // and a 30× crawl on a 48-core server. A mask with no head rows
955        // costs nothing now.
956        let heads_masked = model.masks.masks.iter().any(|m| {
957            m.head_masks.iter().any(|row| {
958                let mut bits = 0usize;
959                for &b in row.iter() {
960                    bits += b.count_ones() as usize;
961                }
962                !row.is_empty() && bits < arch.num_attention_heads
963            })
964        });
965        let force_f32 = heads_masked; // attention only (head masks)
966
967        // ── Tokenizer: embedded → sidecar → byte-level fallback ──
968        let mut tokenizer = if let Some(vocab_bytes) = &model.vocab {
969            Tokenizer::from_bytes(vocab_bytes)
970                .map_err(|e| CmfError::Parse(format!("embedded tokenizer: {e}")))?
971        } else if let Some(g) = &model.header.genome {
972            // A genome binds its tokenizer (the trunk hash covers the VOCAB
973            // section): whatever tokenizer.json lies beside the file is not
974            // part of it (NF-5; core refuses such a file at open as well).
975            return Err(CmfError::Parse(format!(
976                "genome '{}' carries no embedded tokenizer (VOCAB) — a sidecar tokenizer.json \
977                 is not part of the genome; refusing",
978                g.id
979            )));
980        } else {
981            let sidecar = model.path.with_file_name("tokenizer.json");
982            if sidecar.exists() {
983                Tokenizer::from_file(&sidecar)
984                    .map_err(|e| CmfError::Parse(format!("sidecar tokenizer: {e}")))?
985            } else {
986                tracing::warn!("no tokenizer in file or sidecar — using byte-level fallback");
987                Tokenizer::byte_level()
988            }
989        };
990        // Chat/eos bundle (spec §6.1): the FILE defines chat behavior.
991        if let Some(tc) = &model.header.tokenizer_config {
992            tokenizer.chat_template = tc.chat_template.clone();
993            tokenizer.extra_eos.extend(tc.eos_token_ids.iter().copied());
994            if tokenizer.bos_token_id.is_none() {
995                tokenizer.bos_token_id = tc.bos_token_id;
996            }
997            tracing::info!(
998                "chat bundle: template {} chars, {} stop ids",
999                tc.chat_template.as_deref().map(str::len).unwrap_or(0),
1000                tc.eos_token_ids.len()
1001            );
1002        }
1003        // Gemma's contract requires <bos> at sequence start, but its
1004        // tokenizer.json post-processor does not add it (the chat
1005        // template does). Raw prompts need it too — word salad without.
1006        if arch.arch_name.to_lowercase().contains("gemma") && tokenizer.bos_token_id.is_some() {
1007            tokenizer.add_bos = true;
1008        }
1009
1010        // ── Top-level weights (never masked → always quantized) ──
1011        let embed_tokens = load_matrix(model, "model.embed_tokens.weight", false, ov)?;
1012        let final_norm = if arch.qwen4_exp.is_some() {
1013            // qwen4_exp folds its four streams through a learned global
1014            // hyper mixer and has no conventional final RMSNorm.
1015            vec![0.0; arch.hidden_size]
1016        } else {
1017            load_f32(model, "model.norm.weight", ov).map_err(err)?
1018        };
1019        let lm_head = if model.tensor("lm_head.weight").is_some() {
1020            load_matrix(model, "lm_head.weight", false, ov)?
1021        } else if arch.tie_word_embeddings {
1022            // Tied: reuse the embedding matrix (re-open, cheap for Mapped).
1023            load_matrix(model, "model.embed_tokens.weight", false, ov)?
1024        } else {
1025            return Err(CmfError::MissingTensor(
1026                "lm_head.weight (and tie_word_embeddings is false)".into(),
1027            ));
1028        };
1029
1030        // ── Linear-core geometry (required if any linear layer exists) ──
1031        let has_linear = arch
1032            .layer_types
1033            .iter()
1034            .any(|t| matches!(t, LayerType::LinearAttention));
1035        let mut vmf_cfg = None;
1036        let mut gdn_cfg = None;
1037        // `Some` means this file uses the versioned Phase-Delta operator;
1038        // the sorted indices are threaded into each layer's weights below.
1039        // Legacy kinds must not carry selector metadata: accepting and
1040        // ignoring it would recreate the original silent semantic break.
1041        let mut phase_delta_layers: Option<Vec<usize>> = None;
1042        if !has_linear
1043            && arch
1044                .linear_core
1045                .as_ref()
1046                .is_some_and(|lc| lc.phase_delta_layers.is_some())
1047        {
1048            return Err(CmfError::Parse(
1049                "phase_delta_layers is only valid with LinearAttention and \
1050                 vmf_phase_delta_v1"
1051                    .into(),
1052            ));
1053        }
1054        if has_linear {
1055            let lc = arch.linear_core.as_ref().ok_or_else(|| {
1056                CmfError::Parse(
1057                    "model has LinearAttention layers but no arch.linear_core — \
1058                     reconvert with the current converter"
1059                        .into(),
1060                )
1061            })?;
1062            let need = |v: Option<usize>, name: &str| {
1063                v.ok_or_else(|| CmfError::Parse(format!("linear core needs arch.{name}")))
1064            };
1065            match lc.kind.as_str() {
1066                "vmf_phase" => {
1067                    if lc.phase_delta_layers.is_some() {
1068                        return Err(CmfError::Parse(
1069                            "legacy vmf_phase cannot carry phase_delta_layers; use vmf_phase_delta_v1"
1070                                .into(),
1071                        ));
1072                    }
1073                    vmf_cfg = Some(VmfPhaseCfg {
1074                        num_heads: lc.num_heads,
1075                        nphase: need(lc.nphase, "linear_core.nphase")?,
1076                        value_head_dim: lc.value_head_dim,
1077                        hidden_size: arch.hidden_size,
1078                        // Phase-mass correction: default 0 (disabled); CMF_PHASE_MASS
1079                        // widens the phase kernel for folded-unhealed models.
1080                        phase_mass: std::env::var("CMF_PHASE_MASS")
1081                            .ok()
1082                            .and_then(|v| v.parse().ok())
1083                            .unwrap_or(0.0),
1084                    });
1085                }
1086                "vmf_phase_delta_v1" => {
1087                    let mut selected = lc.phase_delta_layers.clone().ok_or_else(|| {
1088                        CmfError::Parse(
1089                            "vmf_phase_delta_v1 requires a non-empty phase_delta_layers selector"
1090                                .into(),
1091                        )
1092                    })?;
1093                    if selected.is_empty() {
1094                        return Err(CmfError::Parse(
1095                            "vmf_phase_delta_v1 phase_delta_layers is empty".into(),
1096                        ));
1097                    }
1098                    if arch.layer_types.len() != arch.num_layers {
1099                        return Err(CmfError::Parse(format!(
1100                            "phase_delta layer schedule has {} entries, expected {}",
1101                            arch.layer_types.len(),
1102                            arch.num_layers
1103                        )));
1104                    }
1105                    selected.sort_unstable();
1106                    for pair in selected.windows(2) {
1107                        if pair[0] == pair[1] {
1108                            return Err(CmfError::Parse(format!(
1109                                "phase_delta_layers contains duplicate layer {}",
1110                                pair[0]
1111                            )));
1112                        }
1113                    }
1114                    for &li in &selected {
1115                        if li >= arch.num_layers {
1116                            return Err(CmfError::Parse(format!(
1117                                "phase_delta layer {li} out of range for {} layers",
1118                                arch.num_layers
1119                            )));
1120                        }
1121                        if !matches!(arch.layer_types[li], LayerType::LinearAttention) {
1122                            return Err(CmfError::Parse(format!(
1123                                "phase_delta layer {li} is not a LinearAttention layer"
1124                            )));
1125                        }
1126                    }
1127                    let nphase = need(lc.nphase, "linear_core.nphase")?;
1128                    if lc.num_heads == 0 || nphase == 0 || lc.value_head_dim == 0 {
1129                        return Err(CmfError::Parse(
1130                            "vmf_phase_delta_v1 requires positive heads, nphase, and value_head_dim"
1131                                .into(),
1132                        ));
1133                    }
1134                    phase_delta_layers = Some(selected);
1135                    vmf_cfg = Some(VmfPhaseCfg {
1136                        num_heads: lc.num_heads,
1137                        nphase,
1138                        value_head_dim: lc.value_head_dim,
1139                        hidden_size: arch.hidden_size,
1140                        // Phase-Delta's normalized features are a closed
1141                        // operator contract. CMF_PHASE_MASS is a legacy
1142                        // tuning knob and is intentionally ignored by the
1143                        // phase_delta branch in linear_core.
1144                        phase_mass: 0.0,
1145                    });
1146                }
1147                "gated_delta_net" => {
1148                    if lc.phase_delta_layers.is_some() {
1149                        return Err(CmfError::Parse(
1150                            "gated_delta_net cannot carry phase_delta_layers".into(),
1151                        ));
1152                    }
1153                    gdn_cfg = Some(GdnCfg {
1154                        num_v_heads: lc.num_heads,
1155                        num_k_heads: need(arch.linear_num_key_heads, "linear_num_key_heads")?,
1156                        key_head_dim: need(arch.linear_key_head_dim, "linear_key_head_dim")?,
1157                        value_head_dim: lc.value_head_dim,
1158                        conv_kernel: need(arch.linear_conv_kernel_dim, "linear_conv_kernel_dim")?,
1159                        hidden_size: arch.hidden_size,
1160                        rms_eps: arch.rms_norm_eps,
1161                        output_gate_sigmoid: false,
1162                    });
1163                }
1164                other => {
1165                    return Err(CmfError::Parse(format!(
1166                        "unknown linear core '{other}' (this runtime executes: \
1167                         gated_delta_net, vmf_phase, vmf_phase_delta_v1)"
1168                    )));
1169                }
1170            }
1171        }
1172
1173        // ── KDA geometry (Kimi Linear / Kimi-K3 delta-attention layers) ──
1174        let has_kda = arch.layer_types.iter().any(|t| matches!(t, LayerType::Kda));
1175        let kda_cfg = if has_kda {
1176            let need = |v: Option<usize>, name: &str| {
1177                v.ok_or_else(|| CmfError::Parse(format!("KDA core needs arch.{name}")))
1178            };
1179            Some(crate::linear_core::KdaCfg {
1180                num_heads: need(arch.linear_num_key_heads, "linear_num_key_heads")?,
1181                head_k_dim: need(arch.linear_key_head_dim, "linear_key_head_dim")?,
1182                head_v_dim: need(arch.linear_value_head_dim, "linear_value_head_dim")?,
1183                conv_kernel: need(arch.linear_conv_kernel_dim, "linear_conv_kernel_dim")?,
1184                hidden_size: arch.hidden_size,
1185                rms_eps: arch.rms_norm_eps,
1186            })
1187        } else {
1188            None
1189        };
1190
1191        // ── Short-convolution geometry (LFM2 conv mixer layers) ──
1192        let has_short_conv = arch
1193            .layer_types
1194            .iter()
1195            .any(|t| matches!(t, LayerType::ShortConv));
1196        let short_conv_cfg = if has_short_conv {
1197            Some(ShortConvCfg {
1198                hidden_size: arch.hidden_size,
1199                kernel: arch.linear_conv_kernel_dim.ok_or_else(|| {
1200                    CmfError::Parse(
1201                        "model has ShortConv layers but no arch.linear_conv_kernel_dim — \
1202                         reconvert with the current converter"
1203                            .into(),
1204                    )
1205                })?,
1206            })
1207        } else {
1208            None
1209        };
1210
1211        // ── Layers ──
1212        let load_full_attn = |prefix: &str, layer: Option<usize>| -> Result<AttnKind, CmfError> {
1213            let t = |suffix: &str| load_matrix(model, &format!("{prefix}{suffix}"), force_f32, ov);
1214            let n = |suffix: &str| -> Option<Vec<f32>> {
1215                model
1216                    .tensor(&format!("{prefix}{suffix}"))
1217                    .and_then(|_| load_f32(model, &format!("{prefix}{suffix}"), ov).ok())
1218            };
1219            // DeepSeek-V2 MLA: the latent projections replace the k/v pair.
1220            if let Some(mla) = arch.mla.as_ref() {
1221                // Compressed q (K3/V3): q_a → rms → q_b; direct otherwise.
1222                let (q_proj, q_a, q_a_norm) = if mla.q_lora_rank.is_some() {
1223                    (
1224                        t("self_attn.q_b_proj.weight")?,
1225                        Some(t("self_attn.q_a_proj.weight")?),
1226                        Some(n("self_attn.q_a_layernorm.weight").ok_or_else(|| {
1227                            CmfError::Parse(format!("{prefix}: MLA needs q_a_layernorm"))
1228                        })?),
1229                    )
1230                } else {
1231                    (t("self_attn.q_proj.weight")?, None, None)
1232                };
1233                let hd = mla.qk_rope_head_dim + mla.qk_nope_head_dim;
1234                let nh = q_proj.rows() / hd;
1235                // YaRN mscale²: DeepSeek corrects the softmax scale by
1236                // (0.1·mscale_all_dim·ln(factor)+1)².
1237                let mut scale = 1.0 / (hd as f32).sqrt();
1238                if let Some(y) = arch.yarn.as_ref() {
1239                    if let Some(m) = y.mscale_all_dim.filter(|&m| m > 0.0) {
1240                        let ms = 0.1 * m * y.factor.ln() + 1.0;
1241                        scale *= ms * ms;
1242                    }
1243                }
1244                return Ok(AttnKind::Mla(Box::new(crate::pipeline::MlaWeights {
1245                    q_proj,
1246                    q_a,
1247                    q_a_norm,
1248                    kv_a: t("self_attn.kv_a_proj_with_mqa.weight")?,
1249                    kv_a_norm: n("self_attn.kv_a_layernorm.weight").ok_or_else(|| {
1250                        CmfError::Parse(format!("{prefix}: MLA needs kv_a_layernorm"))
1251                    })?,
1252                    kv_b: t("self_attn.kv_b_proj.weight")?,
1253                    o_proj: t("self_attn.o_proj.weight")?,
1254                    nh,
1255                    qk_rope: mla.qk_rope_head_dim,
1256                    qk_nope: mla.qk_nope_head_dim,
1257                    v_dim: mla.v_head_dim,
1258                    lora: mla.kv_lora_rank,
1259                    scale,
1260                    nope: mla.nope,
1261                })));
1262            }
1263            let wq = t("self_attn.q_proj.weight")?;
1264            let nh = layer
1265                .and_then(|li| {
1266                    arch.attention_heads_per_layer
1267                        .as_ref()
1268                        .and_then(|v| v.get(li).copied())
1269                })
1270                .unwrap_or(arch.num_attention_heads);
1271            // Qwen3.5 output gate: q_proj rows = 2·nh·hd (per-head [q; gate]).
1272            // Gemma-4 global layers legitimately have nh·global_head_dim
1273            // rows (which can equal 2·nh·hd) — never gated.
1274            let output_gate = arch.global_head_dim.is_none() && wq.rows() == 2 * nh * arch.head_dim;
1275            // Gemma-4 global layers run MQA at global_head_dim — their
1276            // q_proj legitimately carries nh·ghd rows.
1277            let is_global_layer = arch.global_head_dim.is_some()
1278                && layer.is_some_and(|li| {
1279                    arch.sliding_window_pattern
1280                        .is_some_and(|p| p > 0 && (li + 1) % p == 0)
1281                });
1282            let expect = if is_global_layer {
1283                nh * arch.global_head_dim.unwrap_or(arch.head_dim)
1284            } else {
1285                nh * arch.head_dim
1286            };
1287            if !output_gate && wq.rows() != expect {
1288                return Err(CmfError::Parse(format!(
1289                    "{prefix}self_attn.q_proj.weight rows={} != heads({nh}) * head_dim({})",
1290                    wq.rows(),
1291                    expect / nh.max(1)
1292                )));
1293            }
1294            let gate_name = format!("{prefix}self_attn.g_proj.weight");
1295            let softplus_gate = if model.tensor(&gate_name).is_some() {
1296                let gate = load_matrix(model, &gate_name, force_f32, ov)?;
1297                if gate.cols() != arch.hidden_size {
1298                    return Err(CmfError::Parse(format!(
1299                        "{gate_name} cols={} != hidden_size ({})",
1300                        gate.cols(),
1301                        arch.hidden_size
1302                    )));
1303                }
1304                let per_head = if gate.rows() == nh {
1305                    true
1306                } else if gate.rows() == nh * arch.head_dim {
1307                    false
1308                } else {
1309                    return Err(CmfError::Parse(format!(
1310                        "{gate_name} rows={} must equal heads ({nh}) or heads*head_dim ({})",
1311                        gate.rows(),
1312                        nh * arch.head_dim
1313                    )));
1314                };
1315                Some((gate, per_head))
1316            } else {
1317                None
1318            };
1319            // Qwen2-family projection biases (by tensor presence).
1320            let bias = match (
1321                n("self_attn.q_proj.bias"),
1322                n("self_attn.k_proj.bias"),
1323                n("self_attn.v_proj.bias"),
1324            ) {
1325                (Some(a), Some(b), Some(c)) => Some((a, b, c)),
1326                _ => None,
1327            };
1328            let wk = t("self_attn.k_proj.weight")?;
1329            let wv = t("self_attn.v_proj.weight")?;
1330            let wo = t("self_attn.o_proj.weight")?;
1331            // K/V/O geometry, checked here and never at run time: a
1332            // mapped matvec accepts any `out`/`x` at least as long as its
1333            // rows/cols, so a V narrower than the cache's head or an
1334            // o_proj wider than nh·v would otherwise run silently wrong.
1335            // KV heads come from the header — per layer when the model
1336            // varies them (MiMo-V2: 4 full / 8 sliding), Gemma-4 global
1337            // layers carry their own (heads, width).
1338            let (nkv_l, hd_l) = if is_global_layer {
1339                (
1340                    arch.num_global_kv_heads.unwrap_or(arch.num_kv_heads),
1341                    arch.global_head_dim.unwrap_or(arch.head_dim),
1342                )
1343            } else {
1344                (
1345                    layer
1346                        .and_then(|li| {
1347                            arch.kv_heads_per_layer
1348                                .as_ref()
1349                                .and_then(|v| v.get(li).copied())
1350                        })
1351                        .unwrap_or(arch.num_kv_heads),
1352                    arch.head_dim,
1353                )
1354            };
1355            let vd = if is_global_layer {
1356                hd_l
1357            } else {
1358                arch.v_head_dim.unwrap_or(hd_l)
1359            };
1360            let shape_err = |what: String| {
1361                CmfError::Parse(format!(
1362                    "{prefix}self_attn: {what} (heads {nh}, kv heads {nkv_l}, head_dim {hd_l}, \
1363                     v_head_dim {vd}) — file and header disagree"
1364                ))
1365            };
1366            if nkv_l == 0 || nh % nkv_l != 0 {
1367                return Err(shape_err(format!(
1368                    "{nkv_l} KV heads must be nonzero and divide {nh} Q heads"
1369                )));
1370            }
1371            if vd == 0 || vd > hd_l {
1372                return Err(shape_err(format!("v_head_dim {vd} must be in 1..={hd_l}")));
1373            }
1374            if wk.rows() != nkv_l * hd_l {
1375                return Err(shape_err(format!(
1376                    "k_proj rows={} != kv_heads*head_dim={}",
1377                    wk.rows(),
1378                    nkv_l * hd_l
1379                )));
1380            }
1381            if wv.rows() != nkv_l * vd {
1382                return Err(shape_err(format!(
1383                    "v_proj rows={} != kv_heads*v_head_dim={}",
1384                    wv.rows(),
1385                    nkv_l * vd
1386                )));
1387            }
1388            if wo.cols() != nh * vd {
1389                return Err(shape_err(format!(
1390                    "o_proj cols={} != heads*v_head_dim={}",
1391                    wo.cols(),
1392                    nh * vd
1393                )));
1394            }
1395            if vd < hd_l && (output_gate || softplus_gate.is_some()) {
1396                return Err(shape_err(
1397                    "an attention output gate with V heads narrower than Q/K is not supported"
1398                        .into(),
1399                ));
1400            }
1401            Ok(AttnKind::Full {
1402                wq,
1403                wk,
1404                wv,
1405                wo,
1406                q_norm: n("self_attn.q_norm.weight"),
1407                k_norm: n("self_attn.k_norm.weight"),
1408                output_gate,
1409                softplus_gate,
1410                bias,
1411            })
1412        };
1413
1414        let load_linear_attn = |prefix: &str, li: usize| -> Result<AttnKind, CmfError> {
1415            if gdn_cfg.is_some() {
1416                // Faithful vendor operator: tensor names 1:1 with the source.
1417                let t = |suffix: &str| {
1418                    load_matrix(
1419                        model,
1420                        &format!("{prefix}linear_attn.{suffix}"),
1421                        force_f32,
1422                        ov,
1423                    )
1424                };
1425                let f = |suffix: &str| {
1426                    load_f32(model, &format!("{prefix}linear_attn.{suffix}"), ov).map_err(err)
1427                };
1428                return Ok(AttnKind::LinearGdn(GdnWeights {
1429                    in_proj_qkv: t("in_proj_qkv.weight")?,
1430                    in_proj_z: t("in_proj_z.weight")?,
1431                    in_proj_a: t("in_proj_a.weight")?,
1432                    in_proj_b: t("in_proj_b.weight")?,
1433                    conv1d: f("conv1d.weight")?,
1434                    a_log: f("A_log")?,
1435                    dt_bias: f("dt_bias")?,
1436                    norm: f("norm.weight")?,
1437                    out_proj: t("out_proj.weight")?,
1438                }));
1439            }
1440            let t = |suffix: &str| {
1441                load_matrix(model, &format!("{prefix}vmf_attn.{suffix}"), force_f32, ov)
1442            };
1443            let a_log = load_f32(model, &format!("{prefix}vmf_attn.A_log"), ov).map_err(err)?;
1444            // Selective-write gate κ (hybrid_k core): optional by tensor
1445            // presence — files without it run the classic phase kernel
1446            // bit-identically.
1447            let k_gate = if model
1448                .tensor(&format!("{prefix}vmf_attn.k_gate.weight"))
1449                .is_some()
1450            {
1451                Some((
1452                    t("k_gate.weight")?,
1453                    load_f32(model, &format!("{prefix}vmf_attn.k_gate.bias"), ov).map_err(err)?,
1454                ))
1455            } else {
1456                None
1457            };
1458            // Optional short causal conv (embryo genomes born with
1459            // --conv-k): `[hidden, 1, k]` taps, flattened.
1460            let conv = if model
1461                .tensor(&format!("{prefix}vmf_attn.conv1d.weight"))
1462                .is_some()
1463            {
1464                Some(load_f32(model, &format!("{prefix}vmf_attn.conv1d.weight"), ov).map_err(err)?)
1465            } else {
1466                None
1467            };
1468            let thq = t("thq.weight")?;
1469            let thk = t("thk.weight")?;
1470            let v_proj = t("v_proj.weight")?;
1471            let out_proj = t("out_proj.weight")?;
1472            let phase_delta = phase_delta_layers
1473                .as_ref()
1474                .is_some_and(|layers| layers.binary_search(&li).is_ok());
1475            if phase_delta {
1476                let cfg = vmf_cfg.ok_or_else(|| {
1477                    CmfError::Parse("phase_delta layer has no VMF geometry".into())
1478                })?;
1479                let expect = |name: &str, got: (usize, usize), want: (usize, usize)| {
1480                    if got != want {
1481                        Err(CmfError::Parse(format!(
1482                            "phase_delta {name} geometry [{}, {}] != [{}, {}]",
1483                            got.0, got.1, want.0, want.1
1484                        )))
1485                    } else {
1486                        Ok(())
1487                    }
1488                };
1489                expect(
1490                    "thq",
1491                    (thq.rows(), thq.cols()),
1492                    (cfg.num_heads * cfg.nphase, cfg.hidden_size),
1493                )?;
1494                expect(
1495                    "thk",
1496                    (thk.rows(), thk.cols()),
1497                    (cfg.num_heads * cfg.nphase, cfg.hidden_size),
1498                )?;
1499                expect(
1500                    "v_proj",
1501                    (v_proj.rows(), v_proj.cols()),
1502                    (cfg.num_heads * cfg.value_head_dim, cfg.hidden_size),
1503                )?;
1504                expect(
1505                    "out_proj",
1506                    (out_proj.rows(), out_proj.cols()),
1507                    (cfg.hidden_size, cfg.num_heads * cfg.value_head_dim),
1508                )?;
1509                if a_log.len() != cfg.num_heads * 2 * cfg.nphase {
1510                    return Err(CmfError::Parse(format!(
1511                        "phase_delta A_log length {} != {}",
1512                        a_log.len(),
1513                        cfg.num_heads * 2 * cfg.nphase
1514                    )));
1515                }
1516                if let Some((kw, kb)) = &k_gate {
1517                    expect(
1518                        "k_gate",
1519                        (kw.rows(), kw.cols()),
1520                        (cfg.num_heads, cfg.hidden_size),
1521                    )?;
1522                    if kb.len() != cfg.num_heads {
1523                        return Err(CmfError::Parse(format!(
1524                            "phase_delta k_gate.bias length {} != {}",
1525                            kb.len(),
1526                            cfg.num_heads
1527                        )));
1528                    }
1529                }
1530                if let Some(c) = &conv {
1531                    let n = model
1532                        .tensor(&format!("{prefix}vmf_attn.conv1d.weight"))
1533                        .ok_or_else(|| {
1534                            CmfError::Parse(
1535                                "phase_delta conv tensor disappeared during load".into(),
1536                            )
1537                        })?;
1538                    if n.shape.len() != 3
1539                        || n.shape[0] != cfg.hidden_size
1540                        || n.shape[1] != 1
1541                        || n.shape[2] < 2
1542                        || c.len() != n.shape.iter().product::<usize>()
1543                    {
1544                        return Err(CmfError::Parse(format!(
1545                            "phase_delta conv geometry {:?} / {} bytes is incompatible",
1546                            n.shape,
1547                            c.len()
1548                        )));
1549                    }
1550                }
1551            }
1552            Ok(AttnKind::Linear(VmfPhaseWeights {
1553                thq,
1554                conv,
1555                thk,
1556                v_proj,
1557                out_proj,
1558                decay: a_log.iter().map(|&a| (-(a as f64).exp()).exp()).collect(),
1559                k_gate,
1560                phase_delta,
1561            }))
1562        };
1563
1564        // LFM2 short-conv mixer: in_proj [3·hidden, hidden], a depthwise
1565        // conv (stored f16 as `[hidden, 1, kernel]` → flattened taps), and
1566        // out_proj [hidden, hidden]. Names canonicalized at convert time.
1567        let load_short_conv = |prefix: &str| -> Result<AttnKind, CmfError> {
1568            let t = |suffix: &str| {
1569                load_matrix(
1570                    model,
1571                    &format!("{prefix}short_conv.{suffix}"),
1572                    force_f32,
1573                    ov,
1574                )
1575            };
1576            Ok(AttnKind::ShortConv(ShortConvWeights {
1577                in_proj: t("in_proj.weight")?,
1578                conv: load_f32(model, &format!("{prefix}short_conv.conv.weight"), ov)
1579                    .map_err(err)?,
1580                out_proj: t("out_proj.weight")?,
1581            }))
1582        };
1583
1584        // KDA layer (Kimi Linear / Kimi-K3): faithful vendor tensors under
1585        // the `kda_attn.` canonical prefix. The output gate is full-rank
1586        // (g_proj, K3) or low-rank (g_a/g_b, Kimi-Linear-48B) by presence.
1587        let load_kda = |prefix: &str| -> Result<AttnKind, CmfError> {
1588            let t = |suffix: &str| {
1589                load_matrix(model, &format!("{prefix}kda_attn.{suffix}"), force_f32, ov)
1590            };
1591            let f = |suffix: &str| {
1592                load_f32(model, &format!("{prefix}kda_attn.{suffix}"), ov).map_err(err)
1593            };
1594            let gate = if model
1595                .tensor(&format!("{prefix}kda_attn.g_proj.weight"))
1596                .is_some()
1597            {
1598                crate::linear_core::KdaOutGate::Full(t("g_proj.weight")?)
1599            } else {
1600                crate::linear_core::KdaOutGate::LowRank(
1601                    t("g_a_proj.weight")?,
1602                    t("g_b_proj.weight")?,
1603                )
1604            };
1605            Ok(AttnKind::Kda(Box::new(crate::linear_core::KdaWeights {
1606                q_proj: t("q_proj.weight")?,
1607                k_proj: t("k_proj.weight")?,
1608                v_proj: t("v_proj.weight")?,
1609                conv_q: f("q_conv1d.weight")?,
1610                conv_k: f("k_conv1d.weight")?,
1611                conv_v: f("v_conv1d.weight")?,
1612                f_a: t("f_a_proj.weight")?,
1613                f_b: t("f_b_proj.weight")?,
1614                dt_bias: f("dt_bias")?,
1615                a_log: f("A_log")?,
1616                b_proj: t("b_proj.weight")?,
1617                gate,
1618                o_norm: f("o_norm.weight")?,
1619                o_proj: t("o_proj.weight")?,
1620                gate_lower_bound: arch.kda_gate_lower_bound.map(|v| v as f32),
1621            })))
1622        };
1623
1624        // Natively bounded anchor (Embryo-O1 `swa_sink_v1`): the plain
1625        // q/k/v/o projections plus the TRAINED sink vectors
1626        // `self_attn.sink_k/sink_v.weight [kvh, S, hd]` (f32, weights —
1627        // not positions). The operator carries no qk-norm, gate or bias:
1628        // their presence is a contract violation, not something to
1629        // silently execute differently from the trainer.
1630        let load_bounded_attn = |prefix: &str, li: usize| -> Result<AttnKind, CmfError> {
1631            let ac = arch.anchor_core.as_ref().ok_or_else(|| {
1632                CmfError::Parse(format!(
1633                    "layer {li} is BoundedAttention but the header carries no anchor_core"
1634                ))
1635            })?;
1636            let t = |suffix: &str| load_matrix(model, &format!("{prefix}{suffix}"), force_f32, ov);
1637            let (nkv, hd, nh) = (arch.num_kv_heads, arch.head_dim, arch.num_attention_heads);
1638            for extra in [
1639                "self_attn.q_norm.weight",
1640                "self_attn.k_norm.weight",
1641                "self_attn.g_proj.weight",
1642                "self_attn.q_proj.bias",
1643                "self_attn.k_proj.bias",
1644                "self_attn.v_proj.bias",
1645            ] {
1646                if model.tensor(&format!("{prefix}{extra}")).is_some() {
1647                    return Err(CmfError::Parse(format!(
1648                        "{prefix}{extra}: the bounded anchor operator carries no qk-norm, \
1649                         gate or bias (docs/EMBRYO_BOUNDED_ANCHOR.md §1)"
1650                    )));
1651                }
1652            }
1653            let sink = |name: &str| -> Result<Vec<f32>, CmfError> {
1654                let full = format!("{prefix}self_attn.{name}.weight");
1655                let e = model.tensor(&full).ok_or_else(|| {
1656                    CmfError::MissingTensor(format!(
1657                        "{full} (a bounded anchor needs its trained sinks)"
1658                    ))
1659                })?;
1660                if e.shape.len() != 3
1661                    || e.shape[0] != nkv
1662                    || e.shape[1] != ac.sink
1663                    || e.shape[2] != hd
1664                {
1665                    return Err(CmfError::Parse(format!(
1666                        "{full}: shape {:?} != [{nkv}, {}, {hd}] (num_kv_heads, anchor_core.sink, head_dim)",
1667                        e.shape, ac.sink
1668                    )));
1669                }
1670                load_f32(model, &full, ov).map_err(err)
1671            };
1672            let wq = t("self_attn.q_proj.weight")?;
1673            let wk = t("self_attn.k_proj.weight")?;
1674            let wv = t("self_attn.v_proj.weight")?;
1675            let wo = t("self_attn.o_proj.weight")?;
1676            let expect = |name: &str, got: (usize, usize), want: (usize, usize)| {
1677                if got != want {
1678                    Err(CmfError::Parse(format!(
1679                        "{prefix}self_attn.{name}.weight is {}x{}, expected {}x{}",
1680                        got.0, got.1, want.0, want.1
1681                    )))
1682                } else {
1683                    Ok(())
1684                }
1685            };
1686            expect("q_proj", (wq.rows(), wq.cols()), (nh * hd, arch.hidden_size))?;
1687            expect("k_proj", (wk.rows(), wk.cols()), (nkv * hd, arch.hidden_size))?;
1688            expect("v_proj", (wv.rows(), wv.cols()), (nkv * hd, arch.hidden_size))?;
1689            expect("o_proj", (wo.rows(), wo.cols()), (arch.hidden_size, nh * hd))?;
1690            Ok(AttnKind::Bounded(Box::new(crate::bounded::BoundedWeights {
1691                wq,
1692                wk,
1693                wv,
1694                wo,
1695                sink_k: sink("sink_k")?,
1696                sink_v: sink("sink_v")?,
1697                sink: ac.sink,
1698                window: ac.window,
1699            })))
1700        };
1701
1702        fn anyhow_like(ok: bool) -> Result<(), ()> {
1703            if ok { Ok(()) } else { Err(()) }
1704        }
1705        let mut layers = Vec::with_capacity(arch.num_layers);
1706        let is_g3n = arch.g3n.is_some();
1707        // Architectures that load their own layer stack below. DeepSeek-V4
1708        // has none of the canonical projections — no q/k/v/o_proj, no
1709        // per-layer gate_proj — so the generic loop would demand
1710        // `self_attn.q_proj.weight` and fail before its own loader ever ran.
1711        let owns_its_layers = is_g3n
1712            || arch.arch_name == "deepseek_v4"
1713            || arch.arch_name == "deepseek_v41"
1714            || arch.qwen4_exp.is_some();
1715        for li in 0..(if owns_its_layers { 0 } else { arch.num_layers }) {
1716            let prefix = format!("model.layers.{li}.");
1717            let attn = match arch.layer_types.get(li) {
1718                Some(LayerType::LinearAttention) => load_linear_attn(&prefix, li)?,
1719                Some(LayerType::Kda) => load_kda(&prefix)?,
1720                Some(LayerType::ShortConv) => load_short_conv(&prefix)?,
1721                Some(LayerType::BoundedAttention) => load_bounded_attn(&prefix, li)?,
1722                _ => load_full_attn(&prefix, Some(li))?,
1723            };
1724            // Gemma-2/3 sandwich: `pre_feedforward_layernorm` present →
1725            // it is the pre-FFN norm, and post_attention/post_feedforward
1726            // норms apply to the branch OUTPUTS before their residuals.
1727            let pre_ffn = format!("{prefix}pre_feedforward_layernorm.weight");
1728            let sandwich = model.tensor(&pre_ffn).is_some();
1729            layers.push(LayerWeights {
1730                input_norm: load_f32(model, &format!("{prefix}input_layernorm.weight"), ov)
1731                    .map_err(err)?,
1732                post_norm: if sandwich {
1733                    load_f32(model, &pre_ffn, ov).map_err(err)?
1734                } else {
1735                    load_f32(
1736                        model,
1737                        &format!("{prefix}post_attention_layernorm.weight"),
1738                        ov,
1739                    )
1740                    .map_err(err)?
1741                },
1742                attn_out_norm: if sandwich {
1743                    Some(
1744                        load_f32(
1745                            model,
1746                            &format!("{prefix}post_attention_layernorm.weight"),
1747                            ov,
1748                        )
1749                        .map_err(err)?,
1750                    )
1751                } else {
1752                    None
1753                },
1754                ffn_out_norm: if sandwich {
1755                    Some(
1756                        load_f32(
1757                            model,
1758                            &format!("{prefix}post_feedforward_layernorm.weight"),
1759                            ov,
1760                        )
1761                        .map_err(err)?,
1762                    )
1763                } else {
1764                    None
1765                },
1766                // Gemma-4: learned scalar multiplying the layer output.
1767                layer_scale: model
1768                    .tensor(&format!("{prefix}layer_scalar"))
1769                    .and_then(|_| {
1770                        load_f32(model, &format!("{prefix}layer_scalar"), ov)
1771                            .ok()
1772                            .and_then(|v| v.first().copied())
1773                    }),
1774                // FFN always quantized — masks run sparse on quant bytes.
1775                ffn: build_layer_ffn(model, &arch, li, false, ov)?,
1776                attn,
1777            });
1778        }
1779
1780        // ── MTP head (optional, spec §2.1) ──
1781        //
1782        // The header declaring an MTP head is not the same as the file
1783        // carrying one. DeepSeek-V4's config announces a next-token predictor
1784        // whose weights the converter does not map (they are spelled `mtp.N.*`
1785        // and have none of the canonical projections), so demanding
1786        // `model.mtp.layers.0.self_attn.q_proj.weight` failed a model that is
1787        // otherwise complete. Presence in the directory decides.
1788        let mtp_present = model
1789            .tensor("model.mtp.layers.0.self_attn.q_proj.weight")
1790            .is_some()
1791            || model.tensor("model.mtp.eh_proj.weight").is_some();
1792        // DeepSeek-V4 writes its own stack under `model.mtp.N.*` — three full
1793        // layers, not a V3-style single block — so it cannot go through the
1794        // path below and is loaded by the dsv4 arm instead. Saying the file
1795        // "carries none" was a false negative worth six gigabytes.
1796        let dsv4_mtp = model.tensor("model.mtp.0.main_proj.weight").is_some();
1797        if arch.mtp.is_some() && !mtp_present && !dsv4_mtp {
1798            tracing::info!(
1799                "header declares an MTP head but the file carries none — \
1800                 loading without it"
1801            );
1802        }
1803        // MiMo-V2 carries its own three-layer stack (sidecar or main file),
1804        // loaded by `mimo_mtp::load_for` below — never this single block.
1805        let mtp = if let Some(cfg) = arch
1806            .mtp
1807            .as_ref()
1808            .filter(|_| mtp_present && arch.arch_name != "mimo_v2")
1809        {
1810            if cfg.num_layers != 1 {
1811                return Err(CmfError::Parse(format!(
1812                    "MTP with {} blocks not supported yet (only 1)",
1813                    cfg.num_layers
1814                )));
1815            }
1816            let p = "model.mtp.";
1817            let attn = load_full_attn("model.mtp.layers.0.", None)?;
1818            Some(MtpModule {
1819                enorm: load_f32(model, &format!("{p}enorm.weight"), ov).map_err(err)?,
1820                hnorm: load_f32(model, &format!("{p}hnorm.weight"), ov).map_err(err)?,
1821                eh_proj: load_matrix(model, &format!("{p}eh_proj.weight"), false, ov)?,
1822                layer: LayerWeights {
1823                    attn_out_norm: None,
1824                    ffn_out_norm: None,
1825                    layer_scale: None,
1826                    input_norm: load_f32(model, &format!("{p}layers.0.input_layernorm.weight"), ov)
1827                        .map_err(err)?,
1828                    post_norm: load_f32(
1829                        model,
1830                        &format!("{p}layers.0.post_attention_layernorm.weight"),
1831                        ov,
1832                    )
1833                    .map_err(err)?,
1834                    // Whatever the block actually carries: DeepSeek's MTP
1835                    // layer is dense, Qwen3.6's is a full MoE (router + 256
1836                    // experts + shared). Same builder as a backbone layer.
1837                    ffn: build_ffn_at(model, &arch, &format!("{p}layers.0."), false, ov)?,
1838                    attn,
1839                },
1840                final_norm: load_f32(model, &format!("{p}norm.weight"), ov).map_err(err)?,
1841                kv: LayerKvCache::new(arch.num_kv_heads, arch.head_dim),
1842            })
1843        } else {
1844            None
1845        };
1846
1847        tracing::info!(
1848            "Pipeline loaded: {} | {}L ({} linear) | {:.2}B params | storage: {} | MTP: {}",
1849            arch.arch_name,
1850            arch.num_layers,
1851            arch.layer_types
1852                .iter()
1853                .filter(|t| matches!(t, LayerType::LinearAttention))
1854                .count(),
1855            model.total_param_count() as f64 / 1e9,
1856            if force_f32 {
1857                "f32 (masked)"
1858            } else {
1859                "quantized mmap"
1860            },
1861            if mtp.is_some() { "yes" } else { "no" }
1862        );
1863
1864        // KV window: the descriptor's max, capped for dev-box safety;
1865        // CMF_MAX_SEQ overrides the cap (long-context runs).
1866        // 8192 was the silent quality cliff of the Qwen3.8 bring-up: at
1867        // the cap the wgpu token graph declines, the host evicts half the
1868        // KV, and a GDN hybrid's recurrent state goes stale — the model
1869        // stays fluent and loses its mind (Django internals, a Turkish
1870        // essay, an em-dash loop; one failure, three costumes). 32768
1871        // covers every long-form run we actually ship while keeping the
1872        // graph's device KV mirror affordable beside the weights;
1873        // CMF_MAX_SEQ still overrides in either direction.
1874        let cap = std::env::var("CMF_MAX_SEQ")
1875            .ok()
1876            .and_then(|v| v.parse::<usize>().ok())
1877            .unwrap_or(32_768);
1878        let max_seq_len = arch.max_position_embeddings.min(cap);
1879
1880        // Looped Transformer: total virtual layers = physical × num_loops.
1881        let total_layers = arch.num_layers * arch.num_loops;
1882
1883        let mut pipeline = Pipeline::new(
1884            tokenizer,
1885            PipelineWeights {
1886                embed_tokens,
1887                layers,
1888                lm_head,
1889                final_norm,
1890            },
1891            arch.hidden_size,
1892            arch.intermediate_size,
1893            arch.num_attention_heads,
1894            arch.num_kv_heads,
1895            arch.head_dim,
1896            total_layers,
1897            arch.num_layers, // physical layers in weights
1898            arch.loop_final_norm,
1899            arch.vocab_size,
1900            arch.rms_norm_eps,
1901            arch.rope_theta as f32,
1902            arch.norm_style,
1903            max_seq_len,
1904            sampler_config,
1905        );
1906        let rotary = ((arch.head_dim as f32 * arch.partial_rotary_factor) as usize).max(2);
1907        pipeline.set_rotary(rotary, arch.rope_theta as f32);
1908        pipeline.attention_heads_per_layer = arch.attention_heads_per_layer.clone();
1909        if let Some(yarn) = &arch.yarn {
1910            pipeline.inv_freq = std::sync::Arc::new(crate::attention::yarn_inv_freq(
1911                rotary,
1912                arch.rope_theta as f32,
1913                yarn.factor,
1914                yarn.original_max_position_embeddings,
1915                yarn.beta_fast,
1916                yarn.beta_slow,
1917            ));
1918            pipeline.rope_scale = yarn.attention_factor;
1919        }
1920        // Gemma-family extras: embedding scale, attention-scale
1921        // override, and (Gemma-3) sliding-window layers with their own
1922        // local RoPE base.
1923        pipeline.embed_multiplier = arch.embed_multiplier;
1924        pipeline.logit_multiplier = arch.logit_multiplier;
1925        if let Some(qpas) = arch.query_pre_attn_scalar {
1926            pipeline.attn_scale = 1.0 / (qpas as f32).sqrt();
1927        }
1928        if let (Some(w), Some(p)) = (arch.sliding_window, arch.sliding_window_pattern) {
1929            pipeline.swa = Some((w, p));
1930            if let Some(base) = arch.rope_local_base_freq {
1931                pipeline.inv_freq_local = Some(std::sync::Arc::new(
1932                    crate::attention::rope_inv_freq(rotary, base as f32),
1933                ));
1934            }
1935        }
1936        let explicit_sliding: Vec<bool> = arch
1937            .layer_types
1938            .iter()
1939            .map(|t| matches!(t, cortiq_core::LayerType::SlidingAttention))
1940            .collect();
1941        if explicit_sliding.iter().any(|&v| v) {
1942            pipeline.sliding_layers = Some(explicit_sliding);
1943            if let Some(w) = arch.sliding_window {
1944                pipeline.swa = Some((w, usize::MAX));
1945            }
1946            let local_rotary = ((arch.head_dim as f32
1947                * arch
1948                    .local_partial_rotary_factor
1949                    .unwrap_or(arch.partial_rotary_factor))
1950                as usize)
1951                .max(2);
1952            pipeline.rotary_dim_local = Some(local_rotary);
1953            if let Some(base) = arch.rope_local_base_freq {
1954                pipeline.inv_freq_local = Some(std::sync::Arc::new(
1955                    crate::attention::rope_inv_freq(local_rotary, base as f32),
1956                ));
1957            }
1958        }
1959        // Gemma-4: global layers run their own geometry (MQA at
1960        // global_head_dim) with a proportional RoPE — the first
1961        // factor·head_dim dims rotate, the zero-padded tail is identity.
1962        if let (Some(ghd), Some(gkv)) = (arch.global_head_dim, arch.num_global_kv_heads) {
1963            pipeline.global_attn = Some((ghd, gkv));
1964            let prf = arch.global_partial_rotary_factor.unwrap_or(1.0);
1965            let half = ghd / 2;
1966            let ra = (((prf * ghd as f32) as usize) / 2).min(half);
1967            let mut f = vec![0.0f32; half];
1968            for (i, slot) in f.iter_mut().enumerate().take(ra) {
1969                *slot = 1.0 / (arch.rope_theta as f32).powf(2.0 * i as f32 / ghd as f32);
1970            }
1971            pipeline.inv_freq_global = Some(std::sync::Arc::new(f));
1972            // Re-shape the global layers' KV storage to their geometry.
1973            // An explicit layer_types map wins over the numeric pattern
1974            // (explicit tags set swa's pattern to usize::MAX, which
1975            // would otherwise leave every global cache mis-shaped).
1976            let global_at = |li: usize| -> bool {
1977                match &pipeline.sliding_layers {
1978                    Some(map) => !map.get(li).copied().unwrap_or(false),
1979                    None => pipeline
1980                        .swa
1981                        .map(|(_, p)| p > 0 && p != usize::MAX && (li + 1) % p == 0)
1982                        .unwrap_or(false),
1983                }
1984            };
1985            for li in 0..arch.num_layers {
1986                if global_at(li) {
1987                    pipeline.kv_cache.layers[li] = crate::kv_cache::LayerKvCache::new(gkv, ghd);
1988                }
1989            }
1990        }
1991        // MLA (DeepSeek-V2): the expand-to-MHA cache holds nh heads of
1992        // rope+nope dims; rotary covers the rope prefix.
1993        if let Some(mla) = arch.mla.as_ref() {
1994            let hd = mla.qk_rope_head_dim + mla.qk_nope_head_dim;
1995            pipeline.head_dim = hd;
1996            pipeline.num_kv_heads = arch.num_attention_heads;
1997            pipeline.rotary_dim = mla.qk_rope_head_dim;
1998            let half = mla.qk_rope_head_dim / 2;
1999            let mut f = vec![0.0f32; half];
2000            for (i, slot) in f.iter_mut().enumerate() {
2001                *slot = 1.0
2002                    / (arch.rope_theta as f32).powf(2.0 * i as f32 / mla.qk_rope_head_dim as f32);
2003            }
2004            pipeline.inv_freq = std::sync::Arc::new(f);
2005            for li in 0..arch.num_layers {
2006                pipeline.kv_cache.layers[li] =
2007                    crate::kv_cache::LayerKvCache::new(arch.num_attention_heads, hd);
2008            }
2009        }
2010        // Per-layer KV geometry (MiMo-V2): KV heads per layer and V heads
2011        // narrower than Q/K. The per-layer shapes were checked against the
2012        // header in load_full_attn; this installs the geometry and gives
2013        // every layer whose KV head count differs its own cache shape.
2014        if !owns_its_layers {
2015            pipeline
2016                .set_attn_geometry(arch.kv_heads_per_layer.clone(), arch.v_head_dim)
2017                .map_err(|e| CmfError::Parse(format!("attention geometry: {e}")))?;
2018        } else if arch.kv_heads_per_layer.is_some() || arch.v_head_dim.is_some() {
2019            return Err(CmfError::Parse(format!(
2020                "{}: kv_heads_per_layer / v_head_dim are not supported by its own layer stack",
2021                arch.arch_name
2022            )));
2023        }
2024        // Learned attention sinks (gpt-oss / MiMo-V2 SWA layers), by tensor
2025        // presence: `model.layers.N.self_attn.sinks`, one f32 logit per Q
2026        // head (`attention_sink_bias` is the HF spelling, accepted too).
2027        // They fold into the grouped softmax; every attention path that
2028        // cannot carry them (chunk GEMM attend, O(1), the GPU graphs)
2029        // refuses the layer.
2030        if !owns_its_layers {
2031            for li in 0..arch.num_layers {
2032                let name = [
2033                    format!("model.layers.{li}.self_attn.sinks"),
2034                    format!("model.layers.{li}.self_attn.attention_sink_bias"),
2035                ]
2036                .into_iter()
2037                .find(|n| model.tensor(n).is_some());
2038                if let Some(name) = name {
2039                    let sinks = load_f32(model, &name, ov).map_err(err)?;
2040                    pipeline
2041                        .set_layer_sinks(li, sinks)
2042                        .map_err(|e| CmfError::Parse(format!("{name}: {e}")))?;
2043                }
2044            }
2045        }
2046        if pipeline.kv_heads_per_layer.is_some()
2047            || pipeline.v_head_dim.is_some()
2048            || pipeline.kv_cache.layers.iter().any(|l| l.sinks.is_some())
2049        {
2050            let per_token: usize = pipeline
2051                .kv_cache
2052                .layers
2053                .iter()
2054                .map(|l| 2 * l.num_kv_heads * l.head_dim * std::mem::size_of::<f32>())
2055                .sum();
2056            tracing::info!(
2057                "attention geometry: kv heads per layer {:?}, v_head_dim {:?}, {} sink layer(s); \
2058                 f32 KV {} B/token (V padded to head_dim), cap {} tokens",
2059                pipeline.kv_heads_per_layer,
2060                pipeline.v_head_dim,
2061                pipeline
2062                    .kv_cache
2063                    .layers
2064                    .iter()
2065                    .filter(|l| l.sinks.is_some())
2066                    .count(),
2067                per_token,
2068                pipeline.kv_cache.max_seq_len
2069            );
2070        }
2071        // Per-frequency rope divisors (MiniCPM3 longrope short_factor):
2072        // served at the native window with the trained per-dim factors.
2073        // Applied after every inv_freq build (plain, YaRN, MLA).
2074        if let Some(fac) = &arch.rope_freq_factors {
2075            let mut f = pipeline.inv_freq.as_ref().clone();
2076            for (i, v) in f.iter_mut().enumerate() {
2077                if let Some(&d) = fac.get(i) {
2078                    *v /= d as f32;
2079                }
2080            }
2081            pipeline.inv_freq = std::sync::Arc::new(f);
2082        }
2083        if let Some(selected) = &phase_delta_layers {
2084            for &li in selected {
2085                pipeline.kv_cache.layers[li].set_linear_wire_allowed(false);
2086            }
2087        }
2088        pipeline.attn_v_norm = arch.attn_v_norm;
2089        pipeline.qk_norm_after_rope = arch.qk_norm_after_rope;
2090        pipeline.final_softcap = arch.final_logit_softcapping.map(|c| c as f32);
2091        // Cortiq Embryo hierarchical head: cluster matrix → two-level log-probs.
2092        if let Some(ncl) = arch.head_clusters {
2093            let cm = load_f32(model, "lm_head.clusters.weight", ov).map_err(err)?;
2094            if cm.len() != ncl * arch.hidden_size {
2095                return Err(CmfError::Parse(format!(
2096                    "lm_head.clusters.weight: {} != {ncl}×{}",
2097                    cm.len(),
2098                    arch.hidden_size
2099                )));
2100            }
2101            pipeline.head_clusters = Some(std::sync::Arc::new(cm));
2102        }
2103        pipeline.attn_softcap = arch.attn_logit_softcapping.unwrap_or(0.0) as f32;
2104        pipeline.vmf_cfg = vmf_cfg;
2105        pipeline.gdn_cfg = gdn_cfg;
2106        pipeline.kda_cfg = kda_cfg;
2107        if arch.qwen4_exp.is_some() {
2108            let (globals, layers, cfg, state) = crate::qwen4_exp::load(model, &arch)?;
2109            pipeline.qwen4_exp = Some(Box::new((globals, layers, cfg, state)));
2110        }
2111        if let Some(gc) = arch.g3n.as_ref() {
2112            use crate::g3n::{G3nAltUp, G3nGlobals, G3nLaurel, G3nLayer};
2113            anyhow_like(gc.altup_num_inputs == crate::g3n::ALTUP_N).map_err(|_| {
2114                CmfError::Parse(format!(
2115                    "g3n: altup_num_inputs {} != supported {}",
2116                    gc.altup_num_inputs,
2117                    crate::g3n::ALTUP_N
2118                ))
2119            })?;
2120            let t = |name: &str| load_matrix(model, name, force_f32, ov);
2121            let f = |name: &str| load_f32(model, name, ov).map_err(err);
2122            let mut altup_proj = Vec::new();
2123            let mut altup_unembed = Vec::new();
2124            for i in 0..crate::g3n::ALTUP_N - 1 {
2125                altup_proj.push(t(&format!("model.altup_projections.{i}.weight"))?);
2126                altup_unembed.push(t(&format!("model.altup_unembed_projections.{i}.weight"))?);
2127            }
2128            let first_shared = arch.num_layers.saturating_sub(gc.num_kv_shared_layers);
2129            let sliding_of = |li: usize| {
2130                matches!(
2131                    arch.layer_types.get(li),
2132                    Some(cortiq_core::LayerType::SlidingAttention)
2133                )
2134            };
2135            let mut g3n_layers = Vec::with_capacity(arch.num_layers);
2136            for li in 0..arch.num_layers {
2137                let pfx = format!("model.layers.{li}.");
2138                let shared = li >= first_shared && first_shared > 0;
2139                let share_src = if shared {
2140                    let want = sliding_of(li);
2141                    (0..first_shared).rev().find(|&j| sliding_of(j) == want)
2142                } else {
2143                    None
2144                };
2145                g3n_layers.push(G3nLayer {
2146                    altup: G3nAltUp {
2147                        router_norm: f(&format!("{pfx}altup.router_norm.weight"))?,
2148                        modality_router: t(&format!("{pfx}altup.modality_router.weight"))?,
2149                        prediction_coefs: t(&format!("{pfx}altup.prediction_coefs.weight"))?,
2150                        correction_coefs: t(&format!("{pfx}altup.correction_coefs.weight"))?,
2151                        correct_output_scale: f(&format!("{pfx}altup.correct_output_scale"))?,
2152                    },
2153                    laurel: G3nLaurel {
2154                        left: t(&format!("{pfx}laurel.linear_left.weight"))?,
2155                        right: t(&format!("{pfx}laurel.linear_right.weight"))?,
2156                        post_norm: f(&format!("{pfx}laurel.post_laurel_norm.weight"))?,
2157                    },
2158                    input_norm: f(&format!("{pfx}input_layernorm.weight"))?,
2159                    post_attn_norm: f(&format!("{pfx}post_attention_layernorm.weight"))?,
2160                    pre_ffw_norm: f(&format!("{pfx}pre_feedforward_layernorm.weight"))?,
2161                    post_ffw_norm: f(&format!("{pfx}post_feedforward_layernorm.weight"))?,
2162                    wq: t(&format!("{pfx}self_attn.q_proj.weight"))?,
2163                    wk: if shared {
2164                        None
2165                    } else {
2166                        Some(t(&format!("{pfx}self_attn.k_proj.weight"))?)
2167                    },
2168                    wv: if shared {
2169                        None
2170                    } else {
2171                        Some(t(&format!("{pfx}self_attn.v_proj.weight"))?)
2172                    },
2173                    wo: t(&format!("{pfx}self_attn.o_proj.weight"))?,
2174                    q_norm: f(&format!("{pfx}self_attn.q_norm.weight"))?,
2175                    k_norm: if shared {
2176                        None
2177                    } else {
2178                        Some(f(&format!("{pfx}self_attn.k_norm.weight"))?)
2179                    },
2180                    kv_share_src: share_src,
2181                    sliding: sliding_of(li),
2182                    gate: t(&format!("{pfx}mlp.gate_proj.weight"))?,
2183                    up: t(&format!("{pfx}mlp.up_proj.weight"))?,
2184                    down: t(&format!("{pfx}mlp.down_proj.weight"))?,
2185                    sparsity: gc.activation_sparsity.get(li).copied().unwrap_or(0.0),
2186                    ple_gate: t(&format!("{pfx}per_layer_input_gate.weight"))?,
2187                    ple_proj: t(&format!("{pfx}per_layer_projection.weight"))?,
2188                    post_ple_norm: f(&format!("{pfx}post_per_layer_input_norm.weight"))?,
2189                });
2190            }
2191            let hd = arch.head_dim;
2192            let globals = G3nGlobals {
2193                altup_proj,
2194                altup_unembed,
2195                ple_embed: t("model.embed_tokens_per_layer.weight")?,
2196                ple_model_proj: t("model.per_layer_model_projection.weight")?,
2197                ple_norm: f("model.per_layer_projection_norm.weight")?,
2198                ple_vocab: gc.ple_vocab,
2199                ple_dim: gc.ple_dim,
2200                num_layers: arch.num_layers,
2201                hidden: arch.hidden_size,
2202                rms_eps: arch.rms_norm_eps,
2203                inv_freq_local: crate::attention::rope_inv_freq(
2204                    hd,
2205                    arch.rope_local_base_freq.unwrap_or(10_000.0) as f32,
2206                ),
2207                inv_freq_global: crate::attention::rope_inv_freq(hd, arch.rope_theta as f32),
2208                window: arch.sliding_window.unwrap_or(512),
2209            };
2210            pipeline.g3n = Some(Box::new((globals, g3n_layers)));
2211        }
2212        // DeepSeek-V4.1: its own stack, selected by the preserved source
2213        // configuration. The generic layer loop cannot represent shared
2214        // CED/CSA2 state or native Engram tables, so loading failure is
2215        // fatal instead of falling back to an unrelated attention layout.
2216        if arch.arch_name == "deepseek_v41" {
2217            let source = arch.deepseek_v41.as_ref().ok_or_else(|| {
2218                CmfError::Parse("deepseek_v41: missing preserved source config".into())
2219            })?;
2220            let tc = source.get("text_config").unwrap_or(source);
2221            let usize_of = |key: &str, fallback: usize| {
2222                tc.get(key)
2223                    .and_then(|v| v.as_u64())
2224                    .map(|v| v as usize)
2225                    .unwrap_or(fallback)
2226            };
2227            let f32_of = |key: &str, fallback: f32| {
2228                tc.get(key)
2229                    .and_then(|v| v.as_f64())
2230                    .map(|v| v as f32)
2231                    .unwrap_or(fallback)
2232            };
2233            let usize_any = |keys: &[&str], fallback: usize| {
2234                keys.iter()
2235                    .find_map(|key| tc.get(key).and_then(|v| v.as_u64()))
2236                    .map(|v| v as usize)
2237                    .unwrap_or(fallback)
2238            };
2239            let f32_any = |keys: &[&str], fallback: f32| {
2240                keys.iter()
2241                    .find_map(|key| tc.get(key).and_then(|v| v.as_f64()))
2242                    .map(|v| v as f32)
2243                    .unwrap_or(fallback)
2244            };
2245            let bool_of = |key: &str, fallback: bool| {
2246                tc.get(key).and_then(|v| v.as_bool()).unwrap_or(fallback)
2247            };
2248            let array_of = |key: &str| -> Vec<usize> {
2249                tc.get(key)
2250                    .and_then(|v| v.as_array())
2251                    .map(|a| {
2252                        a.iter()
2253                            .filter_map(|v| v.as_u64().map(|x| x as usize))
2254                            .collect()
2255                    })
2256                    .unwrap_or_default()
2257            };
2258            let array_alias = |keys: &[&str]| -> Vec<usize> {
2259                keys.iter()
2260                    .find_map(|key| {
2261                        let values = array_of(key);
2262                        (!values.is_empty()).then_some(values)
2263                    })
2264                    .unwrap_or_default()
2265            };
2266            let dim = usize_any(&["hidden_size", "dim"], arch.hidden_size);
2267            let n_layers = usize_any(&["num_hidden_layers", "n_layers"], arch.num_layers);
2268            let n_heads = usize_any(
2269                &["num_attention_heads", "n_heads"],
2270                arch.num_attention_heads,
2271            );
2272            let head_dim = usize_any(&["head_dim"], arch.head_dim);
2273            let rope_head_dim = usize_any(&["rope_head_dim", "qk_rope_head_dim"], 64.min(head_dim));
2274            let moe_inter = usize_any(
2275                &[
2276                    "moe_intermediate_size",
2277                    "moe_inter_dim",
2278                    "intermediate_size",
2279                ],
2280                arch.intermediate_size,
2281            );
2282            let n_experts = usize_any(
2283                &["n_routed_experts"],
2284                arch.moe.as_ref().map(|m| m.num_experts).unwrap_or(384),
2285            );
2286            let top_k = usize_any(
2287                &["num_experts_per_tok", "n_activated_experts"],
2288                arch.moe.as_ref().map(|m| m.top_k).unwrap_or(6),
2289            );
2290            let mut ratios = array_of("compress_ratios");
2291            if ratios.len() >= n_layers {
2292                ratios.truncate(n_layers);
2293            } else {
2294                ratios = (0..n_layers)
2295                    .map(|li| {
2296                        if (2..20).contains(&li) {
2297                            2
2298                        } else if (20..40).contains(&li) {
2299                            1
2300                        } else {
2301                            0
2302                        }
2303                    })
2304                    .collect();
2305            }
2306            let kv_sources = {
2307                let a = array_alias(&["kv_source_layers", "kv_source_layer_ids"]);
2308                if a.is_empty() { vec![2, 8, 14, 20] } else { a }
2309            };
2310            let index_sources = {
2311                let a = array_alias(&["index_source_layers", "index_source_layer_ids"]);
2312                if a.is_empty() {
2313                    vec![2, 8, 14, 20, 24, 28, 32, 36]
2314                } else {
2315                    a
2316                }
2317            };
2318            let engram_layers = array_of("engram_layer_ids");
2319            let engram_embeddings = array_of("engram_num_embeddings");
2320            let cfg = crate::dsv41::Dsv41Cfg {
2321                dim,
2322                n_heads,
2323                head_dim,
2324                rope_head_dim: rope_head_dim.min(head_dim) & !1,
2325                q_lora_rank: usize_of("q_lora_rank", 1280),
2326                o_lora_rank: usize_of("o_lora_rank", 1024),
2327                o_groups: usize_of("o_groups", 8),
2328                hc_mult: usize_of("hc_mult", 4),
2329                hc_sinkhorn_iters: usize_of("hc_sinkhorn_iters", 20),
2330                hc_eps: f32_of("hc_eps", 1e-6),
2331                norm_eps: f32_any(&["norm_eps", "rms_norm_eps"], arch.rms_norm_eps as f32),
2332                n_routed_experts: n_experts,
2333                top_k,
2334                moe_inter,
2335                gate_temp: f32_of("gate_temp", 1.0),
2336                norm_topk_prob: bool_of("norm_topk_prob", true),
2337                route_scale: f32_any(
2338                    &["routed_scaling_factor", "route_scale"],
2339                    arch.moe
2340                        .as_ref()
2341                        .and_then(|m| m.routed_scaling_factor)
2342                        .unwrap_or(1.5),
2343                ),
2344                swiglu_limit: f32_of("swiglu_limit", 10.0),
2345                window: usize_any(
2346                    &["window_size", "sliding_window"],
2347                    arch.sliding_window.unwrap_or(128),
2348                ),
2349                rope_theta: f32_any(&["rope_theta"], arch.rope_theta as f32),
2350                compress_rope_theta: f32_any(&["compress_rope_theta"], 160_000.0),
2351                rope_factor: f32_any(
2352                    &["rope_factor"],
2353                    tc.get("rope_scaling")
2354                        .and_then(|v| v.get("factor"))
2355                        .and_then(|v| v.as_f64())
2356                        .map(|v| v as f32)
2357                        .unwrap_or(16.0),
2358                ),
2359                original_seq_len: usize_any(
2360                    &["original_seq_len", "original_max_position_embeddings"],
2361                    tc.get("rope_scaling")
2362                        .and_then(|v| v.get("original_max_position_embeddings"))
2363                        .and_then(|v| v.as_u64())
2364                        .map(|v| v as usize)
2365                        .unwrap_or(65_536),
2366                ),
2367                beta_fast: f32_any(
2368                    &["beta_fast"],
2369                    tc.get("rope_scaling")
2370                        .and_then(|v| v.get("beta_fast"))
2371                        .and_then(|v| v.as_f64())
2372                        .map(|v| v as f32)
2373                        .unwrap_or(32.0),
2374                ),
2375                beta_slow: f32_any(
2376                    &["beta_slow"],
2377                    tc.get("rope_scaling")
2378                        .and_then(|v| v.get("beta_slow"))
2379                        .and_then(|v| v.as_f64())
2380                        .map(|v| v as f32)
2381                        .unwrap_or(1.0),
2382                ),
2383                index_heads: usize_any(&["index_n_heads", "indexer_n_heads"], 32),
2384                index_head_dim: usize_any(&["index_head_dim", "indexer_head_dim"], 128),
2385                index_topk: usize_any(&["index_topk", "indexer_topk"], 512),
2386                candidate_source: usize_any(
2387                    &["candidate_source_layer", "candidate_source_layer_id"],
2388                    20,
2389                ),
2390                candidate_topk_blocks: usize_any(&["candidate_topk_blocks"], 2048),
2391                candidate_block_size: usize_any(&["candidate_block_size"], 8),
2392                kv_sources,
2393                index_sources,
2394                compress_ratios: ratios,
2395                engram_layers,
2396                engram_vocab: usize_any(&["engram_vocab_size"], 16_000_000),
2397                engram_embeddings,
2398                engram_max_ngram: usize_any(&["engram_max_ngram_size"], 4),
2399                engram_heads: usize_any(&["engram_n_heads"], 8),
2400                engram_head_dim: usize_any(&["engram_head_dim"], 256),
2401                engram_compressed_vocab: usize_any(&["engram_compressed_vocab_size"], 99_092),
2402                engram_pad_id: usize_any(&["engram_pad_id", "engram_pad_token_id"], 2),
2403                vocab: usize_any(&["vocab_size"], arch.vocab_size),
2404            };
2405            let token_map = crate::dsv41::token_map_from_model(model, cfg.vocab);
2406            let (g, dl, hash) = crate::dsv41::load(model, &cfg, n_layers, token_map)
2407                .map_err(|e| CmfError::Parse(format!("deepseek_v41: {e}")))?;
2408            let st = crate::dsv41::Dsv41State::new(&cfg, hash);
2409            // Vision tensors are optional in text-only exports, but when
2410            // present they stay mmap-backed through the dedicated tower.
2411            // Do not make a text-only CMF fail merely because its source
2412            // config still carries the multimodal section.
2413            if let Ok(vision_cfg) = crate::dsv41_vision::VisionConfig::from_source(source) {
2414                if vision_cfg.vision_enabled()
2415                    && model.tensor("vision.patch_embed.proj.weight").is_some()
2416                {
2417                    pipeline.dsv41_vision = Some(
2418                        crate::dsv41_vision::VisionModel::from_model(model, vision_cfg)
2419                            .map_err(|e| CmfError::Parse(format!("deepseek_v41 vision: {e}")))?,
2420                    );
2421                }
2422            }
2423            tracing::info!(
2424                "deepseek_v41: loaded {} layers, {} KV sources, {} index sources, {} Engram layers; experts remain mmap-backed",
2425                dl.len(),
2426                cfg.kv_sources.len(),
2427                cfg.index_sources.len(),
2428                cfg.engram_layers.len()
2429            );
2430            pipeline.dsv41 = Some(Box::new((g, dl, cfg, st)));
2431        }
2432        // DeepSeek-V4: its own stack, selected by the arch name the
2433        // converter wrote. Loading failure is fatal rather than a silent
2434        // fallback — the generic loop cannot represent this model at all,
2435        // so a fallback would decode noise.
2436        if arch.arch_name == "deepseek_v4" {
2437            let moe = arch
2438                .moe
2439                .as_ref()
2440                .ok_or_else(|| CmfError::Parse("deepseek_v4: no moe config".into()))?;
2441            let cfg = crate::dsv4::Dsv4Cfg {
2442                dim: arch.hidden_size,
2443                n_heads: arch.num_attention_heads,
2444                head_dim: arch.head_dim,
2445                // The rope tail: `partial_rotary_factor` carries it when the
2446                // conversion recorded it (rd/head_dim), which the tensors
2447                // cannot reveal. Files converted before that carry 1.0,
2448                // meaning "unset" here rather than "rotate everything" —
2449                // for those the release's 64 stands in, which is what they
2450                // were converted from.
2451                rope_head_dim: if arch.partial_rotary_factor < 1.0 {
2452                    (((arch.head_dim as f32 * arch.partial_rotary_factor) as usize) & !1)
2453                        .clamp(2, arch.head_dim)
2454                } else {
2455                    64.min(arch.head_dim)
2456                },
2457                // The LoRA ranks and the group count ARE visible in the
2458                // weights, and reading them there means a re-tuned
2459                // checkpoint loads without touching this code.
2460                q_lora_rank: 0,
2461                o_lora_rank: 0,
2462                // Derived below from wo_a's shape — the attention output is
2463                // n_heads*head_dim wide and wo_a takes one group of it per
2464                // row block, so groups = width / wo_a.cols(). A pinned 8 is
2465                // right for the release and wrong for anything else, which
2466                // is exactly what made a toy checkpoint impossible to
2467                // compare against the reference.
2468                o_groups: 8,
2469                hc_mult: 4,
2470                hc_sinkhorn_iters: 20,
2471                hc_eps: 1e-6,
2472                norm_eps: arch.rms_norm_eps as f32,
2473                n_routed_experts: moe.num_experts,
2474                top_k: moe.top_k,
2475                moe_inter: moe.moe_intermediate_size,
2476                route_scale: moe.routed_scaling_factor.unwrap_or(1.0),
2477                // config.json's `swiglu_limit`, which the header has no
2478                // field for. The release ships 10.0; a checkpoint that
2479                // retunes it would need this read from the config, so it
2480                // sits next to the other pinned constants rather than
2481                // hiding inside the expert.
2482                swiglu_limit: 10.0,
2483                window: arch.sliding_window.unwrap_or(128),
2484                index_topk: 512,
2485                vocab: arch.vocab_size,
2486            };
2487            let (g, dl) = crate::dsv4::load(model, &cfg, arch.num_layers)
2488                .map_err(|e| CmfError::Parse(format!("deepseek_v4: {e}")))?;
2489            // Read the ranks off the weights that define them: wq_a's
2490            // rows ARE q_lora_rank, and wo_b's columns are groups x
2491            // o_lora_rank. A header field could disagree with the file;
2492            // these cannot.
2493            let mut cfg = cfg;
2494            if let Some(l0) = dl.first() {
2495                cfg.q_lora_rank = l0.wq_a.rows();
2496                let attn_width = arch.num_attention_heads * arch.head_dim;
2497                if l0.wo_a.cols() > 0 && attn_width % l0.wo_a.cols() == 0 {
2498                    cfg.o_groups = (attn_width / l0.wo_a.cols()).max(1);
2499                }
2500                cfg.o_lora_rank = l0.wo_b.cols() / cfg.o_groups.max(1);
2501                cfg.hc_mult = (l0.hc_attn_fn.len() / l0.hc_attn_base.len().max(1)) / cfg.dim.max(1);
2502                if cfg.hc_mult == 0 {
2503                    cfg.hc_mult = 4;
2504                }
2505            }
2506            // RoPE rides only the last `rope_head_dim` of each head, and the
2507            // reference builds its frequencies over THAT width — not over
2508            // head_dim, which is 512 here. The generic path above sized them
2509            // by head_dim, giving 1/base^(2i/512) where 1/base^(2i/64) is
2510            // wanted: every position rotated by the wrong angle.
2511            //
2512            // YaRN is applied unconditionally by the reference (its guard is
2513            // `original_seq_len > 0`, not the sequence length), so it belongs
2514            // in these frequencies too. Older configs spell the key `type`
2515            // rather than `rope_type`; when the header carries no profile the
2516            // release's own numbers stand in, which is better than silently
2517            // decoding with unscaled frequencies.
2518            let (yf, yo, ybf, ybs) = match &arch.yarn {
2519                Some(y) => (
2520                    y.factor,
2521                    y.original_max_position_embeddings,
2522                    y.beta_fast,
2523                    y.beta_slow,
2524                ),
2525                None => {
2526                    tracing::warn!(
2527                        "deepseek_v4: the header carries no YaRN profile — \
2528                         falling back to the release's (factor 16, original \
2529                         65536, beta 32/1). Re-converting with a build that \
2530                         reads rope_scaling.type would make this exact."
2531                    );
2532                    (16.0, 65536, 32.0, 1.0)
2533                }
2534            };
2535            pipeline.inv_freq = std::sync::Arc::new(crate::attention::yarn_inv_freq(
2536                cfg.rope_head_dim,
2537                arch.rope_theta as f32,
2538                yf,
2539                yo,
2540                ybf,
2541                ybs,
2542            ));
2543            // Keep the working set resident. Everything but the routed
2544            // experts is touched by every token, and of the experts only the
2545            // ones the task actually routes to — the page cache cannot know
2546            // that and evicts by age instead.
2547            if let Ok(stats) = std::env::var("CMF_MOE_PIN") {
2548                let cover = std::env::var("CMF_MOE_PIN_COVER")
2549                    .ok()
2550                    .and_then(|v| v.parse::<f64>().ok())
2551                    .filter(|&c| c > 0.0 && c <= 1.0)
2552                    .unwrap_or(0.95);
2553                let hot = crate::pin::hot_experts(&stats, cover);
2554                let mut names: Vec<String> = Vec::new();
2555                for e in &model.tensors {
2556                    let is_expert = e.name.contains(".mlp.experts.");
2557                    if !is_expert {
2558                        names.push(e.name.clone()); // skeleton: always hot
2559                    }
2560                }
2561                let mut kept_experts = 0usize;
2562                if let Some(hot) = &hot {
2563                    for (li, experts) in hot {
2564                        for e in experts {
2565                            for w in ["gate_proj", "up_proj", "down_proj"] {
2566                                names.push(format!("model.layers.{li}.mlp.experts.{e}.{w}.weight"));
2567                            }
2568                            kept_experts += 1;
2569                        }
2570                    }
2571                }
2572                let r = crate::pin::pin_tensors(model, &names);
2573                tracing::info!(
2574                    "закреплено {:.1} ГБ ({} тензоров, горячих экспертов {kept_experts},                      покрытие {cover}); лимит {}",
2575                    r.bytes as f64 / 1e9,
2576                    r.tensors,
2577                    r.limit
2578                        .map(|l| format!("{:.1} ГБ", l as f64 / 1e9))
2579                        .unwrap_or_else(|| "неизвестен".into())
2580                );
2581                if r.skipped > 0 {
2582                    tracing::warn!("не закреплено тензоров: {}", r.skipped);
2583                }
2584            }
2585            let st = crate::dsv4::Dsv4State::new(arch.num_layers);
2586            // The speculation stack, if the file carries one. Reading it is
2587            // metadata only — the expert weights stay in the mapping until a
2588            // draft actually runs — so it costs nothing to know it is there.
2589            let depth = std::env::var("CMF_DSV4_MTP_DEPTH")
2590                .ok()
2591                .and_then(|v| v.parse::<usize>().ok())
2592                .unwrap_or(3);
2593            pipeline.dsv4_mtp = crate::dsv4::load_mtp(model, &cfg, depth);
2594            // Before any trunk pack is built: leave the draft its VRAM.
2595            crate::dsv4::dspark_reserve_note(&pipeline.dsv4_mtp, &cfg, &dl);
2596            pipeline.dsv4 = Some(Box::new((g, dl, cfg, st)));
2597        }
2598        pipeline.short_conv_cfg = short_conv_cfg;
2599        pipeline.mtp = mtp;
2600        if arch.arch_name == "mimo_v2" {
2601            pipeline.mimo_mtp = crate::pipeline::mimo_mtp::load_for(model, &arch)?;
2602        }
2603        pipeline.install_dynamic_routing(model, false);
2604        // Record the load-time overlay so a later set_active_skill(None)
2605        // correctly reverts it (the union-diff assumes dyn_active mirrors
2606        // the live overlay). Blend loads have no single index to revert.
2607        match ov {
2608            Overlay::One(sid) => {
2609                pipeline.dyn_active = model.header.skills.iter().position(|s| &s.id == sid);
2610            }
2611            Overlay::Blend(_) => pipeline.dyn_blend_loaded = true,
2612            Overlay::None => {}
2613        }
2614        // B1: apply the measured confidence-calibration temperature, if the
2615        // file carries one (softmax(logits / T) for reported confidence).
2616        if let Some(c) = &model.header.calibration {
2617            pipeline.set_calib_temp(c.temperature);
2618        }
2619        // Versioned state wire: every layer carries the operator identity
2620        // so a peer holding another operator is refused, not guessed at.
2621        let identity = arch
2622            .linear_core_identity()
2623            .and_then(|v| serde_json::to_vec(&v).ok())
2624            .map(|b| cortiq_core::hash64(&b))
2625            .unwrap_or(0);
2626        pipeline.install_wire_identity(identity);
2627        // Natively bounded anchor: the file's operator, installed from the
2628        // header (rings sized once, not per prompt). The file governs —
2629        // a post-hoc O(1) knob on such a file is an error, not a warning.
2630        if let Some(ac) = &arch.anchor_core {
2631            pipeline.install_bounded(ac).map_err(CmfError::Parse)?;
2632            let why = pipeline.o1_refusal().unwrap_or_default();
2633            let mut knobs: Vec<String> = std::env::vars()
2634                .filter(|(k, v)| {
2635                    (k == "CMF_O1" && !(v == "off" || v == "0")) || k.starts_with("CMF_O1_")
2636                })
2637                .map(|(k, _)| k)
2638                .collect();
2639            knobs.sort();
2640            if !knobs.is_empty() {
2641                return Err(CmfError::Parse(format!("{why} (set: {})", knobs.join(", "))));
2642            }
2643            if model
2644                .header
2645                .provenance
2646                .as_ref()
2647                .and_then(|p| p.get("o1_attn"))
2648                .is_some()
2649            {
2650                return Err(CmfError::Parse(format!(
2651                    "{why} (the header carries a provenance.o1_attn hint)"
2652                )));
2653            }
2654            return Ok(pipeline);
2655        }
2656        // O(1) Nyström attention (runtime-level, no format change):
2657        // env CMF_O1 decides; unset falls through to the converter hint
2658        // in header.provenance.o1_attn (`cortiq convert --o1`), and
2659        // CMF_O1=off force-disables even the hint. CLI flags override
2660        // later via set_o1().
2661        // A genome's attention operator is its arch (hashed into the
2662        // trunk); a provenance hint is not part of the genome and must not
2663        // switch the backbone to Nyström (NF-6 — core refuses the hint on
2664        // a GENOME file at open; this keeps the loader fail-closed too).
2665        let genome_hint = model.header.genome.is_some()
2666            && model
2667                .header
2668                .provenance
2669                .as_ref()
2670                .and_then(|p| p.get("o1_attn"))
2671                .is_some();
2672        if genome_hint {
2673            return Err(CmfError::Parse(
2674                "genome file with a provenance.o1_attn hint: the trunk's attention operator is \
2675                 set by the genome's arch only — refusing"
2676                    .into(),
2677            ));
2678        }
2679        let o1 = match crate::nystrom::o1_from_env() {
2680            crate::nystrom::O1Env::Off => None,
2681            crate::nystrom::O1Env::On(cfg) => Some(cfg),
2682            crate::nystrom::O1Env::Unset => model
2683                .header
2684                .provenance
2685                .as_ref()
2686                .and_then(|p| p.get("o1_attn"))
2687                .and_then(crate::nystrom::O1Cfg::from_json),
2688        };
2689        if o1.is_some() {
2690            if pipeline.attn_softcap > 0.0 {
2691                return Err(CmfError::Parse(
2692                    "--o1 with attention-logit soft-capping (Gemma-2) is not supported: \
2693                     the streaming operator has no capped-score form"
2694                        .into(),
2695                ));
2696            }
2697            pipeline.set_o1(o1);
2698        }
2699        Ok(pipeline)
2700    }
2701
2702    /// Record per-skill dynamic-routing metadata: which FFN layers each
2703    /// skill actually replaces (derived from the tensors present, not
2704    /// the meta `layers` field), and whether the skill is eligible for
2705    /// cheap dynamic switching (FFN-only). Called once at load.
2706    pub(crate) fn install_dynamic_routing(&mut self, model: &Arc<CmfModel>, force_f32: bool) {
2707        self.model = Some(model.clone());
2708        self.dyn_force_f32 = force_f32;
2709        let mut per_skill = Vec::with_capacity(model.header.skills.len());
2710        for sk in &model.header.skills {
2711            // A lookup record replaces nothing: its tensors are the key →
2712            // card table (u64/u32/u8), never weights. No lane, no switch.
2713            if sk.kind.as_deref() == Some(cortiq_core::knowledge::skill_kind::LOOKUP) {
2714                per_skill.push(None);
2715                continue;
2716            }
2717            let mut ffn_layers = std::collections::BTreeSet::new();
2718            let mut non_ffn = false;
2719            let prefix = format!("skill.{}.", sk.id);
2720            for t in model.skill_tensors(&sk.id) {
2721                let rel = &t.name[prefix.len()..]; // e.g. model.layers.20.mlp.down_proj.weight
2722                let toks: Vec<&str> = rel.split('.').collect();
2723                if toks.len() >= 5 && toks[0] == "model" && toks[1] == "layers" && toks[3] == "mlp"
2724                {
2725                    if let Ok(li) = toks[2].parse::<usize>() {
2726                        ffn_layers.insert(li);
2727                        continue;
2728                    }
2729                }
2730                non_ffn = true; // replaces attention / embed / lm_head
2731            }
2732            if non_ffn {
2733                tracing::warn!(
2734                    "skill '{}' replaces non-FFN tensors — excluded from dynamic \
2735                     routing (static overlay still works)",
2736                    sk.id
2737                );
2738                per_skill.push(None);
2739            } else {
2740                per_skill.push(Some(ffn_layers.into_iter().collect::<Vec<_>>()));
2741            }
2742        }
2743        self.dyn_skill_layers = per_skill;
2744    }
2745
2746    /// Switch the overlaid skill for subsequent forwards (dynamic
2747    /// routing). `idx` = index into model.header.skills; None = backbone.
2748    /// Rebuilds the FFN of the union of the old and new skill's touched
2749    /// layers with the new overlay — tensor-source indirection made
2750    /// dynamic. Cheap: Mapped tensors are re-resolved mmap pointers.
2751    /// Result is bit-identical to loading the pipeline with that skill.
2752    ///
2753    /// Asking for the skill that is already active is a true no-op: the
2754    /// weights do not change, so the sequence state (KV, rings, recurrent
2755    /// state, the prefix-reuse key) stays — clearing it here used to wipe
2756    /// a live conversation on every idempotent call. A real change
2757    /// rebuilds the touched FFNs first (all or nothing: on an error the
2758    /// old overlay and state are untouched), then invalidates every state
2759    /// the old weights produced, including `kv_prefix`.
2760    pub fn set_active_skill(&mut self, idx: Option<usize>) -> Result<(), CmfError> {
2761        if self.dyn_active == idx {
2762            return Ok(());
2763        }
2764        let model = self.model.clone().ok_or_else(|| {
2765            CmfError::Parse("dynamic routing needs a model-backed pipeline".into())
2766        })?;
2767        let mut union: std::collections::BTreeSet<usize> = std::collections::BTreeSet::new();
2768        if let Some(old) = self.dyn_active {
2769            if let Some(Some(ls)) = self.dyn_skill_layers.get(old) {
2770                union.extend(ls.iter().copied());
2771            }
2772        }
2773        let new_id: Option<String> = match idx {
2774            Some(n) => match self.dyn_skill_layers.get(n) {
2775                Some(Some(ls)) => {
2776                    union.extend(ls.iter().copied());
2777                    Some(model.header.skills[n].id.clone())
2778                }
2779                _ => {
2780                    return Err(CmfError::Parse(format!(
2781                        "skill index {n} not dynamic-eligible"
2782                    )));
2783                }
2784            },
2785            None => None,
2786        };
2787        let ov = match &new_id {
2788            Some(s) => Overlay::One(s),
2789            None => Overlay::None,
2790        };
2791        let arch = model.arch();
2792        let mut rebuilt = Vec::with_capacity(union.len());
2793        for li in union {
2794            rebuilt.push((
2795                li,
2796                build_layer_ffn(&model, arch, li, self.dyn_force_f32, &ov)?,
2797            ));
2798        }
2799        for (li, ffn) in rebuilt {
2800            self.weights.layers[li].ffn = ffn;
2801        }
2802        // Overlay swap changed weights → every cached state is stale.
2803        self.invalidate_for_weight_change();
2804        self.dyn_active = idx;
2805        Ok(())
2806    }
2807}
2808
2809/// Fixed per-sequence state a file declares, from its header alone (no
2810/// weights loaded): bytes of bounded-anchor rings, bytes of recurrent
2811/// vectors (S + conv rings), and how many layers still hold a growing
2812/// per-position KV. A strictly-O(1) file has `growing_layers == 0`.
2813#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
2814pub struct PerSequenceState {
2815    pub bounded_bytes: usize,
2816    pub recurrent_bytes: usize,
2817    pub growing_layers: usize,
2818}
2819
2820pub fn per_sequence_state_bytes(model: &CmfModel) -> Result<PerSequenceState, String> {
2821    let arch = model.arch();
2822    let f = std::mem::size_of::<f32>();
2823    let mut st = PerSequenceState::default();
2824    let need = |v: Option<usize>, name: &str| {
2825        v.ok_or_else(|| format!("linear core needs arch.{name}"))
2826    };
2827    for li in 0..arch.num_layers {
2828        match arch.layer_types.get(li) {
2829            Some(LayerType::BoundedAttention) => {
2830                let ac = arch
2831                    .anchor_core
2832                    .as_ref()
2833                    .ok_or("BoundedAttention layer without anchor_core")?;
2834                st.bounded_bytes += ac.ring_state_elems(arch.num_kv_heads, arch.head_dim) * f;
2835            }
2836            Some(LayerType::LinearAttention) => {
2837                let lc = arch
2838                    .linear_core
2839                    .as_ref()
2840                    .ok_or("LinearAttention layer without linear_core")?;
2841                let elems = match lc.kind.as_str() {
2842                    "gated_delta_net" => GdnCfg {
2843                        num_v_heads: lc.num_heads,
2844                        num_k_heads: need(arch.linear_num_key_heads, "linear_num_key_heads")?,
2845                        key_head_dim: need(arch.linear_key_head_dim, "linear_key_head_dim")?,
2846                        value_head_dim: lc.value_head_dim,
2847                        conv_kernel: need(arch.linear_conv_kernel_dim, "linear_conv_kernel_dim")?,
2848                        hidden_size: arch.hidden_size,
2849                        rms_eps: arch.rms_norm_eps,
2850                        output_gate_sigmoid: false,
2851                    }
2852                    .state_len(),
2853                    _ => {
2854                        let nphase = need(lc.nphase, "linear_core.nphase")?;
2855                        let mut n = lc.num_heads * 2 * nphase * lc.value_head_dim;
2856                        if let Some(c) = model.tensor(&format!("model.layers.{li}.vmf_attn.conv1d.weight")) {
2857                            if c.shape.len() == 3 && c.shape[2] >= 2 {
2858                                n += (c.shape[2] - 1) * arch.hidden_size;
2859                            }
2860                        }
2861                        n
2862                    }
2863                };
2864                st.recurrent_bytes += elems * f;
2865            }
2866            Some(LayerType::ShortConv) => {
2867                let k = need(arch.linear_conv_kernel_dim, "linear_conv_kernel_dim")?;
2868                st.recurrent_bytes += ShortConvCfg {
2869                    hidden_size: arch.hidden_size,
2870                    kernel: k,
2871                }
2872                .state_len()
2873                    * f;
2874            }
2875            Some(LayerType::Kda) => {
2876                st.recurrent_bytes += crate::linear_core::KdaCfg {
2877                    num_heads: need(arch.linear_num_key_heads, "linear_num_key_heads")?,
2878                    head_k_dim: need(arch.linear_key_head_dim, "linear_key_head_dim")?,
2879                    head_v_dim: need(arch.linear_value_head_dim, "linear_value_head_dim")?,
2880                    conv_kernel: need(arch.linear_conv_kernel_dim, "linear_conv_kernel_dim")?,
2881                    hidden_size: arch.hidden_size,
2882                    rms_eps: arch.rms_norm_eps,
2883                }
2884                .state_len()
2885                    * f;
2886            }
2887            _ => st.growing_layers += 1,
2888        }
2889    }
2890    Ok(st)
2891}