1use crate::kv_cache::LayerKvCache;
15use crate::linear_core::{
16 GdnCfg, GdnWeights, ShortConvCfg, ShortConvWeights, VmfPhaseCfg, VmfPhaseWeights,
17};
18use crate::pipeline::{
19 AttnKind, DenseFfn, FfnKind, LayerWeights, MoeFfn, MtpModule, Pipeline, PipelineWeights,
20};
21use crate::qtensor::QTensor;
22use crate::sampler::SamplerConfig;
23use crate::tokenizer::Tokenizer;
24use cortiq_core::quant::dequant_tensor;
25use cortiq_core::{CmfError, CmfModel, LayerType, ModelArch};
26use std::sync::Arc;
27
28pub enum Overlay<'a> {
31 None,
32 One(&'a str),
33 Blend(&'a [(String, f32)]),
35}
36
37impl Overlay<'_> {
38 fn blend_touches(&self, model: &CmfModel, name: &str) -> bool {
39 match self {
40 Overlay::Blend(list) => list
41 .iter()
42 .any(|(sid, _)| model.tensor(&format!("skill.{sid}.{name}")).is_some()),
43 _ => false,
44 }
45 }
46}
47
48fn dequant_by_name(model: &CmfModel, name: &str) -> Result<Vec<f32>, String> {
49 let entry = model
50 .tensor(name)
51 .ok_or_else(|| format!("tensor '{name}' not found in CMF directory"))?;
52 let mut out = vec![0.0f32; entry.n_elems()];
53 dequant_tensor(entry, model.entry_bytes(entry), &mut out)?;
54 Ok(out)
55}
56
57fn blend_f32(model: &CmfModel, name: &str, list: &[(String, f32)]) -> Result<Vec<f32>, String> {
60 let mut acc: Option<Vec<f32>> = None;
61 for (sid, w) in list {
62 let sname = format!("skill.{sid}.{name}");
63 let src = if model.tensor(&sname).is_some() {
64 &sname
65 } else {
66 name
67 };
68 let t = dequant_by_name(model, src)?;
69 match &mut acc {
70 None => {
71 let mut t = t;
72 for v in t.iter_mut() {
73 *v *= w;
74 }
75 acc = Some(t);
76 }
77 Some(a) => {
78 for (av, tv) in a.iter_mut().zip(&t) {
79 *av += w * tv;
80 }
81 }
82 }
83 }
84 acc.ok_or_else(|| "empty blend".into())
85}
86
87pub(crate) fn load_f32(model: &CmfModel, name: &str, ov: &Overlay) -> Result<Vec<f32>, String> {
89 if ov.blend_touches(model, name) {
90 if let Overlay::Blend(list) = ov {
91 return blend_f32(model, name, list);
92 }
93 }
94 let skill = match ov {
95 Overlay::One(s) => Some(*s),
96 _ => None,
97 };
98 let entry = model
99 .resolve_tensor(name, skill)
100 .ok_or_else(|| format!("tensor '{name}' not found in CMF directory"))?;
101 let bytes = model.entry_bytes(entry);
102 let mut out = vec![0.0f32; entry.n_elems()];
103 dequant_tensor(entry, bytes, &mut out)?;
104 Ok(out)
105}
106
107pub(crate) fn build_layer_ffn(
113 model: &Arc<CmfModel>,
114 arch: &ModelArch,
115 li: usize,
116 force_f32: bool,
117 ov: &Overlay,
118) -> Result<FfnKind, CmfError> {
119 build_ffn_at(model, arch, &format!("model.layers.{li}."), force_f32, ov)
120}
121
122pub(crate) fn build_ffn_at(
127 model: &Arc<CmfModel>,
128 arch: &ModelArch,
129 prefix: &str,
130 force_f32: bool,
131 ov: &Overlay,
132) -> Result<FfnKind, CmfError> {
133 let prefix = prefix.to_string();
134 let load_dense = |p: &str| -> Result<DenseFfn, CmfError> {
135 let gate_proj = load_matrix(model, &format!("{p}gate_proj.weight"), force_f32, ov)?;
136 let up_proj = load_matrix(model, &format!("{p}up_proj.weight"), force_f32, ov)?;
137 let down_proj = load_matrix(model, &format!("{p}down_proj.weight"), force_f32, ov)?;
138 let inter = gate_proj.rows();
142 if up_proj.rows() != inter || down_proj.cols() != inter {
143 return Err(CmfError::Parse(format!(
144 "{p}: FFN dims disagree (gate.rows={inter}, up.rows={}, \
145 down.cols={}); all three must equal inter'",
146 up_proj.rows(),
147 down_proj.cols()
148 )));
149 }
150 if down_proj.rows() != arch.hidden_size {
151 return Err(CmfError::Parse(format!(
152 "{p}: down_proj.rows={} != hidden_size={}",
153 down_proj.rows(),
154 arch.hidden_size
155 )));
156 }
157 Ok(DenseFfn {
158 gate_proj,
159 up_proj,
160 down_proj,
161 act: crate::pipeline::Act::from_arch_full(arch),
162 })
163 };
164 let router_name = format!("{prefix}mlp.gate.weight");
165 if model.tensor(&router_name).is_none() {
166 return Ok(FfnKind::Dense(load_dense(&format!("{prefix}mlp."))?));
167 }
168 let cfg = arch.moe.as_ref().ok_or_else(|| {
169 CmfError::Parse(format!(
170 "{router_name} present but header has no arch.moe block"
171 ))
172 })?;
173 let mut experts = Vec::new();
178 for e in 0..cfg.num_experts {
179 if model
180 .tensor(&format!("{prefix}mlp.experts.{e}.gate_proj.weight"))
181 .is_none()
182 {
183 break;
184 }
185 experts.push(load_dense(&format!("{prefix}mlp.experts.{e}."))?);
186 }
187 if experts.is_empty() {
188 return Err(CmfError::Parse(format!(
189 "{prefix}: router present but no expert tensors"
190 )));
191 }
192 let shared = if model
193 .tensor(&format!("{prefix}mlp.shared_expert.gate_proj.weight"))
194 .is_some()
195 {
196 let gate_name = format!("{prefix}mlp.shared_expert_gate.weight");
197 Some((
198 load_dense(&format!("{prefix}mlp.shared_expert."))?,
199 if model.tensor(&gate_name).is_some() {
200 Some(load_matrix(model, &gate_name, force_f32, ov)?)
201 } else {
202 None
203 },
204 ))
205 } else {
206 None
207 };
208 let bias_name = format!("{prefix}mlp.expert_bias");
211 let expert_bias = if model.tensor(&bias_name).is_some() {
212 Some(load_f32(model, &bias_name, ov).map_err(CmfError::Parse)?)
213 } else {
214 None
215 };
216 let top_k = std::env::var("CMF_MOE_TOPK")
222 .ok()
223 .and_then(|v| v.parse::<usize>().ok())
224 .filter(|&k| k >= 1 && k <= cfg.top_k)
225 .inspect(|k| tracing::info!("MoE top_k override: {} (header {})", k, cfg.top_k))
226 .unwrap_or(cfg.top_k);
227 let route_tau = std::env::var("CMF_MOE_TAU")
229 .ok()
230 .and_then(|v| v.parse::<f32>().ok())
231 .filter(|&t| t > 0.0 && t < 1.0)
232 .inspect(|t| tracing::info!("MoE adaptive routing: tau {t}"));
233 let mask = moe_task_mask(&prefix, experts.len());
234 let router = load_matrix(model, &router_name, force_f32, ov)?;
235 if router.rows() != experts.len() {
236 return Err(CmfError::Parse(format!(
237 "{router_name}: {} rows != {} experts",
238 router.rows(),
239 experts.len()
240 )));
241 }
242 let top_k = top_k.min(experts.len());
243 let pes_name = format!("{prefix}mlp.per_expert_scale");
247 let per_expert_scale = if model.tensor(&pes_name).is_some() {
248 Some(load_f32(model, &pes_name, ov).map_err(CmfError::Parse)?)
249 } else {
250 None
251 };
252 let router_input_norm = per_expert_scale.is_some();
253 let moe = MoeFfn {
254 router,
255 experts,
256 top_k,
257 route_tau,
258 norm_topk_prob: cfg.norm_topk_prob,
259 router_sigmoid: cfg.router_sigmoid,
260 expert_bias,
261 routed_scaling: cfg.routed_scaling_factor.unwrap_or(1.0),
262 shared,
263 stats: std::cell::RefCell::new(Vec::new()),
264 act_sq: std::cell::RefCell::new(Vec::new()),
265 act_rows: std::cell::RefCell::new(Vec::new()),
266 mask,
267 per_expert_scale,
268 router_input_norm,
269 };
270 if model
273 .tensor(&format!("{prefix}mlp.gate_proj.weight"))
274 .is_some()
275 {
276 let norm = |suffix: &str| -> Result<Vec<f32>, CmfError> {
277 load_f32(model, &format!("{prefix}{suffix}.weight"), ov).map_err(CmfError::Parse)
278 };
279 return Ok(FfnKind::DenseMoe(Box::new(crate::pipeline::DenseMoeFfn {
280 dense: load_dense(&format!("{prefix}mlp."))?,
281 moe,
282 post_norm_1: norm("post_feedforward_layernorm_1")?,
283 pre_norm_2: norm("pre_feedforward_layernorm_2")?,
284 post_norm_2: norm("post_feedforward_layernorm_2")?,
285 })));
286 }
287 Ok(FfnKind::Moe(moe))
288}
289
290pub(crate) fn moe_task_mask(prefix: &str, ne: usize) -> Option<Vec<bool>> {
298 use std::sync::OnceLock;
299 static CFG: OnceLock<Option<(std::collections::HashMap<usize, Vec<u64>>, f64)>> =
300 OnceLock::new();
301 let cfg = CFG.get_or_init(|| {
302 let path = std::env::var("CMF_MOE_MASK").ok()?;
303 let cover = std::env::var("CMF_MOE_MASK_COVER")
304 .ok()
305 .and_then(|v| v.parse::<f64>().ok())
306 .filter(|&c| c > 0.0 && c <= 1.0)
307 .unwrap_or(0.9);
308 let text = std::fs::read_to_string(&path)
309 .map_err(|e| tracing::warn!("CMF_MOE_MASK: cannot read {path}: {e}"))
310 .ok()?;
311 let map: std::collections::HashMap<String, Vec<u64>> = serde_json::from_str(&text)
312 .map_err(|e| tracing::warn!("CMF_MOE_MASK: bad JSON in {path}: {e}"))
313 .ok()?;
314 tracing::info!("MoE task mask: {path}, cover {cover}");
315 Some((
316 map.into_iter()
317 .filter_map(|(k, v)| Some((k.parse::<usize>().ok()?, v)))
318 .collect(),
319 cover,
320 ))
321 });
322 let (stats, cover) = cfg.as_ref()?;
323 let li: usize = prefix
325 .split("layers.")
326 .nth(1)?
327 .split('.')
328 .next()?
329 .parse()
330 .ok()?;
331 let counts = stats.get(&li)?;
332 if counts.len() != ne {
333 tracing::warn!(
334 "CMF_MOE_MASK: layer {li} has {} counts, model has {ne} experts — skipped",
335 counts.len()
336 );
337 return None;
338 }
339 let total: u64 = counts.iter().sum();
340 if total == 0 {
341 return None;
342 }
343 let mut order: Vec<usize> = (0..ne).collect();
344 order.sort_unstable_by_key(|&e| std::cmp::Reverse(counts[e]));
345 let mut mask = vec![false; ne];
346 let mut acc = 0u64;
347 let mut kept = 0usize;
348 for &e in &order {
349 mask[e] = true;
350 acc += counts[e];
351 kept += 1;
352 if (acc as f64) >= cover * (total as f64) {
353 break;
354 }
355 }
356 tracing::info!(
357 "MoE task mask L{li}: {kept}/{ne} experts for {:.0}% mass",
358 cover * 100.0
359 );
360 Some(mask)
361}
362
363fn load_matrix(
364 model: &Arc<CmfModel>,
365 name: &str,
366 force_f32: bool,
367 ov: &Overlay,
368) -> Result<QTensor, CmfError> {
369 if ov.blend_touches(model, name) {
373 if let Overlay::Blend(list) = ov {
374 let entry = model
375 .tensor(name)
376 .ok_or_else(|| CmfError::MissingTensor(name.to_string()))?;
377 let data =
378 blend_f32(model, name, list).map_err(|e| CmfError::Parse(format!("blend: {e}")))?;
379 return Ok(QTensor::from_f32(data, entry.shape[0], entry.shape[1]));
380 }
381 }
382 let skill = match ov {
383 Overlay::One(s) => Some(*s),
384 _ => None,
385 };
386 let name: &str = &match skill {
389 Some(sid) if model.tensor(&format!("skill.{sid}.{name}")).is_some() => {
390 format!("skill.{sid}.{name}")
391 }
392 _ => name.to_string(),
393 };
394 let err = |e: String| CmfError::Parse(format!("weight loading: {e}"));
395 if force_f32 {
396 let entry = model
397 .tensor(name)
398 .ok_or_else(|| CmfError::MissingTensor(name.to_string()))?;
399 if entry.shape.len() != 2 {
400 return Err(err(format!("'{name}' is not 2-D")));
401 }
402 let data = load_f32(model, name, &Overlay::None).map_err(err)?;
403 Ok(QTensor::from_f32(data, entry.shape[0], entry.shape[1]))
404 } else {
405 QTensor::from_model(model, name).map_err(err)
406 }
407}
408
409impl Pipeline {
410 pub fn from_model(
412 model: &Arc<CmfModel>,
413 sampler_config: SamplerConfig,
414 ) -> Result<Self, CmfError> {
415 Self::from_model_with_skill(model, sampler_config, None)
416 }
417
418 pub fn from_model_with_skill(
424 model: &Arc<CmfModel>,
425 sampler_config: SamplerConfig,
426 skill: Option<&str>,
427 ) -> Result<Self, CmfError> {
428 match skill {
429 Some(s) => Self::from_model_with_overlay(model, sampler_config, &Overlay::One(s)),
430 None => Self::from_model_with_overlay(model, sampler_config, &Overlay::None),
431 }
432 }
433
434 pub fn from_model_with_blend(
437 model: &Arc<CmfModel>,
438 sampler_config: SamplerConfig,
439 blend: &[(String, f32)],
440 ) -> Result<Self, CmfError> {
441 Self::from_model_with_overlay(model, sampler_config, &Overlay::Blend(blend))
442 }
443
444 fn skill_file_guard(model: &CmfModel) -> Result<(), CmfError> {
445 if model.required_features & cortiq_core::format::features::SKILL_FILE != 0 {
448 return Err(CmfError::Parse(
449 "this file is a standalone SKILL, not a runnable model — attach it: \
450 cortiq skill apply <base.cmf> <this file> -o specialist.cmf"
451 .into(),
452 ));
453 }
454 Ok(())
455 }
456
457 fn from_model_with_overlay(
458 model: &Arc<CmfModel>,
459 sampler_config: SamplerConfig,
460 ov: &Overlay,
461 ) -> Result<Self, CmfError> {
462 if let Some(dir) = model.path.parent() {
467 crate::gpu::set_cache_dir(dir.to_path_buf());
468 }
469 crate::gpu::graph_unsupported_reset();
472 Self::skill_file_guard(model)?;
473 let skill = match ov {
474 Overlay::One(s) => Some(*s),
475 _ => None,
476 };
477 if let Some(sid) = skill {
478 let known = model.header.skills.iter().any(|s| s.id == sid)
479 || model.skill_tensors(sid).next().is_some();
480 if !known {
481 return Err(CmfError::Parse(format!(
482 "skill '{sid}' not in this container (header.skills: {:?})",
483 model
484 .header
485 .skills
486 .iter()
487 .map(|s| &s.id)
488 .collect::<Vec<_>>()
489 )));
490 }
491 tracing::info!(
492 "skill '{sid}': {} replacement tensors overlaid",
493 model.skill_tensors(sid).count()
494 );
495 }
496 let arch = model.arch().clone();
497 let err = |e: String| CmfError::Parse(format!("weight loading: {e}"));
498 if let Some(heads) = &arch.attention_heads_per_layer {
499 if heads.len() != arch.num_layers {
500 return Err(CmfError::Parse(format!(
501 "arch.attention_heads_per_layer has {} entries, expected {}",
502 heads.len(),
503 arch.num_layers
504 )));
505 }
506 if let Some((li, &nh)) = heads
507 .iter()
508 .enumerate()
509 .find(|(_, nh)| **nh == 0 || **nh % arch.num_kv_heads != 0)
510 {
511 return Err(CmfError::Parse(format!(
512 "layer {li} has {nh} Q heads, which must be nonzero and divisible by {} KV heads",
513 arch.num_kv_heads
514 )));
515 }
516 }
517 if arch
518 .layer_types
519 .iter()
520 .any(|t| matches!(t, LayerType::SlidingAttention))
521 && arch.sliding_window.is_none()
522 {
523 return Err(CmfError::Parse(
524 "model has SlidingAttention layers but no arch.sliding_window".into(),
525 ));
526 }
527
528 let heads_masked = model.masks.masks.iter().any(|m| {
538 m.head_masks.iter().any(|row| {
539 let mut bits = 0usize;
540 for &b in row.iter() {
541 bits += b.count_ones() as usize;
542 }
543 !row.is_empty() && bits < arch.num_attention_heads
544 })
545 });
546 let force_f32 = heads_masked; let mut tokenizer = if let Some(vocab_bytes) = &model.vocab {
550 Tokenizer::from_bytes(vocab_bytes)
551 .map_err(|e| CmfError::Parse(format!("embedded tokenizer: {e}")))?
552 } else {
553 let sidecar = model.path.with_file_name("tokenizer.json");
554 if sidecar.exists() {
555 Tokenizer::from_file(&sidecar)
556 .map_err(|e| CmfError::Parse(format!("sidecar tokenizer: {e}")))?
557 } else {
558 tracing::warn!("no tokenizer in file or sidecar — using byte-level fallback");
559 Tokenizer::byte_level()
560 }
561 };
562 if let Some(tc) = &model.header.tokenizer_config {
564 tokenizer.chat_template = tc.chat_template.clone();
565 tokenizer.extra_eos.extend(tc.eos_token_ids.iter().copied());
566 if tokenizer.bos_token_id.is_none() {
567 tokenizer.bos_token_id = tc.bos_token_id;
568 }
569 tracing::info!(
570 "chat bundle: template {} chars, {} stop ids",
571 tc.chat_template.as_deref().map(str::len).unwrap_or(0),
572 tc.eos_token_ids.len()
573 );
574 }
575 if arch.arch_name.to_lowercase().contains("gemma") && tokenizer.bos_token_id.is_some() {
579 tokenizer.add_bos = true;
580 }
581
582 let embed_tokens = load_matrix(model, "model.embed_tokens.weight", false, ov)?;
584 let final_norm = load_f32(model, "model.norm.weight", ov).map_err(err)?;
585 let lm_head = if model.tensor("lm_head.weight").is_some() {
586 load_matrix(model, "lm_head.weight", false, ov)?
587 } else if arch.tie_word_embeddings {
588 load_matrix(model, "model.embed_tokens.weight", false, ov)?
590 } else {
591 return Err(CmfError::MissingTensor(
592 "lm_head.weight (and tie_word_embeddings is false)".into(),
593 ));
594 };
595
596 let has_linear = arch
598 .layer_types
599 .iter()
600 .any(|t| matches!(t, LayerType::LinearAttention));
601 let mut vmf_cfg = None;
602 let mut gdn_cfg = None;
603 if has_linear {
604 let lc = arch.linear_core.as_ref().ok_or_else(|| {
605 CmfError::Parse(
606 "model has LinearAttention layers but no arch.linear_core — \
607 reconvert with the current converter"
608 .into(),
609 )
610 })?;
611 let need = |v: Option<usize>, name: &str| {
612 v.ok_or_else(|| CmfError::Parse(format!("linear core needs arch.{name}")))
613 };
614 match lc.kind.as_str() {
615 "vmf_phase" => {
616 vmf_cfg = Some(VmfPhaseCfg {
617 num_heads: lc.num_heads,
618 nphase: need(lc.nphase, "linear_core.nphase")?,
619 value_head_dim: lc.value_head_dim,
620 hidden_size: arch.hidden_size,
621 phase_mass: std::env::var("CMF_PHASE_MASS")
624 .ok()
625 .and_then(|v| v.parse().ok())
626 .unwrap_or(0.0),
627 });
628 }
629 "gated_delta_net" => {
630 gdn_cfg = Some(GdnCfg {
631 num_v_heads: lc.num_heads,
632 num_k_heads: need(arch.linear_num_key_heads, "linear_num_key_heads")?,
633 key_head_dim: need(arch.linear_key_head_dim, "linear_key_head_dim")?,
634 value_head_dim: lc.value_head_dim,
635 conv_kernel: need(arch.linear_conv_kernel_dim, "linear_conv_kernel_dim")?,
636 hidden_size: arch.hidden_size,
637 rms_eps: arch.rms_norm_eps,
638 });
639 }
640 other => {
641 return Err(CmfError::Parse(format!(
642 "unknown linear core '{other}' (this runtime executes: \
643 gated_delta_net, vmf_phase)"
644 )));
645 }
646 }
647 }
648
649 let has_kda = arch.layer_types.iter().any(|t| matches!(t, LayerType::Kda));
651 let kda_cfg = if has_kda {
652 let need = |v: Option<usize>, name: &str| {
653 v.ok_or_else(|| CmfError::Parse(format!("KDA core needs arch.{name}")))
654 };
655 Some(crate::linear_core::KdaCfg {
656 num_heads: need(arch.linear_num_key_heads, "linear_num_key_heads")?,
657 head_k_dim: need(arch.linear_key_head_dim, "linear_key_head_dim")?,
658 head_v_dim: need(arch.linear_value_head_dim, "linear_value_head_dim")?,
659 conv_kernel: need(arch.linear_conv_kernel_dim, "linear_conv_kernel_dim")?,
660 hidden_size: arch.hidden_size,
661 rms_eps: arch.rms_norm_eps,
662 })
663 } else {
664 None
665 };
666
667 let has_short_conv = arch
669 .layer_types
670 .iter()
671 .any(|t| matches!(t, LayerType::ShortConv));
672 let short_conv_cfg = if has_short_conv {
673 Some(ShortConvCfg {
674 hidden_size: arch.hidden_size,
675 kernel: arch.linear_conv_kernel_dim.ok_or_else(|| {
676 CmfError::Parse(
677 "model has ShortConv layers but no arch.linear_conv_kernel_dim — \
678 reconvert with the current converter"
679 .into(),
680 )
681 })?,
682 })
683 } else {
684 None
685 };
686
687 let load_full_attn = |prefix: &str, layer: Option<usize>| -> Result<AttnKind, CmfError> {
689 let t = |suffix: &str| load_matrix(model, &format!("{prefix}{suffix}"), force_f32, ov);
690 let n = |suffix: &str| -> Option<Vec<f32>> {
691 model
692 .tensor(&format!("{prefix}{suffix}"))
693 .and_then(|_| load_f32(model, &format!("{prefix}{suffix}"), ov).ok())
694 };
695 if let Some(mla) = arch.mla.as_ref() {
697 let (q_proj, q_a, q_a_norm) = if mla.q_lora_rank.is_some() {
699 (
700 t("self_attn.q_b_proj.weight")?,
701 Some(t("self_attn.q_a_proj.weight")?),
702 Some(n("self_attn.q_a_layernorm.weight").ok_or_else(|| {
703 CmfError::Parse(format!("{prefix}: MLA needs q_a_layernorm"))
704 })?),
705 )
706 } else {
707 (t("self_attn.q_proj.weight")?, None, None)
708 };
709 let hd = mla.qk_rope_head_dim + mla.qk_nope_head_dim;
710 let nh = q_proj.rows() / hd;
711 let mut scale = 1.0 / (hd as f32).sqrt();
714 if let Some(y) = arch.yarn.as_ref() {
715 if let Some(m) = y.mscale_all_dim.filter(|&m| m > 0.0) {
716 let ms = 0.1 * m * y.factor.ln() + 1.0;
717 scale *= ms * ms;
718 }
719 }
720 return Ok(AttnKind::Mla(Box::new(crate::pipeline::MlaWeights {
721 q_proj,
722 q_a,
723 q_a_norm,
724 kv_a: t("self_attn.kv_a_proj_with_mqa.weight")?,
725 kv_a_norm: n("self_attn.kv_a_layernorm.weight").ok_or_else(|| {
726 CmfError::Parse(format!("{prefix}: MLA needs kv_a_layernorm"))
727 })?,
728 kv_b: t("self_attn.kv_b_proj.weight")?,
729 o_proj: t("self_attn.o_proj.weight")?,
730 nh,
731 qk_rope: mla.qk_rope_head_dim,
732 qk_nope: mla.qk_nope_head_dim,
733 v_dim: mla.v_head_dim,
734 lora: mla.kv_lora_rank,
735 scale,
736 nope: mla.nope,
737 })));
738 }
739 let wq = t("self_attn.q_proj.weight")?;
740 let nh = layer
741 .and_then(|li| {
742 arch.attention_heads_per_layer
743 .as_ref()
744 .and_then(|v| v.get(li).copied())
745 })
746 .unwrap_or(arch.num_attention_heads);
747 let output_gate = arch.global_head_dim.is_none() && wq.rows() == 2 * nh * arch.head_dim;
751 let is_global_layer = arch.global_head_dim.is_some()
754 && layer.is_some_and(|li| {
755 arch.sliding_window_pattern
756 .is_some_and(|p| p > 0 && (li + 1) % p == 0)
757 });
758 let expect = if is_global_layer {
759 nh * arch.global_head_dim.unwrap_or(arch.head_dim)
760 } else {
761 nh * arch.head_dim
762 };
763 if !output_gate && wq.rows() != expect {
764 return Err(CmfError::Parse(format!(
765 "{prefix}self_attn.q_proj.weight rows={} != heads({nh}) * head_dim({})",
766 wq.rows(),
767 expect / nh.max(1)
768 )));
769 }
770 let gate_name = format!("{prefix}self_attn.g_proj.weight");
771 let softplus_gate = if model.tensor(&gate_name).is_some() {
772 let gate = load_matrix(model, &gate_name, force_f32, ov)?;
773 if gate.cols() != arch.hidden_size {
774 return Err(CmfError::Parse(format!(
775 "{gate_name} cols={} != hidden_size ({})",
776 gate.cols(),
777 arch.hidden_size
778 )));
779 }
780 let per_head = if gate.rows() == nh {
781 true
782 } else if gate.rows() == nh * arch.head_dim {
783 false
784 } else {
785 return Err(CmfError::Parse(format!(
786 "{gate_name} rows={} must equal heads ({nh}) or heads*head_dim ({})",
787 gate.rows(),
788 nh * arch.head_dim
789 )));
790 };
791 Some((gate, per_head))
792 } else {
793 None
794 };
795 let bias = match (
797 n("self_attn.q_proj.bias"),
798 n("self_attn.k_proj.bias"),
799 n("self_attn.v_proj.bias"),
800 ) {
801 (Some(a), Some(b), Some(c)) => Some((a, b, c)),
802 _ => None,
803 };
804 Ok(AttnKind::Full {
805 wq,
806 wk: t("self_attn.k_proj.weight")?,
807 wv: t("self_attn.v_proj.weight")?,
808 wo: t("self_attn.o_proj.weight")?,
809 q_norm: n("self_attn.q_norm.weight"),
810 k_norm: n("self_attn.k_norm.weight"),
811 output_gate,
812 softplus_gate,
813 bias,
814 })
815 };
816
817 let load_linear_attn = |prefix: &str| -> Result<AttnKind, CmfError> {
818 if gdn_cfg.is_some() {
819 let t = |suffix: &str| {
821 load_matrix(
822 model,
823 &format!("{prefix}linear_attn.{suffix}"),
824 force_f32,
825 ov,
826 )
827 };
828 let f = |suffix: &str| {
829 load_f32(model, &format!("{prefix}linear_attn.{suffix}"), ov).map_err(err)
830 };
831 return Ok(AttnKind::LinearGdn(GdnWeights {
832 in_proj_qkv: t("in_proj_qkv.weight")?,
833 in_proj_z: t("in_proj_z.weight")?,
834 in_proj_a: t("in_proj_a.weight")?,
835 in_proj_b: t("in_proj_b.weight")?,
836 conv1d: f("conv1d.weight")?,
837 a_log: f("A_log")?,
838 dt_bias: f("dt_bias")?,
839 norm: f("norm.weight")?,
840 out_proj: t("out_proj.weight")?,
841 }));
842 }
843 let t = |suffix: &str| {
844 load_matrix(model, &format!("{prefix}vmf_attn.{suffix}"), force_f32, ov)
845 };
846 let a_log = load_f32(model, &format!("{prefix}vmf_attn.A_log"), ov).map_err(err)?;
847 let k_gate = if model
851 .tensor(&format!("{prefix}vmf_attn.k_gate.weight"))
852 .is_some()
853 {
854 Some((
855 t("k_gate.weight")?,
856 load_f32(model, &format!("{prefix}vmf_attn.k_gate.bias"), ov).map_err(err)?,
857 ))
858 } else {
859 None
860 };
861 Ok(AttnKind::Linear(VmfPhaseWeights {
862 thq: t("thq.weight")?,
863 thk: t("thk.weight")?,
864 v_proj: t("v_proj.weight")?,
865 out_proj: t("out_proj.weight")?,
866 decay: a_log.iter().map(|&a| (-(a as f64).exp()).exp()).collect(),
867 k_gate,
868 }))
869 };
870
871 let load_short_conv = |prefix: &str| -> Result<AttnKind, CmfError> {
875 let t = |suffix: &str| {
876 load_matrix(
877 model,
878 &format!("{prefix}short_conv.{suffix}"),
879 force_f32,
880 ov,
881 )
882 };
883 Ok(AttnKind::ShortConv(ShortConvWeights {
884 in_proj: t("in_proj.weight")?,
885 conv: load_f32(model, &format!("{prefix}short_conv.conv.weight"), ov)
886 .map_err(err)?,
887 out_proj: t("out_proj.weight")?,
888 }))
889 };
890
891 let load_kda = |prefix: &str| -> Result<AttnKind, CmfError> {
895 let t = |suffix: &str| {
896 load_matrix(model, &format!("{prefix}kda_attn.{suffix}"), force_f32, ov)
897 };
898 let f = |suffix: &str| {
899 load_f32(model, &format!("{prefix}kda_attn.{suffix}"), ov).map_err(err)
900 };
901 let gate = if model
902 .tensor(&format!("{prefix}kda_attn.g_proj.weight"))
903 .is_some()
904 {
905 crate::linear_core::KdaOutGate::Full(t("g_proj.weight")?)
906 } else {
907 crate::linear_core::KdaOutGate::LowRank(
908 t("g_a_proj.weight")?,
909 t("g_b_proj.weight")?,
910 )
911 };
912 Ok(AttnKind::Kda(Box::new(crate::linear_core::KdaWeights {
913 q_proj: t("q_proj.weight")?,
914 k_proj: t("k_proj.weight")?,
915 v_proj: t("v_proj.weight")?,
916 conv_q: f("q_conv1d.weight")?,
917 conv_k: f("k_conv1d.weight")?,
918 conv_v: f("v_conv1d.weight")?,
919 f_a: t("f_a_proj.weight")?,
920 f_b: t("f_b_proj.weight")?,
921 dt_bias: f("dt_bias")?,
922 a_log: f("A_log")?,
923 b_proj: t("b_proj.weight")?,
924 gate,
925 o_norm: f("o_norm.weight")?,
926 o_proj: t("o_proj.weight")?,
927 gate_lower_bound: arch.kda_gate_lower_bound.map(|v| v as f32),
928 })))
929 };
930
931 fn anyhow_like(ok: bool) -> Result<(), ()> {
932 if ok { Ok(()) } else { Err(()) }
933 }
934 let mut layers = Vec::with_capacity(arch.num_layers);
935 let is_g3n = arch.g3n.is_some();
936 let owns_its_layers = is_g3n || arch.arch_name == "deepseek_v4";
941 for li in 0..(if owns_its_layers { 0 } else { arch.num_layers }) {
942 let prefix = format!("model.layers.{li}.");
943 let attn = match arch.layer_types.get(li) {
944 Some(LayerType::LinearAttention) => load_linear_attn(&prefix)?,
945 Some(LayerType::Kda) => load_kda(&prefix)?,
946 Some(LayerType::ShortConv) => load_short_conv(&prefix)?,
947 _ => load_full_attn(&prefix, Some(li))?,
948 };
949 let pre_ffn = format!("{prefix}pre_feedforward_layernorm.weight");
953 let sandwich = model.tensor(&pre_ffn).is_some();
954 layers.push(LayerWeights {
955 input_norm: load_f32(model, &format!("{prefix}input_layernorm.weight"), ov)
956 .map_err(err)?,
957 post_norm: if sandwich {
958 load_f32(model, &pre_ffn, ov).map_err(err)?
959 } else {
960 load_f32(
961 model,
962 &format!("{prefix}post_attention_layernorm.weight"),
963 ov,
964 )
965 .map_err(err)?
966 },
967 attn_out_norm: if sandwich {
968 Some(
969 load_f32(
970 model,
971 &format!("{prefix}post_attention_layernorm.weight"),
972 ov,
973 )
974 .map_err(err)?,
975 )
976 } else {
977 None
978 },
979 ffn_out_norm: if sandwich {
980 Some(
981 load_f32(
982 model,
983 &format!("{prefix}post_feedforward_layernorm.weight"),
984 ov,
985 )
986 .map_err(err)?,
987 )
988 } else {
989 None
990 },
991 layer_scale: model
993 .tensor(&format!("{prefix}layer_scalar"))
994 .and_then(|_| {
995 load_f32(model, &format!("{prefix}layer_scalar"), ov)
996 .ok()
997 .and_then(|v| v.first().copied())
998 }),
999 ffn: build_layer_ffn(model, &arch, li, false, ov)?,
1001 attn,
1002 });
1003 }
1004
1005 let mtp_present = model
1014 .tensor("model.mtp.layers.0.self_attn.q_proj.weight")
1015 .is_some()
1016 || model.tensor("model.mtp.eh_proj.weight").is_some();
1017 let dsv4_mtp = model.tensor("model.mtp.0.main_proj.weight").is_some();
1022 if arch.mtp.is_some() && !mtp_present && !dsv4_mtp {
1023 tracing::info!(
1024 "header declares an MTP head but the file carries none — \
1025 loading without it"
1026 );
1027 }
1028 let mtp = if let Some(cfg) = arch.mtp.as_ref().filter(|_| mtp_present) {
1029 if cfg.num_layers != 1 {
1030 return Err(CmfError::Parse(format!(
1031 "MTP with {} blocks not supported yet (only 1)",
1032 cfg.num_layers
1033 )));
1034 }
1035 let p = "model.mtp.";
1036 let attn = load_full_attn("model.mtp.layers.0.", None)?;
1037 Some(MtpModule {
1038 enorm: load_f32(model, &format!("{p}enorm.weight"), ov).map_err(err)?,
1039 hnorm: load_f32(model, &format!("{p}hnorm.weight"), ov).map_err(err)?,
1040 eh_proj: load_matrix(model, &format!("{p}eh_proj.weight"), false, ov)?,
1041 layer: LayerWeights {
1042 attn_out_norm: None,
1043 ffn_out_norm: None,
1044 layer_scale: None,
1045 input_norm: load_f32(model, &format!("{p}layers.0.input_layernorm.weight"), ov)
1046 .map_err(err)?,
1047 post_norm: load_f32(
1048 model,
1049 &format!("{p}layers.0.post_attention_layernorm.weight"),
1050 ov,
1051 )
1052 .map_err(err)?,
1053 ffn: build_ffn_at(model, &arch, &format!("{p}layers.0."), false, ov)?,
1057 attn,
1058 },
1059 final_norm: load_f32(model, &format!("{p}norm.weight"), ov).map_err(err)?,
1060 kv: LayerKvCache::new(arch.num_kv_heads, arch.head_dim),
1061 })
1062 } else {
1063 None
1064 };
1065
1066 tracing::info!(
1067 "Pipeline loaded: {} | {}L ({} linear) | {:.2}B params | storage: {} | MTP: {}",
1068 arch.arch_name,
1069 arch.num_layers,
1070 arch.layer_types
1071 .iter()
1072 .filter(|t| matches!(t, LayerType::LinearAttention))
1073 .count(),
1074 model.total_param_count() as f64 / 1e9,
1075 if force_f32 {
1076 "f32 (masked)"
1077 } else {
1078 "quantized mmap"
1079 },
1080 if mtp.is_some() { "yes" } else { "no" }
1081 );
1082
1083 let cap = std::env::var("CMF_MAX_SEQ")
1086 .ok()
1087 .and_then(|v| v.parse::<usize>().ok())
1088 .unwrap_or(8192);
1089 let max_seq_len = arch.max_position_embeddings.min(cap);
1090
1091 let total_layers = arch.num_layers * arch.num_loops;
1093
1094 let mut pipeline = Pipeline::new(
1095 tokenizer,
1096 PipelineWeights {
1097 embed_tokens,
1098 layers,
1099 lm_head,
1100 final_norm,
1101 },
1102 arch.hidden_size,
1103 arch.intermediate_size,
1104 arch.num_attention_heads,
1105 arch.num_kv_heads,
1106 arch.head_dim,
1107 total_layers,
1108 arch.num_layers, arch.loop_final_norm,
1110 arch.vocab_size,
1111 arch.rms_norm_eps,
1112 arch.rope_theta as f32,
1113 arch.norm_style,
1114 max_seq_len,
1115 sampler_config,
1116 );
1117 let rotary = ((arch.head_dim as f32 * arch.partial_rotary_factor) as usize).max(2);
1118 pipeline.set_rotary(rotary, arch.rope_theta as f32);
1119 pipeline.attention_heads_per_layer = arch.attention_heads_per_layer.clone();
1120 if let Some(yarn) = &arch.yarn {
1121 pipeline.inv_freq = std::sync::Arc::new(crate::attention::yarn_inv_freq(
1122 rotary,
1123 arch.rope_theta as f32,
1124 yarn.factor,
1125 yarn.original_max_position_embeddings,
1126 yarn.beta_fast,
1127 yarn.beta_slow,
1128 ));
1129 pipeline.rope_scale = yarn.attention_factor;
1130 }
1131 pipeline.embed_multiplier = arch.embed_multiplier;
1135 pipeline.logit_multiplier = arch.logit_multiplier;
1136 if let Some(qpas) = arch.query_pre_attn_scalar {
1137 pipeline.attn_scale = 1.0 / (qpas as f32).sqrt();
1138 }
1139 if let (Some(w), Some(p)) = (arch.sliding_window, arch.sliding_window_pattern) {
1140 pipeline.swa = Some((w, p));
1141 if let Some(base) = arch.rope_local_base_freq {
1142 pipeline.inv_freq_local = Some(std::sync::Arc::new(
1143 crate::attention::rope_inv_freq(rotary, base as f32),
1144 ));
1145 }
1146 }
1147 let explicit_sliding: Vec<bool> = arch
1148 .layer_types
1149 .iter()
1150 .map(|t| matches!(t, cortiq_core::LayerType::SlidingAttention))
1151 .collect();
1152 if explicit_sliding.iter().any(|&v| v) {
1153 pipeline.sliding_layers = Some(explicit_sliding);
1154 if let Some(w) = arch.sliding_window {
1155 pipeline.swa = Some((w, usize::MAX));
1156 }
1157 let local_rotary = ((arch.head_dim as f32
1158 * arch
1159 .local_partial_rotary_factor
1160 .unwrap_or(arch.partial_rotary_factor))
1161 as usize)
1162 .max(2);
1163 pipeline.rotary_dim_local = Some(local_rotary);
1164 if let Some(base) = arch.rope_local_base_freq {
1165 pipeline.inv_freq_local = Some(std::sync::Arc::new(
1166 crate::attention::rope_inv_freq(local_rotary, base as f32),
1167 ));
1168 }
1169 }
1170 if let (Some(ghd), Some(gkv)) = (arch.global_head_dim, arch.num_global_kv_heads) {
1174 pipeline.global_attn = Some((ghd, gkv));
1175 let prf = arch.global_partial_rotary_factor.unwrap_or(1.0);
1176 let half = ghd / 2;
1177 let ra = (((prf * ghd as f32) as usize) / 2).min(half);
1178 let mut f = vec![0.0f32; half];
1179 for (i, slot) in f.iter_mut().enumerate().take(ra) {
1180 *slot = 1.0 / (arch.rope_theta as f32).powf(2.0 * i as f32 / ghd as f32);
1181 }
1182 pipeline.inv_freq_global = Some(std::sync::Arc::new(f));
1183 let global_at = |li: usize| -> bool {
1188 match &pipeline.sliding_layers {
1189 Some(map) => !map.get(li).copied().unwrap_or(false),
1190 None => pipeline
1191 .swa
1192 .map(|(_, p)| p > 0 && p != usize::MAX && (li + 1) % p == 0)
1193 .unwrap_or(false),
1194 }
1195 };
1196 for li in 0..arch.num_layers {
1197 if global_at(li) {
1198 pipeline.kv_cache.layers[li] = crate::kv_cache::LayerKvCache::new(gkv, ghd);
1199 }
1200 }
1201 }
1202 if let Some(mla) = arch.mla.as_ref() {
1205 let hd = mla.qk_rope_head_dim + mla.qk_nope_head_dim;
1206 pipeline.head_dim = hd;
1207 pipeline.num_kv_heads = arch.num_attention_heads;
1208 pipeline.rotary_dim = mla.qk_rope_head_dim;
1209 let half = mla.qk_rope_head_dim / 2;
1210 let mut f = vec![0.0f32; half];
1211 for (i, slot) in f.iter_mut().enumerate() {
1212 *slot = 1.0
1213 / (arch.rope_theta as f32).powf(2.0 * i as f32 / mla.qk_rope_head_dim as f32);
1214 }
1215 pipeline.inv_freq = std::sync::Arc::new(f);
1216 for li in 0..arch.num_layers {
1217 pipeline.kv_cache.layers[li] =
1218 crate::kv_cache::LayerKvCache::new(arch.num_attention_heads, hd);
1219 }
1220 }
1221 if let Some(fac) = &arch.rope_freq_factors {
1225 let mut f = pipeline.inv_freq.as_ref().clone();
1226 for (i, v) in f.iter_mut().enumerate() {
1227 if let Some(&d) = fac.get(i) {
1228 *v /= d as f32;
1229 }
1230 }
1231 pipeline.inv_freq = std::sync::Arc::new(f);
1232 }
1233 pipeline.attn_v_norm = arch.attn_v_norm;
1234 pipeline.final_softcap = arch.final_logit_softcapping.map(|c| c as f32);
1235 pipeline.attn_softcap = arch.attn_logit_softcapping.unwrap_or(0.0) as f32;
1236 pipeline.vmf_cfg = vmf_cfg;
1237 pipeline.gdn_cfg = gdn_cfg;
1238 pipeline.kda_cfg = kda_cfg;
1239 if let Some(gc) = arch.g3n.as_ref() {
1240 use crate::g3n::{G3nAltUp, G3nGlobals, G3nLaurel, G3nLayer};
1241 anyhow_like(gc.altup_num_inputs == crate::g3n::ALTUP_N).map_err(|_| {
1242 CmfError::Parse(format!(
1243 "g3n: altup_num_inputs {} != supported {}",
1244 gc.altup_num_inputs,
1245 crate::g3n::ALTUP_N
1246 ))
1247 })?;
1248 let t = |name: &str| load_matrix(model, name, force_f32, ov);
1249 let f = |name: &str| load_f32(model, name, ov).map_err(err);
1250 let mut altup_proj = Vec::new();
1251 let mut altup_unembed = Vec::new();
1252 for i in 0..crate::g3n::ALTUP_N - 1 {
1253 altup_proj.push(t(&format!("model.altup_projections.{i}.weight"))?);
1254 altup_unembed.push(t(&format!("model.altup_unembed_projections.{i}.weight"))?);
1255 }
1256 let first_shared = arch.num_layers.saturating_sub(gc.num_kv_shared_layers);
1257 let sliding_of = |li: usize| {
1258 matches!(
1259 arch.layer_types.get(li),
1260 Some(cortiq_core::LayerType::SlidingAttention)
1261 )
1262 };
1263 let mut g3n_layers = Vec::with_capacity(arch.num_layers);
1264 for li in 0..arch.num_layers {
1265 let pfx = format!("model.layers.{li}.");
1266 let shared = li >= first_shared && first_shared > 0;
1267 let share_src = if shared {
1268 let want = sliding_of(li);
1269 (0..first_shared).rev().find(|&j| sliding_of(j) == want)
1270 } else {
1271 None
1272 };
1273 g3n_layers.push(G3nLayer {
1274 altup: G3nAltUp {
1275 router_norm: f(&format!("{pfx}altup.router_norm.weight"))?,
1276 modality_router: t(&format!("{pfx}altup.modality_router.weight"))?,
1277 prediction_coefs: t(&format!("{pfx}altup.prediction_coefs.weight"))?,
1278 correction_coefs: t(&format!("{pfx}altup.correction_coefs.weight"))?,
1279 correct_output_scale: f(&format!("{pfx}altup.correct_output_scale"))?,
1280 },
1281 laurel: G3nLaurel {
1282 left: t(&format!("{pfx}laurel.linear_left.weight"))?,
1283 right: t(&format!("{pfx}laurel.linear_right.weight"))?,
1284 post_norm: f(&format!("{pfx}laurel.post_laurel_norm.weight"))?,
1285 },
1286 input_norm: f(&format!("{pfx}input_layernorm.weight"))?,
1287 post_attn_norm: f(&format!("{pfx}post_attention_layernorm.weight"))?,
1288 pre_ffw_norm: f(&format!("{pfx}pre_feedforward_layernorm.weight"))?,
1289 post_ffw_norm: f(&format!("{pfx}post_feedforward_layernorm.weight"))?,
1290 wq: t(&format!("{pfx}self_attn.q_proj.weight"))?,
1291 wk: if shared {
1292 None
1293 } else {
1294 Some(t(&format!("{pfx}self_attn.k_proj.weight"))?)
1295 },
1296 wv: if shared {
1297 None
1298 } else {
1299 Some(t(&format!("{pfx}self_attn.v_proj.weight"))?)
1300 },
1301 wo: t(&format!("{pfx}self_attn.o_proj.weight"))?,
1302 q_norm: f(&format!("{pfx}self_attn.q_norm.weight"))?,
1303 k_norm: if shared {
1304 None
1305 } else {
1306 Some(f(&format!("{pfx}self_attn.k_norm.weight"))?)
1307 },
1308 kv_share_src: share_src,
1309 sliding: sliding_of(li),
1310 gate: t(&format!("{pfx}mlp.gate_proj.weight"))?,
1311 up: t(&format!("{pfx}mlp.up_proj.weight"))?,
1312 down: t(&format!("{pfx}mlp.down_proj.weight"))?,
1313 sparsity: gc.activation_sparsity.get(li).copied().unwrap_or(0.0),
1314 ple_gate: t(&format!("{pfx}per_layer_input_gate.weight"))?,
1315 ple_proj: t(&format!("{pfx}per_layer_projection.weight"))?,
1316 post_ple_norm: f(&format!("{pfx}post_per_layer_input_norm.weight"))?,
1317 });
1318 }
1319 let hd = arch.head_dim;
1320 let globals = G3nGlobals {
1321 altup_proj,
1322 altup_unembed,
1323 ple_embed: t("model.embed_tokens_per_layer.weight")?,
1324 ple_model_proj: t("model.per_layer_model_projection.weight")?,
1325 ple_norm: f("model.per_layer_projection_norm.weight")?,
1326 ple_vocab: gc.ple_vocab,
1327 ple_dim: gc.ple_dim,
1328 num_layers: arch.num_layers,
1329 hidden: arch.hidden_size,
1330 rms_eps: arch.rms_norm_eps,
1331 inv_freq_local: crate::attention::rope_inv_freq(
1332 hd,
1333 arch.rope_local_base_freq.unwrap_or(10_000.0) as f32,
1334 ),
1335 inv_freq_global: crate::attention::rope_inv_freq(hd, arch.rope_theta as f32),
1336 window: arch.sliding_window.unwrap_or(512),
1337 };
1338 pipeline.g3n = Some(Box::new((globals, g3n_layers)));
1339 }
1340 if arch.arch_name == "deepseek_v4" {
1345 let moe = arch
1346 .moe
1347 .as_ref()
1348 .ok_or_else(|| CmfError::Parse("deepseek_v4: no moe config".into()))?;
1349 let cfg = crate::dsv4::Dsv4Cfg {
1350 dim: arch.hidden_size,
1351 n_heads: arch.num_attention_heads,
1352 head_dim: arch.head_dim,
1353 rope_head_dim: if arch.partial_rotary_factor < 1.0 {
1360 (((arch.head_dim as f32 * arch.partial_rotary_factor) as usize) & !1)
1361 .clamp(2, arch.head_dim)
1362 } else {
1363 64.min(arch.head_dim)
1364 },
1365 q_lora_rank: 0,
1369 o_lora_rank: 0,
1370 o_groups: 8,
1377 hc_mult: 4,
1378 hc_sinkhorn_iters: 20,
1379 hc_eps: 1e-6,
1380 norm_eps: arch.rms_norm_eps as f32,
1381 n_routed_experts: moe.num_experts,
1382 top_k: moe.top_k,
1383 moe_inter: moe.moe_intermediate_size,
1384 route_scale: moe.routed_scaling_factor.unwrap_or(1.0) as f32,
1385 swiglu_limit: 10.0,
1391 window: arch.sliding_window.unwrap_or(128),
1392 index_topk: 512,
1393 vocab: arch.vocab_size,
1394 };
1395 let (g, dl) = crate::dsv4::load(model, &cfg, arch.num_layers)
1396 .map_err(|e| CmfError::Parse(format!("deepseek_v4: {e}")))?;
1397 let mut cfg = cfg;
1402 if let Some(l0) = dl.first() {
1403 cfg.q_lora_rank = l0.wq_a.rows();
1404 let attn_width = arch.num_attention_heads * arch.head_dim;
1405 if l0.wo_a.cols() > 0 && attn_width % l0.wo_a.cols() == 0 {
1406 cfg.o_groups = (attn_width / l0.wo_a.cols()).max(1);
1407 }
1408 cfg.o_lora_rank = l0.wo_b.cols() / cfg.o_groups.max(1);
1409 cfg.hc_mult = (l0.hc_attn_fn.len() / l0.hc_attn_base.len().max(1)) / cfg.dim.max(1);
1410 if cfg.hc_mult == 0 {
1411 cfg.hc_mult = 4;
1412 }
1413 }
1414 let (yf, yo, ybf, ybs) = match &arch.yarn {
1427 Some(y) => (
1428 y.factor,
1429 y.original_max_position_embeddings,
1430 y.beta_fast,
1431 y.beta_slow,
1432 ),
1433 None => {
1434 tracing::warn!(
1435 "deepseek_v4: the header carries no YaRN profile — \
1436 falling back to the release's (factor 16, original \
1437 65536, beta 32/1). Re-converting with a build that \
1438 reads rope_scaling.type would make this exact."
1439 );
1440 (16.0, 65536, 32.0, 1.0)
1441 }
1442 };
1443 pipeline.inv_freq = std::sync::Arc::new(crate::attention::yarn_inv_freq(
1444 cfg.rope_head_dim,
1445 arch.rope_theta as f32,
1446 yf,
1447 yo,
1448 ybf,
1449 ybs,
1450 ));
1451 if let Ok(stats) = std::env::var("CMF_MOE_PIN") {
1456 let cover = std::env::var("CMF_MOE_PIN_COVER")
1457 .ok()
1458 .and_then(|v| v.parse::<f64>().ok())
1459 .filter(|&c| c > 0.0 && c <= 1.0)
1460 .unwrap_or(0.95);
1461 let hot = crate::pin::hot_experts(&stats, cover);
1462 let mut names: Vec<String> = Vec::new();
1463 for e in &model.tensors {
1464 let is_expert = e.name.contains(".mlp.experts.");
1465 if !is_expert {
1466 names.push(e.name.clone()); }
1468 }
1469 let mut kept_experts = 0usize;
1470 if let Some(hot) = &hot {
1471 for (li, experts) in hot {
1472 for e in experts {
1473 for w in ["gate_proj", "up_proj", "down_proj"] {
1474 names.push(format!("model.layers.{li}.mlp.experts.{e}.{w}.weight"));
1475 }
1476 kept_experts += 1;
1477 }
1478 }
1479 }
1480 let r = crate::pin::pin_tensors(model, &names);
1481 tracing::info!(
1482 "закреплено {:.1} ГБ ({} тензоров, горячих экспертов {kept_experts}, покрытие {cover}); лимит {}",
1483 r.bytes as f64 / 1e9,
1484 r.tensors,
1485 r.limit
1486 .map(|l| format!("{:.1} ГБ", l as f64 / 1e9))
1487 .unwrap_or_else(|| "неизвестен".into())
1488 );
1489 if r.skipped > 0 {
1490 tracing::warn!("не закреплено тензоров: {}", r.skipped);
1491 }
1492 }
1493 let st = crate::dsv4::Dsv4State::new(arch.num_layers);
1494 let depth = std::env::var("CMF_DSV4_MTP_DEPTH")
1498 .ok()
1499 .and_then(|v| v.parse::<usize>().ok())
1500 .unwrap_or(3);
1501 pipeline.dsv4_mtp = crate::dsv4::load_mtp(model, &cfg, depth);
1502 crate::dsv4::dspark_reserve_note(&pipeline.dsv4_mtp, &cfg, &dl);
1504 pipeline.dsv4 = Some(Box::new((g, dl, cfg, st)));
1505 }
1506 pipeline.short_conv_cfg = short_conv_cfg;
1507 pipeline.mtp = mtp;
1508 pipeline.install_dynamic_routing(model, false);
1509 match ov {
1513 Overlay::One(sid) => {
1514 pipeline.dyn_active = model.header.skills.iter().position(|s| &s.id == sid);
1515 }
1516 Overlay::Blend(_) => pipeline.dyn_blend_loaded = true,
1517 Overlay::None => {}
1518 }
1519 if let Some(c) = &model.header.calibration {
1522 pipeline.set_calib_temp(c.temperature);
1523 }
1524 let o1 = match crate::nystrom::o1_from_env() {
1530 crate::nystrom::O1Env::Off => None,
1531 crate::nystrom::O1Env::On(cfg) => Some(cfg),
1532 crate::nystrom::O1Env::Unset => model
1533 .header
1534 .provenance
1535 .as_ref()
1536 .and_then(|p| p.get("o1_attn"))
1537 .and_then(crate::nystrom::O1Cfg::from_json),
1538 };
1539 if o1.is_some() {
1540 if pipeline.attn_softcap > 0.0 {
1541 return Err(CmfError::Parse(
1542 "--o1 with attention-logit soft-capping (Gemma-2) is not supported: \
1543 the streaming operator has no capped-score form"
1544 .into(),
1545 ));
1546 }
1547 pipeline.set_o1(o1);
1548 }
1549 Ok(pipeline)
1550 }
1551
1552 pub(crate) fn install_dynamic_routing(&mut self, model: &Arc<CmfModel>, force_f32: bool) {
1557 self.model = Some(model.clone());
1558 self.dyn_force_f32 = force_f32;
1559 let mut per_skill = Vec::with_capacity(model.header.skills.len());
1560 for sk in &model.header.skills {
1561 let mut ffn_layers = std::collections::BTreeSet::new();
1562 let mut non_ffn = false;
1563 let prefix = format!("skill.{}.", sk.id);
1564 for t in model.skill_tensors(&sk.id) {
1565 let rel = &t.name[prefix.len()..]; let toks: Vec<&str> = rel.split('.').collect();
1567 if toks.len() >= 5 && toks[0] == "model" && toks[1] == "layers" && toks[3] == "mlp"
1568 {
1569 if let Ok(li) = toks[2].parse::<usize>() {
1570 ffn_layers.insert(li);
1571 continue;
1572 }
1573 }
1574 non_ffn = true; }
1576 if non_ffn {
1577 tracing::warn!(
1578 "skill '{}' replaces non-FFN tensors — excluded from dynamic \
1579 routing (static overlay still works)",
1580 sk.id
1581 );
1582 per_skill.push(None);
1583 } else {
1584 per_skill.push(Some(ffn_layers.into_iter().collect::<Vec<_>>()));
1585 }
1586 }
1587 self.dyn_skill_layers = per_skill;
1588 }
1589
1590 pub fn set_active_skill(&mut self, idx: Option<usize>) -> Result<(), CmfError> {
1597 self.kv_cache.clear();
1599 self.kv_history.clear();
1600 if self.dyn_active == idx {
1601 return Ok(());
1602 }
1603 let model = self.model.clone().ok_or_else(|| {
1604 CmfError::Parse("dynamic routing needs a model-backed pipeline".into())
1605 })?;
1606 let mut union: std::collections::BTreeSet<usize> = std::collections::BTreeSet::new();
1607 if let Some(old) = self.dyn_active {
1608 if let Some(Some(ls)) = self.dyn_skill_layers.get(old) {
1609 union.extend(ls.iter().copied());
1610 }
1611 }
1612 let new_id: Option<String> = match idx {
1613 Some(n) => match self.dyn_skill_layers.get(n) {
1614 Some(Some(ls)) => {
1615 union.extend(ls.iter().copied());
1616 Some(model.header.skills[n].id.clone())
1617 }
1618 _ => {
1619 return Err(CmfError::Parse(format!(
1620 "skill index {n} not dynamic-eligible"
1621 )));
1622 }
1623 },
1624 None => None,
1625 };
1626 let ov = match &new_id {
1627 Some(s) => Overlay::One(s),
1628 None => Overlay::None,
1629 };
1630 let arch = model.arch();
1631 for li in union {
1632 self.weights.layers[li].ffn =
1633 build_layer_ffn(&model, arch, li, self.dyn_force_f32, &ov)?;
1634 }
1635 self.dyn_active = idx;
1636 Ok(())
1637 }
1638}