Skip to main content

cortiq_engine/
lib.rs

1//! Cortiq inference engine — sparse forward pass, attention, tokenization, sampling.
2
3pub mod attention;
4pub mod chat_template;
5pub mod cpuprof;
6pub mod audiovae;
7pub mod bounded;
8pub mod dit;
9pub mod dsv4;
10pub mod dsv41;
11pub mod dsv41_encoding;
12pub mod dsv41_vision;
13pub(crate) mod expert_store;
14pub mod fcd;
15pub mod fcd_ops;
16pub mod f32_backend;
17pub mod g3n;
18pub mod gptq_capture;
19pub mod gpu;
20#[cfg(target_os = "macos")]
21pub mod gpu_metal;
22#[cfg(feature = "gpu")]
23pub mod gpu_wgpu;
24pub mod imagegen;
25pub mod inference;
26pub mod kv_cache;
27pub mod linear_core;
28pub mod loader;
29pub mod lookup;
30pub mod mimo_moe;
31pub mod ltxaudio;
32pub mod ltxdit;
33pub mod ltxdur;
34pub mod ltxenc;
35pub mod ltxlora;
36pub mod ltxpipe;
37pub mod ltxte;
38pub mod ltxups;
39pub mod ltxvae;
40pub mod media;
41pub mod mimo_audio;
42pub mod mimo_mm;
43pub mod mimo_ingress;
44pub mod mimo_vision;
45pub mod mm_ab;
46pub mod mmh3;
47pub mod mmh3ups;
48pub mod music3;
49pub mod nystrom;
50pub mod pin;
51pub mod pipeline;
52pub mod pool;
53pub mod prism;
54pub mod qtensor;
55pub mod qwen3te;
56pub mod qwen3vis;
57pub mod qwen4_exp;
58pub mod qwen_image;
59pub mod qwen_image21;
60pub mod qwen_image21_vae;
61pub mod qwen_image21gen;
62pub mod qwen_image_encoder;
63pub mod qwen_image_ops;
64pub mod qwen_image_vae;
65pub mod qwen_image_vision;
66pub mod qwen_imagegen;
67pub mod router;
68pub mod runtime;
69pub mod sampler;
70pub mod skillbake;
71pub mod swarm;
72pub mod textenc;
73pub mod tokenizer;
74pub mod vae;
75pub mod vae3d;
76pub mod videogen;
77/// Native Whisper encoder/decoder inference over CMF checkpoints.
78pub mod whisper;
79pub mod zimage;
80pub mod zimagegen;
81/// The native Vulkan lane — an accelerator behind a capability probe,
82/// present only where Vulkan is.
83#[cfg(all(
84    feature = "gpu",
85    any(target_os = "linux", target_os = "windows", target_os = "android")
86))]
87pub use nystrom::NystromState;
88pub use pipeline::{GenerateResult, Pipeline, TokenCallback, TokenTrace};
89pub use runtime::CortiqRuntime;
90
91/// Test-only: N empty Metal command-buffer round trips, total seconds.
92#[doc(hidden)]
93#[cfg(target_os = "macos")]
94pub fn gpu_empty_submit_for_test(n: usize) -> f64 {
95    gpu_metal::empty_submit_bench(n)
96}
97
98/// Test-only: N pipelined empty submits, one final wait.
99#[doc(hidden)]
100#[cfg(target_os = "macos")]
101pub fn gpu_pipelined_submit_for_test(n: usize) -> f64 {
102    gpu_metal::pipelined_submit_bench(n)
103}
104
105/// Test-only: build a q1 MoeJob trio (weight 1.0).
106#[doc(hidden)]
107#[cfg(target_os = "macos")]
108pub fn gpu_moe_job_for_test(
109    gi: usize,
110    ui: usize,
111    di: usize,
112    inter: usize,
113    hidden: usize,
114    x: Vec<f32>,
115) -> gpu::MoeJob<'static> {
116    gpu::MoeJob {
117        gate: (gi, inter, hidden, &[]),
118        up: (ui, inter, hidden, &[]),
119        down: (di, hidden, inter, &[]),
120        xs_gate: x.clone(),
121        xs_up: x,
122        down_col: &[],
123        w: 1.0,
124        q1: true,
125        q4t: false,
126        q4tp: false,
127        gu_q2: false,
128        swiglu_limit: 0.0,
129    }
130}
131
132/// Test-only: run the metal moe_block on one job.
133#[doc(hidden)]
134#[cfg(target_os = "macos")]
135pub fn gpu_moe_block_for_test(
136    model: &std::sync::Arc<cortiq_core::CmfModel>,
137    job: gpu::MoeJob<'_>,
138    out: &mut [f32],
139) -> bool {
140    gpu_metal::moe_block(model, &[job], out)
141}
142
143/// Test-only: q1 matvec_batch — jobs (idx, rows, cols); first two share
144/// x, the third takes xi.
145#[doc(hidden)]
146#[cfg(target_os = "macos")]
147pub fn gpu_batch_q1_for_test(
148    model: &std::sync::Arc<cortiq_core::CmfModel>,
149    shapes: &[(usize, usize, usize)],
150    x: &[f32],
151    xi: &[f32],
152    outs: &mut [&mut [f32]],
153) -> bool {
154    let jobs: Vec<gpu::BatchJob> = shapes
155        .iter()
156        .enumerate()
157        .map(|(k, &(idx, rows, cols))| gpu::BatchJob {
158            idx,
159            rows,
160            cols,
161            row_scale: &[],
162            xs: if k < 2 { x.to_vec() } else { xi.to_vec() },
163            layout: gpu::BatchLayout::Q1,
164        })
165        .collect();
166    gpu_metal::matvec_batch(model, &jobs, outs)
167}
168
169/// Test-only direct handle to the Metal q1 matvec (micro-benchmarks).
170#[doc(hidden)]
171#[cfg(target_os = "macos")]
172pub fn gpu_q1_matvec_for_test(
173    model: &std::sync::Arc<cortiq_core::CmfModel>,
174    idx: usize,
175    xs: &[f32],
176    rows: usize,
177    cols: usize,
178    out: &mut [f32],
179) -> bool {
180    gpu_metal::q1_matvec(model, idx, xs, rows, cols, out)
181}
182pub use sampler::SamplerConfig;