Skip to main content

cortiq_engine/
lib.rs

1//! Cortiq inference engine — sparse forward pass, attention, tokenization, sampling.
2
3pub mod attention;
4pub mod chat_template;
5pub mod cpuprof;
6pub mod audiovae;
7pub mod bounded;
8pub mod dit;
9pub mod dsv4;
10pub mod dsv41;
11pub mod dsv41_encoding;
12pub mod dsv41_vision;
13pub mod fcd;
14pub mod fcd_ops;
15pub mod f32_backend;
16pub mod g3n;
17pub mod gptq_capture;
18pub mod gpu;
19#[cfg(target_os = "macos")]
20pub mod gpu_metal;
21#[cfg(feature = "gpu")]
22pub mod gpu_wgpu;
23pub mod imagegen;
24pub mod inference;
25pub mod kv_cache;
26pub mod linear_core;
27pub mod loader;
28pub mod lookup;
29pub mod mimo_moe;
30pub mod ltxaudio;
31pub mod ltxdit;
32pub mod ltxdur;
33pub mod ltxenc;
34pub mod ltxlora;
35pub mod ltxpipe;
36pub mod ltxte;
37pub mod ltxups;
38pub mod ltxvae;
39pub mod media;
40pub mod mimo_audio;
41pub mod mimo_mm;
42pub mod mimo_ingress;
43pub mod mimo_vision;
44pub mod mm_ab;
45pub mod mmh3;
46pub mod mmh3ups;
47pub mod music3;
48pub mod nystrom;
49pub mod pin;
50pub mod pipeline;
51pub mod pool;
52pub mod prism;
53pub mod qtensor;
54pub mod qwen3te;
55pub mod qwen3vis;
56pub mod qwen4_exp;
57pub mod qwen_image;
58pub mod qwen_image_encoder;
59pub mod qwen_image_ops;
60pub mod qwen_image_vae;
61pub mod qwen_image_vision;
62pub mod qwen_imagegen;
63pub mod router;
64pub mod runtime;
65pub mod sampler;
66pub mod skillbake;
67pub mod swarm;
68pub mod textenc;
69pub mod tokenizer;
70pub mod vae;
71pub mod vae3d;
72pub mod videogen;
73pub mod zimage;
74pub mod zimagegen;
75/// The native Vulkan lane — an accelerator behind a capability probe,
76/// present only where Vulkan is.
77#[cfg(all(
78    feature = "gpu",
79    any(target_os = "linux", target_os = "windows", target_os = "android")
80))]
81pub use nystrom::NystromState;
82pub use pipeline::{GenerateResult, Pipeline, TokenCallback, TokenTrace};
83pub use runtime::CortiqRuntime;
84
85/// Test-only: N empty Metal command-buffer round trips, total seconds.
86#[doc(hidden)]
87#[cfg(target_os = "macos")]
88pub fn gpu_empty_submit_for_test(n: usize) -> f64 {
89    gpu_metal::empty_submit_bench(n)
90}
91
92/// Test-only: N pipelined empty submits, one final wait.
93#[doc(hidden)]
94#[cfg(target_os = "macos")]
95pub fn gpu_pipelined_submit_for_test(n: usize) -> f64 {
96    gpu_metal::pipelined_submit_bench(n)
97}
98
99/// Test-only: build a q1 MoeJob trio (weight 1.0).
100#[doc(hidden)]
101#[cfg(target_os = "macos")]
102pub fn gpu_moe_job_for_test(
103    gi: usize,
104    ui: usize,
105    di: usize,
106    inter: usize,
107    hidden: usize,
108    x: Vec<f32>,
109) -> gpu::MoeJob<'static> {
110    gpu::MoeJob {
111        gate: (gi, inter, hidden, &[]),
112        up: (ui, inter, hidden, &[]),
113        down: (di, hidden, inter, &[]),
114        xs_gate: x.clone(),
115        xs_up: x,
116        down_col: &[],
117        w: 1.0,
118        q1: true,
119        q4t: false,
120        q4tp: false,
121        gu_q2: false,
122        swiglu_limit: 0.0,
123    }
124}
125
126/// Test-only: run the metal moe_block on one job.
127#[doc(hidden)]
128#[cfg(target_os = "macos")]
129pub fn gpu_moe_block_for_test(
130    model: &std::sync::Arc<cortiq_core::CmfModel>,
131    job: gpu::MoeJob<'_>,
132    out: &mut [f32],
133) -> bool {
134    gpu_metal::moe_block(model, &[job], out)
135}
136
137/// Test-only: q1 matvec_batch — jobs (idx, rows, cols); first two share
138/// x, the third takes xi.
139#[doc(hidden)]
140#[cfg(target_os = "macos")]
141pub fn gpu_batch_q1_for_test(
142    model: &std::sync::Arc<cortiq_core::CmfModel>,
143    shapes: &[(usize, usize, usize)],
144    x: &[f32],
145    xi: &[f32],
146    outs: &mut [&mut [f32]],
147) -> bool {
148    let jobs: Vec<gpu::BatchJob> = shapes
149        .iter()
150        .enumerate()
151        .map(|(k, &(idx, rows, cols))| gpu::BatchJob {
152            idx,
153            rows,
154            cols,
155            row_scale: &[],
156            xs: if k < 2 { x.to_vec() } else { xi.to_vec() },
157            layout: gpu::BatchLayout::Q1,
158        })
159        .collect();
160    gpu_metal::matvec_batch(model, &jobs, outs)
161}
162
163/// Test-only direct handle to the Metal q1 matvec (micro-benchmarks).
164#[doc(hidden)]
165#[cfg(target_os = "macos")]
166pub fn gpu_q1_matvec_for_test(
167    model: &std::sync::Arc<cortiq_core::CmfModel>,
168    idx: usize,
169    xs: &[f32],
170    rows: usize,
171    cols: usize,
172    out: &mut [f32],
173) -> bool {
174    gpu_metal::q1_matvec(model, idx, xs, rows, cols, out)
175}
176pub use sampler::SamplerConfig;