cera 0.3.0

Rust-native LLM inference engine
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
//! End-to-end smoke test for the Phase-1 VL loader against a real
//! `LiquidAI/LeapBundles` VL bundle.
//!
//! Asserts:
//! 1. Both GGUFs (main LLM + mmproj) download via `BundleRepo`.
//! 2. `CeraEngine::from_files` accepts the VL pair (the gate is
//!    open) and constructs an engine without error.
//! 3. The mmproj is mmaped and exposed via
//!    `CeraEngine::vision_encoder_gguf()`.
//! 4. The metadata's `max_seq_len` is non-zero so we know the
//!    underlying LFM2 LLM parsed correctly. Greedy decode is
//!    deferred to PR 2+ (vision-encoder forward pass) and to the
//!    larger-coverage `bundle_download.rs`.
//!
//! Gating: `#[ignore]` + `CERA_TEST_DOWNLOAD=1`. Same shared cache
//! path (`target/tmp/cera-test-models`) as the other gated tests, so
//! CI runs amortise the download.
//!
//! ```sh
//! CERA_TEST_DOWNLOAD=1 cargo test -p cera --features remote \
//!     --test vl_bundle_load -- --ignored
//! ```

#![cfg(feature = "remote")]

mod common;

use cera::engine::{BackendPreference, CeraEngine, EngineConfig, ModelFiles};
use cera::manifest::InferenceType;
use cera::tokenizer::{ChatMessage, ChatMessageMultimodal, ContentItem};
use cera::{FinishReason, GenerateOpts, ModalitySink};

const MAIN_URL: &str =
    "https://huggingface.co/LiquidAI/LFM2.5-VL-450M-GGUF/resolve/main/LFM2.5-VL-450M-Q4_0.gguf";
const MAIN_FILE: &str = "LFM2.5-VL-450M-Q4_0.gguf";
const MMPROJ_URL: &str = "https://huggingface.co/LiquidAI/LFM2.5-VL-450M-GGUF/resolve/main/mmproj-LFM2.5-VL-450m-Q8_0.gguf";
const MMPROJ_FILE: &str = "mmproj-LFM2.5-VL-450m-Q8_0.gguf";

#[test]
#[ignore = "downloads ~310 MB across two GGUFs; set CERA_TEST_DOWNLOAD=1 and pass --ignored"]
fn vl_bundle_loads_text_only() {
    if std::env::var("CERA_TEST_DOWNLOAD").is_err() {
        eprintln!("skipping: CERA_TEST_DOWNLOAD not set");
        return;
    }

    let main = common::download::ensure_cached(MAIN_URL, MAIN_FILE);
    let mmproj = common::download::ensure_cached(MMPROJ_URL, MMPROJ_FILE);
    assert!(main.exists(), "main GGUF missing at {}", main.display());
    assert!(
        mmproj.exists(),
        "mmproj GGUF missing at {}",
        mmproj.display()
    );

    // Construct the engine via `from_files` with the VL pair. The
    // explicit `inference_type` is necessary because auto-detect would
    // see the main GGUF's `architecture = "lfm2"` and pick text
    // (correct for the LLM half but skips the eager mmproj load).
    // Real callers reach this path through a manifest;
    // `from_bundle_id` populates these fields automatically.
    let mut files = ModelFiles::text(&main);
    files.multimodal_projector = Some(mmproj.clone());
    files.inference_type = Some(InferenceType::LlamaCppImageToText);

    let engine = CeraEngine::from_files(
        files,
        EngineConfig {
            context_size: 256,
            backend: BackendPreference::Cpu,
            ..Default::default()
        },
    )
    .expect("VL bundle should load with the Phase-1 gate open");

    // Sanity: the LFM2 LLM half parsed cleanly.
    let meta = engine.metadata();
    assert!(
        meta.max_seq_len > 0,
        "engine metadata missing max_seq_len — main GGUF parse failed silently"
    );
    assert_eq!(
        meta.architecture, "lfm2",
        "main GGUF arch should be plain `lfm2`; got `{}`",
        meta.architecture
    );

    // The mmproj must have been mmapped and exposed for Phase 2+.
    let mmproj_gguf = engine
        .vision_encoder_gguf()
        .expect("VL bundles must expose the mmproj GGUF through vision_encoder_gguf()");
    // `clip` arch with `clip.has_vision_encoder = true` is the
    // shape every published VL mmproj uses. Asserting both lets a
    // future schema change surface here rather than silently
    // misbehave in PR 2.
    let arch = mmproj_gguf
        .architecture()
        .expect("mmproj should expose general.architecture");
    assert_eq!(arch, "clip", "mmproj arch should be `clip`; got `{arch}`");
    let has_vision = mmproj_gguf
        .get_bool("clip.has_vision_encoder")
        .unwrap_or(false);
    assert!(
        has_vision,
        "mmproj should set `clip.has_vision_encoder = true`"
    );

    // Phase-2 typed loader. The mmproj must have been parsed into
    // `VisionEncoderWeights`; spec from `project_vl_architecture.md`
    // (LFM2.5-VL-450M ViT-12-768).
    let ve = engine
        .vision_encoder()
        .expect("VL bundles must expose typed VisionEncoderWeights via vision_encoder()");
    assert_eq!(ve.config.n_layer, 12, "ViT block count");
    assert_eq!(ve.config.n_embd, 768, "ViT hidden dim");
    assert_eq!(ve.config.n_head, 12, "ViT head count");
    assert_eq!(ve.config.n_ff, 3072, "ViT FFN dim");
    assert_eq!(ve.config.image_size, 256);
    assert_eq!(ve.config.patch_size, 16);
    assert_eq!(
        ve.config.n_trained_patches, 256,
        "trained 16×16 position-grid"
    );
    assert_eq!(ve.config.image_min_pixels, 65_536, "LFM2-VL min-pixel band");
    assert_eq!(
        ve.config.image_max_pixels, 262_144,
        "LFM2-VL max-pixel band"
    );
    assert_eq!(ve.config.projection_dim, 1024, "matches LFM2 embed dim");
    assert_eq!(ve.config.scale_factor, 2, "pixel-shuffle factor");
    assert_eq!(ve.blocks.len(), 12);

    // Sanity-check the preprocessing constants. The loader reads
    // `clip.vision.image_{mean,std}` from f32-array metadata; if
    // the key got renamed or the array length drifted, the
    // loader bails — but a wrong-but-loadable replacement (e.g.
    // a key returning all-zeros) would slip through. CLIP-family
    // mean/std fall in (0, 1); std must be non-zero to avoid
    // divide-by-zero in the future preprocessor.
    for (i, m) in ve.config.image_mean.iter().enumerate() {
        assert!(*m > 0.0 && *m < 1.0, "image_mean[{i}] = {m} outside (0, 1)");
    }
    for (i, s) in ve.config.image_std.iter().enumerate() {
        assert!(*s > 0.0 && *s < 1.0, "image_std[{i}] = {s} outside (0, 1)");
    }
    assert!(
        ve.config.eps > 0.0 && ve.config.eps < 1e-3,
        "layer_norm_epsilon = {} outside (0, 1e-3)",
        ve.config.eps
    );

    // Per-block shapes — every block must have the same
    // ViT-12-768 layout. Looping catches a hypothetical
    // off-by-one or partial-load bug that would leave a later
    // block in a degenerate state. Loader's `anyhow::ensure!`
    // already runs at parse time but asserting here encodes the
    // contract.
    for (i, blk) in ve.blocks.iter().enumerate() {
        assert_eq!(blk.q_w.rows, 768, "block {i} q_w rows");
        assert_eq!(blk.q_w.cols, 768, "block {i} q_w cols");
        assert_eq!(blk.k_w.rows, 768, "block {i} k_w rows");
        assert_eq!(blk.v_w.rows, 768, "block {i} v_w rows");
        assert_eq!(blk.o_w.rows, 768, "block {i} o_w rows");
        assert_eq!(blk.ffn_up_w.rows, 3072, "block {i} ffn_up_w rows");
        assert_eq!(blk.ffn_up_w.cols, 768, "block {i} ffn_up_w cols");
        assert_eq!(blk.ffn_down_w.rows, 768, "block {i} ffn_down_w rows");
        assert_eq!(blk.ffn_down_w.cols, 3072, "block {i} ffn_down_w cols");
        assert_eq!(blk.ln1_w.len(), 768, "block {i} ln1_w len");
        assert_eq!(blk.ln2_w.len(), 768, "block {i} ln2_w len");
    }

    // Position embeddings cover every patch.
    assert_eq!(ve.position_embed.len(), 256 * 768);
    // Projector dims: mm.1 is [3072 → 2048], mm.2 is [2048 → 1024].
    assert_eq!(ve.projector.mm1_w.rows, 2048);
    assert_eq!(ve.projector.mm1_w.cols, 3072);
    assert_eq!(ve.projector.mm2_w.rows, 1024);
    assert_eq!(ve.projector.mm2_w.cols, 2048);

    // End-to-end: render the chat template (exercises the
    // `{% generation %}` strip in tokenizer.rs), tokenize, prefill,
    // and greedy-decode a few tokens. Catches regressions in either
    // the template fix or the LFM2 forward pass on a VL bundle —
    // the load-only path above doesn't exercise those.
    let tokenizer = engine.tokenizer();
    let messages = vec![ChatMessage {
        role: "user".to_string(),
        content: "Hi".to_string(),
    }];
    let formatted = cera::tokenizer::apply_chat_template(tokenizer, &messages, true)
        .expect("LFM2.5-VL chat template should render after generation-block strip");
    let prompt_tokens = tokenizer.encode(&formatted);
    assert!(
        !prompt_tokens.is_empty(),
        "rendered chat template tokenized to nothing — encoder is broken"
    );

    let mut session = engine.new_session(Default::default()).unwrap();
    session
        .append_tokens(&prompt_tokens)
        .expect("prefill should succeed against a VL bundle's LFM2 LLM");

    struct Collect(Vec<u32>);
    impl ModalitySink for Collect {
        fn on_text_tokens(&mut self, t: &[u32]) {
            self.0.extend_from_slice(t);
        }
        fn on_done(&mut self, _: FinishReason) {}
    }
    let mut sink = Collect(Vec::new());
    let opts = GenerateOpts {
        max_tokens: 8,
        temperature: 0.0,
        ..Default::default()
    };
    session
        .generate(&opts, &mut sink)
        .expect("greedy decode should succeed against a VL bundle");
    assert!(
        !sink.0.is_empty(),
        "VL bundle produced zero tokens — chat template / forward / sink wiring broken"
    );

    // ── Phase-2 slice 2+3: ViT + projector forward smoke. ──
    //
    // Feed a synthetic constant image (`[3, 256, 256]` of 0.0 —
    // exercises every stage without needing real preprocessor
    // output) through the full vision encoder and assert the
    // image-token output is well-formed:
    //   * length = n_image_tokens × projection_dim = 64 × 1024
    //   * all values finite (catches NaN/Inf from broken
    //     softmax / norm / projector arithmetic)
    //   * not all zero (catches a forward path that short-
    //     circuits without doing any real work)
    //   * magnitudes in a sane range (catches numerical
    //     blow-up that would still be finite)
    //
    // **TODO(parity):** these checks are deliberately weak —
    // they would pass even on a forward pass that's wrong-but-
    // plausible (e.g. a kernel-stride bug producing scrambled
    // values). Strong correctness gate (parity vs llama.cpp's
    // `clip.cpp` on a real image, with strict numerical
    // tolerance) lands in Phase 3 alongside the image
    // preprocessor + real fixture. Two known assumptions also
    // need verification then: (1) GELU variant — cera uses
    // `cpu::gelu_erf_inplace` (erf-form), llama.cpp's
    // `ggml_gelu` is the tanh approximation; ~1e-5
    // per-element drift if mismatched. (2) Pixel-shuffle 2×2
    // ordering — cera concatenates source patches in
    // `(sr·sf + sc)` row-major order; clip.cpp's traversal
    // direction needs a reference vector to confirm.
    let ve = engine
        .vision_encoder()
        .expect("vision_encoder still attached");
    // Drive `encode_image` at the trained 16×16 grid (no
    // interpolation path); zeros input verifies forward stays
    // numerically sane.
    let trained_side = (ve.config.n_trained_patches as f64).sqrt().round() as usize;
    let n_pix = 3 * (trained_side * ve.config.patch_size).pow(2);
    let zeros = vec![0.0f32; n_pix];
    let img_tokens = ve
        .encode_image(&zeros, trained_side, trained_side)
        .expect("encode_image should succeed on a zero-input image");
    let expected_n_tokens = (trained_side / ve.config.scale_factor).pow(2);
    let expected_n = expected_n_tokens * ve.config.projection_dim;
    assert_eq!(img_tokens.len(), expected_n, "image-token output length");
    assert!(
        img_tokens.iter().all(|v| v.is_finite()),
        "encode_image produced non-finite values"
    );
    let max_abs = img_tokens.iter().fold(0.0f32, |a, &v| a.max(v.abs()));
    assert!(
        max_abs > 0.0,
        "encode_image returned all zeros — forward likely short-circuited"
    );
    assert!(
        max_abs < 1e3,
        "encode_image returned implausibly large values (max abs = {max_abs}) — \
         numerical blow-up somewhere in the pipeline"
    );
}

/// **Phase 3 slice 1 end-to-end smoke.** Synthesises an in-memory
/// PNG, drives `Session::append_image` end-to-end (preprocess →
/// ViT → projector → splice into LLM prefill via
/// `append_embeddings`), generates a few tokens, and asserts the
/// LLM produced non-degenerate text. Catches "completely broken"
/// — pipeline runs, image embeddings get into the LLM stream, the
/// LLM doesn't crash on them. Strong correctness gate (parity vs
/// `clip.cpp` on a real image with strict numerical tolerance) is
/// deferred; the LLM-output check rules out gross structural
/// issues like NaN propagation or pixel-shuffle row/col swaps
/// causing the LLM to produce gibberish.
#[test]
#[ignore = "downloads ~310 MB across two GGUFs; set CERA_TEST_DOWNLOAD=1 and pass --ignored"]
fn vl_bundle_appends_synthetic_image() {
    if std::env::var("CERA_TEST_DOWNLOAD").is_err() {
        eprintln!("skipping: CERA_TEST_DOWNLOAD not set");
        return;
    }

    let main = common::download::ensure_cached(MAIN_URL, MAIN_FILE);
    let mmproj = common::download::ensure_cached(MMPROJ_URL, MMPROJ_FILE);
    let mut files = ModelFiles::text(&main);
    files.multimodal_projector = Some(mmproj);
    files.inference_type = Some(InferenceType::LlamaCppImageToText);
    let engine = CeraEngine::from_files(
        files,
        EngineConfig {
            // Dynamic-resolution preprocessor produces up to 256
            // image tokens at the LFM2-VL band's upper edge.
            // Budget: prefix tokens + 256 image + suffix + 32
            // generation. 512 covers the worst case while keeping
            // the test fast on CPU.
            context_size: 512,
            backend: BackendPreference::Cpu,
            ..Default::default()
        },
    )
    .expect("VL bundle load");

    // Image input: prefer a real fixture at
    // `cera/tests/fixtures/pug.jpg` (committed for end-to-end
    // demos against an actual recognisable subject). Fall back
    // to a synthesised solid-red 256² PNG when the fixture is
    // missing — keeps the smoke runnable in environments that
    // haven't checked out the fixture (also avoids hard-failing
    // when this test runs through `--test-threads=1` on a
    // shallow clone). The smoke only asserts the LLM produces
    // non-degenerate text, so either input is fine for that
    // gate; a real image just gives nicer manual output.
    let fixture = std::path::Path::new(env!("CARGO_MANIFEST_DIR"))
        .join("tests")
        .join("fixtures")
        .join("pug.jpg");
    let img_bytes: Vec<u8> = if fixture.exists() {
        std::fs::read(&fixture).expect("read pug.jpg fixture")
    } else {
        eprintln!(
            "note: {} not found — falling back to a synthetic red PNG. \
             Save a small image fixture at that path for richer manual output.",
            fixture.display()
        );
        use image::{ImageBuffer, Rgb};
        let img = ImageBuffer::<Rgb<u8>, _>::from_fn(256, 256, |_, _| Rgb([255u8, 0, 0]));
        let mut png = Vec::new();
        image::DynamicImage::ImageRgb8(img)
            .write_to(&mut std::io::Cursor::new(&mut png), image::ImageFormat::Png)
            .expect("encode synthetic png");
        png
    };

    // Capabilities should now report image_in for VL bundles.
    let caps = engine.capabilities();
    assert!(caps.image_in, "VL capabilities must report image_in=true");

    // Drive the LFM2-VL chat template via the helper that walks
    // `<image>` markers automatically (slice 3 — replaces the
    // manual `<|im_start|>user\n` find-and-splice pattern earlier
    // versions of this test used). The expected envelope shape
    // remains:
    //   <bos><|im_start|>user\n<|image_start|>[N image embeds]<|image_end|>TEXT<|im_end|>\n<|im_start|>assistant\n
    // The helper's marker-position match is enforced here by the
    // post-image generation passing — a wrong-position envelope
    // produces non-image-conditioned generic descriptions even on
    // an image-tuned bundle.
    let tokenizer = engine.tokenizer();
    let messages = vec![ChatMessageMultimodal {
        role: "user".to_string(),
        content: vec![
            ContentItem::Image,
            ContentItem::Text {
                text: "Describe what you see.".to_string(),
            },
        ],
    }];

    let mut session = engine.new_session(Default::default()).unwrap();
    session
        .append_chat_with_images(&messages, &[&img_bytes], true)
        .expect("append_chat_with_images should succeed end-to-end");

    struct Collect(Vec<u32>);
    impl ModalitySink for Collect {
        fn on_text_tokens(&mut self, t: &[u32]) {
            self.0.extend_from_slice(t);
        }
        fn on_done(&mut self, _: FinishReason) {}
    }
    let mut sink = Collect(Vec::new());
    let opts = GenerateOpts {
        max_tokens: 32,
        temperature: 0.0,
        ..Default::default()
    };
    session
        .generate(&opts, &mut sink)
        .expect("generate after image+prompt");
    assert!(
        !sink.0.is_empty(),
        "generated zero tokens — append_image / forward / decode wiring is broken"
    );
    let decoded = tokenizer.decode(&sink.0);
    eprintln!("vl smoke decoded output: {decoded:?}");
    // Post-image output must look like English: at least 60% ASCII
    // letters/spaces (catches "lots of unicode + punctuation"
    // garbage from a broken encoder), and at least one space
    // (catches a single long unbroken token sequence). These bars
    // would have failed the pre-fix output ("complex and abstract
    // scene" passed because it *was* English — the fix that
    // mattered was switching from generic to scene-specific). The
    // live test's main correctness gate remains the eprintln
    // dump + manual review on first run; the assertion only
    // catches gross structural regressions.
    let total = decoded.chars().count() as f32;
    let alpha_or_space = decoded
        .chars()
        .filter(|c| c.is_ascii_alphabetic() || *c == ' ')
        .count() as f32;
    assert!(total > 0.0, "decoded text is empty");
    assert!(
        alpha_or_space / total >= 0.6,
        "post-image output mostly non-letters: {decoded:?}"
    );
    assert!(
        decoded.contains(' '),
        "post-image output has no spaces — single unbroken token: {decoded:?}"
    );
    let alpha_count = decoded.chars().filter(char::is_ascii_alphabetic).count();
    assert!(
        alpha_count >= 4,
        "post-image generation looks degenerate (got {decoded:?}); image embeddings \
         likely poisoned the LLM stream"
    );

    // Validate `max_long_size` end-to-end through the RECOMMENDED path
    // (`append_chat_with_images`, which honors the session-default cap)
    // on real weights. Compare against an uncapped baseline: the text
    // tokens are identical across both, so a working cap must yield
    // strictly FEWER image tokens, i.e. a lower KV position after
    // prefill. This is the load-bearing check — it fails if the cap is
    // a no-op, if the chat path ignores the session default, or if the
    // old cascaded downscale→upscale re-inflated the grid back.
    let uncapped_pos = {
        let mut s = engine.new_session(Default::default()).unwrap();
        s.append_chat_with_images(&messages, &[&img_bytes], true)
            .expect("uncapped chat append");
        s.position()
    };
    let mut capped = engine.new_session(Default::default()).unwrap();
    capped.set_image_max_long_size(Some(128));
    capped
        .append_chat_with_images(&messages, &[&img_bytes], true)
        .expect("capped chat append should run end-to-end via the session default");
    let capped_pos = capped.position();
    assert!(
        capped_pos < uncapped_pos,
        "max_long_size must reduce image tokens via the chat path: \
         capped KV {capped_pos} should be < uncapped {uncapped_pos}"
    );

    // The capped, down-sampled embeddings must still be usable: a 128px
    // cap takes the grid below the model's min_pixels floor, so this
    // exercises the interpolated position-embedding path at a small
    // grid. Generation must still produce tokens.
    let mut capped_sink = Collect(Vec::new());
    capped
        .generate(&opts, &mut capped_sink)
        .expect("generate after capped image prefill");
    assert!(
        !capped_sink.0.is_empty(),
        "capped (sub-min_pixels) image prefill produced no generation — \
         small-grid pos-embed interpolation may be broken"
    );
}

/// GPU-path end-to-end smoke (M4 wiring): build the same VL bundle with
/// `BackendPreference::Auto` so the engine builds a GPU vision encoder
/// (Metal on Apple Silicon, else wgpu) and the session runs the ViT on the
/// GPU. Asserts the encoder was actually built (on gpu/metal builds) and that
/// image-conditioned generation still produces non-degenerate English — i.e.
/// the GPU encoder's embeddings splice into the LLM stream correctly, matching
/// the CPU path's behavior. Skipped without `CERA_TEST_DOWNLOAD`.
#[test]
#[ignore = "downloads ~310 MB across two GGUFs; set CERA_TEST_DOWNLOAD=1 and pass --ignored"]
fn vl_bundle_gpu_path_generates() {
    if std::env::var("CERA_TEST_DOWNLOAD").is_err() {
        eprintln!("skipping: CERA_TEST_DOWNLOAD not set");
        return;
    }

    let main = common::download::ensure_cached(MAIN_URL, MAIN_FILE);
    let mmproj = common::download::ensure_cached(MMPROJ_URL, MMPROJ_FILE);
    let mut files = ModelFiles::text(&main);
    files.multimodal_projector = Some(mmproj);
    files.inference_type = Some(InferenceType::LlamaCppImageToText);
    let engine = CeraEngine::from_files(
        files,
        EngineConfig {
            context_size: 512,
            // Auto → Metal/wgpu when compiled in; CPU otherwise.
            backend: BackendPreference::Auto,
            ..Default::default()
        },
    )
    .expect("VL bundle load (Auto backend)");

    // On a GPU/Metal build with a working device the engine should have built a
    // GPU vision encoder — but runtime device/context creation can still fail
    // (headless CI, missing drivers/permissions), in which case it correctly
    // falls back to the CPU encoder. Skip rather than fail when that happens, so
    // this test asserts the GPU path only when a GPU was actually selected.
    #[cfg(any(feature = "gpu", all(feature = "metal", target_os = "macos")))]
    if !engine.has_gpu_vision_encoder() {
        eprintln!("skipping GPU-path assertions: no usable GPU device; encoder fell back to CPU");
        return;
    }

    let fixture = std::path::Path::new(env!("CARGO_MANIFEST_DIR"))
        .join("tests")
        .join("fixtures")
        .join("pug.jpg");
    let img_bytes: Vec<u8> = if fixture.exists() {
        std::fs::read(&fixture).expect("read pug.jpg fixture")
    } else {
        use image::{ImageBuffer, Rgb};
        let img = ImageBuffer::<Rgb<u8>, _>::from_fn(256, 256, |_, _| Rgb([255u8, 0, 0]));
        let mut png = Vec::new();
        image::DynamicImage::ImageRgb8(img)
            .write_to(&mut std::io::Cursor::new(&mut png), image::ImageFormat::Png)
            .expect("encode synthetic png");
        png
    };

    let tokenizer = engine.tokenizer();
    let messages = vec![ChatMessageMultimodal {
        role: "user".to_string(),
        content: vec![
            ContentItem::Image,
            ContentItem::Text {
                text: "Describe what you see.".to_string(),
            },
        ],
    }];

    let mut session = engine.new_session(Default::default()).unwrap();
    session
        .append_chat_with_images(&messages, &[&img_bytes], true)
        .expect("append_chat_with_images (GPU path) should succeed end-to-end");

    struct Collect(Vec<u32>);
    impl ModalitySink for Collect {
        fn on_text_tokens(&mut self, t: &[u32]) {
            self.0.extend_from_slice(t);
        }
        fn on_done(&mut self, _: FinishReason) {}
    }
    let mut sink = Collect(Vec::new());
    let opts = GenerateOpts {
        max_tokens: 32,
        temperature: 0.0,
        ..Default::default()
    };
    session
        .generate(&opts, &mut sink)
        .expect("generate after GPU image prefill");
    let decoded = tokenizer.decode(&sink.0);
    eprintln!("vl GPU-path decoded output: {decoded:?}");

    let total = decoded.chars().count() as f32;
    let alpha_or_space = decoded
        .chars()
        .filter(|c| c.is_ascii_alphabetic() || *c == ' ')
        .count() as f32;
    assert!(total > 0.0, "decoded text is empty");
    assert!(
        alpha_or_space / total >= 0.6,
        "GPU-path output mostly non-letters: {decoded:?}"
    );
    assert!(
        decoded.contains(' '),
        "GPU-path output has no spaces — single unbroken token: {decoded:?}"
    );
}

/// Real-weights embedding parity: the GPU vision encoder must match the CPU
/// encoder on the actual quantized mmproj (the synthetic F32 parity test in
/// `vision_encoder_gpu` can't exercise quantized weights or the real
/// 12-layer / 768-dim / GELU-saturating activations). Asserts no NaN/Inf and a
/// small mean absolute difference vs the CPU reference. This is the regression
/// guard for bugs like GPU `tanh` overflow in GELU.
#[test]
#[ignore = "downloads ~310 MB across two GGUFs; set CERA_TEST_DOWNLOAD=1 and pass --ignored"]
fn vl_bundle_gpu_embeddings_match_cpu() {
    if std::env::var("CERA_TEST_DOWNLOAD").is_err() {
        eprintln!("skipping: CERA_TEST_DOWNLOAD not set");
        return;
    }
    let main = common::download::ensure_cached(MAIN_URL, MAIN_FILE);
    let mmproj = common::download::ensure_cached(MMPROJ_URL, MMPROJ_FILE);
    let mut files = ModelFiles::text(&main);
    files.multimodal_projector = Some(mmproj);
    files.inference_type = Some(InferenceType::LlamaCppImageToText);
    let engine = CeraEngine::from_files(
        files,
        EngineConfig {
            context_size: 512,
            backend: BackendPreference::Cpu,
            ..Default::default()
        },
    )
    .expect("load");
    let cpu_enc = engine.vision_encoder().expect("vision encoder").clone();

    // A non-trivial gradient image so activations span a realistic range
    // (constant images don't exercise the large-activation GELU tail).
    use image::{ImageBuffer, Rgb};
    let img = ImageBuffer::<Rgb<u8>, _>::from_fn(256, 256, |x, y| {
        Rgb([(x % 256) as u8, (y % 256) as u8, ((x + y) % 256) as u8])
    });
    let mut png = Vec::new();
    image::DynamicImage::ImageRgb8(img)
        .write_to(&mut std::io::Cursor::new(&mut png), image::ImageFormat::Png)
        .unwrap();

    let pre =
        cera::model::vision_preprocessor::preprocess_image_with_opts(&png, &cpu_enc.config, None)
            .expect("preprocess");

    let cpu_emb = cpu_enc
        .encode_image(&pre.pixels, pre.grid_w, pre.grid_h)
        .expect("cpu encode");

    // Test whichever GPU backend is compiled in (Auto: Metal then wgpu).
    let gpu = cera::model::vision_encoder_gpu::build_gpu_vision_encoder(
        cpu_enc.as_ref(),
        BackendPreference::Auto,
    );
    let Some(gpu) = gpu else {
        eprintln!("no GPU backend available; skipping");
        return;
    };
    let gpu_emb = gpu
        .encode_image(&pre.pixels, pre.grid_w, pre.grid_h)
        .expect("gpu encode");

    assert_eq!(cpu_emb.len(), gpu_emb.len(), "embedding length mismatch");
    let mut sum_abs = 0.0f64;
    for (i, (c, g)) in cpu_emb.iter().zip(gpu_emb.iter()).enumerate() {
        assert!(
            g.is_finite(),
            "GPU embedding non-finite at {i}: {g} (cpu={c})"
        );
        sum_abs += (c - g).abs() as f64;
    }
    let mean_abs = sum_abs / cpu_emb.len() as f64;
    // GPU dequantizes weights to f32 and sums in a different order than the CPU
    // quantized vec_dot, accumulated over 12 layers — a few-percent mean drift
    // is expected and the LLM tolerates it (see vl_bundle_gpu_path_generates).
    assert!(
        mean_abs < 0.5,
        "GPU vs CPU embedding mean_abs={mean_abs:.4} too large — likely a kernel bug"
    );
    eprintln!(
        "GPU vs CPU embedding parity: mean_abs={mean_abs:.4}, n={}",
        cpu_emb.len()
    );
}