aprender-serve 0.70.2

Pure Rust ML inference engine built from scratch - model serving for GGUF and safetensors
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
1451
1452
1453
1454
1455
1456
1457
1458
1459
1460
1461
1462
1463
1464
1465
1466
1467
1468
1469
1470
1471
1472
1473
1474
1475
1476
1477
1478
1479
1480
1481
1482
1483
1484
1485
1486
1487
1488
1489
1490
1491
1492
1493
1494
1495
1496
1497
1498
1499
1500
1501
1502
1503
1504
1505
1506
1507
1508
1509
1510
1511
1512
1513
1514
1515
1516
1517
1518
1519
1520
1521
1522
1523
1524
1525
1526
1527
1528
1529
1530
1531
1532
1533
1534
1535
1536
1537
1538
1539
1540
1541
1542
1543
1544
1545
1546
1547
1548
1549
1550
1551
1552
1553
1554
1555
1556
1557
1558
1559
1560
1561
1562
1563
1564
1565
1566
1567
1568
1569
1570
1571
1572
1573
1574
1575
1576
1577
1578
1579
1580
1581
1582
1583
1584
1585
1586
1587
1588
1589
1590
1591
1592
1593
1594
1595
1596
1597
1598
1599
1600
1601
1602
1603
1604
1605
1606
1607
1608
1609
1610
1611
1612
1613
1614
1615
1616
1617
1618
1619
1620
1621
1622
1623
1624
1625
1626
1627
1628
1629
1630
1631
1632
1633
1634
1635
1636
1637
1638
1639
1640
1641
1642
1643
1644
1645
1646
1647
1648
1649
1650
1651
1652
1653
1654
1655
1656
1657
1658
1659
1660
1661
1662
1663
1664
1665
1666
1667
1668
1669
1670
1671
1672
1673
1674
1675
1676
1677
1678
1679
1680
1681
1682
1683
1684
1685
1686
1687
1688
1689
1690
1691
1692
1693
1694
1695
1696
1697
1698
1699
1700
1701
1702
1703
1704
1705
1706
1707
1708
1709
1710
1711
1712
1713
1714
1715
1716
1717
1718

/// FALSIFY-CPU-GPU-003 jidoka tag emitted on stderr when a GPU init/parity
/// rejection forces a fallback. Locked in by `tests::cuda_fallback_log_prefix_is_contract_tagged`
/// to prevent regression to the verbose-only behaviour that v6 fixed.
///
/// See `contracts/apr-cpu-vs-gpu-output-parity-v1.yaml` and
/// `evidence/ship-007-layer-0-oracle-bisection-2026-05-03/findings-v6-parity-gate-fires-but-fallback-is-silent.md`.
///
/// The tag is deliberately prose: it is read by users, not by ticket triage.
/// Until #2405 it read `[apr-cpu-vs-gpu-output-parity-v1] CUDA path rejected`,
/// which told the user a contract filename and nothing they could act on.
pub(crate) const CUDA_FALLBACK_LOG_PREFIX: &str = "warning: GPU (CUDA) path rejected";

/// FALSIFY-CPU-GPU-005 jidoka tag emitted on stderr when wgpu init/forward
/// rejection forces a fallback. Locked in by
/// `tests::wgpu_fallback_log_prefix_is_contract_tagged` to prevent the same
/// silent-fallback regression class that #1428 closed for CUDA — the v1.2.0
/// contract predicts this tag at `gguf_gpu_generate.rs:317`-style rejection
/// points so users always see which backend was rejected without --verbose.
///
/// See `contracts/apr-cpu-vs-gpu-output-parity-v1.yaml` § FALSIFY-CPU-GPU-005.
///
/// Prose, not a contract filename — see [`CUDA_FALLBACK_LOG_PREFIX`] and #2405.
pub(crate) const WGPU_FALLBACK_LOG_PREFIX: &str = "warning: GPU (wgpu) path rejected";

/// f64-accumulated cosine similarity for FALSIFY-CPU-GPU-005 part b.
///
/// Numerically-stable companion to `cuda::mod_parity_gate::cosine_similarity` (which
/// lives behind `cfg(feature = "cuda")`). Lifted to this module so the future wgpu
/// cosine gate (predicted by contract `apr-cpu-vs-gpu-output-parity-v1` v1.2.0
/// FALSIFY-CPU-GPU-005 part b) can compare a wgpu single-step decode against a
/// CPU reference forward at init without taking a `--features cuda` build dependency.
///
/// Returns 0.0 when either input is zero-norm or the inputs differ in length —
/// this is the conservative "fail-closed" default that triggers fallback to CPU.
///
/// See `contracts/apr-cpu-vs-gpu-output-parity-v1.yaml` § FALSIFY-CPU-GPU-005
/// implementation_evidence line 201 for the gate algorithm.
pub(crate) fn cpu_vs_gpu_cosine_similarity(a: &[f32], b: &[f32]) -> f32 {
    if a.len() != b.len() || a.is_empty() {
        return 0.0;
    }
    let mut dot: f64 = 0.0;
    let mut norm_a: f64 = 0.0;
    let mut norm_b: f64 = 0.0;
    for (x, y) in a.iter().zip(b.iter()) {
        let x = f64::from(*x);
        let y = f64::from(*y);
        dot += x * y;
        norm_a += x * x;
        norm_b += y * y;
    }
    let denom = norm_a.sqrt() * norm_b.sqrt();
    if denom < 1e-12 {
        0.0
    } else {
        (dot / denom) as f32
    }
}

/// Printed when a sampled request skips a greedy-only wgpu decoder (#3760).
pub const WGPU_SAMPLING_NOTICE: &str = "[wgpu: the wgpu decoder is greedy-only; \
     sampling (--temperature > 0 with --top-k != 1) runs on the CPU (#3760)]";

/// Whether the greedy-only wgpu decoders may serve this request (#3760): a greedy one
/// yes; a sampled one no, with the notice printed, so the CPU loop that draws runs it.
#[cfg(feature = "gpu")]
fn wgpu_can_serve(temperature: f32, top_k: usize) -> bool {
    if crate::sampling::is_greedy(temperature, top_k) {
        return true;
    }
    eprintln!("{WGPU_SAMPLING_NOTICE}");
    false
}

/// #3757: may the GH-559 wgpu fallback be ATTEMPTED for this request?
///
/// Pure so the rule has a case table instead of living inline in two `if`s that
/// drifted. `FALSIFY-BACKEND-CUDA-HONESTY-001` covers `--backend cuda` on a
/// non-cuda build and asserts the run never prints `Backend: wgpu`; nothing
/// covered the BARE `apr run model.gguf`, which is the invocation that shipped
/// broken.
///
/// `accel_forced` is the decisive term and it is the same signal
/// `reconcile_accelerator` already consumes — the user explicitly asked for an
/// accelerator (`--gpu`, or `--backend cuda|wgpu|gpu`). Without it in this
/// predicate, `realizar`'s `default = [… "gpu"]` (which every
/// `cargo install aprender` gets, because apr-cli depends on realizar without
/// `default-features = false`) made the default path dequantize the model to
/// F32, fail wgpu's own cpu-parity gate, and fall back to CPU anyway.
#[cfg(feature = "gpu")]
#[must_use]
pub(crate) fn wgpu_fallback_allowed(
    no_gpu: bool,
    accel_forced: bool,
    has_legacy_quant: bool,
) -> bool {
    !no_gpu && accel_forced && !has_legacy_quant
}

/// GH-559: Try wgpu (Vulkan) generation as fallback when CUDA JIT fails.
/// Uses trueno's WgslForwardPass with dequantized F32 weights.
///
/// #3827: this said "Proven: cosine=0.999863 on Blackwell sm_121" and that is
/// WITHDRAWN, because #3757 measured the same path failing its own cpu-parity
/// gate on four GPUs including Blackwell — the architecture the claim named:
///
///   intel (Vulkan, no working driver)  0.955376
///   gx10  (GB10 — this IS Blackwell)   0.955046
///   mini  (Apple M4, Metal)            0.955169
///   RTX 4090 (sm_89)                   0.955376
///
/// The same value to four significant figures on four different GPUs and
/// drivers, so this is the wgpu path's own numerics rather than any one
/// driver — which also rules out "the proof holds and gx10 is special".
///
/// A proof comment that contradicts a measurement on the architecture it names
/// is worse than no comment, because it stops the next reader looking. Whether
/// 0.999863 was a different model, a different build, or a since-regressed
/// path is NOT known; #3827 owns deciding that, and whether wgpu is repaired or
/// removed. Until then the honest statement is the one above.
///
/// #3757 made this path opt-in (`accel_forced`), so a user no longer pays for
/// it unasked — that bounds the cost, it does not make the path correct.
/// The wgpu greedy loop over a KV-cached forward (#4264).
///
/// `forward(token, position)` runs one token through every layer at `position`,
/// writing that position's K/V, and returns its final hidden state; `pick` maps a
/// hidden state to the next token. Every prompt token before the last is
/// prefilled at its own position first. Before #4264 only the LAST prompt token
/// was fed, so decode attended over an empty cache and ignored the prompt.
#[cfg_attr(not(feature = "gpu"), allow(dead_code))]
fn wgpu_greedy_decode<E>(
    prompt: &[u32],
    max_tokens: usize,
    stop_tokens: &[u32],
    mut forward: impl FnMut(u32, usize) -> std::result::Result<Vec<f32>, E>,
    mut pick: impl FnMut(&[f32]) -> u32,
) -> std::result::Result<Vec<u32>, E> {
    let mut output_tokens = prompt.to_vec();
    let Some((_, head)) = prompt.split_last().filter(|_| max_tokens > 0) else {
        return Ok(output_tokens);
    };
    for (position, &token) in head.iter().enumerate() {
        forward(token, position)?;
    }
    for _ in 0..max_tokens {
        let position = output_tokens.len() - 1;
        let hidden = forward(output_tokens[position], position)?;
        let next = pick(&hidden);
        output_tokens.push(next);
        if crate::sampling::is_stop(next, stop_tokens) {
            break;
        }
    }
    Ok(output_tokens)
}

#[cfg(feature = "gpu")]
fn try_wgpu_generate(
    model: &crate::gguf::OwnedQuantizedModel,
    input_tokens: &[u32],
    gen_config: &crate::gguf::QuantizedGenerateConfig,
    verbose: bool,
) -> Result<(Vec<u32>, bool)> {
    use crate::gpu::adapters::wgpu_adapter;

    if !trueno::backends::gpu::GpuDevice::is_available() {
        return Err(RealizarError::InferenceError("wgpu not available".into()));
    }

    let gpu = trueno::backends::gpu::GpuDevice::new()
        .map_err(|e| RealizarError::InferenceError(format!("wgpu init: {e}")))?;

    // FALSIFY-CPU-GPU-005: wgpu lifecycle visible without --verbose so users
    // see which backend actually serves their tokens after CUDA fallback.
    let _ = verbose;
    eprintln!("Backend: wgpu (Vulkan)");

    let config = model.config();
    let hidden_dim = config.hidden_dim;
    let num_layers = config.num_layers;
    let num_heads = config.num_heads;
    let num_kv_heads = config.num_kv_heads;
    let head_dim = hidden_dim / num_heads;
    let intermediate_dim = config.intermediate_dim;
    let vocab_size = config.vocab_size;
    let eps = config.eps;
    let kv_dim = num_kv_heads * head_dim;

    // Create forward pass and upload dequantized weights
    let mut fwd = trueno::backends::gpu::WgslForwardPass::new(
        gpu.device, gpu.queue,
        hidden_dim, num_heads, num_kv_heads, head_dim, intermediate_dim,
    );
    // #4056: the WGSL RMSNorm takes the model's eps (it hardcoded 1e-6).
    fwd.set_rms_norm_eps(eps);

    // C-WGPU-Q4K-001: Upload raw Q4K bytes for projection weights.
    // encode_matmul() auto-selects Q4K GEMV when M=1 and Q4K weights exist.
    let raw_q4k = wgpu_adapter::raw_q4k_weights(model);
    let q4k_names: std::collections::HashSet<String> =
        raw_q4k.iter().map(|(n, _, _, _)| n.clone()).collect();
    for (name, data, _rows, _cols) in &raw_q4k {
        fwd.upload_q4k_weight(name, data);
    }

    // Upload F32 weights for norms, biases, and non-Q4K tensors.
    //
    // #2378 finding 8: this used to call `dequant_model_weights(model)` and then
    // skip the UPLOAD for names already sent as raw Q4K. The skip was too late --
    // an F32 Vec had already been materialized for every one of those tensors and
    // was then dropped. On a Q4_K model that is the bulk of the weights.
    // `_except` skips the dequantization itself, so the allocation never happens.
    let weights = wgpu_adapter::dequant_model_weights_except(model, &q4k_names)?;
    for (name, data, _rows, _cols) in &weights {
        fwd.upload_weight(name, data);
    }

    // Get output norm and LM head weights
    let output_norm = model.output_norm_weight();
    let lm_head_f32: Vec<f32> = weights.iter()
        .find(|(n, _, _, _)| n == "lm_head")
        .map(|(_, d, _, _)| d.clone())
        .unwrap_or_default();

    // KV caches
    let max_seq = gen_config.max_tokens + input_tokens.len() + 16;
    let mut kv_caches: Vec<(Vec<f32>, Vec<f32>)> = (0..num_layers)
        .map(|_| (Vec::with_capacity(max_seq * kv_dim), Vec::with_capacity(max_seq * kv_dim)))
        .collect();

    // FALSIFY-CPU-GPU-006 (#1864): multi-step CPU vs wgpu parity gate.
    //
    // The pre-#1864 GGUF wgpu path had NO parity gate at all — it loaded
    // weights, ran the autoregressive loop, and returned. Qwen2.5-7B Q4K
    // shipped "ampiezza"-style gibberish straight to the user with exit 0.
    // The .apr wgpu path had a single-step gate (FALSIFY-CPU-GPU-005) but
    // that's also insufficient: 7B's wgpu KV cache drifts each step even
    // when step 0 cosine ≥ 0.99.
    //
    // This gate runs CPU vs wgpu in lockstep for N steps (default 3, override
    // via APR_WGPU_PARITY_STEPS in [1, 16]). Both paths advance through the
    // same deterministic token sequence (CPU argmax), and we cosine-compare
    // the full vocab-size logit vectors at every step. ANY cosine < 0.99
    // aborts with the WGPU_FALLBACK_LOG_PREFIX tag so the caller falls back
    // to CPU rather than ship silent drift.
    //
    // Cost: N forward passes at init (~0.5-2s on 7B Q4K) — paid once per
    // `apr run`, not per token. See contracts/apr-cpu-vs-gpu-output-parity-v1.yaml.
    {
        const MULTI_STEP_PROBE_DEFAULT: usize = 3;
        let multi_step_probe: usize = std::env::var("APR_WGPU_PARITY_STEPS")
            .ok()
            .and_then(|s| s.parse::<usize>().ok())
            .filter(|&n| (1..=16).contains(&n))
            .unwrap_or(MULTI_STEP_PROBE_DEFAULT);

        let probe_max_seq = multi_step_probe + 1;
        let mut cpu_cache = crate::gguf::OwnedQuantizedKVCache::from_config(&config, probe_max_seq);
        let mut probe_kv_caches: Vec<(Vec<f32>, Vec<f32>)> = (0..num_layers)
            .map(|_| (Vec::with_capacity(probe_max_seq * kv_dim), Vec::with_capacity(probe_max_seq * kv_dim)))
            .collect();
        let mut probe_token = *input_tokens.first().unwrap_or(&0);

        for probe_step in 0..multi_step_probe {
            let cpu_logits = match model.forward_single_with_cache(probe_token, &mut cpu_cache, probe_step) {
                Ok(l) => l,
                Err(e) => {
                    eprintln!(
                        "{}, attempting fallback: CPU probe step {} forward failed: {}",
                        WGPU_FALLBACK_LOG_PREFIX, probe_step, e
                    );
                    return Err(RealizarError::InferenceError(format!("wgpu parity gate: CPU probe step {probe_step} failed: {e}")));
                }
            };

            let mut hidden = model.embed(&[probe_token]);
            for layer_idx in 0..num_layers {
                let prefix = format!("layer.{layer_idx}");
                let (ref mut kv_k, ref mut kv_v) = probe_kv_caches[layer_idx];
                if let Err(e) = fwd.forward_layer(&mut hidden, &prefix, probe_step, kv_k, kv_v) {
                    eprintln!(
                        "{}, attempting fallback: wgpu probe step {} layer {} failed: {}",
                        WGPU_FALLBACK_LOG_PREFIX, probe_step, layer_idx, e
                    );
                    return Err(RealizarError::InferenceError(format!("wgpu parity gate: step {probe_step} layer {layer_idx} failed: {e}")));
                }
            }
            let sq_sum: f32 = hidden.iter().map(|x| x * x).sum();
            let rms = (sq_sum / hidden.len() as f32 + eps).sqrt();
            let normed: Vec<f32> = hidden
                .iter()
                .zip(output_norm.iter())
                .map(|(x, g)| (x / rms) * g)
                .collect();
            let mut wgpu_logits = vec![0.0_f32; vocab_size];
            for i in 0..vocab_size {
                let row = &lm_head_f32[i * hidden_dim..(i + 1) * hidden_dim];
                wgpu_logits[i] = row.iter().zip(normed.iter()).map(|(w, x)| w * x).sum();
            }

            let cos = cpu_vs_gpu_cosine_similarity(&cpu_logits, &wgpu_logits);
            if !(cos.is_finite() && cos >= 0.99) {
                eprintln!(
                    "{}, attempting fallback: cosine vs CPU = {:.6} (< 0.99) at step {}/{}",
                    WGPU_FALLBACK_LOG_PREFIX, cos, probe_step + 1, multi_step_probe
                );
                return Err(RealizarError::InferenceError(format!(
                    "wgpu parity gate: cosine={cos:.6} < 0.99 at step {}/{}",
                    probe_step + 1, multi_step_probe
                )));
            }

            // Advance both paths via CPU argmax (deterministic).
            let mut best_idx: u32 = 0;
            let mut best_val = f32::NEG_INFINITY;
            for (i, &v) in cpu_logits.iter().enumerate() {
                if v > best_val {
                    best_val = v;
                    best_idx = i as u32;
                }
            }
            probe_token = best_idx;
        }
    }

    // Prefill, then autoregressive generation (#4264).
    let output_tokens = wgpu_greedy_decode::<RealizarError>(
        input_tokens,
        gen_config.max_tokens,
        &gen_config.stop_tokens,
        |token_id, position| {
            let mut hidden = model.embed(&[token_id]);
            for (layer_idx, (kv_k, kv_v)) in kv_caches.iter_mut().enumerate() {
                let prefix = format!("layer.{layer_idx}");
                fwd.forward_layer(&mut hidden, &prefix, position, kv_k, kv_v)
                    .map_err(|e| RealizarError::InferenceError(format!("wgpu layer {layer_idx}: {e}")))?;
            }
            Ok(hidden)
        },
        |hidden| {
            // Output norm + LM head (CPU — small cost), greedy: the FIRST maximum wins.
            let sq_sum: f32 = hidden.iter().map(|x| x * x).sum();
            let rms = (sq_sum / hidden.len() as f32 + eps).sqrt();
            let normed: Vec<f32> = hidden.iter().zip(output_norm.iter())
                .map(|(x, g)| (x / rms) * g)
                .collect();
            let mut best_idx = 0u32;
            let mut best_val = f32::NEG_INFINITY;
            for i in 0..vocab_size {
                let row = &lm_head_f32[i * hidden_dim..(i + 1) * hidden_dim];
                let logit: f32 = row.iter().zip(normed.iter()).map(|(w, x)| w * x).sum();
                if logit > best_val {
                    best_val = logit;
                    best_idx = i as u32;
                }
            }
            best_idx
        },
    )?;

    Ok((output_tokens, true)) // true = used GPU (wgpu)
}

/// Try GGUF GPU generation. Takes model by value to avoid expensive clone (~1GB).
/// Returns `Ok(result)` on GPU success, `Err(model)` to return model for CPU fallback.
#[cfg(feature = "cuda")]
fn try_gguf_gpu_generate(
    model: crate::gguf::OwnedQuantizedModel,
    input_tokens: &[u32],
    gen_config: &crate::gguf::QuantizedGenerateConfig,
    verbose: bool,
) -> std::result::Result<Result<(Vec<u32>, bool)>, Box<crate::gguf::OwnedQuantizedModel>> {
    use crate::gguf::OwnedQuantizedModelCuda;

    // #4268: the device KV holds the whole turn. It was a flat 2048, and the
    // engine refuses to run past it on the device, so a longer turn would
    // leave the GPU for the CPU instead of being served.
    let kv_len = device_kv_len(&model, input_tokens.len(), gen_config.max_tokens);
    let mut cuda_model = match OwnedQuantizedModelCuda::with_max_seq_len(model, 0, kv_len) {
        Ok(m) => m,
        Err(e) => {
            if verbose {
                eprintln!("Backend: CPU (GPU unavailable: {})", e);
            }
            // Model is preserved inside CudaInitError for CPU fallback.
            // Boxed to keep the `Err` variant small (clippy::result_large_err).
            return Err(Box::new(e.into_model()));
        },
    };

    if verbose {
        eprintln!(
            "Backend: GPU ({}, {} MB VRAM)",
            cuda_model.device_name(),
            cuda_model.vram_mb()
        );
    }

    // #3973: three states, each handled here. Routing is unchanged: only a MISMATCH
    // leaves the GPU. A not-measured probe proceeds, and says it is unvalidated.
    match validate_gpu_first_token(&mut cuda_model, gen_config, input_tokens) {
        F2Outcome::Mismatch => {
            // Validation failed — extract model back for CPU fallback
            return Err(Box::new(cuda_model.into_model()));
        },
        F2Outcome::NotMeasured { reason } => {
            eprintln!("[GH-480] F2 validation NOT MEASURED — {reason}. GPU output is UNVALIDATED (#3973)");
        },
        F2Outcome::Validated { .. } => {},
    }

    // Reuse existing CUDA model — generate_gpu_resident() creates fresh KV cache
    // and resets GPU KV positions internally, so validation doesn't "consume" it.
    mark_generation_start(); // #3981: setup (upload + F2) ends here
    if gen_config.trace {
        // #4268: `--trace` keeps the instrumented loop; the engine has no brick profiler.
        let result = cuda_model
            .generate_gpu_resident(input_tokens, gen_config)
            .map(|tokens| (tokens, true))
            .map_err(|e| RealizarError::InferenceError(format!("GPU generation failed: {}", e)));
        return Ok(result);
    }
    // #4268: the one engine. A forward failure no longer ends the run: the
    // session says so on stderr, moves to the CPU and replays the turn there,
    // and `used_gpu` comes back false.
    let mut session = crate::gguf::dense_session::DenseSession::new(
        crate::gguf::dense_session::DenseForward::cuda(cuda_model),
    );
    Ok(crate::gguf::dense_session::dense_turn(
        &mut session,
        input_tokens,
        gen_config,
    ))
}

/// #4268: the device KV length for a turn: the prompt plus its budget, held to
/// the model's context, and never below the 2048 every CUDA build used before.
#[cfg(feature = "cuda")]
fn device_kv_len(
    model: &crate::gguf::OwnedQuantizedModel,
    prompt_len: usize,
    max_tokens: usize,
) -> usize {
    prompt_len
        .saturating_add(max_tokens)
        .min(model.config.context_length)
        .max(2048)
}

/// Run GGUF generation with GPU or CPU
#[allow(unused_variables)] // config used only in CUDA feature
fn run_gguf_generate(
    model: crate::gguf::OwnedQuantizedModel,
    input_tokens: &[u32],
    gen_config: &crate::gguf::QuantizedGenerateConfig,
    config: &InferenceConfig,
) -> Result<(Vec<u32>, bool, bool)> {
    // #3826: the third element is `gpu_attempted` — whether a GPU backend was
    // ENTERED, whatever the outcome. `used_gpu` alone collapses "never tried"
    // and "tried and refused" into one `false`, which is how
    // `apr run --format json` reported `"fell_back": false` on a run whose own
    // stderr said `attempting fallback`.
    // M32c.2.1: short-circuit MoE forward attempts BEFORE any GPU/CPU
    // dispatch. M32c.2 made `from_gguf` succeed for qwen3_moe by routing
    // to `from_gguf_for_moe` (which leaves dense FFN tensor refs as
    // zero-byte placeholders). Without this guard, the wgpu/CUDA forward
    // path tries to bind those zero-byte buffers and panics deep in
    // `wgpu_core::create_bind_group` with `Buffer with 'layer.0.up_proj'
    // label binding size is zero`. M32c.2.2 will replace this guard with
    // an actual MoE forward via `moe_forward_token`. See
    // contracts/qwen3-moe-forward-v1.yaml § FALSIFY-QW3-MOE-FORWARD-003.
    let canonical_arch =
        crate::tensor_names::normalize_architecture(&model.config.architecture);
    if canonical_arch == "qwen3_moe" {
        return Err(RealizarError::UnsupportedOperation {
            operation: "moe_forward_dispatch".to_string(),
            reason: format!(
                "Architecture '{}' (canonical 'qwen3_moe') uses Mixture-of-Experts FFN. \
                 Load step succeeded via QuantizedGGUFTransformer::from_gguf_for_moe (M32c.2) \
                 with all 4 contract-declared MoE tensors per layer present, but the \
                 forward dispatch is not yet wired to moe_forward_token in \
                 gpu/scheduler/moe_dispatch.rs. Tracked under contract qwen3-moe-forward-v1 \
                 (M32 staged plan: M32a/b/c.1/c.2 SHIPPED; M32c.2.1 forward-refusal \
                 IN PROGRESS; M32c.2.2 forward-wiring + M32d numerical parity PENDING). \
                 See contracts/qwen3-moe-forward-v1.yaml.",
                model.config.architecture
            ),
        });
    }

    let has_legacy_quant = model_has_legacy_quant(&model);

    // GPU path: pass model by value (zero-clone) — model is returned on failure for CPU fallback
    // #3826: an ATTEMPT is recorded before the outcome is known. Both arms below
    // are attempts: `Ok` produced the tokens, `Err` had its result refused and
    // handed the model back for the CPU to redo. Only the second is a fallback,
    // and it is the one that reported nothing.
    #[allow(unused_mut)]
    let mut gpu_attempted = false;
    #[cfg(feature = "cuda")]
    let model = if !config.no_gpu && !has_legacy_quant {
        gpu_attempted = true;
        match try_gguf_gpu_generate(model, input_tokens, gen_config, config.verbose) {
            Ok(result) => return result.map(|(t, u)| (t, u, true)),
            Err(returned_model) => *returned_model, // GPU failed, use returned model for CPU
        }
    } else {
        model
    };

    // GH-559: wgpu fallback — try Vulkan compute before CPU.
    // #3827: the "Proven: wgpu cosine=0.999863 on Blackwell sm_121" claim that
    // stood here is WITHDRAWN — #3757 measured 0.955046 on gx10, which IS
    // Blackwell. See try_wgpu_generate's doc comment for the four-host table.
    // #3760: the wgpu decode loop is greedy-only (an inline argmax over the LM head);
    // a sampled request runs on the CPU loop, which draws, and says so.
    //
    // #3757: attempted ONLY when the user explicitly asked for an accelerator.
    // `realizar`'s own `default = ["server", "cli", "gpu"]` reaches every
    // `cargo install aprender`, because apr-cli depends on it without
    // `default-features = false` — so this block is compiled into the nominally
    // CPU-only default binary and, gated on `!no_gpu` alone, ran on the BARE
    // `apr run model.gguf`. It dequantized the model to F32 (1726.8 MB on
    // qwen2.5-coder-1.5b), failed wgpu's own cpu-parity gate at cosine 0.9554,
    // and fell back — 7607 ms against `--no-gpu`'s 3035 ms for the same answer
    // from the same CPU backend. The same cosine to four figures on intel, gx10,
    // mini and an RTX 4090, so it is the wgpu path's numerics, not a driver.
    //
    // `--gpu` on such a build already REFUSES ("no GPU backend compiled in") and
    // `--backend wgpu` already refuses to report a fallback as success. The bare
    // default was the only path that paid for wgpu silently.
    #[cfg(feature = "gpu")]
    if wgpu_fallback_allowed(config.no_gpu, config.accel_forced, has_legacy_quant)
        && wgpu_can_serve(gen_config.temperature, gen_config.top_k)
    {
        gpu_attempted = true;
        match try_wgpu_generate(&model, input_tokens, gen_config, config.verbose) {
            Ok((t, u)) => return Ok((t, u, true)),
            Err(e) => {
                if config.verbose {
                    eprintln!("Backend: CPU (wgpu unavailable: {})", e);
                }
            }
        }
    }

    log_cpu_backend(config.verbose, has_legacy_quant);
    mark_generation_start(); // #3981
    let tokens = if gen_config.trace {
        // #4268: `--trace` keeps the instrumented loop; the engine has no brick profiler.
        model.generate_with_cache(input_tokens, gen_config)
    } else {
        let mut session = crate::gguf::dense_session::DenseSession::new(
            crate::gguf::dense_session::DenseForward::cpu(std::sync::Arc::new(model)),
        );
        crate::gguf::dense_session::dense_turn(&mut session, input_tokens, gen_config)
            .map(|(tokens, _)| tokens)
    }
    .map_err(|e| RealizarError::InferenceError(format!("CPU generation failed: {}", e)))?;
    // #3826: the CPU answered. `gpu_attempted` distinguishes "CPU because
    // nothing else was tried" from "CPU because the accelerator's result was
    // refused" — the second is the fallback a consumer needs to see.
    Ok((tokens, false, gpu_attempted))
}

/// Run APR model inference (PAR-302, PMAT-APR-CUDA-001)
///
/// Uses AprV2ModelCuda for GPU acceleration when available, falls back to
/// AprTransformer (CPU with proper RoPE and SwiGLU) otherwise.
/// PMAT-237: APR inference now uses PreparedTokens (compile-time enforced chat template).
/// Previously bypassed PreparedTokens entirely via prepare_apr_input_tokens().
fn run_apr_inference(
    config: &InferenceConfig,
    prepared: &PreparedTokens,
) -> Result<InferenceResult> {
    if config.verbose {
        eprintln!("Loading APR model: {}", config.model_path.display());
    }

    let load_start = Instant::now();
    let input_tokens = prepared.tokens();
    let input_token_count = prepared.input_count();

    // Try GPU path first
    #[cfg(feature = "cuda")]
    if !config.no_gpu {
        if let Some(result) =
            try_apr_cuda_inference(config, input_tokens, input_token_count, load_start)
        {
            return result;
        }
    }

    // GH-559: wgpu fallback for APR models — try Vulkan before CPU.
    // #3757: explicit accelerator request only — see the GGUF path above.
    #[cfg(feature = "gpu")]
    if wgpu_fallback_allowed(config.no_gpu, config.accel_forced, false)
        && wgpu_can_serve(config.temperature, config.top_k)
    {
        match try_apr_wgpu_inference(config, input_tokens, input_token_count, load_start) {
            Some(Ok(result)) => return Ok(result),
            Some(Err(e)) => {
                if config.verbose {
                    eprintln!("Backend: CPU (wgpu failed: {})", e);
                }
            }
            None => {
                if config.verbose {
                    eprintln!("Backend: CPU (wgpu not available)");
                }
            }
        }
    }

    // CPU fallback: AprTransformer with RoPE and SwiGLU
    run_apr_cpu_inference(config, input_tokens, input_token_count, load_start)
}

/// GH-559: Try wgpu (Vulkan) inference for APR models.
/// Returns None if wgpu not available, Some(Result) if attempted.
#[cfg(feature = "gpu")]
fn try_apr_wgpu_inference(
    config: &InferenceConfig,
    input_tokens: &[u32],
    input_token_count: usize,
    load_start: Instant,
) -> Option<Result<InferenceResult>> {
    use crate::apr::MappedAprModel;
    use crate::gpu::adapters::wgpu_adapter;
    use trueno::backends::gpu::GpuDevice;

    if !GpuDevice::is_available() {
        return None;
    }

    let gpu = match GpuDevice::new() {
        Ok(g) => g,
        Err(e) => {
            // FALSIFY-CPU-GPU-005: wgpu init failure is a backend-fallback decision —
            // user must see why this backend was rejected without --verbose.
            // Emit BOTH the contract-tagged prefix (greppable, locked in by
            // `wgpu_fallback_log_prefix_is_contract_tagged` test) AND the
            // existing [GH-559] tag (preserved for runbook continuity).
            eprintln!("{}, attempting fallback: {}", WGPU_FALLBACK_LOG_PREFIX, e);
            eprintln!("[GH-559] wgpu init failed: {}", e);
            return None;
        }
    };

    // FALSIFY-CPU-GPU-005: wgpu lifecycle visible without --verbose. Symmetric to
    // FALSIFY-CPU-GPU-003's CUDA-fallback log so users always know which backend
    // actually serves their tokens.
    eprintln!("Backend: wgpu (Vulkan)");

    // Load model
    let mapped = match MappedAprModel::from_path(&config.model_path) {
        Ok(m) => m,
        Err(_) => return None,
    };
    let model = match crate::gguf::OwnedQuantizedModel::from_apr(&mapped) {
        Ok(m) => m,
        Err(_) => return None,
    };

    let cfg = model.config();
    let hidden_dim = cfg.hidden_dim;
    let num_layers = cfg.num_layers;
    let num_heads = cfg.num_heads;
    let num_kv_heads = cfg.num_kv_heads;
    let head_dim = hidden_dim / num_heads;
    let intermediate_dim = cfg.intermediate_dim;
    let vocab_size = cfg.vocab_size;
    let eps = cfg.eps;
    let kv_dim = num_kv_heads * head_dim;
    // Resolve stop tokens from model config + sibling tokenizer
    let mut stop_toks: Vec<u32> = cfg.eos_token_id.into_iter().collect();
    let extra = crate::infer::resolve_apr_stop_tokens(
        cfg.eos_token_id, &[], &config.model_path,
    );
    for t in &extra {
        if !stop_toks.contains(t) { stop_toks.push(*t); }
    }
    let mut gen_config = crate::gguf::QuantizedGenerateConfig {
        max_tokens: config.max_tokens,
        stop_tokens: stop_toks,
        trace: config.trace,
        ..Default::default()
    };
    // PMAT-823: APR wgpu path — forward all sampling params (was hardcoded greedy).
    config.apply_sampling_to(&mut gen_config);

    // Dequantize and upload weights
    let weights = match wgpu_adapter::dequant_model_weights(&model) {
        Ok(w) => w,
        Err(e) => return Some(Err(e)),
    };

    let mut fwd = trueno::backends::gpu::WgslForwardPass::new(
        gpu.device, gpu.queue,
        hidden_dim, num_heads, num_kv_heads, head_dim, intermediate_dim,
    );
    // #4056: the WGSL RMSNorm takes the model's eps (it hardcoded 1e-6).
    fwd.set_rms_norm_eps(eps);

    for (name, data, _rows, _cols) in &weights {
        fwd.upload_weight(name, data);
    }
    // KV cache initialized by caller (no init_kv_cache needed — API change)

    let output_norm = model.output_norm_weight();
    let lm_head_f32: Vec<f32> = weights.iter()
        .find(|(n, _, _, _)| n == "lm_head")
        .map(|(_, d, _, _)| d.clone())
        .unwrap_or_default();

    let max_seq = gen_config.max_tokens + input_tokens.len() + 16;
    let mut kv_caches: Vec<(Vec<f32>, Vec<f32>)> = (0..num_layers)
        .map(|_| (Vec::with_capacity(max_seq * kv_dim), Vec::with_capacity(max_seq * kv_dim)))
        .collect();

    // FALSIFY-CPU-GPU-005 part b: wgpu cosine parity gate.
    //
    // Symmetric to FALSIFY-CPU-GPU-003's CUDA parity_gate (cuda::mod_parity_gate).
    // Runs CPU vs wgpu side-by-side for MULTI_STEP_PROBE tokens, advancing both
    // KV caches in lockstep. Cosine-compares logits at every step. If any step
    // fails the 0.99 threshold, emit `WGPU_FALLBACK_LOG_PREFIX` and return None
    // so we fall back to CPU rather than ship silent wgpu gibberish.
    //
    // **Why multi-step?** Single-step (the pre-#1864 design) caught divergence
    // on the first forward but missed autoregressive drift in the KV cache.
    // Qwen2.5-7B Q4K shipped "ampiezza"-style gibberish via wgpu in the v0.34.0
    // window because the first-token cosine was ≥ 0.99 but every subsequent
    // step diverged as the KV cache accumulated error. The multi-step gate
    // catches that without paying for a full max-tokens probe.
    //
    // **Cost.** Each extra step ~ 5-50ms on a 1.5B Q4K, ~30-200ms on 7B Q4K.
    // MULTI_STEP_PROBE=3 keeps init overhead under 1s for the common case;
    // tunable via APR_WGPU_PARITY_STEPS env var (1..16) for diagnostic runs.
    //
    // See contracts/apr-cpu-vs-gpu-output-parity-v1.yaml § FALSIFY-CPU-GPU-006
    // (multi_step_parity_gate) for the formal invariant.
    {
        const MULTI_STEP_PROBE_DEFAULT: usize = 3;
        let multi_step_probe: usize = std::env::var("APR_WGPU_PARITY_STEPS")
            .ok()
            .and_then(|s| s.parse::<usize>().ok())
            .filter(|&n| (1..=16).contains(&n))
            .unwrap_or(MULTI_STEP_PROBE_DEFAULT);

        // Reuse a single CPU cache + wgpu KV cache across all probe steps so
        // both paths see identical autoregressive state. max_seq sized to fit
        // the probe.
        let probe_max_seq = multi_step_probe + 1;
        let mut cpu_cache = crate::gguf::OwnedQuantizedKVCache::from_config(cfg, probe_max_seq);
        let mut probe_kv_caches: Vec<(Vec<f32>, Vec<f32>)> = (0..num_layers)
            .map(|_| (Vec::with_capacity(probe_max_seq * kv_dim), Vec::with_capacity(probe_max_seq * kv_dim)))
            .collect();
        let mut probe_token = *input_tokens.first().unwrap_or(&0);

        for step in 0..multi_step_probe {
            // CPU reference logits at this step.
            let cpu_logits = match model.forward_single_with_cache(probe_token, &mut cpu_cache, step) {
                Ok(l) => l,
                Err(e) => {
                    eprintln!(
                        "{}, attempting fallback: CPU probe step {} forward failed: {}",
                        WGPU_FALLBACK_LOG_PREFIX, step, e
                    );
                    return None;
                }
            };

            // wgpu single-step replay at the same position.
            let mut hidden = model.embed(&[probe_token]);
            for layer_idx in 0..num_layers {
                let prefix = format!("layer.{layer_idx}");
                let (ref mut kv_k, ref mut kv_v) = probe_kv_caches[layer_idx];
                if let Err(e) = fwd.forward_layer(&mut hidden, &prefix, step, kv_k, kv_v) {
                    eprintln!(
                        "{}, attempting fallback: wgpu probe step {} layer {} failed: {}",
                        WGPU_FALLBACK_LOG_PREFIX, step, layer_idx, e
                    );
                    return None;
                }
            }
            // Output norm + LM head (mirrors the autoregressive loop body).
            let sq_sum: f32 = hidden.iter().map(|x| x * x).sum();
            let rms = (sq_sum / hidden.len() as f32 + eps).sqrt();
            let normed: Vec<f32> = hidden
                .iter()
                .zip(output_norm.iter())
                .map(|(x, g)| (x / rms) * g)
                .collect();
            let mut wgpu_logits = vec![0.0_f32; vocab_size];
            for i in 0..vocab_size {
                let row = &lm_head_f32[i * hidden_dim..(i + 1) * hidden_dim];
                wgpu_logits[i] = row.iter().zip(normed.iter()).map(|(w, x)| w * x).sum();
            }

            let cos = cpu_vs_gpu_cosine_similarity(&cpu_logits, &wgpu_logits);
            if !(cos.is_finite() && cos >= 0.99) {
                eprintln!(
                    "{}, attempting fallback: cosine vs CPU = {:.6} (< 0.99) at step {}/{}",
                    WGPU_FALLBACK_LOG_PREFIX, cos, step + 1, multi_step_probe
                );
                return None;
            }

            // Advance both paths deterministically via CPU argmax.
            // (Greedy choice; matches what the autoregressive loop will do for
            // step 0 in the common case. Probe is contract verification, not
            // user-visible generation.)
            let mut best_idx: u32 = 0;
            let mut best_val = f32::NEG_INFINITY;
            for (i, &v) in cpu_logits.iter().enumerate() {
                if v > best_val {
                    best_val = v;
                    best_idx = i as u32;
                }
            }
            probe_token = best_idx;
        }
    }

    let model_load_ms = load_start.elapsed().as_millis() as f64;

    // Autoregressive generation
    let infer_start = Instant::now();
    // Prefill, then autoregressive generation (#4264).
    let decoded = wgpu_greedy_decode::<RealizarError>(
        input_tokens,
        gen_config.max_tokens,
        &gen_config.stop_tokens,
        |token_id, position| {
            let mut hidden = model.embed(&[token_id]);
            for (layer_idx, (kv_k, kv_v)) in kv_caches.iter_mut().enumerate() {
                let prefix = format!("layer.{layer_idx}");
                fwd.forward_layer(&mut hidden, &prefix, position, kv_k, kv_v)
                    .map_err(|e| RealizarError::InferenceError(format!("wgpu layer {layer_idx}: {e}")))?;
            }
            Ok(hidden)
        },
        |hidden| {
            // Output norm (RMSNorm with output_norm gamma), then the LM head
            // argmax on CPU; the FIRST maximum wins.
            let sq_sum: f32 = hidden.iter().map(|x| x * x).sum();
            let rms = (sq_sum / hidden.len() as f32 + eps).sqrt();
            let normed: Vec<f32> = hidden.iter().zip(output_norm.iter())
                .map(|(x, g)| (x / rms) * g)
                .collect();
            let mut best_idx = 0u32;
            let mut best_val = f32::NEG_INFINITY;
            for i in 0..vocab_size {
                let row = &lm_head_f32[i * hidden_dim..(i + 1) * hidden_dim];
                let logit: f32 = row.iter().zip(normed.iter()).map(|(w, x)| w * x).sum();
                if logit > best_val {
                    best_val = logit;
                    best_idx = i as u32;
                }
            }
            best_idx
        },
    );
    let output_tokens = match decoded {
        Ok(t) => t,
        Err(e) => return Some(Err(e)),
    };

    let inference_ms = infer_start.elapsed().as_millis() as f64;
    let tokens_generated = output_tokens.len() - input_token_count;

    // Decode tokens
    let text = crate::infer::decode_apr_tokens(&config.model_path, &output_tokens[input_token_count..]);

    Some(Ok(InferenceResult {
        text,
        tokens: output_tokens,
        input_token_count,
        generated_token_count: tokens_generated,
        inference_ms,
        load_ms: model_load_ms,
        tok_per_sec: if inference_ms > 0.0 { tokens_generated as f64 / (inference_ms / 1000.0) } else { 0.0 },
        generation_ms: Some(inference_ms), // #3981: this path starts its clock AFTER setup, right before generation
        format: "APR".to_string(),
        used_gpu: true,
        gpu_attempted: true,
    }))
}

/// GH-318: Map APR architecture string to chat template hint using contract.
///
/// Uses `normalize_architecture()` from tensor-names-v1.yaml — no fallback.
/// Unknown architectures default to "llama" (safest default per contract).
fn apr_arch_to_template_hint(apr_arch: &str, _model_name: &str) -> &'static str {
    crate::tensor_names::normalize_architecture(apr_arch)
}

/// Metadata captured from the model config before it is moved into CUDA.
#[cfg(feature = "cuda")]
struct AprCudaModelInfo {
    arch: String,
    num_layers: usize,
    vocab_size: usize,
    hidden_dim: usize,
}

/// The notice a CUDA decline owes the user when the model's quantization has no verified
/// GPU kernel — `None` when every inspected projection is GPU-eligible (#3908).
///
/// A FUNCTION rather than an inline `eprintln!` so the verdict and its wording can be
/// asserted without capturing stderr or owning a GPU.
///
/// WHY THIS EXIT NEEDS AN UNCONDITIONAL NOTICE AND THE TWO ABOVE IT DO NOT. The standard
/// its sibling carries — "CUDA init failure MUST be visible without --verbose … user saw
/// downstream wgpu gibberish without ever knowing CUDA was rejected" — applies where the
/// condition is CUDA-SPECIFIC, because then the other backends proceed and may SUCCEED and
/// the rejection becomes unobservable. Verified rather than assumed:
/// `run_apr_cpu_inference` delegates to `run_apr_quantized_cpu_inference`, which loads
/// through the SAME `OwnedQuantizedModel`, and `try_apr_wgpu_inference` re-reads the same
/// `MappedAprModel`. So a `from_path` or `from_apr` failure fails EVERY backend and
/// surfaces loudly on its own; only the quant whitelist is a CUDA-only verdict that leaves
/// the run going.
///
/// Measured cost of its absence: a bf16 `.apr` declines here, execution continues, wgpu
/// fails with "Unsupported quantization type 30 for WGPU dequant", and two readers
/// concluded the defect was wgpu routing. CUDA had already declined, correctly, silently.
#[cfg(feature = "cuda")]
fn apr_cuda_decline_notice(model: &crate::gguf::OwnedQuantizedModel) -> Option<String> {
    let qtype = model.first_gpu_unsupported_quant()?;
    Some(format!(
        "{CUDA_FALLBACK_LOG_PREFIX}: no verified GPU kernel for quantization type {qtype} — \
         CUDA declined this model before wgpu or CPU was tried. Any backend error after this \
         line is downstream of THIS decision, not its cause. Convert with \
         `apr convert --quantize fp16` (type 1 is GPU-eligible), or run with --no-gpu (#3908)."
    ))
}

/// Load an APR model and initialize it on CUDA, returning None on any failure.
#[cfg(feature = "cuda")]
fn load_apr_cuda_model(
    model_path: &std::path::Path,
    verbose: bool,
) -> Option<(crate::gguf::OwnedQuantizedModelCuda, AprCudaModelInfo)> {
    use crate::apr::MappedAprModel;
    use crate::gguf::{OwnedQuantizedModel, OwnedQuantizedModelCuda};

    let mapped = MappedAprModel::from_path(model_path).map_err(|e| {
        if verbose { eprintln!("[APR-CUDA] MappedAprModel::from_path failed: {}", e); }
    }).ok()?;

    let model = OwnedQuantizedModel::from_apr(&mapped).map_err(|e| {
        if verbose { eprintln!("[APR-CUDA] OwnedQuantizedModel::from_apr failed: {}", e); }
    }).ok()?;

    if let Some(notice) = apr_cuda_decline_notice(&model) {
        eprintln!("{notice}");
        return None;
    }

    let info = AprCudaModelInfo {
        arch: model.config.architecture.clone(),
        num_layers: model.config.num_layers,
        vocab_size: model.config.vocab_size,
        hidden_dim: model.config.hidden_dim,
    };

    // FALSIFY-CPU-GPU-003: CUDA init failure (e.g. parity_gate cosine < 0.99 on
    // a broken GPU build, or ILLEGAL_ADDRESS during the gate's GPU forward) MUST
    // be visible without --verbose. Silent fallback was the SHIP-007 jidoka gap:
    // user saw downstream wgpu gibberish without ever knowing CUDA was rejected.
    let cuda_model = OwnedQuantizedModelCuda::with_max_seq_len(model, 0, 2048).map_err(|e| {
        eprintln!("{}, attempting fallback: {}", CUDA_FALLBACK_LOG_PREFIX, e);
    }).ok()?;

    Some((cuda_model, info))
}

#[cfg(feature = "cuda")]
fn log_apr_cuda_info(
    info: &AprCudaModelInfo,
    cuda_model: &crate::gguf::OwnedQuantizedModelCuda,
    load_ms: f64,
) {
    eprintln!(
        "Architecture: {} ({} layers, vocab_size={})",
        info.arch, info.num_layers, info.vocab_size
    );
    // #4006: the loaded weights' qtypes, not a backend name in the quant field.
    let m = cuda_model.model();
    eprintln!(
        "Config: hidden_size={}, quant={}, backend=CUDA+KVCache, threads=1 (GPU)",
        info.hidden_dim,
        body_quant_label(&model_body_qtypes(m), m.lm_head_weight.qtype)
    );
    eprintln!("Model loaded in {:.1}ms", load_ms);
    eprintln!(
        "Backend: GPU ({}, {} MB VRAM)",
        cuda_model.device_name(),
        cuda_model.vram_mb()
    );
}

/// Try APR CUDA inference, returning None to fall through to CPU.
///
/// Converts APR Q4K model to `OwnedQuantizedModel` and uses the proven GGUF CUDA
/// pipeline (same path as `try_gguf_gpu_generate`). The previous wgpu path used
/// `AprF32ToGpuAdapter` which only reads F32 fields — empty for Q4K models → garbage.
#[cfg(feature = "cuda")]
fn try_apr_cuda_inference(
    config: &InferenceConfig,
    input_tokens: &[u32],
    input_token_count: usize,
    load_start: Instant,
) -> Option<Result<InferenceResult>> {
    use crate::gguf::QuantizedGenerateConfig;

    let (mut cuda_model, info) = load_apr_cuda_model(&config.model_path, config.verbose)?;

    let load_ms = load_start.elapsed().as_secs_f64() * 1000.0;

    if config.verbose {
        log_apr_cuda_info(&info, &cuda_model, load_ms);
    }
    eprintln!("[GH-480-TRACE] try_apr_cuda_inference: model loaded OK, about to resolve stop tokens");

    // GH-373: EOS from model config + caller stop tokens + sibling tokenizer
    let stop_tokens = resolve_apr_stop_tokens(
        cuda_model.model().config.eos_token_id,
        &config.stop_tokens,
        &config.model_path,
    );
    let mut gen_config = QuantizedGenerateConfig {
        max_tokens: config.max_tokens,
        stop_tokens,
        trace: config.trace,
        ..Default::default()
    };
    // PMAT-823: APR CUDA path — forward all sampling params (was hardcoded
    // greedy). The F2 validation probe below clones this then overrides
    // temperature/top_k/max_tokens for a greedy argmax probe, so the probe is
    // unaffected.
    config.apply_sampling_to(&mut gen_config);

    eprintln!("[GH-480] F2 validation starting...");
    // #3973: "PASSED" is printed ONLY for a comparison that ran and agreed, and it
    // now carries the measured cosine. It used to print for every non-failure,
    // including the two branches that compared nothing.
    let f2 = validate_gpu_first_token(&mut cuda_model, &gen_config, input_tokens);
    eprintln!("{}", f2_status_line(&f2));
    if f2 == F2Outcome::Mismatch {
        return None;
    }

    let infer_start = Instant::now();

    let tokens = match cuda_model.generate_gpu_resident(input_tokens, &gen_config) {
        Ok(t) => t,
        Err(e) => {
            let msg = e.to_string();
            eprintln!("[GH-480] generate_gpu_resident FAILED: {msg}");
            // GH-278: Fall back to CPU for unsupported architectures (GPT-2 has no SwiGLU/RMSNorm)
            if msg.contains("not supported") || msg.contains("architecture") {
                if config.verbose {
                    eprintln!("[APR-CUDA] GPU-resident not supported, falling back to CPU: {msg}");
                }
                return None;
            }
            return Some(Err(RealizarError::InferenceError(format!(
                "GPU generation failed: {}",
                e
            ))));
        },
    };

    let inference_ms = infer_start.elapsed().as_secs_f64() * 1000.0;
    let generated_tokens = &tokens[input_token_count..];
    let text = decode_apr_tokens(&config.model_path, generated_tokens);
    let generated_token_count = generated_tokens.len();

    Some(Ok(InferenceResult {
        text,
        tokens,
        input_token_count,
        generated_token_count,
        inference_ms,
        tok_per_sec: tok_per_sec(generated_token_count, inference_ms),
        generation_ms: Some(inference_ms), // #3981: this path starts its clock AFTER setup, right before generation
        load_ms,
        format: "APR".to_string(),
        used_gpu: true,
        gpu_attempted: true,
    }))
}

/// Run APR inference on CPU.
///
/// GH-479: Delegates unconditionally to `run_apr_quantized_cpu_inference`,
/// which uses `OwnedQuantizedModel` with per-tensor scratch dequant (GH-478).
/// The previous F32 `AprTransformer` path required eager dequant of the entire
/// model (peak memory ≈ file_size × 8) and has been removed.
fn run_apr_cpu_inference(
    config: &InferenceConfig,
    input_tokens: &[u32],
    input_token_count: usize,
    load_start: Instant,
) -> Result<InferenceResult> {
    run_apr_quantized_cpu_inference(config, input_tokens, input_token_count, load_start)
}

/// GH-278: CPU inference for APR models using OwnedQuantizedModel
///
/// Used for architectures not supported by AprTransformer (GPT-2, etc.).
/// AprTransformer only supports LLaMA-style (RoPE + SwiGLU).
fn run_apr_quantized_cpu_inference(
    config: &InferenceConfig,
    input_tokens: &[u32],
    input_token_count: usize,
    load_start: Instant,
) -> Result<InferenceResult> {
    use crate::apr::MappedAprModel;
    use crate::gguf::{OwnedQuantizedModel, QuantizedGenerateConfig};

    let mapped = MappedAprModel::from_path(&config.model_path)?;
    let model = OwnedQuantizedModel::from_apr(&mapped)?;
    let load_ms = load_start.elapsed().as_secs_f64() * 1000.0;

    if config.verbose {
        eprintln!(
            "Architecture: {} ({} layers, vocab_size={})",
            model.config.architecture, model.config.num_layers, model.config.vocab_size
        );
        eprintln!(
            "Config: hidden_size={}, quant={} (OwnedQuantizedModel CPU), threads={}",
            model.config.hidden_dim,
            // #4006: a BF16 .apr printed Q4_K here; name what loaded.
            body_quant_label(&model_body_qtypes(&model), model.lm_head_weight.qtype),
            rayon::current_num_threads()
        );
        eprintln!("Model loaded in {:.1}ms", load_ms);
        eprintln!("Backend: CPU (OwnedQuantizedModel fallback for non-LLaMA arch)");
    }

    // GH-373: Resolve stop tokens for quantized path
    let stop_tokens = resolve_apr_stop_tokens(
        model.config.eos_token_id,
        &config.stop_tokens,
        &config.model_path,
    );

    let mut gen_config = QuantizedGenerateConfig {
        max_tokens: config.max_tokens,
        stop_tokens,
        trace: config.trace,
        ..Default::default()
    };
    // PMAT-823: APR quantized CPU path — forward all sampling params (was only
    // temperature+top_k; top_p/seed/repeat_penalty/repeat_last_n were dropped).
    config.apply_sampling_to(&mut gen_config);

    let infer_start = Instant::now();
    let tokens = model.generate_with_cache(input_tokens, &gen_config)?;
    let inference_ms = infer_start.elapsed().as_secs_f64() * 1000.0;
    let generated_tokens = &tokens[input_token_count..];
    let text = decode_apr_tokens(&config.model_path, generated_tokens);
    let generated_token_count = generated_tokens.len();

    Ok(InferenceResult {
        text,
        tokens,
        input_token_count,
        generated_token_count,
        inference_ms,
        tok_per_sec: tok_per_sec(generated_token_count, inference_ms),
        generation_ms: Some(inference_ms), // #3981: this path starts its clock AFTER setup, right before generation
        load_ms,
        format: "APR".to_string(),
        used_gpu: false,
        gpu_attempted: false,
    })
}

/// GH-373: Resolve stop tokens from model config, caller, and sibling tokenizer.
///
/// Merges EOS tokens from three sources:
/// 1. Model config (`eos_token_id` from APR/GGUF metadata)
/// 2. Caller-provided stop tokens (`InferenceConfig.stop_tokens`)
/// 3. Sibling tokenizer (ChatML markers like `<|im_end|>`, `<|endoftext|>`)
fn resolve_apr_stop_tokens(
    model_eos: Option<u32>,
    caller_stop_tokens: &[u32],
    model_path: &std::path::Path,
) -> Vec<u32> {
    let mut tokens: Vec<u32> = model_eos.into_iter().collect();

    // Caller-provided stop tokens
    for &t in caller_stop_tokens {
        if !tokens.contains(&t) {
            tokens.push(t);
        }
    }

    // Sibling tokenizer fallback (GH-373)
    if tokens.is_empty() {
        tokens = resolve_stop_tokens_from_tokenizer(model_path);
    }

    tokens
}

/// Load stop tokens from sibling tokenizer.json (GH-373 helper)
fn resolve_stop_tokens_from_tokenizer(model_path: &std::path::Path) -> Vec<u32> {
    let tokenizer = match crate::apr::AprV2Model::load_tokenizer(model_path) {
        Some(t) => t,
        None => return Vec::new(),
    };

    let mut tokens: Vec<u32> = tokenizer.eos_id.into_iter().collect();

    // ChatML stop tokens for instruct models
    for marker in &["<|im_end|>", "<|endoftext|>"] {
        let id = tokenizer
            .special_tokens
            .get(*marker)
            .or_else(|| tokenizer.token_to_id.get(*marker));
        if let Some(&id) = id {
            if !tokens.contains(&id) {
                tokens.push(id);
            }
        }
    }

    tokens
}

/// Decode APR output tokens using available tokenizer (GH-156)
fn decode_apr_tokens(model_path: &std::path::Path, tokens: &[u32]) -> String {
    use crate::apr::AprV2Model;

    let text = if let Some(tokenizer) = AprV2Model::load_tokenizer(model_path) {
        tokenizer.decode(tokens)
    } else if let Some(tokenizer) = find_fallback_tokenizer(model_path) {
        tokenizer.decode(tokens)
    } else {
        format!("[{} tokens generated, tokenizer not found]", tokens.len())
    };
    clean_model_output(&text)
}

/// Compute tokens per second from count and elapsed milliseconds
fn tok_per_sec(count: usize, ms: f64) -> f64 {
    if ms > 0.0 {
        count as f64 / (ms / 1000.0)
    } else {
        0.0
    }
}

/// Run SafeTensors model inference (PAR-301, PMAT-129)
///
/// PMAT-236: Accepts `PreparedTokens` (compile-time enforced chat template).
/// Previously, this function raw-encoded prompts WITHOUT chat template,
/// producing garbage output for instruct models.
/// Printed when a sampled request skips the greedy-only SafeTensors CUDA decoder (#3760).
pub const SAFETENSORS_CUDA_SAMPLING_NOTICE: &str = "[safetensors: the CUDA decoder is greedy-only; \
     sampling (--temperature > 0 with --top-k != 1) runs on the CPU (#3760)]";

fn run_safetensors_inference(
    config: &InferenceConfig,
    prepared: &PreparedTokens,
) -> Result<InferenceResult> {
    if config.verbose {
        eprintln!("Loading SafeTensors model: {}", config.model_path.display());
    }

    // PMAT-236: Use PreparedTokens (chat template already applied by prepare_tokens)
    let input_tokens = prepared.tokens().to_vec();
    let input_token_count = prepared.input_count();

    // PMAT-129: Try GPU path first.
    //
    // #3760: `SafeTensorsCudaModel::generate(input, max_tokens, eos_id)` is greedy-only;
    // it takes no sampling parameters. A sampled request used to go there and silently
    // decode greedily. It now runs on the CPU loop, which draws, and says so.
    #[cfg(feature = "cuda")]
    if !config.no_gpu {
        if crate::sampling::is_greedy(config.temperature, config.top_k) {
            if let Some(result) =
                try_safetensors_cuda_inference(config, &input_tokens, input_token_count)
            {
                return result;
            }
        } else {
            eprintln!("{SAFETENSORS_CUDA_SAMPLING_NOTICE}");
        }
    }

    // CPU fallback: SafeTensors → AprTransformer conversion
    run_safetensors_cpu_inference(config, &input_tokens, input_token_count)
}

/// Try SafeTensors CUDA inference, returning None to fall through to CPU
#[cfg(feature = "cuda")]
fn try_safetensors_cuda_inference(
    config: &InferenceConfig,
    input_tokens: &[u32],
    input_token_count: usize,
) -> Option<Result<InferenceResult>> {
    use crate::safetensors_cuda::SafeTensorsCudaModel;

    let load_start = Instant::now();
    let mut cuda_model = match SafeTensorsCudaModel::load(&config.model_path, 0) {
        Ok(m) => m,
        Err(e) => {
            if config.verbose {
                eprintln!("Backend: CPU (GPU init failed: {})", e);
            }
            return None;
        },
    };

    let load_ms = load_start.elapsed().as_secs_f64() * 1000.0;

    if config.verbose {
        eprintln!(
            "Architecture: SafeTensors ({} layers, vocab_size={})",
            cuda_model.config().num_layers,
            cuda_model.config().vocab_size
        );
        eprintln!(
            "Config: hidden_size={}, context_length={}, quant={}, threads=1 (GPU)",
            cuda_model.config().hidden_dim,
            cuda_model.config().context_length,
            // #4006: read from the header, not an either/or guess.
            safetensors_quant_label(&config.model_path)
        );
        eprintln!("Model loaded in {:.1}ms", load_ms);
        eprintln!(
            "Backend: GPU ({}, {} MB VRAM)",
            cuda_model.device_name(),
            cuda_model.vram_mb()
        );
    }

    let infer_start = Instant::now();
    // GH-330: EOS from model config (Design by Contract)
    let eos_id = cuda_model.config().eos_token_id.unwrap_or(0);
    let tokens = match cuda_model.generate(input_tokens, config.max_tokens, eos_id) {
        Ok(t) => t,
        Err(e) => {
            return Some(Err(RealizarError::InferenceError(format!(
                "GPU generation failed: {}",
                e
            ))))
        },
    };

    let inference_ms = infer_start.elapsed().as_secs_f64() * 1000.0;
    let generated_tokens = &tokens[input_token_count..];
    let text = decode_apr_tokens(&config.model_path, generated_tokens);
    let generated_token_count = generated_tokens.len();

    Some(Ok(InferenceResult {
        text,
        tokens,
        input_token_count,
        generated_token_count,
        inference_ms,
        tok_per_sec: tok_per_sec(generated_token_count, inference_ms),
        generation_ms: Some(inference_ms), // #3981: this path starts its clock AFTER setup, right before generation
        load_ms,
        format: "SafeTensors".to_string(),
        used_gpu: true,
        gpu_attempted: true,
    }))
}

#[cfg(test)]
mod tests {
    use super::{wgpu_greedy_decode, CUDA_FALLBACK_LOG_PREFIX, WGPU_FALLBACK_LOG_PREFIX};

    /// #4264: a fake KV-cached forward that records every (token, position) it
    /// is fed and returns the position as the "hidden state".
    fn decode_trace(prompt: &[u32], max_tokens: usize, stop: &[u32]) -> (Vec<u32>, Vec<(u32, usize)>) {
        let mut fed = Vec::new();
        let out = wgpu_greedy_decode::<()>(
            prompt,
            max_tokens,
            stop,
            |t, p| {
                fed.push((t, p));
                Ok(vec![p as f32])
            },
            |h| 100 + h[0] as u32,
        )
        .expect("the fake forward never fails");
        (out, fed)
    }

    #[test]
    fn wgpu_decode_prefills_every_prompt_position_before_decoding() {
        let (out, fed) = decode_trace(&[7, 8, 9], 2, &[]);
        assert_eq!(
            fed,
            [(7, 0), (8, 1), (9, 2), (102, 3)],
            "each prompt token enters the cache at its own position, then decode continues"
        );
        assert_eq!(out, [7, 8, 9, 102, 103]);
    }

    #[test]
    fn wgpu_decode_stops_and_handles_edges() {
        let (out, fed) = decode_trace(&[5], 4, &[101]);
        assert_eq!(out, [5, 100, 101], "the stop token is kept, then decode ends");
        assert_eq!(fed, [(5, 0), (100, 1)]);
        assert_eq!(decode_trace(&[], 4, &[]).0, Vec::<u32>::new(), "empty prompt");
        assert!(decode_trace(&[1, 2], 0, &[]).1.is_empty(), "max_tokens 0 runs no forward");
        let err = wgpu_greedy_decode(&[1, 2], 1, &[], |_, p| if p == 0 { Err("boom") } else { Ok(vec![]) }, |_| 0);
        assert_eq!(err, Err("boom"), "a prefill failure is returned, not skipped");
    }

    /// A message the user can act on names the backend and says something went
    /// wrong. A contract filename does neither.
    fn assert_reads_as_a_warning_to_a_human(prefix: &str, backend: &str) {
        assert!(
            !prefix.contains("apr-cpu-vs-gpu-output-parity-v1"),
            "the fallback message addresses the user in a contract ID: {prefix}"
        );
        assert!(
            !prefix.starts_with('['),
            "the fallback message opens with a bracketed internal tag: {prefix}"
        );
        assert!(
            prefix.starts_with("warning:"),
            "the fallback message must announce itself as a warning: {prefix}"
        );
        assert!(
            prefix.contains(backend),
            "fallback message must say which backend was rejected; got: {prefix}"
        );
        assert!(
            prefix.ends_with("path rejected"),
            "fallback message must say the path was rejected; got: {prefix}"
        );
    }

    /// FALSIFY-CPU-GPU-003 (PR #1428) established that the CUDA fallback
    /// decision must reach the user WITHOUT `--verbose`, so a broken GPU can
    /// never ship silent gibberish. #1429 pinned that by asserting the tag was
    /// the contract ID `[apr-cpu-vs-gpu-output-parity-v1]` — which pinned the
    /// wrong half. Visibility is the property worth locking in; the contract
    /// filename was never something the user could act on, and the dogfood
    /// audit (#2405) recorded it leaking into ordinary `apr run` output.
    ///
    /// This test now asserts what #1428 actually bought: an unconditional,
    /// human-readable warning that names the rejected backend.
    #[test]
    fn cuda_fallback_log_prefix_warns_the_user_in_prose() {
        assert_reads_as_a_warning_to_a_human(CUDA_FALLBACK_LOG_PREFIX, "CUDA");
    }

    /// Same property for FALSIFY-CPU-GPU-005: when CUDA falls through to wgpu
    /// and wgpu is itself rejected, the user is told, in English, without
    /// `--verbose`. Reverting to verbose-only, or to a bare contract ID, is the
    /// regression this guards.
    #[test]
    fn wgpu_fallback_log_prefix_warns_the_user_in_prose() {
        assert_reads_as_a_warning_to_a_human(WGPU_FALLBACK_LOG_PREFIX, "wgpu");
    }

    /// Symmetry guard: both hops of the fallback chain must read the same way,
    /// so a user watching CUDA → wgpu → CPU sees one consistent shape rather
    /// than two dialects. Locks in the symmetry PR #1428 (CUDA) and #1442
    /// (wgpu) established, minus the ticket number.
    #[test]
    fn cuda_and_wgpu_fallback_log_prefixes_share_their_shape() {
        for prefix in [CUDA_FALLBACK_LOG_PREFIX, WGPU_FALLBACK_LOG_PREFIX] {
            assert!(prefix.starts_with("warning: GPU ("), "{prefix}");
            assert!(prefix.ends_with(") path rejected"), "{prefix}");
        }
        assert_ne!(
            CUDA_FALLBACK_LOG_PREFIX, WGPU_FALLBACK_LOG_PREFIX,
            "the two hops must remain distinguishable"
        );
    }

    /// FALSIFY-CPU-GPU-005 part b cosine helper — parallel vectors return 1.
    ///
    /// Locks in the gate's positive case: when wgpu produces logits identical
    /// to CPU, the gate must NOT trigger fallback (cosine = 1.0 ≥ 0.99 floor).
    #[test]
    fn cpu_vs_gpu_cosine_similarity_parallel_returns_one() {
        let a = vec![1.0_f32, 2.0, 3.0, 4.0];
        let b = a.clone();
        let cos = super::cpu_vs_gpu_cosine_similarity(&a, &b);
        assert!(
            (cos - 1.0).abs() < 1e-6,
            "parallel vectors must yield cosine 1.0, got {cos}"
        );
    }

    /// FALSIFY-CPU-GPU-005 part b cosine helper — orthogonal returns 0.
    ///
    /// Negative case: orthogonal vectors must yield cosine 0.0 which is well
    /// below the 0.99 gate floor → fallback triggers.
    #[test]
    fn cpu_vs_gpu_cosine_similarity_orthogonal_returns_zero() {
        let a = vec![1.0_f32, 0.0, 0.0, 0.0];
        let b = vec![0.0_f32, 1.0, 0.0, 0.0];
        let cos = super::cpu_vs_gpu_cosine_similarity(&a, &b);
        assert!(
            cos.abs() < 1e-6,
            "orthogonal vectors must yield cosine 0.0, got {cos}"
        );
    }

    /// FALSIFY-CPU-GPU-005 part b cosine helper — fail-closed on bad input.
    ///
    /// Zero-norm or mismatched-length inputs MUST return 0.0 so the future gate
    /// triggers fallback rather than dividing by zero or panicking. This is the
    /// "conservative default" that closes the silent-gibberish loophole even
    /// when the probe forward itself emits NaN/zeros.
    #[test]
    fn cpu_vs_gpu_cosine_similarity_fails_closed() {
        // Zero-norm input
        let zero = vec![0.0_f32; 4];
        let nonzero = vec![1.0_f32, 2.0, 3.0, 4.0];
        assert_eq!(
            super::cpu_vs_gpu_cosine_similarity(&zero, &nonzero),
            0.0,
            "zero-norm input must fail closed"
        );
        // Length mismatch
        let short = vec![1.0_f32, 2.0];
        let long = vec![1.0_f32, 2.0, 3.0, 4.0];
        assert_eq!(
            super::cpu_vs_gpu_cosine_similarity(&short, &long),
            0.0,
            "length mismatch must fail closed"
        );
        // Empty input
        let empty: Vec<f32> = Vec::new();
        assert_eq!(
            super::cpu_vs_gpu_cosine_similarity(&empty, &empty),
            0.0,
            "empty input must fail closed"
        );
    }
}

/// #3757: the case table for `wgpu_fallback_allowed`.
///
/// The row that shipped broken is `bare_apr_run_does_not_attempt_wgpu`. Deleting
/// the `accel_forced` conjunct — the state `release/0.69.1-batch-2` @ `9f8836c71`
/// was in — turns it RED.
#[cfg(all(test, feature = "gpu"))]
mod pmat3757_wgpu_attempt_gate {
    use super::wgpu_fallback_allowed;

    /// `(no_gpu, accel_forced, has_legacy_quant, allowed, what this invocation is)`
    const CASES: &[(bool, bool, bool, bool, &str)] = &[
        (
            false, false, false, false,
            "bare `apr run model.gguf` — the #3757 defect: on a default \
             (non-cuda) install this dequantized 1726.8 MB to F32, failed wgpu's \
             cpu-parity gate at cosine 0.9554 and fell back to CPU, costing \
             7607 ms against --no-gpu's 3035 ms for the identical answer",
        ),
        (
            false, true, false, true,
            "`--backend wgpu` — an explicit request is still served, and still \
             refuses to report a fallback as success (rc=14)",
        ),
        (
            true, true, false, false,
            "`--no-gpu --backend wgpu` — an explicit opt-out wins over an \
             explicit request",
        ),
        (
            true, false, false, false,
            "`--no-gpu` — nothing to attempt",
        ),
        (
            false, true, true, false,
            "a legacy-quant model with `--gpu`: no GPU kernel exists for it, so \
             the attempt would dequantize and fail",
        ),
    ];

    #[test]
    fn bare_apr_run_does_not_attempt_wgpu() {
        let (no_gpu, accel_forced, legacy, _, why) = CASES[0];
        assert!(
            !wgpu_fallback_allowed(no_gpu, accel_forced, legacy),
            "#3757 REGRESSION: {why}"
        );
    }

    #[test]
    fn an_explicit_accelerator_request_is_still_served() {
        let (no_gpu, accel_forced, legacy, _, why) = CASES[1];
        assert!(
            wgpu_fallback_allowed(no_gpu, accel_forced, legacy),
            "#3757 OVER-CORRECTION: the fix removed the backend instead of \
             making it opt-in. {why}"
        );
    }

    /// Every row at once, so a regression names each invocation it broke rather
    /// than stopping at the first.
    #[test]
    fn the_whole_attempt_table_holds() {
        let wrong: Vec<String> = CASES
            .iter()
            .filter(|(n, a, l, want, _)| wgpu_fallback_allowed(*n, *a, *l) != *want)
            .map(|(n, a, l, want, why)| {
                format!(
                    "\n  - no_gpu={n} accel_forced={a} has_legacy_quant={l}: \
                     expected allowed={want}, got {}. {why}",
                    !*want
                )
            })
            .collect();
        assert!(
            wrong.is_empty(),
            "{} of {} wgpu-attempt cases are wrong:{}",
            wrong.len(),
            CASES.len(),
            wrong.join("")
        );
    }

    // ── #3908: a CUDA decline on the quant whitelist must ANNOUNCE itself ──────────

    /// The notice exists and NAMES THE TYPE, both directions.
    ///
    /// A verdict test alone would not have caught the defect this fixes: the old code's
    /// verdict was already correct — it declined bf16, rightly. What was missing was that
    /// it said so. Hence the emission test below as well.
    #[cfg(feature = "cuda")]
    #[test]
    fn apr_cuda_decline_names_the_type_and_is_silent_when_eligible() {
        // Built the same way the PMAT-785 whitelist tests build theirs.
        let cfg = crate::gguf::GGUFConfig {
            architecture: "test".to_string(),
            constraints: crate::gguf::ArchConstraints::from_architecture("test"),
            hidden_dim: 64,
            intermediate_dim: 128,
            num_layers: 1,
            num_heads: 4,
            num_kv_heads: 4,
            vocab_size: 100,
            context_length: 256,
            rope_theta: 10000.0,
            eps: 1e-5,
            rope_type: 0,
            explicit_head_dim: None,
            query_pre_attn_scalar: None,
            bos_token_id: None,
            eos_token_id: None,
        };
        let eligible = crate::gguf::test_helpers::create_test_model_with_config(&cfg);

        // Was 30 (BF16), the type that produced #3908 - until #3908 gave BF16 a
        // measured kernel and admitted it. IQ1_M(29) still has none.
        let mut bad = crate::gguf::test_helpers::create_test_model_with_config(&cfg);
        bad.lm_head_weight.qtype = 29;
        let notice = super::apr_cuda_decline_notice(&bad)
            .expect("a model whose lm_head has no verified GPU kernel owes the user a notice");
        assert!(notice.contains("29"), "the notice must NAME the declining type: {notice}");
        assert!(
            notice.starts_with(super::CUDA_FALLBACK_LOG_PREFIX),
            "the notice must announce which backend was rejected: {notice}"
        );
        assert!(
            notice.contains("downstream of THIS decision"),
            "the notice must say later backend errors are downstream, since misattributing \
             them to wgpu is the defect it exists to prevent: {notice}"
        );

        assert!(
            super::apr_cuda_decline_notice(&eligible).is_none(),
            "a fully GPU-eligible model must produce no decline notice"
        );
    }

    /// MUST-RED: the notice is actually EMITTED at the exit, unconditionally.
    ///
    /// Read from source because the alternative is capturing stderr from a path that needs
    /// a GPU. Delete the `eprintln!` at that exit and this goes red — which is the point:
    /// the previous code's verdict was right and its silence was the bug, so a test that
    /// only checks the verdict would have passed on the broken version.
    #[cfg(feature = "cuda")]
    #[test]
    fn the_decline_notice_is_emitted_unconditionally_not_behind_verbose() {
        let src = std::fs::read_to_string(concat!(
            env!("CARGO_MANIFEST_DIR"),
            "/src/infer/gguf_gpu_generate.rs"
        ))
        .expect("own source readable");
        let at = src
            .find("if let Some(notice) = apr_cuda_decline_notice(&model)")
            .expect("the quant-whitelist exit must consult apr_cuda_decline_notice");
        let tail = &src[at..at + 220];
        assert!(
            tail.contains("eprintln!(\"{notice}\")"),
            "the quant-whitelist exit must PRINT the notice before returning None — a silent \
             decline here is #3908, and the sibling exit's own comment says a CUDA rejection \
             MUST be visible without --verbose"
        );
        assert!(
            !tail.contains("if verbose"),
            "the decline notice must not be behind --verbose: the user who needs it is the \
             one who did not pass it"
        );
    }
}