coremlit 0.1.1

Safe, synchronous CoreML runtime for macOS (CPU/GPU/Neural Engine) with opt-in on-device multimodal pipelines: speech (Whisper STT, forced alignment, speaker diarization, Silero VAD), AudioSet sound-event tagging, and audio/text/image embeddings (CLAP, granite, SigLIP)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
1451
1452
1453
1454
1455
1456
1457
1458
1459
1460
1461
1462
1463
1464
1465
1466
1467
1468
1469
1470
1471
1472
1473
1474
1475
1476
1477
1478
1479
1480
1481
1482
1483
1484
1485
1486
1487
1488
1489
1490
1491
1492
1493
1494
1495
1496
1497
1498
1499
1500
1501
1502
1503
1504
1505
1506
1507
1508
1509
1510
1511
1512
1513
1514
1515
1516
1517
1518
1519
1520
1521
1522
1523
1524
1525
1526
1527
1528
1529
1530
1531
1532
1533
1534
1535
1536
1537
1538
1539
1540
1541
1542
1543
1544
1545
1546
1547
1548
1549
1550
1551
1552
1553
1554
1555
1556
1557
1558
1559
1560
1561
1562
1563
1564
1565
1566
1567
1568
1569
1570
1571
1572
1573
1574
1575
1576
1577
1578
1579
1580
1581
1582
1583
1584
1585
1586
1587
1588
1589
1590
1591
1592
1593
1594
1595
1596
1597
1598
1599
1600
1601
1602
1603
1604
1605
1606
1607
1608
1609
1610
1611
1612
1613
1614
1615
1616
1617
1618
1619
1620
1621
1622
1623
1624
1625
1626
1627
1628
1629
1630
1631
1632
1633
1634
1635
1636
1637
1638
1639
1640
1641
1642
1643
1644
1645
1646
1647
1648
1649
1650
1651
1652
1653
1654
1655
1656
1657
1658
1659
1660
1661
1662
1663
1664
1665
1666
1667
1668
1669
1670
1671
1672
1673
1674
1675
1676
1677
1678
1679
1680
1681
1682
1683
1684
1685
1686
1687
1688
1689
1690
1691
1692
1693
1694
1695
1696
1697
1698
1699
1700
1701
1702
1703
1704
1705
1706
1707
1708
1709
1710
1711
1712
1713
1714
1715
1716
1717
1718
1719
1720
1721
1722
1723
1724
1725
1726
1727
1728
1729
1730
1731
1732
1733
1734
1735
1736
1737
1738
1739
1740
1741
1742
1743
1744
1745
1746
1747
1748
1749
1750
1751
1752
1753
1754
1755
1756
1757
1758
1759
1760
1761
1762
1763
1764
1765
1766
1767
1768
1769
1770
1771
1772
1773
1774
1775
1776
1777
1778
1779
1780
1781
1782
1783
1784
1785
1786
1787
1788
1789
1790
1791
1792
1793
1794
1795
1796
1797
1798
1799
1800
1801
1802
1803
1804
1805
1806
1807
1808
1809
1810
1811
1812
1813
1814
1815
1816
1817
1818
1819
1820
1821
1822
1823
1824
1825
1826
1827
1828
1829
1830
1831
1832
1833
1834
1835
1836
1837
1838
1839
1840
1841
1842
1843
1844
1845
1846
1847
1848
1849
1850
1851
1852
1853
1854
1855
1856
1857
1858
1859
1860
1861
1862
1863
1864
1865
1866
1867
1868
1869
1870
1871
1872
1873
1874
1875
1876
1877
1878
1879
1880
1881
1882
1883
1884
1885
1886
1887
1888
1889
1890
1891
1892
1893
1894
1895
1896
1897
1898
1899
1900
1901
1902
1903
1904
1905
1906
1907
1908
1909
1910
1911
1912
1913
1914
1915
1916
1917
1918
1919
1920
1921
1922
1923
1924
1925
1926
1927
1928
1929
1930
1931
1932
1933
1934
1935
1936
1937
1938
1939
1940
1941
1942
1943
1944
1945
1946
1947
1948
1949
1950
1951
1952
1953
1954
1955
1956
1957
1958
1959
1960
1961
1962
1963
1964
1965
1966
1967
1968
1969
1970
1971
1972
1973
1974
1975
1976
1977
1978
1979
1980
1981
1982
1983
1984
1985
1986
1987
1988
1989
1990
1991
1992
1993
1994
1995
1996
1997
1998
1999
2000
2001
2002
2003
2004
2005
2006
2007
2008
2009
2010
2011
2012
2013
2014
2015
//! The CoreML face embedder: aligned template faces in, L2-normalised
//! embeddings out.
//!
//! # The embedder is pure, and preprocessing is DATA
//!
//! Channel order, scale, bias and tensor layout live in [`FaceModel`] — the
//! per-artifact manifest — and never in a constant at a call site. Issue #115's
//! census is the reason: six ArcFace-family artifacts use **four different
//! divisors, two channel orders and one per-channel mean**, and two "official"
//! releases of one model disagree about RGB vs BGR with no warning on either.
//! A wrong divisor costs `1 − cos ≈ 0.083` and a wrong channel order `0.151`,
//! against an ANE fp16 noise floor of `0.0015`. None of it raises an error;
//! all of it is silent degradation. Making it a value the caller supplies
//! alongside the weights is what keeps a second model from becoming a second
//! code path.
//!
//! | model | order | normalisation |
//! |---|---|---|
//! | `w600k_r50`, `glintr100` | RGB | `(x − 127.5) / 127.5` |
//! | ONNX-zoo ArcFace, OpenCV SFace | RGB | `(x − 128) / 128` (often fused) |
//! | AdaFace | **BGR** | `(x − 127.5) / 127.5` |
//! | FaceNet | RGB | `(x − 127.5) / 128` |
//! | dlib | RGB | `(x − [122.782, 117.001, 104.298]) / 256` |
//!
//! Every row of that table is expressible as [`Preprocessing`]'s `scale` plus a
//! per-channel `bias`.
//!
//! # Batch is the unit
//!
//! [`FaceEmbedder::embed`] takes a slice: a keyframe with N faces is ONE call,
//! whatever the graph's own batch dimension turns out to be. The capacity is
//! read off the loaded model's input feature — after the load contract below
//! has established that the feature admits exactly one shape, so it is the
//! graph's ONLY batch and not the default a flexible one would also report
//! ([`FaceEmbedder::batch_capacity`]) — and the slice is chunked to it, so a
//! batch-1 export and a batch-8 export are the same call site.
//!
//! That capacity is the ARTIFACT's number and nothing bounds it, so the buffers
//! it sizes are the one place this door's arithmetic is over a value it did not
//! choose. Both per-prediction element counts are `checked_mul`'d at load and
//! carried (`TensorElements`).
//!
//! # EVERY allocation on this path is fallible
//!
//! Not every allocation sized by the artifact, and not every allocation sized
//! by the manifest — every allocation between [`FaceEmbedder::embed`] and the
//! embeddings it returns, whatever its size came from. Four reservations cover
//! it, each `Vec::try_reserve_exact`: the result vector through
//! `result_buffer`, the two per-prediction tensors through `zeroed_tensor`,
//! the per-row embedding through `embedding_buffer`, and the de-aliasing
//! gather `Features::from_provider` may run through `MultiArray::deep_copy`.
//!
//! There is no list of sites this class excludes, and the absence is the
//! point. A wrap and an abort are the two ways an accepted model ends the
//! caller's process instead of returning an error, and a fallible reservation
//! on SOME of the buffers fixes neither — the abort simply happens at
//! whichever site is still infallible, one step later. Twice a site was left
//! out under an argument about where its size came from ("the caller supplied
//! that length", "it is a cut of a buffer that already succeeded"); each such
//! argument has to be made afresh per site, which is how one survives a sweep.
//! The rule is the whole path instead. Between `embed` and its result, a
//! length known only at run time is reserved fallibly, and a length fixed at
//! compile time is not — the rank-shaped `Vec`s `input_shape`, `input_dims`
//! and `output_dims` build are three or four elements long by their own
//! construction, so nothing about an artifact can move them. Off that path the
//! rule does not reach, and each site there says so where it sits rather than
//! in a list something has to keep up to date: `feature_names` is the only
//! run-time length in this module that is not on the path, and the digest
//! walk's are all bounded by constants of its own.
//!
//! # The load contract is a value, and a type proves it was checked
//!
//! [`FaceEmbedder`] holds a `Checked` model, never a bare [`Model`]: the only
//! constructor of that wrapper takes this door's `LoadContract` and runs it,
//! so removing the load check is a compile error rather than a mutation that
//! survives every test. The contract is BUILT at load rather than written down
//! as a constant, because two of its numbers are not this module's — the
//! embedding width is the caller's [`FaceModel::dim`], and the batch is the
//! artifact's, read back off the checked model.
//!
//! Three declarations used to load clean here and then fail, or degrade, at
//! predict time. Each is a clause now:
//!
//! - a graph declaring the manifest's input **plus another REQUIRED input**,
//!   which [`FaceEmbedder::embed`] never sends;
//! - a graph declaring an **`MLState` buffer**, which is not an input at all —
//!   it lives in its own dictionary, so a stateful graph naming exactly these
//!   two features cleared every check this door used to make;
//! - a **flexible** input whose DEFAULT shape reads `[n, 3, 112, 112]`.
//!   [`crate::FeatureInfo::shape`] reports the default of a `RangeDim` or
//!   enumerated feature rather than a bound, so its numbers are
//!   indistinguishable from a pinned graph's and the batch read off it would be
//!   a default rather than a fact.
//!
//! ## A legacy `neuralNetwork` export is refused at load, deliberately
//!
//! An earlier version of this door accepted an EMPTY declared shape on either
//! feature — the legacy `neuralnetwork` specification leaves shapes undeclared
//! — by guessing a batch-one graph and leaving the guess to be caught at
//! predict time. The guess is gone: a feature this door cannot read a rank off
//! is refused when the model is loaded, and so is one whose geometry is not
//! [`crate::ShapeConstraint::Fixed`].
//!
//! The refusal is wider than the empty shape it started from, and that is
//! worth stating rather than discovering. [`crate::ShapeConstraint`]'s measured
//! table records that **every output of a `neuralnetwork` export reports
//! `Unspecified`, even when its input is fixed** — so no artifact in that
//! format loads here, whatever it declares. Fail-closed is the choice: a shape
//! this door guesses is a shape nothing measured. If a real legacy artifact
//! ever matters it arrives as a contract variant with a measurement behind it,
//! not as an arm with a guess in it.

use std::path::Path;

use crate::{
  ComputeUnits, DataType, Model, ModelDescription, MultiArray,
  embeddings::face::{
    align::{AlignedFace, TEMPLATE_BYTES, TEMPLATE_SIZE},
    artifact::{ArtifactDigest, digest_artifact},
    error::{
      AllocationFailed, BatchRow, ContractMismatch, ElementCountOverflow, EmbeddingSpaceField,
      Error, IncomparableEmbeddings, NonFiniteOutput, NonFinitePreprocessing, OutputElementCount,
      OutputShape, PredictionTensor, PreprocessingField, PreprocessingMap, Result,
      ResultAllocationFailed, ZeroEmbeddingWidth,
    },
  },
  model::contract::{
    Checked, ContractViolation, Dim, FeatureContract, LoadContract, Rendered, StateContract,
  },
};

/// Elements one face occupies in the input tensor: `3 · 112 · 112`.
///
/// Numerically [`TEMPLATE_BYTES`] — [`write_row`] maps every byte of the RGB8
/// template to exactly one tensor element — and spelled as that constant rather
/// than re-multiplied, so the row stride [`FaceEmbedder::build_input`] slices by
/// cannot drift from the template it is slicing.
const FACE_ELEMENTS: usize = TEMPLATE_BYTES;

/// The channel order a model's input tensor expects.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
#[cfg_attr(feature = "serde", serde(rename_all = "kebab-case"))]
pub enum ChannelOrder {
  /// Red, green, blue — the order [`AlignedFace`] stores.
  Rgb,
  /// Blue, green, red — OpenCV's order, and AdaFace's original checkpoints'.
  Bgr,
}

/// The axis order a model's input tensor expects.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
#[cfg_attr(feature = "serde", serde(rename_all = "kebab-case"))]
pub enum TensorLayout {
  /// `[batch, channel, height, width]` — the PyTorch/ONNX convention every
  /// ArcFace export in the census uses.
  Nchw,
  /// `[batch, height, width, channel]` — the TensorFlow convention.
  Nhwc,
}

/// One `f32` preprocessing field as its SEMANTIC identity: the bit pattern,
/// with the representations that mean the same thing folded onto one.
///
/// Two foldings, and each closes a case where a raw `to_bits` comparison says
/// "different" about preprocessing that is not:
///
/// - **`−0.0` and `+0.0` are one value.** They are the same real number, so
///   `byte · scale + bias` is the same function either way — and the produced
///   TENSOR agrees, because [`write_row`] normalises every zero it writes to
///   `+0.0`. That parenthetical used to read the other way: the tensor *could*
///   differ in the sign of a zero (with a negative scale, byte `0` gives
///   `−0.0 + +0.0 = +0.0` against `−0.0 + −0.0 = −0.0`) and "nothing
///   downstream reads the sign of a zero" was the argument for tolerating it.
///   A graph can read it — `sign`, `copysign`, and `1/x` as `+∞` against `−∞`
///   — so one space had two tensors. The producer canonicalises now, and this
///   fold is the whole truth rather than half of it.
/// - **Every NaN is one value.** This serves [`Preprocessing`]'s own [`Eq`]
///   lawfulness and nothing else. The type is public and both its constructors
///   are `const`, so a NaN `Preprocessing` can be built, and without the fold
///   it would not equal itself. It does not serve a broken manifest reaching a
///   comparison: [`FaceEmbedder::load`] refuses a preprocessing whose map does
///   not stay in `f32` ([`Error::NonFinitePreprocessing`]) — the two fields
///   AND the map they make, at both ends of the byte range — so no stamped
///   [`EmbeddingSpace`] carries a NaN.
///
/// Both foldings are reflexive, symmetric and transitive, which is what lets
/// [`Preprocessing`] and [`EmbeddingSpace`] be [`Eq`] at all — `f32`'s own
/// `PartialEq` is not an equivalence relation.
fn canonical_bits(value: f32) -> u32 {
  if value.is_nan() {
    f32::NAN.to_bits()
  } else if value == 0.0 {
    0
  } else {
    value.to_bits()
  }
}

/// One model's host-side preprocessing: `value = byte · scale + bias[channel]`.
///
/// `scale` and `bias` are in the MODEL's channel order, so a BGR model's
/// per-channel bias is written blue-first. Both forms in the module table
/// reduce to this: a divisor `d` and a mean `m` are `scale = 1/d`,
/// `bias = −m/d`.
///
/// Equality is `canonical_bits` on the two float fields rather than `f32`'s
/// own `==` or a raw bit comparison, which is why this is [`Eq`] and [`Hash`].
/// It is the SAME relation [`EmbeddingSpace`] decides a cosine by — one type
/// must not carry two equalities that disagree.
#[derive(Debug, Clone, Copy)]
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
pub struct Preprocessing {
  order: ChannelOrder,
  layout: TensorLayout,
  scale: f32,
  bias: [f32; 3],
}

impl PartialEq for Preprocessing {
  #[inline]
  fn eq(&self, other: &Self) -> bool {
    self.order == other.order
      && self.layout == other.layout
      && canonical_bits(self.scale) == canonical_bits(other.scale)
      && self.bias.map(canonical_bits) == other.bias.map(canonical_bits)
  }
}

impl Eq for Preprocessing {}

impl core::hash::Hash for Preprocessing {
  #[inline]
  fn hash<H: core::hash::Hasher>(&self, state: &mut H) {
    core::hash::Hash::hash(&self.order, state);
    core::hash::Hash::hash(&self.layout, state);
    core::hash::Hash::hash(&canonical_bits(self.scale), state);
    core::hash::Hash::hash(&self.bias.map(canonical_bits), state);
  }
}

impl Preprocessing {
  /// The ArcFace family's own preprocessing: RGB, NCHW, `(x − 127.5) / 127.5`
  /// — the `[−1, 1]` mapping `w600k_r50` and `glintr100` are trained and
  /// exported against.
  pub const ARCFACE: Self = Self {
    order: ChannelOrder::Rgb,
    layout: TensorLayout::Nchw,
    scale: 1.0 / 127.5,
    bias: [-1.0, -1.0, -1.0],
  };

  /// Preprocessing from its four parts.
  #[inline]
  pub const fn new(order: ChannelOrder, layout: TensorLayout, scale: f32, bias: [f32; 3]) -> Self {
    Self {
      order,
      layout,
      scale,
      bias,
    }
  }

  /// Preprocessing written as the census states it: a per-channel `mean`
  /// subtracted, then a `divisor`.
  ///
  /// `mean` and `divisor` are in the model's own channel order. Equivalent to
  /// [`Self::new`] with `scale = 1/divisor` and `bias = −mean/divisor`.
  #[inline]
  pub const fn from_mean_and_divisor(
    order: ChannelOrder,
    layout: TensorLayout,
    mean: [f32; 3],
    divisor: f32,
  ) -> Self {
    Self {
      order,
      layout,
      scale: 1.0 / divisor,
      bias: [-mean[0] / divisor, -mean[1] / divisor, -mean[2] / divisor],
    }
  }

  /// The channel order the model's input tensor expects.
  #[inline]
  pub const fn order(&self) -> ChannelOrder {
    self.order
  }

  /// The axis order the model's input tensor expects.
  #[inline]
  pub const fn layout(&self) -> TensorLayout {
    self.layout
  }

  /// The multiplier applied to each raw 0–255 byte.
  #[inline]
  pub const fn scale(&self) -> f32 {
    self.scale
  }

  /// The per-channel offset added after [`Self::scale`], in the model's own
  /// channel order.
  #[inline]
  pub const fn bias(&self) -> [f32; 3] {
    self.bias
  }
}

/// The space one embedder's vectors live in: everything that decides what the
/// NUMBERS are.
///
/// # Which fields, and why each one is here
///
/// - **`artifact`** — the [`ArtifactDigest`] of the bytes
///   [`FaceEmbedder::load`] read. The trained parameters ARE most of the
///   function that produced a vector, and every other field is schema two
///   unrelated exports are free to agree on.
/// - **`output`** — the feature the tensor was read from. For a graph with two
///   `[batch, dim]` heads the output name selects *which function produced the
///   numbers*, so it is not routing.
/// - **`input`** — the feature the pixels were written to, for the same
///   reason on the other side.
/// - **`dim`** and **`preprocessing`** — the width, and the pixels-to-tensor
///   map the host applied before inference.
///
/// A previous round removed the two names as "IO routing" and stated the
/// remaining hole — "two distinct artifacts with one schema are one space" —
/// as a residual. Both halves of that were the same mistake one level apart,
/// and `artifact` closes it: **two `FaceEmbedding`s compare only if
/// byte-identical artifacts produced them, read from the same output feature,
/// fed through the same input feature, with the same host preprocessing.**
///
/// # Produced, never assembled
///
/// There is no public constructor. The only value of this type a caller can
/// obtain came from [`FaceEmbedder::space`] or off a [`FaceEmbedding`], and in
/// both cases it is the space an embedder this crate loaded actually ran in.
/// A caller still chooses which artifact to load and what preprocessing to
/// declare — but not what the loaded bytes hash to.
#[derive(Debug, Clone, Copy)]
pub struct EmbeddingSpace {
  artifact: ArtifactDigest,
  input: &'static str,
  output: &'static str,
  dim: usize,
  preprocessing: Preprocessing,
}

impl EmbeddingSpace {
  /// The space a loaded artifact and its manifest name together.
  ///
  /// **The one place the projection happens**, so "which fields are the space"
  /// has a single answer with a single definition — and so a unit gate builds
  /// a space exactly the way [`FaceEmbedder::load`] builds one, rather than
  /// through a second spelling that could drift from it.
  #[inline]
  const fn of(artifact: ArtifactDigest, manifest: &FaceModel) -> Self {
    Self {
      artifact,
      input: manifest.input,
      output: manifest.output,
      dim: manifest.dim,
      preprocessing: manifest.preprocessing,
    }
  }

  /// The SHA-256 identity of the artifact's bytes.
  #[inline]
  pub const fn artifact(&self) -> ArtifactDigest {
    self.artifact
  }

  /// The input feature the pixels were written to.
  #[inline]
  pub const fn input(&self) -> &'static str {
    self.input
  }

  /// The output feature the embedding was read from.
  #[inline]
  pub const fn output(&self) -> &'static str {
    self.output
  }

  /// The embedding width — [`FaceModel::dim`], reconciled against the
  /// artifact's declared output at load.
  #[inline]
  pub const fn dim(&self) -> usize {
    self.dim
  }

  /// The host-side preprocessing the pixels went through.
  #[inline]
  pub const fn preprocessing(&self) -> Preprocessing {
    self.preprocessing
  }
}

impl PartialEq for EmbeddingSpace {
  /// **Defined as `space_difference` finding nothing** — the very walk
  /// [`FaceEmbedding::dot`] refuses on, so `a == b` and `a.dot(b)` cannot
  /// disagree about whether two vectors are comparable. One relation, written
  /// once.
  #[inline]
  fn eq(&self, other: &Self) -> bool {
    space_difference(*self, *other).is_none()
  }
}

impl Eq for EmbeddingSpace {}

impl core::hash::Hash for EmbeddingSpace {
  #[inline]
  fn hash<H: core::hash::Hasher>(&self, state: &mut H) {
    core::hash::Hash::hash(&self.artifact, state);
    core::hash::Hash::hash(&self.input, state);
    core::hash::Hash::hash(&self.output, state);
    core::hash::Hash::hash(&self.dim, state);
    core::hash::Hash::hash(&self.preprocessing, state);
  }
}

/// One face-embedding artifact's contract: what its input and output features
/// are called, how it wants its pixels, and how wide its embedding is.
///
/// A manifest is a VALUE, so wiring a second artifact with different
/// preprocessing is a different manifest at the same call site rather than a
/// second code path.
/// **No serde.** The feature names are `&'static str`, a compile-time contract
/// with the artifact rather than a runtime setting, and `Deserialize` cannot
/// produce a `&'static str` at all. The part that genuinely varies between
/// artifacts — [`Preprocessing`] — is serialisable on its own.
///
/// Equality here is equality of all four fields, with the floats compared by
/// `canonical_bits` as [`Preprocessing`] compares them — the SAME relation
/// [`EmbeddingSpace`] decides a cosine by, on the four fields the two types
/// share. It is deliberately a different type's relation nonetheless, because
/// the two types answer different questions: a manifest is what a caller
/// declares about an artifact, and a space is what an embedder actually ran
/// in. The space carries a fifth field a manifest cannot know — the
/// [`ArtifactDigest`] of the bytes [`FaceEmbedder::load`] read — so two equal
/// manifests name one space only when one artifact produced both.
/// `manifest_equality_and_space_identity_are_one_relation` pins that they
/// never disagree on the four they share.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
pub struct FaceModel {
  input: &'static str,
  output: &'static str,
  dim: usize,
  preprocessing: Preprocessing,
}

impl FaceModel {
  /// A manifest for an artifact with the given feature names and embedding
  /// width, preprocessed as [`Preprocessing::ARCFACE`].
  ///
  /// `dim` is 512 for every ArcFace-family artifact in issue #115's census.
  ///
  /// **A `dim` of zero is accepted HERE and refused at
  /// [`FaceEmbedder::load`]**, with [`Error::ZeroEmbeddingWidth`]. This
  /// constructor is `const` and total, so it cannot refuse anything; the refusal
  /// is at the one place that can, and it is a load-time refusal rather than a
  /// clause of the contract because the contract a zero width builds is
  /// satisfiable — see that error for the walk.
  #[inline]
  pub const fn new(input: &'static str, output: &'static str, dim: usize) -> Self {
    Self {
      input,
      output,
      dim,
      preprocessing: Preprocessing::ARCFACE,
    }
  }

  /// The model's input feature name.
  #[inline]
  pub const fn input(&self) -> &'static str {
    self.input
  }

  /// The model's output feature name.
  #[inline]
  pub const fn output(&self) -> &'static str {
    self.output
  }

  /// The embedding width the artifact produces.
  #[inline]
  pub const fn dim(&self) -> usize {
    self.dim
  }

  /// The artifact's host-side preprocessing.
  #[inline]
  pub const fn preprocessing(&self) -> Preprocessing {
    self.preprocessing
  }

  /// Builder form of [`Self::set_preprocessing`].
  #[must_use]
  #[inline]
  pub const fn with_preprocessing(mut self, preprocessing: Preprocessing) -> Self {
    self.set_preprocessing(preprocessing);
    self
  }

  /// Sets [`Self::preprocessing`] in place.
  #[inline]
  pub const fn set_preprocessing(&mut self, preprocessing: Preprocessing) -> &mut Self {
    self.preprocessing = preprocessing;
    self
  }
}

/// Default [`FaceEmbedderOptions::compute`]: [`ComputeUnits::All`].
///
/// **Deliberately not a measured pin, unlike `siglip`'s and `clap`'s** — and
/// no longer for want of a measurement. Those defaults belong to doors that
/// load ONE artifact, so characterising every arm on it decides the default.
/// This door loads whatever artifact its caller supplies, and a placement
/// measured on one is not a fact about another: the ArcFace sweep put the ANE
/// 3x ahead of `CpuOnly`, while `siglip`'s vision tower COLLAPSES on that same
/// arm. CoreML's own planner choice is the honest answer for an artifact this
/// crate has never seen.
///
/// The artifact it HAS seen carries its own answer beside its manifest:
/// `arcface::RECOMMENDED_COMPUTE` (behind `commercial-face-arcface`) is
/// [`ComputeUnits::CpuAndNeuralEngine`], measured at 287 faces/s against this
/// default's 224 with a third of the spread. A caller staging that bundle
/// should pass it; a caller staging something else should measure.
pub const DEFAULT_FACE_COMPUTE: ComputeUnits = ComputeUnits::All;

#[cfg(feature = "serde")]
fn default_face_compute() -> ComputeUnits {
  DEFAULT_FACE_COMPUTE
}

/// Construction options for [`FaceEmbedder`] (rust-options-pattern): a single
/// `compute` knob with one source of truth shared by `const new`/`Default`.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
pub struct FaceEmbedderOptions {
  #[cfg_attr(feature = "serde", serde(default = "default_face_compute"))]
  compute: ComputeUnits,
}

impl Default for FaceEmbedderOptions {
  fn default() -> Self {
    Self::new()
  }
}

impl FaceEmbedderOptions {
  /// Options matching the module default: [`DEFAULT_FACE_COMPUTE`].
  #[inline]
  pub const fn new() -> Self {
    Self {
      compute: DEFAULT_FACE_COMPUTE,
    }
  }

  /// Which hardware CoreML may schedule the graph on.
  #[inline]
  pub const fn compute(&self) -> ComputeUnits {
    self.compute
  }

  /// Builder form of [`Self::set_compute`].
  #[must_use]
  #[inline]
  pub const fn with_compute(mut self, compute: ComputeUnits) -> Self {
    self.set_compute(compute);
    self
  }

  /// Sets [`Self::compute`] in place.
  #[inline]
  pub const fn set_compute(&mut self, compute: ComputeUnits) -> &mut Self {
    self.compute = compute;
    self
  }
}

/// One face's L2-normalised embedding, carrying the space it belongs to.
///
/// Unit norm, so [`Self::cosine`] is a dot product and a threshold means the
/// same thing for every face. The width is the ARTIFACT's
/// ([`FaceModel::dim`]), not a code constant — 512 for every ArcFace-family
/// model in issue #115's census, but a second family is a different manifest,
/// not a different type.
///
/// # The space travels with the vector
///
/// A cosine is only meaningful between two vectors in the same space, and the
/// widths agreeing does not establish that: two 512-wide ArcFace-family
/// artifacts, or one artifact fed BGR where it was trained on RGB, produce
/// vectors whose dot product lands in `[−1, 1]` looking exactly like a
/// measurement. So each embedding carries the [`EmbeddingSpace`] it was
/// produced in and [`Self::dot`] REFUSES a pair that disagrees.
///
/// **A value here is produced, never assembled.** There is no public
/// constructor for a `FaceEmbedding`: the only way to obtain one is
/// [`FaceEmbedder::embed`], which stamps the space of its own bound manifest
/// onto every row it returns.
///
/// # What that guarantee IS, and what two previous rounds claimed it was
///
/// It was written down here that "there is no public `FaceModel` constructor,
/// so a caller-built `FaceModel` can never be stamped on a vector". **That was
/// false**: [`FaceModel::new`] and [`Preprocessing::new`] are both public
/// `const fn`. It was then written down that every field of the space is
/// therefore a value the caller chose — **and that is false now**, which is
/// the point of [`EmbeddingSpace::artifact`]. Both sentences are kept here
/// inverted rather than deleted, because the shape of the mistake is the
/// useful part: a guarantee is worth what its unforgeable half is worth, and
/// twice the unforgeable half was named wrongly.
///
/// What is actually true, stated as narrowly as it holds:
///
/// - a `FaceEmbedding`'s components came out of a real prediction by an
///   embedder this crate loaded — no caller can assemble one;
/// - the space stamped on it is the space that embedder actually ran in, not a
///   claim made later at the comparison site;
/// - its `artifact` is the SHA-256 of the bytes [`FaceEmbedder::load`] read.
///   A caller chooses which artifact to load; they do not choose what it
///   hashes to, and neither [`crate::embeddings::face::ArtifactDigest`] nor
///   [`EmbeddingSpace`] has a public constructor;
/// - `dim` was reconciled against the artifact's own declared output width at
///   load, with no exception left: the legacy `neuralNetwork` form that used to
///   declare no shape and carry the caller's claim to the first prediction is
///   now refused at load (see the module doc);
/// - the preprocessing half is caller-stated and cannot be otherwise: the
///   artifact does not declare its own normalisation, and the preprocessing
///   really is what the host did to the pixels, so comparing it is sound.
///
/// The strongest claim the type can now make, and it makes it: **two
/// `FaceEmbedding`s compare only if they were produced by byte-identical
/// artifacts as read at [`FaceEmbedder::load`], from the same output feature,
/// fed through the same input feature, with the same host preprocessing.
/// [`Self::dot`] cannot return a score across different weights.**
///
/// That claim carries one precondition, and it is the first residual below:
/// **the artifact must not be modified in place while an embedder holds it;
/// replace a model by an atomic `rename` and load a new embedder.**
///
/// # What this takes from `audio::speaker::calibrate`, and what it does not
///
/// The `SpeakerToken` work converged, over six rounds, on making a wrong
/// pairing **unrepresentable** rather than refused: the caller's key type was
/// removed from the cohort surface entirely, because the defect was never the
/// road that reached it but that an identity could be *resolved* from
/// caller-owned state at all.
///
/// Taken: the identity rides on the value rather than being an argument at the
/// call site, nothing is *resolved* at the comparison site, and the part this
/// crate cannot refute is stated below instead of claimed away.
///
/// **Not taken: a minted, process-unique token, and the reason is a
/// difference in the two problems rather than a lighter standard.** A cohort is
/// one object, so minting inside it costs nothing. A face embedding space is
/// legitimately produced by MORE than one producer — `&self` inference means
/// fan-out is one [`FaceEmbedder`] per worker over the same artifact (see that
/// type's doc), and a per-load token would refuse the cross-worker comparisons
/// those workers exist to make. **The digest sidesteps that because it is an
/// identity of the BYTES rather than of the load**: same bundle, same value,
/// on every worker and every machine. There is also no lookup here to remove:
/// `calibrate`'s defect was a question that could be answered twice
/// differently, and [`Self::dot`] asks nothing of caller-owned state — it
/// compares two values' own recorded spaces.
///
/// # The residuals, stated
///
/// Four, and each is a different kind of thing:
///
/// - **the artifact must not be modified in place while an embedder holds
///   it.** This is a PRECONDITION rather than something the crate checks, and
///   it is the same precondition CoreML itself has for a model it has mapped:
///   the digest above names the bytes as read at `load`, and under a quiescent
///   filesystem those are the bytes every prediction runs on. A model is
///   replaced by an atomic `rename` followed by loading a new embedder — the
///   live mapping keeps the old inode's bytes, which is what every macOS
///   updater relies on. Overwriting a mapped bundle underneath a live embedder
///   is a bug in the writer, and a digest is not a defence against a hostile
///   artifact or a hostile filesystem: neither is in this library's scope,
///   because whoever can rewrite the bundle can already choose which bytes it
///   loads. The digest exists against *confusion*;
/// - **a numerically identical re-export is REFUSED.** One set of weights
///   written out twice — recompiled, or renamed — is two artifacts by digest,
///   so their embeddings do not compare and a caller who re-exports has to
///   re-embed. That is loud and correct under this crate's provenance model,
///   where `MODELS_LOCK` already treats bundle bytes as identity: two files
///   that are not the same bytes are not the same artifact. It is a real cost,
///   and it is the price of the guarantee above rather than an oversight;
/// - **a caller who states the wrong preprocessing gets a consistent,
///   off-distribution space.** That is misuse rather than conflation — the
///   vectors are all wrong the same way, so they still compare with each other
///   — and it is unclosable here, because the artifact declares no
///   normalisation for this crate to check the claim against. The one part of
///   the claim that does not need the artifact IS checked: a preprocessing
///   whose map `byte ↦ byte · scale + bias` leaves `f32` is refused at load
///   ([`Error::NonFinitePreprocessing`]), evaluated at both ends of the byte
///   range rather than on the two fields alone, so no stamped space carries a
///   NaN;
/// - **[`AlignedFace::from_template_pixels`] keeps its documented hole on the
///   pixel side.** Bring-your-own-alignment cannot be checked: pixels aligned
///   to some other template, or not aligned at all, pass that constructor and
///   degrade every cosine silently. See its own doc.
///
/// What is NOT a residual any more is the one the round before last recorded
/// here: "two distinct artifacts with one schema are one space as far as this
/// type can see". They are two spaces, and
/// `two_artifacts_with_one_schema_are_two_spaces` is the gate.
#[derive(Debug, Clone, PartialEq)]
pub struct FaceEmbedding {
  /// The unit-norm components. Always `space.dim()` of them: the only
  /// constructor fills this from a row the space's own width cut.
  ///
  /// **A `Vec` rather than the `Box<[f32]>` this was, because there is no
  /// fallible way to reach the boxed slice.** The width is the manifest's and
  /// the buffer is reserved with `try_reserve_exact` (see
  /// [`embedding_buffer`]); `Vec::into_boxed_slice` then documents that it
  /// "discards excess capacity like `shrink_to_fit`", and `try_reserve_exact`
  /// documents that the allocator "may give the collection more space than it
  /// requests" — so the conversion is a REALLOCATION the standard library is
  /// free to perform, and its failure mode is `handle_alloc_error`, the abort
  /// this whole path exists to remove. Keeping the `Vec` is what makes the
  /// fallible reservation the last allocation on the path. Nothing public
  /// changes: the field is private and every accessor reads it as a slice.
  values: Vec<f32>,
  /// The space of the embedder that produced this vector.
  space: EmbeddingSpace,
}

impl FaceEmbedding {
  /// The embedding width.
  #[inline]
  pub fn dim(&self) -> usize {
    self.values.len()
  }

  /// The unit-norm components.
  #[inline]
  pub fn as_slice(&self) -> &[f32] {
    &self.values
  }

  /// An owned copy of the components.
  ///
  /// **Off the embed path rather than an exception to it.** Every allocation
  /// between [`FaceEmbedder::embed`] and the embeddings it returns is fallible,
  /// with no site excluded — this one is not on that path at all. It is an
  /// accessor the caller reaches for afterwards, and it duplicates a vector
  /// that already exists, so it asks the allocator for a length it has just
  /// served; the same is true of this type's derived [`Clone`], which is
  /// infallible for the same reason and could not be otherwise. It is a `Vec`
  /// method with no fallible spelling, and giving it one would put a `Result`
  /// on a copy of memory the caller already owns. [`Self::as_slice`] borrows
  /// the same components and allocates nothing, for a caller that does not
  /// need the copy.
  #[inline]
  pub fn to_vec(&self) -> Vec<f32> {
    self.values.to_vec()
  }

  /// The [`EmbeddingSpace`] this vector lives in — the space of the embedder
  /// that produced it.
  ///
  /// Readable so a caller storing embeddings can group them, and so a stored
  /// vector can be checked against a freshly loaded embedder before a batch of
  /// comparisons rather than one at a time.
  ///
  /// **Reading it forges nothing, and neither does stating one.** The four
  /// manifest fields are values the caller handed to [`FaceEmbedder::load`];
  /// the fifth is the digest of the bytes that door read, which the caller
  /// does not choose. And an [`EmbeddingSpace`] cannot be assembled at all —
  /// nor can a [`FaceEmbedding`]. See this type's doc for what that does and
  /// does not establish.
  #[inline]
  pub const fn space(&self) -> EmbeddingSpace {
    self.space
  }

  /// The dot product with `other`, which for two unit vectors in one space is
  /// their cosine.
  ///
  /// # Errors
  /// [`Error::IncomparableEmbeddings`] if the two came from different spaces,
  /// naming the first field that differs.
  ///
  /// **Fallible rather than a sentinel.** This used to return `0.0` for a
  /// width mismatch, which is also what a measured orthogonal pair returns —
  /// so a caller could not tell an incompatible model migration from a face
  /// that did not match. A width mismatch is now one arm of the space check,
  /// reported as [`EmbeddingSpaceField::Dim`], and the arm nothing could
  /// report before — equal widths, different spaces — is the rest of it.
  ///
  /// # Accumulated in `f64`, clamped, narrowed once
  ///
  /// A `f32` accumulation returns scores a cosine cannot have. The width-10
  /// witness in `a_unit_vector_never_scores_above_one_against_itself` scored
  /// `1.0000001192` against ITSELF — one `f32` ulp above one, which makes
  /// `acos` NaN, a `1 − cos` distance negative, and a threshold sweep produce
  /// a bucket that should be empty.
  ///
  /// **The clamp is correct here, and an error would be wrong.** Every stored
  /// component is the `f32` rounding of an exact unit component, so the sum is
  /// `Σ uᵢvᵢ(1 + εᵢ)(1 + δᵢ)` with `|εᵢ|, |δᵢ| ≤ 2⁻²⁴`: it can exceed one by
  /// at most `(1 + 2⁻²⁴)² − 1 = 1.19e-7` (measured worst case `8.8e-8`), and
  /// that excess is NARROWING ERROR rather than anything about the two faces.
  /// There is nothing to report, so the value is put back inside the interval
  /// its type promises instead of being turned into a refusal.
  #[inline]
  pub fn dot(&self, other: &Self) -> Result<f32> {
    if let Some(field) = space_difference(self.space, other.space) {
      return Err(Error::IncomparableEmbeddings(IncomparableEmbeddings::new(
        field,
      )));
    }
    let sum: f64 = self
      .values
      .iter()
      .zip(other.values.iter())
      .map(|(x, y)| f64::from(*x) * f64::from(*y))
      .sum();
    Ok(sum.clamp(-1.0, 1.0) as f32)
  }

  /// The cosine similarity with `other` — an alias for [`Self::dot`], since
  /// both operands are unit norm by construction.
  ///
  /// # Errors
  /// As [`Self::dot`].
  #[inline]
  pub fn cosine(&self, other: &Self) -> Result<f32> {
    self.dot(other)
  }
}

/// The first field of two spaces that puts their embeddings in different
/// spaces, or `None` when they are one space.
///
/// **The single definition of that relation.** [`EmbeddingSpace`]'s
/// [`PartialEq`] is this function, so the equality a caller can test and the
/// refusal [`FaceEmbedding::dot`] raises cannot disagree; there is no second
/// walk to drift from this one.
///
/// The `f32`s are compared by [`canonical_bits`] — neither `f32`'s `==`, which
/// makes a NaN scale unequal to itself, nor a raw `to_bits`, which makes `−0.0`
/// a different space from `+0.0`. Both of those are relations that answer a
/// question about the SPELLING where the question asked is about the function.
///
/// Every field is compared, not short-circuited — the array is built before
/// `find_map` walks it — so the order decides only WHICH field is named when
/// several differ at once. [`EmbeddingSpaceField::Artifact`] leads because
/// when the WEIGHTS differ every other agreement is coincidence: reporting a
/// matching width or a matching divisor would be true and would point a reader
/// at the wrong thing.
fn space_difference(left: EmbeddingSpace, right: EmbeddingSpace) -> Option<EmbeddingSpaceField> {
  let (lp, rp) = (left.preprocessing(), right.preprocessing());
  let bias = |b: [f32; 3]| b.map(canonical_bits);
  [
    (
      EmbeddingSpaceField::Artifact,
      left.artifact() != right.artifact(),
    ),
    (
      EmbeddingSpaceField::InputFeature,
      left.input() != right.input(),
    ),
    (
      EmbeddingSpaceField::OutputFeature,
      left.output() != right.output(),
    ),
    (EmbeddingSpaceField::Dim, left.dim() != right.dim()),
    (EmbeddingSpaceField::ChannelOrder, lp.order() != rp.order()),
    (
      EmbeddingSpaceField::TensorLayout,
      lp.layout() != rp.layout(),
    ),
    (
      EmbeddingSpaceField::PreprocessingScale,
      canonical_bits(lp.scale()) != canonical_bits(rp.scale()),
    ),
    (
      EmbeddingSpaceField::PreprocessingBias,
      bias(lp.bias()) != bias(rp.bias()),
    ),
  ]
  .into_iter()
  .find_map(|(field, differs)| differs.then_some(field))
}

/// The CoreML face embedder: a batch of [`AlignedFace`]s in, one
/// [`FaceEmbedding`] each out.
///
/// `&self` inference — the per-call input tensor is local, so fan-out means one
/// embedder per worker over a `Send` (but deliberately `!Sync`) [`Model`],
/// matching every other kit in this crate.
#[derive(Debug)]
pub struct FaceEmbedder {
  /// A [`Checked`], never a bare [`Model`]: [`load_contract`] builds the only
  /// contract this door states and [`Checked::new`] is the only way a model is
  /// wrapped in one, so deleting the check from [`Self::load`] does not
  /// compile.
  model: Checked,
  manifest: FaceModel,
  /// The space every vector this embedder produces is stamped with, built at
  /// load from the manifest AND the digest of the bytes that were read.
  space: EmbeddingSpace,
  /// The graph's own batch dimension AND declared rank. The rank is what
  /// decided the contract; the batch is READ BACK off the checked model, which
  /// is what makes [`Dim::AnyFixed`] a fact rather than a claim. The rank is
  /// carried, not just the capacity: a model that declares the unbatched
  /// rank-3 form has to be fed a rank-3 tensor.
  input: InputContract,
  /// The output form the graph declared, so a predicted tensor is checked
  /// against the axes it promised and not merely against an element count.
  output: OutputContract,
  /// The element counts one prediction allocates, established at load with
  /// `checked_mul` and carried so no inference-time site multiplies the
  /// artifact's batch again.
  elements: TensorElements,
}

impl FaceEmbedder {
  /// Loads a compiled `.mlmodelc` and binds it to `manifest`.
  ///
  /// The manifest's feature names and embedding width are reconciled against
  /// the model's declared contract here, so a manifest that names the wrong
  /// features, or claims the wrong width, fails at load rather than producing
  /// a plausible-looking wrong vector.
  ///
  /// # Whose licence these bytes carry
  ///
  /// **This door loads whatever artifact the caller names, and the licence of
  /// those bytes is the caller's.** Both arguments are the caller's values —
  /// the path, and a [`FaceModel`] they wrote — so a product holding a
  /// commercially licensed ArcFace-shaped model of its own writes its manifest
  /// and loads it here under the plain `face` feature, exactly as intended.
  /// Every refusal below is a CONTRACT refusal. The [`ArtifactDigest`] computed
  /// here is an identity STAMPED on the vectors, never a decision — no clause
  /// compares it against anything — and nothing here is a licence check.
  ///
  /// coremlit's own registered research-only artifact is a different thing, and
  /// it is **wired** — manifested, staged by `MODELS_LOCK`, and tested — only
  /// under the `commercial-face-arcface` feature, which is in no other
  /// feature's closure. That is the whole of what this crate's licence register
  /// (`tests/model_licences.rs`, whose module doc states the guarantee and its
  /// residual) governs; the residual itself is issue #138 §8's: the register
  /// governs what the crate's *features* wire, and `Model::load` is public, so
  /// any consumer can load any bytes they hold.
  ///
  /// # The contract, and where each of its numbers comes from
  ///
  /// The model is checked against a crate-internal `LoadContract` and held as
  /// a `Checked` whose only constructor runs that check, so there is no
  /// separate list of validations here to fall out of step with what the door
  /// needs:
  ///
  /// ```text
  /// input   manifest.input()   f32  [n, 3, 112, 112]  n AnyFixed, the rest Exactly
  ///                            f32  [3, 112, 112]     the unbatched form
  ///                     NHWC:  f32  [n, 112, 112, 3] / [112, 112, 3]
  /// output  manifest.output()  f32  [n, dim]          n Exactly the input's batch
  ///                            f32  [dim]             only where that batch is 1
  /// state   none
  /// ```
  ///
  /// `dim` is [`FaceModel::dim`] and the layout is
  /// [`Preprocessing::layout`] — both the caller's. `n` is the ARTIFACT's: the
  /// declared RANK of the input feature picks which of the two forms the
  /// contract states, and the batch axis is an "any one fixed size" axis —
  /// this door does not require a batch, it reads back whichever one the graph
  /// pins. It reads it off the CHECKED model, so the number
  /// [`Self::batch_capacity`] reports came from a description established to
  /// admit exactly one shape, rather than from the default a flexible graph
  /// also reports.
  ///
  /// **A batch the door reads is a batch the door has to size two buffers
  /// from, and that is checked here rather than trusted.** `n` is the
  /// artifact's, so nothing bounds it: `n = usize::MAX / 1000` is a well-formed
  /// pinned shape that used to load clean and then wrap `n · 112 · 112 · 3` on
  /// the way to an allocation, panicking on the first row slice out of the
  /// too-short buffer that resulted. `TensorElements` computes both counts
  /// with `checked_mul` at load, refuses on overflow
  /// ([`Error::ElementCountOverflow`]) and is CARRIED, so `embed` never
  /// multiplies the artifact's batch again. There is no cap — see that type for
  /// why a proof is the right shape and a cap is not.
  ///
  /// The OUTPUT's batch axis is `Exactly` that same number rather than
  /// `AnyFixed`, because this door does more than read it: [`Self::embed`]
  /// sends `n` faces and cuts `n` rows out of what comes back, so a graph that
  /// takes `n` and emits some other row count is one this door cannot use, and
  /// refusing it at load is the difference between a mismatch and a batch of
  /// silently wrong vectors.
  ///
  /// # What is refused, and why a list of feature checks was not enough
  ///
  /// The contract is complete over the three members of
  /// [`crate::ModelDescription`] that can make an otherwise-conformant
  /// prediction fail, not just over the two features this door names: a graph
  /// carrying the manifest's input plus another REQUIRED input clears every
  /// per-feature clause and then fails every prediction, and a STATE buffer is
  /// not an input at all — it lives in its own dictionary, so a stateful graph
  /// declaring exactly these two features clears the input set too and only
  /// then meets [`Self::embed`], which predicts through the stateless API
  /// CoreML does not let a stateful model be called with.
  ///
  /// Both features must be `float32` MULTI-ARRAYS with a PINNED shape.
  /// Inference supplies and extracts nothing else, so an f16 export — or an
  /// `ImageType` feature, which carries no shape and no element type at all
  /// and is what both third-party CoreML ArcFace builds this module's doc
  /// surveys declare — is refused here rather than loading clean and failing
  /// every prediction. A FLEXIBLE feature is refused for a different reason,
  /// and the module doc carries it along with the deliberate refusal of every
  /// legacy `neuralNetwork` export.
  ///
  /// # The digest names the bytes read here, and the artifact must hold still
  ///
  /// [`Self::space`] — and therefore every [`FaceEmbedding`] this embedder
  /// produces — carries the [`ArtifactDigest`] of the directory this path
  /// names. That is what makes the space an identity of the WEIGHTS rather
  /// than of a schema, and it is computed here because this is the only place
  /// that knows both the bytes and the manifest. Same bundle ⇒ same digest,
  /// on every worker and every machine, so the cross-worker comparisons
  /// `&self` inference exists to allow are unaffected.
  ///
  /// **One walk, of the same path handed to [`crate::Model::load`], taken
  /// LAST** — after the model has loaded and the contract has been checked, so
  /// a manifest the artifact refuses pays no walk at all. The digest identifies
  /// the bytes **as read at `load`**.
  ///
  /// **The precondition, stated rather than defended against: the artifact
  /// must not be modified in place while this embedder holds it.** That is the
  /// same precondition CoreML itself has for a model it has mapped — a model is
  /// replaced by an atomic `rename` followed by loading a new embedder, and the
  /// live mapping keeps the old inode's bytes. A digest is not a defence
  /// against a hostile artifact or a hostile filesystem, and neither is in this
  /// library's scope; the digest exists against *confusion*, and one walk
  /// catches that.
  ///
  /// # Errors
  /// [`Error::NonFinitePreprocessing`] if the manifest's map
  /// `byte ↦ byte · scale + bias` does not stay in `f32` — either field, or
  /// the map itself at an end of the byte range, which two finite fields can
  /// still fail; [`Error::ZeroEmbeddingWidth`] if the manifest's width is zero,
  /// which is refused at the manifest because no contract clause can refuse it
  /// and the failure is otherwise a PANIC in [`Self::embed`]'s row split;
  /// [`Error::Load`] if CoreML rejects the model;
  /// [`Error::ContractMismatch`]
  /// if the model declares no feature by the manifest's name, if the declared
  /// rank of either feature is one no contract of this door's can be built
  /// from (an undeclared shape included), if a named feature's element
  /// type, rank, shape flexibility or any one axis is not the contract's, or if
  /// the model declares the manifest's OUTPUT optional — a graph free to omit
  /// the feature this door reads;
  /// [`Error::UnsatisfiableInput`] if it requires an input this door never
  /// sends; [`Error::UnsatisfiableState`] if it declares a state buffer;
  /// [`Error::ElementCountOverflow`] if the batch the graph pins makes either
  /// the input or the output tensor's element count leave `usize`;
  /// [`Error::ArtifactDigest`] if the artifact's bytes cannot be read — which
  /// fails the load rather than producing vectors with no identity.
  pub fn load(
    model_path: impl AsRef<Path>,
    manifest: FaceModel,
    options: FaceEmbedderOptions,
  ) -> Result<Self> {
    let model_path = model_path.as_ref();
    let model = Model::load(model_path, options.compute())?;
    let resolved = load_contract(model.description(), &manifest)?;
    let model = Checked::new(model, &resolved.contract).map_err(contract_violation)?;
    let input = InputContract::read_back(model.description(), manifest.input(), resolved.rank);
    // Last, and over the same path CoreML was given: the digest names the bytes
    // as read here, and a manifest the artifact refuses never pays for a walk.
    let artifact = digest_artifact(model_path)?;
    let space = EmbeddingSpace::of(artifact, &manifest);
    Ok(Self {
      model,
      manifest,
      space,
      input,
      output: resolved.output,
      elements: resolved.elements,
    })
  }

  /// Loads with [`FaceEmbedderOptions::new`].
  ///
  /// # Errors
  /// As [`Self::load`].
  pub fn from_file(model_path: impl AsRef<Path>, manifest: FaceModel) -> Result<Self> {
    Self::load(model_path, manifest, FaceEmbedderOptions::new())
  }

  /// The manifest this embedder was bound to.
  #[inline]
  pub const fn manifest(&self) -> &FaceModel {
    &self.manifest
  }

  /// The [`EmbeddingSpace`] every vector from this embedder is stamped with.
  ///
  /// **The only public producer of a space**, and the reason it is here rather
  /// than on [`FaceModel`]: half of a space is the manifest and the other half
  /// is the digest of the bytes that were loaded, which a manifest does not
  /// know. Read it to group stored embeddings, or to check a stored vector
  /// against a freshly loaded embedder once instead of on every comparison.
  #[inline]
  pub const fn space(&self) -> EmbeddingSpace {
    self.space
  }

  /// The graph's own batch dimension, read off its input feature at load —
  /// after the load contract established that the feature admits exactly one
  /// shape, so this is the graph's only batch rather than its default one.
  ///
  /// [`Self::embed`] chunks any slice to this, so it is a throughput fact
  /// rather than a call-site constraint.
  ///
  /// **Not bounded, and it does not need to be.** No cap is imposed on what a
  /// graph may pin here; what [`Self::load`] establishes instead is that the
  /// two element counts this number sizes fit `usize` (`TensorElements`), and
  /// what `zeroed_tensor` establishes is that a buffer the allocator will not
  /// give is an error rather than an abort. A batch too large to be useful is
  /// the artifact's business; a batch that makes this door misbehave is not
  /// loadable.
  #[inline]
  pub const fn batch_capacity(&self) -> usize {
    self.input.batch
  }

  /// The embedding width — [`FaceModel::dim`], reconciled against the model at
  /// load, and never zero: [`Self::load`] refuses a zero-width manifest.
  #[inline]
  pub const fn dim(&self) -> usize {
    self.manifest.dim()
  }

  /// Embeds a batch of aligned faces, one [`FaceEmbedding`] per input, in
  /// order.
  ///
  /// An empty slice yields an empty vector without touching the model. Longer
  /// slices are chunked to [`Self::batch_capacity`]; a short final chunk is
  /// zero-padded and the padding rows are discarded, so the result length
  /// always equals `faces.len()`.
  ///
  /// # Errors
  /// [`Error::ResultAllocationFailed`] if the returned vector itself — one
  /// [`FaceEmbedding`] per face — is one the allocator will not give;
  /// [`Error::AllocationFailed`] if a buffer the graph's batch or the
  /// manifest's width sizes cannot be allocated — either per-prediction tensor,
  /// or any one of the per-row embeddings a chunk is cut into — an error rather
  /// than an abort, which is the whole reason every one of them is reserved
  /// fallibly; [`Error::Tensor`] / [`Error::Prediction`] on a tensor or CoreML
  /// failure, which includes
  /// [`PredictionError::AliasCopyFailed`](crate::PredictionError::AliasCopyFailed)
  /// carrying [`TensorError::AllocationFailed`](crate::TensorError::AllocationFailed)
  /// when a graph echoes its input back as its output and the de-aliasing copy
  /// cannot be allocated;
  /// [`Error::OutputShape`] if a predicted tensor's axes diverge from the
  /// contract resolved at load, or [`Error::OutputElementCount`] if only its
  /// element count does; [`Error::NonFiniteOutput`] if the model emits
  /// a NaN or infinite component; [`Error::EmbeddingZero`] if a (finite)
  /// output row has zero magnitude and cannot be normalised.
  pub fn embed(&self, faces: &[AlignedFace]) -> Result<Vec<FaceEmbedding>> {
    // The ONE reservation the whole call makes for its result, and the only
    // one: `predict_chunk` pushes its rows straight in here rather than
    // building a chunk-sized `Vec` of its own to be extended from. Two
    // allocations sized at run time became one, and that one is fallible.
    let mut out = result_buffer(faces.len())?;
    let batch = self.input.batch;
    for (chunk_index, chunk) in faces.chunks(batch).enumerate() {
      self.predict_chunk(chunk, chunk_index * batch, &mut out)?;
    }
    Ok(out)
  }

  /// Predicts one chunk of at most [`Self::batch_capacity`] faces, APPENDING
  /// its embeddings to `out`.
  ///
  /// `first_row` is the chunk's offset into the caller's slice, so every error
  /// names the caller's own index rather than a position inside a chunk the
  /// caller never saw.
  ///
  /// **Every buffer here is reserved fallibly, and the peak is why that has to
  /// include the per-row one.** The flat gather buffer is `elements.output`
  /// long, and the `chunk.len()` rows it is then cut into are each `dim` long
  /// and all live at once, so this function's high-water mark is `batch · dim`
  /// TWICE over on the Rust side, beside both native tensors. Reserving the
  /// flat buffer fallibly and the rows infallibly would move the abort rather
  /// than remove it — the rows are the larger half once the chunk is full.
  ///
  /// **The pushes cannot allocate, which is why appending is what this takes
  /// rather than a `Vec` of its own.** [`Self::embed`] reserved exactly
  /// `faces.len()`; `slice::chunks` PARTITIONS `faces`, so the chunk lengths
  /// sum to that; and this appends exactly `chunk.len()` rows, because `flat`
  /// is `batch · dim` long so `chunks_exact(dim)` yields `batch` rows and
  /// `take(chunk.len())` cuts that to the chunk's own count. `Vec::push`
  /// allocates only when the length has reached the capacity, which no prefix
  /// of that sum does. Returning a chunk-sized `Vec` instead put a second
  /// run-time-sized allocation on the path for the caller to extend from,
  /// which is one more site than the class needs.
  ///
  /// # Panics
  /// Never, and the one that could is `chunks_exact(dim)`, which panics on a
  /// chunk size of zero. `dim` is the MANIFEST's, so nothing about the graph
  /// bounds it; [`load_contract`] refuses a zero-width manifest before this
  /// door exists, which is what makes the split total.
  fn predict_chunk(
    &self,
    chunk: &[AlignedFace],
    first_row: usize,
    out: &mut Vec<FaceEmbedding>,
  ) -> Result<()> {
    let dim = self.manifest.dim();
    let tensor = self.build_input(chunk)?;
    let mut outputs = self
      .model
      .predict_with(&[(self.manifest.input(), &tensor)])?;
    let features = outputs
      .take(self.manifest.output())
      .ok_or_else(|| crate::PredictionError::MissingOutput(self.manifest.output().to_string()))?;
    let batch = self.input.batch;
    check_predicted_shape(
      features.shape(),
      features.count(),
      self.output,
      batch,
      dim,
      self.elements.output,
    )?;

    let mut flat = zeroed_tensor(PredictionTensor::Output, self.elements.output)?;
    features.copy_into::<f32>(&mut flat)?;
    let space = self.space;
    for (offset, row) in flat.chunks_exact(dim).take(chunk.len()).enumerate() {
      out.push(normalise_row(row, first_row + offset, space)?);
    }
    Ok(())
  }

  /// Builds the `[batch, …]` input tensor for one chunk, zero-padding the tail
  /// rows a short chunk leaves.
  ///
  /// The length is the one [`TensorElements`] proved at load, not
  /// `batch · 3 · pixels` recomputed here, and it is reserved through
  /// [`zeroed_tensor`] so an allocation the artifact's batch makes impossible
  /// is an error rather than an abort.
  fn build_input(&self, chunk: &[AlignedFace]) -> Result<MultiArray> {
    let preprocessing = self.manifest.preprocessing();
    let mut data = zeroed_tensor(PredictionTensor::Input, self.elements.input)?;
    for (row, face) in chunk.iter().enumerate() {
      write_row(
        &mut data[row * FACE_ELEMENTS..(row + 1) * FACE_ELEMENTS],
        face,
        preprocessing,
      );
    }
    let shape = input_shape(self.input, preprocessing.layout());
    Ok(MultiArray::from_slice(&shape, &data)?)
  }
}

/// The shape of the tensor [`FaceEmbedder::build_input`] hands the graph, for
/// a contract resolved at load.
///
/// The rank is the CONTRACT's, not a constant. A graph that declares
/// `[3, 112, 112]` is handed `[3, 112, 112]`; only a graph that declared the
/// batch axis is handed one.
///
/// Infallible, and the shape of every site on this path that is: the `Vec` is
/// three or four `usize`s by construction — a RANK, which both arms write out
/// literally — so no artifact number reaches its length and there is no
/// allocator refusal to report. See the module doc for the rule.
fn input_shape(contract: InputContract, layout: TensorLayout) -> Vec<usize> {
  let face = match layout {
    TensorLayout::Nchw => [3, TEMPLATE_SIZE, TEMPLATE_SIZE],
    TensorLayout::Nhwc => [TEMPLATE_SIZE, TEMPLATE_SIZE, 3],
  };
  match contract.rank {
    InputRank::Unbatched => face.to_vec(),
    InputRank::Batched => {
      let mut shape = Vec::with_capacity(1 + face.len());
      shape.push(contract.batch);
      shape.extend_from_slice(&face);
      shape
    }
  }
}

/// The declared feature names, for a contract-mismatch message.
///
/// The one allocation in this module whose length is a run-time number and
/// which is still infallible — how many features the description declares —
/// for two reasons no site on the embed path can claim. It runs at LOAD, and
/// only on a refusal; and it borrows names out of a `[FeatureInfo]` slice that
/// is already materialised, so each `&str` it collects is 16 bytes against a
/// `FeatureInfo` several times that. It cannot ask for memory the caller is not
/// already holding.
fn feature_names(features: &[crate::FeatureInfo]) -> Vec<&str> {
  features.iter().map(crate::FeatureInfo::name).collect()
}

/// Whether a model's input feature declares the batch axis.
///
/// Kept, rather than collapsed into the numeric capacity, because it decides
/// the RANK of the tensor [`input_shape`] then builds. Resolving
/// `[3, 112, 112]` to "batch 1" and building `[1, 3, 112, 112]` from it means
/// a model that loads as supported fails every prediction.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
enum InputRank {
  /// `[n, 3, 112, 112]` / `[n, 112, 112, 3]` — the batch axis is declared.
  Batched,
  /// `[3, 112, 112]` / `[112, 112, 3]` — no batch axis, so capacity 1.
  Unbatched,
}

/// The input contract resolved from a model's declared feature at load.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
struct InputContract {
  /// How many faces one prediction consumes.
  batch: usize,
  /// The rank the graph declared, and therefore the rank it must be fed.
  rank: InputRank,
}

impl InputContract {
  /// The contract the door runs on, taken off a model that has ALREADY been
  /// checked against the [`LoadContract`] `rank` came from.
  ///
  /// # Why the batch is read here and not kept from the declaration
  ///
  /// [`Dim::AnyFixed`] is specified as an axis whose value the door reads back
  /// after the check, and the two moments are not the same fact. Before it,
  /// [`crate::FeatureInfo::shape`] can be the DEFAULT shape of a flexible
  /// feature — a `RangeDim` or enumerated graph reports one it will accept
  /// others beside. After it, the feature is
  /// [`crate::ShapeConstraint::Fixed`], which is what an `AnyFixed` axis
  /// requires, so the number is the graph's only batch rather than a reading of
  /// its declaration. [`load_contract`] therefore returns the RANK and not the
  /// batch: the rank is what the contract is built from, the batch is what the
  /// door then runs on.
  ///
  /// # Panics
  /// Never, for a description [`Checked::new`] accepted against that contract:
  /// the check established that `feature` is declared and has exactly this
  /// rank.
  fn read_back(description: &ModelDescription, feature: &str, rank: InputRank) -> Self {
    let batch = match rank {
      InputRank::Unbatched => 1,
      InputRank::Batched => description
        .input(feature)
        .and_then(|declared| declared.shape().first().copied())
        .expect("the load contract established this feature and its rank"),
    };
    Self { batch, rank }
  }
}

/// The element counts one prediction allocates, PROVED at load to fit `usize`.
///
/// # Why a proof, and why it is kept rather than recomputed
///
/// Both counts are products with the ARTIFACT's batch in them, and that batch
/// is not a number this crate or its caller chose: the input contract states
/// the batch axis as [`Dim::AnyFixed`] and [`InputContract::read_back`] reads
/// back whatever the graph pins, and nothing in a `.mlmodelc` bounds it. A
/// `usize::MAX / 1000` batch declares a perfectly well-formed fixed shape.
///
/// Computed with `*` at the point of use, `batch · 112 · 112 · 3` then wraps
/// silently in a release build; `build_input` allocates the wrapped length and
/// panics on the first `row * FACE_ELEMENTS ..` slice, so a model the door
/// ACCEPTED terminates the caller. Computed here with `checked_mul` it is
/// [`Error::ElementCountOverflow`] — a refusal at load, from a value the door
/// then carries, so no inference-time site multiplies an artifact-derived
/// number again and there is no second spelling to drift.
///
/// **No cap.** A cap would be an enumeration of how big is too big; the product
/// either fits `usize` or it does not. Fitting `usize` is also strictly weaker
/// than the memory existing, which is why the buffers themselves are still
/// reserved fallibly — see [`zeroed_tensor`].
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
struct TensorElements {
  /// `batch · 112 · 112 · 3` — what [`FaceEmbedder::build_input`] allocates.
  input: usize,
  /// `batch · dim` — what [`FaceEmbedder::predict_chunk`] allocates, and the
  /// count [`check_predicted_shape`] measures the predicted tensor against.
  output: usize,
}

impl TensorElements {
  /// Both counts for a graph of this batch under a manifest of this width.
  ///
  /// # Errors
  /// [`Error::ElementCountOverflow`] naming the tensor whose product leaves
  /// `usize`, the batch, and that tensor's per-row count. The input is checked
  /// first only so that a description overflowing both reports one thing; each
  /// is refused on its own, and `dim` is the manifest's, so the output count
  /// can overflow where the input's does not.
  fn of(batch: usize, dim: usize) -> Result<Self> {
    let input = batch
      .checked_mul(FACE_ELEMENTS)
      .ok_or(Error::ElementCountOverflow(ElementCountOverflow::new(
        PredictionTensor::Input,
        batch,
        FACE_ELEMENTS,
      )))?;
    let output =
      batch
        .checked_mul(dim)
        .ok_or(Error::ElementCountOverflow(ElementCountOverflow::new(
          PredictionTensor::Output,
          batch,
          dim,
        )))?;
    Ok(Self { input, output })
  }
}

/// A zeroed `f32` buffer of `elements`, reserved FALLIBLY.
///
/// `vec![0.0f32; n]` on an artifact-derived `n` answers "the allocator will not
/// give me that" by aborting the process, which is not a thing a caller can
/// handle and not a thing a `Result`-returning door should do.
/// [`TensorElements`] has already proved the count fits `usize`, and that is a
/// strictly weaker fact than the memory existing — a batch of `2⁵⁵` counts fine
/// and asks for petabytes — so the reservation is `try_reserve_exact` and the
/// failure is [`Error::AllocationFailed`], naming the tensor and the length
/// that was refused.
///
/// The `resize` cannot allocate: the reservation above it already secured
/// capacity for exactly `elements`.
///
/// This covers the two PER-PREDICTION tensors only. The per-ROW buffer
/// [`normalise_row`] fills has the same provenance and the same failure mode,
/// and is reserved through [`embedding_buffer`].
fn zeroed_tensor(tensor: PredictionTensor, elements: usize) -> Result<Vec<f32>> {
  let mut data: Vec<f32> = Vec::new();
  data
    .try_reserve_exact(elements)
    .map_err(|_| Error::AllocationFailed(AllocationFailed::new(tensor, elements)))?;
  data.resize(elements, 0.0);
  Ok(data)
}

/// What [`load_contract`] resolves off a description: the contract
/// [`Checked::new`] then checks, and the three facts the door runs on once it
/// has passed.
struct Resolved {
  /// The contract to check the description against.
  contract: LoadContract,
  /// The rank the graph declared, and therefore the rank it must be fed.
  rank: InputRank,
  /// The output form a predicted tensor is later measured against.
  output: OutputContract,
  /// The two element counts one prediction allocates.
  elements: TensorElements,
}

/// The load contract this door states for `manifest`, with the two forms
/// [`FaceEmbedder`] carries: the RANK it must feed, and the output form a
/// predicted tensor is later measured against.
///
/// **Pure over a [`ModelDescription`], so every clause is drivable with no
/// model present.** [`FaceEmbedder::load`] runs exactly this and then
/// [`Checked::new`], which is [`crate::model::contract::check_load_contract`]
/// over the same description; this module's fixtures run that same pair. It is
/// the seam that lets a door staging no artifact still gate its own load path.
///
/// # Why the description is read BEFORE it is checked
///
/// A contract cannot be checked before it exists, and this one's SHAPE comes
/// off the artifact: the declared rank of each feature decides which of the two
/// forms the contract states. That reading is not trusted — it is what the
/// check then confirms. A declaration that lies about its geometry, such as a
/// flexible input whose default shape reads exactly `[n, 3, 112, 112]`, builds
/// a contract it then fails; the reading never becomes a fact without passing.
///
/// The order the clauses are checked in does not change that verdict, only
/// which clause is named: the batch this reads off the input is used to state
/// the OUTPUT's row count, and if the input's own declaration is not pinned,
/// the input clause refuses the model whether it is consulted first or last.
///
/// # The manifest's own numbers are refused before the graph is read
///
/// A manifest carries exactly two kinds of number, and both are checked here
/// rather than where they are used:
///
///   - **`preprocessing`'s `scale` and `bias`** — refused when the affine MAP
///     they make leaves `f32`, which also covers the degenerate constructions
///     that reach them: [`Preprocessing::from_mean_and_divisor`] divides by its
///     `divisor`, and a zero one makes `scale` `±inf` while a NaN one makes it
///     NaN. Either way this clause is what stops it, and it stops it on the
///     resulting `scale` rather than on the divisor, so there is no second
///     predicate to drift. A `scale` of zero is NOT degenerate: it writes a
///     constant tensor, which is well defined if useless, and an output row
///     that then comes back all-zero meets [`Error::EmbeddingZero`] like any
///     other.
///   - **`dim`** — refused at zero, see below.
///
/// Nothing else the door multiplies, divides or chunks by is the manifest's.
/// `TEMPLATE_SIZE` and [`FACE_ELEMENTS`] are compile-time constants, and the
/// only other number in the arithmetic is the ARTIFACT's batch, whose zero
/// [`input_form`] refuses and whose overflow [`TensorElements`] does.
///
/// # Errors
/// [`Error::NonFinitePreprocessing`] for a manifest whose map leaves `f32`;
/// [`Error::ZeroEmbeddingWidth`] for a manifest of zero width, which no clause
/// of the contract it would otherwise build can refuse — see
/// [`ZeroEmbeddingWidth`];
/// [`Error::ContractMismatch`] naming the feature whose declaration no contract
/// of this door's can be built from: absent, or of a rank that is neither the
/// batched nor the unbatched form — an undeclared (empty) shape included, and a
/// declared batch of zero with it;
/// [`Error::ElementCountOverflow`] if the batch the input feature declares
/// makes either tensor's element count leave `usize` — see [`TensorElements`].
fn load_contract(description: &ModelDescription, manifest: &FaceModel) -> Result<Resolved> {
  // Before anything about the graph: a manifest whose MAP does not stay in
  // `f32` makes elements of the input tensor non-finite, and the manifest is
  // copied verbatim into the `EmbeddingSpace` stamped on the vectors. Refusing
  // it here is what keeps a NaN out of a produced space, and therefore what
  // makes `canonical_bits`' NaN fold a statement about `Preprocessing`'s own
  // `Eq` and nothing more. The check is on the map at both ends of the byte
  // range, not on the two fields alone — `f32::MAX` and `0.0` are both finite
  // and their map is not.
  if let Some(field) = non_finite_preprocessing(manifest.preprocessing()) {
    return Err(Error::NonFinitePreprocessing(NonFinitePreprocessing::new(
      field,
    )));
  }
  // Also before the graph, and before any contract is built: a manifest of ZERO
  // width. This is the door's only refusal of a manifest NUMBER, and it has to
  // live here because nothing after it can make it — every clause a zero width
  // reaches is satisfied by it, and the failure lands in `predict_chunk`'s
  // `chunks_exact(dim)` as a panic. See [`ZeroEmbeddingWidth`] for the walk.
  if manifest.dim() == 0 {
    return Err(Error::ZeroEmbeddingWidth(ZeroEmbeddingWidth::new(
      manifest.output(),
    )));
  }
  let layout = manifest.preprocessing().layout();
  let declared_input = description.input(manifest.input()).ok_or_else(|| {
    Error::ContractMismatch(ContractMismatch::new(
      manifest.input().to_string(),
      "a declared input feature".to_string(),
      format!("inputs {:?}", feature_names(description.inputs())),
    ))
  })?;
  let (rank, batch) = input_form(declared_input.shape()).ok_or_else(|| {
    Error::ContractMismatch(ContractMismatch::new(
      manifest.input().to_string(),
      format!(
        "{layout:?} shaped [n, 3, {TEMPLATE_SIZE}, {TEMPLATE_SIZE}] (or without the batch axis)"
      ),
      format!("{:?}", declared_input.shape()),
    ))
  })?;

  // The batch is now a number, and both tensors are sized from it. Prove the
  // two products fit `usize` HERE, where the batch first becomes one, rather
  // than at the two allocation sites where a wrap would be silent.
  let elements = TensorElements::of(batch, manifest.dim())?;

  let declared_output = description.output(manifest.output()).ok_or_else(|| {
    Error::ContractMismatch(ContractMismatch::new(
      manifest.output().to_string(),
      "a declared output feature".to_string(),
      format!("outputs {:?}", feature_names(description.outputs())),
    ))
  })?;
  let form = output_form(declared_output.shape(), batch).ok_or_else(|| {
    Error::ContractMismatch(ContractMismatch::new(
      manifest.output().to_string(),
      format!(
        "shaped [{batch}, {}] (or [{}] for a batch-one graph)",
        manifest.dim(),
        manifest.dim()
      ),
      format!("{:?}", declared_output.shape()),
    ))
  })?;

  let contract = LoadContract::new(
    vec![FeatureContract::new(
      manifest.input(),
      DataType::F32,
      input_dims(rank, layout),
    )],
    vec![FeatureContract::new(
      manifest.output(),
      DataType::F32,
      output_dims(form, batch, manifest.dim()),
    )],
    StateContract::None,
  );
  Ok(Resolved {
    contract,
    rank,
    output: form,
    elements,
  })
}

/// The first thing about `preprocessing` that does not stay in `f32` — the
/// MAP, evaluated at both ends of the byte range, attributed to a field where
/// a field is what is wrong.
///
/// # The map, not the fields — and the endpoints are a PROOF
///
/// Checking `scale` and `bias` for finiteness one at a time was an enumeration
/// of what can go wrong that missed the thing the fields are for:
/// `scale = f32::MAX` with `bias = 0` is two perfectly finite numbers whose
/// map writes `+inf` for every byte from 2 upwards, so the input tensor was
/// non-finite from a manifest the load had blessed.
///
/// `byte ↦ byte · scale + bias[channel]` is **affine in `byte`**. The exact
/// value at any byte therefore lies between the exact values at `0` and `255`,
/// and rounding to nearest is monotone — so if both endpoints round to finite
/// `f32`, every byte between them does. That is a proof over the whole domain,
/// not a sample of it, which is why two evaluations per channel are enough and
/// why no third byte needs a rule of its own. NaN cannot arise at all once
/// `scale` and `bias` are finite.
///
/// The expression evaluated here is [`write_row`]'s own `mul_add`, so what is
/// proved is what will be written rather than a differently associated
/// stand-in for it.
///
/// # Which end fires, and which end cannot
///
/// Only the far one, and the reason is arithmetic rather than a gap in the
/// gate: `byte 0 · scale + bias` is exactly `bias` for any finite `scale`, so
/// once the field checks have passed, the byte-0 endpoint is finite by
/// construction. It is evaluated because the PAIR is what proves the 254 bytes
/// between them, and the gate says so where it pins the mutation that drops
/// the far one.
///
/// # The FIRST, not all of them
///
/// One thing that is definitely wrong is more actionable than a list assembled
/// to look thorough, and the same rule [`space_difference`] follows. The
/// failure is attributed to the most specific thing about it: a non-finite
/// `scale` is [`PreprocessingField::Scale`], a non-finite `bias` is
/// [`PreprocessingField::Bias`], and a map that leaves `f32` out of two finite
/// fields is [`PreprocessingField::Map`] naming the channel and the endpoint.
fn non_finite_preprocessing(preprocessing: Preprocessing) -> Option<PreprocessingField> {
  let scale = preprocessing.scale();
  for (channel, offset) in preprocessing.bias().iter().enumerate() {
    for byte in [u8::MIN, u8::MAX] {
      if f32::from(byte).mul_add(scale, *offset).is_finite() {
        continue;
      }
      if !scale.is_finite() {
        return Some(PreprocessingField::Scale);
      }
      if !offset.is_finite() {
        return Some(PreprocessingField::Bias(channel));
      }
      return Some(PreprocessingField::Map(PreprocessingMap::new(
        channel, byte,
      )));
    }
  }
  None
}

/// Map a [`ContractViolation`] into this module's error vocabulary.
///
/// **A newtype variant over the violation itself would be the house shape, and
/// it is not available here.** [`ContractViolation`] is `pub(crate)`, so a
/// public [`Error`] variant carrying one would export a private type; widening
/// the whole contract vocabulary to `pub` for one door's error message is a
/// larger change to a shared type than this door's convenience earns. So the
/// violation is RENDERED, the way `audio::identity` renders it: the
/// per-feature clauses land in [`Error::ContractMismatch`], which already
/// carries a feature name and an expected/actual pair, and the two
/// "unsatisfiable" clauses keep newtype variants of their own, because they are
/// about what the door cannot SUPPLY rather than about a feature's shape. An
/// output the model declares OPTIONAL is a per-feature clause: it is a fact
/// about that feature's declaration, which is what the expected/actual pair is
/// for.
///
/// `ContractViolation::rendered` performs that reduction once for every door,
/// so a clause added to the checker later lands in the `Feature` arm rather
/// than breaking this function and its five siblings at once.
fn contract_violation(violation: ContractViolation) -> Error {
  match violation.rendered() {
    Rendered::UnsatisfiableInput(name) => Error::UnsatisfiableInput(name),
    Rendered::UnsatisfiableState(name) => Error::UnsatisfiableState(name),
    Rendered::Feature(feature) => Error::ContractMismatch(ContractMismatch::new(
      feature.feature().to_string(),
      feature.clone().expected(),
      feature.actual(),
    )),
  }
}

/// Which of the two forms a model's input feature declares, and the batch that
/// form implies — or `None` for a declared rank that is neither.
///
/// **The RANK is all this decides**, and that is the narrowing this door's
/// adoption of the load contract bought. The element type, and that the three
/// trailing axes really are a `3 × 112 × 112` template face in the manifest's
/// layout, are clauses of the contract [`load_contract`] then builds — checked
/// once by [`Checked::new`] rather than twice in two spellings that could
/// disagree.
///
/// An EMPTY shape is refused, and that is deliberate rather than an oversight:
/// the legacy `neuralnetwork` specification declares none, and this used to
/// resolve one to a batch-one guess. The module doc carries the argument.
///
/// A declared batch of ZERO is refused here even though [`Dim::AnyFixed`]'s own
/// clause refuses the same zero, because this number is read BEFORE the contract
/// exists and is an ARGUMENT to building it: the batch becomes
/// [`Dim::Exactly`]`(batch)` on the output feature, picks [`output_form`]'s
/// batch-one arm, and sizes [`TensorElements::of`] — and a contract built from
/// zero states `Exactly(0)` and is satisfied by a graph that embeds nothing.
/// Reading the number here is not a licence to TRUST it: the input's batch axis
/// is `AnyFixed` in the contract [`Checked::new`] then runs, so every batch this
/// door goes on to allocate from is one that check established.
/// [`FaceEmbedder::embed`] would divide its work into chunks of zero and never
/// terminate, so it is a contract mismatch and not a capacity.
fn input_form(shape: &[usize]) -> Option<(InputRank, usize)> {
  match shape.len() {
    3 => Some((InputRank::Unbatched, 1)),
    4 if shape[0] > 0 => Some((InputRank::Batched, shape[0])),
    _ => None,
  }
}

/// The contract axes for an input feature of this rank and layout — the
/// per-axis form of the shape [`input_shape`] later builds, so the geometry the
/// door STATES and the geometry it FEEDS come from one pair of arms.
///
/// Infallible for [`input_shape`]'s reason: three or four axes, written out
/// literally, with no artifact number in the length.
fn input_dims(rank: InputRank, layout: TensorLayout) -> Vec<Dim> {
  let face = match layout {
    TensorLayout::Nchw => [
      Dim::Exactly(3),
      Dim::Exactly(TEMPLATE_SIZE),
      Dim::Exactly(TEMPLATE_SIZE),
    ],
    TensorLayout::Nhwc => [
      Dim::Exactly(TEMPLATE_SIZE),
      Dim::Exactly(TEMPLATE_SIZE),
      Dim::Exactly(3),
    ],
  };
  match rank {
    InputRank::Unbatched => face.to_vec(),
    InputRank::Batched => {
      let mut dims = Vec::with_capacity(1 + face.len());
      // Not `Exactly`: this door does not require a batch, it reads back
      // whichever one the graph pins.
      dims.push(Dim::AnyFixed);
      dims.extend_from_slice(&face);
      dims
    }
  }
}

/// The output form a model declared — and therefore the EXACT shape its
/// predicted tensor must have.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
enum OutputContract {
  /// The model declared `[batch, dim]`.
  Batched,
  /// The model declared `[dim]`, which only a batch-one graph can mean.
  Flat,
}

/// Which output form a model's output feature declares, or `None` for a
/// declared rank that is neither.
///
/// A `[dim]` shape is a batch-one form. Declared against a batch of 4 it is not
/// a shorthand, it is a contradiction, so it is refused here rather than
/// allowed to build a contract the predicted tensor could never satisfy. An
/// EMPTY shape is refused for the reason [`input_form`] refuses one — this is
/// where the `Undeclared` arm used to be.
fn output_form(shape: &[usize], batch: usize) -> Option<OutputContract> {
  match shape.len() {
    1 if batch == 1 => Some(OutputContract::Flat),
    2 => Some(OutputContract::Batched),
    _ => None,
  }
}

/// The contract axes for the output feature: `[batch, dim]`, or the bare
/// `[dim]` a batch-one graph may declare instead.
///
/// The batch axis is [`Dim::Exactly`] rather than [`Dim::AnyFixed`] because
/// this door does more than read it back — see [`FaceEmbedder::load`].
///
/// Infallible for [`input_shape`]'s reason: one axis or two. The artifact's
/// batch appears as a VALUE inside a `Dim`, never as the length.
fn output_dims(form: OutputContract, batch: usize, dim: usize) -> Vec<Dim> {
  match form {
    OutputContract::Flat => vec![Dim::Exactly(dim)],
    OutputContract::Batched => vec![Dim::Exactly(batch), Dim::Exactly(dim)],
  }
}

/// Checks a PREDICTED tensor against the contract resolved at load — its AXES,
/// not merely its element count.
///
/// A count-only check is close to no check here. `[dim, batch]` holds exactly
/// as many elements as `[batch, dim]`, so it passes, and the
/// `chunks_exact(dim)` that follows then slices across the wrong axis: every
/// returned embedding is a mixture of components from different faces,
/// unit-norm and plausible, and nothing downstream — no shape check, no
/// finiteness scan, no cosine — can tell. So the shape must equal the resolved
/// contract exactly, with the bare `[dim]` form allowed only where the
/// contract really is batch-one.
///
/// The count is still checked alongside it: [`MultiArray::count`] is CoreML's
/// own answer rather than a product of the cached shape, and the copy that
/// follows sizes its destination from the contract. It gets an error of its
/// OWN, because it is a different failure: with the axes equal there is no
/// shape mismatch to report, and [`Error::OutputShape`] could only report one
/// by naming the same vector twice.
///
/// `elements` is that required count — [`TensorElements::output`], proved to
/// fit `usize` at load — and is PASSED rather than recomputed as `batch · dim`,
/// which is the same artifact-derived product that would wrap here as silently
/// as it would at an allocation.
///
/// This is a DIFFERENT moment from the load contract, which is why it survives
/// the adoption of [`Checked`]: the contract established what the graph
/// DECLARES, and this measures what one prediction actually produced.
fn check_predicted_shape(
  shape: &[usize],
  count: usize,
  contract: OutputContract,
  batch: usize,
  dim: usize,
  elements: usize,
) -> Result<()> {
  let batched = [batch, dim];
  let flat = [dim];
  let expected: &[usize] = match contract {
    OutputContract::Batched => &batched,
    OutputContract::Flat => &flat,
  };
  if shape != expected {
    return Err(Error::OutputShape(OutputShape::new(
      shape.to_vec(),
      expected.to_vec(),
    )));
  }
  if count != elements {
    return Err(Error::OutputElementCount(OutputElementCount::new(
      count, elements,
    )));
  }
  Ok(())
}

/// Writes one aligned face into `row` as `3 · 112 · 112` preprocessed floats.
///
/// **Every zero written is `+0.0`.** [`Preprocessing`]'s equality folds `±0`
/// onto one value, so two manifests differing only in the sign of a zero bias
/// are ONE [`EmbeddingSpace`] and their embeddings compare — but `byte · scale
/// + bias` does not fold it: pixel `0` with `scale = −1` gives `−0.0` from the
/// multiply, and `+0.0` and `−0.0` as the bias then write two different bit
/// patterns. A graph can read that difference (`sign`, `copysign`, and `1/x`
/// as `+∞` against `−∞`), so one space could produce two tensors. It is
/// canonicalised here, at the producer, rather than compared away downstream:
/// see `a_written_zero_is_positive_zero_whichever_sign_the_bias_carries`.
///
/// **Every value written is finite, and that is the LOAD's guarantee rather
/// than this function's.** `non_finite_preprocessing` evaluates the `mul_add`
/// below at byte `0` and byte `255` for each channel, and the map is affine in
/// `byte`, so the two endpoints being finite proves all 256 are. A check here
/// would add nothing the load has not already established, and one per pixel
/// would pay 37 632 branches to re-derive it.
fn write_row(row: &mut [f32], face: &AlignedFace, preprocessing: Preprocessing) {
  let pixels = TEMPLATE_SIZE * TEMPLATE_SIZE;
  let source = face.pixels();
  let (scale, bias) = (preprocessing.scale(), preprocessing.bias());
  for pixel in 0..pixels {
    for (channel, offset) in bias.iter().enumerate() {
      // `channel` indexes the MODEL's channel; `source_channel` is where that
      // channel's byte lives in the RGB-interleaved template.
      let source_channel = match preprocessing.order() {
        ChannelOrder::Rgb => channel,
        ChannelOrder::Bgr => 2 - channel,
      };
      // `+ 0.0` normalises `−0.0` to `+0.0` and leaves every other value
      // alone: under round-to-nearest the sum of two zeros of opposite sign
      // is `+0.0`, and `x + 0.0` is exactly `x` for any nonzero finite `x`
      // (and for `±∞`). One add, no branch, and the sign of a zero cannot
      // reach the tensor.
      let value = f32::from(source[pixel * 3 + source_channel]).mul_add(scale, *offset) + 0.0;
      let index = match preprocessing.layout() {
        TensorLayout::Nchw => channel * pixels + pixel,
        TensorLayout::Nhwc => pixel * 3 + channel,
      };
      row[index] = value;
    }
  }
}

/// L2-normalises one model output row, classifying a non-finite component and
/// a zero magnitude separately.
///
/// The squared norm accumulates in `f64`, and the division happens there too.
/// In `f32` it could not: `v * v` overflows to infinity for a large component
/// and underflows to zero for a small one, and BOTH used to be reported as
/// [`Error::EmbeddingZero`] — "this row has no direction" — for a row with a
/// perfectly good direction. Nothing in the contract says an artifact's
/// pre-normalisation output is near unit scale, so its magnitude is the
/// model's business and not a reason to refuse it.
///
/// In `f64` the accumulator cannot leave the type: every component is a finite
/// `f32` by the scan above, so each square is at most `f32::MAX²` (≈1.2e77)
/// and even a very wide embedding sums far below `f64::MAX`, while the
/// smallest nonzero `f32` squares to ≈2e-90 — comfortably normal. The norm is
/// therefore zero if and only if every component is exactly zero, which is the
/// only row that genuinely has no direction.
///
/// # Errors
/// [`Error::NonFiniteOutput`] naming the row and the first non-finite
/// component; [`Error::EmbeddingZero`] naming a row whose (finite) components
/// are all exactly zero; [`Error::AllocationFailed`] if the row's own buffer —
/// `dim` `f32`s, the manifest's width, reserved through [`embedding_buffer`]
/// — is one the allocator will not serve.
fn normalise_row(row: &[f32], index: usize, space: EmbeddingSpace) -> Result<FaceEmbedding> {
  if let Some(component) = row.iter().position(|v| !v.is_finite()) {
    return Err(Error::NonFiniteOutput(NonFiniteOutput::new(
      index, component,
    )));
  }
  let norm = row
    .iter()
    .map(|v| f64::from(*v) * f64::from(*v))
    .sum::<f64>()
    .sqrt();
  if norm == 0.0 {
    return Err(Error::EmbeddingZero(BatchRow::new(index)));
  }
  let mut values = embedding_buffer(row.len())?;
  // `extend` cannot allocate: `Vec::reserve` is documented to do nothing when
  // the capacity is already sufficient, and the reservation above secured
  // exactly `row.len()`.
  //
  // Divided in `f64` and narrowed once, at the end: scaling in `f32` would put
  // back the overflow this widening exists to remove.
  values.extend(row.iter().map(|v| (f64::from(*v) / norm) as f32));
  Ok(FaceEmbedding {
    values,
    // The one place a space is attached, and it is the space of the embedder
    // that just ran — the function these numbers actually came out of — rather
    // than one stated about them afterwards at a comparison site.
    space,
  })
}

/// A buffer for ONE normalised output row, reserved FALLIBLY.
///
/// The sibling of [`zeroed_tensor`], and the reason that one was not the whole
/// class. `zeroed_tensor` covers the two PER-PREDICTION buffers;
/// [`normalise_row`] allocates once PER ROW, and it used to do it with a
/// `collect` into a `Box<[f32]>` — which for a `TrustedLen` iterator is
/// `Vec::with_capacity(row.len())` under the covers, so `handle_alloc_error`
/// and an abort when the allocator refuses.
///
/// The width is the MANIFEST's `dim`, the same number `elements.output` is
/// half of, so the same "fits `usize`, may not fit memory" gap applies; and
/// this one MULTIPLIES. Across a chunk the rows duplicate the whole output
/// tensor while the flat gather buffer and both native tensors are still live,
/// so the peak is `batch · dim` twice over — which is exactly the regime where
/// an artifact large enough to matter would abort a caller AFTER the fallibly
/// reserved flat buffer had succeeded.
///
/// Returns an EMPTY `Vec` with the capacity secured, rather than a filled one:
/// [`normalise_row`] then `extend`s it, which cannot allocate because
/// `Vec::reserve` is documented to do nothing when the capacity already
/// suffices.
fn embedding_buffer(elements: usize) -> Result<Vec<f32>> {
  let mut values: Vec<f32> = Vec::new();
  values.try_reserve_exact(elements).map_err(|_| {
    Error::AllocationFailed(AllocationFailed::new(PredictionTensor::Output, elements))
  })?;
  Ok(values)
}

/// The vector [`FaceEmbedder::embed`] returns, `faces` long, reserved FALLIBLY.
///
/// The last member of the class, and the one an enumeration left out: this
/// length is the CALLER's `faces.len()` rather than a number read off the
/// artifact, and "the caller supplied it" was taken as a reason to keep
/// `Vec::with_capacity` here. It is not one. `with_capacity` answers a refusal
/// by aborting the process, so the door still ended its caller's process under
/// memory pressure — at this site rather than at the tensor buffers already
/// converted, which is the same failure one step earlier. A caller's slice of
/// N faces is also not a bound on N embeddings: an [`AlignedFace`] is a
/// pointer and an optional transform, while a [`FaceEmbedding`] is a `Vec` and
/// a whole [`EmbeddingSpace`], so the result is several times the slice that
/// sized it before a single `dim`-wide row is allocated on top.
///
/// Returns an EMPTY `Vec` with the capacity secured, which is what lets
/// `FaceEmbedder::predict_chunk` append into it: `Vec::push` allocates only on
/// reaching the capacity, and the chunk lengths sum to exactly `faces`.
fn result_buffer(faces: usize) -> Result<Vec<FaceEmbedding>> {
  let mut out: Vec<FaceEmbedding> = Vec::new();
  out
    .try_reserve_exact(faces)
    .map_err(|_| Error::ResultAllocationFailed(ResultAllocationFailed::new(faces)))?;
  Ok(out)
}

#[cfg(test)]
mod tests;