coremlit 0.1.2

Safe, synchronous CoreML runtime for macOS (CPU/GPU/Neural Engine) with opt-in on-device multimodal pipelines: speech (Whisper STT, forced alignment, speaker diarization, Silero VAD), AudioSet sound-event tagging, and audio/text/image embeddings (CLAP, granite, SigLIP)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
// Not every integration-test binary that includes `mod common;` uses every
// helper below (`model_io.rs` uses only the model-path helpers; the parity
// suites use the audio/golden/metric helpers). Each test file is its own
// crate, so an unused helper is dead code *in that crate* — allow it here so
// the shared module compiles clean under the workspace's `-D warnings` gate.
#![allow(dead_code)]

// The workspace-root anchor every `models_dir()` below resolves against, and
// the sibling-checkout anchor the oracle gates read. FOUND by searching upward
// for the `[workspace]` manifest, never counted in `../` hops — see its module
// doc for why a count is the wrong shape here. Re-exported so the binaries
// that pull this `common` in share the one resolver.
// The committed per-table file manifests the artifact byte-pins are read from.
// Two revisions land in `Models/speakerkit/`, base then overlay, and each has
// its own manifest — see its module doc for the grammar.
#[path = "../../support/models_lock_manifest.rs"]
#[allow(dead_code)]
pub mod models_lock_manifest;

#[path = "../../support/workspace_root.rs"]
#[allow(dead_code)]
mod workspace_root;
#[allow(unused_imports)]
pub use workspace_root::{checkout_parent, models_root, workspace_root};

use std::{collections::BTreeMap, path::PathBuf};

use coremlit::audio::speaker::{
  extract::EXCLUDE_OVERLAP_MIN_FRAMES,
  segment::{POWERSET_CLASSES, SEG_CHUNK_SAMPLES, SEG_NUM_SLOTS, multilabel},
  window::DEFAULT_ONSET,
};
use sha2::{Digest, Sha256};

// The `coremlit_dir` hop that keeps every crate-relative fixture path below
// correct from BOTH packages that compile this shared module (`coremlit`'s own
// test binaries and `coremlit-parity`'s oracle binaries, which `#[path]`-include
// this very file). Kept in one place — see its module doc.
#[path = "../../support/coremlit_dir.rs"]
#[allow(dead_code)]
mod coremlit_dir;
use coremlit_dir::coremlit_path;

/// Directory containing the downloaded speakerkit model artifacts.
///
/// Overridable via `SPEAKERKIT_TEST_MODELS`; otherwise falls back to
/// `<workspace>/Models/speakerkit` — gitignored, fetched dev-time per the
/// design spec §4 (mirrors whisperkit's `WHISPERKIT_TEST_MODELS`/`Models/`
/// convention, one directory level down for this crate's own model set).
/// The `MODELS_LOCK` `local-dir` BOTH speaker tables stage into, with `Models/`
/// removed. Half of the key into `MODELS_LOCK.d/`; the revision below is the
/// other half, and it is what tells the two layers apart.
#[allow(dead_code)]
pub const VENDOR_DIR: &str = "speakerkit";

/// The OVERLAY table's revision (`FinDIT-Studio/speakerkit-coreml`) — the
/// issue-#15 fp16-guard-repaired re-conversions the pipeline ships.
#[allow(dead_code)]
pub const OVERLAY_LOCK_REVISION: &str = "3db69988bf2de12bab250614d6ac2b03d35132a2";

/// The BASE table's revision (`FluidInference/speaker-diarization-coreml`) —
/// the seven bundles nothing else publishes, plus the pre-repair copies of the
/// two the overlay overwrites.
#[allow(dead_code)]
pub const BASE_LOCK_REVISION: &str = "1ed7a662fdc7109e36d822db793ee6eebdaf8594";

/// Exact per-file SHA-256 of one bundle staged by the OVERLAY table, read from
/// `MODELS_LOCK.d/speakerkit@3db69988….sha256` — the committed copy of that
/// repository's own `CHECKSUMS.sha256`, verbatim.
#[allow(dead_code)]
pub fn overlay_sha256(bundle: &str) -> Vec<(String, String)> {
  models_lock_manifest::bundle_manifest(
    &workspace_root::workspace_root(),
    VENDOR_DIR,
    OVERLAY_LOCK_REVISION,
    bundle,
  )
}

/// Exact per-file SHA-256 of one bundle staged by the BASE table, read from
/// `MODELS_LOCK.d/speakerkit@1ed7a66….sha256`.
///
/// FluidInference's repository ships no `CHECKSUMS.sha256`, so that manifest is
/// coremlit's own `shasum -a 256` over the tree the selector stages at the
/// pinned revision — which is what these gates used to hold inline, written
/// once into a data file instead of three times into Rust source.
#[allow(dead_code)]
pub fn base_sha256(bundle: &str) -> Vec<(String, String)> {
  models_lock_manifest::bundle_manifest(
    &workspace_root::workspace_root(),
    VENDOR_DIR,
    BASE_LOCK_REVISION,
    bundle,
  )
}

pub fn models_dir() -> PathBuf {
  std::env::var_os("SPEAKERKIT_TEST_MODELS").map_or_else(
    || workspace_root::models_root().join("speakerkit"),
    PathBuf::from,
  )
}

/// Path to the decided segmentation artifact.
///
/// See `tests/model_io.rs`'s `// DECISION:` comment for the introspection
/// that picked `pyannote_segmentation.mlmodelc` over `Segmentation.mlmodelc`.
pub fn seg_path() -> PathBuf {
  models_dir().join("pyannote_segmentation.mlmodelc")
}

/// Directory containing the downloaded argmax `speakerkit-coreml` model
/// artifacts (`argmaxinc/speakerkit-coreml` on HuggingFace) — the SECOND
/// `ModelSource` this crate targets (Task 3's `ArgmaxSource`), acquired and
/// pinned independently of the FluidAudio-sourced artifacts [`models_dir`]
/// resolves.
///
/// Overridable via `ARGMAX_TEST_MODELS`; otherwise falls back to
/// `<workspace>/Models/argmax-speakerkit` — gitignored, fetched dev-time via
/// `hf download argmaxinc/speakerkit-coreml --local-dir
/// Models/argmax-speakerkit`. Sibling convention to `models_dir`'s
/// `SPEAKERKIT_TEST_MODELS`/`Models/speakerkit`: a distinct env var and a
/// distinct default directory, since Task 3 loads both sources side by side
/// and each needs its own independently overridable path. See
/// `tests/argmax_model_io.rs`'s module doc for the pinned revision and the
/// full artifact/SHA-256 table.
pub fn argmax_models_dir() -> PathBuf {
  std::env::var_os("ARGMAX_TEST_MODELS").map_or_else(
    || workspace_root::models_root().join("argmax-speakerkit"),
    PathBuf::from,
  )
}

/// Path to the RETIRED int8 embedding artifact (`wespeaker_v2.mlmodelc`, the
/// raw-waveform, in-graph-fbank WeSpeaker model).
///
/// This was the shipping artifact until issue #15 measured its palettization
/// silently collapsing 8-speaker audio; the shipping embedder is now the fp32
/// [`embed_fp32_path`] (see `tests/model_io.rs`'s DECISION). The int8 bytes
/// stay on disk and byte-pinned because the factorial and mechanism records
/// (`coremlit-parity`'s `tests/speaker/backend_factorial.rs`) run on them.
pub fn embed_path() -> PathBuf {
  models_dir().join("wespeaker_v2.mlmodelc")
}

/// Path to the **fp32 SHIPPING** embedding artifact, `wespeaker.mlmodelc`
/// (27 MB uncompressed float32 weights — `tests/model_io.rs`'s
/// `wespeaker_fp32_io_matches_spec`; the shipping selection since issue #15).
/// Contract-equal to the retired int8 `wespeaker_v2.mlmodelc`.
///
/// Gate 2 (embedding conversion fidelity, cosine ≥ 0.9999) is only
/// meaningful at MATCHED precision: dia-ort runs the fp32
/// `wespeaker_resnet34_lm.onnx` (26.7 MB float32), so the precision-matched
/// CoreML side is THIS fp32 artifact, not the int8 shipping one. The int8
/// path is measured separately for context (T3 recorded ~0.90-0.92 int8 vs
/// fp32, i.e. quantization cost, NOT a conversion defect).
pub fn embed_fp32_path() -> PathBuf {
  models_dir().join("wespeaker.mlmodelc")
}

/// A committed parity fixture: a short 16 kHz mono clip copied verbatim from
/// the `diarization` (dia) oracle repo's parity corpus, plus its provenance.
pub struct Fixture {
  /// Basename (no extension) of the committed WAV under `fixtures/audio/`
  /// and the golden JSON under `fixtures/golden/`.
  pub name: &'static str,
  /// Path within the dia repo this clip was copied from (provenance).
  pub source: &'static str,
  /// SHA-256 of the committed WAV, matching the dia-repo source byte-for-byte.
  pub sha256: &'static str,
  /// Human note on why this clip is in the set (coverage rationale).
  pub note: &'static str,
}

/// The parity fixture set (spec §6 Gates 1-2). Two short clips reused verbatim
/// from dia's `tests/parity/fixtures/*/clip_16k.wav`, chosen for ≤ 30 s length
/// (commit budget ~ `whisperkit/tests/fixtures/audio/ted_60.wav`) and to
/// exercise both the whole-window and the zero-padded final-chunk paths.
pub const FIXTURES: &[Fixture] = &[
  Fixture {
    name: "02_pyannote_sample",
    source: "diarization/tests/parity/fixtures/02_pyannote_sample/clip_16k.wav",
    sha256: "c319b4abca767b124e41432d364fd7df006cb26bb79d09326c487d606a134e6e",
    note: "pyannote's canonical 30.0 s sample → exactly 3 full 10 s chunks (no padding)",
  },
  Fixture {
    name: "07_yuhewei_dongbei_english",
    source: "diarization/tests/parity/fixtures/07_yuhewei_dongbei_english/clip_16k.wav",
    sha256: "096890ba8ffbaf10ca770c5373bf6c6664777f9421595c2cb7780af8cb2e46ff",
    note: "25.26 s clip → 2 full chunks + 1 partial (exercises final-chunk zero-padding)",
  },
];

/// The exact `seg_model` provenance string frozen into every committed golden
/// (`tests/speaker/fixtures/golden/*.json`) AND the single source of truth that
/// `tests/generate_goldens.rs` writes when it regenerates one. Pinned here so
/// the two can never silently drift apart; `tests/golden_metadata.rs` asserts
/// each committed golden still carries this exact string in the ordinary
/// `cargo test` suite.
///
/// The phrase **"raw powerset logits" is a legacy misnomer kept verbatim** to
/// match the committed oracle — the identical decision `tests/parity_seg.rs`
/// documents for the golden's `seg_logits` field name: renaming it would churn
/// every committed golden (each ~270 KB) for zero behavioral gain, since no
/// gate reads this string. The values are in fact powerset **log-probabilities**
/// (`pyannote_segmentation.mlmodelc`'s MIL ends `reduce_log_sum_exp` → `sub`,
/// see `coremlit::audio::speaker::segment`'s module doc; the committed ORT
/// golden agrees — every value `<= 0`, every 7-class row
/// `sum(exp(row)) == 1`). Read the string as a provenance tag, not a claim
/// about the tensor's calibration.
pub const SEG_MODEL_LABEL: &str =
  "segmentation-3.0.onnx (dia bundled, ort CPU EP, raw powerset logits)";

/// Directory holding the committed parity fixtures (`audio/` + `golden/`).
pub fn fixtures_dir() -> PathBuf {
  coremlit_path("tests/speaker/fixtures")
}

/// Committed WAV path for a fixture `name`.
pub fn audio_path(name: &str) -> PathBuf {
  fixtures_dir().join("audio").join(format!("{name}.wav"))
}

/// Committed golden-JSON path for a fixture `name`.
pub fn golden_path(name: &str) -> PathBuf {
  fixtures_dir().join("golden").join(format!("{name}.json"))
}

/// Loads a 16 kHz mono WAV as `f32` samples, the single source of truth for
/// both golden generation and parity replay so the two sides feed the models
/// byte-identical audio (the alignkit Gate-1 lesson: prove the inputs match).
///
/// 16-bit PCM is scaled by `1 / 32768`; float WAVs pass through. Asserts the
/// 16 kHz mono contract (pyannote/segmentation-3.0 and WeSpeaker are 16 kHz).
///
/// # Panics
/// If the file is missing, not 16 kHz mono, or an unsupported bit depth.
pub fn load_wav_16k_mono(path: &std::path::Path) -> Vec<f32> {
  let mut reader =
    hound::WavReader::open(path).unwrap_or_else(|e| panic!("open {}: {e}", path.display()));
  let spec = reader.spec();
  assert_eq!(spec.sample_rate, 16_000, "{}: not 16 kHz", path.display());
  assert_eq!(spec.channels, 1, "{}: not mono", path.display());
  match spec.sample_format {
    hound::SampleFormat::Int => {
      assert_eq!(
        spec.bits_per_sample,
        16,
        "{}: only 16-bit int PCM supported",
        path.display()
      );
      reader
        .samples::<i16>()
        .map(|s| f32::from(s.expect("read i16 sample")) / 32_768.0)
        .collect()
    }
    hound::SampleFormat::Float => reader
      .samples::<f32>()
      .map(|s| s.expect("read f32 sample"))
      .collect(),
  }
}

/// Splits `samples` into non-overlapping [`SEG_CHUNK_SAMPLES`]-sample windows,
/// zero-padding the final partial chunk to a full window.
///
/// This is the exact per-chunk contract both models require (dia's
/// `SegmentModel::infer` / `EmbedModel::embed_chunk_with_frame_mask` both take
/// exactly `WINDOW_SAMPLES = 160_000` samples and reject other lengths; dia's
/// offline pipeline zero-pads the short tail, `owned.rs:469-475`). It is a
/// deliberate simplification of dia's *overlapping* sliding-window GRID
/// (`step_samples = 16_000`, `crate::window`) down to `step = window`: the
/// grid geometry is a pipeline concern (speakerkit's `window`/`extract`
/// modules and their own parity), whereas Gates 1-2 isolate the MODELS on
/// identical per-chunk inputs — a 160 000-sample window is processed
/// bit-identically by each model regardless of how the grid spaces windows.
///
/// Used verbatim by BOTH `tests/generate_goldens.rs` (feeding dia-ort) and the
/// parity suites (feeding CoreML), so the two sides are input-identical by
/// construction; [`fnv1a_f32`] hashes recorded in the golden re-prove it at
/// replay time.
pub fn chunk_and_pad(samples: &[f32]) -> Vec<Vec<f32>> {
  let n = samples.len().div_ceil(SEG_CHUNK_SAMPLES).max(1);
  (0..n)
    .map(|c| {
      let start = c * SEG_CHUNK_SAMPLES;
      let mut chunk = vec![0.0f32; SEG_CHUNK_SAMPLES];
      if start < samples.len() {
        let end = (start + SEG_CHUNK_SAMPLES).min(samples.len());
        chunk[..end - start].copy_from_slice(&samples[start..end]);
      }
      chunk
    })
    .collect()
}

/// FNV-1a-64 over the little-endian bytes of `samples` — a stable,
/// platform-independent fingerprint of an exact `f32` buffer.
///
/// The input-match proof: golden generation records this hash of the precise
/// slice it hands to `ort::Session::run`; the parity suites recompute it on
/// the slice they hand to CoreML `predict` and assert equality, proving both
/// sides fed the model element-identical audio (a divergence on mismatched
/// inputs is a harness bug, not a model finding — the alignkit Gate-1 lesson).
pub fn fnv1a_f32(samples: &[f32]) -> u64 {
  let mut h: u64 = 0xcbf2_9ce4_8422_2325;
  for &s in samples {
    for b in s.to_le_bytes() {
      h ^= u64::from(b);
      h = h.wrapping_mul(0x0000_0100_0000_01b3);
    }
  }
  h
}

/// Lowercase 16-hex-digit rendering of a [`fnv1a_f32`] hash for JSON storage.
pub fn fnv_hex(h: u64) -> String {
  format!("{h:016x}")
}

/// SHA-256 (FIPS 180-4) of a byte slice via the `sha2` crate, rendered as a
/// lowercase 64-hex-digit digest.
///
/// Used by `tests/argmax_model_io.rs` to pin the exact bytes of the
/// downloaded argmax model artifacts. Previously hand-rolled here on the
/// belief that a dependency addition was out of scope for that task; it
/// wasn't (root `Cargo.toml`'s `[workspace.dependencies]` is this
/// workspace's one place every dependency is declared, and editing it for
/// this is in scope), and shipping a hand-rolled cryptographic primitive as
/// reusable test infrastructure was unnecessary risk/maintenance for no
/// benefit over the well-tested upstream crate. Unlike [`fnv1a_f32`] below —
/// a non-cryptographic fingerprint with no standard-crate equivalent this
/// concise, so it stays hand-rolled.
pub fn sha256_hex(data: &[u8]) -> String {
  Sha256::digest(data)
    .iter()
    .map(|byte| format!("{byte:02x}"))
    .collect()
}

/// `<bundle>.mlmodelc` name -> the sha256 MODELS_LOCK's `speaker`-kit OVERLAY table
/// (`FinDIT-Studio/speakerkit-coreml`, module doc "ORDER WITHIN THE `speaker` KIT") pins its
/// `model.mil` at, once staged last as the lock requires.
///
/// A THIRD copy of the two hashes `tests/speaker/model_io.rs`'s
/// `fp16_safe_segmentation_matches_pinned_sha256`/`fp16_safe_wespeaker_fp32_matches_pinned_sha256`
/// already duplicate from ci.yml's `overlay-pins` (module doc there: "a deliberate SECOND copy").
/// `ci_stages_the_speakerkit_overlay_last_and_proves_it_won` (tests/whisper/models_lock.rs) is
/// extended to hold all three copies together, the same way it already held the first two.
const OVERLAY_MODEL_MIL_PINS: &[(&str, &str)] = &[
  (
    "pyannote_segmentation.mlmodelc",
    "ded0d1ee11d77976b5c706ce667d0c8cb49977d3fe4367cccbd7b582bdb86dec",
  ),
  (
    "wespeaker.mlmodelc",
    "cff0cfe914078e9336754a9b38a68c2cdd88ca7b6bf97568ad551ab03ae1b666",
  ),
];

/// Reports and returns `true` when `gate` must not run because the staged `bundle` (at
/// `bundle_root`, i.e. [`seg_path`] or [`embed_fp32_path`]) does not hash to MODELS_LOCK's pinned
/// overlay `model.mil` ([`OVERLAY_MODEL_MIL_PINS`]).
///
/// Both the FluidInference base layer and the FinDIT-Studio overlay ship a contract-identical
/// `model.mil` under this same filename — same I/O shapes, same dtypes — so a numeric parity gate
/// loads and runs either one to completion (module doc, "ORDER WITHIN THE `speaker` KIT"). Only
/// the overlay's bytes carry the issue-#15 fp16-guard repair; the base layer's silently produce a
/// plausible-looking but wrong number (an inert `log(epsilon = 0)` / vanished pooling guards)
/// instead of the loud failure a missing or malformed file would give. A byte mismatch here means
/// this host's cache was never re-staged in MODELS_LOCK's lock order — re-download rather than
/// trust this run.
///
/// A missing file is NOT this function's concern: it returns `false` and lets the caller's own
/// `Model::from_file_with(...).expect(...)` a few lines down give the (already loud) "no such
/// file" panic — conflating "wrong bytes" with "no bytes" would blur two different repairs behind
/// one message.
///
/// Follows `tests/lid/common/mod.rs`'s `skipped_for_the_default_shape_refusal` convention: the
/// line goes to the inherited stderr descriptor, because libtest discards a PASSING test's output
/// unless the reader remembers `--nocapture`, and a skip nobody can see is the silent pass this
/// exists to prevent.
///
/// # Panics
/// If `bundle` is not one of [`OVERLAY_MODEL_MIL_PINS`]'s two names — a caller error, not a host
/// or artifact state this function is meant to report on.
#[allow(dead_code)]
pub fn skipped_for_stale_overlay(gate: &str, bundle_root: &std::path::Path, bundle: &str) -> bool {
  use std::io::Write;

  let expected = OVERLAY_MODEL_MIL_PINS
    .iter()
    .find_map(|(name, hash)| (*name == bundle).then_some(*hash))
    .unwrap_or_else(|| {
      panic!("skipped_for_stale_overlay: {bundle:?} is not a pinned overlay bundle")
    });
  let mil = bundle_root.join("model.mil");
  let Ok(bytes) = std::fs::read(&mil) else {
    return false;
  };
  let actual = sha256_hex(&bytes);
  if actual == expected {
    return false;
  }
  let line = format!(
    "model-gates | SKIPPED {gate}: {} sha256 {actual} != MODELS_LOCK overlay pin {expected} — \
     this host staged the FluidInference pre-repair build, not the fp16-guard-repaired \
     FinDIT-Studio overlay; re-run the download in MODELS_LOCK's lock order before trusting this \
     gate\n",
    mil.display()
  );
  // SAFETY: fd 2 is open for the whole life of the process (libtest redirects the Rust-level
  // handles, never the descriptor), it is only written to here, and `ManuallyDrop` keeps the
  // `File` from closing a descriptor it does not own.
  let mut fd2 = std::mem::ManuallyDrop::new(unsafe {
    <std::fs::File as std::os::fd::FromRawFd>::from_raw_fd(2)
  });
  let _ = fd2.write_all(line.as_bytes());
  true
}

/// Cosine similarity of two equal-length vectors, accumulated in `f64` for
/// precision (Gate 2's per-`(chunk, slot)` metric).
///
/// Rejects the two degenerate inputs that silently poison the metric rather
/// than returning a `NaN` a downstream fold would discard: a non-finite
/// element (which propagates `NaN` straight into the result) and a zero-norm
/// vector (`0 / 0 == NaN`). Gate 2 folds per-slot cosines with
/// `worst.min(cos)`, and `f64::min` KEEPS the non-`NaN` operand — so a
/// shape-compatible all-zero (or otherwise degenerate) embedder would leave
/// `worst == 1.0` and report PERFECT parity from garbage (M1). A loud panic
/// naming the offending vector turns that silent pass into a failure.
///
/// # Panics
/// If the lengths differ, either vector contains a non-finite element, or
/// either vector has a zero L2 norm.
pub fn cosine(a: &[f32], b: &[f32]) -> f64 {
  assert_eq!(a.len(), b.len(), "cosine: length mismatch");
  assert!(
    a.iter().all(|v| v.is_finite()),
    "cosine: vector `a` contains a non-finite element"
  );
  assert!(
    b.iter().all(|v| v.is_finite()),
    "cosine: vector `b` contains a non-finite element"
  );
  let (mut dot, mut na, mut nb) = (0.0f64, 0.0f64, 0.0f64);
  for (&x, &y) in a.iter().zip(b) {
    let (x, y) = (f64::from(x), f64::from(y));
    dot += x * y;
    na += x * x;
    nb += y * y;
  }
  assert!(na > 0.0, "cosine: vector `a` has zero norm");
  assert!(nb > 0.0, "cosine: vector `b` has zero norm");
  dot / (na.sqrt() * nb.sqrt())
}

/// Numerically stable softmax over one [`POWERSET_CLASSES`] logit row (the
/// same shape dia applies downstream, `segment::powerset::softmax_row`).
/// Diagnostic only: powerset segmentation's pipeline-relevant output is the
/// softmax probability (then onset/argmax), so softmax max-abs characterizes a
/// raw-logit divergence's actual downstream impact.
///
/// # Panics
/// If `row.len() != POWERSET_CLASSES`.
pub fn softmax_row(row: &[f32]) -> [f32; POWERSET_CLASSES] {
  assert_eq!(row.len(), POWERSET_CLASSES, "softmax_row: bad row length");
  let max = row.iter().copied().fold(f32::NEG_INFINITY, f32::max);
  let mut out = [0f32; POWERSET_CLASSES];
  let mut sum = 0f32;
  for (o, &l) in out.iter_mut().zip(row) {
    *o = (l - max).exp();
    sum += *o;
  }
  for o in &mut out {
    *o /= sum;
  }
  out
}

/// Maximum absolute per-element difference between two equal-length vectors,
/// in `f64` (Gate 1's segmentation-logit metric).
///
/// # Panics
/// If the lengths differ.
pub fn max_abs_diff(a: &[f32], b: &[f32]) -> f64 {
  assert_eq!(a.len(), b.len(), "max_abs_diff: length mismatch");
  a.iter()
    .zip(b)
    .map(|(&x, &y)| (f64::from(x) - f64::from(y)).abs())
    .fold(0.0, f64::max)
}

/// Max `|Σexp(row) − 1|` tolerated when validating a powerset log-softmax row.
/// The committed goldens sit at ≤ 2.3e-7 (an f32 `softmax → log` round-trip);
/// raw logits miss by many orders of magnitude, so this distinguishes the two
/// with a wide margin while never flaking on f32 rounding.
pub const SEG_ROW_SUM_EXP_TOL: f64 = 1e-4;

/// Validates one chunk's flattened `[num_frames * POWERSET_CLASSES]` powerset
/// segmentation output as LOG-PROBABILITIES: every element finite and `≤ 0`, and
/// each [`POWERSET_CLASSES`]-wide row normalized so `Σ exp = 1` (within
/// [`SEG_ROW_SUM_EXP_TOL`]).
///
/// dia-ort's segmentation graph ends `softmax → log` and the CoreML side emits
/// the same quantity through the fused `reduce_log_sum_exp → sub` tail; the
/// committed goldens store it under the legacy `seg_logits` name.
/// A future model emitting RAW logits (positive values, rows that do not sum-exp
/// to 1) with the argmax ORDERING preserved would decode to the same speakers yet
/// break this invariant — which `generate_goldens.rs`'s prose used to only
/// assert, never check (codex r7 F4). This runs BOTH in the generator before
/// serialization and against the committed goldens in the ordinary suite, so that
/// drift cannot land silently.
///
/// # Errors
/// Returns the first violation as a message: a length not matching
/// `num_frames * POWERSET_CLASSES`, a non-finite or positive element, or a row
/// whose `Σ exp` departs from 1 by more than [`SEG_ROW_SUM_EXP_TOL`].
pub fn check_seg_log_probs(seg_logits: &[f32], num_frames: usize) -> Result<(), String> {
  let expected = num_frames * POWERSET_CLASSES;
  if seg_logits.len() != expected {
    return Err(format!(
      "seg_logits length {} != num_frames*POWERSET_CLASSES ({num_frames}*{POWERSET_CLASSES}={expected})",
      seg_logits.len()
    ));
  }
  for (f, row) in seg_logits
    .as_chunks::<POWERSET_CLASSES>()
    .0
    .iter()
    .enumerate()
  {
    let mut sum_exp = 0.0_f64;
    for (k, &v) in row.iter().enumerate() {
      if !v.is_finite() {
        return Err(format!("frame {f} class {k}: non-finite log-prob {v}"));
      }
      if v > 0.0 {
        return Err(format!(
          "frame {f} class {k}: log-prob {v} > 0 — a probability's log is ≤ 0; this looks like a \
           raw logit"
        ));
      }
      sum_exp += f64::from(v).exp();
    }
    let dev = (sum_exp - 1.0).abs();
    if dev > SEG_ROW_SUM_EXP_TOL {
      return Err(format!(
        "frame {f}: Σexp(row) = {sum_exp:.9}, off 1.0 by {dev:.3e} (> {SEG_ROW_SUM_EXP_TOL:.0e}) — \
         the row is not a normalized log-softmax (raw logits, or a broken normalization)"
      ));
    }
  }
  Ok(())
}

/// Hard argmax over one frame's [`POWERSET_CLASSES`] logits, ties toward the
/// lowest index (`>` seeded at class 0) — the exact rule speakerkit's shipping
/// `segment::multilabel` and dia's `powerset_to_speakers_hard` both use. Gate
/// 1's multilabel-flip check argmaxes both models' logits with this and counts
/// per-frame disagreements.
///
/// # Panics
/// If `row.len() != POWERSET_CLASSES`.
pub fn powerset_argmax(row: &[f32]) -> usize {
  assert_eq!(
    row.len(),
    POWERSET_CLASSES,
    "powerset_argmax: bad row length"
  );
  let mut argmax = 0usize;
  let mut max = row[0];
  for (k, &v) in row.iter().enumerate().skip(1) {
    if v > max {
      max = v;
      argmax = k;
    }
  }
  argmax
}

/// Encodes a per-frame boolean mask as a compact `'0'`/`'1'` string for JSON.
pub fn mask_to_string(mask: &[bool]) -> String {
  mask.iter().map(|&b| if b { '1' } else { '0' }).collect()
}

/// Strictly decodes a [`mask_to_string`] bit string into `Vec<bool>`,
/// hard-failing unless it is EXACTLY `expected_len` characters and every
/// character is `'0'` or `'1'`. `context` names the offending subject (e.g.
/// `"<fixture>: chunk C slot S: mask"`) for the panic.
///
/// This is the SINGLE strict decoder for both committed golden-mask families —
/// the embedding golden's per-frame pooling `mask` ([`load_golden`]) and the
/// argmax-Swift golden's `activeFrames` (`parity_argmax_swift::parse_active_frames`).
/// The old lenient decode (`c == '1'`, any length, every non-`'1'` char →
/// `false`) is what let a gate pass vacuously: a truncated, empty, or junk mask
/// decoded to a short or all-`false` `Vec<bool>` while the comparison still
/// REPORTED full coverage — and an OVER-long mask was silently truncated back by
/// the model's frame padding (`embed::repeat_pad_f32`), producing byte-identical
/// model input. A wrong length or an unexpected character is a MALFORMED golden,
/// never a slot with fewer active frames, so it is rejected here before a single
/// fidelity number is read.
///
/// # Panics
/// If any character is not `'0'`/`'1'`, or the decoded length is not `expected_len`.
pub fn parse_bit_mask(context: &str, expected_len: usize, raw: &str) -> Vec<bool> {
  let bits: Vec<bool> = raw
    .chars()
    .map(|c| match c {
      '0' => false,
      '1' => true,
      other => panic!(
        "{context} contains {other:?} — a golden bit mask is hard-binary, so only '0'/'1' are \
         valid; an unknown character is a malformed golden, not an inactive frame."
      ),
    })
    .collect();
  assert_eq!(
    bits.len(),
    expected_len,
    "{context} has {} characters, expected exactly {expected_len}. A short, empty, or over-long \
     mask makes the comparison vacuous (or is truncated back by the model's frame padding) while \
     the gate still reports full coverage.",
    bits.len()
  );
  bits
}

/// One embedded speaker slot within a golden chunk: the per-frame mask fed to
/// the embedding model and the resulting dia-ort reference embedding.
pub struct GoldenSlot {
  /// Speaker-slot index (0..[`coremlit::audio::speaker::segment::SEG_NUM_SLOTS`]).
  pub slot: usize,
  /// The per-frame pooling mask (length = segmentation frame count) fed to
  /// BOTH backends — stored verbatim so the parity side is mask-identical.
  pub mask: Vec<bool>,
  /// dia-ort's raw (un-normalized) 256-d WeSpeaker embedding for this slot.
  pub embedding: Vec<f32>,
}

/// One chunk's dia-ort reference outputs plus its input fingerprint.
pub struct GoldenChunk {
  /// Sample count of the (zero-padded) chunk fed to the models.
  pub input_len: usize,
  /// [`fnv1a_f32`] of the exact chunk samples dia-ort was run on.
  pub input_fnv1a: u64,
  /// dia-ort's flattened `[num_frames * POWERSET_CLASSES]` powerset segmentation
  /// LOG-PROBABILITIES (frame-major; its MIL ends `softmax → log`, kept under the
  /// legacy `seg_logits` name) — the Gate 1 reference. Validated as log-probs by
  /// [`check_seg_log_probs`].
  pub seg_logits: Vec<f32>,
  /// The embedded slots for this chunk (skipped/degenerate slots omitted).
  pub slots: Vec<GoldenSlot>,
}

/// A parity golden: dia-ort reference tensors for one fixture, produced by
/// `tests/generate_goldens.rs` and committed as the pinned oracle.
pub struct Golden {
  /// Fixture name (matches [`Fixture::name`]).
  pub fixture: String,
  /// Number of non-overlapping chunks (`== chunks.len()`).
  pub num_chunks: usize,
  /// Segmentation frames per chunk (589 for pyannote/segmentation-3.0).
  pub num_frames: usize,
  /// Per-chunk reference outputs.
  pub chunks: Vec<GoldenChunk>,
}

/// Loads and parses a committed golden JSON for fixture `name`.
///
/// # Panics
/// If the file is missing or malformed — a committed golden is a hard
/// dependency of the parity suites, not an optional input.
pub fn load_golden(name: &str) -> Golden {
  let path = golden_path(name);
  let bytes =
    std::fs::read(&path).unwrap_or_else(|e| panic!("read golden {}: {e}", path.display()));
  let v: serde_json::Value =
    serde_json::from_slice(&bytes).unwrap_or_else(|e| panic!("parse golden {name}: {e}"));

  let num_frames = v["num_frames"].as_u64().expect("num_frames") as usize;
  let chunks = v["chunks"]
    .as_array()
    .expect("chunks array")
    .iter()
    .enumerate()
    .map(|(c_idx, c)| {
      let seg_logits = c["seg_logits"]
        .as_array()
        .expect("seg_logits array")
        .iter()
        .map(|x| x.as_f64().expect("logit f64") as f32)
        .collect();
      let slots = c["slots"]
        .as_array()
        .expect("slots array")
        .iter()
        .map(|s| {
          let slot = s["slot"].as_u64().expect("slot") as usize;
          GoldenSlot {
            slot,
            // Strict decode: exactly `num_frames` chars, `{'0','1'}` only. A
            // malformed mask is a hard error, not a lenient reinterpretation.
            mask: parse_bit_mask(
              &format!("{name}: chunk {c_idx} slot {slot}: mask"),
              num_frames,
              s["mask"].as_str().expect("mask string"),
            ),
            embedding: s["embedding"]
              .as_array()
              .expect("embedding array")
              .iter()
              .map(|x| x.as_f64().expect("embed f64") as f32)
              .collect(),
          }
        })
        .collect();
      let hex = c["input_fnv1a"].as_str().expect("input_fnv1a hex");
      GoldenChunk {
        input_len: c["input_len"].as_u64().expect("input_len") as usize,
        input_fnv1a: u64::from_str_radix(hex, 16).expect("parse fnv hex"),
        seg_logits,
        slots,
      }
    })
    .collect();

  Golden {
    fixture: v["fixture"].as_str().expect("fixture").to_string(),
    num_chunks: v["num_chunks"].as_u64().expect("num_chunks") as usize,
    num_frames,
    chunks,
  }
}

/// Independently re-derives one chunk's expected per-slot embedding masks from
/// its committed powerset segmentation log-probabilities, reproducing the golden
/// generator's rule (`generate_goldens::derive_slot_masks`) WITHOUT reading the
/// golden's stored slot list. `None` for a slot = it has no active frame, so the
/// generator emitted no golden slot for it.
///
/// The decode is speakerkit's shipping [`multilabel`] (direct argmax over the
/// log-probs). The generator decodes via dia's `softmax`-then-argmax; the two
/// coincide on every committed golden row (pinned by
/// `parity_seg::golden_direct_and_dia_decode_agree`), so this reproduces the
/// generator's exact roster and masks on committed data while depending on
/// neither dia nor the stored slots. The overlap-exclusion below is the same
/// `embedding_exclude_overlap` port speakerkit's `extract::derive_slot_plans`
/// runs, keyed on the SAME production [`DEFAULT_ONSET`] and
/// [`EXCLUDE_OVERLAP_MIN_FRAMES`] (a slot is active at a frame iff its hard 0/1
/// value is `>= DEFAULT_ONSET`; the overlap-excluded mask falls back to raw when
/// `<= EXCLUDE_OVERLAP_MIN_FRAMES` clean frames remain).
///
/// # Panics
/// If `seg_logits.len() != num_frames * POWERSET_CLASSES` (via [`multilabel`]).
pub fn derive_expected_slot_masks(
  seg_logits: &[f32],
  num_frames: usize,
) -> [Option<Vec<bool>>; SEG_NUM_SLOTS] {
  let slab = multilabel(seg_logits, num_frames);
  let onset = f64::from(DEFAULT_ONSET);

  // Per-frame "clean" indicator: fewer than 2 of the slots active (dia's
  // `owned.rs:536-549`). Computed once over all slots, before the per-slot loop.
  let mut clean = vec![false; num_frames];
  for (f, clean_f) in clean.iter_mut().enumerate() {
    let active = (0..SEG_NUM_SLOTS)
      .filter(|&s| slab[f * SEG_NUM_SLOTS + s] >= onset)
      .count();
    *clean_f = active < 2;
  }

  core::array::from_fn(|s| {
    let mut frame_mask = vec![false; num_frames];
    let mut any = false;
    for (f, m) in frame_mask.iter_mut().enumerate() {
      *m = slab[f * SEG_NUM_SLOTS + s] >= onset;
      any |= *m;
    }
    if !any {
      return None; // dia drops a slot with no active frame (owned.rs:561-571).
    }
    // Overlap-excluded mask = raw AND clean; fall back to raw when too few clean
    // frames remain (`<=`, owned.rs:573-591).
    let mut used = vec![false; num_frames];
    let mut clean_count = 0usize;
    for (f, u) in used.iter_mut().enumerate() {
      *u = frame_mask[f] && clean[f];
      if *u {
        clean_count += 1;
      }
    }
    if clean_count <= EXCLUDE_OVERLAP_MIN_FRAMES {
      used = frame_mask;
    }
    Some(used)
  })
}

/// Asserts a golden's stored `(chunk, slot)` roster and every per-frame mask
/// EXACTLY match the roster independently re-derived from its committed
/// `seg_logits` (via [`derive_expected_slot_masks`]), spanning EVERY chunk, and
/// returns the total number of `(chunk, slot)` slots in that roster — the count
/// Gate 2 must fold in.
///
/// This closes the embedding half of the golden-loader leniency class. The Gate
/// 2 replay used to compare only the slots the golden happened to carry, gated
/// on a merely global `n > 0`, so deleting one fixture's slots (or an entire
/// fixture) left the other fixture's count non-zero and stayed green. Here the
/// expected roster is derived from a DIFFERENT part of the golden (the
/// segmentation tensor) than the part under test (the slot list), so a missing
/// slot, an extra slot, a moved mask bit, a duplicate slot, or a dropped fixture
/// all fail — per chunk, named.
///
/// # Panics
/// On any roster/mask divergence, or a duplicate slot within a chunk.
pub fn assert_golden_roster(golden: &Golden) -> usize {
  let mut total = 0usize;
  for (c_idx, chunk) in golden.chunks.iter().enumerate() {
    let derived = derive_expected_slot_masks(&chunk.seg_logits, golden.num_frames);
    let expected: BTreeMap<usize, &Vec<bool>> = derived
      .iter()
      .enumerate()
      .filter_map(|(s, m)| m.as_ref().map(|mask| (s, mask)))
      .collect();

    let mut stored: BTreeMap<usize, &Vec<bool>> = BTreeMap::new();
    for slot in &chunk.slots {
      assert!(
        stored.insert(slot.slot, &slot.mask).is_none(),
        "{}: chunk {c_idx}: slot {} appears more than once in the golden",
        golden.fixture,
        slot.slot
      );
    }

    let stored_roster: Vec<usize> = stored.keys().copied().collect();
    let expected_roster: Vec<usize> = expected.keys().copied().collect();
    assert_eq!(
      stored_roster, expected_roster,
      "{}: chunk {c_idx}: stored (chunk, slot) roster {stored_roster:?} != roster \
       {expected_roster:?} independently derived from seg_logits — a golden slot was added or \
       dropped",
      golden.fixture
    );

    for (s, exp_mask) in &expected {
      let got = stored
        .get(s)
        .expect("roster equality checked directly above");
      assert_eq!(
        got,
        exp_mask,
        "{}: chunk {c_idx} slot {s}: stored mask ({} active frame(s)) != mask independently \
         derived from seg_logits ({} active frame(s))",
        golden.fixture,
        got.iter().filter(|&&b| b).count(),
        exp_mask.iter().filter(|&&b| b).count()
      );
    }
    total += expected.len();
  }
  total
}

// ── Host-class provenance for the committed-Swift-golden parity gate ────────
//
// The gate stamps each golden with the host-class it was generated on and
// enforces its tight bound only against a matching host, because CoreML floats
// are not contracted portable across macOS builds or chips (#36). The
// predicate, the verdict enum and the two diagnosis strings live in ONE file
// under `tests/support/` — the `coremlit_dir` convention — because the speaker
// and vad `common/mod.rs` used to carry byte-identical copies of them and
// whisper needed a third. Re-exported here so `common::HostClass` and friends
// keep their existing spelling; this suite's hermetic tests drive the shared
// copy.
#[path = "../../support/host_class.rs"]
#[allow(dead_code)]
mod host_class;
// Named rather than glob-re-exported so the shared surface is legible here, and
// `allow`ed for the same reason this file's top-level `allow(dead_code)` exists:
// every test binary that says `mod common;` compiles this whole module, and the
// ones with no golden to host-gate reference none of these four.
#[allow(unused_imports)]
pub use host_class::{HostClass, HostVerdict, RecordedHost, check_host_class, legacy_failure_note};

// ── Model-gate visibility (#61) ─────────────────────────────────────────────
//
// NOT `#[ignore]`d, deliberately. This is the ordinary-run half of the gate
// accounting: an ignored-ONLY run (`-- --ignored`, what every CI gate uses)
// never selects it, and it never appears in an ignored-only `--list`, so the
// anti-vacuum counts those gates take are unchanged. What it adds is the case
// no gate covers — a plain, modelless run — where the skipped gates otherwise
// say nothing but `ignored`. Mechanism, and what it does and does not refuse,
// in the shared module.
#[path = "../../support/model_gate_report.rs"]
mod model_gate_report;

/// Reports how many of this binary's tests are `#[ignore]`d speakerkit model gates
/// that did not run, and whether the models root they read is on disk.
#[test]
fn model_gate_report() {
  model_gate_report::report(&[
    ("SPEAKERKIT_TEST_MODELS", models_dir()),
    ("ARGMAX_TEST_MODELS", argmax_models_dir()),
  ]);
}