coremlit 0.1.1

Safe, synchronous CoreML runtime for macOS (CPU/GPU/Neural Engine) with opt-in on-device multimodal pipelines: speech (Whisper STT, forced alignment, speaker diarization, Silero VAD), AudioSet sound-event tagging, and audio/text/image embeddings (CLAP, granite, SigLIP)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
1451
1452
1453
1454
1455
1456
1457
1458
1459
1460
1461
1462
1463
1464
1465
1466
1467
1468
1469
1470
1471
1472
1473
1474
1475
1476
1477
1478
1479
1480
1481
1482
1483
1484
1485
1486
1487
1488
1489
1490
1491
1492
1493
1494
1495
1496
1497
1498
1499
1500
1501
1502
1503
1504
1505
1506
1507
1508
1509
1510
1511
//! Seek-point and segment slicing: turns one decoded window's
//! [`DecodingResult`] into zero or more [`TranscriptionSegment`]s and the
//! sample offset the next window should start decoding from. Ports
//! `SegmentSeeker.findSeekPointAndSegments`
//! (`argmax-oss-swift/Sources/WhisperKit/Core/Text/SegmentSeeker.swift:41-189`).
//!
//! Swift threads `timeToken`/`specialToken`/`sampleRate` in as three
//! separate parameters (`SegmentSeeker.swift:47-49`); this port reads the
//! first two off `tokenizer.special_tokens()` (already a parameter here)
//! and the third off [`crate::audio::whisper::constants::SAMPLE_RATE`], collapsing three
//! parameters into the one `tokenizer` this module already needs for
//! decoding slice text.
//!
//! Also home to the word-timestamp core: [`dynamic_time_warping`] (ports
//! `dynamicTimeWarping(withMatrix:)`, `SegmentSeeker.swift:195-278`,
//! tie-breaking in the private `min_cost_and_trace`, `:239-251`),
//! [`find_alignment`] (`:340-408`), [`merge_punctuations`] (`:280-338`),
//! [`calculate_word_duration_constraints`] and
//! [`truncate_long_words_at_sentence_boundaries`] (`:498-526`),
//! [`update_segments_with_word_timings`] (`:528-659`) — the final
//! re-anchoring step that walks a word-index cursor shared across every
//! segment, applies the short-word pull-back and pause/boundary
//! heuristics, and writes the results back onto [`TranscriptionSegment`]s
//! — and [`add_word_timestamps`] (`:410-496`), the orchestration wrapper
//! that flattens each segment's tokens/log-probs into the flat index list
//! used to filter the raw CoreML alignment-weights array, then threads
//! that through [`find_alignment`] -> the duration/truncation hack ->
//! [`merge_punctuations`] -> [`update_segments_with_word_timings`] in
//! sequence. [`crate::audio::whisper::transcribe::TranscribeTask::run`]'s window loop
//! calls `add_word_timestamps` directly (`TranscribeTask.swift:196-233`)
//! when `options.word_timestamps()` is set.

use unicode_categories::UnicodeCategories;

use crate::{
  MultiArray,
  audio::whisper::{
    backend::{AlignmentMatrix, AlignmentView},
    constants::{SAMPLE_RATE, SECONDS_PER_TIME_TOKEN},
    error::{
      AlignmentPitchUnavailable, AlignmentPitchUnexpectedLayout, InvalidAlignmentShape,
      SegmentError,
    },
    options::{AlignmentGather, DecodingOptions, WordGrouping},
    result::{DecodingResult, TranscriptionSegment, WordTiming},
    tokenizer::WhisperTokenizer,
  },
};

/// The seconds span of one decode window's own audio: `samples` unpadded
/// samples starting `seek` samples into the run.
///
/// ONE rounding contract, shared by [`find_seek_point_and_segments`]'s segment
/// timing and by the language observation
/// [`crate::audio::whisper::transcribe::TranscribeTask::run`] records for the
/// same window, so the two can never disagree about the window they both
/// describe (coremlit issue #107).
///
/// The start is `seek` converted once; the end is that start PLUS the
/// duration, never the sum `seek + samples` converted as a whole. Above 2^24
/// samples (~17.5 min) adjacent sample indices share one f32, so converting
/// the sum collapses a short final window onto its own start — a one-sample
/// window at 1050 s reports `1050.0 .. 1050.0` although the decoder saw audio.
///
/// A nonempty window always gets `end > start`. From 2048 s on, one f32 ulp
/// of the start exceeds a one-sample duration and even the addition absorbs
/// it, so the end is nudged to the next representable value instead. Swift's
/// `SegmentSeeker` computes the addition without that guard and reports the
/// zero-length span (documented deviation, in a regime where the window did
/// consume audio and Swift's own `:217-218` filter deletes what it produced).
pub(crate) fn window_span(seek: usize, samples: usize) -> (f32, f32) {
  let start = seek as f32 / SAMPLE_RATE as f32;
  let end = start + samples as f32 / SAMPLE_RATE as f32;
  if samples > 0 && end <= start {
    return (start, start.next_up());
  }
  (start, end)
}

/// The same span re-anchored `offset_seconds` later — the ONE re-anchoring
/// contract, shared by every span
/// [`crate::audio::whisper::audio::chunker::apply_result_seek_offset`] lifts
/// out of a VAD chunk's own timeline into the original one.
///
/// [`window_span`] guarantees a window that held audio reports `end > start`;
/// adding the offset to each endpoint on its own throws that guarantee away
/// again, because the two additions round independently. From 2048 s on, one
/// f32 ulp of the shifted start (2.44e-4 s) exceeds a short window's whole
/// duration, so a valid local span `0.0 .. 1/16000` shifted by 32_768_000
/// samples lands on `2048.0 .. 2048.0` — a kept probe, or a lump segment,
/// whose evidence is zero-width although the decoder saw audio. This helper
/// preserves the guarantee across the shift: a span that came in nonempty
/// leaves nonempty, its end nudged to the next representable value above the
/// shifted start rather than onto it. Observations and segments both re-anchor
/// through it, so the two can no more disagree about the window they describe
/// after the shift than [`window_span`] lets them before it.
///
/// A span that came in EMPTY (`end == start`) is left exactly as shifted, with
/// no nudge. The segment path can legitimately hold one — a zero-length
/// segment survives until
/// [`crate::audio::whisper::transcribe::TranscribeTask::run`]'s `:217-218`
/// filter drops it — and inventing an extent here would hide it from that
/// filter.
pub(crate) fn shift_span(start: f32, end: f32, offset_seconds: f32) -> (f32, f32) {
  let shifted_start = start + offset_seconds;
  let shifted_end = end + offset_seconds;
  if end > start && shifted_end <= shifted_start {
    return (shifted_start, shifted_start.next_up());
  }
  (shifted_start, shifted_end)
}

/// Turns `decoding` — the just-decoded window starting at `current_seek`
/// samples — into the next seek offset and, unless the window was silent,
/// the [`TranscriptionSegment`]s it contains. `all_segments_count` seeds
/// each new segment's [`TranscriptionSegment::id`] so ids stay unique
/// across every window a caller has already processed; `segment_size` is
/// the window's length in samples (normally
/// [`crate::audio::whisper::constants::WINDOW_SAMPLES`], smaller for a final short
/// window).
///
/// Three phases, ported structure-preserving from `findSeekPointAndSegments`:
///
/// 1. **Silence skip** (`SegmentSeeker.swift:57-74`): if
///    `options.no_speech_threshold()` is set and `decoding.no_speech_prob()`
///    exceeds it, the whole window is dropped — seek advances by
///    `segment_size` and `None` is returned — *unless*
///    `options.logprob_threshold()` is also set and `decoding.avg_logprob()`
///    exceeds *that*, which overrides the skip (confident text beats a
///    high no-speech probability).
/// 2. **Consecutive-timestamp slicing** (`:79-148`): otherwise, adjacent
///    timestamp-token pairs in `decoding.tokens_slice()` mark segment
///    boundaries. A lone trailing timestamp (single-timestamp ending) or a
///    trailing run of plain tokens (no-timestamp ending) each contribute
///    one final boundary of their own. Each resulting slice becomes a
///    segment spanning its first-to-last timestamp token; seek advances to
///    the last timestamp found (or by `segment_size` on a no-timestamp
///    ending).
/// 3. **Lump fallback** (`:149-186`): if no consecutive timestamp pair
///    exists at all, the whole window becomes one segment, its end time
///    refined by the last timestamp token above `<|0.00|>` if any exists;
///    seek always advances by `segment_size` on this path.
///
/// # Errors
/// [`SegmentError::Tokenizer`] if decoding a slice's tokens back to text
/// fails.
pub fn find_seek_point_and_segments(
  decoding: &DecodingResult,
  options: &DecodingOptions,
  all_segments_count: usize,
  current_seek: usize,
  segment_size: usize,
  tokenizer: &WhisperTokenizer,
) -> Result<(usize, Option<Vec<TranscriptionSegment>>), SegmentError> {
  let special = tokenizer.special_tokens();
  let time_token = special.time_token_begin();
  let special_token_begin = special.special_token_begin();
  let mut seek = current_seek;
  let (time_offset, window_end) = window_span(current_seek, segment_size);

  // :57-74 — silence skip: no-speech probability above threshold skips the
  // window entirely, unless overridden by high average confidence.
  if let Some(threshold) = options.no_speech_threshold() {
    let mut should_skip = decoding.no_speech_prob() > threshold;
    if let Some(logprob_threshold) = options.logprob_threshold()
      && decoding.avg_logprob() > logprob_threshold
    {
      should_skip = false;
    }
    if should_skip {
      return Ok((seek + segment_size, None));
    }
  }

  let current_tokens = decoding.tokens_slice();
  let current_log_probs = decoding.token_log_probs_slice();
  let is_timestamp_token: Vec<bool> = current_tokens.iter().map(|&t| t >= time_token).collect();

  // :84-86 — the ending shape decides whether/how a trailing boundary is
  // synthesized below. A slice with fewer than 3 tokens can match neither
  // pattern, exactly like Swift's `Array == [Bool]` on a short `suffix`.
  let single_timestamp_ending = matches!(is_timestamp_token.as_slice(), [.., false, true, false]);
  let no_timestamp_ending = matches!(is_timestamp_token.as_slice(), [.., false, false, false]);

  // :88-97 — end index of every consecutive timestamp-token pair.
  let mut slice_indexes: Vec<usize> = Vec::new();
  let mut previous_is_timestamp = false;
  for (index, &is_timestamp) in is_timestamp_token.iter().enumerate() {
    if previous_is_timestamp && is_timestamp {
      slice_indexes.push(index);
    }
    previous_is_timestamp = is_timestamp;
  }

  let mut segments: Vec<TranscriptionSegment> = Vec::new();

  if slice_indexes.is_empty() {
    // :149-186 — no consecutive timestamps anywhere: lump the whole window
    // into one segment.
    // The window's own span end, unless a timestamp token below refines it.
    let mut segment_end = window_end;
    let timestamp_tokens: Vec<u32> = current_tokens
      .iter()
      .copied()
      .filter(|&t| t > time_token)
      .collect();
    if let Some(&last_timestamp) = timestamp_tokens.last() {
      segment_end = time_offset + (last_timestamp - time_token) as f32 * SECONDS_PER_TIME_TOKEN;
    }

    let word_tokens: Vec<u32> = current_tokens
      .iter()
      .copied()
      .filter(|&t| t < special_token_begin)
      .collect();
    let segment_text_tokens: &[u32] = if options.skip_special_tokens() {
      &word_tokens
    } else {
      current_tokens
    };
    let segment_text = tokenizer.decode(segment_text_tokens, false)?;

    segments.push(
      TranscriptionSegment::new()
        .with_id(all_segments_count + segments.len())
        .with_seek(seek)
        .with_start(time_offset)
        .with_end(segment_end)
        .with_text(segment_text)
        .with_tokens(current_tokens)
        .with_token_log_probs(current_log_probs)
        .with_temperature(decoding.temperature())
        .with_avg_logprob(decoding.avg_logprob())
        .with_compression_ratio(decoding.compression_ratio())
        .with_no_speech_prob(decoding.no_speech_prob()),
    );

    // Model gave no consecutive-timestamp boundary, so the whole window is
    // consumed regardless of the refined end above (Swift's own
    // upstream TODO at `:184-185` notes the more accurate
    // `durationSeconds`-based advance is not yet used either).
    seek += segment_size;
  } else {
    // :101-107 — a lone trailing timestamp or trailing run of plain tokens
    // each need one more boundary appended beyond what the main loop above
    // found, to cover the window's tail as a final slice.
    if single_timestamp_ending {
      let single_ending_index = is_timestamp_token
        .iter()
        .rposition(|&t| t)
        .expect("single_timestamp_ending's pattern requires a `true` entry");
      slice_indexes.push(single_ending_index + 1);
    } else if no_timestamp_ending {
      slice_indexes.push(current_tokens.len());
    }

    let mut last_slice_start = 0usize;
    for &current_slice_end in &slice_indexes {
      let sliced_tokens = &current_tokens[last_slice_start..current_slice_end];
      let sliced_log_probs = &current_log_probs[last_slice_start..current_slice_end];

      // Every slice here is bounded by a detected timestamp-pair boundary
      // (this loop's own start) or ends at one (the main loop above only
      // ever records the second index of a `true, true` pair), so it
      // always contains at least one timestamp token — the same invariant
      // Swift trusts with `timestampTokens.first!`/`.last!`.
      let timestamp_tokens: Vec<u32> = sliced_tokens
        .iter()
        .copied()
        .filter(|&t| t >= time_token)
        .collect();
      let start_ts = *timestamp_tokens
        .first()
        .expect("slice bounded by a timestamp pair contains a timestamp token");
      let end_ts = *timestamp_tokens
        .last()
        .expect("slice bounded by a timestamp pair contains a timestamp token");
      let start_seconds = (start_ts - time_token) as f32 * SECONDS_PER_TIME_TOKEN;
      let end_seconds = (end_ts - time_token) as f32 * SECONDS_PER_TIME_TOKEN;

      let word_tokens: Vec<u32> = sliced_tokens
        .iter()
        .copied()
        .filter(|&t| t < special_token_begin)
        .collect();
      let sliced_text_tokens: &[u32] = if options.skip_special_tokens() {
        &word_tokens
      } else {
        sliced_tokens
      };
      let slice_text = tokenizer.decode(sliced_text_tokens, false)?;

      segments.push(
        TranscriptionSegment::new()
          .with_id(all_segments_count + segments.len())
          .with_seek(seek)
          .with_start(time_offset + start_seconds)
          .with_end(time_offset + end_seconds)
          .with_text(slice_text)
          .with_tokens(sliced_tokens)
          .with_token_log_probs(sliced_log_probs)
          .with_temperature(decoding.temperature())
          .with_avg_logprob(decoding.avg_logprob())
          .with_compression_ratio(decoding.compression_ratio())
          .with_no_speech_prob(decoding.no_speech_prob()),
      );

      last_slice_start = current_slice_end;
    }

    // :140-148 — seek to the last timestamp found, unless the tail was an
    // unbounded run of plain tokens (no-timestamp ending), which instead
    // consumes the full window like the lump branch does.
    if no_timestamp_ending {
      seek += segment_size;
    } else {
      let last_index = last_slice_start - usize::from(single_timestamp_ending);
      let last_timestamp_token = current_tokens[last_index] - time_token;
      let last_timestamp_seconds = last_timestamp_token as f32 * SECONDS_PER_TIME_TOKEN;
      seek += (last_timestamp_seconds * SAMPLE_RATE as f32) as usize;
    }
  }

  Ok((seek, Some(segments)))
}

// ---------------------------------------------------------------------
// Word timestamps: DTW alignment, merge_punctuations, find_alignment
// ---------------------------------------------------------------------

/// One decoded-token/audio-frame alignment path out of
/// [`dynamic_time_warping`]'s cost-matrix backtrace: parallel
/// `text_indices`/`time_indices` sequences of equal length, walking from
/// the matrix's first aligned position to `(rows - 1, cols - 1)`. `isize`
/// mirrors the `-1` entries Swift's border-walk code shape permits
/// (`SegmentSeeker.swift:260-262`); in practice both cursors reach `0`
/// together via the unique `(1, 1) -> (0, 0)` step, so valid inputs never
/// produce one.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct DtwPath {
  text_indices: Vec<isize>,
  time_indices: Vec<isize>,
}

impl DtwPath {
  /// Per-step decoded-token-row index (parallel to
  /// [`Self::time_indices_slice`]).
  #[inline(always)]
  pub fn text_indices_slice(&self) -> &[isize] {
    self.text_indices.as_slice()
  }

  /// Per-step audio-frame-column index (parallel to
  /// [`Self::text_indices_slice`]).
  #[inline(always)]
  pub fn time_indices_slice(&self) -> &[isize] {
    self.time_indices.as_slice()
  }
}

/// The three-way step cost and its winning direction for one DTW cell —
/// `0` diagonal, `1` up, `2` left. Ports `minCostAndTrace`
/// (`SegmentSeeker.swift:239-251`) exactly, including its tie-break order:
/// diagonal wins only by being strictly less than BOTH alternatives, then
/// up wins only by being strictly less than BOTH alternatives, and
/// everything else — including every exact tie — falls to the final
/// `else` and picks left. This `if`/`else if`/`else` shape (not a
/// three-way `min`) is what makes left the tie winner; it must not be
/// reordered or loosened to `<=`.
///
/// Swift computes `c0 = diagonal + value`, `c1 = up + value`, `c2 = left +
/// value` up front and compares THOSE (`SegmentSeeker.swift:239-251`) —
/// and this port does the same, because adding a common finite value is
/// not order-preserving in floating point: a large-magnitude `value` can
/// round distinct incoming costs into exact ties, and exact ties fall to
/// left. Comparing the bare incoming costs picked a different winner on
/// such inputs (phase-gate finding; pinned by
/// `dtw_add_before_compare_matches_swift_rounding_ties`).
fn min_cost_and_trace(diagonal: f64, up: f64, left: f64, value: f64) -> (f64, i8) {
  let c0 = diagonal + value;
  let c1 = up + value;
  let c2 = left + value;
  if c0 < c1 && c0 < c2 {
    (c0, 0)
  } else if c1 < c0 && c1 < c2 {
    (c1, 1)
  } else {
    (c2, 2)
  }
}

/// Dynamic time warping over a decoded-token x audio-frame cross-attention
/// alignment matrix. Builds a `(rows + 1) x (cols + 1)` cost/trace
/// matrix — Swift's nested `[[Double]]`/`[[Int]]`, flattened here into
/// row-major `Vec<f64>`/`Vec<i8>` indexed `row * (cols + 1) + col` for the
/// same values without per-row allocation — then backtraces it into a
/// [`DtwPath`]. Ports `dynamicTimeWarping(withMatrix:)`
/// (`SegmentSeeker.swift:195-237`, backtrace `:253-278`); cell
/// tie-breaking is the private `min_cost_and_trace` (`:239-251`).
///
/// # Errors
/// [`SegmentError::InvalidAlignmentShape`] if `matrix` has zero rows or
/// columns. Swift has no equivalent guard: `1...numberOfColumns`/
/// `1...numberOfRows` over a zero dimension is an invalid `ClosedRange`
/// and traps at runtime. This is a deliberate improvement over that
/// crash — a typed, recoverable error on the same malformed input.
pub fn dynamic_time_warping(matrix: &AlignmentView<'_>) -> Result<DtwPath, SegmentError> {
  let (rows, cols) = (matrix.rows(), matrix.cols());
  if rows == 0 || cols == 0 {
    return Err(SegmentError::InvalidAlignmentShape(
      InvalidAlignmentShape::new(rows, cols, matrix.data().len()),
    ));
  }

  let width = cols + 1;
  let mut cost = vec![f64::INFINITY; (rows + 1) * width];
  let mut trace = vec![-1i8; (rows + 1) * width];
  cost[0] = 0.0;
  for cell in &mut trace[1..=cols] {
    *cell = 2; // :208-210 -- top border backtraces LEFT.
  }
  for i in 1..=rows {
    trace[i * width] = 1; // :211-213 -- left border backtraces UP.
  }

  for row in 1..=rows {
    for column in 1..=cols {
      // :217 -- MLMultiArray's flat linear index; equivalent to this
      // AlignmentView's row-major `row(row - 1)[column - 1]`.
      let value = -f64::from(matrix.row(row - 1)[column - 1]);
      let diagonal = cost[(row - 1) * width + column - 1];
      let up = cost[(row - 1) * width + column];
      let left = cost[row * width + column - 1];
      let (best, direction) = min_cost_and_trace(diagonal, up, left, value);
      cost[row * width + column] = best;
      trace[row * width + column] = direction;
    }
  }

  // :253-278 -- backtrace from the bottom-right corner to the origin.
  let (mut i, mut j) = (rows, cols);
  let mut text_indices = Vec::new();
  let mut time_indices = Vec::new();
  while i > 0 || j > 0 {
    text_indices.push(i as isize - 1);
    time_indices.push(j as isize - 1);
    match trace[i * width + j] {
      0 => {
        i -= 1;
        j -= 1;
      }
      1 => i -= 1,
      2 => j -= 1,
      // Unreachable for any (i, j) this loop actually visits: every cell
      // but (0, 0) -- never read, since the loop condition stops there --
      // was written by the border init or the main loop above to 0/1/2.
      // Kept as a defensive exit; Swift's `default: break` only exits the
      // `switch` there, which would spin forever instead if this were
      // ever hit.
      _ => break,
    }
  }
  text_indices.reverse();
  time_indices.reverse();

  Ok(DtwPath {
    text_indices,
    time_indices,
  })
}

/// True where Swift's `String.contains(String)` is: substring search that
/// treats an empty needle as never found (`String`'s/`NSString`'s
/// `range(of: "")` is documented to return no match), unlike
/// `str::contains`, for which an empty pattern matches everywhere.
/// [`merge_punctuations`]'s punctuation-membership checks need Swift's
/// behavior to stay exact for a word that trims to nothing.
fn swift_contains(haystack: &str, needle: &str) -> bool {
  !needle.is_empty() && haystack.contains(needle)
}

/// Trims Swift's `.whitespaces` `CharacterSet` (Unicode general category
/// `Zs` plus U+0009 CHARACTER TABULATION; no newlines) off both ends of
/// `s`. Same predicate as `tokenizer::is_single_punctuation_scalar`'s trim
/// step, duplicated here because that helper is private to its module.
fn trim_swift_whitespaces(s: &str) -> &str {
  s.trim_matches(|c: char| c.is_separator_space() || c == '\u{0009}')
}

/// Merges leading/trailing punctuation-only words in `alignment` onto
/// their neighboring word, then drops the words that end up empty or are
/// themselves bare merged-away punctuation. Ports `mergePunctuations`
/// (`SegmentSeeker.swift:280-338`) in its exact two-pass shape: characters
/// in `prepended` glue onto the FOLLOWING word (`:291-315`), characters in
/// `appended` glue onto the PRECEDING word (`:322-333`), then a final
/// filter drops the leftovers (`:336`).
///
/// Both passes replicate a Swift quirk rather than smoothing it over: each
/// iteration reads its merge neighbor from the *original* pre-merge
/// slice — `alignment[i - 1]` in the prepend pass, `prependedAlignment[i -
/// 1]` in the append pass — never from the tail of the list actually being
/// built. A run of three or more consecutive punctuation-only words
/// therefore does not fully chain together in either this port or
/// upstream Swift: only the immediately preceding original word survives
/// a second merge. Whisper's fixed single-character punctuation
/// vocabularies make three consecutive punctuation-only *words*
/// essentially unreachable in practice, so this port keeps Swift's exact
/// indexing rather than silently changing the observable behavior.
pub fn merge_punctuations(
  alignment: &[WordTiming],
  prepended: &str,
  appended: &str,
) -> Vec<WordTiming> {
  if alignment.is_empty() {
    return Vec::new();
  }

  // :291-315 -- merge PREPEND punctuation onto the following word.
  let mut prepended_alignment: Vec<WordTiming> = Vec::new();
  if !swift_contains(prepended, trim_swift_whitespaces(alignment[0].word())) {
    prepended_alignment.push(alignment[0].clone());
  }
  for pair in alignment.windows(2) {
    let previous = &pair[0];
    let current = &pair[1];
    let previous_starts_with_whitespace = previous
      .word()
      .chars()
      .next()
      .is_some_and(|c| c.is_separator_space() || c == '\u{0009}');
    if previous_starts_with_whitespace
      && swift_contains(prepended, trim_swift_whitespaces(previous.word()))
    {
      let mut word = previous.word().to_string();
      word.push_str(current.word());
      let mut tokens = previous.tokens_slice().to_vec();
      tokens.extend_from_slice(current.tokens_slice());
      let merged = WordTiming::new(
        word,
        tokens,
        current.start(),
        current.end(),
        current.probability(),
      );
      if prepended_alignment.is_empty() {
        prepended_alignment.push(merged);
      } else {
        let last = prepended_alignment.len() - 1;
        prepended_alignment[last] = merged;
      }
    } else {
      prepended_alignment.push(current.clone());
    }
  }

  // :317-333 -- merge APPEND punctuation onto the preceding word.
  let mut appended_alignment: Vec<WordTiming> = Vec::new();
  if let Some(first) = prepended_alignment.first() {
    appended_alignment.push(first.clone());
  }
  for pair in prepended_alignment.windows(2) {
    let previous = &pair[0];
    let current = &pair[1];
    if !previous.word().ends_with(' ')
      && swift_contains(appended, trim_swift_whitespaces(current.word()))
    {
      let mut word = previous.word().to_string();
      word.push_str(current.word());
      let mut tokens = previous.tokens_slice().to_vec();
      tokens.extend_from_slice(current.tokens_slice());
      let merged = WordTiming::new(
        word,
        tokens,
        previous.start(),
        previous.end(),
        previous.probability(),
      );
      let last = appended_alignment.len() - 1;
      appended_alignment[last] = merged;
    } else {
      appended_alignment.push(current.clone());
    }
  }

  // :336 -- drop empties and bare merged-away punctuation words.
  appended_alignment
    .into_iter()
    .filter(|w| {
      !w.word().is_empty()
        && !swift_contains(appended, w.word())
        && !swift_contains(prepended, w.word())
    })
    .collect()
}

/// Word-level timestamps for one window's decoded tokens: runs
/// [`dynamic_time_warping`] over `alignment`, groups `word_token_ids` into
/// words via [`WhisperTokenizer::split_to_word_tokens`], and reads each
/// word's start/end time off the DTW path's per-token-row boundaries plus
/// its mean sampled-token log probability. Ports `findAlignment`
/// (`SegmentSeeker.swift:340-408`); `language_code` is threaded straight
/// into `split_to_word_tokens` in place of Swift's internal
/// `NLLanguageRecognizer` detection (spec §5.3; see
/// [`WhisperTokenizer::split_to_word_tokens`]'s own doc for the same
/// substitution there), and `grouping` chooses how that splitter groups
/// (coremlit issue #14 — [`WordGrouping::SwiftParity`] is the default after
/// #41; [`WordGrouping::FineGrained`] is this port's long-standing opt-in).
///
/// Returns an empty vec when `split_to_word_tokens` groups `word_token_ids`
/// into one word or fewer (`:351-353`) — DTW timing is meaningless for a
/// single undivided span. DTW itself still runs first regardless (matching
/// Swift's own unconditional call order), so a malformed `alignment`
/// still errors even on that trivial path.
///
/// # Errors
/// [`SegmentError::InvalidAlignmentShape`] if `alignment` has zero rows or
/// columns (from [`dynamic_time_warping`]); [`SegmentError::Tokenizer`] if
/// `split_to_word_tokens` fails to decode `word_token_ids`.
pub fn find_alignment(
  word_token_ids: &[u32],
  alignment: &AlignmentView<'_>,
  token_log_probs: &[f32],
  tokenizer: &WhisperTokenizer,
  language_code: &str,
  grouping: WordGrouping,
) -> Result<Vec<WordTiming>, SegmentError> {
  let path = dynamic_time_warping(alignment)?;
  let text_indices = path.text_indices_slice();
  let time_indices = path.time_indices_slice();

  let word_tokens = tokenizer.split_to_word_tokens(word_token_ids, language_code, grouping)?;
  if word_tokens.len() <= 1 {
    return Ok(Vec::new());
  }

  // :356-371 -- per-decoded-token-row start/end times: one boundary each
  // time the DTW path's aligned row changes.
  let mut start_times: Vec<f32> = vec![0.0];
  let mut end_times: Vec<f32> = Vec::new();
  let mut current_text_index = text_indices.first().copied().unwrap_or(0);
  for (index, &text_index) in text_indices.iter().enumerate() {
    if text_index != current_text_index {
      current_text_index = text_index;
      let time = time_indices[index] as f32 * SECONDS_PER_TIME_TOKEN;
      start_times.push(time);
      end_times.push(time);
    }
  }
  end_times.push(time_indices.last().copied().unwrap_or(1500) as f32 * SECONDS_PER_TIME_TOKEN);

  // :373-405 -- walk word groups; each consumes `tokens.len()` rows of
  // start_times/end_times/token_log_probs.
  let mut word_timings: Vec<WordTiming> = Vec::with_capacity(word_tokens.len());
  let mut current_token_index = 0usize;
  for (word, tokens) in word_tokens {
    let start_index = current_token_index;
    let word_start_time = start_times[current_token_index];
    current_token_index += tokens.len() - 1;
    let word_end_time = end_times[current_token_index];
    current_token_index += 1;

    let probs = &token_log_probs[start_index..current_token_index];
    let mean_log_prob = probs.iter().sum::<f32>() / probs.len() as f32;

    word_timings.push(WordTiming::new(
      word,
      tokens,
      word_start_time,
      word_end_time,
      mean_log_prob.exp(),
    ));
  }

  Ok(word_timings)
}

// ---------------------------------------------------------------------
// Word-duration constraints and sentence-boundary truncation
// ---------------------------------------------------------------------

/// The capped-median/max word-duration pair [`calculate_word_duration_constraints`]
/// computes over one window's word alignment — Swift's anonymous
/// `(median: Float, max: Float)` tuple return (`SegmentSeeker.swift:
/// 498-507`). `Copy`, and deliberately has no constructor: unlike an
/// options type, whose fields a caller assembles piecewise before the
/// fact, both fields here only ever come out of that one function
/// together, and `max` is always exactly `median * 2` — a public
/// constructor would let a caller build a pair that violates that
/// invariant. This type is a computation RESULT, not configuration; that
/// is a deliberate departure from the options pattern, not an oversight.
#[derive(Debug, Clone, Copy, PartialEq)]
pub struct WordDurationConstraints {
  median: f32,
  max: f32,
}

impl WordDurationConstraints {
  /// The capped median word duration: `min(0.7, raw median)`, in seconds,
  /// or `0.0` when the source alignment had no positive-duration word
  /// (`SegmentSeeker.swift:502-503`).
  #[inline(always)]
  pub const fn median(&self) -> f32 {
    self.median
  }

  /// The overlong-word threshold: always exactly twice [`Self::median`]
  /// (`SegmentSeeker.swift:504`), consumed by
  /// [`truncate_long_words_at_sentence_boundaries`].
  #[inline(always)]
  pub const fn max_duration(&self) -> f32 {
    self.max
  }
}

/// Computes the capped-median/max word-duration pair used to flag overlong
/// words. Ports `calculateWordDurationConstraints`
/// (`SegmentSeeker.swift:498-507`): every word whose `duration` is not
/// strictly positive is dropped before the median is taken (`:499-500`,
/// `filter { $0 > 0 }`); the median is the sorted list's UPPER middle
/// element — `sorted[count / 2]`, integer division, NOT an average of the
/// two middle values on an even count (`:502`); that raw median is then
/// capped at `0.7` s (`:503`), and `max` is always twice the CAPPED
/// value, never the raw one (`:504`). An empty `alignment`, or one where
/// every word has zero or negative duration, yields `median = max = 0.0`.
pub fn calculate_word_duration_constraints(alignment: &[WordTiming]) -> WordDurationConstraints {
  let mut durations: Vec<f32> = alignment
    .iter()
    .map(WordTiming::duration)
    .filter(|&duration| duration > 0.0)
    .collect();
  durations.sort_by(f32::total_cmp);

  let raw_median = durations.get(durations.len() / 2).copied().unwrap_or(0.0);
  let median = raw_median.min(0.7);
  let max = median * 2.0;

  WordDurationConstraints { median, max }
}

/// Sentence-ending marks [`truncate_long_words_at_sentence_boundaries`]
/// matches a word's text against EXACTLY — no substring or trimmed match
/// (Swift `sentenceEndMarks`, `SegmentSeeker.swift:510`; includes the CJK
/// full-width punctuation forms alongside the ASCII ones).
const SENTENCE_END_MARKS: [&str; 6] = [".", "。", "!", "!", "?", "?"];

/// Clips words whose duration exceeds `max_duration` back down to it, but
/// only where a sentence boundary justifies the clip — guards against a
/// single misaligned DTW timestamp stretching one word far past its real
/// span. Ports `truncateLongWordsAtSentenceBoundaries`
/// (`SegmentSeeker.swift:509-526`) structure-preserving: index `0` is
/// never inspected or modified (the loop runs `1..alignment.len()`,
/// `:514`; Rust's half-open `Range` is simply empty rather than panicking
/// when `alignment` itself is empty, so no separate emptiness guard is
/// needed the way Swift's `1..<0` `ClosedRange` would require). For each
/// later word whose `duration()` exceeds `max_duration`:
/// - if the word's own text EXACTLY matches a mark in the internal
///   `SENTENCE_END_MARKS` table (whole-word — a caller passing `" ."` with
///   a leading space does not qualify; note that words reaching here from
///   [`find_alignment`] never carry that space, because
///   [`WhisperTokenizer::decode`]'s tokenization-space cleanup strips it,
///   just as Swift's does), its `end` is pulled back to
///   `start + max_duration`;
/// - otherwise, if the PRECEDING word's text exactly matches a mark, this
///   word's `start` is pushed forward to `end - max_duration`.
///
/// These are `if`/`else if` branches (`:516-520`), so a word that is
/// itself a mark AND immediately follows another mark only ever takes the
/// first branch: its `end` is adjusted, its `start` never is.
pub fn truncate_long_words_at_sentence_boundaries(
  mut alignment: Vec<WordTiming>,
  max_duration: f32,
) -> Vec<WordTiming> {
  for i in 1..alignment.len() {
    if alignment[i].duration() > max_duration {
      if SENTENCE_END_MARKS.contains(&alignment[i].word()) {
        let start = alignment[i].start();
        alignment[i].set_end(start + max_duration);
      } else if SENTENCE_END_MARKS.contains(&alignment[i - 1].word()) {
        let end = alignment[i].end();
        alignment[i].set_start(end - max_duration);
      }
    }
  }
  alignment
}

/// Re-anchors DTW word-level `merged_alignment` timings onto `segments`,
/// applying Swift's short-word pull-back and pause/boundary heuristics
/// along the way. Ports `updateSegmentsWithWordTimings`
/// (`SegmentSeeker.swift:528-659`) — the last stage of
/// [`add_word_timestamps`] (`:410-496`), which runs it after
/// `findAlignment` -> the duration/truncation hack -> `mergePunctuations`.
///
/// `merged_alignment` is walked with a cursor SHARED across every
/// `segments` entry (`:538`, Swift's `wordIndex`): once an alignment entry
/// is consumed by one segment it is never revisited by a later one, even
/// if that segment's own token budget is not fully accounted for (the
/// cursor simply runs out and that segment's word list ends up short).
/// `last_speech_timestamp` is similarly threaded across every segment,
/// seeded by the caller's initial value and updated to each segment's own
/// final `end` once it gets at least one word (`:651`, guarded by the same
/// non-empty check as the pause/boundary hack itself — a wordless segment
/// leaves `last_speech_timestamp` untouched for the next one).
///
/// # Errors
/// [`SegmentError::Tokenizer`] if retokenizing a partially-special-filtered
/// alignment entry's surviving tokens fails (`:556-559`).
pub fn update_segments_with_word_timings(
  segments: &[TranscriptionSegment],
  merged_alignment: &[WordTiming],
  seek: usize,
  last_speech_timestamp: f32,
  constrained_median_duration: f32,
  max_duration: f32,
  tokenizer: &WhisperTokenizer,
) -> Result<Vec<TranscriptionSegment>, SegmentError> {
  // :537 -- this window's seek offset, in seconds.
  let time_offset = seek as f32 / SAMPLE_RATE as f32;
  let special_begin = tokenizer.special_tokens().special_token_begin();
  // :538 -- cursor into `merged_alignment`, shared across every segment
  // below; never reset per segment.
  let mut word_index = 0usize;
  let mut last_speech_timestamp = last_speech_timestamp;
  let mut updated_segments: Vec<TranscriptionSegment> = Vec::with_capacity(segments.len());

  for (segment_index, segment) in segments.iter().enumerate() {
    let mut saved_tokens = 0usize;
    // :544 -- only text tokens count toward this segment's word budget;
    // special/timestamp tokens already in `segment.tokens` never do.
    let text_token_count = segment
      .tokens_slice()
      .iter()
      .filter(|&&token| token < special_begin)
      .count();
    let mut words_in_segment: Vec<WordTiming> = Vec::new();

    // :547's `where savedTokens < textTokens.count` guards each element in
    // Swift's `for timing in mergedAlignment[wordIndex...]`, skipping
    // (not necessarily stopping at) elements while false. `break` here is
    // behaviorally identical: `saved_tokens` and `word_index` both only
    // ever advance inside this loop body, so once the bound trips false it
    // stays false for every later element too, and Swift's `where` never
    // lets the body run again either. `.min(merged_alignment.len())`
    // guards a slice a Swift `mergedAlignment[wordIndex...]` has no
    // equivalent for (it would trap on an out-of-range `wordIndex`); the
    // invariant that `word_index` never exceeds `merged_alignment.len()`
    // holds by construction (each increment consumes one element of the
    // shrinking remaining slice), so this is a zero-cost safety net, not a
    // behavior change.
    for timing in &merged_alignment[word_index.min(merged_alignment.len())..] {
      if saved_tokens >= text_token_count {
        break;
      }
      word_index += 1;

      // :551-554 -- drop special/timestamp tokens from this timing; an
      // all-special entry is consumed from the cursor but emits no word.
      let timing_tokens: Vec<u32> = timing
        .tokens_slice()
        .iter()
        .copied()
        .filter(|&token| token < special_begin)
        .collect();
      if timing_tokens.is_empty() {
        continue;
      }

      // :556-559 -- retokenize only when some (not all) of this timing's
      // tokens were filtered out; otherwise reuse its own decoded word.
      let timing_tokens_len = timing_tokens.len();
      let word = if timing_tokens_len < timing.tokens_slice().len() {
        tokenizer.decode(&timing_tokens, false)?
      } else {
        timing.word().to_string()
      };

      // :561-562.
      let mut start = rounded_to_places(time_offset + timing.start(), 2);
      let end = rounded_to_places(time_offset + timing.end(), 2);

      // :564-596 -- a short-duration word gets its start pulled back into
      // any gap before it: against the previous word in THIS segment if
      // there is one, else (only for a segment's own first word) against
      // the previous segment's already-finalized end.
      if end - start < constrained_median_duration / 4.0 {
        if let Some(previous) = words_in_segment.last() {
          let previous_end = previous.end();
          if start > previous_end {
            let space_available = start - previous_end;
            let desired_duration = space_available.min(constrained_median_duration / 2.0);
            start = rounded_to_places(start - desired_duration, 2);
          }
        } else if segment_index > 0
          && updated_segments.len() > segment_index - 1
          && start > updated_segments[segment_index - 1].end()
        {
          let previous_end = updated_segments[segment_index - 1].end();
          let space_available = start - previous_end;
          let desired_duration = space_available.min(constrained_median_duration / 2.0);
          start = rounded_to_places(start - desired_duration, 2);
        }
      }

      // :598.
      let probability = rounded_to_places(timing.probability(), 2);
      words_in_segment.push(WordTiming::new(
        word,
        timing_tokens,
        start,
        end,
        probability,
      ));
      // :606 -- Swift re-reads `timingTokens.count`, the local filtered
      // vec, not the just-pushed word's own token slice; captured above
      // before `timing_tokens` moved into the `WordTiming`.
      saved_tokens += timing_tokens_len;
    }

    let mut updated_segment = segment.clone();

    // :615-652 -- only a segment that got at least one word runs the
    // pause/boundary hack and advances `last_speech_timestamp`; a wordless
    // segment leaves both `updated_segment`'s bounds and
    // `last_speech_timestamp` untouched.
    if !words_in_segment.is_empty() {
      // :616-620 -- read BEFORE any mutation below, matching Swift's own
      // `firstWord` copy.
      let pause_length = words_in_segment[0].end() - last_speech_timestamp;
      let first_word_too_long = words_in_segment[0].duration() > max_duration;
      let both_words_too_long = words_in_segment.len() > 1
        && words_in_segment[1].end() - words_in_segment[0].start() > max_duration * 2.0;

      // :621-633 -- after an over-long pause, clamp the first word (and,
      // if it is also too long, re-split the 0/1 boundary first) so
      // neither word spans more than `max_duration`.
      if pause_length > constrained_median_duration * 4.0
        && (first_word_too_long || both_words_too_long)
      {
        if words_in_segment.len() > 1 && words_in_segment[1].duration() > max_duration {
          let w1_end = words_in_segment[1].end();
          let boundary = (w1_end / 2.0).max(w1_end - max_duration);
          words_in_segment[0].set_end(boundary);
          words_in_segment[1].set_start(boundary);
        }
        // Reads `words_in_segment[0].end()` LIVE: the boundary re-split
        // just above, if it fired, already changed it.
        let w0_end = words_in_segment[0].end();
        words_in_segment[0].set_start(last_speech_timestamp.max(w0_end - max_duration));
      }

      // :635-640 -- prefer the segment-level start over the (possibly
      // hack-adjusted) first word's start when the word has drifted more
      // than half a second earlier than the segment itself began.
      let w0_start = words_in_segment[0].start();
      let w0_end = words_in_segment[0].end();
      if segment.start() < w0_end && segment.start() - 0.5 > w0_start {
        let clamped = (w0_end - constrained_median_duration)
          .min(segment.start())
          .max(0.0);
        words_in_segment[0].set_start(clamped);
      } else {
        updated_segment.set_start(words_in_segment[0].start());
      }

      // :642-649 -- symmetric preference for the segment-level end over
      // the last word's end. Swift's `wordsInSegment.last` is always
      // non-nil here (guarded by the outer non-empty check already); when
      // there is exactly one word this is the SAME element the
      // start-preference block above just wrote, so `last_start` below
      // can already reflect that mutation.
      let last_index = words_in_segment.len() - 1;
      let last_start = words_in_segment[last_index].start();
      let last_end = words_in_segment[last_index].end();
      if updated_segment.end() > last_start && segment.end() + 0.5 < last_end {
        let clamped = (last_start + constrained_median_duration).max(segment.end());
        words_in_segment[last_index].set_end(clamped);
      } else {
        updated_segment.set_end(last_end);
      }

      // :651.
      last_speech_timestamp = updated_segment.end();
    }

    // :654-655.
    updated_segment.set_words(words_in_segment);
    updated_segments.push(updated_segment);
  }

  Ok(updated_segments)
}

/// Rounds `value` to `decimal_places` decimal digits, half-away-from-zero.
/// Ports `Float.rounded(_:)` (`ArgmaxCore/FoundationExtensions.swift:
/// 9-13`: `(self * divisor).rounded() / divisor`, where Swift's
/// no-argument `.rounded()` defaults to rule `.toNearestOrAwayFromZero`).
/// Rust's [`f32::round`] documents that exact rule (round half-way cases
/// away from `0.0`), so this is a direct, unadjusted port. `pub(crate)`:
/// no caller outside this crate needs it yet — [`update_segments_with_word_timings`]
/// is the first non-test consumer.
pub(crate) fn rounded_to_places(value: f32, decimal_places: i32) -> f32 {
  let divisor = 10f32.powi(decimal_places);
  (value * divisor).round() / divisor
}

// ---------------------------------------------------------------------
// add_word_timestamps: the orchestration wrapper
// ---------------------------------------------------------------------

/// Measures **this host's** CoreVideo row pitch, in Float16 elements, for
/// the IOSurface-backed `MLMultiArray` Swift's alignment gather allocates —
/// `MLMultiArray(shape: [rows, cols], dataType: .float16, initialValue: 0)`
/// (`ArgmaxCore/MLMultiArrayExtensions.swift:11-53`), which for `.float16`
/// is a `kCVPixelFormatType_OneComponent16Half` `CVPixelBuffer`
/// (`:121-136`) — by allocating the equivalent array through
/// [`MultiArray::f16_surface`] and reading the strides CoreVideo actually
/// chose.
///
/// **Only [`AlignmentGather::SwiftParity`] — the opt-in mode — calls this.**
/// The default gather ([`AlignmentGather::Complete`]) allocates no surface,
/// measures nothing and depends on no property of this host.
///
/// **Not a constant, deliberately.** Apple documents `CVPixelBuffer` row
/// alignment as hardware-dependent and requires it to be queried (QA1829,
/// "Understanding CVPixelBuffer memory alignment"), and
/// [`MultiArray::f16_surface`]'s own contract calls the padding
/// platform-chosen. The reference host for whisper #41 (M1 Max, macOS 26.5)
/// aligns rows to 64 bytes — 32 Float16 elements — which is what
/// `tests/whisper_swift_probes/probe_alignment_stride.out` recorded (cols
/// 8 -> 32, 9 -> 32, 100 -> 128, 1496/1500/1504 -> 1504); that table is
/// evidence of the layout, not a rule about it. A host that pads
/// differently would have Swift's gather truncate *different* cells, so a
/// compiled-in quantum would zero the wrong ones and make
/// [`AlignmentGather::SwiftParity`] silently non-parity there. Querying the
/// same allocator Swift's array comes from is what makes the mode correct
/// on every host rather than on one.
///
/// The probe uses the caller's exact `[rows, cols]` shape rather than a
/// cheaper 1-row stand-in, so nothing here assumes the pitch is independent
/// of the row count — which is why [`gather_swift_parity_into`] asks for the
/// SOURCE and DESTINATION shapes separately instead of reusing one answer for
/// both. One `CVPixelBufferCreate` plus its zero-fill per shape per
/// word-timestamped window is immaterial next to that window's encoder and
/// decoder runs.
///
/// # What this establishes, and what it does not
///
/// **Established:** the row pitch of the surface THIS call allocated, read
/// back from CoreVideo's own strides.
///
/// **Not established:** the row pitch of the surface Swift's process
/// allocated. Nothing in CoreVideo promises two allocations of one shape
/// share a layout, and this port cannot observe Swift's. `SwiftParity`
/// assumes they agree — assumption **A1** on
/// [`AlignmentGather::SwiftParity`], stated there rather than left implicit.
///
/// **What would settle it:** a documented CoreVideo/IOSurface guarantee that
/// row alignment is a pure function of pixel format and dimensions. No such
/// guarantee is published; QA1829 says the opposite in spirit by requiring
/// the value to be queried per buffer.
///
/// # Errors
/// [`SegmentError::AlignmentPitchUnavailable`] if the probe allocation
/// fails (e.g. [`TensorError::SurfaceUnsupported`](crate::TensorError) below
/// macOS 12, where `MLMultiArray` has no pixel-buffer initializer);
/// [`SegmentError::AlignmentPitchUnexpectedLayout`] if it succeeds but
/// reports strides other than `[pitch, 1]` with `pitch >= cols`. Both are
/// fail-closed: see [`SegmentError::AlignmentPitchUnavailable`] for why a
/// silent fall back to [`AlignmentGather::Complete`] is not offered.
// `pub(crate)`, not private: `transcribe::tests`' own gather fixture
// (`swift_parity_gather_moves_the_first_windows_end_and_the_next_seek`) is built around where
// this host's pitch cuts the gather, and has to ask the same helper the
// pipeline asks rather than restate a number.
pub(crate) fn coreml_f16_row_pitch(rows: usize, cols: usize) -> Result<usize, SegmentError> {
  let probe = MultiArray::f16_surface(&[rows, cols]).map_err(|source| {
    SegmentError::AlignmentPitchUnavailable(AlignmentPitchUnavailable::new(rows, cols, source))
  })?;
  row_pitch_of(&probe, rows, cols)
}

/// The row pitch a probe surface reports, rejecting any layout the gather
/// cannot describe.
///
/// CoreML pads only BETWEEN rows (the invariant `MultiArray::copy_into`'s
/// padded-gather already relies on), so the only layout [`gather_swift_rows`]
/// reproduces is a unit last-dimension stride under a row pitch that is at
/// least the logical width. Anything else is a layout this port has never
/// seen and cannot claim to replicate.
fn row_pitch_of(probe: &MultiArray, rows: usize, cols: usize) -> Result<usize, SegmentError> {
  let strides = probe.strides();
  match *strides {
    [pitch, 1] if pitch >= cols => Ok(pitch),
    _ => Err(SegmentError::AlignmentPitchUnexpectedLayout(
      AlignmentPitchUnexpectedLayout::new(rows, cols, strides.to_vec()),
    )),
  }
}

/// The row-pitch measurement [`gather_swift_parity_into`] runs, as a
/// parameter rather than a hard-wired call.
///
/// Production passes [`coreml_f16_row_pitch`], which measures the running
/// host. `segment::tests` passes layouts this host will never produce — above
/// all a host that pitches Swift's 224-row source array differently from this
/// port's 225-row accumulator, which is the ONLY way to check that the probe
/// asks CoreVideo about Swift's height rather than the port's (whisper #41,
/// codex round 3, F2).
type RowPitchProbe<'a> = &'a dyn Fn(usize, usize) -> Result<usize, SegmentError>;

/// Reproduces Swift's alignment gather into `out` (supplied zero-filled),
/// measuring both surfaces' row pitches through `row_pitch`.
///
/// # The two heights are NOT the same number
///
/// `alignment.rows()` is this port's accumulator height,
/// `max_token_context + 1` — 225 — because
/// [`commit_alignment_row`](crate::audio::whisper::backend::InferenceBackend::commit_alignment_row)
/// commits step `position`'s row at `position + 1`, so the last trait-legal
/// position needs one slot of headroom past the KV dimension. Swift has no
/// such headroom: `alignmentWeights` is allocated at
/// `[kvCacheMaxSequenceLength, n_audio_ctx]` (`TextDecoder.swift:141`) — the
/// **224**-slot KV dimension itself.
///
/// The pitch is a function of the shape CoreVideo is handed, and this branch
/// exists precisely because pitch may vary with height. Probing `[225, cols]`
/// would therefore decode Swift's storage offsets with a pitch Swift's array
/// never had wherever `pitch(224, cols) != pitch(225, cols)`, splicing the
/// wrong weights (or the wrong padding) into DTW's input while still
/// reporting parity. `swift_source_rows` is Swift's physical height, passed
/// down from the model's own `max_token_context` — never derived from the
/// view — and it is the shape this probes.
///
/// The reproduction's read bound is the smaller of the two: an offset past
/// `swift_source_rows` is past Swift's array (Swift would read off its end),
/// and an offset past `alignment.rows()` is past this port's data. Neither is
/// reachable for a real window (`needed <= 224 <= min(224, 225)`); the bound
/// is the same defensive branch the prefix take has always carried.
///
/// # Errors
/// [`SegmentError::AlignmentPitchUnavailable`] /
/// [`SegmentError::AlignmentPitchUnexpectedLayout`] from `row_pitch`, for
/// either surface — see [`coreml_f16_row_pitch`].
fn gather_swift_parity_into(
  out: &mut [f32],
  alignment: &AlignmentView<'_>,
  needed: usize,
  cols: usize,
  swift_source_rows: usize,
  row_pitch: RowPitchProbe<'_>,
) -> Result<(), SegmentError> {
  let src_rows = alignment.rows().min(swift_source_rows);
  // A row-less source has no surface for Swift to have allocated, and the
  // gather reads nothing from it whatever CoreVideo would have pitched it at
  // (every offset lands past row `src_rows`), so the reproduction is all
  // zeros. Skipping the probe keeps a caller-built empty view out of
  // `AlignmentPitchUnavailable`, which reports an unmeasurable HOST rather
  // than an empty input. No backend produces one: both build the FIXED
  // `max_token_context + 1` accumulator against a nonzero `max_token_context`.
  let src_pitch = if src_rows == 0 {
    cols
  } else {
    row_pitch(swift_source_rows, cols)?
  };
  let dst_pitch = row_pitch(needed, cols)?;
  gather_swift_rows(
    out,
    alignment.data(),
    src_rows,
    needed,
    cols,
    src_pitch,
    dst_pitch,
  );
  Ok(())
}

/// Reproduces Swift's alignment gather: what `dynamicTimeWarping` actually
/// reads out of the destination array `addWordTimestamps` builds, given the
/// source's logical rows and the two surfaces' measured row pitches.
///
/// Writes `needed * cols` row-major f32 into `out`, which the caller supplies
/// zero-filled.
///
/// # The reproduction, cell by cell
///
/// Swift's loop (`SegmentSeeker.swift:452-460`) binds both arrays' strides
/// and then ignores them, offsetting BOTH sides by `index * columnCount`.
/// Its `filteredIndices` are `0..needed` by construction (`:429-432`,
/// `:441`), so the copies tile `[0, needed * cols)` contiguously on each
/// side and the whole loop is one verbatim copy: destination STORAGE offset
/// `o` holds source STORAGE offset `o`, for every `o < needed * cols`.
/// `dynamicTimeWarping` reads logical row `r`, column `c` through
/// `MLMultiArray`'s stride-aware flat subscript (`:217`), i.e. destination
/// storage `o = r * dst_pitch + c`. So:
///
/// - `o >= needed * cols` — past the copy. Neither the `memcpy` nor the
///   `initialValue:` fill (logical `count` elements, `:450`) ever wrote
///   there, so it is the destination allocation's untouched tail, and what
///   DTW reads from it is whatever `CVPixelBufferCreate` returned. This
///   function substitutes ZERO, which is assumption **A2** of the opt-in
///   mode — see [`AlignmentGather::SwiftParity`], where it is stated in full
///   rather than presented as established. It is not verified here and
///   cannot be: an earlier revision sampled a *different*, disposable
///   allocation's tail, which proves nothing about the one Swift's process
///   gathered into (codex round 3, F3), and the sampling itself read
///   uninitialized memory (F1).
/// - `o < needed * cols` — source storage `o`, which is source logical row
///   `o / src_pitch`, column `o % src_pitch`. A column below `cols` is a
///   real weight; at or above it, the source's own inter-row padding, for
///   which this function likewise substitutes ZERO.
///
/// **What the source-padding zero rests on.** `alignmentWeights` is allocated
/// once (`TextDecoder.swift:141`) through the same `initialValue:
/// FloatType(0)` initializer, whose `initialize(repeating:count:)` runs over
/// the LOGICAL element count — flat, ignoring the pitch — and so zeroes
/// storage `[0, src_rows * cols)` at construction, padding cells included. It
/// is never a model input or output backing (`TextDecoder.swift:394-401`,
/// `:414`) and `DecodingInputs.reset` never clears it
/// (`Models.swift:312-322`); its only writer is `updateAlignmentWeights`
/// (`:272-295`), which is stride-aware and so writes no padding cell. The
/// gather reads storage below `needed * cols <= src_rows * cols`, so every
/// padding cell it can reach was zero *at construction*.
///
/// That its construction-time value SURVIVES is a separate step, and an
/// unverified one: both the writer and the gather reach the array through
/// `withUnsafeMutableBytes`, whose contract states the strides handed to the
/// closure may differ from the value before invocation — so no cited contract
/// rules out a relayout or a replacement of the backing storage introducing
/// padding that construction never zeroed. That is assumption **A3** of the
/// opt-in mode, stated on [`AlignmentGather::SwiftParity`] (codex round 3,
/// F4).
///
/// # Why the two pitches are measured separately
///
/// `src_pitch == dst_pitch` collapses this to "row `r` keeps its first
/// `min(cols, needed * cols - r * dst_pitch)` columns and reads zeros after
/// them", the form the shipping host takes. But row-count invariance is a
/// property this host was measured to have, not one CoreVideo promises, and
/// where the pitches differ the reproduction is not a truncation at all: row
/// `r` reads a SHIFTED window of the source, mixing the tail of one logical
/// row with the head of the next and with zeroed padding. Modelling only the
/// equal-pitch case would have made [`AlignmentGather::SwiftParity`]
/// silently wrong there while still reporting parity, so both shapes are
/// probed and the general mapping is what runs.
///
/// Rows at or past `src_rows` read zeros: Swift's gather would run off the
/// end of its source array there. `needed <= src_rows` for every real window
/// (the accumulator is `max_token_context + 1` rows and `needed <= 224`), so
/// this is the same defensive branch [`add_word_timestamps`]'s prefix take
/// has always had, not a modelled behavior.
///
/// Taking both pitches as parameters (rather than measuring inside) is what
/// lets the reproduction be tested at pitches this host will never choose —
/// see `segment::tests`.
fn gather_swift_rows(
  out: &mut [f32],
  source: &[f32],
  src_rows: usize,
  needed: usize,
  cols: usize,
  src_pitch: usize,
  dst_pitch: usize,
) {
  debug_assert!(src_pitch >= cols && dst_pitch >= cols && cols > 0);
  let copied = needed * cols;
  for (row, out_row) in out.chunks_mut(cols).enumerate().take(needed) {
    // Destination storage this logical row reads, clipped to the copy.
    let mut offset = row * dst_pitch;
    let end = (offset + cols).min(copied);
    let mut column = 0usize;
    while offset < end {
      let source_row = offset / src_pitch;
      if source_row >= src_rows {
        // Past the source array; every later offset is too, so the rest of
        // this row and every row after it keeps `out`'s zero.
        break;
      }
      let source_column = offset % src_pitch;
      let run = if source_column < cols {
        let run = (cols - source_column).min(end - offset);
        let start = source_row * cols + source_column;
        out_row[column..column + run].copy_from_slice(&source[start..start + run]);
        run
      } else {
        // Source-side inter-row padding: zero under assumption A3, per this
        // function's doc, and `out` already holds zero.
        (src_pitch - source_column).min(end - offset)
      };
      offset += run;
      column += run;
    }
    // `[column, cols)` is the destination tail: zero under assumption A2,
    // per this function's doc, and already zero in `out`.
  }
}

/// Assembles one window's word-level timestamps end to end. Ports
/// `SegmentSeeker.addWordTimestamps` (`SegmentSeeker.swift:410-496`):
/// flattens `segments`' tokens/log-probs into the flat list
/// [`find_alignment`] needs (`:427-442`), builds a prefix-take,
/// zero-padded [`AlignmentMatrix`] from `alignment` (`:444-461`), then
/// threads that through [`find_alignment`] (`:465-472`) -> the
/// duration-constraint/sentence-boundary truncation hack (`:474-477`) ->
/// [`merge_punctuations`] when non-empty (`:479-482`) ->
/// [`update_segments_with_word_timings`] (`:484-493`).
///
/// `language_code` and `grouping` are threaded straight into
/// `find_alignment` -> `split_to_word_tokens`: the same
/// `NLLanguageRecognizer` replacement documented on [`find_alignment`] and
/// [`WhisperTokenizer::split_to_word_tokens`] (spec §5.3), plus the
/// explicit word-grouping mode from coremlit issue #14
/// ([`WordGrouping::SwiftParity`] is the default after #41;
/// [`WordGrouping::FineGrained`] is this port's long-standing opt-in).
///
/// `gather` selects what the prefix take hands to DTW.
/// [`AlignmentGather::Complete`] — **the default** — gathers every row whole:
/// a plain prefix take, allocating no surface, measuring nothing about the
/// host and assuming nothing about it.
/// [`AlignmentGather::SwiftParity`] is the opt-in that instead reproduces
/// Swift's own gather: its `memcpy` loop (`SegmentSeeker.swift:452-460`) binds
/// both `MLMultiArray`s' strides and then indexes with `columnCount`, while
/// `dynamicTimeWarping`'s flat subscript (`:217`) *is* stride-aware and
/// CoreVideo pads the Float16 backing's rows
/// (`ArgmaxCore/MLMultiArrayExtensions.swift:121-136`), so what DTW reads back
/// is a function of BOTH surfaces' row pitches — each measured on the running
/// host by this module's `coreml_f16_row_pitch`, at each surface's own height.
/// This module's `gather_swift_parity_into`/`gather_swift_rows` derive the
/// mapping; see [`AlignmentGather::SwiftParity`] for the three assumptions
/// that mode carries and why they are acceptable only because it is opt-in
/// (whisper #41). `swift_source_rows` is Swift's PHYSICAL source height —
/// `kvCacheMaxSequenceLength`, i.e. the model's `max_token_context` — which
/// is one row SHORTER than `alignment.rows()`; it is ignored under `Complete`
/// and load-bearing under `SwiftParity` (see `gather_swift_parity_into`).
/// Swift's
/// `segmentSize` and `options` parameters are unused in the function body
/// (verified against `SegmentSeeker.swift:410-496`), and `timings` is
/// only passed through to `findAlignment` (`:471`), which ignores it —
/// all three are dropped here: `timings`' duration/run-count bookkeeping
/// (`TranscribeTask.swift:214-215`) is the caller's responsibility, same
/// as at Swift's own call site.
///
/// # Errors
/// [`SegmentError::InvalidAlignmentShape`] if the prefix-take alignment
/// ends up with zero rows or columns — notably an empty (or
/// all-empty-tokens) `segments` input, which Swift's own unguarded
/// `1...0` range would instead crash on (see [`dynamic_time_warping`]'s
/// doc); [`SegmentError::Tokenizer`] if `split_to_word_tokens` or a
/// partial-special retokenize fails;
/// [`SegmentError::AlignmentPitchUnavailable`] /
/// [`SegmentError::AlignmentPitchUnexpectedLayout`] under
/// [`AlignmentGather::SwiftParity`] only, if either surface's CoreVideo row
/// pitch cannot be measured or is not a row-padded row-major layout — parity
/// cannot be claimed against a layout this port cannot describe, so it is
/// refused rather than approximated (see this module's
/// `coreml_f16_row_pitch`). The default gather returns none of the three.
#[allow(clippy::too_many_arguments)] // Mirrors Swift's addWordTimestamps argument
// surface (mirroring decode_text's own precedent for this exact lint, per its
// doc comment); no natural subset of these forms a cohesive struct without
// inventing one purely to dodge the lint.
pub fn add_word_timestamps(
  segments: &[TranscriptionSegment],
  alignment: &AlignmentView<'_>,
  tokenizer: &WhisperTokenizer,
  language_code: &str,
  grouping: WordGrouping,
  gather: AlignmentGather,
  swift_source_rows: usize,
  seek: usize,
  prepended: &str,
  appended: &str,
  last_speech_timestamp: f32,
) -> Result<Vec<TranscriptionSegment>, SegmentError> {
  // :427-442 -- flatten every segment's tokens, in order; pair each with
  // its logged log-prob only when Swift's dictionary probe would have
  // found one (`segment.tokenLogProbs[index][token] != nil`): this
  // position's logged token id equals the token actually being gathered.
  // `.get` (rather than a direct index) additionally tolerates a shorter
  // `token_log_probs_slice` -- every `TranscriptionSegment` this crate
  // constructs keeps the two parallel, so that is a defensive no-op here,
  // not an intentional behavior difference from Swift's dictionary array.
  let mut word_token_ids: Vec<u32> = Vec::new();
  let mut filtered_log_probs: Vec<f32> = Vec::new();
  for segment in segments {
    let log_probs = segment.token_log_probs_slice();
    for (index, &token) in segment.tokens_slice().iter().enumerate() {
      word_token_ids.push(token);
      if let Some(&(logged_token, log_prob)) = log_probs.get(index)
        && logged_token == token
      {
        filtered_log_probs.push(log_prob);
      }
    }
  }

  // :444-461 -- Swift's `filteredIndices` are consecutive `0..N` by
  // construction (`:432` unconditionally appends `index + indexOffset`,
  // and `:441` advances `indexOffset` by exactly the previous segment's
  // token count), so "filtering" the alignment weights collapses to a
  // prefix take over rows `0..word_token_ids.len()`. The view is the
  // backend's FIXED-size accumulator now, so `alignment.rows() == max_ctx +
  // 1 >= needed` for every real window (`needed <= 224 < 225`): the
  // `.min(needed)` below always copies all `needed` rows and this `data`
  // zero-init is defensive-only. Whichever rows this window did not commit
  // carry an earlier window's bytes or the construction-time zero, and that
  // stale/zero content is the parity-bearing payload (whisper #41) -- exactly
  // like Swift's memcpy (`:454-459`) out of its once-allocated
  // `alignmentWeights` tensor whose uncommitted rows read back as the
  // previous window's values, into the per-call zero-initialized destination
  // (`initialValue: FloatType(0)`, `:450`).
  let needed = word_token_ids.len();
  let cols = alignment.cols();
  // The doc's zero-columns promise, honored HERE: `chunks_mut(0)` below
  // panics even over an empty buffer, and `dynamic_time_warping`'s own
  // zero-shape rejection sits after this construction — too late.
  if cols == 0 {
    return Err(SegmentError::InvalidAlignmentShape(
      InvalidAlignmentShape::new(alignment.rows(), cols, alignment.data().len()),
    ));
  }
  let mut data = vec![0.0f32; needed * cols];

  // Under the opt-in `SwiftParity` the gather is REPRODUCED rather than
  // copied: what `dynamicTimeWarping` reads back out of Swift's destination
  // array is a function of BOTH surfaces' CoreVideo row pitches, and neither
  // is a constant -- see `gather_swift_parity_into` for the two heights and
  // `coreml_f16_row_pitch` for why a compiled-in alignment quantum would make
  // this mode silently non-parity on a host CoreVideo pads differently. The
  // source shape is the accumulator Swift allocates once
  // (`TextDecoder.swift:141`, `swift_source_rows` rows -- NOT this port's
  // `alignment.rows()`, which carries one extra commit slot), the destination
  // the per-call `[needed, cols]` one (`SegmentSeeker.swift:450`).
  //
  // `needed == 0` skips the probes (and the no-op gather): there is nothing to
  // reproduce, and the empty `segments` input it comes from is already
  // `SegmentError::InvalidAlignmentShape`'s -- reported below by
  // `dynamic_time_warping`, as this function's `# Errors` doc promises.
  if gather == AlignmentGather::SwiftParity && needed > 0 {
    gather_swift_parity_into(
      &mut data,
      alignment,
      needed,
      cols,
      swift_source_rows,
      &coreml_f16_row_pitch,
    )?;
  } else {
    // `Complete`, the DEFAULT: the prefix take with no gather artifacts at
    // all -- and, just as importantly, no surface allocation, no pitch
    // measurement and no host-dependent assumption anywhere on this path.
    for (row_index, row) in data
      .chunks_mut(cols)
      .enumerate()
      .take(alignment.rows().min(needed))
    {
      row.copy_from_slice(alignment.row(row_index));
    }
  }

  let filtered = AlignmentMatrix::new(data, needed, cols);

  // :465-472. The construction above guarantees `filtered.rows() ==
  // word_token_ids.len()` always -- the invariant `find_alignment` needs
  // to index `start_times`/`end_times`/`token_log_probs` in lockstep with
  // `word_token_ids` -- regardless of how `alignment.rows()` compares to
  // `word_token_ids.len()`. When `word_token_ids` is empty this makes
  // `filtered.rows() == 0`; `dynamic_time_warping` checks that
  // unconditionally, before `find_alignment`'s own `<= 1 word` early
  // return (see that function's doc), so an empty `segments` input
  // surfaces `SegmentError::InvalidAlignmentShape` here rather than
  // degrading to word-less segments.
  let mut merged = find_alignment(
    &word_token_ids,
    &filtered.view(),
    &filtered_log_probs,
    tokenizer,
    language_code,
    grouping,
  )?;

  // :474-477 -- the upstream "hack" Swift's own comment flags (reference,
  // Swift's own citation at `:474-475`: openai/whisper
  // `whisper/timing.py#L305`, commit `ba3f3cd`): constrain the
  // median/max word duration, then truncate overlong words at sentence
  // boundaries, before merging punctuation.
  let word_durations = calculate_word_duration_constraints(&merged);
  merged = truncate_long_words_at_sentence_boundaries(merged, word_durations.max_duration());

  // :480-482 -- gated on the merged ALIGNMENT being non-empty, not on
  // `prepended`/`appended` (a correction to this task's brief: Swift's
  // `if !alignment.isEmpty` reads the alignment array, not the
  // punctuation-string parameters). `merge_punctuations` is already a
  // no-op on an empty slice (see its own doc), so this gate changes
  // nothing observable -- kept only to mirror Swift's exact shape.
  if !merged.is_empty() {
    merged = merge_punctuations(&merged, prepended, appended);
  }

  // :484-493.
  update_segments_with_word_timings(
    segments,
    &merged,
    seek,
    last_speech_timestamp,
    word_durations.median(),
    word_durations.max_duration(),
    tokenizer,
  )
}

#[cfg(test)]
mod tests;