coremlit 0.2.0

Safe, synchronous CoreML runtime for macOS (CPU/GPU/Neural Engine) with opt-in on-device multimodal pipelines: speech (Whisper STT, forced alignment, speaker diarization, Silero VAD), AudioSet sound-event tagging, and audio/text/image embeddings (CLAP, granite, SigLIP)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
1451
1452
1453
1454
1455
1456
1457
1458
1459
1460
1461
1462
1463
1464
1465
1466
1467
1468
1469
1470
1471
1472
1473
1474
1475
1476
1477
1478
1479
1480
1481
1482
1483
1484
1485
1486
1487
1488
1489
1490
1491
1492
1493
1494
1495
1496
1497
1498
1499
1500
1501
1502
1503
1504
1505
1506
1507
1508
1509
1510
1511
1512
1513
1514
1515
1516
1517
1518
1519
1520
1521
1522
1523
1524
1525
1526
1527
1528
1529
1530
1531
1532
1533
1534
1535
1536
1537
1538
1539
1540
1541
1542
1543
1544
1545
1546
1547
1548
1549
1550
1551
1552
1553
1554
1555
1556
1557
1558
1559
1560
1561
1562
1563
1564
1565
1566
1567
1568
1569
1570
1571
1572
1573
1574
1575
1576
1577
1578
1579
1580
1581
1582
1583
1584
1585
1586
1587
1588
1589
1590
1591
1592
1593
1594
1595
1596
1597
1598
1599
1600
1601
1602
1603
1604
1605
1606
1607
1608
1609
1610
1611
1612
1613
1614
1615
1616
1617
1618
1619
1620
1621
1622
1623
1624
1625
1626
1627
1628
1629
1630
1631
1632
1633
1634
1635
1636
1637
1638
1639
1640
1641
1642
1643
1644
1645
1646
1647
1648
1649
1650
1651
1652
1653
1654
1655
1656
1657
1658
1659
1660
1661
1662
1663
1664
1665
1666
1667
1668
1669
1670
1671
1672
1673
1674
1675
1676
1677
1678
1679
1680
1681
1682
1683
1684
1685
1686
1687
1688
1689
1690
1691
1692
1693
1694
1695
1696
1697
1698
1699
1700
1701
1702
1703
1704
1705
1706
1707
1708
1709
1710
1711
1712
1713
1714
1715
1716
1717
1718
1719
1720
1721
1722
1723
1724
1725
1726
1727
1728
1729
1730
1731
1732
1733
1734
1735
1736
1737
1738
1739
1740
1741
1742
1743
1744
1745
1746
1747
1748
1749
1750
1751
1752
1753
1754
1755
1756
1757
1758
1759
1760
1761
1762
1763
1764
1765
1766
1767
1768
1769
1770
1771
1772
1773
1774
1775
1776
1777
1778
1779
1780
1781
1782
1783
1784
1785
1786
1787
1788
1789
1790
1791
1792
1793
1794
1795
1796
1797
1798
1799
1800
1801
1802
1803
1804
1805
1806
1807
1808
1809
1810
1811
1812
1813
1814
1815
1816
1817
1818
1819
1820
1821
1822
1823
1824
1825
1826
1827
1828
1829
1830
1831
1832
1833
1834
1835
1836
1837
1838
1839
1840
1841
1842
1843
1844
1845
1846
1847
1848
1849
1850
1851
1852
1853
1854
1855
1856
1857
1858
1859
1860
1861
1862
1863
1864
1865
1866
1867
1868
1869
1870
1871
1872
1873
1874
1875
1876
1877
1878
1879
1880
1881
1882
1883
1884
1885
1886
1887
1888
1889
1890
1891
1892
1893
1894
1895
1896
1897
1898
1899
1900
1901
1902
1903
1904
1905
1906
1907
1908
1909
1910
1911
1912
1913
1914
1915
1916
1917
1918
1919
1920
1921
1922
1923
1924
1925
1926
1927
1928
1929
1930
1931
1932
1933
1934
1935
1936
1937
1938
1939
1940
1941
1942
1943
1944
1945
1946
1947
1948
1949
1950
1951
1952
1953
1954
1955
1956
1957
1958
1959
1960
1961
1962
1963
1964
1965
1966
1967
1968
1969
1970
1971
1972
1973
1974
1975
1976
1977
1978
1979
1980
1981
1982
1983
1984
1985
1986
1987
1988
1989
1990
1991
1992
1993
1994
1995
1996
1997
1998
1999
2000
2001
2002
2003
2004
2005
2006
2007
2008
2009
2010
2011
2012
2013
2014
2015
2016
2017
2018
2019
2020
2021
2022
2023
2024
2025
2026
2027
2028
2029
2030
2031
2032
2033
2034
2035
2036
2037
2038
2039
2040
2041
2042
2043
2044
2045
2046
2047
2048
2049
2050
2051
2052
2053
2054
2055
2056
2057
2058
2059
2060
2061
2062
2063
2064
2065
2066
2067
2068
2069
2070
2071
2072
2073
2074
2075
2076
2077
2078
2079
2080
2081
2082
2083
2084
2085
2086
2087
2088
2089
2090
2091
2092
2093
2094
2095
2096
2097
2098
2099
2100
2101
2102
2103
2104
2105
2106
2107
2108
2109
2110
2111
2112
2113
2114
2115
2116
2117
2118
2119
2120
2121
2122
2123
2124
2125
2126
2127
2128
2129
2130
2131
2132
2133
2134
2135
2136
2137
2138
2139
2140
2141
2142
2143
2144
2145
2146
2147
2148
2149
2150
2151
2152
2153
2154
2155
2156
2157
2158
2159
2160
2161
2162
2163
2164
2165
2166
2167
2168
2169
2170
2171
2172
2173
2174
2175
2176
2177
2178
2179
2180
2181
2182
2183
2184
2185
2186
2187
2188
2189
2190
2191
2192
2193
2194
2195
2196
2197
2198
2199
2200
2201
2202
2203
2204
2205
2206
2207
2208
2209
2210
2211
2212
2213
2214
2215
2216
2217
2218
2219
2220
2221
2222
2223
2224
2225
2226
2227
2228
2229
2230
2231
2232
2233
2234
2235
2236
2237
2238
2239
2240
2241
2242
2243
2244
2245
2246
2247
2248
2249
2250
2251
2252
2253
2254
2255
2256
2257
2258
2259
2260
2261
2262
2263
2264
2265
2266
2267
//! Seek-point and segment slicing: turns one decoded window's
//! [`DecodingResult`] into zero or more [`TranscriptionSegment`]s and the
//! sample offset the next window should start decoding from. Ports
//! `SegmentSeeker.findSeekPointAndSegments`
//! (`argmax-oss-swift/Sources/WhisperKit/Core/Text/SegmentSeeker.swift:41-189`).
//!
//! Swift threads `timeToken`/`specialToken`/`sampleRate` in as three
//! separate parameters (`SegmentSeeker.swift:47-49`); this port reads the
//! first two off `tokenizer.special_tokens()` (already a parameter here)
//! and the third off [`crate::audio::whisper::constants::SAMPLE_RATE`], collapsing three
//! parameters into the one `tokenizer` this module already needs for
//! decoding slice text.
//!
//! Also home to the word-timestamp core: [`dynamic_time_warping`] (ports
//! `dynamicTimeWarping(withMatrix:)`, `SegmentSeeker.swift:195-278`,
//! tie-breaking in the private `min_cost_and_trace`, `:239-251`),
//! [`find_alignment`] (`:340-408`), [`merge_punctuations`] (`:280-338`),
//! [`calculate_word_duration_constraints`] and
//! [`truncate_long_words_at_sentence_boundaries`] (`:498-526`),
//! [`update_segments_with_word_timings`] (`:528-659`) — the final
//! re-anchoring step that walks a word-index cursor shared across every
//! segment, applies the short-word pull-back and pause/boundary
//! heuristics, and writes the results back onto [`TranscriptionSegment`]s
//! — and [`add_word_timestamps`] (`:410-496`), the orchestration wrapper
//! that flattens each segment's tokens/log-probs into the flat index list
//! used to filter the raw CoreML alignment-weights array, then threads
//! that through [`find_alignment`] -> the duration/truncation hack ->
//! [`merge_punctuations`] -> [`update_segments_with_word_timings`] in
//! sequence. [`crate::audio::whisper::transcribe::TranscribeTask::run`]'s window loop
//! calls `add_word_timestamps` directly (`TranscribeTask.swift:196-233`)
//! when `options.word_timestamps()` is set.

use unicode_categories::UnicodeCategories;

use crate::{
  MultiArray,
  audio::whisper::{
    backend::{AlignmentMatrix, AlignmentView},
    constants::{SAMPLE_RATE, SECONDS_PER_TIME_TOKEN},
    decode::AlignmentRows,
    error::{
      AlignmentPitchUnavailable, AlignmentPitchUnexpectedLayout, InvalidAlignmentShape,
      SegmentError,
    },
    options::{AlignmentGather, DecodingOptions, WordGrouping},
    result::{DecodingResult, TranscriptionSegment, WordTiming},
    tokenizer::WhisperTokenizer,
  },
};

/// The seconds span of one decode window's own audio: `samples` unpadded
/// samples starting `seek` samples into the run.
///
/// ONE rounding contract, shared by [`find_seek_point_and_segments`]'s segment
/// timing and by the language observation
/// [`crate::audio::whisper::transcribe::TranscribeTask::run`] records for the
/// same window, so the two can never disagree about the window they both
/// describe (coremlit issue #107).
///
/// The start is `seek` converted once; the end is that start PLUS the
/// duration, never the sum `seek + samples` converted as a whole. Above 2^24
/// samples (~17.5 min) adjacent sample indices share one f32, so converting
/// the sum collapses a short final window onto its own start — a one-sample
/// window at 1050 s reports `1050.0 .. 1050.0` although the decoder saw audio.
///
/// A nonempty window always gets `end > start`. From 2048 s on, one f32 ulp
/// of the start exceeds a one-sample duration and even the addition absorbs
/// it, so the end is nudged to the next representable value instead. Swift's
/// `SegmentSeeker` computes the addition without that guard and reports the
/// zero-length span (documented deviation, in a regime where the window did
/// consume audio and Swift's own `:217-218` filter deletes what it produced).
pub(crate) fn window_span(seek: usize, samples: usize) -> (f32, f32) {
  let start = seek as f32 / SAMPLE_RATE as f32;
  let end = start + samples as f32 / SAMPLE_RATE as f32;
  if samples > 0 && end <= start {
    return (start, start.next_up());
  }
  (start, end)
}

/// The same span re-anchored `offset_seconds` later — the ONE re-anchoring
/// contract, shared by every span
/// [`crate::audio::whisper::audio::chunker::apply_result_seek_offset`] lifts
/// out of a VAD chunk's own timeline into the original one.
///
/// [`window_span`] guarantees a window that held audio reports `end > start`;
/// adding the offset to each endpoint on its own throws that guarantee away
/// again, because the two additions round independently. From 2048 s on, one
/// f32 ulp of the shifted start (2.44e-4 s) exceeds a short window's whole
/// duration, so a valid local span `0.0 .. 1/16000` shifted by 32_768_000
/// samples lands on `2048.0 .. 2048.0` — a kept probe, or a lump segment,
/// whose evidence is zero-width although the decoder saw audio. This helper
/// preserves the guarantee across the shift: a span that came in nonempty
/// leaves nonempty, its end nudged to the next representable value above the
/// shifted start rather than onto it. Observations and segments both re-anchor
/// through it, so the two can no more disagree about the window they describe
/// after the shift than [`window_span`] lets them before it.
///
/// A span that came in EMPTY (`end == start`) is left exactly as shifted, with
/// no nudge. The segment path can legitimately hold one — a zero-length
/// segment survives until
/// [`crate::audio::whisper::transcribe::TranscribeTask::run`]'s `:217-218`
/// filter drops it — and inventing an extent here would hide it from that
/// filter.
pub(crate) fn shift_span(start: f32, end: f32, offset_seconds: f32) -> (f32, f32) {
  let shifted_start = start + offset_seconds;
  let shifted_end = end + offset_seconds;
  if end > start && shifted_end <= shifted_start {
    return (shifted_start, shifted_start.next_up());
  }
  (shifted_start, shifted_end)
}

/// Turns `decoding` — the just-decoded window starting at `current_seek`
/// samples — into the next seek offset and, unless the window was silent,
/// the [`TranscriptionSegment`]s it contains. `all_segments_count` seeds
/// each new segment's [`TranscriptionSegment::id`] so ids stay unique
/// across every window a caller has already processed; `segment_size` is
/// the window's length in samples (normally
/// [`crate::audio::whisper::constants::WINDOW_SAMPLES`], smaller for a final short
/// window).
///
/// Three phases, ported structure-preserving from `findSeekPointAndSegments`:
///
/// 1. **Silence skip** (`SegmentSeeker.swift:57-74`): if
///    `options.no_speech_threshold()` is set and `decoding.no_speech_prob()`
///    exceeds it, the whole window is dropped — seek advances by
///    `segment_size` and `None` is returned — *unless*
///    `options.logprob_threshold()` is also set and `decoding.avg_logprob()`
///    exceeds *that*, which overrides the skip (confident text beats a
///    high no-speech probability).
/// 2. **Consecutive-timestamp slicing** (`:79-148`): otherwise, adjacent
///    timestamp-token pairs in `decoding.tokens_slice()` mark segment
///    boundaries. A lone trailing timestamp (single-timestamp ending) or a
///    trailing run of plain tokens (no-timestamp ending) each contribute
///    one final boundary of their own. Each resulting slice becomes a
///    segment spanning its first-to-last timestamp token; seek advances to
///    the last timestamp found (or by `segment_size` on a no-timestamp
///    ending).
/// 3. **Lump fallback** (`:149-186`): if no consecutive timestamp pair
///    exists at all, the whole window becomes one segment, its end time
///    refined by the last timestamp token above `<|0.00|>` if any exists;
///    seek always advances by `segment_size` on this path.
///
/// # Errors
/// [`SegmentError::Tokenizer`] if decoding a slice's tokens back to text
/// fails.
pub fn find_seek_point_and_segments(
  decoding: &DecodingResult,
  options: &DecodingOptions,
  all_segments_count: usize,
  current_seek: usize,
  segment_size: usize,
  tokenizer: &WhisperTokenizer,
) -> Result<(usize, Option<Vec<TranscriptionSegment>>), SegmentError> {
  let special = tokenizer.special_tokens();
  let time_token = special.time_token_begin();
  let special_token_begin = special.special_token_begin();
  let mut seek = current_seek;
  let (time_offset, window_end) = window_span(current_seek, segment_size);

  // :57-74 — silence skip: no-speech probability above threshold skips the
  // window entirely, unless overridden by high average confidence.
  if let Some(threshold) = options.no_speech_threshold() {
    let mut should_skip = decoding.no_speech_prob() > threshold;
    if let Some(logprob_threshold) = options.logprob_threshold()
      && decoding.avg_logprob() > logprob_threshold
    {
      should_skip = false;
    }
    if should_skip {
      return Ok((seek + segment_size, None));
    }
  }

  let current_tokens = decoding.tokens_slice();
  let current_log_probs = decoding.token_log_probs_slice();
  let is_timestamp_token: Vec<bool> = current_tokens.iter().map(|&t| t >= time_token).collect();

  // :84-86 — the ending shape decides whether/how a trailing boundary is
  // synthesized below. A slice with fewer than 3 tokens can match neither
  // pattern, exactly like Swift's `Array == [Bool]` on a short `suffix`.
  let single_timestamp_ending = matches!(is_timestamp_token.as_slice(), [.., false, true, false]);
  let no_timestamp_ending = matches!(is_timestamp_token.as_slice(), [.., false, false, false]);

  // :88-97 — end index of every consecutive timestamp-token pair.
  let mut slice_indexes: Vec<usize> = Vec::new();
  let mut previous_is_timestamp = false;
  for (index, &is_timestamp) in is_timestamp_token.iter().enumerate() {
    if previous_is_timestamp && is_timestamp {
      slice_indexes.push(index);
    }
    previous_is_timestamp = is_timestamp;
  }

  let mut segments: Vec<TranscriptionSegment> = Vec::new();

  if slice_indexes.is_empty() {
    // :149-186 — no consecutive timestamps anywhere: lump the whole window
    // into one segment.
    // The window's own span end, unless a timestamp token below refines it.
    let mut segment_end = window_end;
    let timestamp_tokens: Vec<u32> = current_tokens
      .iter()
      .copied()
      .filter(|&t| t > time_token)
      .collect();
    if let Some(&last_timestamp) = timestamp_tokens.last() {
      segment_end = time_offset + (last_timestamp - time_token) as f32 * SECONDS_PER_TIME_TOKEN;
    }

    let word_tokens: Vec<u32> = current_tokens
      .iter()
      .copied()
      .filter(|&t| t < special_token_begin)
      .collect();
    let segment_text_tokens: &[u32] = if options.skip_special_tokens() {
      &word_tokens
    } else {
      current_tokens
    };
    let segment_text = tokenizer.decode(segment_text_tokens, false)?;

    segments.push(
      TranscriptionSegment::new()
        .with_id(all_segments_count + segments.len())
        .with_seek(seek)
        .with_start(time_offset)
        .with_end(segment_end)
        .with_text(segment_text)
        .with_tokens(current_tokens)
        .with_token_log_probs(current_log_probs)
        .with_temperature(decoding.temperature())
        .with_avg_logprob(decoding.avg_logprob())
        .with_compression_ratio(decoding.compression_ratio())
        .with_no_speech_prob(decoding.no_speech_prob()),
    );

    // Model gave no consecutive-timestamp boundary, so the whole window is
    // consumed regardless of the refined end above (Swift's own
    // upstream TODO at `:184-185` notes the more accurate
    // `durationSeconds`-based advance is not yet used either).
    seek += segment_size;
  } else {
    // :101-107 — a lone trailing timestamp or trailing run of plain tokens
    // each need one more boundary appended beyond what the main loop above
    // found, to cover the window's tail as a final slice.
    if single_timestamp_ending {
      let single_ending_index = is_timestamp_token
        .iter()
        .rposition(|&t| t)
        .expect("single_timestamp_ending's pattern requires a `true` entry");
      slice_indexes.push(single_ending_index + 1);
    } else if no_timestamp_ending {
      slice_indexes.push(current_tokens.len());
    }

    let mut last_slice_start = 0usize;
    for &current_slice_end in &slice_indexes {
      let sliced_tokens = &current_tokens[last_slice_start..current_slice_end];
      let sliced_log_probs = &current_log_probs[last_slice_start..current_slice_end];

      // Every slice here is bounded by a detected timestamp-pair boundary
      // (this loop's own start) or ends at one (the main loop above only
      // ever records the second index of a `true, true` pair), so it
      // always contains at least one timestamp token — the same invariant
      // Swift trusts with `timestampTokens.first!`/`.last!`.
      let timestamp_tokens: Vec<u32> = sliced_tokens
        .iter()
        .copied()
        .filter(|&t| t >= time_token)
        .collect();
      let start_ts = *timestamp_tokens
        .first()
        .expect("slice bounded by a timestamp pair contains a timestamp token");
      let end_ts = *timestamp_tokens
        .last()
        .expect("slice bounded by a timestamp pair contains a timestamp token");
      let start_seconds = (start_ts - time_token) as f32 * SECONDS_PER_TIME_TOKEN;
      let end_seconds = (end_ts - time_token) as f32 * SECONDS_PER_TIME_TOKEN;

      let word_tokens: Vec<u32> = sliced_tokens
        .iter()
        .copied()
        .filter(|&t| t < special_token_begin)
        .collect();
      let sliced_text_tokens: &[u32] = if options.skip_special_tokens() {
        &word_tokens
      } else {
        sliced_tokens
      };
      let slice_text = tokenizer.decode(sliced_text_tokens, false)?;

      segments.push(
        TranscriptionSegment::new()
          .with_id(all_segments_count + segments.len())
          .with_seek(seek)
          .with_start(time_offset + start_seconds)
          .with_end(time_offset + end_seconds)
          .with_text(slice_text)
          .with_tokens(sliced_tokens)
          .with_token_log_probs(sliced_log_probs)
          .with_temperature(decoding.temperature())
          .with_avg_logprob(decoding.avg_logprob())
          .with_compression_ratio(decoding.compression_ratio())
          .with_no_speech_prob(decoding.no_speech_prob()),
      );

      last_slice_start = current_slice_end;
    }

    // :140-148 — seek to the last timestamp found, unless the tail was an
    // unbounded run of plain tokens (no-timestamp ending), which instead
    // consumes the full window like the lump branch does.
    if no_timestamp_ending {
      seek += segment_size;
    } else {
      let last_index = last_slice_start - usize::from(single_timestamp_ending);
      let last_timestamp_token = current_tokens[last_index] - time_token;
      let last_timestamp_seconds = last_timestamp_token as f32 * SECONDS_PER_TIME_TOKEN;
      seek += (last_timestamp_seconds * SAMPLE_RATE as f32) as usize;
    }
  }

  Ok((seek, Some(segments)))
}

// ---------------------------------------------------------------------
// Word timestamps: DTW alignment, merge_punctuations, find_alignment
// ---------------------------------------------------------------------

/// One decoded-token/audio-frame alignment path out of
/// [`dynamic_time_warping`]'s cost-matrix backtrace: parallel
/// `text_indices`/`time_indices` sequences of equal length, walking from
/// the matrix's first aligned position to `(rows - 1, cols - 1)`. `isize`
/// mirrors the `-1` entries Swift's border-walk code shape permits
/// (`SegmentSeeker.swift:260-262`); in practice both cursors reach `0`
/// together via the unique `(1, 1) -> (0, 0)` step, so valid inputs never
/// produce one.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct DtwPath {
  text_indices: Vec<isize>,
  time_indices: Vec<isize>,
}

impl DtwPath {
  /// Per-step decoded-token-row index (parallel to
  /// [`Self::time_indices_slice`]).
  #[inline(always)]
  pub fn text_indices_slice(&self) -> &[isize] {
    self.text_indices.as_slice()
  }

  /// Per-step audio-frame-column index (parallel to
  /// [`Self::text_indices_slice`]).
  #[inline(always)]
  pub fn time_indices_slice(&self) -> &[isize] {
    self.time_indices.as_slice()
  }
}

/// The three-way step cost and its winning direction for one DTW cell —
/// `0` diagonal, `1` up, `2` left. Ports `minCostAndTrace`
/// (`SegmentSeeker.swift:239-251`) exactly, including its tie-break order:
/// diagonal wins only by being strictly less than BOTH alternatives, then
/// up wins only by being strictly less than BOTH alternatives, and
/// everything else — including every exact tie — falls to the final
/// `else` and picks left. This `if`/`else if`/`else` shape (not a
/// three-way `min`) is what makes left the tie winner; it must not be
/// reordered or loosened to `<=`.
///
/// Swift computes `c0 = diagonal + value`, `c1 = up + value`, `c2 = left +
/// value` up front and compares THOSE (`SegmentSeeker.swift:239-251`) —
/// and this port does the same, because adding a common finite value is
/// not order-preserving in floating point: a large-magnitude `value` can
/// round distinct incoming costs into exact ties, and exact ties fall to
/// left. Comparing the bare incoming costs picked a different winner on
/// such inputs (phase-gate finding; pinned by
/// `dtw_add_before_compare_matches_swift_rounding_ties`).
fn min_cost_and_trace(diagonal: f64, up: f64, left: f64, value: f64) -> (f64, i8) {
  let c0 = diagonal + value;
  let c1 = up + value;
  let c2 = left + value;
  if c0 < c1 && c0 < c2 {
    (c0, 0)
  } else if c1 < c0 && c1 < c2 {
    (c1, 1)
  } else {
    (c2, 2)
  }
}

/// Dynamic time warping over a decoded-token x audio-frame cross-attention
/// alignment matrix. Builds a `(rows + 1) x (cols + 1)` cost/trace
/// matrix — Swift's nested `[[Double]]`/`[[Int]]`, flattened here into
/// row-major `Vec<f64>`/`Vec<i8>` indexed `row * (cols + 1) + col` for the
/// same values without per-row allocation — then backtraces it into a
/// [`DtwPath`]. Ports `dynamicTimeWarping(withMatrix:)`
/// (`SegmentSeeker.swift:195-237`, backtrace `:253-278`); cell
/// tie-breaking is the private `min_cost_and_trace` (`:239-251`).
///
/// # Errors
/// [`SegmentError::InvalidAlignmentShape`] if `matrix` has zero rows or
/// columns. Swift has no equivalent guard: `1...numberOfColumns`/
/// `1...numberOfRows` over a zero dimension is an invalid `ClosedRange`
/// and traps at runtime. This is a deliberate improvement over that
/// crash — a typed, recoverable error on the same malformed input.
pub fn dynamic_time_warping(matrix: &AlignmentView<'_>) -> Result<DtwPath, SegmentError> {
  let (rows, cols) = (matrix.rows(), matrix.cols());
  if rows == 0 || cols == 0 {
    return Err(SegmentError::InvalidAlignmentShape(
      InvalidAlignmentShape::new(rows, cols, matrix.data().len()),
    ));
  }

  let width = cols + 1;
  let mut cost = vec![f64::INFINITY; (rows + 1) * width];
  let mut trace = vec![-1i8; (rows + 1) * width];
  cost[0] = 0.0;
  for cell in &mut trace[1..=cols] {
    *cell = 2; // :208-210 -- top border backtraces LEFT.
  }
  for i in 1..=rows {
    trace[i * width] = 1; // :211-213 -- left border backtraces UP.
  }

  for row in 1..=rows {
    for column in 1..=cols {
      // :217 -- MLMultiArray's flat linear index; equivalent to this
      // AlignmentView's row-major `row(row - 1)[column - 1]`.
      let value = -f64::from(matrix.row(row - 1)[column - 1]);
      let diagonal = cost[(row - 1) * width + column - 1];
      let up = cost[(row - 1) * width + column];
      let left = cost[row * width + column - 1];
      let (best, direction) = min_cost_and_trace(diagonal, up, left, value);
      cost[row * width + column] = best;
      trace[row * width + column] = direction;
    }
  }

  // :253-278 -- backtrace from the bottom-right corner to the origin.
  let (mut i, mut j) = (rows, cols);
  let mut text_indices = Vec::new();
  let mut time_indices = Vec::new();
  while i > 0 || j > 0 {
    text_indices.push(i as isize - 1);
    time_indices.push(j as isize - 1);
    match trace[i * width + j] {
      0 => {
        i -= 1;
        j -= 1;
      }
      1 => i -= 1,
      2 => j -= 1,
      // Unreachable for any (i, j) this loop actually visits: every cell
      // but (0, 0) -- never read, since the loop condition stops there --
      // was written by the border init or the main loop above to 0/1/2.
      // Kept as a defensive exit; Swift's `default: break` only exits the
      // `switch` there, which would spin forever instead if this were
      // ever hit.
      _ => break,
    }
  }
  text_indices.reverse();
  time_indices.reverse();

  Ok(DtwPath {
    text_indices,
    time_indices,
  })
}

/// True where Swift's `String.contains(String)` is: substring search that
/// treats an empty needle as never found (`String`'s/`NSString`'s
/// `range(of: "")` is documented to return no match), unlike
/// `str::contains`, for which an empty pattern matches everywhere.
/// [`merge_punctuations`]'s punctuation-membership checks need Swift's
/// behavior to stay exact for a word that trims to nothing.
fn swift_contains(haystack: &str, needle: &str) -> bool {
  !needle.is_empty() && haystack.contains(needle)
}

/// Trims Swift's `.whitespaces` `CharacterSet` (Unicode general category
/// `Zs` plus U+0009 CHARACTER TABULATION; no newlines) off both ends of
/// `s`. Same predicate as `tokenizer::is_single_punctuation_scalar`'s trim
/// step, duplicated here because that helper is private to its module.
fn trim_swift_whitespaces(s: &str) -> &str {
  s.trim_matches(|c: char| c.is_separator_space() || c == '\u{0009}')
}

/// Merges leading/trailing punctuation-only words in `alignment` onto
/// their neighboring word, then drops the words that end up empty or are
/// themselves bare merged-away punctuation. Ports `mergePunctuations`
/// (`SegmentSeeker.swift:280-338`) in its exact two-pass shape: characters
/// in `prepended` glue onto the FOLLOWING word (`:291-315`), characters in
/// `appended` glue onto the PRECEDING word (`:322-333`), then a final
/// filter drops the leftovers (`:336`).
///
/// Both passes replicate a Swift quirk rather than smoothing it over: each
/// iteration reads its merge neighbor from the *original* pre-merge
/// slice — `alignment[i - 1]` in the prepend pass, `prependedAlignment[i -
/// 1]` in the append pass — never from the tail of the list actually being
/// built. A run of three or more consecutive punctuation-only words
/// therefore does not fully chain together in either this port or
/// upstream Swift: only the immediately preceding original word survives
/// a second merge. Whisper's fixed single-character punctuation
/// vocabularies make three consecutive punctuation-only *words*
/// essentially unreachable in practice, so this port keeps Swift's exact
/// indexing rather than silently changing the observable behavior.
pub fn merge_punctuations(
  alignment: &[WordTiming],
  prepended: &str,
  appended: &str,
) -> Vec<WordTiming> {
  if alignment.is_empty() {
    return Vec::new();
  }

  // :291-315 -- merge PREPEND punctuation onto the following word.
  let mut prepended_alignment: Vec<WordTiming> = Vec::new();
  if !swift_contains(prepended, trim_swift_whitespaces(alignment[0].word())) {
    prepended_alignment.push(alignment[0].clone());
  }
  for pair in alignment.windows(2) {
    let previous = &pair[0];
    let current = &pair[1];
    let previous_starts_with_whitespace = previous
      .word()
      .chars()
      .next()
      .is_some_and(|c| c.is_separator_space() || c == '\u{0009}');
    if previous_starts_with_whitespace
      && swift_contains(prepended, trim_swift_whitespaces(previous.word()))
    {
      let mut word = previous.word().to_string();
      word.push_str(current.word());
      let mut tokens = previous.tokens_slice().to_vec();
      tokens.extend_from_slice(current.tokens_slice());
      let merged = WordTiming::new(
        word,
        tokens,
        current.start(),
        current.end(),
        current.probability(),
      );
      if prepended_alignment.is_empty() {
        prepended_alignment.push(merged);
      } else {
        let last = prepended_alignment.len() - 1;
        prepended_alignment[last] = merged;
      }
    } else {
      prepended_alignment.push(current.clone());
    }
  }

  // :317-333 -- merge APPEND punctuation onto the preceding word.
  let mut appended_alignment: Vec<WordTiming> = Vec::new();
  if let Some(first) = prepended_alignment.first() {
    appended_alignment.push(first.clone());
  }
  for pair in prepended_alignment.windows(2) {
    let previous = &pair[0];
    let current = &pair[1];
    if !previous.word().ends_with(' ')
      && swift_contains(appended, trim_swift_whitespaces(current.word()))
    {
      let mut word = previous.word().to_string();
      word.push_str(current.word());
      let mut tokens = previous.tokens_slice().to_vec();
      tokens.extend_from_slice(current.tokens_slice());
      let merged = WordTiming::new(
        word,
        tokens,
        previous.start(),
        previous.end(),
        previous.probability(),
      );
      let last = appended_alignment.len() - 1;
      appended_alignment[last] = merged;
    } else {
      appended_alignment.push(current.clone());
    }
  }

  // :336 -- drop empties and bare merged-away punctuation words.
  appended_alignment
    .into_iter()
    .filter(|w| {
      !w.word().is_empty()
        && !swift_contains(appended, w.word())
        && !swift_contains(prepended, w.word())
    })
    .collect()
}

/// Word-level timestamps for one window's decoded tokens: runs
/// [`dynamic_time_warping`] over `alignment`, groups `word_token_ids` into
/// words via [`WhisperTokenizer::split_to_word_tokens`], and reads each
/// word's start/end time off the DTW path's per-token-row boundaries plus
/// its mean sampled-token log probability. Ports `findAlignment`
/// (`SegmentSeeker.swift:340-408`); `language_code` is threaded straight
/// into `split_to_word_tokens` in place of Swift's internal
/// `NLLanguageRecognizer` detection (spec §5.3; see
/// [`WhisperTokenizer::split_to_word_tokens`]'s own doc for the same
/// substitution there), and `grouping` chooses how that splitter groups
/// (coremlit issue #14 — [`WordGrouping::SwiftParity`] is the default after
/// #41; [`WordGrouping::FineGrained`] is this port's long-standing opt-in).
///
/// Returns an empty vec when `split_to_word_tokens` groups `word_token_ids`
/// into one word or fewer (`:351-353`) — DTW timing is meaningless for a
/// single undivided span. DTW itself still runs first regardless (matching
/// Swift's own unconditional call order), so a malformed `alignment`
/// still errors even on that trivial path.
///
/// # Errors
/// [`SegmentError::InvalidAlignmentShape`] if `alignment` has zero rows or
/// columns (from [`dynamic_time_warping`]); [`SegmentError::Tokenizer`] if
/// `split_to_word_tokens` fails to decode `word_token_ids`.
pub fn find_alignment(
  word_token_ids: &[u32],
  alignment: &AlignmentView<'_>,
  token_log_probs: &[f32],
  tokenizer: &WhisperTokenizer,
  language_code: &str,
  grouping: WordGrouping,
) -> Result<Vec<WordTiming>, SegmentError> {
  Ok(
    find_alignment_spanned(
      word_token_ids,
      alignment,
      token_log_probs,
      tokenizer,
      language_code,
      grouping,
    )?
    .into_iter()
    .map(|(word, _)| word)
    .collect(),
  )
}

/// A word of an alignment with the half-open span of positions its tokens
/// came from in the tokens the alignment read.
pub(crate) type SpannedWord = (WordTiming, (usize, usize));

/// [`find_alignment`], each word with the half-open span of positions in
/// `word_token_ids` its tokens came from.
pub(crate) fn find_alignment_spanned(
  word_token_ids: &[u32],
  alignment: &AlignmentView<'_>,
  token_log_probs: &[f32],
  tokenizer: &WhisperTokenizer,
  language_code: &str,
  grouping: WordGrouping,
) -> Result<Vec<SpannedWord>, SegmentError> {
  Ok(
    find_alignment_timed(
      word_token_ids,
      alignment,
      token_log_probs,
      tokenizer,
      language_code,
      grouping,
    )?
    .words,
  )
}

/// An alignment's words, each with its span ([`SpannedWord`]), beside what
/// the alignment gave each token it read: its start and end in seconds of
/// the window, by the token's position — the boundaries a word's own times
/// are read from (`SegmentSeeker.swift:356-371`). Empty where the alignment
/// has no words.
pub(crate) struct TimedAlignment {
  words: Vec<SpannedWord>,
  starts: Vec<f32>,
  ends: Vec<f32>,
}

/// [`find_alignment_spanned`], with each token's start and end beside the
/// words ([`TimedAlignment`]).
fn find_alignment_timed(
  word_token_ids: &[u32],
  alignment: &AlignmentView<'_>,
  token_log_probs: &[f32],
  tokenizer: &WhisperTokenizer,
  language_code: &str,
  grouping: WordGrouping,
) -> Result<TimedAlignment, SegmentError> {
  let path = dynamic_time_warping(alignment)?;
  let text_indices = path.text_indices_slice();
  let time_indices = path.time_indices_slice();

  let word_tokens = tokenizer.split_to_word_tokens(word_token_ids, language_code, grouping)?;
  if word_tokens.len() <= 1 {
    return Ok(TimedAlignment {
      words: Vec::new(),
      starts: Vec::new(),
      ends: Vec::new(),
    });
  }

  // :356-371 -- per-decoded-token-row start/end times: one boundary each
  // time the DTW path's aligned row changes.
  let mut start_times: Vec<f32> = vec![0.0];
  let mut end_times: Vec<f32> = Vec::new();
  let mut current_text_index = text_indices.first().copied().unwrap_or(0);
  for (index, &text_index) in text_indices.iter().enumerate() {
    if text_index != current_text_index {
      current_text_index = text_index;
      let time = time_indices[index] as f32 * SECONDS_PER_TIME_TOKEN;
      start_times.push(time);
      end_times.push(time);
    }
  }
  end_times.push(time_indices.last().copied().unwrap_or(1500) as f32 * SECONDS_PER_TIME_TOKEN);

  // :373-405 -- walk word groups; each consumes `tokens.len()` rows of
  // start_times/end_times/token_log_probs.
  let mut word_timings: Vec<SpannedWord> = Vec::with_capacity(word_tokens.len());
  let mut current_token_index = 0usize;
  for (word, tokens) in word_tokens {
    let start_index = current_token_index;
    let word_start_time = start_times[current_token_index];
    current_token_index += tokens.len() - 1;
    let word_end_time = end_times[current_token_index];
    current_token_index += 1;

    let probs = &token_log_probs[start_index..current_token_index];
    let mean_log_prob = probs.iter().sum::<f32>() / probs.len() as f32;

    word_timings.push((
      WordTiming::new(
        word,
        tokens,
        word_start_time,
        word_end_time,
        mean_log_prob.exp(),
      ),
      (start_index, current_token_index),
    ));
  }

  Ok(TimedAlignment {
    words: word_timings,
    starts: start_times,
    ends: end_times,
  })
}

// ---------------------------------------------------------------------
// Word-duration constraints and sentence-boundary truncation
// ---------------------------------------------------------------------

/// The capped-median/max word-duration pair [`calculate_word_duration_constraints`]
/// computes over one window's word alignment — Swift's anonymous
/// `(median: Float, max: Float)` tuple return (`SegmentSeeker.swift:
/// 498-507`). `Copy`, and deliberately has no constructor: unlike an
/// options type, whose fields a caller assembles piecewise before the
/// fact, both fields here only ever come out of that one function
/// together, and `max` is always exactly `median * 2` — a public
/// constructor would let a caller build a pair that violates that
/// invariant. This type is a computation RESULT, not configuration; that
/// is a deliberate departure from the options pattern, not an oversight.
#[derive(Debug, Clone, Copy, PartialEq)]
pub struct WordDurationConstraints {
  median: f32,
  max: f32,
}

impl WordDurationConstraints {
  /// The capped median word duration: `min(0.7, raw median)`, in seconds,
  /// or `0.0` when the source alignment had no positive-duration word
  /// (`SegmentSeeker.swift:502-503`).
  #[inline(always)]
  pub const fn median(&self) -> f32 {
    self.median
  }

  /// The overlong-word threshold: always exactly twice [`Self::median`]
  /// (`SegmentSeeker.swift:504`), consumed by
  /// [`truncate_long_words_at_sentence_boundaries`].
  #[inline(always)]
  pub const fn max_duration(&self) -> f32 {
    self.max
  }
}

/// Computes the capped-median/max word-duration pair used to flag overlong
/// words. Ports `calculateWordDurationConstraints`
/// (`SegmentSeeker.swift:498-507`): every word whose `duration` is not
/// strictly positive is dropped before the median is taken (`:499-500`,
/// `filter { $0 > 0 }`); the median is the sorted list's UPPER middle
/// element — `sorted[count / 2]`, integer division, NOT an average of the
/// two middle values on an even count (`:502`); that raw median is then
/// capped at `0.7` s (`:503`), and `max` is always twice the CAPPED
/// value, never the raw one (`:504`). An empty `alignment`, or one where
/// every word has zero or negative duration, yields `median = max = 0.0`.
pub fn calculate_word_duration_constraints(alignment: &[WordTiming]) -> WordDurationConstraints {
  let mut durations: Vec<f32> = alignment
    .iter()
    .map(WordTiming::duration)
    .filter(|&duration| duration > 0.0)
    .collect();
  durations.sort_by(f32::total_cmp);

  let raw_median = durations.get(durations.len() / 2).copied().unwrap_or(0.0);
  let median = raw_median.min(0.7);
  let max = median * 2.0;

  WordDurationConstraints { median, max }
}

/// Sentence-ending marks [`truncate_long_words_at_sentence_boundaries`]
/// matches a word's text against EXACTLY — no substring or trimmed match
/// (Swift `sentenceEndMarks`, `SegmentSeeker.swift:510`; includes the CJK
/// full-width punctuation forms alongside the ASCII ones).
const SENTENCE_END_MARKS: [&str; 6] = [".", "。", "!", "!", "?", "?"];

/// Clips words whose duration exceeds `max_duration` back down to it, but
/// only where a sentence boundary justifies the clip — guards against a
/// single misaligned DTW timestamp stretching one word far past its real
/// span. Ports `truncateLongWordsAtSentenceBoundaries`
/// (`SegmentSeeker.swift:509-526`) structure-preserving: index `0` is
/// never inspected or modified (the loop runs `1..alignment.len()`,
/// `:514`; Rust's half-open `Range` is simply empty rather than panicking
/// when `alignment` itself is empty, so no separate emptiness guard is
/// needed the way Swift's `1..<0` `ClosedRange` would require). For each
/// later word whose `duration()` exceeds `max_duration`:
/// - if the word's own text EXACTLY matches a mark in the internal
///   `SENTENCE_END_MARKS` table (whole-word — a caller passing `" ."` with
///   a leading space does not qualify; note that words reaching here from
///   [`find_alignment`] never carry that space, because
///   [`WhisperTokenizer::decode`]'s tokenization-space cleanup strips it,
///   just as Swift's does), its `end` is pulled back to
///   `start + max_duration`;
/// - otherwise, if the PRECEDING word's text exactly matches a mark, this
///   word's `start` is pushed forward to `end - max_duration`.
///
/// These are `if`/`else if` branches (`:516-520`), so a word that is
/// itself a mark AND immediately follows another mark only ever takes the
/// first branch: its `end` is adjusted, its `start` never is.
pub fn truncate_long_words_at_sentence_boundaries(
  mut alignment: Vec<WordTiming>,
  max_duration: f32,
) -> Vec<WordTiming> {
  for i in 1..alignment.len() {
    if alignment[i].duration() > max_duration {
      if SENTENCE_END_MARKS.contains(&alignment[i].word()) {
        let start = alignment[i].start();
        alignment[i].set_end(start + max_duration);
      } else if SENTENCE_END_MARKS.contains(&alignment[i - 1].word()) {
        let end = alignment[i].end();
        alignment[i].set_start(end - max_duration);
      }
    }
  }
  alignment
}

/// Re-anchors DTW word-level `merged_alignment` timings onto `segments`,
/// applying Swift's short-word pull-back and pause/boundary heuristics
/// along the way. Ports `updateSegmentsWithWordTimings`
/// (`SegmentSeeker.swift:528-659`) — the last stage of
/// [`add_word_timestamps`] (`:410-496`), which runs it after
/// `findAlignment` -> the duration/truncation hack -> `mergePunctuations`.
///
/// `merged_alignment` is walked with a cursor SHARED across every
/// `segments` entry (`:538`, Swift's `wordIndex`): once an alignment entry
/// is consumed by one segment it is never revisited by a later one, even
/// if that segment's own token budget is not fully accounted for (the
/// cursor simply runs out and that segment's word list ends up short).
/// `last_speech_timestamp` is similarly threaded across every segment,
/// seeded by the caller's initial value and updated to each segment's own
/// final `end` once it gets at least one word (`:651`, guarded by the same
/// non-empty check as the pause/boundary hack itself — a wordless segment
/// leaves `last_speech_timestamp` untouched for the next one).
///
/// # Errors
/// [`SegmentError::Tokenizer`] if retokenizing a partially-special-filtered
/// alignment entry's surviving tokens fails (`:556-559`).
pub fn update_segments_with_word_timings(
  segments: &[TranscriptionSegment],
  merged_alignment: &[WordTiming],
  seek: usize,
  last_speech_timestamp: f32,
  constrained_median_duration: f32,
  max_duration: f32,
  tokenizer: &WhisperTokenizer,
) -> Result<Vec<TranscriptionSegment>, SegmentError> {
  // :537 -- this window's seek offset, in seconds.
  let time_offset = seek as f32 / SAMPLE_RATE as f32;
  // :538 -- cursor into `merged_alignment`, shared across every segment
  // below; never reset per segment. `.min(merged_alignment.len())` guards a
  // slice a Swift `mergedAlignment[wordIndex...]` has no equivalent for (it
  // would trap on an out-of-range `wordIndex`); the invariant that
  // `word_index` never exceeds `merged_alignment.len()` holds by
  // construction (a segment consumes elements of the remaining slice only),
  // so this is a zero-cost safety net, not a behavior change.
  let mut word_index = 0usize;
  let mut last_speech_timestamp = last_speech_timestamp;
  let mut updated_segments: Vec<TranscriptionSegment> = Vec::with_capacity(segments.len());
  for segment in segments {
    let (updated_segment, consumed) = update_segment_with_word_timings(
      segment,
      &merged_alignment[word_index.min(merged_alignment.len())..],
      updated_segments.last().map(TranscriptionSegment::end),
      time_offset,
      &mut last_speech_timestamp,
      constrained_median_duration,
      max_duration,
      tokenizer,
    )?;
    word_index += consumed;
    updated_segments.push(updated_segment);
  }
  Ok(updated_segments)
}

/// One segment's pass of [`update_segments_with_word_timings`]
/// (`SegmentSeeker.swift:540-655`): it takes words from the front of
/// `alignment` while its text-token budget lasts and answers the updated
/// segment with how many entries it consumed. `previous_end` is the end of
/// the segment updated before it in the window, if any (the short-word
/// pull-back of a segment's first word reads it); `time_offset` is the
/// window's seek offset in seconds; `last_speech_timestamp` is read and, when
/// the segment gets a word, advanced to its end.
///
/// A short clip's window derives each segment's visible words from its own
/// words alone through this one pass (`derive_visible_words`): handed only
/// its own words, a segment can take no other segment's.
///
/// # Errors
/// [`SegmentError::Tokenizer`] if retokenizing a partially-special-filtered
/// alignment entry's surviving tokens fails (`:556-559`).
#[allow(clippy::too_many_arguments)] // The loop's own state, threaded per segment.
fn update_segment_with_word_timings(
  segment: &TranscriptionSegment,
  alignment: &[WordTiming],
  previous_end: Option<f32>,
  time_offset: f32,
  last_speech_timestamp: &mut f32,
  constrained_median_duration: f32,
  max_duration: f32,
  tokenizer: &WhisperTokenizer,
) -> Result<(TranscriptionSegment, usize), SegmentError> {
  let special_begin = tokenizer.special_tokens().special_token_begin();
  let mut saved_tokens = 0usize;
  // :544 -- only text tokens count toward this segment's word budget;
  // special/timestamp tokens already in `segment.tokens` never do.
  let text_token_count = segment
    .tokens_slice()
    .iter()
    .filter(|&&token| token < special_begin)
    .count();
  let mut words_in_segment: Vec<WordTiming> = Vec::new();

  // :547's `where savedTokens < textTokens.count` guards each element in
  // Swift's `for timing in mergedAlignment[wordIndex...]`, skipping
  // (not necessarily stopping at) elements while false. `break` here is
  // behaviorally identical: `saved_tokens` and `consumed` both only ever
  // advance inside this loop body, so once the bound trips false it stays
  // false for every later element too, and Swift's `where` never lets the
  // body run again either.
  let mut consumed = 0usize;
  for timing in alignment {
    if saved_tokens >= text_token_count {
      break;
    }
    consumed += 1;

    // :551-554 -- drop special/timestamp tokens from this timing; an
    // all-special entry is consumed from the cursor but emits no word.
    let timing_tokens: Vec<u32> = timing
      .tokens_slice()
      .iter()
      .copied()
      .filter(|&token| token < special_begin)
      .collect();
    if timing_tokens.is_empty() {
      continue;
    }

    // :556-559 -- retokenize only when some (not all) of this timing's
    // tokens were filtered out; otherwise reuse its own decoded word.
    let timing_tokens_len = timing_tokens.len();
    let word = if timing_tokens_len < timing.tokens_slice().len() {
      tokenizer.decode(&timing_tokens, false)?
    } else {
      timing.word().to_string()
    };

    // :561-562.
    let mut start = rounded_to_places(time_offset + timing.start(), 2);
    let end = rounded_to_places(time_offset + timing.end(), 2);

    // :564-596 -- a short-duration word gets its start pulled back into
    // any gap before it: against the previous word in THIS segment if
    // there is one, else (only for a segment's own first word) against
    // the previous segment's already-finalized end.
    if end - start < constrained_median_duration / 4.0 {
      if let Some(previous) = words_in_segment.last() {
        let previous_end = previous.end();
        if start > previous_end {
          let space_available = start - previous_end;
          let desired_duration = space_available.min(constrained_median_duration / 2.0);
          start = rounded_to_places(start - desired_duration, 2);
        }
      } else if let Some(previous_end) = previous_end
        && start > previous_end
      {
        let space_available = start - previous_end;
        let desired_duration = space_available.min(constrained_median_duration / 2.0);
        start = rounded_to_places(start - desired_duration, 2);
      }
    }

    // :598.
    let probability = rounded_to_places(timing.probability(), 2);
    words_in_segment.push(WordTiming::new(
      word,
      timing_tokens,
      start,
      end,
      probability,
    ));
    // :606 -- Swift re-reads `timingTokens.count`, the local filtered
    // vec, not the just-pushed word's own token slice; captured above
    // before `timing_tokens` moved into the `WordTiming`.
    saved_tokens += timing_tokens_len;
  }

  let mut updated_segment = segment.clone();

  // :615-652 -- only a segment that got at least one word runs the
  // pause/boundary hack and advances `last_speech_timestamp`; a wordless
  // segment leaves both `updated_segment`'s bounds and
  // `last_speech_timestamp` untouched.
  if !words_in_segment.is_empty() {
    // :616-620 -- read BEFORE any mutation below, matching Swift's own
    // `firstWord` copy.
    let pause_length = words_in_segment[0].end() - *last_speech_timestamp;
    let first_word_too_long = words_in_segment[0].duration() > max_duration;
    let both_words_too_long = words_in_segment.len() > 1
      && words_in_segment[1].end() - words_in_segment[0].start() > max_duration * 2.0;

    // :621-633 -- after an over-long pause, clamp the first word (and,
    // if it is also too long, re-split the 0/1 boundary first) so
    // neither word spans more than `max_duration`.
    if pause_length > constrained_median_duration * 4.0
      && (first_word_too_long || both_words_too_long)
    {
      if words_in_segment.len() > 1 && words_in_segment[1].duration() > max_duration {
        let w1_end = words_in_segment[1].end();
        let boundary = (w1_end / 2.0).max(w1_end - max_duration);
        words_in_segment[0].set_end(boundary);
        words_in_segment[1].set_start(boundary);
      }
      // Reads `words_in_segment[0].end()` LIVE: the boundary re-split
      // just above, if it fired, already changed it.
      let w0_end = words_in_segment[0].end();
      words_in_segment[0].set_start(last_speech_timestamp.max(w0_end - max_duration));
    }

    // :635-640 -- prefer the segment-level start over the (possibly
    // hack-adjusted) first word's start when the word has drifted more
    // than half a second earlier than the segment itself began.
    let w0_start = words_in_segment[0].start();
    let w0_end = words_in_segment[0].end();
    if segment.start() < w0_end && segment.start() - 0.5 > w0_start {
      let clamped = (w0_end - constrained_median_duration)
        .min(segment.start())
        .max(0.0);
      words_in_segment[0].set_start(clamped);
    } else {
      updated_segment.set_start(words_in_segment[0].start());
    }

    // :642-649 -- symmetric preference for the segment-level end over
    // the last word's end. Swift's `wordsInSegment.last` is always
    // non-nil here (guarded by the outer non-empty check already); when
    // there is exactly one word this is the SAME element the
    // start-preference block above just wrote, so `last_start` below
    // can already reflect that mutation.
    let last_index = words_in_segment.len() - 1;
    let last_start = words_in_segment[last_index].start();
    let last_end = words_in_segment[last_index].end();
    if updated_segment.end() > last_start && segment.end() + 0.5 < last_end {
      let clamped = (last_start + constrained_median_duration).max(segment.end());
      words_in_segment[last_index].set_end(clamped);
    } else {
      updated_segment.set_end(last_end);
    }

    // :651.
    *last_speech_timestamp = updated_segment.end();
  }

  // :654-655.
  updated_segment.set_words(words_in_segment);
  Ok((updated_segment, consumed))
}

/// Rounds `value` to `decimal_places` decimal digits, half-away-from-zero.
/// Ports `Float.rounded(_:)` (`ArgmaxCore/FoundationExtensions.swift:
/// 9-13`: `(self * divisor).rounded() / divisor`, where Swift's
/// no-argument `.rounded()` defaults to rule `.toNearestOrAwayFromZero`).
/// Rust's [`f32::round`] documents that exact rule (round half-way cases
/// away from `0.0`), so this is a direct, unadjusted port. `pub(crate)`:
/// no caller outside this crate needs it yet — [`update_segments_with_word_timings`]
/// is the first non-test consumer.
pub(crate) fn rounded_to_places(value: f32, decimal_places: i32) -> f32 {
  let divisor = 10f32.powi(decimal_places);
  (value * divisor).round() / divisor
}

// ---------------------------------------------------------------------
// add_word_timestamps: the orchestration wrapper
// ---------------------------------------------------------------------

/// Measures **this host's** CoreVideo row pitch, in Float16 elements, for
/// the IOSurface-backed `MLMultiArray` Swift's alignment gather allocates —
/// `MLMultiArray(shape: [rows, cols], dataType: .float16, initialValue: 0)`
/// (`ArgmaxCore/MLMultiArrayExtensions.swift:11-53`), which for `.float16`
/// is a `kCVPixelFormatType_OneComponent16Half` `CVPixelBuffer`
/// (`:121-136`) — by allocating the equivalent array through
/// [`MultiArray::f16_surface`] and reading the strides CoreVideo actually
/// chose.
///
/// **Only [`AlignmentGather::SwiftParity`] — the opt-in mode — calls this.**
/// The default gather ([`AlignmentGather::Complete`]) allocates no surface,
/// measures nothing and depends on no property of this host.
///
/// **Not a constant, deliberately.** Apple documents `CVPixelBuffer` row
/// alignment as hardware-dependent and requires it to be queried (QA1829,
/// "Understanding CVPixelBuffer memory alignment"), and
/// [`MultiArray::f16_surface`]'s own contract calls the padding
/// platform-chosen. The reference host for whisper #41 (M1 Max, macOS 26.5)
/// aligns rows to 64 bytes — 32 Float16 elements — which is what
/// `tests/whisper_swift_probes/probe_alignment_stride.out` recorded (cols
/// 8 -> 32, 9 -> 32, 100 -> 128, 1496/1500/1504 -> 1504); that table is
/// evidence of the layout, not a rule about it. A host that pads
/// differently would have Swift's gather truncate *different* cells, so a
/// compiled-in quantum would zero the wrong ones and make
/// [`AlignmentGather::SwiftParity`] silently non-parity there. Querying the
/// same allocator Swift's array comes from is what makes the mode correct
/// on every host rather than on one.
///
/// The probe uses the caller's exact `[rows, cols]` shape rather than a
/// cheaper 1-row stand-in, so nothing here assumes the pitch is independent
/// of the row count — which is why [`gather_swift_parity_into`] asks for the
/// SOURCE and DESTINATION shapes separately instead of reusing one answer for
/// both. One `CVPixelBufferCreate` plus its zero-fill per shape per
/// word-timestamped window is immaterial next to that window's encoder and
/// decoder runs.
///
/// # What this establishes, and what it does not
///
/// **Established:** the row pitch of the surface THIS call allocated, read
/// back from CoreVideo's own strides.
///
/// **Not established:** the row pitch of the surface Swift's process
/// allocated. Nothing in CoreVideo promises two allocations of one shape
/// share a layout, and this port cannot observe Swift's. `SwiftParity`
/// assumes they agree — assumption **A1** on
/// [`AlignmentGather::SwiftParity`], stated there rather than left implicit.
///
/// **What would settle it:** a documented CoreVideo/IOSurface guarantee that
/// row alignment is a pure function of pixel format and dimensions. No such
/// guarantee is published; QA1829 says the opposite in spirit by requiring
/// the value to be queried per buffer.
///
/// # Errors
/// [`SegmentError::AlignmentPitchUnavailable`] if the probe allocation
/// fails (e.g. [`TensorError::SurfaceUnsupported`](crate::TensorError) below
/// macOS 12, where `MLMultiArray` has no pixel-buffer initializer);
/// [`SegmentError::AlignmentPitchUnexpectedLayout`] if it succeeds but
/// reports strides other than `[pitch, 1]` with `pitch >= cols`. Both are
/// fail-closed: see [`SegmentError::AlignmentPitchUnavailable`] for why a
/// silent fall back to [`AlignmentGather::Complete`] is not offered.
// `pub(crate)`, not private: `transcribe::tests`' own gather fixture
// (`swift_parity_gather_moves_the_first_windows_end_and_the_next_seek`) is built around where
// this host's pitch cuts the gather, and has to ask the same helper the
// pipeline asks rather than restate a number.
pub(crate) fn coreml_f16_row_pitch(rows: usize, cols: usize) -> Result<usize, SegmentError> {
  let probe = MultiArray::f16_surface(&[rows, cols]).map_err(|source| {
    SegmentError::AlignmentPitchUnavailable(AlignmentPitchUnavailable::new(rows, cols, source))
  })?;
  row_pitch_of(&probe, rows, cols)
}

/// The row pitch a probe surface reports, rejecting any layout the gather
/// cannot describe.
///
/// CoreML pads only BETWEEN rows (the invariant `MultiArray::copy_into`'s
/// padded-gather already relies on), so the only layout [`gather_swift_rows`]
/// reproduces is a unit last-dimension stride under a row pitch that is at
/// least the logical width. Anything else is a layout this port has never
/// seen and cannot claim to replicate.
fn row_pitch_of(probe: &MultiArray, rows: usize, cols: usize) -> Result<usize, SegmentError> {
  let strides = probe.strides();
  match *strides {
    [pitch, 1] if pitch >= cols => Ok(pitch),
    _ => Err(SegmentError::AlignmentPitchUnexpectedLayout(
      AlignmentPitchUnexpectedLayout::new(rows, cols, strides.to_vec()),
    )),
  }
}

/// The row-pitch measurement [`gather_swift_parity_into`] runs, as a
/// parameter rather than a hard-wired call.
///
/// Production passes [`coreml_f16_row_pitch`], which measures the running
/// host. `segment::tests` passes layouts this host will never produce — above
/// all a host that pitches Swift's 224-row source array differently from this
/// port's 225-row accumulator, which is the ONLY way to check that the probe
/// asks CoreVideo about Swift's height rather than the port's (whisper #41,
/// codex round 3, F2).
type RowPitchProbe<'a> = &'a dyn Fn(usize, usize) -> Result<usize, SegmentError>;

/// Reproduces Swift's alignment gather into `out` (supplied zero-filled),
/// measuring both surfaces' row pitches through `row_pitch`.
///
/// # The two heights are NOT the same number
///
/// `alignment.rows()` is this port's accumulator height,
/// `max_token_context + 1` — 225 — because
/// [`commit_alignment_row`](crate::audio::whisper::backend::InferenceBackend::commit_alignment_row)
/// commits step `position`'s row at `position + 1`, so the last trait-legal
/// position needs one slot of headroom past the KV dimension. Swift has no
/// such headroom: `alignmentWeights` is allocated at
/// `[kvCacheMaxSequenceLength, n_audio_ctx]` (`TextDecoder.swift:141`) — the
/// **224**-slot KV dimension itself.
///
/// The pitch is a function of the shape CoreVideo is handed, and this branch
/// exists precisely because pitch may vary with height. Probing `[225, cols]`
/// would therefore decode Swift's storage offsets with a pitch Swift's array
/// never had wherever `pitch(224, cols) != pitch(225, cols)`, splicing the
/// wrong weights (or the wrong padding) into DTW's input while still
/// reporting parity. `swift_source_rows` is Swift's physical height, passed
/// down from the model's own `max_token_context` — never derived from the
/// view — and it is the shape this probes.
///
/// The reproduction's read bound is the smaller of the two: an offset past
/// `swift_source_rows` is past Swift's array (Swift would read off its end),
/// and an offset past `alignment.rows()` is past this port's data. Neither is
/// reachable for a real window (`needed <= 224 <= min(224, 225)`); the bound
/// is the same defensive branch the prefix take has always carried.
///
/// # Errors
/// [`SegmentError::AlignmentPitchUnavailable`] /
/// [`SegmentError::AlignmentPitchUnexpectedLayout`] from `row_pitch`, for
/// either surface — see [`coreml_f16_row_pitch`].
fn gather_swift_parity_into(
  out: &mut [f32],
  alignment: &AlignmentView<'_>,
  needed: usize,
  cols: usize,
  swift_source_rows: usize,
  row_pitch: RowPitchProbe<'_>,
) -> Result<(), SegmentError> {
  let src_rows = alignment.rows().min(swift_source_rows);
  // A row-less source has no surface for Swift to have allocated, and the
  // gather reads nothing from it whatever CoreVideo would have pitched it at
  // (every offset lands past row `src_rows`), so the reproduction is all
  // zeros. Skipping the probe keeps a caller-built empty view out of
  // `AlignmentPitchUnavailable`, which reports an unmeasurable HOST rather
  // than an empty input. No backend produces one: both build the FIXED
  // `max_token_context + 1` accumulator against a nonzero `max_token_context`.
  let src_pitch = if src_rows == 0 {
    cols
  } else {
    row_pitch(swift_source_rows, cols)?
  };
  let dst_pitch = row_pitch(needed, cols)?;
  gather_swift_rows(
    out,
    alignment.data(),
    src_rows,
    needed,
    cols,
    src_pitch,
    dst_pitch,
  );
  Ok(())
}

/// Reproduces Swift's alignment gather: what `dynamicTimeWarping` actually
/// reads out of the destination array `addWordTimestamps` builds, given the
/// source's logical rows and the two surfaces' measured row pitches.
///
/// Writes `needed * cols` row-major f32 into `out`, which the caller supplies
/// zero-filled.
///
/// # The reproduction, cell by cell
///
/// Swift's loop (`SegmentSeeker.swift:452-460`) binds both arrays' strides
/// and then ignores them, offsetting BOTH sides by `index * columnCount`.
/// Its `filteredIndices` are `0..needed` by construction (`:429-432`,
/// `:441`), so the copies tile `[0, needed * cols)` contiguously on each
/// side and the whole loop is one verbatim copy: destination STORAGE offset
/// `o` holds source STORAGE offset `o`, for every `o < needed * cols`.
/// `dynamicTimeWarping` reads logical row `r`, column `c` through
/// `MLMultiArray`'s stride-aware flat subscript (`:217`), i.e. destination
/// storage `o = r * dst_pitch + c`. So:
///
/// - `o >= needed * cols` — past the copy. Neither the `memcpy` nor the
///   `initialValue:` fill (logical `count` elements, `:450`) ever wrote
///   there, so it is the destination allocation's untouched tail, and what
///   DTW reads from it is whatever `CVPixelBufferCreate` returned. This
///   function substitutes ZERO, which is assumption **A2** of the opt-in
///   mode — see [`AlignmentGather::SwiftParity`], where it is stated in full
///   rather than presented as established. It is not verified here and
///   cannot be: an earlier revision sampled a *different*, disposable
///   allocation's tail, which proves nothing about the one Swift's process
///   gathered into (codex round 3, F3), and the sampling itself read
///   uninitialized memory (F1).
/// - `o < needed * cols` — source storage `o`, which is source logical row
///   `o / src_pitch`, column `o % src_pitch`. A column below `cols` is a
///   real weight; at or above it, the source's own inter-row padding, for
///   which this function likewise substitutes ZERO.
///
/// **What the source-padding zero rests on.** `alignmentWeights` is allocated
/// once (`TextDecoder.swift:141`) through the same `initialValue:
/// FloatType(0)` initializer, whose `initialize(repeating:count:)` runs over
/// the LOGICAL element count — flat, ignoring the pitch — and so zeroes
/// storage `[0, src_rows * cols)` at construction, padding cells included. It
/// is never a model input or output backing (`TextDecoder.swift:394-401`,
/// `:414`) and `DecodingInputs.reset` never clears it
/// (`Models.swift:312-322`); its only writer is `updateAlignmentWeights`
/// (`:272-295`), which is stride-aware and so writes no padding cell. The
/// gather reads storage below `needed * cols <= src_rows * cols`, so every
/// padding cell it can reach was zero *at construction*.
///
/// That its construction-time value SURVIVES is a separate step, and an
/// unverified one: both the writer and the gather reach the array through
/// `withUnsafeMutableBytes`, whose contract states the strides handed to the
/// closure may differ from the value before invocation — so no cited contract
/// rules out a relayout or a replacement of the backing storage introducing
/// padding that construction never zeroed. That is assumption **A3** of the
/// opt-in mode, stated on [`AlignmentGather::SwiftParity`] (codex round 3,
/// F4).
///
/// # Why the two pitches are measured separately
///
/// `src_pitch == dst_pitch` collapses this to "row `r` keeps its first
/// `min(cols, needed * cols - r * dst_pitch)` columns and reads zeros after
/// them", the form the shipping host takes. But row-count invariance is a
/// property this host was measured to have, not one CoreVideo promises, and
/// where the pitches differ the reproduction is not a truncation at all: row
/// `r` reads a SHIFTED window of the source, mixing the tail of one logical
/// row with the head of the next and with zeroed padding. Modelling only the
/// equal-pitch case would have made [`AlignmentGather::SwiftParity`]
/// silently wrong there while still reporting parity, so both shapes are
/// probed and the general mapping is what runs.
///
/// Rows at or past `src_rows` read zeros: Swift's gather would run off the
/// end of its source array there. `needed <= src_rows` for every real window
/// (the accumulator is `max_token_context + 1` rows and `needed <= 224`), so
/// this is the same defensive branch [`add_word_timestamps`]'s prefix take
/// has always had, not a modelled behavior.
///
/// Taking both pitches as parameters (rather than measuring inside) is what
/// lets the reproduction be tested at pitches this host will never choose —
/// see `segment::tests`.
fn gather_swift_rows(
  out: &mut [f32],
  source: &[f32],
  src_rows: usize,
  needed: usize,
  cols: usize,
  src_pitch: usize,
  dst_pitch: usize,
) {
  debug_assert!(src_pitch >= cols && dst_pitch >= cols && cols > 0);
  let copied = needed * cols;
  for (row, out_row) in out.chunks_mut(cols).enumerate().take(needed) {
    // Destination storage this logical row reads, clipped to the copy.
    let mut offset = row * dst_pitch;
    let end = (offset + cols).min(copied);
    let mut column = 0usize;
    while offset < end {
      let source_row = offset / src_pitch;
      if source_row >= src_rows {
        // Past the source array; every later offset is too, so the rest of
        // this row and every row after it keeps `out`'s zero.
        break;
      }
      let source_column = offset % src_pitch;
      let run = if source_column < cols {
        let run = (cols - source_column).min(end - offset);
        let start = source_row * cols + source_column;
        out_row[column..column + run].copy_from_slice(&source[start..start + run]);
        run
      } else {
        // Source-side inter-row padding: zero under assumption A3, per this
        // function's doc, and `out` already holds zero.
        (src_pitch - source_column).min(end - offset)
      };
      offset += run;
      column += run;
    }
    // `[column, cols)` is the destination tail: zero under assumption A2,
    // per this function's doc, and already zero in `out`.
  }
}

/// Assembles one window's word-level timestamps end to end. Ports
/// `SegmentSeeker.addWordTimestamps` (`SegmentSeeker.swift:410-496`):
/// flattens `segments`' tokens/log-probs into the flat list
/// [`find_alignment`] needs (`:427-442`), builds a prefix-take,
/// zero-padded [`AlignmentMatrix`] from `alignment` (`:444-461`), then
/// threads that through [`find_alignment`] (`:465-472`) -> the
/// duration-constraint/sentence-boundary truncation hack (`:474-477`) ->
/// [`merge_punctuations`] when non-empty (`:479-482`) ->
/// [`update_segments_with_word_timings`] (`:484-493`).
///
/// `language_code` and `grouping` are threaded straight into
/// `find_alignment` -> `split_to_word_tokens`: the same
/// `NLLanguageRecognizer` replacement documented on [`find_alignment`] and
/// [`WhisperTokenizer::split_to_word_tokens`] (spec §5.3), plus the
/// explicit word-grouping mode from coremlit issue #14
/// ([`WordGrouping::SwiftParity`] is the default after #41;
/// [`WordGrouping::FineGrained`] is this port's long-standing opt-in).
///
/// `gather` selects what the prefix take hands to DTW.
/// [`AlignmentGather::Complete`] — **the default** — gathers every row whole:
/// a plain prefix take, allocating no surface, measuring nothing about the
/// host and assuming nothing about it.
/// [`AlignmentGather::SwiftParity`] is the opt-in that instead reproduces
/// Swift's own gather: its `memcpy` loop (`SegmentSeeker.swift:452-460`) binds
/// both `MLMultiArray`s' strides and then indexes with `columnCount`, while
/// `dynamicTimeWarping`'s flat subscript (`:217`) *is* stride-aware and
/// CoreVideo pads the Float16 backing's rows
/// (`ArgmaxCore/MLMultiArrayExtensions.swift:121-136`), so what DTW reads back
/// is a function of BOTH surfaces' row pitches — each measured on the running
/// host by this module's `coreml_f16_row_pitch`, at each surface's own height.
/// This module's `gather_swift_parity_into`/`gather_swift_rows` derive the
/// mapping; see [`AlignmentGather::SwiftParity`] for the three assumptions
/// that mode carries and why they are acceptable only because it is opt-in
/// (whisper #41). `swift_source_rows` is Swift's PHYSICAL source height —
/// `kvCacheMaxSequenceLength`, i.e. the model's `max_token_context` — which
/// is one row SHORTER than `alignment.rows()`; it is ignored under `Complete`
/// and load-bearing under `SwiftParity` (see `gather_swift_parity_into`).
/// Swift's
/// `segmentSize` and `options` parameters are unused in the function body
/// (verified against `SegmentSeeker.swift:410-496`), and `timings` is
/// only passed through to `findAlignment` (`:471`), which ignores it —
/// all three are dropped here: `timings`' duration/run-count bookkeeping
/// (`TranscribeTask.swift:214-215`) is the caller's responsibility, same
/// as at Swift's own call site.
///
/// # Errors
/// [`SegmentError::InvalidAlignmentShape`] if the prefix-take alignment
/// ends up with zero rows or columns — notably an empty (or
/// all-empty-tokens) `segments` input, which Swift's own unguarded
/// `1...0` range would instead crash on (see [`dynamic_time_warping`]'s
/// doc); [`SegmentError::Tokenizer`] if `split_to_word_tokens` or a
/// partial-special retokenize fails;
/// [`SegmentError::AlignmentPitchUnavailable`] /
/// [`SegmentError::AlignmentPitchUnexpectedLayout`] under
/// [`AlignmentGather::SwiftParity`] only, if either surface's CoreVideo row
/// pitch cannot be measured or is not a row-padded row-major layout — parity
/// cannot be claimed against a layout this port cannot describe, so it is
/// refused rather than approximated (see this module's
/// `coreml_f16_row_pitch`). The default gather returns none of the three.
#[allow(clippy::too_many_arguments)] // Mirrors Swift's addWordTimestamps argument
// surface (mirroring decode_text's own precedent for this exact lint, per its
// doc comment); no natural subset of these forms a cohesive struct without
// inventing one purely to dodge the lint.
pub fn add_word_timestamps(
  segments: &[TranscriptionSegment],
  alignment: &AlignmentView<'_>,
  tokenizer: &WhisperTokenizer,
  language_code: &str,
  grouping: WordGrouping,
  gather: AlignmentGather,
  swift_source_rows: usize,
  seek: usize,
  prepended: &str,
  appended: &str,
  last_speech_timestamp: f32,
) -> Result<Vec<TranscriptionSegment>, SegmentError> {
  let aligned = aligned_window(
    segments,
    alignment,
    tokenizer,
    language_code,
    grouping,
    gather,
    swift_source_rows,
  )?;
  visible_words(
    segments,
    aligned.into_iter().map(|(word, _)| word).collect(),
    tokenizer,
    seek,
    prepended,
    appended,
    last_speech_timestamp,
  )
}

/// The visible words of the segments a clip-back kept, DERIVED from what
/// survived it by the pass every window's words come from — the caller's
/// [`WordGrouping`] and [`AlignmentGather`] over the window's alignment, then
/// Swift's derivation (the duration hack and sentence-boundary truncation,
/// punctuation merged, the word-timing pass) — so a clipped window's visible
/// words are the words a window that is not clipped would have for the same
/// text, and there is one source of truth: no visible word holds a token the
/// clip removed, and the words and the text never disagree.
///
/// `window` is the window's segments before the clip-back, and `kept` each
/// segment it kept, with the raw words it kept and the positions of its
/// tokens in `window`'s flattened tokens (`crate::audio::whisper::transcribe`'s
/// `clip_back_to_window`). The window is aligned whole, exactly as
/// [`add_word_timestamps`] aligns a window that is not clipped — the same
/// rows, gathered as `gather` says, the same tokens, grouped as `grouping`
/// says — so each word is timed as that window times it, and each word then
/// keeps only its surviving tokens: a word the clip removed is gone, and one
/// it cut keeps the tokens before the cut, its text decoded from them, timed
/// as the word. Aligning the survivors alone would not time them so: a word
/// the clip ends would reach across the removed text to the next token the
/// alignment still has. The clip-back's raw words — the fine-grained units,
/// every special token split off, read from the rows the decode committed —
/// decide what is clipped, and nothing else.
///
/// **Per segment.** Each segment's visible words are derived from its own
/// words alone: the punctuation merge and the word-timing pass
/// ([`update_segment_with_word_timings`]) run over that segment and the
/// words its tokens make, never over the window's words at once — a merge
/// across a segment boundary changes how many tokens a word holds, and the
/// word-timing pass's token budget then hands a segment a word whose tokens
/// it does not hold. A word whose tokens two segments hold is cut where they
/// meet, each part its own segment's word. What the window shares is what
/// every window's pass shares, in window order: the alignment, the duration
/// constraints — computed over every word the alignment gives, before the
/// clip removes or cuts any, as a window that is not clipped computes them
/// ([`kept_words`]) — and the sentence-boundary truncation, which moves
/// times and no word ([`segment_words`]); then the end of the segment
/// before, and the last speech timestamp, threaded from segment to segment.
///
/// **The tokens a segment kept, as it kept them.** The clip-back restates a
/// surviving boundary timestamp (`<|0.50|>` read as `<|0.00|>`), and the
/// window's alignment and log probabilities are of the tokens before it did.
/// A word holds the tokens its segment kept, and is weighed by the log
/// probabilities of the kept tokens that are the ones the model sampled —
/// a restated timestamp's is not ([`kept_words`]).
///
/// # Errors
/// [`SegmentError::Tokenizer`] if a word's tokens fail to decode; as
/// [`add_word_timestamps`] for the alignment, under
/// [`AlignmentGather::SwiftParity`] the pitch errors included.
#[allow(clippy::too_many_arguments)] // add_word_timestamps' surface, and what the clip kept.
pub(crate) fn derive_visible_words(
  window: &[TranscriptionSegment],
  kept: &[KeptSegment],
  alignment: &AlignmentView<'_>,
  tokenizer: &WhisperTokenizer,
  language_code: &str,
  grouping: WordGrouping,
  gather: AlignmentGather,
  swift_source_rows: usize,
  seek: usize,
  prepended: &str,
  appended: &str,
  last_speech_timestamp: f32,
) -> Result<Vec<TranscriptionSegment>, SegmentError> {
  if kept.is_empty() {
    return Ok(Vec::new());
  }
  let timed = aligned_window_timed(
    window,
    alignment,
    tokenizer,
    language_code,
    grouping,
    gather,
    swift_source_rows,
  )?;
  let (durations, own_words) = kept_words(window, kept, &timed, |tokens| {
    Ok(tokenizer.decode(tokens, false)?)
  })?;
  derive_per_segment(
    kept.iter().map(|(segment, _, _)| segment),
    own_words,
    durations,
    tokenizer,
    seek,
    prepended,
    appended,
    last_speech_timestamp,
  )
}

/// The window's aligned words made each of the `kept` segments' own, beside
/// the window's duration constraints.
///
/// - **The constraints are the window's**, computed over every word of
///   `timed` before any is removed or cut — the statistics a window that is
///   not clipped computes ([`visible_words`]). Computed over what survived,
///   they moved what survived: a clip that removed the long words left a
///   median of the short ones, and the truncation then cut a surviving word
///   the window's own constraints leave alone.
/// - **A word keeps the tokens its segment kept**, at the window positions
///   the clip-back answered with, by the segment holding them; a word two
///   segments hold is cut where they meet. A word a segment kept whole, every
///   token the one the window aligned, is the word the alignment gave;
///   any other part — cut, or holding a token the clip-back restated — is
///   built of its kept tokens ([`word_part`]), its text decoded from them by
///   `decode`.
/// - **Its probability is of the tokens the model sampled.** A position
///   weighs in only where the token the segment kept there is the one the
///   window logged and aligned there — Swift's probe, a logged pair read
///   only where its token is the one at its index — so a restated
///   timestamp, which the model never sampled, never does: a word grouping
///   `<|0.50|>中文`, restated `<|0.00|>中文`, is weighed by `中文` alone,
///   never by the whole word's probability, which the old timestamp's
///   weighed.
///
/// The surviving words are truncated once at sentence boundaries, in window
/// order, under the window's constraints ([`segment_words`]).
///
/// # Errors
/// What `decode` answers for a part's tokens.
fn kept_words<D>(
  window: &[TranscriptionSegment],
  kept: &[KeptSegment],
  timed: &TimedAlignment,
  decode: D,
) -> Result<(WordDurationConstraints, Vec<Vec<WordTiming>>), SegmentError>
where
  D: Fn(&[u32]) -> Result<String, SegmentError>,
{
  let words: Vec<WordTiming> = timed.words.iter().map(|(word, _)| word.clone()).collect();
  let durations = calculate_word_duration_constraints(&words);
  // Each window position's logged log probability, where the logged token
  // is the token there.
  let logged: Vec<Option<f32>> = window
    .iter()
    .flat_map(|segment| {
      let pairs = segment.token_log_probs_slice();
      segment
        .tokens_slice()
        .iter()
        .enumerate()
        .map(move |(index, &token)| match pairs.get(index) {
          Some(&(logged, log_prob)) if logged == token => Some(log_prob),
          _ => None,
        })
    })
    .collect();
  // The kept segment each window position survives in, and the token it
  // kept there.
  let mut owner: Vec<Option<(usize, u32)>> = vec![None; logged.len()];
  for (index, (segment, _, positions)) in kept.iter().enumerate() {
    for (&position, &token) in positions.iter().zip(segment.tokens_slice()) {
      if let Some(slot) = owner.get_mut(position) {
        *slot = Some((index, token));
      }
    }
  }
  // The window's surviving words in window order, each with the segment
  // holding it.
  let mut surviving: Vec<(usize, WordTiming)> = Vec::new();
  for (word, (from, to)) in &timed.words {
    // The word's kept tokens, by the segment holding them, in order: their
    // positions in the window, whether the segment restated any, and the log
    // probabilities of those the model sampled.
    let mut parts: Vec<KeptPart> = Vec::new();
    for (position, &aligned) in (*from..*to).zip(word.tokens_slice()) {
      let Some((index, token)) = owner.get(position).copied().flatten() else {
        continue;
      };
      let restated = token != aligned;
      let sampled = if restated {
        None
      } else {
        logged.get(position).copied().flatten()
      };
      match parts.last_mut() {
        Some(part) if part.segment == index => {
          part.tokens.push(token);
          part.positions.push(position);
          part.restated |= restated;
          part.own.extend(sampled);
        }
        _ => parts.push(KeptPart {
          segment: index,
          tokens: vec![token],
          positions: vec![position],
          restated,
          own: sampled.into_iter().collect(),
        }),
      }
    }
    for part in parts {
      if !part.restated && part.tokens.as_slice() == word.tokens_slice() {
        surviving.push((part.segment, word.clone()));
      } else {
        let text = decode(&part.tokens)?;
        surviving.push((part.segment, word_part(word, text, &part, timed)));
      }
    }
  }
  Ok((durations, segment_words(surviving, kept.len(), durations)))
}

/// What one segment kept of an aligned word ([`kept_words`]).
struct KeptPart {
  /// The kept segment holding it.
  segment: usize,
  /// The tokens the segment kept, in order.
  tokens: Vec<u32>,
  /// Their positions among the window's flattened tokens.
  positions: Vec<usize>,
  /// Whether the segment restated any of them — a boundary timestamp the
  /// clip-back moved.
  restated: bool,
  /// The log probabilities of those the model sampled: kept as the window
  /// aligned them, their logged pair their own.
  own: Vec<f32>,
}

/// The `part` of the aligned `word` a segment kept: its kept tokens, its
/// `text` decoded from them, timed from its own first token's start to its
/// own last token's end ([`TimedAlignment`]), its probability the mean of
/// the log probabilities of its kept tokens the model sampled,
/// exponentiated, as [`find_alignment`] weighs a word. Copying the whole
/// word's times and probability stretched a kept prefix through the suffix
/// the clip removed and weighed it by tokens it no longer holds, or by a
/// timestamp the clip-back restated. Where no kept token has a probability
/// of its own, the word's own value stands for a part the clip-back
/// restated nothing in; a part of restated timestamps alone — the clip-back
/// restates nothing else — has none, weighs 0, and the word-timing pass,
/// which makes no word of special tokens alone, drops it.
fn word_part(
  word: &WordTiming,
  text: String,
  part: &KeptPart,
  timed: &TimedAlignment,
) -> WordTiming {
  let start = part
    .positions
    .first()
    .and_then(|&first| timed.starts.get(first).copied())
    .unwrap_or(word.start());
  let end = part
    .positions
    .last()
    .and_then(|&last| timed.ends.get(last).copied())
    .unwrap_or(word.end());
  let probability = if !part.own.is_empty() {
    (part.own.iter().sum::<f32>() / part.own.len() as f32).exp()
  } else if part.restated {
    0.0
  } else {
    word.probability()
  };
  WordTiming::new(text, part.tokens.clone(), start, end, probability)
}

/// The window's surviving words — each with the segment holding it, in
/// window order — made each of `segments` segments' own: the sentence-boundary
/// truncation once over them in window order under the window's `durations`
/// (`SegmentSeeker.swift:474-477`), exactly as a window that is not clipped
/// truncates its words — a segment's first word reads the word before it,
/// its previous segment's last — then each word handed to its segment.
/// Truncating per segment left a segment's first word unread: the
/// truncation starts at a list's second word.
fn segment_words(
  surviving: Vec<(usize, WordTiming)>,
  segments: usize,
  durations: WordDurationConstraints,
) -> Vec<Vec<WordTiming>> {
  let (owners, words): (Vec<usize>, Vec<WordTiming>) = surviving.into_iter().unzip();
  let words = truncate_long_words_at_sentence_boundaries(words, durations.max_duration());
  let mut own_words: Vec<Vec<WordTiming>> = vec![Vec::new(); segments];
  for (owner, word) in owners.into_iter().zip(words) {
    if let Some(own) = own_words.get_mut(owner) {
      own.push(word);
    }
  }
  own_words
}

/// Each of `segments` — the segments a clip-back kept — with the visible
/// words derived from `own_words`, its own words in order, already
/// truncated in window order under the window's `durations`
/// ([`segment_words`]): per segment the punctuation merge and the
/// word-timing pass — the rest of `SegmentSeeker.swift:474-493`'s
/// derivation, with no step that moves a word between segments.
///
/// A segment keeps the bounds the clip-back gave it — what survived states
/// them, and its timestamp tokens say them — and its words are held inside
/// them: the word-timing pass re-times a segment from its words, and a word
/// it timed past the clip's end (a pause before it, the alignment reaching
/// into the padding) moved the segment out of the clip, where the clamp
/// left it no length and the zero-length filter dropped text the clip-back
/// kept. The segment's end so held is the last speech the next one reads.
///
/// The clip-back hands no segment whose end precedes its start
/// (`crate::audio::whisper::transcribe`'s `clip_back_to_window` retimes one
/// from the raw words it kept); a word is held between the bounds without
/// `f32::clamp`, which panics on bounds out of order.
#[allow(clippy::too_many_arguments)] // The derivation's own state, and the window's.
fn derive_per_segment<'a>(
  segments: impl IntoIterator<Item = &'a TranscriptionSegment>,
  own_words: Vec<Vec<WordTiming>>,
  durations: WordDurationConstraints,
  tokenizer: &WhisperTokenizer,
  seek: usize,
  prepended: &str,
  appended: &str,
  last_speech_timestamp: f32,
) -> Result<Vec<TranscriptionSegment>, SegmentError> {
  // :537 -- the window's seek offset, in seconds, as
  // `update_segments_with_word_timings` takes it.
  let time_offset = seek as f32 / SAMPLE_RATE as f32;
  let mut last_speech_timestamp = last_speech_timestamp;
  let mut derived: Vec<TranscriptionSegment> = Vec::with_capacity(own_words.len());
  for (segment, mut words) in segments.into_iter().zip(own_words) {
    // :480-482, over this segment's words alone.
    if !words.is_empty() {
      words = merge_punctuations(&words, prepended, appended);
    }
    let (mut updated, _) = update_segment_with_word_timings(
      segment,
      &words,
      derived.last().map(TranscriptionSegment::end),
      time_offset,
      &mut last_speech_timestamp,
      durations.median(),
      durations.max_duration(),
      tokenizer,
    )?;
    let (start, end) = (segment.start(), segment.end());
    updated.set_start(start).set_end(end);
    let hold = |time: f32| time.max(start).min(end);
    for word in updated.words_slice_mut() {
      let (from, to) = (word.start(), word.end());
      word.set_start(hold(from)).set_end(hold(to));
    }
    if !updated.words_slice().is_empty() {
      last_speech_timestamp = end;
    }
    derived.push(updated);
  }
  Ok(derived)
}

/// The window's raw attribution — every segment's [`RawWord`]s — read from
/// the rows the window's own decode committed ([`AlignmentRows`]) and built
/// from its text alone ([`text_units`]): each of the segments' tokens takes
/// the row the decoder committed for it, a token with none takes no part
/// ([`token_rows`]), and no special token starts, ends or cuts a raw word.
/// Empty where no alignment can be had of them (no token with a committed
/// row): the clip-back's rule for unattributed text then applies.
pub(crate) fn attribute_window(
  segments: &[TranscriptionSegment],
  alignment: &AlignmentView<'_>,
  rows: &AlignmentRows,
  tokenizer: &WhisperTokenizer,
  language_code: &str,
) -> Vec<Vec<RawWord>> {
  match text_units(segments, alignment, rows, tokenizer, language_code) {
    Ok(units) => own_raw_words(segments, &units),
    Err(_) => Vec::new(),
  }
}

/// The row of the window's alignment snapshot each of `segments`' tokens
/// reads, in their flattened order: the row the window's own decode
/// committed for it, or `None` where it committed none — the result's
/// `<|startoftranscript|>` without a prompt before it, and the end of text.
/// The segments are the seeker's slices of the decode's result, contiguous
/// from its first token, so the `i`-th token is the result's `i`-th.
pub(crate) fn token_rows(
  segments: &[TranscriptionSegment],
  rows: &AlignmentRows,
) -> Vec<Option<usize>> {
  let count = segments
    .iter()
    .map(|segment| segment.tokens_slice().len())
    .sum();
  (0..count).map(|index| rows.row_of(index)).collect()
}

/// One unit of a window's text as its alignment places it: the half-open
/// span of the segments' flattened tokens it holds — text tokens alone — and
/// the 20 ms frames of the window at which it starts and ends.
pub(crate) type TextUnit = ((usize, usize), usize, usize);

/// The units of the text of the window `segments` make, read through
/// `rows`: every token with a committed row, in order, its own row gathered
/// and aligned by [`dynamic_time_warping`]; the tokens grouped as the
/// fine-grained splitter groups them — Unicode-complete units for the
/// scripts written without spaces (`zh`, `yue`, `ja`, `th`, `lo`, `my`),
/// space-delimited words otherwise; and each group split into its runs of
/// text, every special token split off, each run timed by its own tokens.
///
/// So no special token starts, ends or cuts a unit of text. The default
/// grouping for Chinese is Swift's space splitter, which starts a word at a
/// timestamp and appends every character after it: such a word started
/// where the timestamp did, inside a short clip, and read as one crossing
/// its end however many of its characters the alignment placed in the
/// padding. A token without a committed row takes no part, and a run breaks
/// where one lay between two that have.
fn text_units(
  segments: &[TranscriptionSegment],
  alignment: &AlignmentView<'_>,
  rows: &AlignmentRows,
  tokenizer: &WhisperTokenizer,
  language_code: &str,
) -> Result<Vec<TextUnit>, SegmentError> {
  let cols = alignment.cols();
  if cols == 0 {
    return Err(SegmentError::InvalidAlignmentShape(
      InvalidAlignmentShape::new(alignment.rows(), cols, alignment.data().len()),
    ));
  }
  let token_rows = token_rows(segments, rows);
  let mut tokens: Vec<u32> = Vec::new();
  let mut positions: Vec<usize> = Vec::new();
  let mut data: Vec<f32> = Vec::new();
  let flattened = segments
    .iter()
    .flat_map(|segment| segment.tokens_slice().iter().copied());
  for (position, token) in flattened.enumerate() {
    if let Some(row) = token_rows[position]
      && row < alignment.rows()
    {
      tokens.push(token);
      positions.push(position);
      data.extend_from_slice(alignment.row(row));
    }
  }
  let matrix = AlignmentMatrix::new(data, tokens.len(), cols);
  let (starts, ends) = token_frames(&dynamic_time_warping(&matrix.view())?);
  let special_begin = tokenizer.special_tokens().special_token_begin();
  let unit = |from: usize, to: usize| -> TextUnit {
    (
      (positions[from], positions[to - 1] + 1),
      starts[from],
      ends[to - 1],
    )
  };
  let mut units = Vec::new();
  let mut at = 0usize;
  for (_, group) in
    tokenizer.split_to_word_tokens(&tokens, language_code, WordGrouping::FineGrained)?
  {
    let end = (at + group.len()).min(tokens.len());
    let mut from: Option<usize> = None;
    for index in at..end {
      let text = tokens[index] < special_begin;
      if let Some(start) = from
        && (!text || positions[index] != positions[index - 1] + 1)
      {
        units.push(unit(start, index));
        from = None;
      }
      if text && from.is_none() {
        from = Some(index);
      }
    }
    if let Some(start) = from {
      units.push(unit(start, end));
    }
    at = end;
  }
  Ok(units)
}

/// The frame at which each aligned token starts and ends along `path`: a
/// boundary each time the path's row changes, the last token ending where the
/// path does — `find_alignment`'s own reading of the path (`:356-371`), in
/// frames.
fn token_frames(path: &DtwPath) -> (Vec<usize>, Vec<usize>) {
  let text_indices = path.text_indices_slice();
  let time_indices = path.time_indices_slice();
  let frame = |time: isize| usize::try_from(time).unwrap_or(0);
  let mut starts = vec![0usize];
  let mut ends = Vec::new();
  let mut current = text_indices.first().copied().unwrap_or(0);
  for (index, &text_index) in text_indices.iter().enumerate() {
    if text_index != current {
      current = text_index;
      starts.push(frame(time_indices[index]));
      ends.push(frame(time_indices[index]));
    }
  }
  ends.push(time_indices.last().map_or(1500, |&time| frame(time)));
  (starts, ends)
}

/// The alignment of the window `segments` make: their tokens flattened in
/// order, the alignment rows gathered for them, and [`find_alignment_spanned`]
/// run over both — every word with its span in the flattened tokens.
fn aligned_window(
  segments: &[TranscriptionSegment],
  alignment: &AlignmentView<'_>,
  tokenizer: &WhisperTokenizer,
  language_code: &str,
  grouping: WordGrouping,
  gather: AlignmentGather,
  swift_source_rows: usize,
) -> Result<Vec<SpannedWord>, SegmentError> {
  Ok(
    aligned_window_timed(
      segments,
      alignment,
      tokenizer,
      language_code,
      grouping,
      gather,
      swift_source_rows,
    )?
    .words,
  )
}

/// [`aligned_window`], with each token's start and end beside the words
/// ([`TimedAlignment`]).
fn aligned_window_timed(
  segments: &[TranscriptionSegment],
  alignment: &AlignmentView<'_>,
  tokenizer: &WhisperTokenizer,
  language_code: &str,
  grouping: WordGrouping,
  gather: AlignmentGather,
  swift_source_rows: usize,
) -> Result<TimedAlignment, SegmentError> {
  // :427-442 -- flatten every segment's tokens, in order; pair each with
  // its logged log-prob only when Swift's dictionary probe would have
  // found one (`segment.tokenLogProbs[index][token] != nil`): this
  // position's logged token id equals the token actually being gathered.
  // `.get` (rather than a direct index) additionally tolerates a shorter
  // `token_log_probs_slice` -- every `TranscriptionSegment` this crate
  // constructs keeps the two parallel, so that is a defensive no-op here,
  // not an intentional behavior difference from Swift's dictionary array.
  let mut word_token_ids: Vec<u32> = Vec::new();
  let mut filtered_log_probs: Vec<f32> = Vec::new();
  for segment in segments {
    let log_probs = segment.token_log_probs_slice();
    for (index, &token) in segment.tokens_slice().iter().enumerate() {
      word_token_ids.push(token);
      if let Some(&(logged_token, log_prob)) = log_probs.get(index)
        && logged_token == token
      {
        filtered_log_probs.push(log_prob);
      }
    }
  }

  // :444-461 -- Swift's `filteredIndices` are consecutive `0..N` by
  // construction (`:432` unconditionally appends `index + indexOffset`,
  // and `:441` advances `indexOffset` by exactly the previous segment's
  // token count), so "filtering" the alignment weights collapses to a
  // prefix take over rows `0..word_token_ids.len()`. The view is the
  // backend's FIXED-size accumulator now, so `alignment.rows() == max_ctx +
  // 1 >= needed` for every real window (`needed <= 224 < 225`): the
  // `.min(needed)` below always copies all `needed` rows and this `data`
  // zero-init is defensive-only. Whichever rows this window did not commit
  // carry an earlier window's bytes or the construction-time zero, and that
  // stale/zero content is the parity-bearing payload (whisper #41) -- exactly
  // like Swift's memcpy (`:454-459`) out of its once-allocated
  // `alignmentWeights` tensor whose uncommitted rows read back as the
  // previous window's values, into the per-call zero-initialized destination
  // (`initialValue: FloatType(0)`, `:450`).
  let needed = word_token_ids.len();
  let cols = alignment.cols();
  // The doc's zero-columns promise, honored HERE: `chunks_mut(0)` below
  // panics even over an empty buffer, and `dynamic_time_warping`'s own
  // zero-shape rejection sits after this construction — too late.
  if cols == 0 {
    return Err(SegmentError::InvalidAlignmentShape(
      InvalidAlignmentShape::new(alignment.rows(), cols, alignment.data().len()),
    ));
  }
  let mut data = vec![0.0f32; needed * cols];

  // Under the opt-in `SwiftParity` the gather is REPRODUCED rather than
  // copied: what `dynamicTimeWarping` reads back out of Swift's destination
  // array is a function of BOTH surfaces' CoreVideo row pitches, and neither
  // is a constant -- see `gather_swift_parity_into` for the two heights and
  // `coreml_f16_row_pitch` for why a compiled-in alignment quantum would make
  // this mode silently non-parity on a host CoreVideo pads differently. The
  // source shape is the accumulator Swift allocates once
  // (`TextDecoder.swift:141`, `swift_source_rows` rows -- NOT this port's
  // `alignment.rows()`, which carries one extra commit slot), the destination
  // the per-call `[needed, cols]` one (`SegmentSeeker.swift:450`).
  //
  // `needed == 0` skips the probes (and the no-op gather): there is nothing to
  // reproduce, and the empty `segments` input it comes from is already
  // `SegmentError::InvalidAlignmentShape`'s -- reported below by
  // `dynamic_time_warping`, as this function's `# Errors` doc promises.
  if gather == AlignmentGather::SwiftParity && needed > 0 {
    gather_swift_parity_into(
      &mut data,
      alignment,
      needed,
      cols,
      swift_source_rows,
      &coreml_f16_row_pitch,
    )?;
  } else {
    // `Complete`, the DEFAULT: the prefix take with no gather artifacts at
    // all -- and, just as importantly, no surface allocation, no pitch
    // measurement and no host-dependent assumption anywhere on this path.
    for (row_index, row) in data
      .chunks_mut(cols)
      .enumerate()
      .take(alignment.rows().min(needed))
    {
      row.copy_from_slice(alignment.row(row_index));
    }
  }

  let filtered = AlignmentMatrix::new(data, needed, cols);

  // :465-472. The construction above guarantees `filtered.rows() ==
  // word_token_ids.len()` always -- the invariant `find_alignment` needs
  // to index `start_times`/`end_times`/`token_log_probs` in lockstep with
  // `word_token_ids` -- regardless of how `alignment.rows()` compares to
  // `word_token_ids.len()`. When `word_token_ids` is empty this makes
  // `filtered.rows() == 0`; `dynamic_time_warping` checks that
  // unconditionally, before `find_alignment`'s own `<= 1 word` early
  // return (see that function's doc), so an empty `segments` input
  // surfaces `SegmentError::InvalidAlignmentShape` here rather than
  // degrading to word-less segments.
  find_alignment_timed(
    &word_token_ids,
    &filtered.view(),
    &filtered_log_probs,
    tokenizer,
    language_code,
    grouping,
  )
}

/// The visible word list of `segments`, exactly Swift's: the duration hack
/// and sentence-boundary truncation over the alignment's words, punctuation
/// merged, then the word-timing pass (`SegmentSeeker.swift:474-493`).
fn visible_words(
  segments: &[TranscriptionSegment],
  mut merged: Vec<WordTiming>,
  tokenizer: &WhisperTokenizer,
  seek: usize,
  prepended: &str,
  appended: &str,
  last_speech_timestamp: f32,
) -> Result<Vec<TranscriptionSegment>, SegmentError> {
  // :474-477 -- the upstream "hack" Swift's own comment flags (reference,
  // Swift's own citation at `:474-475`: openai/whisper
  // `whisper/timing.py#L305`, commit `ba3f3cd`): constrain the
  // median/max word duration, then truncate overlong words at sentence
  // boundaries, before merging punctuation.
  let word_durations = calculate_word_duration_constraints(&merged);
  merged = truncate_long_words_at_sentence_boundaries(merged, word_durations.max_duration());

  // :480-482 -- gated on the merged ALIGNMENT being non-empty, not on
  // `prepended`/`appended` (a correction to this task's brief: Swift's
  // `if !alignment.isEmpty` reads the alignment array, not the
  // punctuation-string parameters). `merge_punctuations` is already a
  // no-op on an empty slice (see its own doc), so this gate changes
  // nothing observable -- kept only to mirror Swift's exact shape.
  if !merged.is_empty() {
    merged = merge_punctuations(&merged, prepended, appended);
  }

  // :484-493.
  update_segments_with_word_timings(
    segments,
    &merged,
    seek,
    last_speech_timestamp,
    word_durations.median(),
    word_durations.max_duration(),
    tokenizer,
  )
}

/// One unit of a window's text as its alignment places it ([`TextUnit`]):
/// the half-open span of its segment's own tokens it came from, and where it
/// starts and ends, in samples from the window's start.
///
/// The clip-back of a window a short clip decodes reads its provenance from
/// these, never from the visible word list — whose merging replaces and
/// filters words away, and whose word-timing pass assigns them to segments
/// by token counts. Its times stay in the window's own samples from the
/// moment the alignment yields them — a frame is 20 ms, `320` samples, so
/// they are exact — and no time of the audio in seconds is ever compared:
/// an hour into a file, an f32 second is a quarter of a millisecond wide.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(crate) struct RawWord {
  span: (usize, usize),
  start: usize,
  end: usize,
}

impl RawWord {
  /// A raw word over its segment's tokens `span`, from sample `start` to
  /// sample `end` of its window.
  pub(crate) const fn new(span: (usize, usize), start: usize, end: usize) -> Self {
    Self { span, start, end }
  }

  /// The half-open span of its segment's tokens it came from.
  pub(crate) const fn span(&self) -> (usize, usize) {
    self.span
  }

  /// Its start, in samples from the window's start.
  pub(crate) const fn start(&self) -> usize {
    self.start
  }

  /// Its end, in samples from the window's start.
  pub(crate) const fn end(&self) -> usize {
    self.end
  }
}

/// Samples per frame of a window's alignment: 20 ms.
const SAMPLES_PER_FRAME: usize = SAMPLE_RATE as usize / 50;

/// A segment a short clip's clip-back kept: the segment, the raw words it
/// kept, and the positions of the tokens it kept among its window's
/// flattened tokens, in order.
pub(crate) type KeptSegment = (TranscriptionSegment, Vec<RawWord>, Vec<usize>);

/// Every segment's raw words: each text unit ([`TextUnit`]) belongs to the
/// segment holding its FIRST token in the window's flattened tokens — the
/// intersection of its span with the segments' token ranges, never a count
/// of tokens — its span restated in that segment's own tokens, and its frames
/// made samples of the window. A unit that straddles two segments belongs to
/// the first; its span may run past that segment's tokens.
pub(crate) fn own_raw_words(
  segments: &[TranscriptionSegment],
  units: &[TextUnit],
) -> Vec<Vec<RawWord>> {
  let mut ranges = Vec::with_capacity(segments.len());
  let mut at = 0usize;
  for segment in segments {
    let len = segment.tokens_slice().len();
    ranges.push(at..at + len);
    at += len;
  }
  let mut owned: Vec<Vec<RawWord>> = segments.iter().map(|_| Vec::new()).collect();
  for &((from, to), start, end) in units {
    let Some(index) = ranges.iter().position(|range| range.contains(&from)) else {
      continue;
    };
    let offset = ranges[index].start;
    owned[index].push(RawWord::new(
      (from - offset, to.saturating_sub(offset)),
      start * SAMPLES_PER_FRAME,
      end * SAMPLES_PER_FRAME,
    ));
  }
  owned
}

#[cfg(test)]
mod tests;